You are viewing a plain text version of this content. The canonical link for it is here.
Posted to commits@hudi.apache.org by vi...@apache.org on 2019/03/26 17:12:51 UTC
[incubator-hudi] branch master updated: [HUDI-63] Removed unused BucketedIndex code

This is an automated email from the ASF dual-hosted git repository.

vinoth pushed a commit to branch master
in repository https://gitbox.apache.org/repos/asf/incubator-hudi.git


The following commit(s) were added to refs/heads/master by this push:
     new 395806f  [HUDI-63] Removed unused BucketedIndex code
395806f is described below

commit 395806fc684d8f6c3fa1b6b0f29e601cb6f4bbd0
Author: ambition119 <12...@qq.com>
AuthorDate: Wed Mar 20 12:25:51 2019 +0800

    [HUDI-63] Removed unused BucketedIndex code
---
 .../java/com/uber/hoodie/index/HoodieIndex.java    |   5 +-
 .../uber/hoodie/index/bucketed/BucketedIndex.java  | 115 ---------------------
 2 files changed, 1 insertion(+), 119 deletions(-)

diff --git a/hoodie-client/src/main/java/com/uber/hoodie/index/HoodieIndex.java b/hoodie-client/src/main/java/com/uber/hoodie/index/HoodieIndex.java
index 1789784..5eea915 100644
--- a/hoodie-client/src/main/java/com/uber/hoodie/index/HoodieIndex.java
+++ b/hoodie-client/src/main/java/com/uber/hoodie/index/HoodieIndex.java
@@ -25,7 +25,6 @@ import com.uber.hoodie.config.HoodieWriteConfig;
 import com.uber.hoodie.exception.HoodieIndexException;
 import com.uber.hoodie.index.bloom.HoodieBloomIndex;
 import com.uber.hoodie.index.bloom.HoodieGlobalBloomIndex;
-import com.uber.hoodie.index.bucketed.BucketedIndex;
 import com.uber.hoodie.index.hbase.HBaseIndex;
 import com.uber.hoodie.table.HoodieTable;
 import java.io.Serializable;
@@ -56,8 +55,6 @@ public abstract class HoodieIndex<T extends HoodieRecordPayload> implements Seri
         return new HoodieBloomIndex<>(config);
       case GLOBAL_BLOOM:
         return new HoodieGlobalBloomIndex<>(config);
-      case BUCKETED:
-        return new BucketedIndex<>(config);
       default:
         throw new HoodieIndexException("Index type unspecified, set " + config.getIndexType());
     }
@@ -119,6 +116,6 @@ public abstract class HoodieIndex<T extends HoodieRecordPayload> implements Seri
 
 
   public enum IndexType {
-    HBASE, INMEMORY, BLOOM, GLOBAL_BLOOM, BUCKETED
+    HBASE, INMEMORY, BLOOM, GLOBAL_BLOOM
   }
 }
diff --git a/hoodie-client/src/main/java/com/uber/hoodie/index/bucketed/BucketedIndex.java b/hoodie-client/src/main/java/com/uber/hoodie/index/bucketed/BucketedIndex.java
deleted file mode 100644
index dcb84ad..0000000
--- a/hoodie-client/src/main/java/com/uber/hoodie/index/bucketed/BucketedIndex.java
+++ /dev/null
@@ -1,115 +0,0 @@
-/*
- *  Copyright (c) 2017 Uber Technologies, Inc. (hoodie-dev-group@uber.com)
- *
- *  Licensed under the Apache License, Version 2.0 (the "License");
- *  you may not use this file except in compliance with the License.
- *  You may obtain a copy of the License at
- *
- *           http://www.apache.org/licenses/LICENSE-2.0
- *
- * Unless required by applicable law or agreed to in writing, software
- * distributed under the License is distributed on an "AS IS" BASIS,
- * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
- * See the License for the specific language governing permissions and
- * limitations under the License.
- *
- *
- */
-
-package com.uber.hoodie.index.bucketed;
-
-import com.google.common.base.Optional;
-import com.uber.hoodie.WriteStatus;
-import com.uber.hoodie.common.model.HoodieKey;
-import com.uber.hoodie.common.model.HoodieRecord;
-import com.uber.hoodie.common.model.HoodieRecordLocation;
-import com.uber.hoodie.common.model.HoodieRecordPayload;
-import com.uber.hoodie.config.HoodieWriteConfig;
-import com.uber.hoodie.exception.HoodieIndexException;
-import com.uber.hoodie.index.HoodieIndex;
-import com.uber.hoodie.table.HoodieTable;
-import org.apache.log4j.LogManager;
-import org.apache.log4j.Logger;
-import org.apache.spark.api.java.JavaPairRDD;
-import org.apache.spark.api.java.JavaRDD;
-import org.apache.spark.api.java.JavaSparkContext;
-import scala.Tuple2;
-
-/**
- * An `stateless` index implementation that will using a deterministic mapping function to determine
- * the fileID for a given record.
- * <p>
- * Pros: - Fast
- * <p>
- * Cons : - Need to tune the number of buckets per partition path manually (FIXME: Need to autotune
- * this) - Could increase write amplification on copy-on-write storage since inserts always rewrite
- * files - Not global.
- */
-public class BucketedIndex<T extends HoodieRecordPayload> extends HoodieIndex<T> {
-
-  private static Logger logger = LogManager.getLogger(BucketedIndex.class);
-
-  public BucketedIndex(HoodieWriteConfig config) {
-    super(config);
-  }
-
-  private String getBucket(String recordKey) {
-    return String.valueOf(recordKey.hashCode() % config.getNumBucketsPerPartition());
-  }
-
-  @Override
-  public JavaPairRDD<HoodieKey, Optional<String>> fetchRecordLocation(JavaRDD<HoodieKey> hoodieKeys,
-      JavaSparkContext jsc, HoodieTable<T> hoodieTable) {
-    return hoodieKeys.mapToPair(hk -> new Tuple2<>(hk, Optional.of(getBucket(hk.getRecordKey()))));
-  }
-
-  @Override
-  public JavaRDD<HoodieRecord<T>> tagLocation(JavaRDD<HoodieRecord<T>> recordRDD, JavaSparkContext jsc,
-      HoodieTable<T> hoodieTable)
-      throws HoodieIndexException {
-    return recordRDD.map(record -> {
-      String bucket = getBucket(record.getRecordKey());
-      //HACK(vc) a non-existent commit is provided here.
-      record.setCurrentLocation(new HoodieRecordLocation("000", bucket));
-      return record;
-    });
-  }
-
-  @Override
-  public JavaRDD<WriteStatus> updateLocation(JavaRDD<WriteStatus> writeStatusRDD, JavaSparkContext jsc,
-      HoodieTable<T> hoodieTable)
-      throws HoodieIndexException {
-    return writeStatusRDD;
-  }
-
-  @Override
-  public boolean rollbackCommit(String commitTime) {
-    // nothing to rollback in the index.
-    return true;
-  }
-
-  /**
-   * Bucketing is still done within each partition.
-   */
-  @Override
-  public boolean isGlobal() {
-    return false;
-  }
-
-  /**
-   * Since indexing is just a deterministic hash, we can identify file group correctly even without
-   * an index on the actual log file.
-   */
-  @Override
-  public boolean canIndexLogFiles() {
-    return true;
-  }
-
-  /**
-   * Indexing is just a hash function.
-   */
-  @Override
-  public boolean isImplicitWithStorage() {
-    return true;
-  }
-}