View Javadoc

1   /**
2    *
3    * Licensed to the Apache Software Foundation (ASF) under one
4    * or more contributor license agreements.  See the NOTICE file
5    * distributed with this work for additional information
6    * regarding copyright ownership.  The ASF licenses this file
7    * to you under the Apache License, Version 2.0 (the
8    * "License"); you may not use this file except in compliance
9    * with the License.  You may obtain a copy of the License at
10   *
11   *     http://www.apache.org/licenses/LICENSE-2.0
12   *
13   * Unless required by applicable law or agreed to in writing, software
14   * distributed under the License is distributed on an "AS IS" BASIS,
15   * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
16   * See the License for the specific language governing permissions and
17   * limitations under the License.
18   */
19  package org.apache.hadoop.hbase.mapreduce;
20  
21  import java.util.Iterator;
22  import java.util.List;
23  import java.util.TreeSet;
24  
25  import org.apache.hadoop.hbase.classification.InterfaceAudience;
26  import org.apache.hadoop.hbase.classification.InterfaceStability;
27  import org.apache.hadoop.hbase.Cell;
28  import org.apache.hadoop.hbase.CellComparator;
29  import org.apache.hadoop.hbase.KeyValue;
30  import org.apache.hadoop.hbase.KeyValueUtil;
31  import org.apache.hadoop.hbase.client.Put;
32  import org.apache.hadoop.hbase.io.ImmutableBytesWritable;
33  import org.apache.hadoop.mapreduce.Reducer;
34  import org.apache.hadoop.util.StringUtils;
35  
36  /**
37   * Emits sorted Puts.
38   * Reads in all Puts from passed Iterator, sorts them, then emits
39   * Puts in sorted order.  If lots of columns per row, it will use lots of
40   * memory sorting.
41   * @see HFileOutputFormat
42   * @see KeyValueSortReducer
43   */
44  @InterfaceAudience.Public
45  @InterfaceStability.Stable
46  public class PutSortReducer extends
47      Reducer<ImmutableBytesWritable, Put, ImmutableBytesWritable, KeyValue> {
48    
49    @Override
50    protected void reduce(
51        ImmutableBytesWritable row,
52        java.lang.Iterable<Put> puts,
53        Reducer<ImmutableBytesWritable, Put,
54                ImmutableBytesWritable, KeyValue>.Context context)
55        throws java.io.IOException, InterruptedException
56    {
57      // although reduce() is called per-row, handle pathological case
58      long threshold = context.getConfiguration().getLong(
59          "putsortreducer.row.threshold", 1L * (1<<30));
60      Iterator<Put> iter = puts.iterator();
61      while (iter.hasNext()) {
62        TreeSet<KeyValue> map = new TreeSet<KeyValue>(CellComparator.COMPARATOR);
63        long curSize = 0;
64        // stop at the end or the RAM threshold
65        while (iter.hasNext() && curSize < threshold) {
66          Put p = iter.next();
67          for (List<Cell> cells: p.getFamilyCellMap().values()) {
68            for (Cell cell: cells) {
69              KeyValue kv = KeyValueUtil.ensureKeyValue(cell);
70              map.add(kv);
71              curSize += kv.heapSize();
72            }
73          }
74        }
75        context.setStatus("Read " + map.size() + " entries of " + map.getClass()
76            + "(" + StringUtils.humanReadableInt(curSize) + ")");
77        int index = 0;
78        for (KeyValue kv : map) {
79          context.write(row, kv);
80          if (++index % 100 == 0)
81            context.setStatus("Wrote " + index);
82        }
83  
84        // if we have more entries to process
85        if (iter.hasNext()) {
86          // force flush because we cannot guarantee intra-row sorted order
87          context.write(null, null);
88        }
89      }
90    }
91  }