mirror of https://github.com/apache/cassandra
replace JenkinsHash w/ MurmurHash. its hash distribution is just as good, and it's faster
git-svn-id: https://svn.apache.org/repos/asf/incubator/cassandra/trunk@766138 13f79535-47bb-0310-9956-ffa450edef68
This commit is contained in:
parent
7f256c32cb
commit
04344cbb03
|
|
@ -1,185 +0,0 @@
|
|||
package org.apache.cassandra.utils;
|
||||
|
||||
class JenkinsHash
|
||||
{
|
||||
|
||||
// max value to limit it to 4 bytes
|
||||
private static final long MAX_VALUE = 0xFFFFFFFFL;
|
||||
|
||||
// internal variables used in the various calculations
|
||||
private long a_;
|
||||
private long b_;
|
||||
private long c_;
|
||||
|
||||
/**
|
||||
* Convert a byte into a long value without making it negative.
|
||||
*/
|
||||
private long byteToLong(byte b)
|
||||
{
|
||||
long val = b & 0x7F;
|
||||
if ((b & 0x80) != 0)
|
||||
{
|
||||
val += 128;
|
||||
}
|
||||
return val;
|
||||
}
|
||||
|
||||
/**
|
||||
* Do addition and turn into 4 bytes.
|
||||
*/
|
||||
private long add(long val, long add)
|
||||
{
|
||||
return (val + add) & MAX_VALUE;
|
||||
}
|
||||
|
||||
/**
|
||||
* Do subtraction and turn into 4 bytes.
|
||||
*/
|
||||
private long subtract(long val, long subtract)
|
||||
{
|
||||
return (val - subtract) & MAX_VALUE;
|
||||
}
|
||||
|
||||
/**
|
||||
* Left shift val by shift bits and turn in 4 bytes.
|
||||
*/
|
||||
private long xor(long val, long xor)
|
||||
{
|
||||
return (val ^ xor) & MAX_VALUE;
|
||||
}
|
||||
|
||||
/**
|
||||
* Left shift val by shift bits. Cut down to 4 bytes.
|
||||
*/
|
||||
private long leftShift(long val, int shift)
|
||||
{
|
||||
return (val << shift) & MAX_VALUE;
|
||||
}
|
||||
|
||||
/**
|
||||
* Convert 4 bytes from the buffer at offset into a long value.
|
||||
*/
|
||||
private long fourByteToLong(byte[] bytes, int offset)
|
||||
{
|
||||
return (byteToLong(bytes[offset + 0])
|
||||
+ (byteToLong(bytes[offset + 1]) << 8)
|
||||
+ (byteToLong(bytes[offset + 2]) << 16) + (byteToLong(bytes[offset + 3]) << 24));
|
||||
}
|
||||
|
||||
/**
|
||||
* Mix up the values in the hash function.
|
||||
*/
|
||||
private void hashMix()
|
||||
{
|
||||
a_ = subtract(a_, b_);
|
||||
a_ = subtract(a_, c_);
|
||||
a_ = xor(a_, c_ >> 13);
|
||||
b_ = subtract(b_, c_);
|
||||
b_ = subtract(b_, a_);
|
||||
b_ = xor(b_, leftShift(a_, 8));
|
||||
c_ = subtract(c_, a_);
|
||||
c_ = subtract(c_, b_);
|
||||
c_ = xor(c_, (b_ >> 13));
|
||||
a_ = subtract(a_, b_);
|
||||
a_ = subtract(a_, c_);
|
||||
a_ = xor(a_, (c_ >> 12));
|
||||
b_ = subtract(b_, c_);
|
||||
b_ = subtract(b_, a_);
|
||||
b_ = xor(b_, leftShift(a_, 16));
|
||||
c_ = subtract(c_, a_);
|
||||
c_ = subtract(c_, b_);
|
||||
c_ = xor(c_, (b_ >> 5));
|
||||
a_ = subtract(a_, b_);
|
||||
a_ = subtract(a_, c_);
|
||||
a_ = xor(a_, (c_ >> 3));
|
||||
b_ = subtract(b_, c_);
|
||||
b_ = subtract(b_, a_);
|
||||
b_ = xor(b_, leftShift(a_, 10));
|
||||
c_ = subtract(c_, a_);
|
||||
c_ = subtract(c_, b_);
|
||||
c_ = xor(c_, (b_ >> 15));
|
||||
}
|
||||
|
||||
/**
|
||||
* Hash a variable-length key into a 32-bit value. Every bit of the key
|
||||
* affects every bit of the return value. Every 1-bit and 2-bit delta
|
||||
* achieves avalanche. The best hash table sizes are powers of 2.
|
||||
*
|
||||
* @param buffer
|
||||
* Byte array that we are hashing on.
|
||||
* @param initialValue
|
||||
* Initial value of the hash if we are continuing from a previous
|
||||
* run. 0 if none.
|
||||
* @return Hash value for the buffer.
|
||||
*/
|
||||
public long hash(byte[] buffer, long initialValue)
|
||||
{
|
||||
int len, pos;
|
||||
|
||||
// set up the internal state
|
||||
// the golden ratio; an arbitrary value
|
||||
a_ = 0x09e3779b9L;
|
||||
// the golden ratio; an arbitrary value
|
||||
b_ = 0x09e3779b9L;
|
||||
// the previous hash value
|
||||
c_ = initialValue;
|
||||
|
||||
// handle most of the key
|
||||
pos = 0;
|
||||
for (len = buffer.length; len >= 12; len -= 12)
|
||||
{
|
||||
a_ = add(a_, fourByteToLong(buffer, pos));
|
||||
b_ = add(b_, fourByteToLong(buffer, pos + 4));
|
||||
c_ = add(c_, fourByteToLong(buffer, pos + 8));
|
||||
hashMix();
|
||||
pos += 12;
|
||||
}
|
||||
|
||||
c_ += buffer.length;
|
||||
|
||||
// all the case statements fall through to the next on purpose
|
||||
switch (len)
|
||||
{
|
||||
case 11:
|
||||
c_ = add(c_, leftShift(byteToLong(buffer[pos + 10]), 24));
|
||||
case 10:
|
||||
c_ = add(c_, leftShift(byteToLong(buffer[pos + 9]), 16));
|
||||
case 9:
|
||||
c_ = add(c_, leftShift(byteToLong(buffer[pos + 8]), 8));
|
||||
// the first byte of c is reserved for the length
|
||||
case 8:
|
||||
b_ = add(b_, leftShift(byteToLong(buffer[pos + 7]), 24));
|
||||
case 7:
|
||||
b_ = add(b_, leftShift(byteToLong(buffer[pos + 6]), 16));
|
||||
case 6:
|
||||
b_ = add(b_, leftShift(byteToLong(buffer[pos + 5]), 8));
|
||||
case 5:
|
||||
b_ = add(b_, byteToLong(buffer[pos + 4]));
|
||||
case 4:
|
||||
a_ = add(a_, leftShift(byteToLong(buffer[pos + 3]), 24));
|
||||
case 3:
|
||||
a_ = add(a_, leftShift(byteToLong(buffer[pos + 2]), 16));
|
||||
case 2:
|
||||
a_ = add(a_, leftShift(byteToLong(buffer[pos + 1]), 8));
|
||||
case 1:
|
||||
a_ = add(a_, byteToLong(buffer[pos + 0]));
|
||||
// case 0: nothing left to add
|
||||
}
|
||||
hashMix();
|
||||
|
||||
return c_;
|
||||
}
|
||||
|
||||
/**
|
||||
* See hash(byte[] buffer, long initialValue)
|
||||
*
|
||||
* @param buffer
|
||||
* Byte array that we are hashing on.
|
||||
* @return Hash value for the buffer.
|
||||
*/
|
||||
public long hash(byte[] buffer)
|
||||
{
|
||||
return hash(buffer, 0);
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -0,0 +1,77 @@
|
|||
/**
|
||||
* Licensed to the Apache Software Foundation (ASF) under one
|
||||
* or more contributor license agreements. See the NOTICE file
|
||||
* distributed with this work for additional information
|
||||
* regarding copyright ownership. The ASF licenses this file
|
||||
* to you under the Apache License, Version 2.0 (the
|
||||
* "License"); you may not use this file except in compliance
|
||||
* with the License. You may obtain a copy of the License at
|
||||
*
|
||||
* http://www.apache.org/licenses/LICENSE-2.0
|
||||
*
|
||||
* Unless required by applicable law or agreed to in writing, software
|
||||
* distributed under the License is distributed on an "AS IS" BASIS,
|
||||
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
* See the License for the specific language governing permissions and
|
||||
* limitations under the License.
|
||||
*/
|
||||
|
||||
package org.apache.cassandra.utils;
|
||||
|
||||
/**
|
||||
* This is a very fast, non-cryptographic hash suitable for general hash-based
|
||||
* lookup. See http://murmurhash.googlepages.com/ for more details.
|
||||
*
|
||||
* <p>The C version of MurmurHash 2.0 found at that site was ported
|
||||
* to Java by Andrzej Bialecki (ab at getopt org).</p>
|
||||
*/
|
||||
public class MurmurHash {
|
||||
public int hash(byte[] data, int length, int seed) {
|
||||
int m = 0x5bd1e995;
|
||||
int r = 24;
|
||||
|
||||
int h = seed ^ length;
|
||||
|
||||
int len_4 = length >> 2;
|
||||
|
||||
for (int i = 0; i < len_4; i++) {
|
||||
int i_4 = i << 2;
|
||||
int k = data[i_4 + 3];
|
||||
k = k << 8;
|
||||
k = k | (data[i_4 + 2] & 0xff);
|
||||
k = k << 8;
|
||||
k = k | (data[i_4 + 1] & 0xff);
|
||||
k = k << 8;
|
||||
k = k | (data[i_4 + 0] & 0xff);
|
||||
k *= m;
|
||||
k ^= k >>> r;
|
||||
k *= m;
|
||||
h *= m;
|
||||
h ^= k;
|
||||
}
|
||||
|
||||
// avoid calculating modulo
|
||||
int len_m = len_4 << 2;
|
||||
int left = length - len_m;
|
||||
|
||||
if (left != 0) {
|
||||
if (left >= 3) {
|
||||
h ^= (int) data[length - 3] << 16;
|
||||
}
|
||||
if (left >= 2) {
|
||||
h ^= (int) data[length - 2] << 8;
|
||||
}
|
||||
if (left >= 1) {
|
||||
h ^= (int) data[length - 1];
|
||||
}
|
||||
|
||||
h *= m;
|
||||
}
|
||||
|
||||
h ^= h >>> 13;
|
||||
h *= m;
|
||||
h ^= h >>> 15;
|
||||
|
||||
return h;
|
||||
}
|
||||
}
|
||||
Loading…
Reference in New Issue