令我惊讶的是,这个速度较慢,但不管它值多少钱,这里有一个c++解决方案,它可以实现您所指的功能—将每一行作为一组字节进行散列。“诀窍”是获取元素的地址
<char*>&a[i, 0]
-其他一切都是簿记。
我可能做了一些明显的次优和/或使用不同的哈希表impl性能可能更好。
编辑:
回复:如何从行创建字符串
我认为你能做的最好的就是——构建一个
bytes
对象。这必然涉及到行的副本参见c api
docs
.
%%cython
from numpy cimport *
cimport numpy as np
import numpy as np
from cpython.bytes cimport PyBytes_FromStringAndSize
def unique_int_string(ndarray[np.int64_t, ndim=2] a):
cdef int i, len_before
cdef int nr = a.shape[0]
cdef int nc = a.shape[1]
cdef set s = set()
cdef ndarray[np.uint8_t, cast = True] idx = np.zeros(nr, dtype='bool')
cdef bytes string
for i in range(nr):
len_before = len(s)
string = PyBytes_FromStringAndSize(<char*>&a[i, 0], sizeof(np.int64_t) * nc)
s.add(string)
if len(s) > len_before:
idx[i] = True
return idx
//时间安排
In [9]: from unique import unique_ints
In [10]: %timeit unique_int_tuple4(a)
100 loops, best of 3: 10.1 ms per loop
In [11]: %timeit unique_ints(a)
100 loops, best of 3: 11.9 ms per loop
In [12]: (unique_ints(a) == unique_int_tuple4(a)).all()
Out[12]: True
//助手。h类
#include <unordered_set>
#include <cstring>
struct Hasher {
size_t size;
size_t operator()(char* buf) const {
// https://github.com/yt-project/yt/blob/c1569367c6e3d8d0a02e10d0f3d0bd701d2e2114/yt/utilities/lib/fnv_hash.pyx
size_t hash_val = 2166136261;
for (int i = 0; i < size; ++i) {
hash_val ^= buf[i];
hash_val *= 16777619;
}
return hash_val;
}
};
struct Comparer {
size_t size;
bool operator()(char* lhs, char* rhs) const {
return (std::memcmp(lhs, rhs, size) == 0) ? true : false;
}
};
struct ArraySet {
std::unordered_set<char*, Hasher, Comparer> set;
ArraySet (size_t size) : set(0, Hasher{size}, Comparer{size}) {}
ArraySet () {}
bool add(char* buf) {
auto p = set.insert(buf);
return p.second;
}
};
//独一无二。pyx公司
from numpy cimport int64_t, uint8_t
import numpy as np
cdef extern from 'helper.h' nogil:
cdef cppclass ArraySet:
ArraySet()
ArraySet(size_t)
bint add(char*)
def unique_ints(int64_t[:, :] a):
cdef:
Py_ssize_t i, nr = a.shape[0], nc = a.shape[1]
ArraySet s = ArraySet(sizeof(int64_t) * nc)
uint8_t[:] idx = np.zeros(nr, dtype='uint8')
bint found;
for i in range(nr):
found = s.add(<char*>&a[i, 0])
if found:
idx[i] = True
return idx
//设置。py公司
from setuptools import setup, Extension
from Cython.Build import cythonize
import numpy as np
exts = [
Extension('unique', ['unique.pyx'], language='c++', include_dirs=[np.get_include()])
]
setup(name='test', ext_modules=cythonize(exts))