如何获得比numpy.dot更快的矩阵乘法代码？

n_row=1000 n_col=1000 n_batch=10 def test_plain_numpy(): A=np.random.rand(n_row,n_col)# float by default? B=np.random.rand(n_col,n_row) t0= time.time() res= np.dot(A,B) print (time.time()-t0) #A,B in RAM, C on disk def test_hdf5_ram(): rows = n_row cols = n_col batches = n_batch #using numpy array A=np.random.rand(n_row,n_col) B=np.random.rand(n_col,n_row) #settings for all hdf5 files atom = tables.Float32Atom() #if store uint8 less memory? filters = tables.Filters(complevel=9, complib='blosc') # tune parameters Nchunk = 128 # ? chunkshape = (Nchunk, Nchunk) chunk_multiple = 1 block_size = chunk_multiple * Nchunk #using hdf5 fileName_C = 'CArray_C.h5' shape = (A.shape[0], B.shape[1]) h5f_C = tables.open_file(fileName_C, 'w') C = h5f_C.create_carray(h5f_C.root, 'CArray', atom, shape, chunkshape=chunkshape, filters=filters) sz= block_size t0= time.time() for i in range(0, A.shape[0], sz): for j in range(0, B.shape[1], sz): for k in range(0, A.shape[1], sz): C[i:i+sz,j:j+sz] += np.dot(A[i:i+sz,k:k+sz],B[k:k+sz,j:j+sz]) print (time.time()-t0) h5f_C.close() def test_hdf5_disk(): rows = n_row cols = n_col batches = n_batch #settings for all hdf5 files atom = tables.Float32Atom() #if store uint8 less memory? filters = tables.Filters(complevel=9, complib='blosc') # tune parameters Nchunk = 128 # ? chunkshape = (Nchunk, Nchunk) chunk_multiple = 1 block_size = chunk_multiple * Nchunk fileName_A = 'carray_A.h5' shape_A = (n_row*n_batch, n_col) # predefined size h5f_A = tables.open_file(fileName_A, 'w') A = h5f_A.create_carray(h5f_A.root, 'CArray', atom, shape_A, chunkshape=chunkshape, filters=filters) for i in range(batches): data = np.random.rand(n_row, n_col) A[i*n_row:(i+1)*n_row]= data[:] rows = n_col cols = n_row batches = n_batch fileName_B = 'carray_B.h5' shape_B = (rows, cols*batches) # predefined size h5f_B = tables.open_file(fileName_B, 'w') B = h5f_B.create_carray(h5f_B.root, 'CArray', atom, shape_B, chunkshape=chunkshape, filters=filters) sz= rows/batches for i in range(batches): data = np.random.rand(sz, cols*batches) B[i*sz:(i+1)*sz]= data[:] fileName_C = 'CArray_C.h5' shape = (A.shape[0], B.shape[1]) h5f_C = tables.open_file(fileName_C, 'w') C = h5f_C.create_carray(h5f_C.root, 'CArray', atom, shape, chunkshape=chunkshape, filters=filters) sz= block_size t0= time.time() for i in range(0, A.shape[0], sz): for j in range(0, B.shape[1], sz): for k in range(0, A.shape[1], sz): C[i:i+sz,j:j+sz] += np.dot(A[i:i+sz,k:k+sz],B[k:k+sz,j:j+sz]) print (time.time()-t0) h5f_A.close() h5f_B.close() h5f_C.close()

1条回答

网友

1楼 · 发布于 2024-09-23 22:25:00

当np.dot发送到BLAS时

NumPy被编译成使用BLAS
运行时可以使用BLAS实现
您的数据有一个数据类型float32、float64、complex32或complex64，并且
数据在内存中适当对齐。

否则，它默认使用自己的、缓慢的矩阵乘法例程。

检查BLAS链接的说明如下here。简而言之，检查NumPy安装中是否有文件_dotblas.so或类似文件。如果有的话，请检查它链接到哪个BLAS库；参考BLAS速度慢，ATLAS速度快，OpenBLAS和特定于供应商的版本（如Intel MKL）甚至更快。注意多线程BLAS实现，因为它们是Python的multiprocessing的don't play nicely。

接下来，通过检查数组的flags来检查数据对齐。在1.7.2之前的NumPy版本中，np.dot的两个参数都应该是C顺序的。在NumPy>；=1.7.2中，这不再像Fortran数组的特殊情况那样重要。

>>> X = np.random.randn(10, 4)
>>> Y = np.random.randn(7, 4).T
>>> X.flags
  C_CONTIGUOUS : True
  F_CONTIGUOUS : False
  OWNDATA : True
  WRITEABLE : True
  ALIGNED : True
  UPDATEIFCOPY : False
>>> Y.flags
  C_CONTIGUOUS : False
  F_CONTIGUOUS : True
  OWNDATA : False
  WRITEABLE : True
  ALIGNED : True
  UPDATEIFCOPY : False

如果您的NumPy没有链接到BLAS，可以（简单地）重新安装它，或者（硬）使用SciPy中的BLASgemm（广义矩阵乘法）函数：

>>> from scipy.linalg import get_blas_funcs
>>> gemm = get_blas_funcs("gemm", [X, Y])
>>> np.all(gemm(1, X, Y) == np.dot(X, Y))
True

这看起来很简单，但几乎没有任何错误检查，所以你必须真正知道你在做什么。

相关问题更多 >

编程相关推荐

热门问题

热门文章