哪位大神给看看这个矩阵向量乘法的CUDA程序为什么不对

哪位大神给看看这个矩阵向量乘法的CUDA程序为什么不对,里面Nd的大小是随意设的。大神帮帮忙,非常感谢。或者哪位大神有矩阵向量相乘比较好的代码发一份也非常感谢
global void matXvector_kernel(const float * Md, const float * Vd, float* Pd, int colsize, int pitchItem)
{
/* 函数功能: 矩阵乘向量kernel函数
参数: (矩阵指针,向量指针,结果向量指针,矩阵的列数,矩阵行主元的个数)
*/
shared float Mds[TILE_WIDTH][TILE_WIDTH];
shared float Vds[TILE_WIDTH];
float Nd[2000][2000] = {0};
int bx = blockIdx.x; int by = blockIdx.y;
int tx = threadIdx.x; int ty = threadIdx.y;
int Row = by*blockDim.y + ty;
float Pvalue = 0.0;
if ((by*blockDim.y + ty) < pitchItem && (bx*blockDim.x + tx) < colsize){
Mds[ty][tx] = Md[(by*blockDim.y + ty)*colsize + bx*blockDim.x + tx];
Vds[tx] = Vd[bx*blockDim.x + tx];
}
else
{
Mds[ty][tx] = 0;
Vds[tx] = 0;
}
__syncthreads();
for (int k = 0; k < blockDim.x; ++k)
{
Nd[Row][bx] += Mds[ty][k] * Vds[k];
}
__syncthreads();
if (Row < pitchItem && tx < 1)
{

for (int k = 0; k < gridDim.x; ++k)
{
Pd[Row] += Nd[Row][k];
}
}
}

挺简单的,你懂啊 http://www.yeyelujiaduolu.com