Пытаюсь создать 2D texture object, 4x4 uint8_t.
Код следующий:
__global__ void kernel(cudaTextureObject_t tex)
{
int x = threadIdx.x;
int y = threadIdx.y;
uint8_t val = tex2D<uint8_t>(tex, x, y);
printf("%d,", val);
return;
}
int main(int argc, char **argv)
{
cudaTextureObject_t tex;
uint8_t dataIn[16] = {0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15};
uint8_t* dataDev = 0;
cudaMalloc((void**)&dataDev, 16);
struct cudaResourceDesc resDesc;
memset(&resDesc, 0, sizeof(resDesc));
resDesc.resType = cudaResourceTypePitch2D;
resDesc.res.pitch2D.devPtr = dataDev;
resDesc.res.pitch2D.desc.x = 8;
resDesc.res.pitch2D.desc.y = 8;
resDesc.res.pitch2D.desc.f = cudaChannelFormatKindUnsigned;
resDesc.res.pitch2D.width = 4;
resDesc.res.pitch2D.height = 4;
resDesc.res.pitch2D.pitchInBytes = 0;
struct cudaTextureDesc texDesc;
memset(&texDesc, 0, sizeof(texDesc));
cudaCreateTextureObject(&tex, &resDesc, &texDesc, NULL);
cudaMemcpy(dataDev, &dataIn[0], 16, cudaMemcpyHostToDevice);
dim3 threads(4, 4);
kernel<<<1, threads>>>(tex);
cudaDeviceSynchronize();
return 0;
}
Результат ожидаю увидеть такой:
0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15
Но вместо этого получаю:
0,2,4,6,0,2,4,6,0,2,4,6,0,2,4,6
Подскажите, что я делаю не так?