fix bug
This commit is contained in:
@@ -631,7 +631,7 @@ __global__ void VecQuant4TransposeMatMulHalfKernel(
|
|||||||
#if __CUDA_ARCH__ < 700 && __CUDA_ARCH__ > 600
|
#if __CUDA_ARCH__ < 700 && __CUDA_ARCH__ > 600
|
||||||
atomicAddHalf(&mul2[n_cols * height * 8 + n_rows], res);
|
atomicAddHalf(&mul2[n_cols * height * 8 + n_rows], res);
|
||||||
#else
|
#else
|
||||||
atomicAddHalf(&mul2[n_cols * height * 8 + n_rows], res);
|
atomicAdd(&mul2[n_cols * height * 8 + n_rows], res);
|
||||||
#endif
|
#endif
|
||||||
#endif
|
#endif
|
||||||
}
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user