Buckets:
| static_assert(MATRIX_ROW_PADDING % CUDA_QUANTIZE_BLOCK_SIZE == 0, "Risk of out-of-bounds access."); | |
| static_assert(MATRIX_ROW_PADDING % (4*CUDA_QUANTIZE_BLOCK_SIZE_MMQ) == 0, "Risk of out-of-bounds access."); | |
| typedef void (*quantize_cuda_t)( | |
| const float * x, const int32_t * ids, void * vy, | |
| ggml_type type_src0, int64_t ne00, int64_t s01, int64_t s02, int64_t s03, | |
| int64_t ne0, int64_t ne1, int64_t ne2, int64_t ne3, cudaStream_t stream); | |
| void quantize_row_q8_1_cuda( | |
| const float * x, const int32_t * ids, void * vy, | |
| ggml_type type_src0, int64_t ne00, int64_t s01, int64_t s02, int64_t s03, | |
| int64_t ne0, int64_t ne1, int64_t ne2, int64_t ne3, cudaStream_t stream); | |
| void quantize_mmq_q8_1_cuda( | |
| const float * x, const int32_t * ids, void * vy, | |
| ggml_type type_src0, int64_t ne00, int64_t s01, int64_t s02, int64_t s03, | |
| int64_t ne0, int64_t ne1, int64_t ne2, int64_t ne3, cudaStream_t stream); | |
| void quantize_mmq_fp4_cuda(const float * x, | |
| const int32_t * ids, | |
| void * vy, | |
| ggml_type type_src0, | |
| int64_t ne00, | |
| int64_t s01, | |
| int64_t s02, | |
| int64_t s03, | |
| int64_t ne0, | |
| int64_t ne1, | |
| int64_t ne2, | |
| int64_t ne3, | |
| cudaStream_t stream); | |
Xet Storage Details
- Size:
- 1.79 kB
- Xet hash:
- b74b279c5bceeaf29e4634fdb0137dd1f3cb8e033ba07d75284da615dff20e47
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.