mirror of
https://github.com/microsoft/BitNet.git
synced 2026-08-23 17:36:31 +00:00
feat: add I2_S GGUF conversion for bitnet-b1.58-2B-4T and refactor T-MAC LUT path
- Add quantize_to_i2_s() for direct ternary-to-I2_S packing in conversion script - Support offline-quantized models (uint8 packed weights + weight_scale) - Fix weight_quant double-quantization bug for offline-quantized models - Fix I2_S scale computation to use first nonzero absolute value - Add I2_S ftype mapping and BitNetForCausalLM registration - Refactor ggml-bitnet-lut T-MAC wrapper with proper mul_mat implementation - Update llama.cpp submodule with I2_S ftype and 2B model type support
This commit is contained in:
File diff suppressed because it is too large
Load Diff
@@ -14,6 +14,8 @@ typedef float bitnet_float_type;
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
struct ggml_compute_params;
|
||||
|
||||
struct bitnet_tensor_extra {
|
||||
int lut_scales_size;
|
||||
int BK;
|
||||
@@ -33,6 +35,7 @@ GGML_API size_t ggml_bitnet_mul_mat_get_wsize(const struct ggml_tensor * src0, c
|
||||
GGML_API void ggml_bitnet_mul_mat_task_init(void * src1, void * qlut, void * lut_scales, void * lut_biases, int n, int k, int m, int bits);
|
||||
GGML_API void ggml_bitnet_mul_mat_task_compute(void * src0, void * scales, void * qlut, void * lut_scales, void * lut_biases, void * dst, int n, int k, int m, int bits);
|
||||
GGML_API void ggml_bitnet_transform_tensor(struct ggml_tensor * tensor);
|
||||
GGML_API void ggml_bitnet_mul_mat(const struct ggml_compute_params * params, struct ggml_tensor * dst);
|
||||
GGML_API int ggml_bitnet_get_type_bits(enum ggml_type type);
|
||||
GGML_API void ggml_bitnet_set_n_threads(int n_threads);
|
||||
#if defined(GGML_BITNET_ARM_TL1)
|
||||
|
||||
@@ -0,0 +1,21 @@
|
||||
[Kernels_0]
|
||||
m = 3200
|
||||
k = 8640
|
||||
bm = 160
|
||||
bk = 96
|
||||
bmm = 32
|
||||
|
||||
[Kernels_1]
|
||||
m = 3200
|
||||
k = 3200
|
||||
bm = 320
|
||||
bk = 96
|
||||
bmm = 32
|
||||
|
||||
[Kernels_2]
|
||||
m = 8640
|
||||
k = 3200
|
||||
bm = 320
|
||||
bk = 96
|
||||
bmm = 32
|
||||
|
||||
Reference in New Issue
Block a user