Browse Source
skip quantizing per_layer_token_embd (#11207)
this tensor isn't compatible with cuda when quantized to q4_K so skip it
mxyng/quant
Michael Yang
1 year ago
committed by
GitHub
No known key found for this signature in database
GPG Key ID: B5690EEEBB952194
1 changed files with
2 additions and
0 deletions
-
server/quantization.go
|
|
|
@ -231,6 +231,8 @@ func newType(t *fsggml.Tensor, kv fsggml.KV, qs *quantizeState, ftype fsggml.Fil |
|
|
|
// do not quantize relative position bias (T5)
|
|
|
|
quantize = quantize && !strings.Contains(name, "attn_rel_b.weight") |
|
|
|
|
|
|
|
quantize = quantize && !strings.Contains(name, "per_layer_token_embd.weight") |
|
|
|
|
|
|
|
newType := fsggml.TensorType(t.Kind) |
|
|
|
if quantize { |
|
|
|
// get more optimal quantization type based on the tensor shape, layer, etc.
|
|
|
|
|