ExLlamaV3: Add kv cache quantization (#6903)

2026-04-05 06:35:15 +00:00 · 2025-04-25 21:32:00 -03:00 · 2025-04-25 21:32:00 -03:00 · d4017fbb6d
commit d4017fbb6d
parent d4b1e31c49
4 changed files with 32 additions and 3 deletions
--- a/modules/loaders.py
+++ b/modules/loaders.py
@ -13,6 +13,7 @@ loaders_and_params = OrderedDict({
        'cache_type',
        'tensor_split',
        'extra_flags',
+        'streaming_llm',
        'rope_freq_base',
        'compress_pos_emb',
        'flash_attn',
@ -49,6 +50,7 @@ loaders_and_params = OrderedDict({
    ],
    'ExLlamav3_HF': [
        'ctx_size',
+        'cache_type',
        'gpu_split',
        'cfg_cache',
        'trust_remote_code',