"""
MK3 Quantization Module

Comprehensive quantization and efficiency improvements for MK3.

Supported Methods:
1. QLoRA: 4-bit NF4 quantization with LoRA fine-tuning
2. LoRA: Low-Rank Adaptation for efficient fine-tuning
3. DoRA: Weight-Decomposed Low-Rank Adaptation
4. GPTQ: Post-training quantization with optimal weight rounding
5. AWQ: Activation-aware weight quantization
6. KV Cache Quantization: 8-bit and 4-bit KV cache compression

Memory Savings:
- QLoRA: ~75% reduction (4-bit weights + small LoRA adapters)
- LoRA: ~60-70% reduction (frozen weights + low-rank adapters)
- DoRA: ~65-75% reduction (improved LoRA variant)
- GPTQ: ~75% reduction (4-bit weights, no training overhead)
- AWQ: ~75% reduction (4-bit weights with better accuracy)
- KV Cache (8-bit): ~50% cache memory reduction
- KV Cache (4-bit): ~75% cache memory reduction
"""

from .config import QuantizationConfig, LoRAConfig, QLoRAConfig, GPTQConfig, AWQConfig
from .lora import LoRALayer, DoRALayer, apply_lora_to_model
from .qlora import QLoRALinear, apply_qlora_to_model
from .gptq import GPTQuantizer, quantize_model_gptq
from .awq import AWQQuantizer, quantize_model_awq
from .kv_cache_quant import QuantizedKVCache, KVCacheQuantizer
from .memory_utils import estimate_model_memory, get_quantization_memory_savings
from .quantized_model import QuantizedMK3Model

__all__ = [
    # Configuration
    'QuantizationConfig',
    'LoRAConfig',
    'QLoRAConfig',
    'GPTQConfig',
    'AWQConfig',

    # LoRA/DoRA
    'LoRALayer',
    'DoRALayer',
    'apply_lora_to_model',

    # QLoRA
    'QLoRALinear',
    'apply_qlora_to_model',

    # GPTQ
    'GPTQuantizer',
    'quantize_model_gptq',

    # AWQ
    'AWQQuantizer',
    'quantize_model_awq',

    # KV Cache
    'QuantizedKVCache',
    'KVCacheQuantizer',

    # Utilities
    'estimate_model_memory',
    'get_quantization_memory_savings',
    'QuantizedMK3Model',
]
