V5: W8A8 aprimorado (α aprendível via STE) + bug fix apply_w8a8_to_model + 4 hiperparâmetros auto-ajustáveis (T, τ, λ_ent, init_gate) + OOM-Killer fixes + num_layers_hyp=8 fixo
cbf308e verified | """Quantização W8A8 (Lema 3). | |
| V5: módulos novos: | |
| - SmoothQuantW8A8 (α aprendível via STE + sigmoid) | |
| - TrustGuidedW8A8 (redução iterativa de erro, self-contained) | |
| - MultiLayerW8A8Reducer (aplica a múltiplas camadas) | |
| - apply_w8a8_to_model (bug fix crítico — substitui in-place) | |
| """ | |
| from .quantized_linear import ( | |
| QuantizedLinear, | |
| quantize_tensor, | |
| apply_w8a8, | |
| apply_w8a8_to_model, | |
| ) | |
| from .w8a8_smoothquant import ( | |
| SmoothQuantW8A8, | |
| calibrate_smoothquant, | |
| quantize_per_channel_symmetric, | |
| quantize_per_token_symmetric, | |
| STEQuantize, | |
| ste_quantize, | |
| ) | |
| from .w8a8_error_reduction import ( | |
| W8A8ConvergenceError, | |
| analytic_trust, | |
| W8A8ReductionResult, | |
| TrustGuidedW8A8, | |
| MultiLayerW8A8Reducer, | |
| ) | |
| __all__ = [ | |
| # V1-V4 | |
| "QuantizedLinear", | |
| "quantize_tensor", | |
| "apply_w8a8", | |
| # V5 | |
| "apply_w8a8_to_model", | |
| "SmoothQuantW8A8", | |
| "calibrate_smoothquant", | |
| "quantize_per_channel_symmetric", | |
| "quantize_per_token_symmetric", | |
| "STEQuantize", | |
| "ste_quantize", | |
| "W8A8ConvergenceError", | |
| "analytic_trust", | |
| "W8A8ReductionResult", | |
| "TrustGuidedW8A8", | |
| "MultiLayerW8A8Reducer", | |
| ] | |