File size: 1,182 Bytes
cbf308e 3275441 cbf308e | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 | """Quantização W8A8 (Lema 3).
V5: módulos novos:
- SmoothQuantW8A8 (α aprendível via STE + sigmoid)
- TrustGuidedW8A8 (redução iterativa de erro, self-contained)
- MultiLayerW8A8Reducer (aplica a múltiplas camadas)
- apply_w8a8_to_model (bug fix crítico — substitui in-place)
"""
from .quantized_linear import (
QuantizedLinear,
quantize_tensor,
apply_w8a8,
apply_w8a8_to_model,
)
from .w8a8_smoothquant import (
SmoothQuantW8A8,
calibrate_smoothquant,
quantize_per_channel_symmetric,
quantize_per_token_symmetric,
STEQuantize,
ste_quantize,
)
from .w8a8_error_reduction import (
W8A8ConvergenceError,
analytic_trust,
W8A8ReductionResult,
TrustGuidedW8A8,
MultiLayerW8A8Reducer,
)
__all__ = [
# V1-V4
"QuantizedLinear",
"quantize_tensor",
"apply_w8a8",
# V5
"apply_w8a8_to_model",
"SmoothQuantW8A8",
"calibrate_smoothquant",
"quantize_per_channel_symmetric",
"quantize_per_token_symmetric",
"STEQuantize",
"ste_quantize",
"W8A8ConvergenceError",
"analytic_trust",
"W8A8ReductionResult",
"TrustGuidedW8A8",
"MultiLayerW8A8Reducer",
]
|