File size: 1,182 Bytes
cbf308e
3275441
cbf308e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
"""Quantização W8A8 (Lema 3).

V5: módulos novos:
  - SmoothQuantW8A8 (α aprendível via STE + sigmoid)
  - TrustGuidedW8A8 (redução iterativa de erro, self-contained)
  - MultiLayerW8A8Reducer (aplica a múltiplas camadas)
  - apply_w8a8_to_model (bug fix crítico — substitui in-place)
"""
from .quantized_linear import (
    QuantizedLinear,
    quantize_tensor,
    apply_w8a8,
    apply_w8a8_to_model,
)
from .w8a8_smoothquant import (
    SmoothQuantW8A8,
    calibrate_smoothquant,
    quantize_per_channel_symmetric,
    quantize_per_token_symmetric,
    STEQuantize,
    ste_quantize,
)
from .w8a8_error_reduction import (
    W8A8ConvergenceError,
    analytic_trust,
    W8A8ReductionResult,
    TrustGuidedW8A8,
    MultiLayerW8A8Reducer,
)

__all__ = [
    # V1-V4
    "QuantizedLinear",
    "quantize_tensor",
    "apply_w8a8",
    # V5
    "apply_w8a8_to_model",
    "SmoothQuantW8A8",
    "calibrate_smoothquant",
    "quantize_per_channel_symmetric",
    "quantize_per_token_symmetric",
    "STEQuantize",
    "ste_quantize",
    "W8A8ConvergenceError",
    "analytic_trust",
    "W8A8ReductionResult",
    "TrustGuidedW8A8",
    "MultiLayerW8A8Reducer",
]