Commit ·
4ca25fa
1
Parent(s): ceb6a91
QNN: re-export a16w8 with MinMaxObserver activations
Browse filesThe previous files used make_quantizer's MovingAverageMinMaxObserver default,
whose moving average trails the real activation range. Measured on an S26 Ultra
over 500 ImageNetV2 images: 70.00% top-1 before, 71.80% after, against 72.20%
for fp32. Quantization loss drops from 2.20 points to 0.40 at the same size.
Only the v81 file is device-validated; the other four are the same code path and
calibration compiled for Hexagon versions we have no hardware for.
qnn/config.json
CHANGED
|
@@ -34,7 +34,9 @@
|
|
| 34 |
}
|
| 35 |
]
|
| 36 |
}
|
| 37 |
-
}
|
|
|
|
|
|
|
| 38 |
},
|
| 39 |
{
|
| 40 |
"file": "efficientnet_v2_s_qnn_a16w8_v73.pte",
|
|
@@ -62,7 +64,9 @@
|
|
| 62 |
}
|
| 63 |
]
|
| 64 |
}
|
| 65 |
-
}
|
|
|
|
|
|
|
| 66 |
},
|
| 67 |
{
|
| 68 |
"file": "efficientnet_v2_s_qnn_a16w8_v75.pte",
|
|
@@ -90,7 +94,9 @@
|
|
| 90 |
}
|
| 91 |
]
|
| 92 |
}
|
| 93 |
-
}
|
|
|
|
|
|
|
| 94 |
},
|
| 95 |
{
|
| 96 |
"file": "efficientnet_v2_s_qnn_a16w8_v79.pte",
|
|
@@ -118,7 +124,9 @@
|
|
| 118 |
}
|
| 119 |
]
|
| 120 |
}
|
| 121 |
-
}
|
|
|
|
|
|
|
| 122 |
},
|
| 123 |
{
|
| 124 |
"file": "efficientnet_v2_s_qnn_a16w8_v81.pte",
|
|
@@ -146,7 +154,9 @@
|
|
| 146 |
}
|
| 147 |
]
|
| 148 |
}
|
| 149 |
-
}
|
|
|
|
|
|
|
| 150 |
}
|
| 151 |
]
|
| 152 |
}
|
|
|
|
| 34 |
}
|
| 35 |
]
|
| 36 |
}
|
| 37 |
+
},
|
| 38 |
+
"quantized": true,
|
| 39 |
+
"default": false
|
| 40 |
},
|
| 41 |
{
|
| 42 |
"file": "efficientnet_v2_s_qnn_a16w8_v73.pte",
|
|
|
|
| 64 |
}
|
| 65 |
]
|
| 66 |
}
|
| 67 |
+
},
|
| 68 |
+
"quantized": true,
|
| 69 |
+
"default": false
|
| 70 |
},
|
| 71 |
{
|
| 72 |
"file": "efficientnet_v2_s_qnn_a16w8_v75.pte",
|
|
|
|
| 94 |
}
|
| 95 |
]
|
| 96 |
}
|
| 97 |
+
},
|
| 98 |
+
"quantized": true,
|
| 99 |
+
"default": false
|
| 100 |
},
|
| 101 |
{
|
| 102 |
"file": "efficientnet_v2_s_qnn_a16w8_v79.pte",
|
|
|
|
| 124 |
}
|
| 125 |
]
|
| 126 |
}
|
| 127 |
+
},
|
| 128 |
+
"quantized": true,
|
| 129 |
+
"default": false
|
| 130 |
},
|
| 131 |
{
|
| 132 |
"file": "efficientnet_v2_s_qnn_a16w8_v81.pte",
|
|
|
|
| 154 |
}
|
| 155 |
]
|
| 156 |
}
|
| 157 |
+
},
|
| 158 |
+
"quantized": true,
|
| 159 |
+
"default": true
|
| 160 |
}
|
| 161 |
]
|
| 162 |
}
|
qnn/efficientnet_v2_s_qnn_a16w8_v69.pte
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 32839168
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:96552e464af2c400dc6e816be977afa3e81001ed8eb353e83d8e4b651be4f50c
|
| 3 |
size 32839168
|
qnn/efficientnet_v2_s_qnn_a16w8_v73.pte
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 25556480
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:e044a5f69bcbd7c438524d476288fe083371c05e48b9f973d40fe5c7e592e127
|
| 3 |
size 25556480
|
qnn/efficientnet_v2_s_qnn_a16w8_v75.pte
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
-
size
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:e0b1b7521626f263e9775b00a1cdd55a41ae08b4838678d2dd6305959f2d38f5
|
| 3 |
+
size 25552384
|
qnn/efficientnet_v2_s_qnn_a16w8_v79.pte
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
-
size
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:e476c80bf2ccbc69cac20f3ae889e814de9809300a8a59e7f93f169b34a3cec7
|
| 3 |
+
size 25642496
|
qnn/efficientnet_v2_s_qnn_a16w8_v81.pte
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 25712128
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:1ffb1d663490f59a5f14611a6c335dcc284eac7e6cf65ab47fb473b06fc08005
|
| 3 |
size 25712128
|