Artem Plastinkin commited on
Commit ·
8697afc
1
Parent(s): 7b5d384
Remove duplicate benchmark
Browse files
int8/benchmarks/x5h_mwmx_npu.yaml
DELETED
|
@@ -1,69 +0,0 @@
|
|
| 1 |
-
hardware:
|
| 2 |
-
vendor: renesas
|
| 3 |
-
chip: rcar-x5h
|
| 4 |
-
cpu: arm-cortex-a720
|
| 5 |
-
npu: npx6-48k
|
| 6 |
-
npu_count: 2
|
| 7 |
-
npu_cores: 12
|
| 8 |
-
npu_default_freq_mhz: 1066
|
| 9 |
-
accelerator:
|
| 10 |
-
- npu
|
| 11 |
-
|
| 12 |
-
runtime:
|
| 13 |
-
engine: mwmx # Renesas MWMX (Middleware MX) native inference runtime
|
| 14 |
-
toolchain_version: "MWMX SDK v4.35.0"
|
| 15 |
-
format: onnx # input model format (FP32 ONNX)
|
| 16 |
-
execution_provider: npu
|
| 17 |
-
execution_precision: int8 # FP32 model is auto-cast to INT8 by the MWMX toolchain
|
| 18 |
-
|
| 19 |
-
configuration:
|
| 20 |
-
npu_instances: 1 # single NPU
|
| 21 |
-
npu_cores_per_instance: 12 # single AI core
|
| 22 |
-
npu_freq_mhz: 850 # 850 MHz NPU clock frequency
|
| 23 |
-
|
| 24 |
-
benchmark:
|
| 25 |
-
type: hil # Hardware-in-the-loop — measured on physical X5H silicon
|
| 26 |
-
parameters:
|
| 27 |
-
batch_size: 1
|
| 28 |
-
input_resolution: [1, 3, 480, 640]
|
| 29 |
-
|
| 30 |
-
performance:
|
| 31 |
-
fps: null # throughput: 1000 / latency
|
| 32 |
-
latency: 7.30 # median inference latency (ms) over 1000 warm runs
|
| 33 |
-
|
| 34 |
-
metrics:
|
| 35 |
-
accuracy: null
|
| 36 |
-
top5_accuracy: null
|
| 37 |
-
|
| 38 |
-
memory:
|
| 39 |
-
peak_mb: null # peak NPU memory during inference
|
| 40 |
-
|
| 41 |
-
power:
|
| 42 |
-
avg_w: null # not yet characterized
|
| 43 |
-
|
| 44 |
-
# Same config (1 instance, 12 cores, 850 MHz) as x5h_mwmx_npu_apm80_12core.yaml
|
| 45 |
-
# — reusing its verified reproduce steps rather than re-deriving them.
|
| 46 |
-
reproduce:
|
| 47 |
-
steps:
|
| 48 |
-
- title: Activate the Python environment
|
| 49 |
-
command: >-
|
| 50 |
-
Activate the Python virtual environment that has the `hf` CLI
|
| 51 |
-
(huggingface_hub) and the NNAC toolchain installed, e.g.
|
| 52 |
-
`source nnac_venv/bin/activate` -- path depends on your toolchain install.
|
| 53 |
-
kind: note
|
| 54 |
-
- title: Download the ONNX model and compile config
|
| 55 |
-
command: hf download Renesas/RetinaNet-R101-ONNX --repo-type model --include "fp32/*" "compile_config/*" --local-dir ./RetinaNet-R101-ONNX-fp32
|
| 56 |
-
- title: Compile with the NNAC toolchain (INT8 auto-cast from the FP32 graph)
|
| 57 |
-
command: |
|
| 58 |
-
python3 nnac_frontend/legalize.py -d binary/nnx ./RetinaNet-R101-ONNX-fp32/fp32/retinanet-9.onnx --num-core 12 --network-config ./RetinaNet-R101-ONNX-fp32/compile_config/network_config.yaml
|
| 59 |
-
- title: Set up the R-Car X5H board
|
| 60 |
-
command: Configure the board per the AI Compiler (NNAC) "Getting Started" guide, section 3.4 (host TFTP/NFS setup, bootloader flashing, U-Boot, Linux boot, login) -- exact steps depend on your board/network setup.
|
| 61 |
-
kind: note
|
| 62 |
-
- title: Copy the compiled artifact to the board
|
| 63 |
-
command: Copy ${WORKDIR}/binary/nnx (the working directory from the download/compile steps above) to the board -- method may vary (NFS mount, scp, USB, etc.).
|
| 64 |
-
kind: note
|
| 65 |
-
- title: Run on R-Car X5H (single NPU cluster, 12 AI cores)
|
| 66 |
-
command: |
|
| 67 |
-
cd binary
|
| 68 |
-
./host_app ./arc_prog_npus ./nnx/retinanet-9
|
| 69 |
-
expected: hash[n] = 0x...(OK) means the run's output matches the reference hash in hash.txt; latency is the NPX execution time reported in cycles and ms.
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|