Artem Plastinkin commited on
Commit Β·
ba3dd21
1
Parent(s): 41edfa0
Update steps
Browse files
int8/benchmarks/x5h_mwmx_npu_apm80_12core.yaml
CHANGED
|
@@ -1,87 +1,86 @@
|
|
| 1 |
-
hardware:
|
| 2 |
-
vendor: renesas
|
| 3 |
-
chip: rcar-x5h
|
| 4 |
-
cpu: arm-cortex-a720
|
| 5 |
-
npu: npx6-48k
|
| 6 |
-
npu_count: 2
|
| 7 |
-
npu_cores: 12
|
| 8 |
-
npu_default_freq_mhz: 1066
|
| 9 |
-
accelerator:
|
| 10 |
-
- npu
|
| 11 |
-
|
| 12 |
-
runtime:
|
| 13 |
-
engine: mwmx
|
| 14 |
-
toolchain_version: "MWMX SDK v4.35.0"
|
| 15 |
-
format: onnx
|
| 16 |
-
execution_provider: npu
|
| 17 |
-
execution_precision: int8
|
| 18 |
-
|
| 19 |
-
configuration:
|
| 20 |
-
npu_instances: 1
|
| 21 |
-
npu_cores_per_instance: 12
|
| 22 |
-
npu_freq_mhz: 850
|
| 23 |
-
|
| 24 |
-
benchmark:
|
| 25 |
-
type: hil
|
| 26 |
-
parameters:
|
| 27 |
-
batch_size: 1
|
| 28 |
-
input_resolution: [1, 3, 512, 1024] # explicit in the source checkpoint name (512x1024, Cityscapes)
|
| 29 |
-
|
| 30 |
-
performance:
|
| 31 |
-
fps: null
|
| 32 |
-
latency: 38.103941 # APM80 pipeline run (metawaremx_runtime CI); this artifact is ONE segment
|
| 33 |
-
# of a 4-way split model ("custom_seg_split_4_split_2") β not full end-to-end latency.
|
| 34 |
-
# 1-core run for this artifact failed to compile β no figure available for that slice.
|
| 35 |
-
|
| 36 |
-
metrics:
|
| 37 |
-
accuracy: null
|
| 38 |
-
top5_accuracy: null
|
| 39 |
-
|
| 40 |
-
memory:
|
| 41 |
-
peak_mb: null
|
| 42 |
-
|
| 43 |
-
power:
|
| 44 |
-
avg_w: null
|
| 45 |
-
|
| 46 |
-
# Exact commands verified against the NNAC "Getting Started" chapter. Rendered
|
| 47 |
-
# by the AI-Dashboard in place of the generic placeholder flow β see
|
| 48 |
-
# downloadRunHTML() / parse_reproduce() in AI-Dashboard/app.js.
|
| 49 |
-
# CAVEAT: this artifact is segment "split_2" of a 4-way split network (see
|
| 50 |
-
# README) β these steps reproduce only this segment's latency, not an
|
| 51 |
-
# end-to-end DeepLabV3+ result. The 1-AI-core config also fails to compile
|
| 52 |
-
# for this artifact, so no 1-core reproduce block exists.
|
| 53 |
-
reproduce:
|
| 54 |
-
steps:
|
| 55 |
-
- title: Download the ONNX model and compile config
|
| 56 |
-
command: hf download Renesas/DeepLabV3Plus-R50-ONNX --repo-type=model --include "fp32/*" --include "compile_config/*"
|
| 57 |
-
- title: Configure the network for the 12-core build (required β this model has no working 1-core config)
|
| 58 |
-
command: |
|
| 59 |
-
cat >> ${WORKDIR}/nnac_config/config.yaml << 'EOF'
|
| 60 |
-
network_configuration:
|
| 61 |
-
deeplabv3plus_r50_oss_sim_inf:
|
| 62 |
-
optimization_options:
|
| 63 |
-
- -segmentation_json ./compile_config/l2_seg_48k_add_split_4_split_2.json
|
| 64 |
-
- -enable_dma_fusion
|
| 65 |
-
- -enable_stu_multi_channel
|
| 66 |
-
- -stu_multi_channel_config 0x15212420
|
| 67 |
-
- -stu_weight
|
| 68 |
-
- -distributed_coef_stu
|
| 69 |
-
EOF
|
| 70 |
-
# If config.yaml already has a network_configuration: key (e.g. from a
|
| 71 |
-
# previous model), merge this block into it instead of appending a duplicate key.
|
| 72 |
-
- title: Compile with the NNAC toolchain (INT8 auto-cast from the FP32 graph)
|
| 73 |
-
command: |
|
| 74 |
-
python3 nnac_frontend/legalize.py -d binary/nnx ./fp32/deeplabv3plus_r50_oss_sim_inf.onnx
|
| 75 |
-
- title: Set up the R-Car X5H board
|
| 76 |
-
command: Configure the board per the AI Compiler (NNAC) "Getting Started" guide, section 3.4 (host TFTP/NFS setup, bootloader flashing, U-Boot, Linux boot, login) -- exact steps depend on your board/network setup.
|
| 77 |
-
|
| 78 |
-
|
| 79 |
-
|
| 80 |
-
|
| 81 |
-
|
| 82 |
-
|
| 83 |
-
|
| 84 |
-
|
| 85 |
-
|
| 86 |
-
|
| 87 |
-
not whole-model latency.
|
|
|
|
| 1 |
+
hardware:
|
| 2 |
+
vendor: renesas
|
| 3 |
+
chip: rcar-x5h
|
| 4 |
+
cpu: arm-cortex-a720
|
| 5 |
+
npu: npx6-48k
|
| 6 |
+
npu_count: 2
|
| 7 |
+
npu_cores: 12
|
| 8 |
+
npu_default_freq_mhz: 1066
|
| 9 |
+
accelerator:
|
| 10 |
+
- npu
|
| 11 |
+
|
| 12 |
+
runtime:
|
| 13 |
+
engine: mwmx
|
| 14 |
+
toolchain_version: "MWMX SDK v4.35.0"
|
| 15 |
+
format: onnx
|
| 16 |
+
execution_provider: npu
|
| 17 |
+
execution_precision: int8
|
| 18 |
+
|
| 19 |
+
configuration:
|
| 20 |
+
npu_instances: 1
|
| 21 |
+
npu_cores_per_instance: 12
|
| 22 |
+
npu_freq_mhz: 850
|
| 23 |
+
|
| 24 |
+
benchmark:
|
| 25 |
+
type: hil
|
| 26 |
+
parameters:
|
| 27 |
+
batch_size: 1
|
| 28 |
+
input_resolution: [1, 3, 512, 1024] # explicit in the source checkpoint name (512x1024, Cityscapes)
|
| 29 |
+
|
| 30 |
+
performance:
|
| 31 |
+
fps: null
|
| 32 |
+
latency: 38.103941 # APM80 pipeline run (metawaremx_runtime CI); this artifact is ONE segment
|
| 33 |
+
# of a 4-way split model ("custom_seg_split_4_split_2") β not full end-to-end latency.
|
| 34 |
+
# 1-core run for this artifact failed to compile β no figure available for that slice.
|
| 35 |
+
|
| 36 |
+
metrics:
|
| 37 |
+
accuracy: null
|
| 38 |
+
top5_accuracy: null
|
| 39 |
+
|
| 40 |
+
memory:
|
| 41 |
+
peak_mb: null
|
| 42 |
+
|
| 43 |
+
power:
|
| 44 |
+
avg_w: null
|
| 45 |
+
|
| 46 |
+
# Exact commands verified against the NNAC "Getting Started" chapter. Rendered
|
| 47 |
+
# by the AI-Dashboard in place of the generic placeholder flow β see
|
| 48 |
+
# downloadRunHTML() / parse_reproduce() in AI-Dashboard/app.js.
|
| 49 |
+
# CAVEAT: this artifact is segment "split_2" of a 4-way split network (see
|
| 50 |
+
# README) β these steps reproduce only this segment's latency, not an
|
| 51 |
+
# end-to-end DeepLabV3+ result. The 1-AI-core config also fails to compile
|
| 52 |
+
# for this artifact, so no 1-core reproduce block exists.
|
| 53 |
+
reproduce:
|
| 54 |
+
steps:
|
| 55 |
+
- title: Download the ONNX model and compile config
|
| 56 |
+
command: hf download Renesas/DeepLabV3Plus-R50-ONNX --repo-type=model --include "fp32/*" --include "compile_config/*"
|
| 57 |
+
- title: Configure the network for the 12-core build (required β this model has no working 1-core config)
|
| 58 |
+
command: |
|
| 59 |
+
cat >> ${WORKDIR}/nnac_config/config.yaml << 'EOF'
|
| 60 |
+
network_configuration:
|
| 61 |
+
deeplabv3plus_r50_oss_sim_inf:
|
| 62 |
+
optimization_options:
|
| 63 |
+
- -segmentation_json ./compile_config/l2_seg_48k_add_split_4_split_2.json
|
| 64 |
+
- -enable_dma_fusion
|
| 65 |
+
- -enable_stu_multi_channel
|
| 66 |
+
- -stu_multi_channel_config 0x15212420
|
| 67 |
+
- -stu_weight
|
| 68 |
+
- -distributed_coef_stu
|
| 69 |
+
EOF
|
| 70 |
+
# If config.yaml already has a network_configuration: key (e.g. from a
|
| 71 |
+
# previous model), merge this block into it instead of appending a duplicate key.
|
| 72 |
+
- title: Compile with the NNAC toolchain (INT8 auto-cast from the FP32 graph)
|
| 73 |
+
command: |
|
| 74 |
+
python3 nnac_frontend/legalize.py -d binary/nnx ./fp32/deeplabv3plus_r50_oss_sim_inf.onnx
|
| 75 |
+
- title: Set up the R-Car X5H board
|
| 76 |
+
command: Configure the board per the AI Compiler (NNAC) "Getting Started" guide, section 3.4 (host TFTP/NFS setup, bootloader flashing, U-Boot, Linux boot, login) -- exact steps depend on your board/network setup.
|
| 77 |
+
kind: note
|
| 78 |
+
- title: Copy the compiled artifact to the board
|
| 79 |
+
command: Copy ${WORKDIR}/binary/nnx to the board -- method may vary (NFS mount, scp, USB, etc.).
|
| 80 |
+
kind: note
|
| 81 |
+
- title: Run on R-Car X5H (single NPU cluster, 12 AI cores)
|
| 82 |
+
command: |
|
| 83 |
+
cd binary
|
| 84 |
+
./host_app ./arc_prog_npus ./nnx/deeplabv3plus_r50_oss_sim_inf
|
| 85 |
+
expected: hash[n] = 0x...(OK) means the run's output matches the reference hash in hash.txt; latency is the NPX execution time reported in cycles and ms.
|
| 86 |
+
notes: This is one segment (split_2 of 4) of the full segmentation pipeline β not whole-model latency.
|
|
|