File size: 5,477 Bytes
9116984
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
aaae9b0
9116984
 
 
 
 
 
 
 
aaae9b0
9116984
 
 
 
 
 
aaae9b0
9116984
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
"""Fill the model card from measured records."""
import json
from pathlib import Path

from scripts.report import load_records


def _tr(platform, english, chinese):
    return chinese if platform == 'modelscope' else english


def _table(headers, rows):
    lines = ['| ' + ' | '.join(headers) + ' |', '|' + '|'.join('---' for _ in headers) + '|']
    lines.extend('| ' + ' | '.join(row) + ' |' for row in rows)
    return '\n'.join(lines)


def render(evaluation, platform, summary, vram_row):
    evaluation = Path(evaluation)
    baseline = load_records(evaluation / 'bf16')
    env = json.loads((evaluation / 'bf16/environment.json').read_text(encoding='utf-8'))
    vram_env = json.loads((evaluation / 'vram8/environment.json').read_text(encoding='utf-8'))
    notes = (evaluation / 'qualitative.md').read_text(encoding='utf-8').strip()
    pairs = sorted(baseline, key=lambda key: (key[0], key[1]))
    metric_rows = []
    comparison = {}
    import csv
    with (evaluation / 'comparison.csv').open(encoding='utf-8', newline='') as handle:
        for row in csv.DictReader(handle):
            comparison[(row['case_id'], int(row['seed']))] = row
    for case_id, seed in pairs:
        row = comparison[(case_id, seed)]
        metric_rows.append([case_id, str(seed), f'{float(row["bf16_seconds"]):.2f}',
                           f'{float(row["int4_seconds"]):.2f}',
                           f'{float(row["bf16_peak_allocated_gib"]):.2f}',
                           f'{float(row["int4_peak_allocated_gib"]):.2f}',
                           f'{float(row["rgb_mae_white"]):.4f}'])
    int4_records = load_records(evaluation / 'int4')
    samples = []
    for case_id, seed in pairs:
        left = f'evaluation/bf16/{baseline[(case_id, seed)]["image"]}'
        right = f'evaluation/int4/{int4_records[(case_id, seed)]["image"]}'
        prompt = baseline[(case_id, seed)]['prompt']
        samples.append('\n'.join([
            f'### {case_id} / seed {seed}', '',
            prompt, '',
            _table(['BF16', 'INT4'], [[f'![BF16]({left})', f'![INT4]({right})']]),
            '']))
    packages = '\n'.join(f'| {name} | {version} |' for name, version in env['packages'].items())
    cap_gib = vram_env['memory_cap']['cap_gib']
    details = '\n'.join([
        _tr(platform, '### Measured comparison', '### 实测对比'), '',
        _tr(platform,
            f'{summary["pairs"]} pairs, {summary["width"]}×{summary["height"]}, {summary["steps"]} steps, '
            f'CFG 1, KV cache on, model CPU offload, one excluded warmup. Host: {env["gpu"]}.',
            f'{summary["pairs"]} 对,{summary["width"]}×{summary["height"]},{summary["steps"]} 步,'
            f'CFG 1,开启 KV cache,model CPU offload,各有一次不计入的预热。机器:{env["gpu"]}。'),
        '',
        _table(['', 'BF16', 'INT4'], [
            [_tr(platform, 'Weight files (decimal GB)', '权重体积(十进制 GB)'),
             f'{summary["bf16_weight_bytes"]/1e9:.3f}', f'{summary["int4_weight_bytes"]/1e9:.3f}'],
            [_tr(platform, 'Mean call latency (s)', '平均调用耗时(秒)'),
             f'{summary["bf16_mean_seconds"]:.2f}', f'{summary["int4_mean_seconds"]:.2f}'],
            [_tr(platform, 'Maximum allocated CUDA memory (GiB)', '已分配 CUDA 显存最高值(GiB)'),
             f'{summary["bf16_max_allocated_gib"]:.2f}', f'{summary["int4_max_allocated_gib"]:.2f}'],
        ]),
        '',
        _table(['Case', 'Seed', 'BF16 s', 'INT4 s', 'BF16 GiB', 'INT4 GiB', 'RGB MAE'], metric_rows),
        '',
        _tr(platform, '### Package versions', '### 软件包版本'), '',
        '| Package | Version |', '|---|---|', packages, '',
        _tr(platform, '### Visual inspection', '### 目视检查'), '',
        notes, '',
        _tr(platform, '### CUDA allocator-cap measurement', '### CUDA 分配器限额测试'), '',
        _tr(platform,
            f'The saved INT4 checkpoint generated portrait seed 42 at 1024×1024 for 40 steps '
            f'while PyTorch\'s caching allocator was limited to {cap_gib} GiB on this '
            f'{vram_env["gpu"]}. Offload mode: {vram_env["offload"]}: one transformer block or one '
            f'text-encoder leaf on GPU, VAE encode and decode on CPU. '
            f'Allocated peak {vram_row["peak_allocated_bytes"]/2**30:.2f} GiB, '
            f'reserved peak {vram_row["peak_reserved_bytes"]/2**30:.2f} GiB, '
            f'call {vram_row["seconds"]:.2f}s after an excluded warmup. '
            'These measurements were recorded on the RTX 5090, not on the 8GB laptop described above.',
            f'在 {vram_env["gpu"]} 上将 PyTorch 缓存分配器限制为 {cap_gib} GiB 后,'
            f'已保存的 INT4 权重完成了 portrait、种子 42、1024×1024、40 步。'
            f'卸载方式:{vram_env["offload"]},Transformer 一次一个 block、文本编码器一次一个叶子层放在 GPU,VAE 的编码和解码在 CPU。'
            f'已分配峰值 {vram_row["peak_allocated_bytes"]/2**30:.2f} GiB,'
            f'保留峰值 {vram_row["peak_reserved_bytes"]/2**30:.2f} GiB,'
            f'预热之后的本次调用 {vram_row["seconds"]:.2f} 秒。'
            '这些测量值记录于 RTX 5090,并非来自上文提及的 8GB 显存笔记本。'),
        '',
        f'![8GB cap portrait](evaluation/vram8/{vram_row["image"]})',
        ''])
    return {'DETAILS': details, 'SAMPLES': '\n'.join(samples)}