File size: 12,518 Bytes
9116984
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
"""Attach modification notices and stage a platform release directory."""
import argparse
import json
import math
import os
import shutil
import struct
from pathlib import Path

from scripts.integrity import sha256, verify_files
from scripts.provenance import REVISION, model_identity, validate_model_structure, validate_roles
from scripts.quantize import DIFFUSERS_COMMIT
from scripts.report import load_records, summary_from

MODIFICATION = ('Modified by ixim / iximbox: eligible linear weights converted from '
                'Qwen-Image-2.1 to SDNQ UINT4 with SVD rank 32. Built with Qwen. '
                'Non-commercial research/evaluation under the accompanying Qwen Research License.')
GENERATION_CASES = ('portrait', 'english_text', 'chinese_text', 'composition', 'texture', 'rgba')
EDIT_SEEDS = (1000042, 1000123)
VRAM_CAP_BYTES = int(7.2 * 2**30)


def add_safetensors_notice(path):
    path = Path(path)
    temp = path.with_name(path.name + '.notice-tmp')
    try:
        with path.open('rb') as source:
            prefix = source.read(8)
            if len(prefix) != 8:
                raise ValueError(f'Invalid safetensors prefix: {path}')
            length = struct.unpack('<Q', prefix)[0]
            if length > 100_000_000:
                raise ValueError('Unexpected safetensors header size')
            header = json.loads(source.read(length))
            metadata = header.setdefault('__metadata__', {})
            if metadata.get('modification_notice') == MODIFICATION:
                return
            metadata.update(modification_notice=MODIFICATION, base_revision=REVISION)
            encoded = json.dumps(header, ensure_ascii=False, separators=(',', ':')).encode('utf-8')
            encoded += b' ' * ((8 - len(encoded) % 8) % 8)
            with temp.open('wb') as target:
                target.write(struct.pack('<Q', len(encoded)))
                target.write(encoded)
                shutil.copyfileobj(source, target, length=8 * 1024 * 1024)
        os.replace(temp, path)
    finally:
        if temp.exists():
            temp.unlink()


def annotate(root):
    root = Path(root)
    if validate_model_structure(root) != 'int4':
        raise ValueError('Only a complete INT4 pipeline can be annotated')
    for component in ('transformer', 'text_encoder'):
        for path in (root / component).glob('*.safetensors'):
            add_safetensors_notice(path)
        for path in (root / component).glob('*.json'):
            data = json.loads(path.read_text(encoding='utf-8'))
            if 'weight_map' in data:
                data.setdefault('metadata', {})['modification_notice'] = MODIFICATION
            else:
                data['_modification_notice'] = MODIFICATION
            path.write_text(json.dumps(data, indent=2) + '\n', encoding='utf-8')
    (root / 'CHANGES.md').write_text(
        '# Modifications\n\n' + MODIFICATION + '\n\n'
        '- Converted eligible transformer and text-encoder linear layers to SDNQ UINT4.\n'
        '- Stored a rank-32 SVD residual of the quantization error with the weights.\n'
        '- Left the requested sensitive projections, normalization, embeddings, vision tower, '
        'output head and VAE in floating point.\n'
        '- Did not use a calibration set or fine-tuning.\n'
        '- Quantized matmul is off so CUDA and Apple Silicon use the same eager dequantization.\n',
        encoding='utf-8')
    weights = [{'path': path.relative_to(root).as_posix(), 'size': path.stat().st_size, 'sha256': sha256(path)}
               for path in sorted(root.rglob('*.safetensors'))]
    document = {'base_model': 'Qwen/Qwen-Image-2.1', 'base_revision': REVISION,
                'diffusers_commit': DIFFUSERS_COMMIT, 'method': 'sdnq',
                'weights_dtype': 'uint4', 'use_svd': True, 'svd_rank': 32,
                'use_quantized_matmul': False, 'weight_files': weights}
    (root / 'conversion.json').write_text(json.dumps(document, indent=2), encoding='utf-8')


def _expected_keys():
    keys = {(name, seed) for name in GENERATION_CASES for seed in (42, 123)}
    keys |= {('edit', seed) for seed in EDIT_SEEDS}
    return keys


def validate_evaluation(evaluation, model, summary):
    evaluation = Path(evaluation)
    baseline = load_records(evaluation / 'bf16')
    candidate = load_records(evaluation / 'int4')
    expected = _expected_keys()
    if baseline.keys() != expected or candidate.keys() != expected:
        raise ValueError('Incomplete benchmark suite')
    baseline_env = json.loads((evaluation / 'bf16/environment.json').read_text(encoding='utf-8'))
    candidate_env = json.loads((evaluation / 'int4/environment.json').read_text(encoding='utf-8'))
    validate_roles(baseline_env['model_identity'], candidate_env['model_identity'])
    for key in ('benchmark_sha256', 'runtime_helper_sha256', 'device_helper_sha256'):
        if baseline_env.get(key) != candidate_env.get(key):
            raise ValueError(f'Benchmark implementation mismatch: {key}')
    identity = model_identity(model)
    if identity['fingerprint'] != candidate_env['model_identity']['fingerprint']:
        raise ValueError('Evaluated INT4 model differs from the staged model')
    for key in ('gpu', 'cuda', 'packages', 'offload', 'warmup', 'generator_device', 'cases_sha256'):
        if baseline_env[key] != candidate_env[key]:
            raise ValueError(f'Runtime mismatch: {key}')
    if not baseline_env['warmup'] or baseline_env['offload'] != 'model':
        raise ValueError('Release comparison requires warmup and model CPU offload')
    rows = []
    for key in baseline:
        left, right = baseline[key], candidate[key]
        from scripts.integrity import validate_pair
        validate_pair(left, right)
        if (left['steps'], left['width'], left['height'], left['cfg'], left['kv_cache']) != (40, 1024, 1024, 1.0, True):
            raise ValueError(f'Unexpected settings: {key}')
        if key[0] == 'edit':
            if left['source_seed'] != 42:
                raise ValueError('Editing record must name source seed 42')
            if left['input_sha256'] != baseline[('portrait', 42)]['image_sha256']:
                raise ValueError('Editing input is not the BF16 portrait at seed 42')
        rows.append({'case_id': key[0], 'seed': key[1],
                     'bf16_seconds': left['seconds'], 'int4_seconds': right['seconds'],
                     'bf16_peak_allocated_gib': left['peak_allocated_bytes'] / 2**30,
                     'int4_peak_allocated_gib': right['peak_allocated_bytes'] / 2**30})
    computed = summary_from(rows, baseline_env, candidate_env, next(iter(baseline.values())))
    if computed != summary:
        raise ValueError('Summary does not match the raw measurements')
    notes = (evaluation / 'qualitative.md').read_text(encoding='utf-8')
    if 'Pending visual inspection' in notes or len(notes) < 400:
        raise ValueError('qualitative.md still lacks an inspection of the outputs')
    for name in (*GENERATION_CASES, 'edit'):
        if name not in notes:
            raise ValueError(f'qualitative.md does not mention {name}')
    return identity


def validate_vram(evaluation, identity):
    folder = Path(evaluation) / 'vram8'
    env = json.loads((folder / 'environment.json').read_text(encoding='utf-8'))
    if env['model_identity']['fingerprint'] != identity['fingerprint']:
        raise ValueError('8GB proof used a different checkpoint')
    if env['offload'] != 'group' or not env.get('warmup'):
        raise ValueError('8GB proof must warm up and use group offload')
    cap = env.get('memory_cap') or {}
    if cap.get('cap_bytes') != VRAM_CAP_BYTES or not math.isclose(cap.get('cap_gib', 0), 7.2, abs_tol=1e-9):
        raise ValueError('8GB proof must cap the PyTorch allocator at 7.2 GiB')
    records = load_records(folder)
    row = records.get(('portrait', 42))
    if row is None or (row['width'], row['height'], row['steps'], row['offload']) != (1024, 1024, 40, 'group'):
        raise ValueError('8GB proof requires the 1024×1024, 40-step portrait at seed 42')
    if row['peak_allocated_bytes'] > VRAM_CAP_BYTES or row['peak_reserved_bytes'] > VRAM_CAP_BYTES:
        raise ValueError('8GB proof exceeded the allocator cap')
    return row


def stage(model, evaluation, output, platform):
    model, evaluation, output = Path(model), Path(evaluation), Path(output)
    if output.exists():
        raise FileExistsError(f'Refusing to overwrite release directory: {output}')
    summary = json.loads((evaluation / 'summary.json').read_text(encoding='utf-8'))
    if summary['pairs'] != 14 or summary['cases'] != 7 or summary['steps'] != 40 or not summary['warmup']:
        raise ValueError('Release requires the 14-pair, 40-step evaluation')
    identity = validate_evaluation(evaluation, model, summary)
    vram_row = validate_vram(evaluation, identity)
    conversion = json.loads((model / 'conversion.json').read_text(encoding='utf-8'))
    verify_files(model, conversion['weight_files'])
    for component in ('transformer', 'text_encoder'):
        quant = json.loads((model / component / 'config.json').read_text(encoding='utf-8'))['quantization_config']
        if str(quant.get('quant_method', '')).lower() != 'sdnq' or quant.get('weights_dtype') != 'uint4':
            raise ValueError(f'{component} is not saved as SDNQ UINT4')
        if quant.get('use_quantized_matmul'):
            raise ValueError(f'{component} enables a non-portable matmul kernel')
    shutil.copytree(model, output, ignore=shutil.ignore_patterns('.cache', '__pycache__', '*.lock'))
    shutil.copytree(evaluation, output / 'evaluation', ignore=shutil.ignore_patterns('__pycache__'))
    for name in ('scripts', 'benchmarks', 'cards', 'tests'):
        shutil.copytree(name, output / name, ignore=shutil.ignore_patterns('__pycache__', '*.pyc'))
    shutil.copy2('README.md', output / 'REPRODUCE.md')
    for name in ('requirements.txt', 'PUBLISHING.md'):
        shutil.copy2(name, output / name)
    shutil.copy2('artifacts/download-verification.json', output / 'upstream-verification.json')
    from scripts.model_card import render
    card = Path(f'cards/{platform}.md').read_text(encoding='utf-8')
    rendered = render(evaluation, platform, summary, vram_row)
    for key, value in rendered.items():
        card = card.replace('{{' + key + '}}', value)
    if '{{' in card:
        raise ValueError('Unrendered model card placeholder')
    (output / 'README.md').write_text(card, encoding='utf-8')
    if platform == 'modelscope':
        (output / 'configuration.json').write_text(json.dumps(
            {'framework': 'pytorch', 'task': 'text-to-image-synthesis'}, indent=2) + '\n', encoding='utf-8')
    manifest = [{'path': path.relative_to(output).as_posix(), 'size': path.stat().st_size, 'sha256': sha256(path)}
                for path in sorted(output.rglob('*')) if path.is_file()]
    (output / 'MANIFEST.json').write_text(json.dumps(manifest, indent=2), encoding='utf-8')
    print(f'Staged {platform}: {len(manifest)} files, {sum(item["size"] for item in manifest)/1e9:.3f} GB', flush=True)


def validate_release(root):
    root = Path(root)
    rows = json.loads((root / 'MANIFEST.json').read_text(encoding='utf-8'))
    expected = {row['path'] for row in rows} | {'MANIFEST.json'}
    actual = {path.relative_to(root).as_posix() for path in root.rglob('*') if path.is_file()}
    if actual != expected:
        raise ValueError(f'Unexpected or missing release files: {sorted(actual ^ expected)[:12]}')
    verify_files(root, rows)
    summary = json.loads((root / 'evaluation/summary.json').read_text(encoding='utf-8'))
    validate_evaluation(root / 'evaluation', root, summary)
    validate_vram(root / 'evaluation', model_identity(root))
    return rows


def main():
    parser = argparse.ArgumentParser()
    parser.add_argument('action', choices=['annotate', 'stage', 'check'])
    parser.add_argument('--model', default='models/int4')
    parser.add_argument('--evaluation', default='artifacts/eval')
    parser.add_argument('--output', default='release/huggingface')
    parser.add_argument('--platform', choices=['huggingface', 'modelscope'], default='huggingface')
    args = parser.parse_args()
    if args.action == 'annotate':
        annotate(args.model)
    elif args.action == 'stage':
        stage(args.model, args.evaluation, args.output, args.platform)
    else:
        print(f'Validated {len(validate_release(args.output))} files')


if __name__ == '__main__':
    main()