#!/usr/bin/env python3 """Validate current BPW inventory, arithmetic, HF matches, and newly added GGUFs.""" from pathlib import Path import hashlib, json, math, os, re, sys from datetime import datetime, timezone ROOT=Path(__file__).resolve().parent sys.path.insert(0,str(Path.home()/'projects/llama.cpp/gguf-py')) from gguf import GGUFReader, GGMLQuantizationType, GGML_QUANT_SIZES native=json.loads((ROOT/'native-bpw.json').read_text()) hf=json.loads((ROOT/'hf/hf-bpw-estimates.json').read_text()) lookup={tuple(sorted(r['local_paths'])):r for r in hf} files={p['path']:p for p in native['files']} assert len(lookup)==len(hf)==len(native['rows']) count=0 for item in native['files']: stat=Path(item['path']).stat() assert (stat.st_size,stat.st_mtime_ns)==(item['file_bytes'],item['mtime_ns']) for t in item['tensors']: block,size=GGML_QUANT_SIZES[GGMLQuantizationType[t['type'].upper()]] n=math.prod(t['shape']) assert t['shape'][0]%block==0 and n==t['elements'] and n//block*size==t['bytes'],t['name'] count+=1 for row in native['rows']: h=lookup[tuple(sorted(row['files']))] assert h['status']=='comparable' and h['hf_sibling_bytes']==row['file_bytes'] and h['hf_gguf_total']==row['llama_cpp_parameters'] tensors=[t for p in row['files'] for t in files[p]['tensors']] assert len({t['name'] for t in tensors})==len(tensors)==row['tensor_count'] excluded=set(row['embedded_mtp_tensors']) main=[t for t in tensors if t['name'] not in excluded] for group,paramkey,byteskey,bpwkey in [(tensors,'llama_cpp_parameters','llama_cpp_tensor_bytes','llama_cpp_bpw'),(main,'main_parameters','main_tensor_bytes','main_bpw')]: params=sum(t['elements'] for t in group); size=sum(t['bytes'] for t in group) assert params==row[paramkey] and size==row[byteskey] and size*8/params==row[bpwkey] assert not re.search(r'(?:^|[-_.])Q3_K_[MLS](?=[-_.]|$)',Path(row['model']).name,re.I) shapes={} for family in ('qwen3.5-4b','qwen3.6-35b-a3b'): variants=[] for row in [r for r in native['rows'] if r['family']==family]: excluded=set(row['embedded_mtp_tensors']) shape=sorted((t['name'],t['shape']) for p in row['files'] for t in files[p]['tensors'] if t['name'] not in excluded) digest=hashlib.sha256(json.dumps(shape,separators=(',',':')).encode()).hexdigest() variants.append({'model':row['model'],'tensor_count':len(shape),'main_tensor_names_shapes_sha256':digest}) assert len({v['main_tensor_names_shapes_sha256'] for v in variants})==1 shapes[family]=variants (ROOT/'main-tensor-shape-validation.json').write_text(json.dumps({'status':'passed','families':shapes},indent=2)+'\n') actual=set() for folder,dirs,names in os.walk(native['provenance']['models_dir']): dirs[:]=[d for d in dirs if not d.startswith('.')] actual.update(str(Path(folder)/name) for name in names if Path(name).suffix.lower()=='.gguf') assert actual==set(files)|{p['path'] for p in native['excluded']} newchecks=[] # Optional positional paths select files for the independent Python reader check. for filename in sorted({str(Path(arg).expanduser().resolve()) for arg in sys.argv[1:]}): assert filename in files, f"File not in current scan: {filename}" reader=GGUFReader(filename) expected=files[filename]['tensors'] actualt={t.name:(int(t.n_elements),int(t.n_bytes)) for t in reader.tensors} assert actualt=={t['name']:(t['elements'],t['bytes']) for t in expected} newchecks.append({'file':filename,'tensor_count':len(expected),'exact_match':True}) del reader print('GGUFReader verified new file:',filename,flush=True) result={'status':'passed','updated_utc':datetime.now(timezone.utc).isoformat(),'variants':len(native['rows']), 'families':len({r['family'] for r in native['rows']}),'main_files':len(files),'excluded_files':len(native['excluded']), 'visible_files_accounted_for':len(actual),'tensor_descriptors_independently_recalculated':count, 'native_scan_errors':native['errors'],'hf_pairing_and_bpw':'all matched','removed_Q3_K_M_L_S_absent':True, 'main_tensor_shape_check':'both Qwen families passed','added_files_independent_GGUFReader':newchecks} (ROOT/'refresh-validation.json').write_text(json.dumps(result,indent=2)+'\n') print(json.dumps({k:v for k,v in result.items() if k!='added_files_independent_GGUFReader'},indent=2))