Download tinyquery/package_dataset.py from karmx/TinyQuery-140M: direct link, hf CLI and curl.
- Browser
- Download file 3 kB
-
https://huggingface.co/karmx/TinyQuery-140M/resolve/main/tinyquery/package_dataset.py
- Command line
-
hf download hf://karmx/TinyQuery-140M/tinyquery/package_dataset.py
-
curl -L -o package_dataset.py https://huggingface.co/karmx/TinyQuery-140M/resolve/main/tinyquery/package_dataset.py
3 kB
| """Build a Hugging Face viewer table plus lossless compressed source records.""" | |
| import argparse | |
| import gzip | |
| import hashlib | |
| import json | |
| from pathlib import Path | |
| import shutil | |
| import pyarrow as pa | |
| import pyarrow.parquet as pq | |
| def main(): | |
| p=argparse.ArgumentParser(); p.add_argument('--data',required=True);p.add_argument('--out',required=True) | |
| p.add_argument('--teacher-source'); args=p.parse_args(); source=Path(args.data); out=Path(args.out) | |
| (out/'data').mkdir(parents=True,exist_ok=True); (out/'raw').mkdir(exist_ok=True) | |
| counts={} | |
| for split in ['train','validation','test','manual']+(['development'] if (source/'development.jsonl').exists() else []): | |
| writer=None; batch=[]; count=0 | |
| with (source/(split+'.jsonl')).open() as stream, gzip.open(out/'raw'/(split+'.jsonl.gz'),'wt',encoding='utf-8') as raw: | |
| for line in stream: | |
| raw.write(line); row=json.loads(line) | |
| # Heterogeneous runtime JSON schemas remain lossless strings in Arrow. | |
| flat={key:row.get(key,'') for key in ['id','scenario_id','language','backend','operation','question','prompt','response','provenance']} | |
| flat['split']=split | |
| flat['sample_weight']=float(row.get('sample_weight',1)) | |
| for key in ['context','target','slots']: flat[key]=json.dumps(row[key],ensure_ascii=False,separators=(',',':')) | |
| batch.append(flat); count+=1 | |
| if len(batch)==2000: | |
| table=pa.Table.from_pylist(batch) | |
| if writer is None: writer=pq.ParquetWriter(out/'data'/(split+'.parquet'),table.schema,compression='zstd') | |
| writer.write_table(table); batch=[] | |
| if batch: | |
| table=pa.Table.from_pylist(batch) | |
| if writer is None: writer=pq.ParquetWriter(out/'data'/(split+'.parquet'),table.schema,compression='zstd') | |
| writer.write_table(table) | |
| if writer: writer.close() | |
| counts[split]=count | |
| (out/'provenance').mkdir(exist_ok=True) | |
| for name in ['tokenization.json','grounding-stats.json','split-audit.json','full-audit.json']: | |
| if (source/name).exists(): shutil.copy2(source/name,out/'provenance'/name) | |
| if args.teacher_source: | |
| teacher=Path(args.teacher_source) | |
| for name in ['templates.jsonl','verified-concrete.jsonl']: | |
| if (teacher/name).exists(): | |
| with (teacher/name).open('rb') as src,gzip.open(out/'provenance'/(name+'.gz'),'wb') as dest: shutil.copyfileobj(src,dest) | |
| manifest={'splits':counts,'files':{}} | |
| for file in sorted(out.rglob('*')): | |
| if file.is_file() and file.name!='manifest.json': | |
| digest=hashlib.sha256(file.read_bytes()).hexdigest() | |
| manifest['files'][str(file.relative_to(out))]={'bytes':file.stat().st_size,'sha256':digest} | |
| (out/'manifest.json').write_text(json.dumps(manifest,indent=2));print(json.dumps(counts)) | |
| if __name__=='__main__': main() | |