Download preference_datasets_topic/split.py from hulehule/pllm2-full-dump: direct link, hf CLI and curl.
- Browser
- Download file 883 Bytes
-
https://huggingface.co/hulehule/pllm2-full-dump/resolve/main/preference_datasets_topic/split.py
- Command line
-
hf download hf://hulehule/pllm2-full-dump/preference_datasets_topic/split.py
-
curl -L -o split.py https://huggingface.co/hulehule/pllm2-full-dump/resolve/main/preference_datasets_topic/split.py
883 Bytes
| #!/usr/bin/env python3 | |
| import json, os | |
| from collections import defaultdict | |
| BASE = "/opt/tiger/PLLM2.0/preference_datasets_topic" | |
| IN_FILES = [ | |
| "preference_topic_writing_0_4.json", | |
| "preference_topic_writing_5_9.json" | |
| ] | |
| def main(): | |
| buckets = defaultdict(list) | |
| for fname in IN_FILES: | |
| with open(f"{BASE}/{fname}", "r", encoding="utf-8") as f: | |
| data = json.load(f) | |
| for item in data: | |
| cid = int(item.get("cluster", -1)) | |
| if 0 <= cid <= 9: | |
| buckets[cid].append(item) | |
| # generate cluster_0 ~ cluster_9 | |
| for cid in range(10): | |
| out = f"{BASE}/preference_cluster_{cid}.json" | |
| with open(out, "w", encoding="utf-8") as f: | |
| json.dump(buckets[cid], f, ensure_ascii=False, indent=2) | |
| print(f"cluster {cid}: {len(buckets[cid])} → {out}") | |
| if __name__ == "__main__": | |
| main() | |