Download dataloader/processing.py from hulehule/pllm2-full-dump: direct link, hf CLI and curl.
- Browser
- Download file 1.68 kB
-
https://huggingface.co/hulehule/pllm2-full-dump/resolve/main/dataloader/processing.py
- Command line
-
hf download hf://hulehule/pllm2-full-dump/dataloader/processing.py
-
curl -L -o processing.py https://huggingface.co/hulehule/pllm2-full-dump/resolve/main/dataloader/processing.py
1.68 kB
| import json | |
| import random | |
| import os | |
| def process_data(input_path, output_path, sample_ratio=0.1): | |
| """处理数据并只保存指定比例的数据""" | |
| print(f"处理文件: {input_path}") | |
| # 读取原始数据 | |
| with open(input_path, 'r') as f: | |
| data = [json.loads(line) for line in f if line.strip()] | |
| print(f"原始数据量: {len(data)} 条") | |
| # 随机采样10%的数据 | |
| sample_size = int(len(data) * sample_ratio) | |
| sampled_data = random.sample(data, sample_size) | |
| print(f"采样数据量: {len(sampled_data)} 条 ({sample_ratio*100}%)") | |
| # 保存处理后的数据 | |
| with open(output_path, 'w') as f: | |
| json.dump(sampled_data, f, ensure_ascii=False, indent=2) | |
| print(f"数据已保存到: {output_path}") | |
| print("-" * 50) | |
| # 设置随机种子确保结果可重现 | |
| random.seed(42) | |
| # 创建processed_data文件夹 | |
| processed_data_dir = '/opt/tiger/PLLM1.0/dataset/processed_data' | |
| if not os.path.exists(processed_data_dir): | |
| os.makedirs(processed_data_dir) | |
| print(f"✅ 创建文件夹: {processed_data_dir}") | |
| else: | |
| print(f"📁 文件夹已存在: {processed_data_dir}") | |
| # 处理训练数据 | |
| train_input_path = '/opt/tiger/PLLM1.0/dataset/abstract_train.json' | |
| train_output_path = '/opt/tiger/PLLM1.0/dataset/processed_data/abstract_train.json' | |
| process_data(train_input_path, train_output_path, sample_ratio=0.1) | |
| # 处理测试数据 | |
| test_input_path = '/opt/tiger/PLLM1.0/dataset/abstract_test.json' | |
| test_output_path = '/opt/tiger/PLLM1.0/dataset/processed_data/abstract_test.json' | |
| process_data(test_input_path, test_output_path, sample_ratio=0.1) | |
| print("✅ 数据处理完成!") | |