Download scripts/prepare_subset.py from Amitkumar001/Logo_watermark_detection: direct link, hf CLI and curl.
- Browser
- Download file 890 Bytes
-
https://huggingface.co/Amitkumar001/Logo_watermark_detection/resolve/main/scripts/prepare_subset.py
- Command line
-
hf download hf://Amitkumar001/Logo_watermark_detection/scripts/prepare_subset.py
-
curl -L -o prepare_subset.py https://huggingface.co/Amitkumar001/Logo_watermark_detection/resolve/main/scripts/prepare_subset.py
890 Bytes
| import os | |
| import shutil | |
| from pathlib import Path | |
| def prepare_subset(src_dir="dataset", dst_dir="dataset_subset", count=100): | |
| for split in ['train', 'val']: | |
| n = count if split == 'train' else count // 5 | |
| src_img = Path(src_dir) / "images" / split | |
| dst_img = Path(dst_dir) / "images" / split | |
| src_lbl = Path(src_dir) / "labels" / split | |
| dst_lbl = Path(dst_dir) / "labels" / split | |
| dst_img.mkdir(parents=True, exist_ok=True) | |
| dst_lbl.mkdir(parents=True, exist_ok=True) | |
| files = sorted(list(src_img.glob("*.jpg")))[:n] | |
| for f in files: | |
| shutil.copy(f, dst_img / f.name) | |
| lbl = src_lbl / (f.stem + ".txt") | |
| if lbl.exists(): | |
| shutil.copy(lbl, dst_lbl / lbl.name) | |
| print(f"Copied {count} images to {dst_dir}") | |
| if __name__ == "__main__": | |
| prepare_subset() | |