Download deployment/fetch_datasets.py from agenthinkmesh/project-manara-annotate: direct link, hf CLI and curl.
- Browser
- Download file 6.49 kB
-
https://huggingface.co/agenthinkmesh/project-manara-annotate/resolve/main/deployment/fetch_datasets.py
- Command line
-
hf download hf://agenthinkmesh/project-manara-annotate/deployment/fetch_datasets.py
-
curl -L -o fetch_datasets.py https://huggingface.co/agenthinkmesh/project-manara-annotate/resolve/main/deployment/fetch_datasets.py
6.49 kB
| """ | |
| Manara-Annotate: Arabic Dataset Fetching & Cataloging Script | |
| Phase 2 β Identify and fetch public Arabic/GCC dialect datasets | |
| """ | |
| import os | |
| import json | |
| import csv | |
| from pathlib import Path | |
| from datasets import load_dataset | |
| DATA_DIR = Path("/home/ubuntu/manara-annotate/data/raw") | |
| DATA_DIR.mkdir(parents=True, exist_ok=True) | |
| catalog = [] | |
| # βββββββββββββββββββββββββββββββββββββββββββββ | |
| # 1. ArSarcasm-v2 (Arabic Sarcasm Detection) | |
| # βββββββββββββββββββββββββββββββββββββββββββββ | |
| print("=" * 60) | |
| print("[1/3] Fetching ArSarcasm-v2 ...") | |
| try: | |
| ds_sarcasm = load_dataset("iabufarha/ar_sarcasm", trust_remote_code=True) | |
| train_split = ds_sarcasm["train"] | |
| test_split = ds_sarcasm["test"] | |
| # Save samples | |
| sarcasm_path = DATA_DIR / "arsarcasm_train.jsonl" | |
| with open(sarcasm_path, "w", encoding="utf-8") as f: | |
| for row in train_split: | |
| f.write(json.dumps(row, ensure_ascii=False) + "\n") | |
| sarcasm_test_path = DATA_DIR / "arsarcasm_test.jsonl" | |
| with open(sarcasm_test_path, "w", encoding="utf-8") as f: | |
| for row in test_split: | |
| f.write(json.dumps(row, ensure_ascii=False) + "\n") | |
| # Show sample | |
| print(f" Train samples : {len(train_split)}") | |
| print(f" Test samples : {len(test_split)}") | |
| print(f" Columns : {train_split.column_names}") | |
| print(f" Sample row : {train_split[0]}") | |
| catalog.append({ | |
| "name": "ArSarcasm-v1", | |
| "hf_id": "iabufarha/ar_sarcasm", | |
| "task": "Sarcasm Detection / Sentiment", | |
| "language": "Arabic (MSA + Dialectal)", | |
| "train_size": len(train_split), | |
| "test_size": len(test_split), | |
| "columns": train_split.column_names, | |
| "local_path": str(sarcasm_path), | |
| "relevance": "Primary seed for sarcasm + sentiment labeling tasks" | |
| }) | |
| print(" [OK] ArSarcasm saved.") | |
| except Exception as e: | |
| print(f" [ERROR] ArSarcasm: {e}") | |
| # βββββββββββββββββββββββββββββββββββββββββββββ | |
| # 2. QADI β Arabic Dialect Identification | |
| # βββββββββββββββββββββββββββββββββββββββββββββ | |
| print("\n[2/3] Fetching QADI (Arabic Dialect Identification) ...") | |
| try: | |
| ds_qadi = load_dataset("qcri/qadi", trust_remote_code=True) | |
| qadi_split = ds_qadi["train"] | |
| qadi_path = DATA_DIR / "qadi_train.jsonl" | |
| with open(qadi_path, "w", encoding="utf-8") as f: | |
| for row in qadi_split: | |
| f.write(json.dumps(row, ensure_ascii=False) + "\n") | |
| print(f" Train samples : {len(qadi_split)}") | |
| print(f" Columns : {qadi_split.column_names}") | |
| print(f" Sample row : {qadi_split[0]}") | |
| catalog.append({ | |
| "name": "QADI", | |
| "hf_id": "qcri/qadi", | |
| "task": "Dialect Identification (18 country-level Arabic dialects)", | |
| "language": "Arabic Dialectal (incl. Kuwaiti/Gulf)", | |
| "train_size": len(qadi_split), | |
| "test_size": 0, | |
| "columns": qadi_split.column_names, | |
| "local_path": str(qadi_path), | |
| "relevance": "Primary seed for Kuwaiti vs MSA dialect detection" | |
| }) | |
| print(" [OK] QADI saved.") | |
| except Exception as e: | |
| print(f" [ERROR] QADI: {e}") | |
| # Fallback: try alternative dataset ID | |
| try: | |
| ds_qadi = load_dataset("arbml/qadi", trust_remote_code=True) | |
| qadi_split = list(ds_qadi.values())[0] | |
| qadi_path = DATA_DIR / "qadi_train.jsonl" | |
| with open(qadi_path, "w", encoding="utf-8") as f: | |
| for row in qadi_split: | |
| f.write(json.dumps(row, ensure_ascii=False) + "\n") | |
| print(f" [OK] QADI (arbml) saved. Samples: {len(qadi_split)}") | |
| catalog.append({ | |
| "name": "QADI (arbml)", | |
| "hf_id": "arbml/qadi", | |
| "task": "Dialect Identification", | |
| "language": "Arabic Dialectal", | |
| "train_size": len(qadi_split), | |
| "test_size": 0, | |
| "columns": list(qadi_split.features.keys()) if hasattr(qadi_split, 'features') else [], | |
| "local_path": str(qadi_path), | |
| "relevance": "Dialect detection seed data" | |
| }) | |
| except Exception as e2: | |
| print(f" [ERROR] QADI fallback: {e2}") | |
| # βββββββββββββββββββββββββββββββββββββββββββββ | |
| # 3. Arabic Sentiment (ASTD / LABR / SemEval) | |
| # βββββββββββββββββββββββββββββββββββββββββββββ | |
| print("\n[3/3] Fetching Arabic Sentiment / NER datasets ...") | |
| try: | |
| ds_sent = load_dataset("arbml/ASTD", trust_remote_code=True) | |
| sent_split = list(ds_sent.values())[0] | |
| sent_path = DATA_DIR / "astd_sentiment.jsonl" | |
| with open(sent_path, "w", encoding="utf-8") as f: | |
| for row in sent_split: | |
| f.write(json.dumps(row, ensure_ascii=False) + "\n") | |
| print(f" Samples : {len(sent_split)}") | |
| print(f" Columns : {sent_split.column_names}") | |
| print(f" Sample : {sent_split[0]}") | |
| catalog.append({ | |
| "name": "ASTD (Arabic Sentiment Tweets Dataset)", | |
| "hf_id": "arbml/ASTD", | |
| "task": "Sentiment Analysis", | |
| "language": "Arabic (MSA + Egyptian/Gulf)", | |
| "train_size": len(sent_split), | |
| "test_size": 0, | |
| "columns": sent_split.column_names, | |
| "local_path": str(sent_path), | |
| "relevance": "Sentiment nuance labeling; Khaleeji sarcasm baseline" | |
| }) | |
| print(" [OK] ASTD saved.") | |
| except Exception as e: | |
| print(f" [ERROR] ASTD: {e}") | |
| # βββββββββββββββββββββββββββββββββββββββββββββ | |
| # Save catalog | |
| # βββββββββββββββββββββββββββββββββββββββββββββ | |
| catalog_path = DATA_DIR / "dataset_catalog.json" | |
| with open(catalog_path, "w", encoding="utf-8") as f: | |
| json.dump(catalog, f, ensure_ascii=False, indent=2) | |
| print("\n" + "=" * 60) | |
| print(f"Dataset catalog saved to: {catalog_path}") | |
| print(f"Total datasets cataloged: {len(catalog)}") | |
| for d in catalog: | |
| print(f" - {d['name']}: {d.get('train_size', '?')} train samples | Task: {d['task']}") | |