Instructions to use peter2000/laya-vulnerability-groups with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use peter2000/laya-vulnerability-groups with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-classification", model="peter2000/laya-vulnerability-groups")# Load model directly from transformers import AutoModel model = AutoModel.from_pretrained("peter2000/laya-vulnerability-groups", device_map="auto") - Notebooks
- Google Colab
- Kaggle
Download eval_full.py from peter2000/laya-vulnerability-groups: direct link, hf CLI and curl.
- Browser
- Download file 7.06 kB
-
https://huggingface.co/peter2000/laya-vulnerability-groups/resolve/main/eval_full.py
- Command line
-
hf download hf://peter2000/laya-vulnerability-groups/eval_full.py
-
curl -L -o eval_full.py https://huggingface.co/peter2000/laya-vulnerability-groups/resolve/main/eval_full.py
7.06 kB
| import os | |
| os.environ.setdefault("USE_TF", "0") | |
| os.environ.setdefault("USE_TORCH", "1") | |
| os.environ.setdefault("TOKENIZERS_PARALLELISM", "false") | |
| os.environ.setdefault("HF_HUB_DISABLE_XET", "1") | |
| import json | |
| import time | |
| import numpy as np | |
| import pandas as pd | |
| import torch | |
| from huggingface_hub import HfApi, snapshot_download | |
| from laya.agent import _fix_tokenizer_config | |
| from sklearn.metrics import f1_score, hamming_loss, precision_score, recall_score | |
| from sklearn.model_selection import train_test_split | |
| from transformers import AutoTokenizer | |
| import laya | |
| BASE_MODEL_ID = "convaiinnovations/laya" | |
| FT_REPO = "peter2000/laya-vulnerability-groups" | |
| SETFIT_REPO = "peter2000/setfit-vulnerability-groups" | |
| PARQUET_URL = "https://huggingface.co/datasets/GIZ/vulnerability_training_data_full/resolve/refs%2Fconvert%2Fparquet/default/train/0000.parquet" | |
| LABELS = [ | |
| "Agricultural communities", "Coastal communities", "Ethnic, racial or other minorities", | |
| "Fishery communities", "Informal sector workers", "Members of indigenous and local communities", | |
| "Migrants and displaced persons", "Older persons", "Other", "Persons living in poverty", | |
| "Persons with disabilities", "Persons with pre-existing health conditions", | |
| "Residents of drought-prone regions", "Rural populations", "Sexual minorities (LGBTQI+)", | |
| "Urban populations", "Women and other genders", | |
| ] | |
| QIDS = [f"g{i}" for i in range(len(LABELS))] | |
| QUESTIONS = { | |
| qid: { | |
| "type": "noul", | |
| "instructions": f"Does this text indicate that {label} are targeted, supported, or affected as a vulnerable group? Answer true or false.", | |
| } | |
| for qid, label in zip(QIDS, LABELS) | |
| } | |
| def load_data(): | |
| df = pd.read_parquet(PARQUET_URL) | |
| assert len(df) == 475, f"expected 475 rows, got {len(df)}" | |
| Y = df[LABELS].values.astype(np.int64) | |
| nlab = Y.sum(1) | |
| idx_tr, idx_te = train_test_split( | |
| np.arange(len(df)), test_size=0.2, random_state=42, stratify=np.minimum(nlab, 3) | |
| ) | |
| return df, Y, np.asarray(idx_tr), np.asarray(idx_te) | |
| def ece(conf, correct, n_bins=15): | |
| conf = np.asarray(conf, dtype=np.float64) | |
| corr = np.asarray(correct, dtype=np.float64) | |
| bins = np.linspace(0.0, 1.0, n_bins + 1) | |
| e = 0.0 | |
| for lo, hi in zip(bins[:-1], bins[1:]): | |
| m = (conf > lo) & (conf <= hi) | |
| if m.sum() > 0: | |
| e += m.mean() * abs(corr[m].mean() - conf[m].mean()) | |
| return float(e) | |
| def evaluate_full(Y_true, P_pred, threshold=0.5): | |
| pred = (P_pred >= threshold).astype(int) | |
| pl_f1 = f1_score(Y_true, pred, average=None, zero_division=0) | |
| pl_prec = precision_score(Y_true, pred, average=None, zero_division=0) | |
| pl_rec = recall_score(Y_true, pred, average=None, zero_division=0) | |
| per_label_acc = (pred == Y_true).mean(axis=0) | |
| conf = np.where(pred == 1, P_pred, 1.0 - P_pred) | |
| corr = (pred == Y_true).astype(np.float64) | |
| return { | |
| "threshold": threshold, | |
| "exact_match_accuracy": float(((pred == Y_true).all(axis=1)).mean()), | |
| "hamming_accuracy": float(1.0 - hamming_loss(Y_true, pred)), | |
| "hamming_loss": float(hamming_loss(Y_true, pred)), | |
| "macro_f1": float(f1_score(Y_true, pred, average="macro", zero_division=0)), | |
| "micro_f1": float(f1_score(Y_true, pred, average="micro", zero_division=0)), | |
| "weighted_f1": float(f1_score(Y_true, pred, average="weighted", zero_division=0)), | |
| "macro_precision": float(precision_score(Y_true, pred, average="macro", zero_division=0)), | |
| "micro_precision": float(precision_score(Y_true, pred, average="micro", zero_division=0)), | |
| "macro_recall": float(recall_score(Y_true, pred, average="macro", zero_division=0)), | |
| "micro_recall": float(recall_score(Y_true, pred, average="micro", zero_division=0)), | |
| "ece": ece(conf, corr), | |
| "per_label_f1": {LABELS[i]: round(float(pl_f1[i]), 4) for i in range(len(LABELS))}, | |
| "per_label_precision": {LABELS[i]: round(float(pl_prec[i]), 4) for i in range(len(LABELS))}, | |
| "per_label_recall": {LABELS[i]: round(float(pl_rec[i]), 4) for i in range(len(LABELS))}, | |
| "per_label_accuracy": {LABELS[i]: round(float(per_label_acc[i]), 4) for i in range(len(LABELS))}, | |
| "test_positives": {LABELS[i]: int(Y_true[:, i].sum()) for i in range(len(LABELS))}, | |
| } | |
| def probs_from_answers(res): | |
| return np.array([res["answers"][qid]["noul"] for qid in QIDS], dtype=np.float64) | |
| def eval_agent(agent, texts, Y_true): | |
| t0 = time.time() | |
| P = np.stack([probs_from_answers(agent.predict(t, QUESTIONS)) for t in texts]) | |
| m = evaluate_full(Y_true, P) | |
| m["eval_seconds"] = round(time.time() - t0, 1) | |
| m["probabilities"] = P.round(4).tolist() | |
| return m | |
| def main(): | |
| df, Y, idx_tr, idx_te = load_data() | |
| texts = df["text"].tolist() | |
| X_te = [texts[i] for i in idx_te] | |
| Y_te = Y[idx_te] | |
| print(f"test rows: {len(X_te)}, labels: {len(LABELS)}", flush=True) | |
| device = "cuda" | |
| results = {} | |
| print("== laya fine-tuned ==", flush=True) | |
| ft_dir = snapshot_download(FT_REPO, ignore_patterns=["*.py"]) | |
| agent = laya.load(ft_dir, device=device) | |
| m = eval_agent(agent, X_te, Y_te) | |
| results["laya_fine_tuned"] = m | |
| print(json.dumps({k: v for k, v in m.items() if k != "probabilities"}, indent=2), flush=True) | |
| del agent | |
| torch.cuda.empty_cache() | |
| print("== laya base zero-shot ==", flush=True) | |
| base_dir = snapshot_download(BASE_MODEL_ID, ignore_patterns=["multilingual/*", "typed-decisions/*", "assets/*", "eval/*", "*.py"]) | |
| _fix_tokenizer_config(base_dir) | |
| agent = laya.load(base_dir, device=device) | |
| m = eval_agent(agent, X_te, Y_te) | |
| results["laya_base_zero_shot"] = m | |
| print(json.dumps({k: v for k, v in m.items() if k != "probabilities"}, indent=2), flush=True) | |
| del agent | |
| torch.cuda.empty_cache() | |
| print("== setfit ==", flush=True) | |
| from setfit import SetFitModel | |
| sf = SetFitModel.from_pretrained(SETFIT_REPO) | |
| t0 = time.time() | |
| P = np.asarray(sf.predict_proba(X_te)) | |
| m = evaluate_full(Y_te, P) | |
| m["eval_seconds"] = round(time.time() - t0, 1) | |
| m["probabilities"] = P.round(4).tolist() | |
| results["setfit"] = m | |
| print(json.dumps({k: v for k, v in m.items() if k != "probabilities"}, indent=2), flush=True) | |
| out = { | |
| "dataset": "GIZ/vulnerability_training_data_full", | |
| "split": "train_test_split(random_state=42, test_size=0.2, stratify=min(n_labels,3)); n_test=95", | |
| "models": results, | |
| } | |
| api = HfApi(token=os.environ.get("HF_TOKEN")) | |
| for repo in (FT_REPO, SETFIT_REPO): | |
| api.upload_file( | |
| path_or_fileobj=json.dumps(out, indent=2).encode(), | |
| path_in_repo="metrics_full.json", | |
| repo_id=repo, | |
| repo_type="model", | |
| commit_message="Full accuracy+F1 metrics: laya zero-shot, laya fine-tuned, setfit (95 test rows)", | |
| ) | |
| print("uploaded metrics_full.json to", FT_REPO, "and", SETFIT_REPO) | |
| print("DONE") | |
| if __name__ == "__main__": | |
| t0 = time.time() | |
| main() | |
| print(f"elapsed {time.time()-t0:.0f}s") |