Instructions to use peter2000/laya-vulnerability-groups with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use peter2000/laya-vulnerability-groups with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-classification", model="peter2000/laya-vulnerability-groups")# pip install -U transformers accelerate # Load model directly from transformers import AutoModel model = AutoModel.from_pretrained("peter2000/laya-vulnerability-groups", device_map="auto") - Laya
How to use peter2000/laya-vulnerability-groups with Laya:
# No code snippets available yet for this library. # To use this model, check the repository files and the library's documentation. # Want to help? PRs adding snippets are welcome at: # https://github.com/huggingface/huggingface.js
- Notebooks
- Google Colab
- Kaggle
File size: 7,055 Bytes
fe8e23c | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 | import os
os.environ.setdefault("USE_TF", "0")
os.environ.setdefault("USE_TORCH", "1")
os.environ.setdefault("TOKENIZERS_PARALLELISM", "false")
os.environ.setdefault("HF_HUB_DISABLE_XET", "1")
import json
import time
import numpy as np
import pandas as pd
import torch
from huggingface_hub import HfApi, snapshot_download
from laya.agent import _fix_tokenizer_config
from sklearn.metrics import f1_score, hamming_loss, precision_score, recall_score
from sklearn.model_selection import train_test_split
from transformers import AutoTokenizer
import laya
BASE_MODEL_ID = "convaiinnovations/laya"
FT_REPO = "peter2000/laya-vulnerability-groups"
SETFIT_REPO = "peter2000/setfit-vulnerability-groups"
PARQUET_URL = "https://huggingface.co/datasets/GIZ/vulnerability_training_data_full/resolve/refs%2Fconvert%2Fparquet/default/train/0000.parquet"
LABELS = [
"Agricultural communities", "Coastal communities", "Ethnic, racial or other minorities",
"Fishery communities", "Informal sector workers", "Members of indigenous and local communities",
"Migrants and displaced persons", "Older persons", "Other", "Persons living in poverty",
"Persons with disabilities", "Persons with pre-existing health conditions",
"Residents of drought-prone regions", "Rural populations", "Sexual minorities (LGBTQI+)",
"Urban populations", "Women and other genders",
]
QIDS = [f"g{i}" for i in range(len(LABELS))]
QUESTIONS = {
qid: {
"type": "noul",
"instructions": f"Does this text indicate that {label} are targeted, supported, or affected as a vulnerable group? Answer true or false.",
}
for qid, label in zip(QIDS, LABELS)
}
def load_data():
df = pd.read_parquet(PARQUET_URL)
assert len(df) == 475, f"expected 475 rows, got {len(df)}"
Y = df[LABELS].values.astype(np.int64)
nlab = Y.sum(1)
idx_tr, idx_te = train_test_split(
np.arange(len(df)), test_size=0.2, random_state=42, stratify=np.minimum(nlab, 3)
)
return df, Y, np.asarray(idx_tr), np.asarray(idx_te)
def ece(conf, correct, n_bins=15):
conf = np.asarray(conf, dtype=np.float64)
corr = np.asarray(correct, dtype=np.float64)
bins = np.linspace(0.0, 1.0, n_bins + 1)
e = 0.0
for lo, hi in zip(bins[:-1], bins[1:]):
m = (conf > lo) & (conf <= hi)
if m.sum() > 0:
e += m.mean() * abs(corr[m].mean() - conf[m].mean())
return float(e)
def evaluate_full(Y_true, P_pred, threshold=0.5):
pred = (P_pred >= threshold).astype(int)
pl_f1 = f1_score(Y_true, pred, average=None, zero_division=0)
pl_prec = precision_score(Y_true, pred, average=None, zero_division=0)
pl_rec = recall_score(Y_true, pred, average=None, zero_division=0)
per_label_acc = (pred == Y_true).mean(axis=0)
conf = np.where(pred == 1, P_pred, 1.0 - P_pred)
corr = (pred == Y_true).astype(np.float64)
return {
"threshold": threshold,
"exact_match_accuracy": float(((pred == Y_true).all(axis=1)).mean()),
"hamming_accuracy": float(1.0 - hamming_loss(Y_true, pred)),
"hamming_loss": float(hamming_loss(Y_true, pred)),
"macro_f1": float(f1_score(Y_true, pred, average="macro", zero_division=0)),
"micro_f1": float(f1_score(Y_true, pred, average="micro", zero_division=0)),
"weighted_f1": float(f1_score(Y_true, pred, average="weighted", zero_division=0)),
"macro_precision": float(precision_score(Y_true, pred, average="macro", zero_division=0)),
"micro_precision": float(precision_score(Y_true, pred, average="micro", zero_division=0)),
"macro_recall": float(recall_score(Y_true, pred, average="macro", zero_division=0)),
"micro_recall": float(recall_score(Y_true, pred, average="micro", zero_division=0)),
"ece": ece(conf, corr),
"per_label_f1": {LABELS[i]: round(float(pl_f1[i]), 4) for i in range(len(LABELS))},
"per_label_precision": {LABELS[i]: round(float(pl_prec[i]), 4) for i in range(len(LABELS))},
"per_label_recall": {LABELS[i]: round(float(pl_rec[i]), 4) for i in range(len(LABELS))},
"per_label_accuracy": {LABELS[i]: round(float(per_label_acc[i]), 4) for i in range(len(LABELS))},
"test_positives": {LABELS[i]: int(Y_true[:, i].sum()) for i in range(len(LABELS))},
}
def probs_from_answers(res):
return np.array([res["answers"][qid]["noul"] for qid in QIDS], dtype=np.float64)
def eval_agent(agent, texts, Y_true):
t0 = time.time()
P = np.stack([probs_from_answers(agent.predict(t, QUESTIONS)) for t in texts])
m = evaluate_full(Y_true, P)
m["eval_seconds"] = round(time.time() - t0, 1)
m["probabilities"] = P.round(4).tolist()
return m
def main():
df, Y, idx_tr, idx_te = load_data()
texts = df["text"].tolist()
X_te = [texts[i] for i in idx_te]
Y_te = Y[idx_te]
print(f"test rows: {len(X_te)}, labels: {len(LABELS)}", flush=True)
device = "cuda"
results = {}
print("== laya fine-tuned ==", flush=True)
ft_dir = snapshot_download(FT_REPO, ignore_patterns=["*.py"])
agent = laya.load(ft_dir, device=device)
m = eval_agent(agent, X_te, Y_te)
results["laya_fine_tuned"] = m
print(json.dumps({k: v for k, v in m.items() if k != "probabilities"}, indent=2), flush=True)
del agent
torch.cuda.empty_cache()
print("== laya base zero-shot ==", flush=True)
base_dir = snapshot_download(BASE_MODEL_ID, ignore_patterns=["multilingual/*", "typed-decisions/*", "assets/*", "eval/*", "*.py"])
_fix_tokenizer_config(base_dir)
agent = laya.load(base_dir, device=device)
m = eval_agent(agent, X_te, Y_te)
results["laya_base_zero_shot"] = m
print(json.dumps({k: v for k, v in m.items() if k != "probabilities"}, indent=2), flush=True)
del agent
torch.cuda.empty_cache()
print("== setfit ==", flush=True)
from setfit import SetFitModel
sf = SetFitModel.from_pretrained(SETFIT_REPO)
t0 = time.time()
P = np.asarray(sf.predict_proba(X_te))
m = evaluate_full(Y_te, P)
m["eval_seconds"] = round(time.time() - t0, 1)
m["probabilities"] = P.round(4).tolist()
results["setfit"] = m
print(json.dumps({k: v for k, v in m.items() if k != "probabilities"}, indent=2), flush=True)
out = {
"dataset": "GIZ/vulnerability_training_data_full",
"split": "train_test_split(random_state=42, test_size=0.2, stratify=min(n_labels,3)); n_test=95",
"models": results,
}
api = HfApi(token=os.environ.get("HF_TOKEN"))
for repo in (FT_REPO, SETFIT_REPO):
api.upload_file(
path_or_fileobj=json.dumps(out, indent=2).encode(),
path_in_repo="metrics_full.json",
repo_id=repo,
repo_type="model",
commit_message="Full accuracy+F1 metrics: laya zero-shot, laya fine-tuned, setfit (95 test rows)",
)
print("uploaded metrics_full.json to", FT_REPO, "and", SETFIT_REPO)
print("DONE")
if __name__ == "__main__":
t0 = time.time()
main()
print(f"elapsed {time.time()-t0:.0f}s") |