/* Detection back-ends for Subtask 1A. - "standard": rules + corpus scanner (runs in the Python worker). - "camelbert": fine-tuned CAMeLBERT-MSA token classifier (BIO tags: B/I-Ayah, B/I-Hadith, O). * With ICV_CONFIG.HF_MODEL_ID set, the Hugging Face Inference API is called. * Without it, a clearly labelled SIMULATION runs: no model weights are loaded; the spans come from the standard detector and are re-expressed as BIO tags with deterministic pseudo-confidences. */ (function () { const cfg = window.ICV_CONFIG; const sleep = (ms) => new Promise((r) => setTimeout(r, ms)); function bioTags(text, spans) { const tokens = []; const re = /\S+/g; let m; while ((m = re.exec(text))) { const s = spans.find((x) => m.index < x.end && m.index + m[0].length > x.start); let tag = "O"; if (s) tag = (m.index <= s.start || !tokens.length || !String(tokens[tokens.length - 1].tag).endsWith(s.label) ? "B-" : "I-") + s.label; tokens.push({ token: m[0], tag }); } return tokens; } /* The model is trained on diacritic-free text (see research/train_camelbert.py): strip before sending, map offsets back. */ const DIACRITICS = /[\u0640\u064B-\u065F\u0670\u06D6-\u06ED]/; function stripWithMap(text) { let out = ""; const map = []; for (let i = 0; i < text.length; i++) if (!DIACRITICS.test(text[i])) { out += text[i]; map.push(i); } return { out, map }; } async function viaHuggingFace(text) { const headers = { "Content-Type": "application/json" }; if (cfg.HF_TOKEN) headers.Authorization = "Bearer " + cfg.HF_TOKEN; const url = cfg.HF_ENDPOINT_URL || ("https://api-inference.huggingface.co/models/" + cfg.HF_MODEL_ID); const { out, map } = stripWithMap(text); const r = await fetch(url, { method: "POST", headers, body: JSON.stringify({ inputs: out, parameters: { aggregation_strategy: "simple" }, options: { wait_for_model: true } }) }); if (!r.ok) throw new Error("HF " + r.status); const groups = await r.json(); const spans = groups.filter((g) => g.end > g.start).map((g) => { let end = map[g.end - 1] + 1; while (end < text.length && DIACRITICS.test(text[end])) end++; return { label: /hadith/i.test(g.entity_group) ? "Hadith" : "Ayah", start: map[g.start], end, score: g.score }; }); return { spans, simulated: false }; } /* detectSpans: returns { spans, tags, simulated, note }. `standard` is a function returning the standard detector's spans. */ async function detectSpans(text, standard) { if (cfg.HF_MODEL_ID || cfg.HF_ENDPOINT_URL) { try { const out = await viaHuggingFace(text); out.tags = bioTags(text, out.spans); return out; } catch (e) { /* fall through to the simulation so a demo never breaks */ } } await sleep(500 + Math.min(700, text.length / 8)); // model "inference" latency, for a realistic demo const spans = await standard(); const conf = (s) => 0.9 + ((parseInt(window.icvCache.hash(text + s.start), 36) % 90) / 1000); const scored = spans.map((s) => ({ ...s, score: Math.round(conf(s) * 1000) / 1000 })); return { spans: scored, tags: bioTags(text, scored), simulated: true }; } window.icvDetectors = { detectSpans, bioTags }; })();