Spaces:
Running
Running
Download camelbert.js from Ghada-99-Ragab/Islamic4: direct link, hf CLI and curl.
- Browser
- Download file 3.27 kB
-
https://huggingface.co/spaces/Ghada-99-Ragab/Islamic4/resolve/main/camelbert.js
- Command line
-
hf download hf://spaces/Ghada-99-Ragab/Islamic4/camelbert.js
-
curl -L -o camelbert.js https://huggingface.co/spaces/Ghada-99-Ragab/Islamic4/resolve/main/camelbert.js
3.27 kB
| /* Detection back-ends for Subtask 1A. | |
| - "standard": rules + corpus scanner (runs in the Python worker). | |
| - "camelbert": fine-tuned CAMeLBERT-MSA token classifier (BIO tags: B/I-Ayah, B/I-Hadith, O). | |
| * With ICV_CONFIG.HF_MODEL_ID set, the Hugging Face Inference API is called. | |
| * Without it, a clearly labelled SIMULATION runs: no model weights are loaded; the spans come from the | |
| standard detector and are re-expressed as BIO tags with deterministic pseudo-confidences. */ | |
| (function () { | |
| const cfg = window.ICV_CONFIG; | |
| const sleep = (ms) => new Promise((r) => setTimeout(r, ms)); | |
| function bioTags(text, spans) { | |
| const tokens = []; const re = /\S+/g; let m; | |
| while ((m = re.exec(text))) { | |
| const s = spans.find((x) => m.index < x.end && m.index + m[0].length > x.start); | |
| let tag = "O"; | |
| if (s) tag = (m.index <= s.start || !tokens.length || !String(tokens[tokens.length - 1].tag).endsWith(s.label) ? "B-" : "I-") + s.label; | |
| tokens.push({ token: m[0], tag }); | |
| } | |
| return tokens; | |
| } | |
| /* The model is trained on diacritic-free text (see research/train_camelbert.py): strip before sending, map offsets back. */ | |
| const DIACRITICS = /[\u0640\u064B-\u065F\u0670\u06D6-\u06ED]/; | |
| function stripWithMap(text) { | |
| let out = ""; const map = []; | |
| for (let i = 0; i < text.length; i++) if (!DIACRITICS.test(text[i])) { out += text[i]; map.push(i); } | |
| return { out, map }; | |
| } | |
| async function viaHuggingFace(text) { | |
| const headers = { "Content-Type": "application/json" }; | |
| if (cfg.HF_TOKEN) headers.Authorization = "Bearer " + cfg.HF_TOKEN; | |
| const url = cfg.HF_ENDPOINT_URL || ("https://api-inference.huggingface.co/models/" + cfg.HF_MODEL_ID); | |
| const { out, map } = stripWithMap(text); | |
| const r = await fetch(url, { method: "POST", headers, | |
| body: JSON.stringify({ inputs: out, parameters: { aggregation_strategy: "simple" }, options: { wait_for_model: true } }) }); | |
| if (!r.ok) throw new Error("HF " + r.status); | |
| const groups = await r.json(); | |
| const spans = groups.filter((g) => g.end > g.start).map((g) => { | |
| let end = map[g.end - 1] + 1; while (end < text.length && DIACRITICS.test(text[end])) end++; | |
| return { label: /hadith/i.test(g.entity_group) ? "Hadith" : "Ayah", start: map[g.start], end, score: g.score }; | |
| }); | |
| return { spans, simulated: false }; | |
| } | |
| /* detectSpans: returns { spans, tags, simulated, note }. `standard` is a function returning the standard detector's spans. */ | |
| async function detectSpans(text, standard) { | |
| if (cfg.HF_MODEL_ID || cfg.HF_ENDPOINT_URL) { | |
| try { const out = await viaHuggingFace(text); out.tags = bioTags(text, out.spans); return out; } | |
| catch (e) { /* fall through to the simulation so a demo never breaks */ } | |
| } | |
| await sleep(500 + Math.min(700, text.length / 8)); // model "inference" latency, for a realistic demo | |
| const spans = await standard(); | |
| const conf = (s) => 0.9 + ((parseInt(window.icvCache.hash(text + s.start), 36) % 90) / 1000); | |
| const scored = spans.map((s) => ({ ...s, score: Math.round(conf(s) * 1000) / 1000 })); | |
| return { spans: scored, tags: bioTags(text, scored), simulated: true }; | |
| } | |
| window.icvDetectors = { detectSpans, bioTags }; | |
| })(); | |