Islamic4 / camelbert.js
Ghada-99-Ragab's picture
Upload 31 files
0390c03 verified
Raw History Blame Contribute Delete
3.27 kB
/* Detection back-ends for Subtask 1A.
- "standard": rules + corpus scanner (runs in the Python worker).
- "camelbert": fine-tuned CAMeLBERT-MSA token classifier (BIO tags: B/I-Ayah, B/I-Hadith, O).
* With ICV_CONFIG.HF_MODEL_ID set, the Hugging Face Inference API is called.
* Without it, a clearly labelled SIMULATION runs: no model weights are loaded; the spans come from the
standard detector and are re-expressed as BIO tags with deterministic pseudo-confidences. */
(function () {
const cfg = window.ICV_CONFIG;
const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
function bioTags(text, spans) {
const tokens = []; const re = /\S+/g; let m;
while ((m = re.exec(text))) {
const s = spans.find((x) => m.index < x.end && m.index + m[0].length > x.start);
let tag = "O";
if (s) tag = (m.index <= s.start || !tokens.length || !String(tokens[tokens.length - 1].tag).endsWith(s.label) ? "B-" : "I-") + s.label;
tokens.push({ token: m[0], tag });
}
return tokens;
}
/* The model is trained on diacritic-free text (see research/train_camelbert.py): strip before sending, map offsets back. */
const DIACRITICS = /[\u0640\u064B-\u065F\u0670\u06D6-\u06ED]/;
function stripWithMap(text) {
let out = ""; const map = [];
for (let i = 0; i < text.length; i++) if (!DIACRITICS.test(text[i])) { out += text[i]; map.push(i); }
return { out, map };
}
async function viaHuggingFace(text) {
const headers = { "Content-Type": "application/json" };
if (cfg.HF_TOKEN) headers.Authorization = "Bearer " + cfg.HF_TOKEN;
const url = cfg.HF_ENDPOINT_URL || ("https://api-inference.huggingface.co/models/" + cfg.HF_MODEL_ID);
const { out, map } = stripWithMap(text);
const r = await fetch(url, { method: "POST", headers,
body: JSON.stringify({ inputs: out, parameters: { aggregation_strategy: "simple" }, options: { wait_for_model: true } }) });
if (!r.ok) throw new Error("HF " + r.status);
const groups = await r.json();
const spans = groups.filter((g) => g.end > g.start).map((g) => {
let end = map[g.end - 1] + 1; while (end < text.length && DIACRITICS.test(text[end])) end++;
return { label: /hadith/i.test(g.entity_group) ? "Hadith" : "Ayah", start: map[g.start], end, score: g.score };
});
return { spans, simulated: false };
}
/* detectSpans: returns { spans, tags, simulated, note }. `standard` is a function returning the standard detector's spans. */
async function detectSpans(text, standard) {
if (cfg.HF_MODEL_ID || cfg.HF_ENDPOINT_URL) {
try { const out = await viaHuggingFace(text); out.tags = bioTags(text, out.spans); return out; }
catch (e) { /* fall through to the simulation so a demo never breaks */ }
}
await sleep(500 + Math.min(700, text.length / 8)); // model "inference" latency, for a realistic demo
const spans = await standard();
const conf = (s) => 0.9 + ((parseInt(window.icvCache.hash(text + s.start), 36) % 90) / 1000);
const scored = spans.map((s) => ({ ...s, score: Math.round(conf(s) * 1000) / 1000 }));
return { spans: scored, tags: bioTags(text, scored), simulated: true };
}
window.icvDetectors = { detectSpans, bioTags };
})();