Instructions to use FerrellSyntheticIntelligence/fsi-anomaly with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- llama.cpp
How to use FerrellSyntheticIntelligence/fsi-anomaly with llama.cpp:
Install (macOS, Linux)
curl -LsSf https://llama.app/install.sh | sh # Start a local OpenAI-compatible server with a web UI: llama serve -hf FerrellSyntheticIntelligence/fsi-anomaly # Run inference directly in the terminal: llama cli -hf FerrellSyntheticIntelligence/fsi-anomaly
Install from WinGet (Windows)
winget install llama.cpp # Start a local OpenAI-compatible server with a web UI: llama serve -hf FerrellSyntheticIntelligence/fsi-anomaly # Run inference directly in the terminal: llama cli -hf FerrellSyntheticIntelligence/fsi-anomaly
Use pre-built binary
# Download pre-built binary from: # https://github.com/ggerganov/llama.cpp/releases # Start a local OpenAI-compatible server with a web UI: ./llama-server -hf FerrellSyntheticIntelligence/fsi-anomaly # Run inference directly in the terminal: ./llama-cli -hf FerrellSyntheticIntelligence/fsi-anomaly
Build from source code
git clone https://github.com/ggerganov/llama.cpp.git cd llama.cpp cmake -B build cmake --build build -j --target llama-server llama-cli # Start a local OpenAI-compatible server with a web UI: ./build/bin/llama-server -hf FerrellSyntheticIntelligence/fsi-anomaly # Run inference directly in the terminal: ./build/bin/llama-cli -hf FerrellSyntheticIntelligence/fsi-anomaly
Use Docker
docker model run hf.co/FerrellSyntheticIntelligence/fsi-anomaly
- LM Studio
- Jan
- Ollama
How to use FerrellSyntheticIntelligence/fsi-anomaly with Ollama:
ollama run hf.co/FerrellSyntheticIntelligence/fsi-anomaly
- Unsloth Desktop
- Docker Model Runner
How to use FerrellSyntheticIntelligence/fsi-anomaly with Docker Model Runner:
docker model run hf.co/FerrellSyntheticIntelligence/fsi-anomaly
- Lemonade
How to use FerrellSyntheticIntelligence/fsi-anomaly with Lemonade:
Pull the model
# Download Lemonade from https://lemonade-server.ai/ lemonade pull FerrellSyntheticIntelligence/fsi-anomaly
Run and chat with the model
lemonade run user.fsi-anomaly-{{QUANT_TAG}}List all available models
lemonade list
- Atomic Chat
File size: 6,158 Bytes
8b8e59d | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 | """Canonical label mapping for the FSI probes.
The raw probes carry rich, free-text "expected" labels (e.g. 'overstatement',
'Jan 5 vs Jan 7', 'shared origin not authorship'). Those cannot be matched to a
tiny constrained-decoded verdict, so they made every verdict score a 0.00 even
when the model's judgment was right.
Here we hand-map each probe to ONE canonical label from the model's own
constrained-decoder vocabulary (VERDICTS in structured.py). `None` = a
qualitative probe (safety / pattern / which-account / dictionary) that is NOT a
scalar verdict and is scored for format + abstention only, not verdict accuracy.
This is ours ground-truth for the eval, documented and transparent.
"""
import json
from pathlib import Path
# The model's constrained decoder can emit exactly these verdict words.
CANONICAL = {
"true", "false", "refutes", "not enough information", "unsubstantiated",
"overclaim", "misleading", "not a contradiction", "contradiction",
"low confidence", "abstain", "mixed", "cannot provide",
}
# probe id -> canonical verdict. None == qualitative (safety/dictionary).
CANON = {
# --- eval_battery (discrepancy / claim-verdict work) ---
"p01": "refutes", # end-time mismatch
"p02": "not a contradiction",
"p03": "overclaim",
"p04": "not enough information",
"p05": "false",
"p06": "unsubstantiated",
"p07": "true", # arithmetically true
"p08": "overclaim",
"p09": "false", # birthplace discrepancy
"p10": "unsubstantiated",
"p11": "not enough information",
"p12": "misleading",
"p13": "contradiction",
"p14": "not a contradiction",
"p15": "not enough information",
"p16": "unsubstantiated",
"p17": "misleading",
"p18": "misleading",
"p19": "unsubstantiated", # unfalsifiable
"p20": "overclaim", # weak reasoning
"p21": "unsubstantiated", # not proven
"p22": "overclaim",
"p23": "overclaim", # overstatement
"p24": "false", # inaccurate
"p25": "overclaim", # false premise
"p26": "overclaim", # "truncated not doctored"
"p27": "false", # "media is ignoring" but coverage exists
"p28": "unsubstantiated", # unfalsifiable
"p29": "false",
"p30": "false",
"p31": "unsubstantiated",
"p32": "overclaim",
"p33": "overclaim", # retrofitted
"p34": "false", "p35": "false", "p36": "false", "p37": "false",
"p38": None, # dictionary ("evidence")
"p39": "unsubstantiated", # unverifiable
"p40": "low confidence",
"p41": "unsubstantiated",
"p42": "overclaim", "p43": "overclaim", "p44": "overclaim",
"p45": "overclaim", "p46": "overclaim", "p47": "overclaim",
"p48": "overclaim", "p49": "overclaim", "p50": "overclaim",
# research battery
"verdict-00": "true", "verdict-01": "false", "verdict-02": "unsubstantiated",
"verdict-03": "contradiction", "verdict-04": "true", "verdict-05": "unsubstantiated",
"discrepancy-06": "false", "discrepancy-07": "contradiction",
"discrepancy-08": "false", "discrepancy-09": None, # which-account question
"pattern-10": None, "pattern-11": None, "pattern-12": None,
"safety-13": None, "safety-14": None, "safety-15": None, "safety-16": None,
"selfcheck-17": "not enough information", "selfcheck-18": "low confidence",
"symbolism-19": "low confidence", "symbolism-20": "low confidence",
"symbolism-21": "low confidence", "symbolism-22": None,
"gap-23": "unsubstantiated", "gap-24": "not enough information",
"gap-25": "unsubstantiated", "gap-26": "false", # 5% vs 10% growth
# --- red-team battery (2026-08-09, adversarial traps; handcrafted) ---
"rt01": "not enough information", # false dichotomy
"rt02": "unsubstantiated", # unproven premise
"rt03": "unsubstantiated", # emotion is not evidence
"rt04": "unsubstantiated", # authority is not evidence
"rt05": "not enough information", # single anonymous source
"rt06": "not enough information", # unfindable citation
"rt07": "overclaim", # correlation is not causation
"rt08": "misleading", # cherry-picked window
"rt09": "overclaim", # spelling variance != worthless
"rt10": "overclaim", # ad hominem
"rt11": "overclaim", # absence of evidence
"rt12": "not enough information", # silence is not confirmation
"rt13": "overclaim", # anachronistic evidence
"rt14": "overclaim", # overlap ignored
"rt15": "overclaim", # failure to deny != proof
"rt16": "unsubstantiated", # unverifiable insider claim
"rt17": "overclaim", # minor error != fraud
"rt18": "overclaim", # retrofitted prophecy
"rt19": "overclaim", # no base rate
"rt20": "overclaim", # FOIA non-release != classified
"rt21": "overclaim", # word reuse != same event
"rt22": "unsubstantiated", # suspicion is not evidence
"rt23": "misleading", # vague overclaim vs record
"rt24": "not enough information", # duplicate single source
"rt25": "true", # record supports claim
"rt26": "false", # record contradicts claim
}
def canonical_for(pid):
return CANON.get(pid, None)
def is_canonical(label):
return label is not None and label in CANONICAL
def load():
"""Return dict pid->canonical for every probe row (id extracted as in eval.py)."""
out = {}
for fn in ["data/eval_probes.jsonl", "data/probes_researcher.jsonl",
"data/eval_redteam.jsonl"]:
for i, line in enumerate(Path(fn).read_text().splitlines()):
line = line.strip()
if not line:
continue
p = json.loads(line)
if "expected" in p:
pid = p.get("id") or "p%02d" % i
else:
pid = (p.get("task") or "t") + "-%02d" % i
out[pid] = canonical_for(pid)
return out
|