ai-puzzle-maker / moderation.py
Codex
Modal content-safety gate + README/BLOG, demo & architecture
887e561
Raw
History Blame Contribute Delete
5.11 kB
"""
Content moderation client — NVIDIA Nemotron-3.5-Content-Safety, hosted on Modal.
The model (modal_safety.py) classifies the text into the Aegis-V2 taxonomy and
returns the categories it flags. We BLOCK only when a flagged category is in our
blocklist (Option B); everything else passes — so a puzzle of a knight with a
sword, a dragon's fiery battle, profanity, political themes, etc. don't reject
legit puzzle subjects.
What gets checked, BEFORE any artwork is generated: the player's typed puzzle
SUBJECT (the one line they want a jigsaw of).
FAIL-OPEN: any error/timeout/cold-start ALLOWS, so a moderation hiccup never
blocks puzzle generation. No platform gate (remote API) — works anywhere the
URL is set.
Config (Space secrets / env):
MODAL_SAFETY_URL deployed Modal endpoint
MODAL_SAFETY_TOKEN bearer token (same as the modal secret SAFETY_TOKEN)
SAFETY_BLOCK comma-separated categories to block (codes OR names);
overrides the default below — no Modal redeploy needed
MODERATION=0 kill switch
"""
import json
import os
import threading
import urllib.request
URL = os.environ.get("MODAL_SAFETY_URL", "").strip()
TOKEN = os.environ.get("MODAL_SAFETY_TOKEN", "").strip()
ENABLED = os.environ.get("MODERATION", "1") != "0" and bool(URL)
# Aegis-V2 taxonomy: code -> name (for S-code input + friendly block messages).
TAXONOMY = {
"S1": "Violence", "S2": "Sexual", "S3": "Criminal Planning/Confessions",
"S4": "Guns and Illegal Weapons", "S5": "Controlled/Regulated Substances",
"S6": "Suicide and Self Harm", "S7": "Sexual (minor)", "S8": "Hate/Identity Hate",
"S9": "PII/Privacy", "S10": "Harassment", "S11": "Threat", "S12": "Profanity",
"S13": "Needs Caution", "S14": "Other", "S15": "Manipulation",
"S16": "Fraud/Deception", "S17": "Malware", "S18": "High Risk Gov Decision Making",
"S19": "Political/Misinformation/Conspiracy", "S20": "Copyright/Trademark/Plagiarism",
"S21": "Unauthorized Advice", "S22": "Illegal Activity", "S23": "Immoral/Unethical",
"S24": "Economic Harm",
}
# Categories that BLOCK (Option B). Default = serious real-world harm; fictional
# violence/weapons (S1/S4), profanity, political, copyright, etc. are ALLOWED so
# storybook scenes (a knight, a pirate battle, a dragon) play fine.
# Override with SAFETY_BLOCK="S2,S6,S7,..." (codes) or full names.
DEFAULT_BLOCK = "S2,S3,S5,S6,S7,S8,S9,S10,S11,S16,S17,S22"
def _norm(s):
return "".join(ch for ch in str(s).lower() if ch.isalnum())
def _block_set():
raw = os.environ.get("SAFETY_BLOCK", "").strip() or DEFAULT_BLOCK
out = set()
for tok in raw.split(","):
tok = tok.strip()
if tok:
out.add(_norm(TAXONOMY.get(tok.upper(), tok))) # S-code -> name; names pass through
return out
_BLOCK = _block_set()
print(
f"moderation: ON — Nemotron-3.5-Content-Safety via Modal; blocking {len(_BLOCK)} categories"
if ENABLED else "moderation: OFF (set MODAL_SAFETY_URL to enable)"
)
def check(text, timeout=10):
"""-> (safe: bool, categories: str). Fail-open on any error.
Blocks only when the model flags a category in the blocklist; other flagged
categories (fictional violence, profanity, ...) are allowed through."""
if not ENABLED or not str(text).strip():
return True, ""
try:
body = json.dumps({"text": str(text)[:1500]}).encode()
headers = {"Content-Type": "application/json"}
if TOKEN:
headers["Authorization"] = "Bearer " + TOKEN
req = urllib.request.Request(URL, data=body, headers=headers)
with urllib.request.urlopen(req, timeout=timeout) as r:
out = json.loads(r.read())
except Exception as e:
print(f"moderation check failed (allowing): {e}")
return True, ""
overall_safe = bool(out.get("safe", True))
cats = out.get("categories", []) or []
if isinstance(cats, str): # tolerate a plain string
cats = [c.strip() for c in cats.split(",") if c.strip()]
hits = []
for c in cats:
name = TAXONOMY.get(str(c).strip().upper(), str(c).strip()) # S-code -> name
if _norm(name) in _BLOCK:
hits.append(name)
if hits:
return False, ", ".join(dict.fromkeys(hits)) # dedup, keep order
if not cats and not overall_safe:
# flagged unsafe but no parseable category -> be conservative
return False, "policy violation"
return True, ""
def prewarm():
"""Warm the Modal container in a background thread at startup so the first real
check is fast (Modal scales to zero between bursts). No-op if disabled; fail-safe."""
if not ENABLED:
return
def _run():
try:
check("a cozy log cabin in snowy mountains at dusk", timeout=120) # absorbs the cold start
print("moderation: Modal safety endpoint warmed; gate ready")
except Exception as e:
print(f"moderation prewarm failed (will retry lazily): {e}")
threading.Thread(target=_run, daemon=True).start()