"""The Understudy — fine-tuned MiniCPM4-0.5B (GGUF) on llama.cpp, CPU. Powers Tiny Mode and instant slider-drag previews: zero GPU, single-shot. With ACE-Step 2B turbo the whole Tiny Mode pipeline stays ~2.5B params. - fine-tuned & published on the Hub -> Well-Tuned - served as Q4_K_M GGUF via llama.cpp -> Llama Champion - 0.5B lyricist + 2B turbo music -> Tiny Mode The base model is ChatML (<|im_start|>/<|im_end|>, eos <|im_end|>). We feed messages through create_chat_completion, which applies the GGUF's embedded ChatML template AND tokenizes <|im_start|>/<|im_end|> as the real special tokens the fine-tune was trained on. (A hand-built prompt string passed to plain completion mis-tokenizes those markers as literal text — verified to produce degenerate output — so don't go back to that.) """ import os import threading from .prompts import UNDERSTUDY_SYSTEM, build_messages REPO = "tanya8997/openwork-understudy-0.5b" GGUF_FILE = "understudy-Q4_K_M.gguf" # generation_config.json on the Hub: temperature/top_p 0.8, eos <|im_end|> # bumped a touch so the 0.5B stops falling back on its memorised calibration anchors # ("I turn messy signals into decisions…") and actually writes from the resume _TEMPERATURE = 0.95 _TOP_P = 0.92 # a 0.5B loves to loop a hook ("I'm the job candidate" x12) — penalise repeats _REPEAT_PENALTY = 1.3 # presence_penalty nudges it toward fresh wording instead of the stock examples _PRESENCE_PENALTY = 0.4 _STOP = ["<|im_end|>", "<|endoftext|>", ""] _llm = None _lock = threading.Lock() _gguf_path: str | None = None load_error: Exception | None = None # Prefetch the GGUF at import so the first preview is instant on the Space. # OTW_SKIP_PREFETCH lets local dev / tests import without the ~400MB download. if os.environ.get("OTW_SKIP_PREFETCH"): load_error = RuntimeError("prefetch skipped (OTW_SKIP_PREFETCH)") else: try: from huggingface_hub import hf_hub_download _gguf_path = hf_hub_download(REPO, GGUF_FILE) print(f"[understudy] gguf ready: {_gguf_path}") except Exception as e: # offline / no hub — Tiny Mode falls back to stubs load_error = e def _messages(resume_text: str, job_description: str, genre: str, level: int, zone_desc: str, voice: str | None = None) -> list[dict]: """The condensed Understudy system prompt + the same production user payload the fine-tune was trained on (build_messages' user turn).""" user = build_messages(resume_text, job_description, genre, level, zone_desc, voice)[1][ "content" ] return [ {"role": "system", "content": UNDERSTUDY_SYSTEM}, {"role": "user", "content": user}, ] def _get_llm(): """Lazy-load the llama.cpp model once (CPU). Raises if unavailable.""" global _llm if _llm is not None: return _llm if load_error is not None or not _gguf_path: raise RuntimeError(f"understudy unavailable: {load_error!r}") with _lock: if _llm is None: from llama_cpp import Llama _llm = Llama( model_path=_gguf_path, n_ctx=2048, n_threads=os.cpu_count() or 4, verbose=False, ) return _llm # The 0.5B memorised the calibration anchors (it was fine-tuned on gpt-oss data that # few-shot those exact lines), so for data-scientist resumes at levels 1/5/10 it parrots # them instead of writing from the resume. Detect that and regenerate hotter. _ANCHOR_FRAGMENTS = ( "messy signals into decisions", "models with precision", "dream in sql and dashboards", "churn rate fears me", "split-test feelings in the dark", "regression has a hiring arc", ) def _echoes_anchor(text: str) -> bool: t = text.lower() return any(frag in t for frag in _ANCHOR_FRAGMENTS) def write(resume_text: str, job_description: str, genre: str, level: int, zone_desc: str, max_tokens: int = 512, voice: str | None = None) -> str: """Single-shot lyric generation on the CPU Understudy. Returns the raw GENRE/TITLE/LYRICS text (src.lyrics parses it). Raises on failure.""" llm = _get_llm() msgs = _messages(resume_text, job_description, genre, level, zone_desc, voice) text = "" for temp in (_TEMPERATURE, 1.15): # second pass only if it parroted an anchor out = llm.create_chat_completion( messages=msgs, max_tokens=max_tokens, temperature=temp, top_p=_TOP_P, repeat_penalty=_REPEAT_PENALTY, presence_penalty=_PRESENCE_PENALTY, stop=_STOP, ) text = out["choices"][0]["message"]["content"].strip() if not _echoes_anchor(text): break return text