Tanya-Khanna
Reframe Tiny Titan -> Tiny Mode (award is the judges' call; describe the feature)
062f677 | """The Understudy — fine-tuned MiniCPM4-0.5B (GGUF) on llama.cpp, CPU. | |
| Powers Tiny Mode and instant slider-drag previews: zero GPU, single-shot. | |
| With ACE-Step 2B turbo the whole Tiny Mode pipeline stays ~2.5B params. | |
| - fine-tuned & published on the Hub -> Well-Tuned | |
| - served as Q4_K_M GGUF via llama.cpp -> Llama Champion | |
| - 0.5B lyricist + 2B turbo music -> Tiny Mode | |
| The base model is ChatML (<|im_start|>/<|im_end|>, eos <|im_end|>). We feed | |
| messages through create_chat_completion, which applies the GGUF's embedded | |
| ChatML template AND tokenizes <|im_start|>/<|im_end|> as the real special | |
| tokens the fine-tune was trained on. (A hand-built prompt string passed to | |
| plain completion mis-tokenizes those markers as literal text — verified to | |
| produce degenerate output — so don't go back to that.) | |
| """ | |
| import os | |
| import threading | |
| from .prompts import UNDERSTUDY_SYSTEM, build_messages | |
| REPO = "tanya8997/openwork-understudy-0.5b" | |
| GGUF_FILE = "understudy-Q4_K_M.gguf" | |
| # generation_config.json on the Hub: temperature/top_p 0.8, eos <|im_end|> | |
| # bumped a touch so the 0.5B stops falling back on its memorised calibration anchors | |
| # ("I turn messy signals into decisions…") and actually writes from the resume | |
| _TEMPERATURE = 0.95 | |
| _TOP_P = 0.92 | |
| # a 0.5B loves to loop a hook ("I'm the job candidate" x12) — penalise repeats | |
| _REPEAT_PENALTY = 1.3 | |
| # presence_penalty nudges it toward fresh wording instead of the stock examples | |
| _PRESENCE_PENALTY = 0.4 | |
| _STOP = ["<|im_end|>", "<|endoftext|>", "<s>"] | |
| _llm = None | |
| _lock = threading.Lock() | |
| _gguf_path: str | None = None | |
| load_error: Exception | None = None | |
| # Prefetch the GGUF at import so the first preview is instant on the Space. | |
| # OTW_SKIP_PREFETCH lets local dev / tests import without the ~400MB download. | |
| if os.environ.get("OTW_SKIP_PREFETCH"): | |
| load_error = RuntimeError("prefetch skipped (OTW_SKIP_PREFETCH)") | |
| else: | |
| try: | |
| from huggingface_hub import hf_hub_download | |
| _gguf_path = hf_hub_download(REPO, GGUF_FILE) | |
| print(f"[understudy] gguf ready: {_gguf_path}") | |
| except Exception as e: # offline / no hub — Tiny Mode falls back to stubs | |
| load_error = e | |
| def _messages(resume_text: str, job_description: str, genre: str, | |
| level: int, zone_desc: str, voice: str | None = None) -> list[dict]: | |
| """The condensed Understudy system prompt + the same production user | |
| payload the fine-tune was trained on (build_messages' user turn).""" | |
| user = build_messages(resume_text, job_description, genre, level, zone_desc, voice)[1][ | |
| "content" | |
| ] | |
| return [ | |
| {"role": "system", "content": UNDERSTUDY_SYSTEM}, | |
| {"role": "user", "content": user}, | |
| ] | |
| def _get_llm(): | |
| """Lazy-load the llama.cpp model once (CPU). Raises if unavailable.""" | |
| global _llm | |
| if _llm is not None: | |
| return _llm | |
| if load_error is not None or not _gguf_path: | |
| raise RuntimeError(f"understudy unavailable: {load_error!r}") | |
| with _lock: | |
| if _llm is None: | |
| from llama_cpp import Llama | |
| _llm = Llama( | |
| model_path=_gguf_path, | |
| n_ctx=2048, | |
| n_threads=os.cpu_count() or 4, | |
| verbose=False, | |
| ) | |
| return _llm | |
| # The 0.5B memorised the calibration anchors (it was fine-tuned on gpt-oss data that | |
| # few-shot those exact lines), so for data-scientist resumes at levels 1/5/10 it parrots | |
| # them instead of writing from the resume. Detect that and regenerate hotter. | |
| _ANCHOR_FRAGMENTS = ( | |
| "messy signals into decisions", "models with precision", | |
| "dream in sql and dashboards", "churn rate fears me", | |
| "split-test feelings in the dark", "regression has a hiring arc", | |
| ) | |
| def _echoes_anchor(text: str) -> bool: | |
| t = text.lower() | |
| return any(frag in t for frag in _ANCHOR_FRAGMENTS) | |
| def write(resume_text: str, job_description: str, genre: str, | |
| level: int, zone_desc: str, max_tokens: int = 512, voice: str | None = None) -> str: | |
| """Single-shot lyric generation on the CPU Understudy. Returns the raw | |
| GENRE/TITLE/LYRICS text (src.lyrics parses it). Raises on failure.""" | |
| llm = _get_llm() | |
| msgs = _messages(resume_text, job_description, genre, level, zone_desc, voice) | |
| text = "" | |
| for temp in (_TEMPERATURE, 1.15): # second pass only if it parroted an anchor | |
| out = llm.create_chat_completion( | |
| messages=msgs, max_tokens=max_tokens, temperature=temp, top_p=_TOP_P, | |
| repeat_penalty=_REPEAT_PENALTY, presence_penalty=_PRESENCE_PENALTY, stop=_STOP, | |
| ) | |
| text = out["choices"][0]["message"]["content"].strip() | |
| if not _echoes_anchor(text): | |
| break | |
| return text | |