"""End-to-end quality check for the natural-language-directive manager + bot. The three user-facing properties this test checks (per the user's spec): A. **WHY are we discussing this?** Opening renders the plan's `hook` as a participant-facing line above the first question, anchored to the argument's words. B. **TASK COMPLETED → transition + recommendation.** After the participant gives substantive content satisfying the gap, the manager judges Path (A) → transition-due. The bot's reply has the 2-block transition shape with a recap of participant content + a new hook + 2 connected questions. C. **STUCK → transition + revision recommendation from the plan.** After the participant repeatedly signals confusion / asks for the answer, the manager judges Path (B) → transition-due. The bot's reply has Block 1 with a revision recommendation built from the gap, then Block 2 with the new hook + 1 question (from new probe's question_set). Also asserted globally: no framework jargon leaks; no candidate-answer introductions; clear short questions. Run from the socratic/ folder: python -X utf8 -m tests.test_user_facing_quality """ import sys, re from pathlib import Path HERE = Path(__file__).resolve().parent ROOT = HERE.parent sys.path.insert(0, str(ROOT)) from dotenv import load_dotenv # noqa: E402 load_dotenv(ROOT / "../../.env") from core import plan_parsing # noqa: E402 from core.perfect_answer import generate_session_artifacts # noqa: E402 from core.manager_agent import classify # noqa: E402 from core.main_agent import build_plan_control, build_messages, stream # noqa: E402 from core.config_loader import BASE_DIR # noqa: E402 ARGUMENT = ( "i think it is not ethically acceptable because students will lost their " "privacy, and they will feel uncofmtable of being monitred. also their " "tracable activity online does not represent their actually learning, " "and each individuals are different, you cant use one-size-fits-all " "metrics to evalaute everyone. also it may be biased and unfair to students." ) JARGON_FORBIDDEN = ( "stakeholders", "expected_consequences", "evidence_uncertainty", "final_judgement", "consequence framework", "consequence-based", "sub-probe", "sub probe", "the gap", "the rationale", "the directive", ) ACTOR_LEAK_SUSPECTS = ( "decision-makers", "decision makers", "administrators", "advisors", "counselors", "support staff", "parents", "future cohorts", "wrongly-flagged students", "wrongly flagged students", ) def banner(t): print() print("=" * 72) print(f" {t}") print("=" * 72) def keywords(s, k=4, min_len=6): words = [w.strip(".,;:'\"()[]") for w in (s or "").split()] return [w.lower() for w in words if len(w) >= min_len][:k] def has_jargon(text): tl = text.lower() return [j for j in JARGON_FORBIDDEN if j in tl] def actor_leaks(text, allowed_text): tl = text.lower() al = allowed_text.lower() return [w for w in ACTOR_LEAK_SUSPECTS if w in tl and w not in al] def run_bot(verdict, subprobes, plan_text, recent_turns, current_msg): case_path = BASE_DIR / "data" / "case.md" case_text = case_path.read_text(encoding="utf-8") \ if case_path.exists() else "(case unavailable)" plan_control = build_plan_control(verdict, subprobes) full_history = recent_turns + [{"role": "user", "content": current_msg}] msgs = build_messages(full_history, case_text, plan_text, plan_control) for kind, payload in stream(msgs): if kind == "done": return payload.visible, payload.scratch return "", {} def report(name, checks): print(f"\nChecks ({name}):") for label, ok in checks: print(f" [{'PASS' if ok else 'FAIL'}] {label}") return all(ok for _, ok in checks) def last_question(visible): """Return the last sentence ending in `?`, stripped of any leading colon-bridge ("Starting from what you wrote about X: " → "").""" m = list(re.finditer(r"([^.!?\n]*\?)", visible)) q = m[-1].group(1).strip() if m else "" if ":" in q: q = q.split(":", 1)[1].strip() return q def first_words_have_stem(question, allowed=("what", "why", "how", "who", "which")): words = [w.strip(",.;:") for w in question.lstrip().lower().split()[:5]] return any(w in allowed for w in words) # --------------------------------------------------------------------------- def scenario_a_opening(subprobes, plan_text): banner("Scenario A — OPENING: hook + clear Q1 anchored to argument") verdict = classify( discussion_plan_subprobes=subprobes, manager_history=[], recent_turns=[{"role": "user", "content": ARGUMENT}], current_msg=ARGUMENT, initial_argument=ARGUMENT, ) print(f"manager turn_type : {verdict.turn_type}") print(f"manager live_subprobe_id: {verdict.live_subprobe_id}") print(f"manager directive (truncated):\n{verdict.directive[:400]}...\n") live = plan_parsing.by_id(subprobes, verdict.live_subprobe_id) or {} plan_hook = live.get("hook", "") hook_kws = keywords(plan_hook, k=4) visible, scratch = run_bot(verdict, subprobes, plan_text, recent_turns=[], current_msg=ARGUMENT) print("VISIBLE REPLY:\n" + visible + "\n") print(f"SCRATCH: {scratch}\n") q = last_question(visible) q_words = len(q.split()) participant_kws = keywords(ARGUMENT, k=12, min_len=6) q_anchor_hits = sum(1 for w in participant_kws if w in q.lower()) checks = [ ("manager turn_type=opening", verdict.turn_type == "opening"), ("manager directive is non-empty", bool(verdict.directive.strip())), (f"plan.hook content rendered in visible (>=2 of {len(hook_kws)} kws)", sum(1 for w in hook_kws if w in visible.lower()) >= 2), (f"exactly one '?' in visible (got {visible.count('?')})", visible.count("?") == 1), (f"Q <=25 words (got {q_words})", q_words <= 25), (f"Q starts with what/why/how/who/which: {q[:80]!r}", first_words_have_stem(q)), (f"Q anchors to participant's argument words (>=1 of {len(participant_kws)} kws)", q_anchor_hits >= 1), (f"no framework jargon in visible (leaks: {has_jargon(visible)})", not has_jargon(visible)), ] return report("Scenario A", checks) # --------------------------------------------------------------------------- def scenario_b_path_a_transition(subprobes, plan_text): banner("Scenario B — Path (A): gap satisfied → transition + recommendation") # Pre-load 2 substantive turns of engaging-default on sub-probe 2. p_replies = [ "teachers and the university make the decisions about all this", "tech staff that build the system, plus a manager who actually runs it", ] bot_replies = [ "Got it. Starting from what you wrote: who else besides students feels these predictions day to day?", "Right, teachers and the university. Beyond them, which specific staff actually touch the predictions?", ] history = [ { "live_subprobe_id": 2, "advanced_this_turn": False, "turn_type": "opening", "directive": "(opening directive)", "reason": "First turn; opening on probe 2 (stakeholders).", "user_msg_excerpt": ARGUMENT[:120], }, { "live_subprobe_id": 2, "advanced_this_turn": False, "turn_type": "engaging-default", "directive": "(engaging-default directive)", "reason": "Participant named teachers and university; widening the lens.", "user_msg_excerpt": p_replies[0], }, ] recent = [{"role": "user", "content": ARGUMENT}] for b, u in zip(bot_replies, p_replies): recent += [{"role": "assistant", "content": b}, {"role": "user", "content": u}] current = "and probably parents whose kids' grades suffer" verdict = classify( discussion_plan_subprobes=subprobes, manager_history=history, recent_turns=recent, current_msg=current, initial_argument=ARGUMENT, ) print(f"manager turn_type : {verdict.turn_type}") print(f"manager live_subprobe_id: {verdict.live_subprobe_id} (was 2; advanced={verdict.advanced_this_turn})") print(f"manager directive:\n{verdict.directive}\n") visible, scratch = run_bot(verdict, subprobes, plan_text, recent, current) print("VISIBLE REPLY:\n" + visible + "\n") print(f"SCRATCH: {scratch}\n") # The participant said "teachers", "university", "tech staff", "manager", "parents". participant_text = " ".join(p_replies + [current]) # New live sub-probe's hook should be referenced in Block 2. new_live = plan_parsing.by_id(subprobes, verdict.live_subprobe_id) or {} new_hook = new_live.get("hook", "") new_hook_kws = keywords(new_hook, k=4) # Block 2 contains 1 question; Block 1 must have 0. paragraphs = [p.strip() for p in visible.split("\n\n") if p.strip()] block1 = paragraphs[0] if paragraphs else "" # Verify question_set has been generated for the NEW probe and is sane. qset = verdict.question_set or [] pending_count = sum(1 for it in qset if it.get("status") == "pending") asked_count = sum(1 for it in qset if it.get("status") == "asked") checks = [ ("manager advanced (transition-due)", verdict.turn_type == "transition-due"), ("live_subprobe_id moved past 2", verdict.live_subprobe_id > 2), ("directive is natural language (no Q1/Q2/revision_hook field labels)", "Q1:" not in verdict.directive and "revision_hook:" not in verdict.directive), (f"question_set is non-empty (has {len(qset)} items)", len(qset) >= 1), (f"question_set has at least one pending (got {pending_count})", pending_count >= 1), ("transition-due discarded prior probe's question_set " "(asked count should be 0 since this is a fresh set for new probe)", asked_count == 0), (f"visible reply has exactly 1 '?' (Block 2 = 1 question; got {visible.count('?')})", visible.count("?") == 1), (f"new hook content rendered in visible (>=2 of {len(new_hook_kws)} kws)", sum(1 for w in new_hook_kws if w in visible.lower()) >= 2), ("Block 1 has 0 '?'", "?" not in block1), (f"visible reply has at least 2 paragraphs (Block 1 + Block 2)", len(paragraphs) >= 2), (f"no framework jargon in visible (leaks: {has_jargon(visible)})", not has_jargon(visible)), (f"visible introduces no new actor names beyond participant's (leaks: {actor_leaks(visible, participant_text)})", not actor_leaks(visible, participant_text)), ("Block 1 does NOT start with engaging-default reflect-back", not any(block1.lower().startswith(p) for p in ("got it", "right,", "yeah,", "you named"))), ] return report("Scenario B", checks) # --------------------------------------------------------------------------- def scenario_c_path_b_transition(subprobes, plan_text): banner("Scenario C — Path (B): stuck → transition + revision guidance from plan") history = [ { "live_subprobe_id": 2, "advanced_this_turn": False, "turn_type": "opening", "directive": "(opening)", "reason": "First turn.", "user_msg_excerpt": ARGUMENT[:120], }, { "live_subprobe_id": 2, "advanced_this_turn": False, "turn_type": "needs-clarification", "directive": "(re-pose simpler)", "reason": "Participant said 'i don't get what you mean'.", "user_msg_excerpt": "i don't get what you mean", }, { "live_subprobe_id": 2, "advanced_this_turn": False, "turn_type": "needs-clarification", "directive": "(re-pose simpler again)", "reason": "Still confused after first simplification.", "user_msg_excerpt": "i still dont follow", }, ] recent = [ {"role": "user", "content": ARGUMENT}, {"role": "assistant", "content": "Starting from what you wrote: who else besides students feels these predictions?"}, {"role": "user", "content": "i don't get what you mean"}, {"role": "assistant", "content": "Sorry, let me try more concretely. Imagine a teacher seeing 'likely to struggle' on a Monday morning. Who else, besides the student, would notice that label?"}, {"role": "user", "content": "i still dont follow"}, {"role": "assistant", "content": "Let me cut it down. Picture one moment in the course when the system makes a prediction. What other person sees it besides the student?"}, ] current = "i really don't know, just tell me" verdict = classify( discussion_plan_subprobes=subprobes, manager_history=history, recent_turns=recent, current_msg=current, initial_argument=ARGUMENT, ) print(f"manager turn_type : {verdict.turn_type}") print(f"manager live_subprobe_id: {verdict.live_subprobe_id} (was 2; advanced={verdict.advanced_this_turn})") print(f"manager directive:\n{verdict.directive}\n") if verdict.turn_type != "transition-due": # Manager didn't take Path (B). Report just that fact. return report("Scenario C", [ ("manager judged Path (B) and advanced (transition-due)", verdict.turn_type == "transition-due"), ]) visible, scratch = run_bot(verdict, subprobes, plan_text, recent, current) print("VISIBLE REPLY:\n" + visible + "\n") print(f"SCRATCH: {scratch}\n") paragraphs = [p.strip() for p in visible.split("\n\n") if p.strip()] block1 = paragraphs[0] if paragraphs else "" # Path (B) revision recommendation should reference the closing probe's # territory in plain words. Look for "for your revision" / "would # strengthen your draft" / "even a single line" / "would close" / # similar revision-pointing phrases. rev_phrases = ("for your revision", "your draft would", "would strengthen", "even a single line", "even one line", "would close", "would tighten", "would make the harm chain", "would let a reader") block1_l = block1.lower() has_revision_pointer = any(p in block1_l for p in rev_phrases) new_live = plan_parsing.by_id(subprobes, verdict.live_subprobe_id) or {} new_hook = new_live.get("hook", "") new_hook_kws = keywords(new_hook, k=4) checks = [ ("manager judged Path (B) and advanced (transition-due)", verdict.turn_type == "transition-due"), ("live_subprobe_id moved past 2", verdict.live_subprobe_id > 2), (f"Block 1 contains revision-pointing phrase (found: {has_revision_pointer})", has_revision_pointer), (f"visible has 1 '?' for 1-question Block 2 (got {visible.count('?')})", visible.count("?") == 1), (f"new probe's hook rendered in visible (>=2 of {len(new_hook_kws)} kws)", sum(1 for w in new_hook_kws if w in visible.lower()) >= 2), (f"no framework jargon in visible (leaks: {has_jargon(visible)})", not has_jargon(visible)), # Path (B): participant gave no substance, so no actor names should be introduced. (f"visible introduces no new actor names (leaks: {actor_leaks(visible, ARGUMENT)})", not actor_leaks(visible, ARGUMENT)), ] return report("Scenario C", checks) # --------------------------------------------------------------------------- def scenario_d_lookback_pivot(subprobes, plan_text): """Replicates session 20260602_181022_31358 turn 6→7: at transition into expected_consequences, the participant's CURRENT_MSG already names the teacher reaction. The manager's fresh probe-3 question_set must NOT open with a near-paraphrase of what the participant just said. The Look-back check should pivot the entry angle.""" banner("Scenario D — Look-back: don't ask what the participant just said") history = [ {"live_subprobe_id": 2, "advanced_this_turn": False, "turn_type": "opening", "directive": "(opening on stakeholders)", "reason": "First turn; opening on probe 2.", "user_msg_excerpt": ARGUMENT[:120], "question_set": [ {"text": "Besides students, who else would be affected by this system?", "status": "pending"}, {"text": "Who in the university would use the predictions day to day?", "status": "pending"}, {"text": "Whose decisions could change because the system is deployed?", "status": "pending"}, ]}, {"live_subprobe_id": 2, "advanced_this_turn": False, "turn_type": "engaging-default", "directive": "(probe 2 evolution)", "reason": "Participant said teacher; covered Q[0].", "user_msg_excerpt": "teacher", "question_set": [ {"text": "Besides students, who else would be affected by this system?", "status": "asked"}, {"text": "Who in the university would use the predictions day to day?", "status": "pending"}, {"text": "Whose decisions could change because the system is deployed?", "status": "pending"}, ]}, {"live_subprobe_id": 2, "advanced_this_turn": False, "turn_type": "engaging-default", "directive": "(probe 2 evolution)", "reason": "Participant said teacher again.", "user_msg_excerpt": "as i said , teacher", "question_set": [ {"text": "Besides students, who else would be affected by this system?", "status": "asked"}, {"text": "Who would look at the system's predictions in day-to-day work?", "status": "asked"}, {"text": "Whose decisions could change because the system is deployed?", "status": "pending"}, ]}, ] recent = [ {"role": "user", "content": ARGUMENT}, {"role": "assistant", "content": "Starting from what you wrote about privacy: Besides students, who else would be affected by this system?"}, {"role": "user", "content": "teacher"}, {"role": "assistant", "content": "Got it. Who in the university would use the predictions day to day?"}, {"role": "user", "content": "as i said , teacher"}, {"role": "assistant", "content": "Right. What everyday choice or action would teachers or staff make differently because of this system?"}, ] # This is the exact phrasing from the log that bled into probe-3 territory. current = ("teacher will provide support to students, so that when they " "see the flag in system, they may treat students differently") verdict = classify( discussion_plan_subprobes=subprobes, manager_history=history, recent_turns=recent, current_msg=current, initial_argument=ARGUMENT, ) print(f"manager turn_type : {verdict.turn_type}") print(f"manager live_subprobe_id: {verdict.live_subprobe_id} (was 2; advanced={verdict.advanced_this_turn})") print(f"manager directive:\n{verdict.directive}\n") print("manager question_set for new probe:") for i, item in enumerate(verdict.question_set or []): print(f" [{i}] ({item.get('status','?')}) {item.get('text','')}") print() # Manager should advance to probe 3 (expected_consequences). advanced_ok = verdict.turn_type == "transition-due" and verdict.live_subprobe_id >= 3 qset = verdict.question_set or [] first_pending = next((it["text"] for it in qset if it.get("status") == "pending"), "") # Look-back check: the first pending must not be a near-paraphrase of the # participant's CURRENT_MSG. Two heuristic gates: # (1) tokens unique to the question (beyond stopwords) should NOT be a # subset of CURRENT_MSG tokens — meaning the question introduces a # fresh angle. # (2) literal phrase overlap: forbid the question from containing # "treat students differently" or "see the flag" verbatim — those # are the participant's just-said words. STOP = {"the","a","an","to","of","in","on","at","for","by","with","and","or","but", "what","why","how","who","which","when","where","is","are","do","does","did", "you","they","them","their","this","that","these","those","be","been","being", "would","could","should","will","might","may","just","one","two","else","more", "after","before","over","under","also","like","as","if","then","than","there"} def tokenize(s): s = re.sub(r"[^a-zA-Z\s]", " ", s.lower()) return [w for w in s.split() if w and w not in STOP and len(w) > 2] q_tokens = set(tokenize(first_pending)) msg_tokens = set(tokenize(current)) q_unique_to_msg = bool(q_tokens - msg_tokens) # has at least one fresh non-stopword literal_overlap_phrases = ( "treat students differently", "treat them differently", "see the flag", "see a flag", "see the prediction", "provide support", ) literal_leak = [p for p in literal_overlap_phrases if p in first_pending.lower()] # An additional sanity check: the question should reasonably ask about # consequences beyond the immediate reaction (e.g. "first concrete change", # "over time", "downstream", "trajectory", "if wrong", "specifically"). pivot_signal_words = ("first", "concrete", "specifically", "over time", "trajectory", "downstream", "wrong", "instead", "longer", "later", "outside", "missed", "noticed", "trust", "interest", "grade", "experience") has_pivot_signal = any(w in first_pending.lower() for w in pivot_signal_words) checks = [ (f"manager advanced to probe 3 (got {verdict.live_subprobe_id})", advanced_ok), (f"first pending exists (got {first_pending!r})", bool(first_pending)), (f"first pending has tokens beyond CURRENT_MSG (fresh angle)", q_unique_to_msg), (f"first pending has no literal leak of participant phrases (leaks: {literal_leak})", not literal_leak), (f"first pending has a pivot signal (one of {pivot_signal_words})", has_pivot_signal), ] return report("Scenario D", checks) # --------------------------------------------------------------------------- def main(): banner("STEP 0: generate plan") arts = generate_session_artifacts(ARGUMENT) plan_text = arts.get("discussion_plan", "") subprobes = plan_parsing.parse_plan(plan_text) print(f"plan sub-probes: {len(subprobes)}; dominant: {arts.get('dominant_framework')}") for sp in subprobes: print(f" id={sp['id']} component={sp['component']:<22}") print(f" gap : {(sp.get('gap') or '')[:80]}") print(f" hook: {(sp.get('hook') or '')[:80]}") missing = [sp["id"] for sp in subprobes if not sp.get("hook", "").strip()] if missing: print(f"WARNING: plan missing hook on {missing}; scenarios touching those may fail.") a = scenario_a_opening(subprobes, plan_text) b = scenario_b_path_a_transition(subprobes, plan_text) c = scenario_c_path_b_transition(subprobes, plan_text) d = scenario_d_lookback_pivot(subprobes, plan_text) banner(f"OVERALL: A={'PASS' if a else 'FAIL'} " f"B={'PASS' if b else 'FAIL'} " f"C={'PASS' if c else 'FAIL'} " f"D={'PASS' if d else 'FAIL'}") return a and b and c and d if __name__ == "__main__": sys.exit(0 if main() else 1)