A / tests /test_user_facing_quality.py
linxinhua's picture
Clone from v5-AI-ethic-HOT-socratic to v5-AI-ethic-HOT-socratic-frame
068faf9 verified
Raw
History Blame Contribute Delete
23.8 kB
"""End-to-end quality check for the natural-language-directive manager + bot.
The three user-facing properties this test checks (per the user's spec):
A. **WHY are we discussing this?** Opening renders the plan's `hook` as
a participant-facing line above the first question, anchored to the
argument's words.
B. **TASK COMPLETED → transition + recommendation.** After the
participant gives substantive content satisfying the gap, the
manager judges Path (A) → transition-due. The bot's reply has the
2-block transition shape with a recap of participant content + a new
hook + 2 connected questions.
C. **STUCK → transition + revision recommendation from the plan.** After
the participant repeatedly signals confusion / asks for the answer,
the manager judges Path (B) → transition-due. The bot's reply has
Block 1 with a revision recommendation built from the gap, then
Block 2 with the new hook + 1 question (from new probe's question_set).
Also asserted globally: no framework jargon leaks; no candidate-answer
introductions; clear short questions.
Run from the socratic/ folder:
python -X utf8 -m tests.test_user_facing_quality
"""
import sys, re
from pathlib import Path
HERE = Path(__file__).resolve().parent
ROOT = HERE.parent
sys.path.insert(0, str(ROOT))
from dotenv import load_dotenv # noqa: E402
load_dotenv(ROOT / "../../.env")
from core import plan_parsing # noqa: E402
from core.perfect_answer import generate_session_artifacts # noqa: E402
from core.manager_agent import classify # noqa: E402
from core.main_agent import build_plan_control, build_messages, stream # noqa: E402
from core.config_loader import BASE_DIR # noqa: E402
ARGUMENT = (
"i think it is not ethically acceptable because students will lost their "
"privacy, and they will feel uncofmtable of being monitred. also their "
"tracable activity online does not represent their actually learning, "
"and each individuals are different, you cant use one-size-fits-all "
"metrics to evalaute everyone. also it may be biased and unfair to students."
)
JARGON_FORBIDDEN = (
"stakeholders", "expected_consequences", "evidence_uncertainty",
"final_judgement", "consequence framework", "consequence-based",
"sub-probe", "sub probe", "the gap", "the rationale", "the directive",
)
ACTOR_LEAK_SUSPECTS = (
"decision-makers", "decision makers", "administrators",
"advisors", "counselors", "support staff", "parents", "future cohorts",
"wrongly-flagged students", "wrongly flagged students",
)
def banner(t):
print()
print("=" * 72)
print(f" {t}")
print("=" * 72)
def keywords(s, k=4, min_len=6):
words = [w.strip(".,;:'\"()[]") for w in (s or "").split()]
return [w.lower() for w in words if len(w) >= min_len][:k]
def has_jargon(text):
tl = text.lower()
return [j for j in JARGON_FORBIDDEN if j in tl]
def actor_leaks(text, allowed_text):
tl = text.lower()
al = allowed_text.lower()
return [w for w in ACTOR_LEAK_SUSPECTS if w in tl and w not in al]
def run_bot(verdict, subprobes, plan_text, recent_turns, current_msg):
case_path = BASE_DIR / "data" / "case.md"
case_text = case_path.read_text(encoding="utf-8") \
if case_path.exists() else "(case unavailable)"
plan_control = build_plan_control(verdict, subprobes)
full_history = recent_turns + [{"role": "user", "content": current_msg}]
msgs = build_messages(full_history, case_text, plan_text, plan_control)
for kind, payload in stream(msgs):
if kind == "done":
return payload.visible, payload.scratch
return "", {}
def report(name, checks):
print(f"\nChecks ({name}):")
for label, ok in checks:
print(f" [{'PASS' if ok else 'FAIL'}] {label}")
return all(ok for _, ok in checks)
def last_question(visible):
"""Return the last sentence ending in `?`, stripped of any leading
colon-bridge ("Starting from what you wrote about X: <question>" →
"<question>")."""
m = list(re.finditer(r"([^.!?\n]*\?)", visible))
q = m[-1].group(1).strip() if m else ""
if ":" in q:
q = q.split(":", 1)[1].strip()
return q
def first_words_have_stem(question, allowed=("what", "why", "how", "who", "which")):
words = [w.strip(",.;:") for w in question.lstrip().lower().split()[:5]]
return any(w in allowed for w in words)
# ---------------------------------------------------------------------------
def scenario_a_opening(subprobes, plan_text):
banner("Scenario A — OPENING: hook + clear Q1 anchored to argument")
verdict = classify(
discussion_plan_subprobes=subprobes,
manager_history=[],
recent_turns=[{"role": "user", "content": ARGUMENT}],
current_msg=ARGUMENT,
initial_argument=ARGUMENT,
)
print(f"manager turn_type : {verdict.turn_type}")
print(f"manager live_subprobe_id: {verdict.live_subprobe_id}")
print(f"manager directive (truncated):\n{verdict.directive[:400]}...\n")
live = plan_parsing.by_id(subprobes, verdict.live_subprobe_id) or {}
plan_hook = live.get("hook", "")
hook_kws = keywords(plan_hook, k=4)
visible, scratch = run_bot(verdict, subprobes, plan_text,
recent_turns=[], current_msg=ARGUMENT)
print("VISIBLE REPLY:\n" + visible + "\n")
print(f"SCRATCH: {scratch}\n")
q = last_question(visible)
q_words = len(q.split())
participant_kws = keywords(ARGUMENT, k=12, min_len=6)
q_anchor_hits = sum(1 for w in participant_kws if w in q.lower())
checks = [
("manager turn_type=opening", verdict.turn_type == "opening"),
("manager directive is non-empty", bool(verdict.directive.strip())),
(f"plan.hook content rendered in visible (>=2 of {len(hook_kws)} kws)",
sum(1 for w in hook_kws if w in visible.lower()) >= 2),
(f"exactly one '?' in visible (got {visible.count('?')})",
visible.count("?") == 1),
(f"Q <=25 words (got {q_words})", q_words <= 25),
(f"Q starts with what/why/how/who/which: {q[:80]!r}",
first_words_have_stem(q)),
(f"Q anchors to participant's argument words (>=1 of {len(participant_kws)} kws)",
q_anchor_hits >= 1),
(f"no framework jargon in visible (leaks: {has_jargon(visible)})",
not has_jargon(visible)),
]
return report("Scenario A", checks)
# ---------------------------------------------------------------------------
def scenario_b_path_a_transition(subprobes, plan_text):
banner("Scenario B — Path (A): gap satisfied → transition + recommendation")
# Pre-load 2 substantive turns of engaging-default on sub-probe 2.
p_replies = [
"teachers and the university make the decisions about all this",
"tech staff that build the system, plus a manager who actually runs it",
]
bot_replies = [
"Got it. Starting from what you wrote: who else besides students feels these predictions day to day?",
"Right, teachers and the university. Beyond them, which specific staff actually touch the predictions?",
]
history = [
{
"live_subprobe_id": 2, "advanced_this_turn": False, "turn_type": "opening",
"directive": "(opening directive)",
"reason": "First turn; opening on probe 2 (stakeholders).",
"user_msg_excerpt": ARGUMENT[:120],
},
{
"live_subprobe_id": 2, "advanced_this_turn": False, "turn_type": "engaging-default",
"directive": "(engaging-default directive)",
"reason": "Participant named teachers and university; widening the lens.",
"user_msg_excerpt": p_replies[0],
},
]
recent = [{"role": "user", "content": ARGUMENT}]
for b, u in zip(bot_replies, p_replies):
recent += [{"role": "assistant", "content": b}, {"role": "user", "content": u}]
current = "and probably parents whose kids' grades suffer"
verdict = classify(
discussion_plan_subprobes=subprobes,
manager_history=history,
recent_turns=recent,
current_msg=current,
initial_argument=ARGUMENT,
)
print(f"manager turn_type : {verdict.turn_type}")
print(f"manager live_subprobe_id: {verdict.live_subprobe_id} (was 2; advanced={verdict.advanced_this_turn})")
print(f"manager directive:\n{verdict.directive}\n")
visible, scratch = run_bot(verdict, subprobes, plan_text, recent, current)
print("VISIBLE REPLY:\n" + visible + "\n")
print(f"SCRATCH: {scratch}\n")
# The participant said "teachers", "university", "tech staff", "manager", "parents".
participant_text = " ".join(p_replies + [current])
# New live sub-probe's hook should be referenced in Block 2.
new_live = plan_parsing.by_id(subprobes, verdict.live_subprobe_id) or {}
new_hook = new_live.get("hook", "")
new_hook_kws = keywords(new_hook, k=4)
# Block 2 contains 1 question; Block 1 must have 0.
paragraphs = [p.strip() for p in visible.split("\n\n") if p.strip()]
block1 = paragraphs[0] if paragraphs else ""
# Verify question_set has been generated for the NEW probe and is sane.
qset = verdict.question_set or []
pending_count = sum(1 for it in qset if it.get("status") == "pending")
asked_count = sum(1 for it in qset if it.get("status") == "asked")
checks = [
("manager advanced (transition-due)", verdict.turn_type == "transition-due"),
("live_subprobe_id moved past 2", verdict.live_subprobe_id > 2),
("directive is natural language (no Q1/Q2/revision_hook field labels)",
"Q1:" not in verdict.directive and "revision_hook:" not in verdict.directive),
(f"question_set is non-empty (has {len(qset)} items)",
len(qset) >= 1),
(f"question_set has at least one pending (got {pending_count})",
pending_count >= 1),
("transition-due discarded prior probe's question_set "
"(asked count should be 0 since this is a fresh set for new probe)",
asked_count == 0),
(f"visible reply has exactly 1 '?' (Block 2 = 1 question; got {visible.count('?')})",
visible.count("?") == 1),
(f"new hook content rendered in visible (>=2 of {len(new_hook_kws)} kws)",
sum(1 for w in new_hook_kws if w in visible.lower()) >= 2),
("Block 1 has 0 '?'", "?" not in block1),
(f"visible reply has at least 2 paragraphs (Block 1 + Block 2)",
len(paragraphs) >= 2),
(f"no framework jargon in visible (leaks: {has_jargon(visible)})",
not has_jargon(visible)),
(f"visible introduces no new actor names beyond participant's (leaks: {actor_leaks(visible, participant_text)})",
not actor_leaks(visible, participant_text)),
("Block 1 does NOT start with engaging-default reflect-back",
not any(block1.lower().startswith(p) for p in ("got it", "right,", "yeah,", "you named"))),
]
return report("Scenario B", checks)
# ---------------------------------------------------------------------------
def scenario_c_path_b_transition(subprobes, plan_text):
banner("Scenario C — Path (B): stuck → transition + revision guidance from plan")
history = [
{
"live_subprobe_id": 2, "advanced_this_turn": False, "turn_type": "opening",
"directive": "(opening)",
"reason": "First turn.",
"user_msg_excerpt": ARGUMENT[:120],
},
{
"live_subprobe_id": 2, "advanced_this_turn": False, "turn_type": "needs-clarification",
"directive": "(re-pose simpler)",
"reason": "Participant said 'i don't get what you mean'.",
"user_msg_excerpt": "i don't get what you mean",
},
{
"live_subprobe_id": 2, "advanced_this_turn": False, "turn_type": "needs-clarification",
"directive": "(re-pose simpler again)",
"reason": "Still confused after first simplification.",
"user_msg_excerpt": "i still dont follow",
},
]
recent = [
{"role": "user", "content": ARGUMENT},
{"role": "assistant", "content": "Starting from what you wrote: who else besides students feels these predictions?"},
{"role": "user", "content": "i don't get what you mean"},
{"role": "assistant", "content": "Sorry, let me try more concretely. Imagine a teacher seeing 'likely to struggle' on a Monday morning. Who else, besides the student, would notice that label?"},
{"role": "user", "content": "i still dont follow"},
{"role": "assistant", "content": "Let me cut it down. Picture one moment in the course when the system makes a prediction. What other person sees it besides the student?"},
]
current = "i really don't know, just tell me"
verdict = classify(
discussion_plan_subprobes=subprobes,
manager_history=history,
recent_turns=recent,
current_msg=current,
initial_argument=ARGUMENT,
)
print(f"manager turn_type : {verdict.turn_type}")
print(f"manager live_subprobe_id: {verdict.live_subprobe_id} (was 2; advanced={verdict.advanced_this_turn})")
print(f"manager directive:\n{verdict.directive}\n")
if verdict.turn_type != "transition-due":
# Manager didn't take Path (B). Report just that fact.
return report("Scenario C", [
("manager judged Path (B) and advanced (transition-due)",
verdict.turn_type == "transition-due"),
])
visible, scratch = run_bot(verdict, subprobes, plan_text, recent, current)
print("VISIBLE REPLY:\n" + visible + "\n")
print(f"SCRATCH: {scratch}\n")
paragraphs = [p.strip() for p in visible.split("\n\n") if p.strip()]
block1 = paragraphs[0] if paragraphs else ""
# Path (B) revision recommendation should reference the closing probe's
# territory in plain words. Look for "for your revision" / "would
# strengthen your draft" / "even a single line" / "would close" /
# similar revision-pointing phrases.
rev_phrases = ("for your revision", "your draft would", "would strengthen",
"even a single line", "even one line", "would close",
"would tighten", "would make the harm chain",
"would let a reader")
block1_l = block1.lower()
has_revision_pointer = any(p in block1_l for p in rev_phrases)
new_live = plan_parsing.by_id(subprobes, verdict.live_subprobe_id) or {}
new_hook = new_live.get("hook", "")
new_hook_kws = keywords(new_hook, k=4)
checks = [
("manager judged Path (B) and advanced (transition-due)",
verdict.turn_type == "transition-due"),
("live_subprobe_id moved past 2",
verdict.live_subprobe_id > 2),
(f"Block 1 contains revision-pointing phrase (found: {has_revision_pointer})",
has_revision_pointer),
(f"visible has 1 '?' for 1-question Block 2 (got {visible.count('?')})",
visible.count("?") == 1),
(f"new probe's hook rendered in visible (>=2 of {len(new_hook_kws)} kws)",
sum(1 for w in new_hook_kws if w in visible.lower()) >= 2),
(f"no framework jargon in visible (leaks: {has_jargon(visible)})",
not has_jargon(visible)),
# Path (B): participant gave no substance, so no actor names should be introduced.
(f"visible introduces no new actor names (leaks: {actor_leaks(visible, ARGUMENT)})",
not actor_leaks(visible, ARGUMENT)),
]
return report("Scenario C", checks)
# ---------------------------------------------------------------------------
def scenario_d_lookback_pivot(subprobes, plan_text):
"""Replicates session 20260602_181022_31358 turn 6→7: at transition into
expected_consequences, the participant's CURRENT_MSG already names the
teacher reaction. The manager's fresh probe-3 question_set must NOT
open with a near-paraphrase of what the participant just said. The
Look-back check should pivot the entry angle."""
banner("Scenario D — Look-back: don't ask what the participant just said")
history = [
{"live_subprobe_id": 2, "advanced_this_turn": False, "turn_type": "opening",
"directive": "(opening on stakeholders)",
"reason": "First turn; opening on probe 2.",
"user_msg_excerpt": ARGUMENT[:120],
"question_set": [
{"text": "Besides students, who else would be affected by this system?", "status": "pending"},
{"text": "Who in the university would use the predictions day to day?", "status": "pending"},
{"text": "Whose decisions could change because the system is deployed?", "status": "pending"},
]},
{"live_subprobe_id": 2, "advanced_this_turn": False, "turn_type": "engaging-default",
"directive": "(probe 2 evolution)",
"reason": "Participant said teacher; covered Q[0].",
"user_msg_excerpt": "teacher",
"question_set": [
{"text": "Besides students, who else would be affected by this system?", "status": "asked"},
{"text": "Who in the university would use the predictions day to day?", "status": "pending"},
{"text": "Whose decisions could change because the system is deployed?", "status": "pending"},
]},
{"live_subprobe_id": 2, "advanced_this_turn": False, "turn_type": "engaging-default",
"directive": "(probe 2 evolution)",
"reason": "Participant said teacher again.",
"user_msg_excerpt": "as i said , teacher",
"question_set": [
{"text": "Besides students, who else would be affected by this system?", "status": "asked"},
{"text": "Who would look at the system's predictions in day-to-day work?", "status": "asked"},
{"text": "Whose decisions could change because the system is deployed?", "status": "pending"},
]},
]
recent = [
{"role": "user", "content": ARGUMENT},
{"role": "assistant", "content": "Starting from what you wrote about privacy: Besides students, who else would be affected by this system?"},
{"role": "user", "content": "teacher"},
{"role": "assistant", "content": "Got it. Who in the university would use the predictions day to day?"},
{"role": "user", "content": "as i said , teacher"},
{"role": "assistant", "content": "Right. What everyday choice or action would teachers or staff make differently because of this system?"},
]
# This is the exact phrasing from the log that bled into probe-3 territory.
current = ("teacher will provide support to students, so that when they "
"see the flag in system, they may treat students differently")
verdict = classify(
discussion_plan_subprobes=subprobes,
manager_history=history,
recent_turns=recent,
current_msg=current,
initial_argument=ARGUMENT,
)
print(f"manager turn_type : {verdict.turn_type}")
print(f"manager live_subprobe_id: {verdict.live_subprobe_id} (was 2; advanced={verdict.advanced_this_turn})")
print(f"manager directive:\n{verdict.directive}\n")
print("manager question_set for new probe:")
for i, item in enumerate(verdict.question_set or []):
print(f" [{i}] ({item.get('status','?')}) {item.get('text','')}")
print()
# Manager should advance to probe 3 (expected_consequences).
advanced_ok = verdict.turn_type == "transition-due" and verdict.live_subprobe_id >= 3
qset = verdict.question_set or []
first_pending = next((it["text"] for it in qset if it.get("status") == "pending"), "")
# Look-back check: the first pending must not be a near-paraphrase of the
# participant's CURRENT_MSG. Two heuristic gates:
# (1) tokens unique to the question (beyond stopwords) should NOT be a
# subset of CURRENT_MSG tokens — meaning the question introduces a
# fresh angle.
# (2) literal phrase overlap: forbid the question from containing
# "treat students differently" or "see the flag" verbatim — those
# are the participant's just-said words.
STOP = {"the","a","an","to","of","in","on","at","for","by","with","and","or","but",
"what","why","how","who","which","when","where","is","are","do","does","did",
"you","they","them","their","this","that","these","those","be","been","being",
"would","could","should","will","might","may","just","one","two","else","more",
"after","before","over","under","also","like","as","if","then","than","there"}
def tokenize(s):
s = re.sub(r"[^a-zA-Z\s]", " ", s.lower())
return [w for w in s.split() if w and w not in STOP and len(w) > 2]
q_tokens = set(tokenize(first_pending))
msg_tokens = set(tokenize(current))
q_unique_to_msg = bool(q_tokens - msg_tokens) # has at least one fresh non-stopword
literal_overlap_phrases = (
"treat students differently", "treat them differently",
"see the flag", "see a flag", "see the prediction",
"provide support",
)
literal_leak = [p for p in literal_overlap_phrases if p in first_pending.lower()]
# An additional sanity check: the question should reasonably ask about
# consequences beyond the immediate reaction (e.g. "first concrete change",
# "over time", "downstream", "trajectory", "if wrong", "specifically").
pivot_signal_words = ("first", "concrete", "specifically", "over time",
"trajectory", "downstream", "wrong", "instead",
"longer", "later", "outside", "missed",
"noticed", "trust", "interest", "grade", "experience")
has_pivot_signal = any(w in first_pending.lower() for w in pivot_signal_words)
checks = [
(f"manager advanced to probe 3 (got {verdict.live_subprobe_id})", advanced_ok),
(f"first pending exists (got {first_pending!r})", bool(first_pending)),
(f"first pending has tokens beyond CURRENT_MSG (fresh angle)", q_unique_to_msg),
(f"first pending has no literal leak of participant phrases (leaks: {literal_leak})",
not literal_leak),
(f"first pending has a pivot signal (one of {pivot_signal_words})",
has_pivot_signal),
]
return report("Scenario D", checks)
# ---------------------------------------------------------------------------
def main():
banner("STEP 0: generate plan")
arts = generate_session_artifacts(ARGUMENT)
plan_text = arts.get("discussion_plan", "")
subprobes = plan_parsing.parse_plan(plan_text)
print(f"plan sub-probes: {len(subprobes)}; dominant: {arts.get('dominant_framework')}")
for sp in subprobes:
print(f" id={sp['id']} component={sp['component']:<22}")
print(f" gap : {(sp.get('gap') or '')[:80]}")
print(f" hook: {(sp.get('hook') or '')[:80]}")
missing = [sp["id"] for sp in subprobes if not sp.get("hook", "").strip()]
if missing:
print(f"WARNING: plan missing hook on {missing}; scenarios touching those may fail.")
a = scenario_a_opening(subprobes, plan_text)
b = scenario_b_path_a_transition(subprobes, plan_text)
c = scenario_c_path_b_transition(subprobes, plan_text)
d = scenario_d_lookback_pivot(subprobes, plan_text)
banner(f"OVERALL: A={'PASS' if a else 'FAIL'} "
f"B={'PASS' if b else 'FAIL'} "
f"C={'PASS' if c else 'FAIL'} "
f"D={'PASS' if d else 'FAIL'}")
return a and b and c and d
if __name__ == "__main__":
sys.exit(0 if main() else 1)