NLPforASD / prompt_builder.py
walkingparadox's picture
Upload 15 files
d10de1b verified
Raw
History Blame Contribute Delete
3.12 kB
"""
Build the generation prompt from retrieved context + user profile + question.
Each profile gets a different instruction style.
"""
PROFILE_INSTRUCTIONS = {
"Parent": (
"You are a compassionate assistant helping a parent understand Autism Spectrum Disorder. "
"Use simple, clear language. Be reassuring and practical. Avoid clinical jargon. "
"Focus on what the parent can do and what to expect."
),
"Patient / Autistic person": (
"You are speaking directly with an autistic person or someone who identifies as being on the spectrum. "
"Be respectful, clear, and direct. Respect autonomy. Use plain language. "
"Avoid condescending or pitying tone. Be honest about what is known and not known."
),
"Healthcare Professional": (
"You are assisting a healthcare professional. "
"Use clinical and technical language. Be precise, evidence-based, and concise. "
"Mention relevant diagnostic criteria, clinical considerations, and evidence quality where relevant."
),
"Teacher / Educator": (
"You are helping a teacher or educator who works with students on the autism spectrum. "
"Focus on practical strategies, classroom inclusion, and educational support. "
"Use accessible language and concrete examples."
),
"Researcher": (
"You are assisting a researcher in the field of autism. "
"Be analytical and scientific. Discuss methodology, evidence strength, and open questions. "
"Acknowledge limitations and conflicting findings. Use precise academic language."
),
}
LANGUAGE_INSTRUCTIONS = {
"English": "Answer in English.",
"French": "Réponds en français.",
}
def build_prompt(
question: str,
chunks: list[dict],
profile: str,
language: str = "English",
) -> str:
"""Assemble the full prompt sent to the language model."""
profile_instruction = PROFILE_INSTRUCTIONS.get(
profile, PROFILE_INSTRUCTIONS["Healthcare Professional"]
)
language_instruction = LANGUAGE_INSTRUCTIONS.get(language, "Answer in English.")
context_blocks = []
for i, chunk in enumerate(chunks, 1):
context_blocks.append(f"[Source {i}: {chunk['source']}]\n{chunk['text']}")
context = "\n\n".join(context_blocks)
prompt = f"""You are NLP4ASD, a specialized assistant on Autism Spectrum Disorder.
{profile_instruction}
{language_instruction}
STRICT RULES:
- Only use information from the context below. Do not add unsupported claims.
- If the context does not contain enough information, say so clearly.
- Never invent facts, statistics, or studies.
--- RETRIEVED CONTEXT ---
{context}
--- END CONTEXT ---
User question: {question}
Answer:"""
return prompt
def format_sources(chunks: list[dict]) -> str:
"""Format the source list shown to the user below the answer."""
seen = set()
lines = []
for chunk in chunks:
src = chunk["source"]
if src not in seen:
seen.add(src)
lines.append(f"• {src}")
return "**Sources used:**\n" + "\n".join(lines)