form-generatro / src /context_pack_builder.py
GitHub Action
Deploy form-generator from GitHub: eef63a23b3bc1ffad4a0dc85f8592e8481938620
62065d2
Raw
History Blame Contribute Delete
10.6 kB
"""
Docs-to-context-pack mapping and relevance rules for richer AI form generation.
Consumes LLM documentation (from docs API) and builds token-budgeted prompt sections
based on detected intent (conditional logic, document extraction, advanced fields).
"""
from typing import Any, Dict, List, Optional, Tuple
import json
import logging
logger = logging.getLogger(__name__)
# Approximate tokens from chars (conservative for Latin text)
CHARS_PER_TOKEN = 4
# --- Intent detection keywords (plan: Smart Context Strategy) ---
CONDITIONAL_LOGIC_SIGNALS = [
"if", "when", "depends", "show", "hide", "branch", "conditional",
"only when", "based on", "depending on", "skip", "reveal",
]
DOC_EXTRACTION_SIGNALS = [
"resume", "cv", "invoice", "receipt", "id card", "parse", "extract",
"document extraction", "upload and extract", "pdf", "scan",
]
ADVANCED_FIELD_SIGNALS = [
"matrix", "ranking", "nps", "likert", "rating", "score", "slider",
"validation", "validate", "scale", "grid",
]
# --- Pack names and doc section mapping ---
PACK_FIELD_TYPES = "fieldTypes"
PACK_CONDITIONAL_LOGIC = "conditionalLogic"
PACK_DOC_EXTRACTION = "aiFields"
PACK_EXAMPLES = "exampleForms"
PACK_BEST_PRACTICES = "bestPractices"
PACK_FORM_SCHEMA = "formSchema"
# Default character budgets (approx 4 chars per token; ~500 tokens per pack)
MAX_CHARS_PER_PACK = 2000
MAX_TOTAL_CONTEXT_CHARS = 8000
def detect_intent(
description: str = "",
user_request: Optional[str] = None,
current_fields: Optional[List[Dict[str, Any]]] = None,
conversation_context: Optional[List[str]] = None,
) -> Dict[str, bool]:
"""
Detect which context packs are relevant from user text and current form.
Returns flags: needs_conditional_logic, needs_document_extraction, needs_advanced_fields.
"""
text = " ".join(
filter(
None,
[description or "", user_request or ""]
+ (conversation_context or []),
)
).lower()
has_conditional = any(s in text for s in CONDITIONAL_LOGIC_SIGNALS)
has_doc_extraction = any(s in text for s in DOC_EXTRACTION_SIGNALS)
has_advanced = any(s in text for s in ADVANCED_FIELD_SIGNALS)
# If current form already has conditional logic or document-extraction, include those packs for refine
if current_fields:
for f in current_fields:
if isinstance(f, dict):
if (f.get("conditionalLogic") or {}).get("enabled"):
has_conditional = True
if f.get("type") == "document-extraction":
has_doc_extraction = True
return {
"needs_conditional_logic": has_conditional,
"needs_document_extraction": has_doc_extraction,
"needs_advanced_fields": has_advanced,
}
def _truncate(text: str, max_chars: int) -> str:
if len(text) <= max_chars:
return text
return text[: max_chars - 3].rstrip() + "..."
def _pack_field_types(docs: Dict[str, Any], max_chars: int) -> str:
"""Build FieldTypes pack from docs.fieldTypes (summary + key types)."""
field_types = docs.get("fieldTypes") or {}
if not field_types:
return ""
lines = ["## Field types reference\n"]
for name, spec in list(field_types.items())[:25]:
if not isinstance(spec, dict):
continue
desc = spec.get("description", "")
required = spec.get("requiredProperties", [])
lines.append(f"- **{name}**: {desc[:200]}")
if required:
lines.append(f" Required: {', '.join(required)}")
ex = spec.get("exampleJSON")
if ex:
lines.append(f" Example: {json.dumps(ex)[:300]}")
out = "\n".join(lines)
return _truncate(out, max_chars)
def _pack_conditional_logic(docs: Dict[str, Any], max_chars: int) -> str:
"""Build ConditionalLogic pack from docs.conditionalLogic."""
cl = docs.get("conditionalLogic") or {}
if not cl:
return ""
lines = [
"## Conditional logic\n",
(cl.get("overview") or "")[:500],
"\n### Operators (use in conditions): ",
]
ops = cl.get("operators") or {}
for op_name, op_spec in list(ops.items())[:15]:
if isinstance(op_spec, dict):
lines.append(f"- {op_name}: {(op_spec.get('description') or '')[:150]}")
lines.append("\n### Actions: show | hide | validate | cap_responses")
actions = cl.get("actions") or {}
for act_name, act_spec in list(actions.items())[:4]:
if isinstance(act_spec, dict):
ex = act_spec.get("example")
if ex:
lines.append(f"- {act_name}: {json.dumps(ex)[:200]}")
examples = cl.get("examples") or []
for ex in examples[:2]:
if isinstance(ex, dict) and ex.get("json"):
lines.append(f"Example: {json.dumps(ex['json'])[:250]}")
out = "\n".join(lines)
return _truncate(out, max_chars)
def _pack_doc_extraction(docs: Dict[str, Any], max_chars: int) -> str:
"""Build DocExtraction pack from docs.aiFields.documentExtraction."""
ai = docs.get("aiFields") or {}
doc_ext = ai.get("documentExtraction") if isinstance(ai, dict) else None
if not doc_ext:
return ""
lines = [
"## Document extraction field (document-extraction)\n",
(doc_ext.get("description") or "")[:400],
"\n### customFields: array of { id, name, description, fieldType, required, editable }",
"fieldType: single_value | list | number | date",
"Always set acceptedFileTypes (e.g. ['pdf','doc','docx','png','jpg','jpeg']) and maxFileSize in bytes (e.g. 10485760 for 10MB). Do not leave customFields empty.",
"verificationPrompt: descriptive text only (full sentences) stating how to verify the candidate against extracted data. Use the user's must-have/role requirements (e.g. from clarifying answers: years of experience, degree, skills). Never put only a number (e.g. 80) — invalid; use criteria like 'Candidate must have 2+ years experience, masters in CS, and LangGraph expertise.'",
]
config = doc_ext.get("configuration") or {}
if isinstance(config, dict):
cf = config.get("customFields") or {}
if isinstance(cf, dict) and cf.get("fieldTypes"):
lines.append("Field types: " + json.dumps(list((cf["fieldTypes"] or {}).keys())))
example_configs = doc_ext.get("exampleConfigurations") or []
for ex in example_configs[:2]:
if isinstance(ex, dict) and ex.get("json"):
lines.append(f"Example: {json.dumps(ex['json'])[:400]}")
out = "\n".join(lines)
return _truncate(out, max_chars)
def _pack_examples(docs: Dict[str, Any], max_chars: int, intent: Dict[str, bool]) -> str:
"""Build Examples pack: 1–3 examples matching intent."""
examples = docs.get("exampleForms") or []
if not examples:
return ""
chosen = []
for ex in examples:
if not isinstance(ex, dict) or not ex.get("json"):
continue
title = (ex.get("title") or "").lower()
features = ex.get("featuresUsed") or []
if intent.get("needs_document_extraction") and (
"document-extraction" in features or "resume" in title or "extraction" in title
):
chosen.append(ex)
elif intent.get("needs_conditional_logic") and "conditionalLogic" in str(features):
chosen.append(ex)
elif not chosen:
chosen.append(ex)
if len(chosen) >= 3:
break
if not chosen:
chosen = examples[:2]
lines = ["## Example forms (reference only)\n"]
for ex in chosen:
lines.append(f"### {ex.get('title', 'Form')}")
lines.append(json.dumps(ex.get("json") or {}, indent=2)[:800])
out = "\n".join(lines)
return _truncate(out, max_chars)
def build_context_packs(
docs: Dict[str, Any],
intent: Dict[str, bool],
max_chars_per_pack: int = MAX_CHARS_PER_PACK,
max_total_chars: int = MAX_TOTAL_CONTEXT_CHARS,
log_usage: bool = True,
) -> str:
"""
Build a single formatted context string from docs and intent.
Packs are included by relevance; total output is capped by max_total_chars.
When log_usage is True, logs pack names and approximate token count.
"""
sections = []
used = 0
pack_names: List[str] = []
# 1. Form schema (compact, always useful)
form_schema = docs.get("formSchema") or {}
if form_schema:
schema_str = json.dumps(form_schema.get("example") or form_schema)[:800]
block = f"## Form structure\n{schema_str}\n"
if used + len(block) <= max_total_chars:
sections.append(block)
used += len(block)
pack_names.append("formSchema")
# 2. Field types (always include, truncated)
block = _pack_field_types(docs, max_chars_per_pack)
if block and used + len(block) <= max_total_chars:
sections.append(block)
used += len(block)
pack_names.append("fieldTypes")
# 3. Conditional logic (if intent says so)
if intent.get("needs_conditional_logic"):
block = _pack_conditional_logic(docs, max_chars_per_pack)
if block and used + len(block) <= max_total_chars:
sections.append(block)
used += len(block)
pack_names.append("conditionalLogic")
# 4. Document extraction (if intent says so)
if intent.get("needs_document_extraction"):
block = _pack_doc_extraction(docs, max_chars_per_pack)
if block and used + len(block) <= max_total_chars:
sections.append(block)
used += len(block)
pack_names.append("docExtraction")
# 5. Examples (1–3 matching intent)
block = _pack_examples(docs, max_chars_per_pack, intent)
if block and used + len(block) <= max_total_chars:
sections.append(block)
used += len(block)
pack_names.append("examples")
# 6. Best practices (short)
practices = docs.get("bestPractices") or []
if practices and used < max_total_chars:
block = "## Best practices\n" + "\n".join(f"- {p}" for p in practices[:8])
block = _truncate(block, max_total_chars - used)
if block:
sections.append(block)
used += len(block)
pack_names.append("bestPractices")
out = "\n\n".join(sections) if sections else ""
if log_usage and pack_names:
approx_tokens = len(out) // CHARS_PER_TOKEN
logger.info(
"context_packs built packs=%s total_chars=%s approx_tokens=%s",
pack_names,
len(out),
approx_tokens,
)
return out