Spaces:
Sleeping
Sleeping
| """ | |
| Docs-to-context-pack mapping and relevance rules for richer AI form generation. | |
| Consumes LLM documentation (from docs API) and builds token-budgeted prompt sections | |
| based on detected intent (conditional logic, document extraction, advanced fields). | |
| """ | |
| from typing import Any, Dict, List, Optional, Tuple | |
| import json | |
| import logging | |
| logger = logging.getLogger(__name__) | |
| # Approximate tokens from chars (conservative for Latin text) | |
| CHARS_PER_TOKEN = 4 | |
| # --- Intent detection keywords (plan: Smart Context Strategy) --- | |
| CONDITIONAL_LOGIC_SIGNALS = [ | |
| "if", "when", "depends", "show", "hide", "branch", "conditional", | |
| "only when", "based on", "depending on", "skip", "reveal", | |
| ] | |
| DOC_EXTRACTION_SIGNALS = [ | |
| "resume", "cv", "invoice", "receipt", "id card", "parse", "extract", | |
| "document extraction", "upload and extract", "pdf", "scan", | |
| ] | |
| ADVANCED_FIELD_SIGNALS = [ | |
| "matrix", "ranking", "nps", "likert", "rating", "score", "slider", | |
| "validation", "validate", "scale", "grid", | |
| ] | |
| # --- Pack names and doc section mapping --- | |
| PACK_FIELD_TYPES = "fieldTypes" | |
| PACK_CONDITIONAL_LOGIC = "conditionalLogic" | |
| PACK_DOC_EXTRACTION = "aiFields" | |
| PACK_EXAMPLES = "exampleForms" | |
| PACK_BEST_PRACTICES = "bestPractices" | |
| PACK_FORM_SCHEMA = "formSchema" | |
| # Default character budgets (approx 4 chars per token; ~500 tokens per pack) | |
| MAX_CHARS_PER_PACK = 2000 | |
| MAX_TOTAL_CONTEXT_CHARS = 8000 | |
| def detect_intent( | |
| description: str = "", | |
| user_request: Optional[str] = None, | |
| current_fields: Optional[List[Dict[str, Any]]] = None, | |
| conversation_context: Optional[List[str]] = None, | |
| ) -> Dict[str, bool]: | |
| """ | |
| Detect which context packs are relevant from user text and current form. | |
| Returns flags: needs_conditional_logic, needs_document_extraction, needs_advanced_fields. | |
| """ | |
| text = " ".join( | |
| filter( | |
| None, | |
| [description or "", user_request or ""] | |
| + (conversation_context or []), | |
| ) | |
| ).lower() | |
| has_conditional = any(s in text for s in CONDITIONAL_LOGIC_SIGNALS) | |
| has_doc_extraction = any(s in text for s in DOC_EXTRACTION_SIGNALS) | |
| has_advanced = any(s in text for s in ADVANCED_FIELD_SIGNALS) | |
| # If current form already has conditional logic or document-extraction, include those packs for refine | |
| if current_fields: | |
| for f in current_fields: | |
| if isinstance(f, dict): | |
| if (f.get("conditionalLogic") or {}).get("enabled"): | |
| has_conditional = True | |
| if f.get("type") == "document-extraction": | |
| has_doc_extraction = True | |
| return { | |
| "needs_conditional_logic": has_conditional, | |
| "needs_document_extraction": has_doc_extraction, | |
| "needs_advanced_fields": has_advanced, | |
| } | |
| def _truncate(text: str, max_chars: int) -> str: | |
| if len(text) <= max_chars: | |
| return text | |
| return text[: max_chars - 3].rstrip() + "..." | |
| def _pack_field_types(docs: Dict[str, Any], max_chars: int) -> str: | |
| """Build FieldTypes pack from docs.fieldTypes (summary + key types).""" | |
| field_types = docs.get("fieldTypes") or {} | |
| if not field_types: | |
| return "" | |
| lines = ["## Field types reference\n"] | |
| for name, spec in list(field_types.items())[:25]: | |
| if not isinstance(spec, dict): | |
| continue | |
| desc = spec.get("description", "") | |
| required = spec.get("requiredProperties", []) | |
| lines.append(f"- **{name}**: {desc[:200]}") | |
| if required: | |
| lines.append(f" Required: {', '.join(required)}") | |
| ex = spec.get("exampleJSON") | |
| if ex: | |
| lines.append(f" Example: {json.dumps(ex)[:300]}") | |
| out = "\n".join(lines) | |
| return _truncate(out, max_chars) | |
| def _pack_conditional_logic(docs: Dict[str, Any], max_chars: int) -> str: | |
| """Build ConditionalLogic pack from docs.conditionalLogic.""" | |
| cl = docs.get("conditionalLogic") or {} | |
| if not cl: | |
| return "" | |
| lines = [ | |
| "## Conditional logic\n", | |
| (cl.get("overview") or "")[:500], | |
| "\n### Operators (use in conditions): ", | |
| ] | |
| ops = cl.get("operators") or {} | |
| for op_name, op_spec in list(ops.items())[:15]: | |
| if isinstance(op_spec, dict): | |
| lines.append(f"- {op_name}: {(op_spec.get('description') or '')[:150]}") | |
| lines.append("\n### Actions: show | hide | validate | cap_responses") | |
| actions = cl.get("actions") or {} | |
| for act_name, act_spec in list(actions.items())[:4]: | |
| if isinstance(act_spec, dict): | |
| ex = act_spec.get("example") | |
| if ex: | |
| lines.append(f"- {act_name}: {json.dumps(ex)[:200]}") | |
| examples = cl.get("examples") or [] | |
| for ex in examples[:2]: | |
| if isinstance(ex, dict) and ex.get("json"): | |
| lines.append(f"Example: {json.dumps(ex['json'])[:250]}") | |
| out = "\n".join(lines) | |
| return _truncate(out, max_chars) | |
| def _pack_doc_extraction(docs: Dict[str, Any], max_chars: int) -> str: | |
| """Build DocExtraction pack from docs.aiFields.documentExtraction.""" | |
| ai = docs.get("aiFields") or {} | |
| doc_ext = ai.get("documentExtraction") if isinstance(ai, dict) else None | |
| if not doc_ext: | |
| return "" | |
| lines = [ | |
| "## Document extraction field (document-extraction)\n", | |
| (doc_ext.get("description") or "")[:400], | |
| "\n### customFields: array of { id, name, description, fieldType, required, editable }", | |
| "fieldType: single_value | list | number | date", | |
| "Always set acceptedFileTypes (e.g. ['pdf','doc','docx','png','jpg','jpeg']) and maxFileSize in bytes (e.g. 10485760 for 10MB). Do not leave customFields empty.", | |
| "verificationPrompt: descriptive text only (full sentences) stating how to verify the candidate against extracted data. Use the user's must-have/role requirements (e.g. from clarifying answers: years of experience, degree, skills). Never put only a number (e.g. 80) — invalid; use criteria like 'Candidate must have 2+ years experience, masters in CS, and LangGraph expertise.'", | |
| ] | |
| config = doc_ext.get("configuration") or {} | |
| if isinstance(config, dict): | |
| cf = config.get("customFields") or {} | |
| if isinstance(cf, dict) and cf.get("fieldTypes"): | |
| lines.append("Field types: " + json.dumps(list((cf["fieldTypes"] or {}).keys()))) | |
| example_configs = doc_ext.get("exampleConfigurations") or [] | |
| for ex in example_configs[:2]: | |
| if isinstance(ex, dict) and ex.get("json"): | |
| lines.append(f"Example: {json.dumps(ex['json'])[:400]}") | |
| out = "\n".join(lines) | |
| return _truncate(out, max_chars) | |
| def _pack_examples(docs: Dict[str, Any], max_chars: int, intent: Dict[str, bool]) -> str: | |
| """Build Examples pack: 1–3 examples matching intent.""" | |
| examples = docs.get("exampleForms") or [] | |
| if not examples: | |
| return "" | |
| chosen = [] | |
| for ex in examples: | |
| if not isinstance(ex, dict) or not ex.get("json"): | |
| continue | |
| title = (ex.get("title") or "").lower() | |
| features = ex.get("featuresUsed") or [] | |
| if intent.get("needs_document_extraction") and ( | |
| "document-extraction" in features or "resume" in title or "extraction" in title | |
| ): | |
| chosen.append(ex) | |
| elif intent.get("needs_conditional_logic") and "conditionalLogic" in str(features): | |
| chosen.append(ex) | |
| elif not chosen: | |
| chosen.append(ex) | |
| if len(chosen) >= 3: | |
| break | |
| if not chosen: | |
| chosen = examples[:2] | |
| lines = ["## Example forms (reference only)\n"] | |
| for ex in chosen: | |
| lines.append(f"### {ex.get('title', 'Form')}") | |
| lines.append(json.dumps(ex.get("json") or {}, indent=2)[:800]) | |
| out = "\n".join(lines) | |
| return _truncate(out, max_chars) | |
| def build_context_packs( | |
| docs: Dict[str, Any], | |
| intent: Dict[str, bool], | |
| max_chars_per_pack: int = MAX_CHARS_PER_PACK, | |
| max_total_chars: int = MAX_TOTAL_CONTEXT_CHARS, | |
| log_usage: bool = True, | |
| ) -> str: | |
| """ | |
| Build a single formatted context string from docs and intent. | |
| Packs are included by relevance; total output is capped by max_total_chars. | |
| When log_usage is True, logs pack names and approximate token count. | |
| """ | |
| sections = [] | |
| used = 0 | |
| pack_names: List[str] = [] | |
| # 1. Form schema (compact, always useful) | |
| form_schema = docs.get("formSchema") or {} | |
| if form_schema: | |
| schema_str = json.dumps(form_schema.get("example") or form_schema)[:800] | |
| block = f"## Form structure\n{schema_str}\n" | |
| if used + len(block) <= max_total_chars: | |
| sections.append(block) | |
| used += len(block) | |
| pack_names.append("formSchema") | |
| # 2. Field types (always include, truncated) | |
| block = _pack_field_types(docs, max_chars_per_pack) | |
| if block and used + len(block) <= max_total_chars: | |
| sections.append(block) | |
| used += len(block) | |
| pack_names.append("fieldTypes") | |
| # 3. Conditional logic (if intent says so) | |
| if intent.get("needs_conditional_logic"): | |
| block = _pack_conditional_logic(docs, max_chars_per_pack) | |
| if block and used + len(block) <= max_total_chars: | |
| sections.append(block) | |
| used += len(block) | |
| pack_names.append("conditionalLogic") | |
| # 4. Document extraction (if intent says so) | |
| if intent.get("needs_document_extraction"): | |
| block = _pack_doc_extraction(docs, max_chars_per_pack) | |
| if block and used + len(block) <= max_total_chars: | |
| sections.append(block) | |
| used += len(block) | |
| pack_names.append("docExtraction") | |
| # 5. Examples (1–3 matching intent) | |
| block = _pack_examples(docs, max_chars_per_pack, intent) | |
| if block and used + len(block) <= max_total_chars: | |
| sections.append(block) | |
| used += len(block) | |
| pack_names.append("examples") | |
| # 6. Best practices (short) | |
| practices = docs.get("bestPractices") or [] | |
| if practices and used < max_total_chars: | |
| block = "## Best practices\n" + "\n".join(f"- {p}" for p in practices[:8]) | |
| block = _truncate(block, max_total_chars - used) | |
| if block: | |
| sections.append(block) | |
| used += len(block) | |
| pack_names.append("bestPractices") | |
| out = "\n\n".join(sections) if sections else "" | |
| if log_usage and pack_names: | |
| approx_tokens = len(out) // CHARS_PER_TOKEN | |
| logger.info( | |
| "context_packs built packs=%s total_chars=%s approx_tokens=%s", | |
| pack_names, | |
| len(out), | |
| approx_tokens, | |
| ) | |
| return out | |