Spaces:
Sleeping
Sleeping
| """ | |
| Program dossier — komplet dokumentów + wymagania + gotowość doradcy. | |
| Bez paczki plików (regulamin, RWP, wytyczne, ogłoszenie, akty) jesteśmy „ślepi”: | |
| nie da się sensownie opisać programu, zmatchować firmę ani zasiać sekcji wniosku. | |
| Pipeline: | |
| 1) fetch strony naboru | |
| 2) extract regulation pack | |
| 3) HTTP-verify + persist na grant | |
| 4) multi-doc ingest (snapshot + RAG) | |
| 5) extract advisor brief (sekcje, załączniki, reguły, braki) | |
| """ | |
| from __future__ import annotations | |
| import logging | |
| import os | |
| from datetime import datetime, timezone | |
| from typing import Any, Dict, List, Optional | |
| from sqlalchemy.orm import Session | |
| from core.grants.completeness import grant_dict_from_row | |
| from core.grants.models import Grant | |
| from core.grants.regulation_pack import ( | |
| enrich_item_with_regulation_pack, | |
| merge_regulation_pack, | |
| ) | |
| from core.grants.regulation_url_quality import ( | |
| infer_doc_role, | |
| is_valid_regulation_url, | |
| ) | |
| logger = logging.getLogger(__name__) | |
| # Roles that count as "we can see the rules" | |
| _RULE_ROLES = frozenset({"regulamin", "rwp", "wytyczne", "isap", "eurlex", "pdf"}) | |
| _MIN_PACK_FOR_ADVICE = int(os.environ.get("DOSSIER_MIN_DOCS_FOR_ADVICE", "1")) | |
| def _now() -> str: | |
| return datetime.now(timezone.utc).isoformat() | |
| def find_grant_row(db: Session, grant_ref: str) -> Optional[Grant]: | |
| """Find by source_id or internal UUID.""" | |
| if not grant_ref: | |
| return None | |
| row = db.query(Grant).filter(Grant.source_id == grant_ref).first() | |
| if row: | |
| return row | |
| return db.query(Grant).filter(Grant.id == grant_ref).first() | |
| def dossier_readiness(docs: List[Dict[str, Any]], primary: str) -> Dict[str, Any]: | |
| """ | |
| blind | partial | ready | |
| - blind: brak dokumentów regułowych | |
| - partial: jest primary lub 1+ doc, ale mało ról | |
| - ready: primary regułowy + (pack>=2 lub PDF/rwp/wytyczne) | |
| """ | |
| roles = {(d.get("role") or "").lower() for d in docs} | |
| rule_docs = [ | |
| d | |
| for d in docs | |
| if (d.get("role") or "").lower() in _RULE_ROLES | |
| or str(d.get("url") or "").lower().endswith(".pdf") | |
| ] | |
| has_primary = bool(primary and (is_valid_regulation_url(primary) or primary.lower().endswith(".pdf"))) | |
| has_rules = bool(rule_docs) or has_primary | |
| multi = len(docs) >= 2 | |
| has_rwp_or_wytyczne = bool(roles & {"rwp", "wytyczne", "regulamin"}) | |
| if not has_rules: | |
| level = "blind" | |
| message = ( | |
| "Brak dokumentów regułowych (regulamin/RWP/wytyczne). " | |
| "Doradca nie może wiarygodnie opisać wymagań ani dopasować firmy." | |
| ) | |
| elif has_primary and (multi or has_rwp_or_wytyczne): | |
| level = "ready" | |
| message = "Komplet wystarczający do matchu i szkieletu sekcji wniosku." | |
| else: | |
| level = "partial" | |
| message = ( | |
| "Częściowy pakiet — jest co najmniej jeden dokument, " | |
| "ale warto dociągnąć pełną paczkę ze strony naboru." | |
| ) | |
| return { | |
| "level": level, | |
| "message": message, | |
| "has_primary_regulation": has_primary, | |
| "document_count": len(docs), | |
| "rule_document_count": len(rule_docs), | |
| "roles_present": sorted(roles), | |
| "can_advise": level in ("ready", "partial") and has_rules, | |
| "can_seed_sections": has_rules, | |
| "is_blind": level == "blind", | |
| } | |
| def _advisor_brief_from_snapshots(program: str, name: str, urls: List[str]) -> Dict[str, Any]: | |
| """Pull key_rules, sections, attachments from regulation snapshots + text extract.""" | |
| from core.grants.advisor_brief_extract import ( | |
| build_advisor_brief_from_text, | |
| is_brief_usable, | |
| merge_briefs, | |
| ) | |
| from core.search.regulation_snapshot import regulation_snapshot_store | |
| key_rules: List[str] = [] | |
| required_sections: List[str] = [] | |
| required_attachments: List[str] = [] | |
| legal_ids: List[str] = [] | |
| snapshots_used = 0 | |
| seen_rules: set = set() | |
| seen_sec: set = set() | |
| seen_att: set = set() | |
| text_briefs: List[Dict[str, Any]] = [] | |
| keys = [] | |
| if program and name: | |
| keys.append(f"{program}|{name}"[:80].upper()) | |
| if program: | |
| keys.append(str(program).upper()[:80]) | |
| if name: | |
| keys.append(str(name).upper()[:80]) | |
| for key in keys: | |
| for snap in regulation_snapshot_store.get_snapshots_for_program(key, limit=8): | |
| snapshots_used += 1 | |
| for r in list(getattr(snap, "key_rules", None) or [])[:8]: | |
| s = str(r).strip() | |
| if s and s.lower() not in seen_rules: | |
| seen_rules.add(s.lower()) | |
| key_rules.append(s[:300]) | |
| for s in list(getattr(snap, "required_sections", None) or [])[:12]: | |
| t = str(s).strip() | |
| if t and t.lower() not in seen_sec: | |
| seen_sec.add(t.lower()) | |
| required_sections.append(t[:200]) | |
| for a in list(getattr(snap, "required_attachments", None) or [])[:12]: | |
| t = str(a).strip() | |
| if t and t.lower() not in seen_att: | |
| seen_att.add(t.lower()) | |
| required_attachments.append(t[:200]) | |
| raw = (getattr(snap, "raw_text", None) or "")[:50000] | |
| if raw and len(raw.strip()) >= 40: | |
| text_briefs.append( | |
| build_advisor_brief_from_text( | |
| raw, program=program, name=name | |
| ) | |
| ) | |
| for url in urls[:8]: | |
| snap = regulation_snapshot_store.get_latest_by_source_url(url) | |
| if not snap: | |
| continue | |
| snapshots_used += 1 | |
| text = (getattr(snap, "raw_text", None) or "")[:50000] | |
| if text: | |
| try: | |
| from core.document_intel.legal_citations import extract_legal_citations | |
| leg = extract_legal_citations(text) or {} | |
| for cid in list(leg.get("celex_ids") or [])[:10]: | |
| if cid not in legal_ids: | |
| legal_ids.append(cid) | |
| except Exception: | |
| pass | |
| text_briefs.append( | |
| build_advisor_brief_from_text( | |
| text, source_url=url, program=program, name=name | |
| ) | |
| ) | |
| # Merge snapshot field harvest with pure text extraction | |
| field_brief = { | |
| "key_rules": key_rules[:20], | |
| "required_sections": required_sections[:20], | |
| "required_attachments": required_attachments[:15], | |
| "attention_points": [], | |
| "funding_limits": [], | |
| "eligibility_signals": [], | |
| "legal_ids": legal_ids[:15], | |
| } | |
| merged = merge_briefs(field_brief, *text_briefs) if text_briefs else field_brief | |
| if not merged.get("attention_points"): | |
| from core.grants.advisor_brief_extract import build_attention_points | |
| merged["attention_points"] = build_attention_points( | |
| key_rules=list(merged.get("key_rules") or []), | |
| attachments=list(merged.get("required_attachments") or []), | |
| eligibility=list(merged.get("eligibility_signals") or []), | |
| limits=list(merged.get("funding_limits") or []), | |
| ) | |
| merged["legal_ids"] = legal_ids[:15] or list(merged.get("legal_ids") or [])[:15] | |
| merged["snapshots_consulted"] = snapshots_used | |
| merged["usable"] = is_brief_usable(merged) | |
| return { | |
| "key_rules": list(merged.get("key_rules") or [])[:20], | |
| "required_sections": list(merged.get("required_sections") or [])[:20], | |
| "required_attachments": list(merged.get("required_attachments") or [])[:15], | |
| "legal_ids": list(merged.get("legal_ids") or [])[:15], | |
| "funding_limits": list(merged.get("funding_limits") or [])[:10], | |
| "eligibility_signals": list(merged.get("eligibility_signals") or [])[:10], | |
| "snapshots_consulted": snapshots_used, | |
| "attention_points": list(merged.get("attention_points") or [])[:10], | |
| "usable": bool(merged.get("usable")), | |
| "extraction_method": merged.get("extraction_method") or "advisor_brief_extract.v1", | |
| } | |
| def _attention_points(rules: List[str], attachments: List[str]) -> List[str]: | |
| """Backward-compatible wrapper — prefer advisor_brief_extract.build_attention_points.""" | |
| from core.grants.advisor_brief_extract import build_attention_points | |
| return build_attention_points(key_rules=rules, attachments=attachments) | |
| async def ensure_program_dossier( | |
| db: Session, | |
| *, | |
| grant_ref: str = "", | |
| grant_row: Optional[Grant] = None, | |
| fetch_page: bool = True, | |
| ingest: bool = True, | |
| max_ingest_docs: int = 6, | |
| ) -> Dict[str, Any]: | |
| """ | |
| Główny entry: dociąga paczkę, zapisuje na grant, ingestuje, buduje brief doradcy. | |
| """ | |
| row = grant_row or find_grant_row(db, grant_ref) | |
| if not row: | |
| return {"ok": False, "error": "grant_not_found", "grant_ref": grant_ref} | |
| item = grant_dict_from_row(row) | |
| enriched = await enrich_item_with_regulation_pack(item, fetch_page=fetch_page) | |
| pack = merge_regulation_pack( | |
| enriched, | |
| html="", # already applied in enrich | |
| page_url=str(enriched.get("official_page_url") or ""), | |
| ) | |
| # prefer enriched pack | |
| docs = enriched.get("regulation_documents") or pack.get("regulation_documents") or [] | |
| urls = enriched.get("regulation_urls") or pack.get("regulation_urls") or [] | |
| primary = str( | |
| enriched.get("precise_regulation_url") | |
| or pack.get("precise_regulation_url") | |
| or "" | |
| ) | |
| page = str( | |
| enriched.get("official_page_url") | |
| or pack.get("official_page_url") | |
| or item.get("url") | |
| or "" | |
| ) | |
| raw = dict(row.raw_data or {}) | |
| raw["regulation_documents"] = docs | |
| raw["regulation_urls"] = urls | |
| raw["regulation_pack_size"] = len(docs) | |
| raw["dossier_built_at"] = _now() | |
| if primary: | |
| raw["precise_regulation_url"] = primary | |
| raw["regulation_url"] = primary | |
| row.precise_regulation_url = primary | |
| row.regulation_url = primary | |
| if page and page.startswith("http"): | |
| raw["official_page_url"] = page | |
| try: | |
| if not row.official_page_url: | |
| row.official_page_url = page | |
| except Exception: | |
| pass | |
| row.raw_data = raw | |
| ingest_stats: Dict[str, Any] = {"skipped": True} | |
| if ingest and urls: | |
| try: | |
| from core.grants.regulation_ingest import _ingest_single | |
| program = row.program or row.source or "UNKNOWN" | |
| name = row.name or "regulamin" | |
| gid = row.source_id or row.id | |
| ok = fail = 0 | |
| for d in docs[:max_ingest_docs]: | |
| url = d.get("url") or "" | |
| if not url.startswith("http"): | |
| continue | |
| # skip pure announcement HTML if we already have PDFs | |
| role = (d.get("role") or "").lower() | |
| if role == "ogloszenie" and any( | |
| str(x.get("url") or "").lower().endswith(".pdf") for x in docs | |
| ): | |
| continue | |
| res = await _ingest_single( | |
| url, | |
| program=program, | |
| name=name, | |
| grant_id=str(gid), | |
| doc_role=d.get("role") or infer_doc_role(url), | |
| ) | |
| if res.get("ok"): | |
| ok += 1 | |
| d["ingested"] = True | |
| else: | |
| fail += 1 | |
| d["ingested"] = False | |
| d["ingest_error"] = res.get("reason") | |
| raw["regulation_documents"] = docs | |
| row.raw_data = raw | |
| ingest_stats = {"skipped": False, "ok": ok, "failed": fail, "attempted": ok + fail} | |
| except Exception as e: | |
| logger.warning("[ProgramDossier] ingest failed: %s", e) | |
| ingest_stats = {"skipped": False, "error": str(e)[:160]} | |
| db.commit() | |
| db.refresh(row) | |
| brief = _advisor_brief_from_snapshots( | |
| row.program or "", | |
| row.name or "", | |
| urls or ([primary] if primary else []), | |
| ) | |
| readiness = dossier_readiness(docs, primary) | |
| dossier = { | |
| "ok": True, | |
| "grant_id": row.source_id or row.id, | |
| "internal_id": row.id, | |
| "name": row.name, | |
| "program": row.program, | |
| "operator": row.operator, | |
| "status": row.status, | |
| "deadline": row.deadline, | |
| "official_page_url": page, | |
| "precise_regulation_url": primary, | |
| "regulation_documents": docs, | |
| "regulation_urls": urls, | |
| "document_count": len(docs), | |
| "readiness": readiness, | |
| "advisor_brief": brief, | |
| "ingest": ingest_stats, | |
| "built_at": raw.get("dossier_built_at"), | |
| # Dane potrzebne do matchu / wniosku | |
| "firm_data_needed": _firm_data_needed(brief, readiness), | |
| "section_seed_hints": brief.get("required_sections") or [], | |
| } | |
| raw["program_dossier"] = { | |
| "readiness": readiness, | |
| "document_count": len(docs), | |
| "built_at": dossier["built_at"], | |
| "advisor_brief": { | |
| "key_rules": brief.get("key_rules", [])[:10], | |
| "required_sections": brief.get("required_sections", [])[:12], | |
| "required_attachments": brief.get("required_attachments", [])[:10], | |
| "attention_points": brief.get("attention_points", []), | |
| }, | |
| } | |
| row.raw_data = raw | |
| db.commit() | |
| return dossier | |
| def _firm_data_needed(brief: Dict[str, Any], readiness: Dict[str, Any]) -> List[str]: | |
| needed = [ | |
| "NIP / dane rejestrowe firmy", | |
| "Opis inwestycji (cele, zakres, terminy)", | |
| "Szacunkowy budżet i wkład własny", | |
| "PKD / branża", | |
| "Status MŚP (zatrudnienie, powiązania)", | |
| ] | |
| if brief.get("required_attachments"): | |
| needed.append("Załączniki wskazane w regulaminie (lista w dossier)") | |
| if readiness.get("is_blind"): | |
| needed.insert( | |
| 0, | |
| "UWAGA: brak regulaminu w systemie — dołącz PDF regulaminu ręcznie lub odśwież dossier", | |
| ) | |
| return needed | |
| def dossier_from_grant_dict(item: Dict[str, Any]) -> Dict[str, Any]: | |
| """Lightweight dossier from already-loaded grant dict (no DB write).""" | |
| docs = list(item.get("regulation_documents") or []) | |
| if not docs: | |
| pack = merge_regulation_pack(item) | |
| docs = pack.get("regulation_documents") or [] | |
| primary = pack.get("precise_regulation_url") or "" | |
| urls = pack.get("regulation_urls") or [] | |
| page = pack.get("official_page_url") or item.get("url") or "" | |
| else: | |
| primary = str(item.get("precise_regulation_url") or item.get("regulation_url") or "") | |
| urls = list(item.get("regulation_urls") or [d.get("url") for d in docs]) | |
| page = item.get("official_page_url") or item.get("url") or "" | |
| readiness = dossier_readiness(docs, primary) | |
| stored = item.get("program_dossier") if isinstance(item.get("program_dossier"), dict) else {} | |
| brief = stored.get("advisor_brief") or { | |
| "key_rules": [], | |
| "required_sections": [], | |
| "required_attachments": [], | |
| "attention_points": readiness.get("message") and [readiness["message"]] or [], | |
| } | |
| return { | |
| "ok": True, | |
| "name": item.get("name"), | |
| "program": item.get("program"), | |
| "official_page_url": page, | |
| "precise_regulation_url": primary, | |
| "regulation_documents": docs, | |
| "regulation_urls": urls, | |
| "document_count": len(docs), | |
| "readiness": readiness, | |
| "advisor_brief": brief, | |
| "firm_data_needed": _firm_data_needed(brief if isinstance(brief, dict) else {}, readiness), | |
| "cached": True, | |
| } | |
| async def ensure_dossier_for_project_context( | |
| db: Session, | |
| external_context: Dict[str, Any], | |
| ) -> Dict[str, Any]: | |
| """ | |
| Przy wyborze programu w projekcie — buduje dossier i wrzuca do external_context. | |
| """ | |
| ext = dict(external_context or {}) | |
| selected = ext.get("selected_grant") or {} | |
| grant_ref = ( | |
| selected.get("id") | |
| or selected.get("source_id") | |
| or selected.get("grant_id") | |
| or ext.get("grant_id") | |
| or "" | |
| ) | |
| dossier = await ensure_program_dossier(db, grant_ref=str(grant_ref), fetch_page=True, ingest=True) | |
| if not dossier.get("ok"): | |
| # still store blind flag | |
| ext["program_dossier"] = { | |
| "readiness": {"level": "blind", "is_blind": True, "can_advise": False}, | |
| "error": dossier.get("error"), | |
| } | |
| return ext | |
| ext["program_dossier"] = dossier | |
| ext["precise_regulation_url"] = dossier.get("precise_regulation_url") or ext.get( | |
| "precise_regulation_url" | |
| ) | |
| ext["regulation_url"] = dossier.get("precise_regulation_url") or ext.get("regulation_url") | |
| ext["regulation_urls"] = dossier.get("regulation_urls") or ext.get("regulation_urls") or [] | |
| ext["regulation_documents"] = dossier.get("regulation_documents") or [] | |
| if dossier.get("official_page_url"): | |
| ext["official_page_url"] = dossier["official_page_url"] | |
| # selected_grant enrichment | |
| if isinstance(selected, dict): | |
| selected = dict(selected) | |
| selected["precise_regulation_url"] = dossier.get("precise_regulation_url") | |
| selected["regulation_documents"] = dossier.get("regulation_documents") | |
| selected["regulation_urls"] = dossier.get("regulation_urls") | |
| selected["dossier_readiness"] = (dossier.get("readiness") or {}).get("level") | |
| ext["selected_grant"] = selected | |
| # F1: leave structure_only only while still blind | |
| try: | |
| from core.projects.generation_consent import sync_grounding_after_dossier_update | |
| ext = sync_grounding_after_dossier_update(ext) | |
| except Exception: | |
| pass | |
| # Instrument-first: rebuild schema from dossier brief + grant labels | |
| try: | |
| from core.projects.instrument_schema import ( | |
| build_instrument_schema, | |
| schema_seed_sections, | |
| ) | |
| brief = dossier.get("advisor_brief") if isinstance(dossier.get("advisor_brief"), dict) else {} | |
| reg_text_parts: list[str] = [] | |
| for r in list(brief.get("key_rules") or [])[:12]: | |
| reg_text_parts.append(str(r)) | |
| for s in list(brief.get("required_sections") or [])[:12]: | |
| reg_text_parts.append(str(s)) | |
| schema = build_instrument_schema( | |
| program_type=str( | |
| ext.get("program_type") | |
| or selected.get("type") | |
| or dossier.get("program") | |
| or "" | |
| ), | |
| program_name=str( | |
| ext.get("program_name") | |
| or selected.get("name") | |
| or dossier.get("name") | |
| or "" | |
| ), | |
| grant_id=str(grant_ref or ""), | |
| description=str(ext.get("project_description") or ""), | |
| regulation_text="\n".join(reg_text_parts), | |
| advisor_brief=brief, | |
| ) | |
| # Attach light legal refs from pack roles | |
| legal_refs = [] | |
| for d in dossier.get("regulation_documents") or []: | |
| if not isinstance(d, dict): | |
| continue | |
| role = (d.get("role") or "").lower() | |
| url = d.get("url") or "" | |
| if url and role in ("isap", "eurlex", "regulamin", "rwp", "wytyczne", "pdf"): | |
| legal_refs.append({"url": url, "role": role, "title": d.get("title") or ""}) | |
| if legal_refs: | |
| schema["legal_references"] = legal_refs[:12] | |
| ext["instrument_schema"] = schema | |
| ext["instrument_program_type"] = schema.get("family") | |
| ext["instrument_kind"] = schema.get("instrument_kind") | |
| seeds = schema_seed_sections(schema) | |
| if seeds: | |
| ext["instrument_required_sections"] = seeds | |
| if not ext.get("required_sections"): | |
| ext["required_sections"] = seeds | |
| if brief: | |
| ext["advisor_brief"] = brief | |
| except Exception: | |
| pass | |
| return ext | |