Spaces:
Sleeping
Sleeping
| """Find similar content: semantic matches in the tenant vector index + draft overlaps.""" | |
| from __future__ import annotations | |
| import re | |
| from sqlalchemy.ext.asyncio import AsyncSession | |
| from app.models.schemas import ( | |
| DraftOverlapMatch, | |
| LibrarySimilarMatch, | |
| SimilarContentRequest, | |
| SimilarContentResponse, | |
| ) | |
| from app.services.provenance_enrichment import fetch_doc_filenames | |
| from app.vectorstore.factory import get_vectorstore | |
| _MIN_CHARS_LIBRARY = 12 | |
| _MIN_LINE_CHARS = 14 | |
| _TOKEN_RE = re.compile(r"[a-z0-9]+", re.IGNORECASE) | |
| def _tokens(s: str) -> set[str]: | |
| return {m.group(0).lower() for m in _TOKEN_RE.finditer(s)} | |
| def jaccard_similarity(a: str, b: str) -> float: | |
| """Token Jaccard similarity in ``[0, 1]``.""" | |
| ta, tb = _tokens(a), _tokens(b) | |
| if not ta or not tb: | |
| return 0.0 | |
| inter = len(ta & tb) | |
| union = len(ta | tb) | |
| return inter / union if union else 0.0 | |
| def find_draft_overlaps( | |
| text: str, | |
| section_code: str | None, | |
| peer_sections: dict[str, str], | |
| *, | |
| line_threshold: float = 0.42, | |
| block_threshold: float = 0.38, | |
| ) -> list[DraftOverlapMatch]: | |
| """Detect near-duplicate lines or blocks across section draft notes.""" | |
| text = text.strip() | |
| if not text: | |
| return [] | |
| out: list[DraftOverlapMatch] = [] | |
| seen: set[str] = set() | |
| current_lines = [ | |
| ln.strip() | |
| for ln in text.splitlines() | |
| if len(ln.strip()) >= _MIN_LINE_CHARS | |
| ] | |
| if not current_lines: | |
| current_lines = [text] if len(text) >= _MIN_LINE_CHARS else [] | |
| for other_code, peer_raw in peer_sections.items(): | |
| if section_code and other_code == section_code: | |
| continue | |
| peer_text = (peer_raw or "").strip() | |
| if not peer_text: | |
| continue | |
| blk_sim = jaccard_similarity(text, peer_text) | |
| if blk_sim >= block_threshold and len(text) >= _MIN_LINE_CHARS and len(peer_text) >= _MIN_LINE_CHARS: | |
| bkey = f"b:{section_code or ''}:{other_code}:{text[:80]}:{peer_text[:80]}" | |
| if bkey not in seen: | |
| seen.add(bkey) | |
| short_peer = peer_text if len(peer_text) <= 220 else peer_text[:217] + "…" | |
| short_you = text if len(text) <= 220 else text[:217] + "…" | |
| out.append( | |
| DraftOverlapMatch( | |
| other_section_code=other_code, | |
| overlap_kind="block", | |
| similarity=round(blk_sim, 4), | |
| your_preview=short_you, | |
| other_preview=short_peer, | |
| ) | |
| ) | |
| peer_lines = [ | |
| ln.strip() | |
| for ln in peer_text.splitlines() | |
| if len(ln.strip()) >= _MIN_LINE_CHARS | |
| ] | |
| for cl in current_lines: | |
| for pl in peer_lines: | |
| sim = jaccard_similarity(cl, pl) | |
| if sim < line_threshold: | |
| continue | |
| lkey = f"l:{other_code}:{cl[:60]}:{pl[:60]}" | |
| if lkey in seen: | |
| continue | |
| seen.add(lkey) | |
| out.append( | |
| DraftOverlapMatch( | |
| other_section_code=other_code, | |
| overlap_kind="line", | |
| similarity=round(sim, 4), | |
| your_preview=cl if len(cl) <= 200 else cl[:197] + "…", | |
| other_preview=pl if len(pl) <= 200 else pl[:197] + "…", | |
| ) | |
| ) | |
| out.sort(key=lambda m: m.similarity, reverse=True) | |
| return out[:24] | |
| async def scan_similar_content( | |
| db: AsyncSession, | |
| tenant_id: str, | |
| body: SimilarContentRequest, | |
| ) -> SimilarContentResponse: | |
| """Semantic search over indexed uploads plus lexical overlap across peer sections.""" | |
| draft_overlaps = find_draft_overlaps( | |
| body.text, | |
| body.section_code, | |
| body.peer_sections, | |
| ) | |
| q = body.text.strip() | |
| if len(q) < _MIN_CHARS_LIBRARY: | |
| msg = ( | |
| "Add a little more text (at least 12 characters) to search your uploaded library." | |
| if q | |
| else "Enter some notes to compare against your library." | |
| ) | |
| return SimilarContentResponse( | |
| library_matches=[], | |
| draft_overlaps=draft_overlaps, | |
| message=msg, | |
| ) | |
| from app.retrieval.vector_search import async_vs_search | |
| vs = get_vectorstore() | |
| exclude_docs = set(body.exclude_document_ids or []) | |
| # Pull extra candidates when excluding docs or deduping so we can still fill ``limit``. | |
| mult = 6 if exclude_docs else 4 | |
| fetch_k = min(max(body.limit * mult, 16), 80) | |
| raw = await async_vs_search(vs, q, tenant_id, k=fetch_k) | |
| if not raw: | |
| return SimilarContentResponse( | |
| library_matches=[], | |
| draft_overlaps=draft_overlaps, | |
| message="No indexed reference documents yet for this tenant, or nothing similar was found.", | |
| ) | |
| mx = max(r.score for r in raw) | |
| doc_ids = {r.doc_id for r in raw if r.doc_id} | |
| filenames = await fetch_doc_filenames(db, tenant_id, doc_ids) | |
| library: list[LibrarySimilarMatch] = [] | |
| seen_chunk_ids: set[str] = set() | |
| for r in raw: | |
| if r.doc_id in exclude_docs: | |
| continue | |
| if r.chunk_id in seen_chunk_ids: | |
| continue | |
| pct = 100.0 * r.score / mx if mx > 0 else 0.0 | |
| if pct < body.min_relevance_percent: | |
| continue | |
| seen_chunk_ids.add(r.chunk_id) | |
| snip = (r.text or "").strip() | |
| if len(snip) > 320: | |
| snip = snip[:317] + "…" | |
| library.append( | |
| LibrarySimilarMatch( | |
| chunk_id=r.chunk_id, | |
| document_id=r.doc_id, | |
| filename=filenames.get(r.doc_id), | |
| snippet=snip, | |
| relevance_percent=round(pct, 1), | |
| section_type=r.section_type or "paragraph", | |
| ) | |
| ) | |
| if len(library) >= body.limit: | |
| break | |
| msg = "" | |
| if not library and raw: | |
| msg = ( | |
| "No library matches after applying your filters (excluded documents and/or relevance threshold). " | |
| "Lower min_relevance_percent, clear exclude_document_ids, or upload additional reference files." | |
| ) | |
| return SimilarContentResponse( | |
| library_matches=library, | |
| draft_overlaps=draft_overlaps, | |
| message=msg, | |
| ) | |