RandomZ / app /tests /test_content_similar_api.py
StormShadow308's picture
- Added support for RICS survey levels (1, 2, 3) in document uploads and reports, allowing for better tier management and retrieval filtering.
b76f199
Raw
History Blame Contribute Delete
4.33 kB
"""HTTP tests for POST /content/similar (library + draft overlap)."""
import pytest
from app.db.models import Document, IngestStatus
from app.models.schemas import SearchResult
class _FakeVS:
"""Vector store returning fixed chunks (no FAISS on disk)."""
def __init__(self, rows: list[SearchResult]) -> None:
self._rows = rows
def search(self, query: str, tenant_id: str, k: int = 10, **kwargs: object) -> list[SearchResult]:
return self._rows
@pytest.mark.asyncio
async def test_content_similar_returns_library_and_filename(
async_client: object,
test_db: object,
monkeypatch: pytest.MonkeyPatch,
) -> None:
doc_id = "aaaaaaaa-aaaa-aaaa-aaaa-aaaaaaaaaaaa"
tenant = "tenant_test"
doc = Document(
id=doc_id,
tenant_id=tenant,
filename="rics_guidance.pdf",
file_path="/tmp/rics_guidance.pdf",
status=IngestStatus.complete,
)
test_db.add(doc)
await test_db.commit()
hits = [
SearchResult(
chunk_id="chunk-1",
doc_id=doc_id,
tenant_id=tenant,
text="Approved Document guidance on damp proof courses and minimum heights",
score=0.35,
section_type="paragraph",
),
]
monkeypatch.setattr(
"app.services.content_similarity.get_vectorstore",
lambda: _FakeVS(hits),
)
res = await async_client.post(
"/content/similar",
json={
"text": "damp proof course minimum height regulations above ground level",
"section_code": "E4",
"peer_sections": {},
},
headers={"X-Tenant-ID": tenant},
)
assert res.status_code == 200
data = res.json()
assert len(data["library_matches"]) == 1
assert data["library_matches"][0]["document_id"] == doc_id
assert data["library_matches"][0]["filename"] == "rics_guidance.pdf"
assert "damp" in data["library_matches"][0]["snippet"].lower()
@pytest.mark.asyncio
async def test_content_similar_excludes_document_ids(
async_client: object,
test_db: object,
monkeypatch: pytest.MonkeyPatch,
) -> None:
doc_id = "bbbbbbbb-bbbb-bbbb-bbbb-bbbbbbbbbbbb"
tenant = "tenant_test"
doc = Document(
id=doc_id,
tenant_id=tenant,
filename="survey.docx",
file_path="/tmp/survey.docx",
status=IngestStatus.complete,
)
test_db.add(doc)
await test_db.commit()
hits = [
SearchResult(
chunk_id="c1",
doc_id=doc_id,
tenant_id=tenant,
text="Only chunk from excluded survey file",
score=0.9,
section_type="paragraph",
),
]
monkeypatch.setattr(
"app.services.content_similarity.get_vectorstore",
lambda: _FakeVS(hits),
)
res = await async_client.post(
"/content/similar",
json={
"text": "matching text long enough for library search twelve",
"section_code": "D",
"peer_sections": {},
"exclude_document_ids": [doc_id],
},
headers={"X-Tenant-ID": tenant},
)
assert res.status_code == 200
data = res.json()
assert data["library_matches"] == []
assert "excluded" in data["message"].lower() or "filters" in data["message"].lower()
@pytest.mark.asyncio
async def test_content_similar_draft_overlap_without_library(
async_client: object,
monkeypatch: pytest.MonkeyPatch,
) -> None:
monkeypatch.setattr(
"app.services.content_similarity.get_vectorstore",
lambda: _FakeVS([]),
)
text = (
"planning permission required for loft conversion and rear extension "
"under permitted development rules"
)
peer = (
"planning permission required for loft conversion and rear extension "
"under permitted development rules 2024 update"
)
res = await async_client.post(
"/content/similar",
json={
"text": text,
"section_code": "E2",
"peer_sections": {"I1": peer},
},
headers={"X-Tenant-ID": "tenant_test"},
)
assert res.status_code == 200
data = res.json()
assert data["draft_overlaps"]
assert data["draft_overlaps"][0]["other_section_code"] == "I1"