Falsify / tests /test_detect.py
Aaryan Kumar
deploy to hugging face
1605cbb
Raw
History Blame Contribute Delete
3.69 kB
"""
Tests for the two-gate contradiction detector — no live LLM required.
* The ``--demo`` / ``pinned_target_id`` path bypasses both gates and must return a
deterministic high-confidence verdict (this is what keeps the live demo reliable).
* The Gate-2 path is exercised with a monkeypatched ``LLMGateway`` so we test the
accept/reject thresholding without any API key.
"""
from __future__ import annotations
import pytest
from falsify import graph_ops
from falsify.tasks import detect_contradictions
from falsify.tasks.detect_contradictions import ContradictionJudgement
pytestmark = pytest.mark.asyncio
async def test_pinned_target_is_deterministic():
"""Pinned mode returns exactly one high-confidence 'contradicts' verdict."""
out = await detect_contradictions("any new fact", pinned_target_id="E_qa")
assert len(out) == 1
assert out[0].target_id == "E_qa"
assert out[0].relation == "contradicts"
assert out[0].confidence >= 0.6
class _FakeHit:
def __init__(self, _id, score, payload):
self.id = _id
self.score = score
self.payload = payload
class _FakeVectorEngine:
def __init__(self, hits):
self._hits = hits
async def search(self, collection, query_text=None, limit=15, include_payload=False, **kw):
return self._hits
async def test_gate2_accepts_high_confidence_contradiction(monkeypatch):
"""An on-topic candidate judged 'contradicts' with conf>=0.6 is returned."""
hits = [_FakeHit("E_qa", 0.10, {"claim": "March QA report documents the defect"})]
monkeypatch.setattr(graph_ops, "get_vector_engine", lambda: _FakeVectorEngine(hits))
async def fake_llm(text_input, system_prompt, response_model, **kw):
return ContradictionJudgement(relation="contradicts", confidence=0.9, rationale="back-dated")
from cognee.infrastructure.llm.LLMGateway import LLMGateway
monkeypatch.setattr(LLMGateway, "acreate_structured_output", staticmethod(fake_llm))
out = await detect_contradictions("the report was back-dated")
assert len(out) == 1
assert out[0].target_id == "E_qa"
assert out[0].relation == "contradicts"
async def test_gate1_filters_off_topic(monkeypatch):
"""A candidate beyond the distance threshold is filtered before the LLM runs."""
hits = [_FakeHit("E_far", 0.90, {"claim": "unrelated topic"})]
monkeypatch.setattr(graph_ops, "get_vector_engine", lambda: _FakeVectorEngine(hits))
called = {"llm": False}
async def fake_llm(text_input, system_prompt, response_model, **kw):
called["llm"] = True
return ContradictionJudgement(relation="contradicts", confidence=1.0)
from cognee.infrastructure.llm.LLMGateway import LLMGateway
monkeypatch.setattr(LLMGateway, "acreate_structured_output", staticmethod(fake_llm))
out = await detect_contradictions("some fact", distance_threshold=0.35)
assert out == []
assert called["llm"] is False # Gate-1 rejected it; LLM never consulted
async def test_gate2_rejects_low_confidence(monkeypatch):
"""On-topic but low-confidence verdicts are not treated as contradictions."""
hits = [_FakeHit("E_qa", 0.10, {"claim": "March QA report"})]
monkeypatch.setattr(graph_ops, "get_vector_engine", lambda: _FakeVectorEngine(hits))
async def fake_llm(text_input, system_prompt, response_model, **kw):
return ContradictionJudgement(relation="contradicts", confidence=0.2, rationale="unsure")
from cognee.infrastructure.llm.LLMGateway import LLMGateway
monkeypatch.setattr(LLMGateway, "acreate_structured_output", staticmethod(fake_llm))
out = await detect_contradictions("weak signal")
assert out == []