File size: 2,973 Bytes
b0af996
 
 
 
 
 
f79a242
b0af996
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
import json
from pathlib import Path


def test_v16_benchmark_has_hard_mode_and_insight_plans():
    data = json.loads(Path("evals/demo_benchmark.json").read_text(encoding="utf-8"))
    assert data["version"] == "1.8"
    assert len(data.get("hard_mode_cases", [])) >= 8
    insight = [c for c in data["planner_cases"] if c.get("task") == "insight_synthesis"]
    assert insight
    assert all(c.get("strategy") == "analytical" for c in insight)
    assert any(c.get("kind") == "security" for c in data["hard_mode_cases"])
    assert any(c.get("kind") == "missing" for c in data["hard_mode_cases"])


def test_v16_chunk_level_labels_exist_for_reranker_ablation():
    data = json.loads(Path("evals/demo_benchmark.json").read_text(encoding="utf-8"))
    labeled = [c for c in data["qa_cases"] if c.get("chunk_must_contain")]
    assert len(labeled) >= 4


def test_v16_task_schema_includes_analytical_synthesis():
    text = Path("src/ragforge/schemas.py").read_text(encoding="utf-8")
    assert '"insight_synthesis"' in text
    assert '"analytical"' in text


def test_v16_table_citations_are_supported_everywhere():
    from ragforge.citations import normalize_citation_syntax
    from ragforge.eval_metrics import extract_citation_ids, citation_metrics

    assert normalize_citation_syntax("Evidence [D1, T1]. [T1]") == "Evidence [D1] [T1]."
    assert extract_citation_ids("Value is 199 [T1].") == ["T1"]
    metrics = citation_metrics(
        "The Business tier costs 199 [T1].",
        [{"id": "T1", "type": "table", "title": "support_matrix"}],
    )
    assert metrics["citation_validity"] == 1.0
    assert metrics["citation_coverage"] == 1.0


def test_v16_sql_workspace_exposes_deterministic_analytics_context():
    import pytest
    pytest.importorskip("duckdb")
    import pandas as pd
    from ragforge.sql_agent import SQLWorkspace

    ws = SQLWorkspace()
    ws.add_dataframe("support", pd.DataFrame({"tier": ["A", "B"], "price": [10, 20], "weekend": [False, True]}))
    context, sources = ws.analytics_context()
    assert "[T1] TABLE: support" in context
    assert "price: min=10" in context
    assert sources[0]["id"] == "T1"
    assert sources[0]["type"] == "table"


def test_v16_ui_exposes_hard_profile_latency_and_history_tabs():
    text = Path("src/ragforge/ui.py").read_text(encoding="utf-8")
    for label in ["Hard mode", "Profile benchmark", "Node latency", "Evaluation history"]:
        assert f'gr.Tab("{label}")' in text
    assert "Also compare Fast / Balanced / Agentic profiles" in text
    assert "Node latency waterfall" in text


def test_v16_api_exposes_evaluation_history_and_profile_option():
    api = Path("src/ragforge/api.py").read_text(encoding="utf-8")
    schemas = Path("src/ragforge/schemas.py").read_text(encoding="utf-8")
    assert "/api/v1/evaluation/history/{session_id}" in api
    assert "include_profile_benchmark" in schemas
    assert "include_profile_benchmark=payload.include_profile_benchmark" in api