File size: 10,127 Bytes
fb9f5e1 03b3d27 cbd62a3 03b3d27 fb9f5e1 03b3d27 9ca8ee5 cbd62a3 9ca8ee5 cbd62a3 9ca8ee5 cbd62a3 9ca8ee5 cbd62a3 9ca8ee5 cbd62a3 9ca8ee5 cbd62a3 9ca8ee5 cbd62a3 9ca8ee5 cbd62a3 9ca8ee5 cbd62a3 9ca8ee5 cbd62a3 9ca8ee5 cbd62a3 9ca8ee5 cbd62a3 9ca8ee5 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 | import sqlite3
import pytest
from common.db import (
init_db, upsert_paper, mark_inactive, get_paper, keyword_search,
get_papers_by_ids, get_paper_id_by_faiss_id, update_metadata_fields,
)
@pytest.fixture
def conn(tmp_path):
return init_db(str(tmp_path / "test.db"))
def test_upsert_and_get_paper(conn):
upsert_paper(conn, {
"id": "p1", "title": "Neural Machine Translation", "abstract": "We study NMT.",
"authors": "A. Author", "venue": "ACL", "year": 2023, "url": "http://x", "active": True,
})
result = get_paper(conn, "p1")
assert result["title"] == "Neural Machine Translation"
assert result["active"] == 1
def test_get_paper_missing_returns_none(conn):
assert get_paper(conn, "missing") is None
def test_upsert_is_idempotent_update(conn):
upsert_paper(conn, {"id": "p1", "title": "Old Title", "abstract": "", "authors": "", "venue": "ACL", "year": 2020, "url": "", "active": True})
upsert_paper(conn, {"id": "p1", "title": "New Title", "abstract": "", "authors": "", "venue": "ACL", "year": 2020, "url": "", "active": True})
result = get_paper(conn, "p1")
assert result["title"] == "New Title"
def test_mark_inactive_excludes_from_keyword_search(conn):
upsert_paper(conn, {"id": "p1", "title": "Transformer Networks", "abstract": "", "authors": "", "venue": "ACL", "year": 2021, "url": "", "active": True})
mark_inactive(conn, "p1")
assert keyword_search(conn, "Transformer") == []
def test_keyword_search_matches_title_and_abstract(conn):
upsert_paper(conn, {"id": "p1", "title": "Parsing with Transformers", "abstract": "irrelevant text", "authors": "", "venue": "ACL", "year": 2022, "url": "", "active": True})
upsert_paper(conn, {"id": "p2", "title": "Unrelated Paper", "abstract": "discusses transformer architectures", "authors": "", "venue": "ACL", "year": 2022, "url": "", "active": True})
upsert_paper(conn, {"id": "p3", "title": "Something Else", "abstract": "no match here", "authors": "", "venue": "ACL", "year": 2022, "url": "", "active": True})
results = keyword_search(conn, "transformer")
ids = {r["id"] for r in results}
assert ids == {"p1", "p2"}
def test_get_papers_by_ids_preserves_order_and_skips_missing(conn):
upsert_paper(conn, {"id": "p1", "title": "A", "abstract": "", "authors": "", "venue": "ACL", "year": 2020, "url": "", "active": True})
upsert_paper(conn, {"id": "p2", "title": "B", "abstract": "", "authors": "", "venue": "ACL", "year": 2020, "url": "", "active": True})
results = get_papers_by_ids(conn, ["p2", "missing", "p1"])
assert [r["id"] for r in results] == ["p2", "p1"]
def test_upsert_paper_persists_faiss_id(conn):
upsert_paper(conn, {"id": "p1", "title": "A", "abstract": "", "authors": "", "venue": "ACL", "year": 2020, "url": "", "active": True, "faiss_id": 5})
assert get_paper(conn, "p1")["faiss_id"] == 5
def test_upsert_paper_without_faiss_id_does_not_null_existing(conn):
upsert_paper(conn, {"id": "p1", "title": "A", "abstract": "", "authors": "", "venue": "ACL", "year": 2020, "url": "", "active": True, "faiss_id": 5})
upsert_paper(conn, {"id": "p1", "title": "A2", "abstract": "", "authors": "", "venue": "ACL", "year": 2020, "url": "", "active": True})
result = get_paper(conn, "p1")
assert result["title"] == "A2"
assert result["faiss_id"] == 5
def test_get_paper_id_by_faiss_id_valid(conn):
upsert_paper(conn, {"id": "p1", "title": "A", "abstract": "", "authors": "", "venue": "ACL", "year": 2020, "url": "", "active": True, "faiss_id": 3})
assert get_paper_id_by_faiss_id(conn, 3) == "p1"
def test_get_paper_id_by_faiss_id_stale_returns_none(conn):
upsert_paper(conn, {"id": "p1", "title": "A", "abstract": "", "authors": "", "venue": "ACL", "year": 2020, "url": "", "active": True, "faiss_id": 3})
upsert_paper(conn, {"id": "p1", "title": "A", "abstract": "", "authors": "", "venue": "ACL", "year": 2020, "url": "", "active": True, "faiss_id": 7})
assert get_paper_id_by_faiss_id(conn, 3) is None
assert get_paper_id_by_faiss_id(conn, 7) == "p1"
def test_get_paper_id_by_faiss_id_inactive_returns_none(conn):
upsert_paper(conn, {"id": "p1", "title": "A", "abstract": "", "authors": "", "venue": "ACL", "year": 2020, "url": "", "active": True, "faiss_id": 3})
mark_inactive(conn, "p1")
assert get_paper_id_by_faiss_id(conn, 3) is None
def test_get_paper_id_by_faiss_id_unknown_returns_none(conn):
assert get_paper_id_by_faiss_id(conn, 999) is None
def test_upsert_paper_without_bibtex_defaults_to_empty(conn):
upsert_paper(conn, {"id": "p1", "title": "A", "abstract": "", "authors": "", "venue": "ACL", "year": 2020, "url": "", "active": True})
assert get_paper(conn, "p1")["bibtex"] == ""
def test_upsert_paper_persists_bibtex(conn):
upsert_paper(conn, {"id": "p1", "title": "A", "abstract": "", "authors": "", "venue": "ACL", "year": 2020, "url": "", "active": True, "bibtex": "@inproceedings{p1}"})
assert get_paper(conn, "p1")["bibtex"] == "@inproceedings{p1}"
def test_upsert_paper_updates_bibtex_on_conflict(conn):
upsert_paper(conn, {"id": "p1", "title": "A", "abstract": "", "authors": "", "venue": "ACL", "year": 2020, "url": "", "active": True, "bibtex": "old"})
upsert_paper(conn, {"id": "p1", "title": "A2", "abstract": "", "authors": "", "venue": "ACL", "year": 2020, "url": "", "active": True, "bibtex": "new"})
result = get_paper(conn, "p1")
assert result["title"] == "A2"
assert result["bibtex"] == "new"
def test_init_db_migrates_bibtex_column_into_existing_db(tmp_path):
"""A pre-existing DB snapshot (created before the bibtex/pdf_url columns
existed) must gain them on init_db so the service works against the
current HF snapshot without a full resync."""
path = str(tmp_path / "legacy.db")
legacy_conn = sqlite3.connect(path)
legacy_conn.executescript(
"""
CREATE TABLE papers (
id TEXT PRIMARY KEY, title TEXT NOT NULL, abstract TEXT NOT NULL DEFAULT '',
authors TEXT NOT NULL DEFAULT '', venue TEXT NOT NULL DEFAULT '', year INTEGER,
url TEXT NOT NULL DEFAULT '', active INTEGER NOT NULL DEFAULT 1, faiss_id INTEGER
);
INSERT INTO papers (id, title, abstract, authors, venue, year, url, active)
VALUES ('legacy-1', 'Old', '', '', 'ACL', 2020, '', 1);
"""
)
legacy_conn.commit()
legacy_conn.close()
conn = init_db(path)
columns = {row["name"] for row in conn.execute("PRAGMA table_info(papers)")}
assert "bibtex" in columns
assert "pdf_url" in columns
row = conn.execute("SELECT * FROM papers WHERE id = 'legacy-1'").fetchone()
assert dict(row)["bibtex"] == ""
assert dict(row)["pdf_url"] == ""
def test_upsert_paper_without_pdf_url_defaults_to_empty(conn):
upsert_paper(conn, {"id": "p1", "title": "A", "abstract": "", "authors": "", "venue": "ACL", "year": 2020, "url": "", "active": True})
assert get_paper(conn, "p1")["pdf_url"] == ""
def test_upsert_paper_persists_pdf_url(conn):
upsert_paper(conn, {"id": "p1", "title": "A", "abstract": "", "authors": "", "venue": "ACL", "year": 2020, "url": "", "active": True, "pdf_url": "http://x/p1.pdf"})
assert get_paper(conn, "p1")["pdf_url"] == "http://x/p1.pdf"
def test_update_metadata_fields_fills_empty_and_skips_unchanged(conn):
upsert_paper(conn, {"id": "p1", "title": "A", "abstract": "", "authors": "", "venue": "ACL", "year": 2020, "url": "", "active": True})
upsert_paper(conn, {"id": "p2", "title": "B", "abstract": "", "authors": "", "venue": "ACL", "year": 2020, "url": "", "active": True, "bibtex": "existing", "pdf_url": "existing.pdf"})
update_metadata_fields(conn, [
{"id": "p1", "bibtex": "@inproceedings{p1}", "pdf_url": "http://x/p1.pdf"},
{"id": "p2", "bibtex": "existing", "pdf_url": "existing.pdf"}, # unchanged -> no write
])
assert get_paper(conn, "p1")["bibtex"] == "@inproceedings{p1}"
assert get_paper(conn, "p1")["pdf_url"] == "http://x/p1.pdf"
assert get_paper(conn, "p2")["bibtex"] == "existing"
assert get_paper(conn, "p2")["pdf_url"] == "existing.pdf"
def test_update_metadata_fields_refreshes_when_value_differs(conn):
"""A bibtex/pdf_url change on an otherwise-unchanged paper (e.g. a pages
edit) is picked up without re-embedding — the decoupled freshness path."""
upsert_paper(conn, {"id": "p1", "title": "A", "abstract": "", "authors": "", "venue": "ACL", "year": 2020, "url": "", "active": True, "bibtex": "stale", "pdf_url": "stale.pdf"})
update_metadata_fields(conn, [{"id": "p1", "bibtex": "fresh", "pdf_url": "fresh.pdf"}])
result = get_paper(conn, "p1")
assert result["bibtex"] == "fresh"
assert result["pdf_url"] == "fresh.pdf"
def test_update_metadata_fields_skips_papers_without_either_key(conn):
"""Dicts without a bibtex or pdf_url key (e.g. test mocks) must not raise
and must leave stored values untouched."""
upsert_paper(conn, {"id": "p1", "title": "A", "abstract": "", "authors": "", "venue": "ACL", "year": 2020, "url": "", "active": True, "bibtex": "keep", "pdf_url": "keep.pdf"})
update_metadata_fields(conn, [{"id": "p1"}])
result = get_paper(conn, "p1")
assert result["bibtex"] == "keep"
assert result["pdf_url"] == "keep.pdf"
def test_update_metadata_fields_leaves_field_untouched_when_key_missing(conn):
"""A dict with only one of the two keys must not clobber the other field
with an empty default."""
upsert_paper(conn, {"id": "p1", "title": "A", "abstract": "", "authors": "", "venue": "ACL", "year": 2020, "url": "", "active": True, "bibtex": "keep", "pdf_url": "old.pdf"})
update_metadata_fields(conn, [{"id": "p1", "pdf_url": "new.pdf"}])
result = get_paper(conn, "p1")
assert result["bibtex"] == "keep"
assert result["pdf_url"] == "new.pdf"
def test_update_metadata_fields_no_op_for_unknown_ids(conn):
update_metadata_fields(conn, [{"id": "ghost", "bibtex": "@inproceedings{ghost}", "pdf_url": "http://x/g.pdf"}])
assert get_paper(conn, "ghost") is None
|