AutonomousAgent / test_sources.py
Mathias Heider
Claude Fable 5.1
Candidate queue: a bulk discovery pass fills agent_candidates, cycles drain it; UDSpace + CORE lanes
68b949f unverified
Raw History Blame Contribute Delete
25.7 kB
"""Unit tests for the Sep 2026 "more papers" crawler/orchestrator changes.
No database and no network: HTTP is replaced with scripted responses. Run
with: python test_sources.py
Covers: per-host circuit breaker (budget 429s), arXiv query builder, gate
relief without abstract, OpenAlex repository-first URL ranking + landing
page, Semantic Scholar bulk gating, citation_pdf_url fallback, the
orchestrator skipping a closed source, the datasheet-round schedule, the
topic walk cursors and the query grid.
"""
import json
import os
import sys
import tempfile
import time
from pathlib import Path
os.environ["AGENT_USE_S2"] = "0"
os.environ["AGENT_USE_OPENALEX_TOPICS"] = "0"
os.environ["AGENT_USE_DATASHEETS"] = "0"
import pdf_crawler as pc
from agent import agdb, orchestrator, query_grid
fails = []
def check(name, cond):
print(("PASS " if cond else "FAIL ") + name)
if not cond:
fails.append(name)
# --- fake HTTP ---------------------------------------------------------------
class FakeResp:
def __init__(self, status, headers=None, body=b"", url="", json_body=None):
self.status_code = status
self.headers = headers or {}
self._body = body if isinstance(body, bytes) else body.encode()
self.url = url
self._json = json_body
if json_body is not None:
self._body = json.dumps(json_body).encode()
self.headers.setdefault("Content-Type", "application/json")
@property
def text(self):
return self._body.decode(errors="replace")
def json(self):
if self._json is None:
return json.loads(self._body.decode())
return self._json
def iter_content(self, chunk_size=65536):
yield self._body
SCRIPT: dict = {} # url prefix -> list of responses (popped in order) or single
CALLS: list = []
SLEEPS: list = []
def fake_get(url, timeout=None, **kw):
if kw.get("params"):
import urllib.parse
url = url + ("&" if "?" in url else "?") + urllib.parse.urlencode(kw["params"])
CALLS.append(url)
for prefix, resp in SCRIPT.items():
if url.startswith(prefix):
if isinstance(resp, list):
return resp.pop(0) if len(resp) > 1 else resp[0]
return resp
return FakeResp(404)
pc.SESSION.get = fake_get
pc.time.sleep = lambda s: SLEEPS.append(s)
pc.THROTTLE.wait = lambda url: None
# --- 1. circuit breaker --------------------------------------------------------
pc.reopen_sources()
CALLS.clear(); SLEEPS.clear()
budget_429 = FakeResp(429, {"Retry-After": "65169", "X-RateLimit-Remaining": "0"},
json_body={"error": "Rate limit exceeded",
"message": "Insufficient budget. This request costs $0.001 "
"but you only have $0 remaining. Resets at midnight UTC."})
SCRIPT["https://api.openalex.org/"] = budget_429
r = pc.http_get("https://api.openalex.org/works?search=x")
check("budget 429 returns None without retrying", r is None and len(CALLS) == 1)
check("budget 429 does not sleep", SLEEPS == [])
reason = pc.source_closed("api.openalex.org")
check("host is closed with the server's reason", bool(reason) and "Insufficient budget" in reason)
CALLS.clear()
r = pc.http_get("https://api.openalex.org/works?search=y")
check("closed host short-circuits (no HTTP call)", r is None and CALLS == [])
check("other hosts unaffected", pc.source_closed("export.arxiv.org") is None)
cands = list(pc.search_openalex("thermoplastic composite", 5))
check("search_openalex yields nothing while closed", cands == [] and CALLS == [])
pc.reopen_sources()
check("reopen_sources clears the closure", pc.source_closed("api.openalex.org") is None)
# short Retry-After is still a normal retry
CALLS.clear(); SLEEPS.clear()
SCRIPT["https://api.openalex.org/"] = [FakeResp(429, {"Retry-After": "5"}), FakeResp(200, json_body={"results": []})]
r = pc.http_get("https://api.openalex.org/works?search=z")
check("short Retry-After: slept then retried", SLEEPS == [5.0] and r is not None and r.status_code == 200)
check("short Retry-After does not close the host", pc.source_closed("api.openalex.org") is None)
# a huge Retry-After closes even without the X-RateLimit header
CALLS.clear(); SLEEPS.clear()
SCRIPT["https://api.semanticscholar.org/"] = FakeResp(429, {"Retry-After": "7200"})
r = pc.http_get("https://api.semanticscholar.org/graph/v1/paper/search/bulk?query=a")
check("Retry-After above cap closes the host", r is None and pc.source_closed("api.semanticscholar.org") is not None and SLEEPS == [])
pc.reopen_sources()
# --- 2. arXiv query builder ------------------------------------------------------
q = pc.arxiv_search_query("thermoplastic composite mechanical properties")
check("arXiv: phrase becomes ANDed terms", q == "all:thermoplastic AND all:composite AND all:mechanical")
q = pc.arxiv_search_query("PA66 glass fiber composite mechanical properties")
check("arXiv: matrix and reinforcement beat generic words", q == "all:pa66 AND all:glass AND (all:fiber OR all:fibre)")
q = pc.arxiv_search_query("carbon fiber PEEK composite tensile properties")
check("arXiv: PEEK kept, fibre spelled both ways", "all:peek" in q and "(all:fiber OR all:fibre)" in q and q.count(" AND ") == 2)
q2 = pc.arxiv_search_query("carbon fiber PEEK composite tensile properties", max_terms=2)
check("arXiv: two-term fallback drops the weakest term", q2 == "(all:fiber OR all:fibre) AND all:peek" or q2 == "all:carbon AND all:peek")
check("arXiv: single word passes through", pc.arxiv_search_query("PEEK") == "all:peek")
# search_arxiv retries broader when the specific query is empty
CALLS.clear()
EMPTY = '<?xml version="1.0"?><feed xmlns="http://www.w3.org/2005/Atom"></feed>'
ONE = ('<?xml version="1.0"?><feed xmlns="http://www.w3.org/2005/Atom"><entry>'
'<title>Carbon fibre PEEK laminates: tensile modulus</title><summary>thermoplastic composite '
'carbon fiber PEEK tensile strength</summary><published>2024-01-01</published>'
'<link title="pdf" href="https://arxiv.org/pdf/2401.00001"/></entry></feed>')
SCRIPT["https://export.arxiv.org/"] = [FakeResp(200, body=EMPTY), FakeResp(200, body=ONE)]
got = list(pc.search_arxiv("carbon fiber PEEK composite tensile properties", 5))
check("arXiv: empty result triggers one broader retry", len(CALLS) == 2 and len(got) == 1)
check("arXiv: retry used fewer terms", CALLS[1].count("AND") < CALLS[0].count("AND"))
# --- 3. gate relief ----------------------------------------------------------------
check("gate: full score with abstract", pc.effective_min_score(4, "some abstract") == 4)
check("gate: relief without abstract", pc.effective_min_score(4, "") == 2)
check("gate: floor at 1", pc.effective_min_score(1, "") == 1)
check("_passes_gate: title-only paper accepted with relief",
pc._passes_gate("Fatigue of thermoplastic laminates", "", "https://x/a.pdf", 4))
check("_passes_gate: irrelevant title rejected", not pc._passes_gate("Bird migration", "", "https://x/b.pdf", 4))
# --- 4. OpenAlex URL ranking + candidate ------------------------------------------
work = {
"display_name": "Interlaminar shear strength of carbon fibre PEEK laminates",
"doi": "https://doi.org/10.1/abc", "publication_year": 2023,
"abstract_inverted_index": {"thermoplastic": [0], "composite": [1], "tensile": [2]},
"best_oa_location": {"pdf_url": "https://www.mdpi.com/x/pdf", "landing_page_url": "https://www.mdpi.com/x",
"version": "publishedVersion", "source": {"type": "journal"}},
"locations": [
{"pdf_url": "https://www.mdpi.com/x/pdf", "source": {"type": "journal"}},
{"pdf_url": "https://hal.science/hal-1/document", "landing_page_url": "https://hal.science/hal-1",
"version": "acceptedVersion", "source": {"type": "repository"}},
{"pdf_url": "https://europepmc.org/articles/PMC1?pdf=render", "version": "publishedVersion",
"source": {"type": "repository"}},
{"pdf_url": "https://x.org/slides.pdf", "source": {"type": "repository"}},
],
}
urls, landing = pc._rank_oa_urls(work)
check("ranking: repository publishedVersion first", urls[0].startswith("https://europepmc.org"))
check("ranking: repository acceptedVersion second", urls[1].startswith("https://hal.science"))
check("ranking: publisher URL last", urls[-1] == "https://www.mdpi.com/x/pdf")
check("ranking: junk URL dropped", not any("slides" in u for u in urls))
check("ranking: landing page comes from the first ranked location", landing == "https://hal.science/hal-1")
cand = pc._openalex_work_to_candidate(work, "q", 4)
check("candidate: pdf_url is the top-ranked URL, alts follow", cand.pdf_url == urls[0] and cand.alt_urls == urls[1:])
check("candidate: doi stripped of prefix", cand.doi == "10.1/abc")
work_noabs = dict(work, abstract_inverted_index=None,
display_name="Consolidation quality of AFP-made thermoplastic parts")
check("candidate: title-only work passes with relief",
pc._openalex_work_to_candidate(work_noabs, "q", 4) is not None)
check("candidate: title-only work rejected without relief",
pc._openalex_work_to_candidate(dict(work_noabs, display_name="Bird migration"), "q", 4) is None)
# OpenAlex request carries the key and the OA-status filter
pc.OPENALEX_API_KEY = "k-test"
CALLS.clear()
SCRIPT["https://api.openalex.org/"] = FakeResp(200, json_body={"results": [work]})
got = list(pc.search_openalex("thermoplastic composite", 3))
check("openalex request has api_key, oa_status filter and per-page>=25",
"api_key=k-test" in CALLS[0] and "oa_status:gold%7Chybrid%7Cgreen" in CALLS[0].replace("|", "%7C")
and "per-page=25" in CALLS[0])
check("openalex yields the ranked candidate", len(got) == 1 and got[0].pdf_url.startswith("https://europepmc.org"))
pc.OPENALEX_API_KEY = ""
# topic page call
CALLS.clear()
SCRIPT["https://api.openalex.org/"] = FakeResp(200, json_body={"results": [work], "meta": {"next_cursor": "abc"}})
cands, nxt = pc.list_openalex_topic("T11664", 50, "*", 4, query="topic:x")
check("topic page: filter call (no search=), cursor and sort present",
"search=" not in CALLS[0] and "topics.id:T11664" in CALLS[0] and "cursor=" in CALLS[0]
and "cited_by_count" in CALLS[0])
check("topic page: candidates + next cursor", len(cands) == 1 and nxt == "abc" and cands[0].query == "topic:x")
SCRIPT["https://api.openalex.org/"] = FakeResp(200, json_body={"results": [], "meta": {"next_cursor": None}})
cands, nxt = pc.list_openalex_topic("T11664", 50, "abc", 4)
check("topic page: exhausted → next cursor None", cands == [] and nxt is None)
pc.close_source("api.openalex.org", 100, "test")
cands, nxt = pc.list_openalex_topic("T11664", 50, "abc", 4)
check("topic page: closed source keeps the cursor", cands == [] and nxt == "abc")
pc.reopen_sources()
# --- 5. Semantic Scholar bulk ------------------------------------------------------
papers = [
{"title": "Glass fiber polypropylene composite tensile modulus", "abstract": "thermoplastic composite tensile",
"year": 2020, "externalIds": {"DOI": "10.2/a"}, "openAccessPdf": {"url": "https://www.mdpi.com/a/pdf"}},
{"title": "PEEK carbon fibre laminate fatigue", "abstract": "thermoplastic composite laminate",
"year": 2021, "externalIds": {"DOI": "10.2/b"}, "openAccessPdf": {"url": "https://osti.gov/b.pdf"}},
{"title": "arXiv-only preprint on PEEK composites", "abstract": "thermoplastic composite",
"year": 2022, "externalIds": {"ArXiv": "2201.1"}, "openAccessPdf": {"url": "https://arxiv.org/pdf/2201.1"}},
{"title": "No pdf thermoplastic composite", "abstract": "", "year": 2019, "externalIds": {}, "openAccessPdf": None},
{"title": "Bird migration", "abstract": "", "year": 2018, "externalIds": {"DOI": "10.2/c"},
"openAccessPdf": {"url": "https://x/c.pdf"}},
]
CALLS.clear()
SCRIPT["https://api.semanticscholar.org/"] = FakeResp(200, json_body={"total": 5, "data": papers})
got = list(pc.search_semantic_scholar("thermoplastic composite", 1))
check("S2: bulk endpoint with openAccessPdf flag", "search/bulk" in CALLS[0] and "openAccessPdf=" in CALLS[0])
check("S2: limit keeps the highest-scoring gated paper", len(got) == 1 and got[0].title.startswith("PEEK"))
got = list(pc.search_semantic_scholar("thermoplastic composite", 10))
check("S2: arXiv-only, no-pdf and irrelevant papers dropped", [g.doi for g in got] == ["10.2/b", "10.2/a"])
# --- 6. citation_pdf_url fallback ------------------------------------------------------
pc.robots_allows = lambda url: True
HTML1 = '<html><head><meta name="citation_pdf_url" content="/files/paper.pdf"></head></html>'
HTML2 = '<html><head><meta content="https://r.org/p.pdf" name="citation_pdf_url"></head></html>'
HTML3 = '<html><head><link rel="alternate" type="application/pdf" href="alt.pdf"></head></html>'
SCRIPT["https://r.org/land1"] = FakeResp(200, {"Content-Type": "text/html"}, HTML1, url="https://r.org/land1")
SCRIPT["https://r.org/land2"] = FakeResp(200, {"Content-Type": "text/html"}, HTML2, url="https://r.org/land2")
SCRIPT["https://r.org/land3"] = FakeResp(200, {"Content-Type": "text/html"}, HTML3, url="https://r.org/land3")
SCRIPT["https://r.org/none"] = FakeResp(200, {"Content-Type": "text/html"}, "<html></html>", url="https://r.org/none")
check("landing: meta name/content order resolved relative", pc.pdf_url_from_landing_page("https://r.org/land1") == "https://r.org/files/paper.pdf")
check("landing: meta content/name order", pc.pdf_url_from_landing_page("https://r.org/land2") == "https://r.org/p.pdf")
check("landing: <link type=application/pdf>", pc.pdf_url_from_landing_page("https://r.org/land3") == "https://r.org/alt.pdf")
check("landing: nothing found → None", pc.pdf_url_from_landing_page("https://r.org/none") is None)
# download_pdf uses the landing page only after every listed URL failed
attempts = []
def fake_download_url(cand, url, pdf_dir, state):
attempts.append(url)
if url == "https://r.org/files/paper.pdf":
return {"filename": "ok.pdf", "url": url}
return None
_real_download_url = pc._download_url
_real_unpaywall = pc.unpaywall_pdf_url
pc._download_url = fake_download_url
pc.unpaywall_pdf_url = lambda doi: None
st = pc.CrawlerState(Path(tempfile.mkdtemp()) / "state.json")
c = pc.Candidate(title="t", pdf_url="https://pub/x.pdf", source="openalex", doi="10.1/x",
alt_urls=["https://repo/x.pdf"], landing_url="https://r.org/land1")
row = pc.download_pdf(c, Path(tempfile.mkdtemp()), st)
check("download_pdf: landing fallback after primary, alt and Unpaywall",
row is not None and attempts == ["https://pub/x.pdf", "https://repo/x.pdf", "https://r.org/files/paper.pdf"])
check("download_pdf: landing page marked final (one look ever)", "https://r.org/land1" in st.seen_urls)
attempts.clear()
row = pc.download_pdf(c, Path(tempfile.mkdtemp()), st)
check("download_pdf: landing page not re-fetched", attempts == ["https://pub/x.pdf", "https://repo/x.pdf"] and row is None)
pc._download_url = _real_download_url
pc.unpaywall_pdf_url = _real_unpaywall
# --- 7. orchestrator: closed source is skipped, one event per cycle -------------------
EVENTS = []
STATE_KV: dict = {}
class Conn:
def close(self):
pass
agdb.connect = lambda: Conn()
agdb.log_event = lambda conn, message, level="info", node="", run_id=None: EVENTS.append((level, node, message))
agdb.get_state = lambda conn, key, default=None: STATE_KV.get(key, default)
agdb.set_state = lambda conn, key, value: STATE_KV.__setitem__(key, value)
agdb.seen_doi = lambda conn, doi: False
agdb.record_download = lambda conn, rec: None
def cand(title, url, doi=""):
return pc.Candidate(title=title, pdf_url=url, source="openalex", doi=doi,
abstract="thermoplastic composite tensile carbon fiber PEEK")
pc.search_openalex = lambda query, limit: iter([cand("oa " + query, f"https://oa/{query}.pdf", "10.9/" + query)])
pc.search_arxiv = lambda query, limit: iter([cand("ax " + query, f"https://ax/{query}.pdf")])
pc.search_semantic_scholar = lambda query, limit: iter([cand("s2 " + query, f"https://s2/{query}.pdf")])
cfg = {"max_per_query": 5, "min_relevance_score": 4, "use_openalex": True, "use_arxiv": True,
"use_semantic_scholar": True, "use_ntrs": False}
state = {"run_id": 1, "cfg": cfg, "metrics": {}}
pc.close_source("api.openalex.org", 600, "HTTP 429: Insufficient budget")
got = orchestrator._discover_round(Conn(), state, [(1, "q1"), (2, "q2")], 1)
check("closed source skipped, other sources still searched",
sorted(c.source for c in got) == ["openalex"] * 4 and all(not c.title.startswith("oa") for c in got))
warns = [e for e in EVENTS if e[0] == "warn" and "openalex skipped" in e[2]]
check("one warn event for the closed source (not per query)", len(warns) == 1)
orchestrator._discover_round(Conn(), state, [(3, "q3")], 2)
warns = [e for e in EVENTS if e[0] == "warn" and "openalex skipped" in e[2]]
check("still one warn event across rounds of the same cycle", len(warns) == 1)
summary = [e for e in EVENTS if e[1] == "discover" and e[2].startswith("round 1:")][0][2]
check("round summary carries a per-source breakdown", "arxiv 2" in summary and "semantic_scholar 2" in summary)
pc.reopen_sources()
# --- 8. datasheet round schedule ------------------------------------------------------
EVENTS.clear(); STATE_KV.clear()
crawled = []
def fake_seed(seed):
crawled.append(seed)
yield pc.Candidate(title="TDS", pdf_url=f"{seed}/tds1.pdf", source="datasheet", query=seed)
yield pc.Candidate(title="TDS2", pdf_url=f"{seed}/tds2.pdf", source="datasheet", query=seed)
pc.crawl_datasheet_seed = fake_seed
downloaded = []
def fake_download(c, pdf_dir, st):
downloaded.append(c.pdf_url)
st.mark_final(c.pdf_url)
return {"filename": c.pdf_url.rsplit("/", 1)[-1], "url": c.pdf_url, "source": "datasheet",
"title": c.title, "doi": "", "sha256": c.pdf_url, "query": c.query}
pc.download_pdf = fake_download
cs = pc.CrawlerState(Path(tempfile.mkdtemp()) / "state.json")
dcfg = {"use_datasheets": True, "datasheet_seed_urls": "https://a; https://b; https://c",
"datasheet_hours": 24, "datasheet_seeds_per_cycle": 2}
dl = []
n = orchestrator._datasheet_round(Conn(), {"run_id": 2, "cfg": dcfg, "metrics": {}}, dcfg, cs, set(), dl, 10)
check("datasheet: two due seeds crawled per cycle", crawled == ["https://a", "https://b"] and n == 4)
n = orchestrator._datasheet_round(Conn(), {"run_id": 3, "cfg": dcfg, "metrics": {}}, dcfg, cs, set(), [], 10)
check("datasheet: third seed next cycle, crawled ones parked for 24 h", crawled[-1] == "https://c" and len(crawled) == 3)
n = orchestrator._datasheet_round(Conn(), {"run_id": 4, "cfg": dcfg, "metrics": {}}, dcfg, cs, set(), [], 10)
check("datasheet: nothing due → no crawl", len(crawled) == 3 and n == 0)
STATE_KV["datasheet_seeds"]["https://a"]["last"] -= 25 * 3600
n = orchestrator._datasheet_round(Conn(), {"run_id": 5, "cfg": dcfg, "metrics": {}}, dcfg, cs, set(), [], 10)
check("datasheet: seed due again after 24 h, already-seen PDFs skipped", crawled[-1] == "https://a" and n == 0)
n = orchestrator._datasheet_round(Conn(), {"run_id": 6, "cfg": dcfg, "metrics": {}}, dcfg, cs, set(), [{}] * 10, 10)
check("datasheet: skipped when the cap is already reached", n == 0 and len(crawled) == 4)
check("datasheet: disabled by config", orchestrator._datasheet_round(
Conn(), {"run_id": 7, "cfg": {}, "metrics": {}}, {"use_datasheets": False}, cs, set(), [], 10) == 0)
# --- 9. topic walk ---------------------------------------------------------------------
EVENTS.clear(); STATE_KV.clear()
pages = []
def fake_topic(tid, per_page, cursor, min_score, query="", sort=""):
pages.append((tid, cursor))
if cursor == "*":
return [cand(f"{tid} p1", f"https://t/{tid}/1.pdf")], "c2"
return [cand(f"{tid} p2", f"https://t/{tid}/2.pdf")], None
pc.list_openalex_topic = fake_topic
tcfg = {"use_openalex_topics": True, "use_openalex": True,
"openalex_topic_names": "T11664 Fiber-reinforced polymer composites; T10219 Mechanical Behavior of Composites",
"openalex_topics_per_cycle": 1, "openalex_topic_page": 50, "min_relevance_score": 4}
got = orchestrator._discover_topics(Conn(), {"run_id": 8, "cfg": tcfg, "metrics": {}}, tcfg)
check("topics: ids parsed from config, one topic per cycle", pages == [("T11664", "*")] and len(got) == 1)
check("topics: cursor persisted", STATE_KV.get("openalex_topic_cursor:T11664") == "c2")
got = orchestrator._discover_topics(Conn(), {"run_id": 9, "cfg": tcfg, "metrics": {}}, tcfg)
check("topics: round-robin to the other topic", pages[-1] == ("T10219", "*"))
got = orchestrator._discover_topics(Conn(), {"run_id": 10, "cfg": tcfg, "metrics": {}}, tcfg)
check("topics: back to the first topic at its saved cursor", pages[-1] == ("T11664", "c2"))
check("topics: exhausted topic parked as done", STATE_KV.get("openalex_topic_cursor:T11664") == "done")
got = orchestrator._discover_topics(Conn(), {"run_id": 11, "cfg": tcfg, "metrics": {}}, tcfg)
got = orchestrator._discover_topics(Conn(), {"run_id": 12, "cfg": tcfg, "metrics": {}}, tcfg)
check("topics: a done topic is skipped", ("T11664", "done") not in pages)
pc.close_source("api.openalex.org", 600, "HTTP 429: budget")
st12 = {"run_id": 13, "cfg": tcfg, "metrics": {}}
got = orchestrator._discover_topics(Conn(), st12, tcfg)
check("topics: closed OpenAlex → no call, one warn", got == [] and any("openalex skipped" in e[2] for e in EVENTS))
pc.reopen_sources()
# --- 10. query grid ------------------------------------------------------------------
qs = query_grid.build()
check("grid: a few hundred distinct intents", 200 <= len(qs) <= 600 and len(set(qs)) == len(qs))
check("grid: deterministic", qs == query_grid.build())
check("grid: covers PEEK, PPS and glass/carbon fibre", any("PEEK" in q for q in qs) and any("PPS" in q for q in qs)
and any("glass fiber" in q for q in qs) and any("carbon fiber" in q for q in qs))
# --- 11. UDSpace lane (DSpace 7 REST: discover → bundles → bitstreams) ----------------
SCRIPT.clear(); CALLS.clear()
_item = {"uuid": "11111111-1111-1111-1111-111111111111", "type": "item",
"name": "Void consolidation of thermoplastic composites via non-autoclave processing",
"metadata": {"dc.description.abstract": [{"value": "carbon fiber PEEK laminate tensile modulus"}],
"dc.date.issued": [{"value": "2021-05-01"}],
"dc.identifier.uri": [{"value": "https://udspace.udel.edu/handle/19716/99999"}]}}
_junk = {"uuid": "22222222-2222-2222-2222-222222222222", "type": "item",
"name": "Medieval poetry and its readers",
"metadata": {"dc.description.abstract": [{"value": "verse"}]}}
SCRIPT["https://udspace.udel.edu/server/api/discover/search/objects"] = FakeResp(200, json_body={
"_embedded": {"searchResult": {"_embedded": {"objects": [
{"_embedded": {"indexableObject": _item}}, {"_embedded": {"indexableObject": _junk}}]}}}})
SCRIPT["https://udspace.udel.edu/server/api/core/items/11111111-1111-1111-1111-111111111111/bundles"] = FakeResp(200, json_body={
"_embedded": {"bundles": [{"uuid": "bbbb", "name": "LICENSE"}, {"uuid": "aaaa", "name": "ORIGINAL"}]}})
SCRIPT["https://udspace.udel.edu/server/api/core/bundles/aaaa/bitstreams"] = FakeResp(200, json_body={
"_embedded": {"bitstreams": [
{"name": "thesis.pdf", "_links": {"content": {"href": "https://udspace.udel.edu/server/api/core/bitstreams/cccc/content"}}}]}})
got = list(pc.search_udspace("thermoplastic composite consolidation", 5))
check("udspace: one candidate (the poetry item fails the gate before any bundle call)",
len(got) == 1 and got[0].source == "udspace" and got[0].year == "2021")
check("udspace: PDF is the ORIGINAL bundle's bitstream content link",
got[0].pdf_url.endswith("/bitstreams/cccc/content"))
check("udspace: handle used as the DOI-like key and landing page",
got[0].doi == "udspace.udel.edu/handle/19716/99999" and got[0].landing_url.endswith("/19716/99999"))
check("udspace: no bundle calls for the rejected item",
not any("22222222" in u for u in CALLS))
# --- 12. CORE lane (key-gated; hosted download URLs) ----------------------------------
SCRIPT.clear(); CALLS.clear()
pc.CORE_API_KEY = ""
check("core: inactive without a key (no call)", list(pc.search_core("thermoplastic composite", 5)) == [] and CALLS == [])
pc.CORE_API_KEY = "test-key"
SCRIPT["https://api.core.ac.uk/v3/search/works"] = FakeResp(200, json_body={"results": [
{"id": 1, "title": "Tensile behaviour of carbon fibre PEEK thermoplastic composites",
"abstract": "thermoplastic composite tensile modulus", "yearPublished": 2020,
"doi": "https://doi.org/10.5/core.1", "downloadUrl": "https://core.ac.uk/download/1.pdf"},
{"id": 2, "title": "Poems", "abstract": "", "downloadUrl": "https://core.ac.uk/download/2.pdf"},
{"id": 3, "title": "Thermoplastic composite welding review", "abstract": "composite fiber",
"yearPublished": 2019, "downloadUrl": ""},
]})
got = list(pc.search_core("thermoplastic composite", 5))
check("core: relevant work with a hosted download URL becomes a candidate; junk and URL-less dropped",
[c.doi for c in got] == ["10.5/core.1"] and got[0].pdf_url == "https://core.ac.uk/download/1.pdf"
and got[0].year == "2020" and got[0].source == "core")
check("core: the key travels as a bearer token", any("core.ac.uk" in u for u in CALLS))
pc.CORE_API_KEY = ""
print()
print("all tests passed" if not fails else f"{len(fails)} FAILED: {fails}")
sys.exit(1 if fails else 0)