"""Unit tests for the Sep 2026 "more papers" crawler/orchestrator changes. No database and no network: HTTP is replaced with scripted responses. Run with: python test_sources.py Covers: per-host circuit breaker (budget 429s), arXiv query builder, gate relief without abstract, OpenAlex repository-first URL ranking + landing page, Semantic Scholar bulk gating, citation_pdf_url fallback, the orchestrator skipping a closed source, the datasheet-round schedule, the topic walk cursors and the query grid. """ import json import os import sys import tempfile import time from pathlib import Path os.environ["AGENT_USE_S2"] = "0" os.environ["AGENT_USE_OPENALEX_TOPICS"] = "0" os.environ["AGENT_USE_DATASHEETS"] = "0" import pdf_crawler as pc from agent import agdb, orchestrator, query_grid fails = [] def check(name, cond): print(("PASS " if cond else "FAIL ") + name) if not cond: fails.append(name) # --- fake HTTP --------------------------------------------------------------- class FakeResp: def __init__(self, status, headers=None, body=b"", url="", json_body=None): self.status_code = status self.headers = headers or {} self._body = body if isinstance(body, bytes) else body.encode() self.url = url self._json = json_body if json_body is not None: self._body = json.dumps(json_body).encode() self.headers.setdefault("Content-Type", "application/json") @property def text(self): return self._body.decode(errors="replace") def json(self): if self._json is None: return json.loads(self._body.decode()) return self._json def iter_content(self, chunk_size=65536): yield self._body SCRIPT: dict = {} # url prefix -> list of responses (popped in order) or single CALLS: list = [] SLEEPS: list = [] def fake_get(url, timeout=None, **kw): if kw.get("params"): import urllib.parse url = url + ("&" if "?" in url else "?") + urllib.parse.urlencode(kw["params"]) CALLS.append(url) for prefix, resp in SCRIPT.items(): if url.startswith(prefix): if isinstance(resp, list): return resp.pop(0) if len(resp) > 1 else resp[0] return resp return FakeResp(404) pc.SESSION.get = fake_get pc.time.sleep = lambda s: SLEEPS.append(s) pc.THROTTLE.wait = lambda url: None # --- 1. circuit breaker -------------------------------------------------------- pc.reopen_sources() CALLS.clear(); SLEEPS.clear() budget_429 = FakeResp(429, {"Retry-After": "65169", "X-RateLimit-Remaining": "0"}, json_body={"error": "Rate limit exceeded", "message": "Insufficient budget. This request costs $0.001 " "but you only have $0 remaining. Resets at midnight UTC."}) SCRIPT["https://api.openalex.org/"] = budget_429 r = pc.http_get("https://api.openalex.org/works?search=x") check("budget 429 returns None without retrying", r is None and len(CALLS) == 1) check("budget 429 does not sleep", SLEEPS == []) reason = pc.source_closed("api.openalex.org") check("host is closed with the server's reason", bool(reason) and "Insufficient budget" in reason) CALLS.clear() r = pc.http_get("https://api.openalex.org/works?search=y") check("closed host short-circuits (no HTTP call)", r is None and CALLS == []) check("other hosts unaffected", pc.source_closed("export.arxiv.org") is None) cands = list(pc.search_openalex("thermoplastic composite", 5)) check("search_openalex yields nothing while closed", cands == [] and CALLS == []) pc.reopen_sources() check("reopen_sources clears the closure", pc.source_closed("api.openalex.org") is None) # short Retry-After is still a normal retry CALLS.clear(); SLEEPS.clear() SCRIPT["https://api.openalex.org/"] = [FakeResp(429, {"Retry-After": "5"}), FakeResp(200, json_body={"results": []})] r = pc.http_get("https://api.openalex.org/works?search=z") check("short Retry-After: slept then retried", SLEEPS == [5.0] and r is not None and r.status_code == 200) check("short Retry-After does not close the host", pc.source_closed("api.openalex.org") is None) # a huge Retry-After closes even without the X-RateLimit header CALLS.clear(); SLEEPS.clear() SCRIPT["https://api.semanticscholar.org/"] = FakeResp(429, {"Retry-After": "7200"}) r = pc.http_get("https://api.semanticscholar.org/graph/v1/paper/search/bulk?query=a") check("Retry-After above cap closes the host", r is None and pc.source_closed("api.semanticscholar.org") is not None and SLEEPS == []) pc.reopen_sources() # --- 2. arXiv query builder ------------------------------------------------------ q = pc.arxiv_search_query("thermoplastic composite mechanical properties") check("arXiv: phrase becomes ANDed terms", q == "all:thermoplastic AND all:composite AND all:mechanical") q = pc.arxiv_search_query("PA66 glass fiber composite mechanical properties") check("arXiv: matrix and reinforcement beat generic words", q == "all:pa66 AND all:glass AND (all:fiber OR all:fibre)") q = pc.arxiv_search_query("carbon fiber PEEK composite tensile properties") check("arXiv: PEEK kept, fibre spelled both ways", "all:peek" in q and "(all:fiber OR all:fibre)" in q and q.count(" AND ") == 2) q2 = pc.arxiv_search_query("carbon fiber PEEK composite tensile properties", max_terms=2) check("arXiv: two-term fallback drops the weakest term", q2 == "(all:fiber OR all:fibre) AND all:peek" or q2 == "all:carbon AND all:peek") check("arXiv: single word passes through", pc.arxiv_search_query("PEEK") == "all:peek") # search_arxiv retries broader when the specific query is empty CALLS.clear() EMPTY = '' ONE = ('' 'Carbon fibre PEEK laminates: tensile modulusthermoplastic composite ' 'carbon fiber PEEK tensile strength2024-01-01' '') SCRIPT["https://export.arxiv.org/"] = [FakeResp(200, body=EMPTY), FakeResp(200, body=ONE)] got = list(pc.search_arxiv("carbon fiber PEEK composite tensile properties", 5)) check("arXiv: empty result triggers one broader retry", len(CALLS) == 2 and len(got) == 1) check("arXiv: retry used fewer terms", CALLS[1].count("AND") < CALLS[0].count("AND")) # --- 3. gate relief ---------------------------------------------------------------- check("gate: full score with abstract", pc.effective_min_score(4, "some abstract") == 4) check("gate: relief without abstract", pc.effective_min_score(4, "") == 2) check("gate: floor at 1", pc.effective_min_score(1, "") == 1) check("_passes_gate: title-only paper accepted with relief", pc._passes_gate("Fatigue of thermoplastic laminates", "", "https://x/a.pdf", 4)) check("_passes_gate: irrelevant title rejected", not pc._passes_gate("Bird migration", "", "https://x/b.pdf", 4)) # --- 4. OpenAlex URL ranking + candidate ------------------------------------------ work = { "display_name": "Interlaminar shear strength of carbon fibre PEEK laminates", "doi": "https://doi.org/10.1/abc", "publication_year": 2023, "abstract_inverted_index": {"thermoplastic": [0], "composite": [1], "tensile": [2]}, "best_oa_location": {"pdf_url": "https://www.mdpi.com/x/pdf", "landing_page_url": "https://www.mdpi.com/x", "version": "publishedVersion", "source": {"type": "journal"}}, "locations": [ {"pdf_url": "https://www.mdpi.com/x/pdf", "source": {"type": "journal"}}, {"pdf_url": "https://hal.science/hal-1/document", "landing_page_url": "https://hal.science/hal-1", "version": "acceptedVersion", "source": {"type": "repository"}}, {"pdf_url": "https://europepmc.org/articles/PMC1?pdf=render", "version": "publishedVersion", "source": {"type": "repository"}}, {"pdf_url": "https://x.org/slides.pdf", "source": {"type": "repository"}}, ], } urls, landing = pc._rank_oa_urls(work) check("ranking: repository publishedVersion first", urls[0].startswith("https://europepmc.org")) check("ranking: repository acceptedVersion second", urls[1].startswith("https://hal.science")) check("ranking: publisher URL last", urls[-1] == "https://www.mdpi.com/x/pdf") check("ranking: junk URL dropped", not any("slides" in u for u in urls)) check("ranking: landing page comes from the first ranked location", landing == "https://hal.science/hal-1") cand = pc._openalex_work_to_candidate(work, "q", 4) check("candidate: pdf_url is the top-ranked URL, alts follow", cand.pdf_url == urls[0] and cand.alt_urls == urls[1:]) check("candidate: doi stripped of prefix", cand.doi == "10.1/abc") work_noabs = dict(work, abstract_inverted_index=None, display_name="Consolidation quality of AFP-made thermoplastic parts") check("candidate: title-only work passes with relief", pc._openalex_work_to_candidate(work_noabs, "q", 4) is not None) check("candidate: title-only work rejected without relief", pc._openalex_work_to_candidate(dict(work_noabs, display_name="Bird migration"), "q", 4) is None) # OpenAlex request carries the key and the OA-status filter pc.OPENALEX_API_KEY = "k-test" CALLS.clear() SCRIPT["https://api.openalex.org/"] = FakeResp(200, json_body={"results": [work]}) got = list(pc.search_openalex("thermoplastic composite", 3)) check("openalex request has api_key, oa_status filter and per-page>=25", "api_key=k-test" in CALLS[0] and "oa_status:gold%7Chybrid%7Cgreen" in CALLS[0].replace("|", "%7C") and "per-page=25" in CALLS[0]) check("openalex yields the ranked candidate", len(got) == 1 and got[0].pdf_url.startswith("https://europepmc.org")) pc.OPENALEX_API_KEY = "" # topic page call CALLS.clear() SCRIPT["https://api.openalex.org/"] = FakeResp(200, json_body={"results": [work], "meta": {"next_cursor": "abc"}}) cands, nxt = pc.list_openalex_topic("T11664", 50, "*", 4, query="topic:x") check("topic page: filter call (no search=), cursor and sort present", "search=" not in CALLS[0] and "topics.id:T11664" in CALLS[0] and "cursor=" in CALLS[0] and "cited_by_count" in CALLS[0]) check("topic page: candidates + next cursor", len(cands) == 1 and nxt == "abc" and cands[0].query == "topic:x") SCRIPT["https://api.openalex.org/"] = FakeResp(200, json_body={"results": [], "meta": {"next_cursor": None}}) cands, nxt = pc.list_openalex_topic("T11664", 50, "abc", 4) check("topic page: exhausted → next cursor None", cands == [] and nxt is None) pc.close_source("api.openalex.org", 100, "test") cands, nxt = pc.list_openalex_topic("T11664", 50, "abc", 4) check("topic page: closed source keeps the cursor", cands == [] and nxt == "abc") pc.reopen_sources() # --- 5. Semantic Scholar bulk ------------------------------------------------------ papers = [ {"title": "Glass fiber polypropylene composite tensile modulus", "abstract": "thermoplastic composite tensile", "year": 2020, "externalIds": {"DOI": "10.2/a"}, "openAccessPdf": {"url": "https://www.mdpi.com/a/pdf"}}, {"title": "PEEK carbon fibre laminate fatigue", "abstract": "thermoplastic composite laminate", "year": 2021, "externalIds": {"DOI": "10.2/b"}, "openAccessPdf": {"url": "https://osti.gov/b.pdf"}}, {"title": "arXiv-only preprint on PEEK composites", "abstract": "thermoplastic composite", "year": 2022, "externalIds": {"ArXiv": "2201.1"}, "openAccessPdf": {"url": "https://arxiv.org/pdf/2201.1"}}, {"title": "No pdf thermoplastic composite", "abstract": "", "year": 2019, "externalIds": {}, "openAccessPdf": None}, {"title": "Bird migration", "abstract": "", "year": 2018, "externalIds": {"DOI": "10.2/c"}, "openAccessPdf": {"url": "https://x/c.pdf"}}, ] CALLS.clear() SCRIPT["https://api.semanticscholar.org/"] = FakeResp(200, json_body={"total": 5, "data": papers}) got = list(pc.search_semantic_scholar("thermoplastic composite", 1)) check("S2: bulk endpoint with openAccessPdf flag", "search/bulk" in CALLS[0] and "openAccessPdf=" in CALLS[0]) check("S2: limit keeps the highest-scoring gated paper", len(got) == 1 and got[0].title.startswith("PEEK")) got = list(pc.search_semantic_scholar("thermoplastic composite", 10)) check("S2: arXiv-only, no-pdf and irrelevant papers dropped", [g.doi for g in got] == ["10.2/b", "10.2/a"]) # --- 6. citation_pdf_url fallback ------------------------------------------------------ pc.robots_allows = lambda url: True HTML1 = '' HTML2 = '' HTML3 = '' SCRIPT["https://r.org/land1"] = FakeResp(200, {"Content-Type": "text/html"}, HTML1, url="https://r.org/land1") SCRIPT["https://r.org/land2"] = FakeResp(200, {"Content-Type": "text/html"}, HTML2, url="https://r.org/land2") SCRIPT["https://r.org/land3"] = FakeResp(200, {"Content-Type": "text/html"}, HTML3, url="https://r.org/land3") SCRIPT["https://r.org/none"] = FakeResp(200, {"Content-Type": "text/html"}, "", url="https://r.org/none") check("landing: meta name/content order resolved relative", pc.pdf_url_from_landing_page("https://r.org/land1") == "https://r.org/files/paper.pdf") check("landing: meta content/name order", pc.pdf_url_from_landing_page("https://r.org/land2") == "https://r.org/p.pdf") check("landing: ", pc.pdf_url_from_landing_page("https://r.org/land3") == "https://r.org/alt.pdf") check("landing: nothing found → None", pc.pdf_url_from_landing_page("https://r.org/none") is None) # download_pdf uses the landing page only after every listed URL failed attempts = [] def fake_download_url(cand, url, pdf_dir, state): attempts.append(url) if url == "https://r.org/files/paper.pdf": return {"filename": "ok.pdf", "url": url} return None _real_download_url = pc._download_url _real_unpaywall = pc.unpaywall_pdf_url pc._download_url = fake_download_url pc.unpaywall_pdf_url = lambda doi: None st = pc.CrawlerState(Path(tempfile.mkdtemp()) / "state.json") c = pc.Candidate(title="t", pdf_url="https://pub/x.pdf", source="openalex", doi="10.1/x", alt_urls=["https://repo/x.pdf"], landing_url="https://r.org/land1") row = pc.download_pdf(c, Path(tempfile.mkdtemp()), st) check("download_pdf: landing fallback after primary, alt and Unpaywall", row is not None and attempts == ["https://pub/x.pdf", "https://repo/x.pdf", "https://r.org/files/paper.pdf"]) check("download_pdf: landing page marked final (one look ever)", "https://r.org/land1" in st.seen_urls) attempts.clear() row = pc.download_pdf(c, Path(tempfile.mkdtemp()), st) check("download_pdf: landing page not re-fetched", attempts == ["https://pub/x.pdf", "https://repo/x.pdf"] and row is None) pc._download_url = _real_download_url pc.unpaywall_pdf_url = _real_unpaywall # --- 7. orchestrator: closed source is skipped, one event per cycle ------------------- EVENTS = [] STATE_KV: dict = {} class Conn: def close(self): pass agdb.connect = lambda: Conn() agdb.log_event = lambda conn, message, level="info", node="", run_id=None: EVENTS.append((level, node, message)) agdb.get_state = lambda conn, key, default=None: STATE_KV.get(key, default) agdb.set_state = lambda conn, key, value: STATE_KV.__setitem__(key, value) agdb.seen_doi = lambda conn, doi: False agdb.record_download = lambda conn, rec: None def cand(title, url, doi=""): return pc.Candidate(title=title, pdf_url=url, source="openalex", doi=doi, abstract="thermoplastic composite tensile carbon fiber PEEK") pc.search_openalex = lambda query, limit: iter([cand("oa " + query, f"https://oa/{query}.pdf", "10.9/" + query)]) pc.search_arxiv = lambda query, limit: iter([cand("ax " + query, f"https://ax/{query}.pdf")]) pc.search_semantic_scholar = lambda query, limit: iter([cand("s2 " + query, f"https://s2/{query}.pdf")]) cfg = {"max_per_query": 5, "min_relevance_score": 4, "use_openalex": True, "use_arxiv": True, "use_semantic_scholar": True, "use_ntrs": False} state = {"run_id": 1, "cfg": cfg, "metrics": {}} pc.close_source("api.openalex.org", 600, "HTTP 429: Insufficient budget") got = orchestrator._discover_round(Conn(), state, [(1, "q1"), (2, "q2")], 1) check("closed source skipped, other sources still searched", sorted(c.source for c in got) == ["openalex"] * 4 and all(not c.title.startswith("oa") for c in got)) warns = [e for e in EVENTS if e[0] == "warn" and "openalex skipped" in e[2]] check("one warn event for the closed source (not per query)", len(warns) == 1) orchestrator._discover_round(Conn(), state, [(3, "q3")], 2) warns = [e for e in EVENTS if e[0] == "warn" and "openalex skipped" in e[2]] check("still one warn event across rounds of the same cycle", len(warns) == 1) summary = [e for e in EVENTS if e[1] == "discover" and e[2].startswith("round 1:")][0][2] check("round summary carries a per-source breakdown", "arxiv 2" in summary and "semantic_scholar 2" in summary) pc.reopen_sources() # --- 8. datasheet round schedule ------------------------------------------------------ EVENTS.clear(); STATE_KV.clear() crawled = [] def fake_seed(seed): crawled.append(seed) yield pc.Candidate(title="TDS", pdf_url=f"{seed}/tds1.pdf", source="datasheet", query=seed) yield pc.Candidate(title="TDS2", pdf_url=f"{seed}/tds2.pdf", source="datasheet", query=seed) pc.crawl_datasheet_seed = fake_seed downloaded = [] def fake_download(c, pdf_dir, st): downloaded.append(c.pdf_url) st.mark_final(c.pdf_url) return {"filename": c.pdf_url.rsplit("/", 1)[-1], "url": c.pdf_url, "source": "datasheet", "title": c.title, "doi": "", "sha256": c.pdf_url, "query": c.query} pc.download_pdf = fake_download cs = pc.CrawlerState(Path(tempfile.mkdtemp()) / "state.json") dcfg = {"use_datasheets": True, "datasheet_seed_urls": "https://a; https://b; https://c", "datasheet_hours": 24, "datasheet_seeds_per_cycle": 2} dl = [] n = orchestrator._datasheet_round(Conn(), {"run_id": 2, "cfg": dcfg, "metrics": {}}, dcfg, cs, set(), dl, 10) check("datasheet: two due seeds crawled per cycle", crawled == ["https://a", "https://b"] and n == 4) n = orchestrator._datasheet_round(Conn(), {"run_id": 3, "cfg": dcfg, "metrics": {}}, dcfg, cs, set(), [], 10) check("datasheet: third seed next cycle, crawled ones parked for 24 h", crawled[-1] == "https://c" and len(crawled) == 3) n = orchestrator._datasheet_round(Conn(), {"run_id": 4, "cfg": dcfg, "metrics": {}}, dcfg, cs, set(), [], 10) check("datasheet: nothing due → no crawl", len(crawled) == 3 and n == 0) STATE_KV["datasheet_seeds"]["https://a"]["last"] -= 25 * 3600 n = orchestrator._datasheet_round(Conn(), {"run_id": 5, "cfg": dcfg, "metrics": {}}, dcfg, cs, set(), [], 10) check("datasheet: seed due again after 24 h, already-seen PDFs skipped", crawled[-1] == "https://a" and n == 0) n = orchestrator._datasheet_round(Conn(), {"run_id": 6, "cfg": dcfg, "metrics": {}}, dcfg, cs, set(), [{}] * 10, 10) check("datasheet: skipped when the cap is already reached", n == 0 and len(crawled) == 4) check("datasheet: disabled by config", orchestrator._datasheet_round( Conn(), {"run_id": 7, "cfg": {}, "metrics": {}}, {"use_datasheets": False}, cs, set(), [], 10) == 0) # --- 9. topic walk --------------------------------------------------------------------- EVENTS.clear(); STATE_KV.clear() pages = [] def fake_topic(tid, per_page, cursor, min_score, query="", sort=""): pages.append((tid, cursor)) if cursor == "*": return [cand(f"{tid} p1", f"https://t/{tid}/1.pdf")], "c2" return [cand(f"{tid} p2", f"https://t/{tid}/2.pdf")], None pc.list_openalex_topic = fake_topic tcfg = {"use_openalex_topics": True, "use_openalex": True, "openalex_topic_names": "T11664 Fiber-reinforced polymer composites; T10219 Mechanical Behavior of Composites", "openalex_topics_per_cycle": 1, "openalex_topic_page": 50, "min_relevance_score": 4} got = orchestrator._discover_topics(Conn(), {"run_id": 8, "cfg": tcfg, "metrics": {}}, tcfg) check("topics: ids parsed from config, one topic per cycle", pages == [("T11664", "*")] and len(got) == 1) check("topics: cursor persisted", STATE_KV.get("openalex_topic_cursor:T11664") == "c2") got = orchestrator._discover_topics(Conn(), {"run_id": 9, "cfg": tcfg, "metrics": {}}, tcfg) check("topics: round-robin to the other topic", pages[-1] == ("T10219", "*")) got = orchestrator._discover_topics(Conn(), {"run_id": 10, "cfg": tcfg, "metrics": {}}, tcfg) check("topics: back to the first topic at its saved cursor", pages[-1] == ("T11664", "c2")) check("topics: exhausted topic parked as done", STATE_KV.get("openalex_topic_cursor:T11664") == "done") got = orchestrator._discover_topics(Conn(), {"run_id": 11, "cfg": tcfg, "metrics": {}}, tcfg) got = orchestrator._discover_topics(Conn(), {"run_id": 12, "cfg": tcfg, "metrics": {}}, tcfg) check("topics: a done topic is skipped", ("T11664", "done") not in pages) pc.close_source("api.openalex.org", 600, "HTTP 429: budget") st12 = {"run_id": 13, "cfg": tcfg, "metrics": {}} got = orchestrator._discover_topics(Conn(), st12, tcfg) check("topics: closed OpenAlex → no call, one warn", got == [] and any("openalex skipped" in e[2] for e in EVENTS)) pc.reopen_sources() # --- 10. query grid ------------------------------------------------------------------ qs = query_grid.build() check("grid: a few hundred distinct intents", 200 <= len(qs) <= 600 and len(set(qs)) == len(qs)) check("grid: deterministic", qs == query_grid.build()) check("grid: covers PEEK, PPS and glass/carbon fibre", any("PEEK" in q for q in qs) and any("PPS" in q for q in qs) and any("glass fiber" in q for q in qs) and any("carbon fiber" in q for q in qs)) # --- 11. UDSpace lane (DSpace 7 REST: discover → bundles → bitstreams) ---------------- SCRIPT.clear(); CALLS.clear() _item = {"uuid": "11111111-1111-1111-1111-111111111111", "type": "item", "name": "Void consolidation of thermoplastic composites via non-autoclave processing", "metadata": {"dc.description.abstract": [{"value": "carbon fiber PEEK laminate tensile modulus"}], "dc.date.issued": [{"value": "2021-05-01"}], "dc.identifier.uri": [{"value": "https://udspace.udel.edu/handle/19716/99999"}]}} _junk = {"uuid": "22222222-2222-2222-2222-222222222222", "type": "item", "name": "Medieval poetry and its readers", "metadata": {"dc.description.abstract": [{"value": "verse"}]}} SCRIPT["https://udspace.udel.edu/server/api/discover/search/objects"] = FakeResp(200, json_body={ "_embedded": {"searchResult": {"_embedded": {"objects": [ {"_embedded": {"indexableObject": _item}}, {"_embedded": {"indexableObject": _junk}}]}}}}) SCRIPT["https://udspace.udel.edu/server/api/core/items/11111111-1111-1111-1111-111111111111/bundles"] = FakeResp(200, json_body={ "_embedded": {"bundles": [{"uuid": "bbbb", "name": "LICENSE"}, {"uuid": "aaaa", "name": "ORIGINAL"}]}}) SCRIPT["https://udspace.udel.edu/server/api/core/bundles/aaaa/bitstreams"] = FakeResp(200, json_body={ "_embedded": {"bitstreams": [ {"name": "thesis.pdf", "_links": {"content": {"href": "https://udspace.udel.edu/server/api/core/bitstreams/cccc/content"}}}]}}) got = list(pc.search_udspace("thermoplastic composite consolidation", 5)) check("udspace: one candidate (the poetry item fails the gate before any bundle call)", len(got) == 1 and got[0].source == "udspace" and got[0].year == "2021") check("udspace: PDF is the ORIGINAL bundle's bitstream content link", got[0].pdf_url.endswith("/bitstreams/cccc/content")) check("udspace: handle used as the DOI-like key and landing page", got[0].doi == "udspace.udel.edu/handle/19716/99999" and got[0].landing_url.endswith("/19716/99999")) check("udspace: no bundle calls for the rejected item", not any("22222222" in u for u in CALLS)) # --- 12. CORE lane (key-gated; hosted download URLs) ---------------------------------- SCRIPT.clear(); CALLS.clear() pc.CORE_API_KEY = "" check("core: inactive without a key (no call)", list(pc.search_core("thermoplastic composite", 5)) == [] and CALLS == []) pc.CORE_API_KEY = "test-key" SCRIPT["https://api.core.ac.uk/v3/search/works"] = FakeResp(200, json_body={"results": [ {"id": 1, "title": "Tensile behaviour of carbon fibre PEEK thermoplastic composites", "abstract": "thermoplastic composite tensile modulus", "yearPublished": 2020, "doi": "https://doi.org/10.5/core.1", "downloadUrl": "https://core.ac.uk/download/1.pdf"}, {"id": 2, "title": "Poems", "abstract": "", "downloadUrl": "https://core.ac.uk/download/2.pdf"}, {"id": 3, "title": "Thermoplastic composite welding review", "abstract": "composite fiber", "yearPublished": 2019, "downloadUrl": ""}, ]}) got = list(pc.search_core("thermoplastic composite", 5)) check("core: relevant work with a hosted download URL becomes a candidate; junk and URL-less dropped", [c.doi for c in got] == ["10.5/core.1"] and got[0].pdf_url == "https://core.ac.uk/download/1.pdf" and got[0].year == "2020" and got[0].source == "core") check("core: the key travels as a bearer token", any("core.ac.uk" in u for u in CALLS)) pc.CORE_API_KEY = "" print() print("all tests passed" if not fails else f"{len(fails)} FAILED: {fails}") sys.exit(1 if fails else 0)