Spaces:
Running
Running
Mathias Heider
Claude Fable 5.1
Candidate queue: a bulk discovery pass fills agent_candidates, cycles drain it; UDSpace + CORE lanes
68b949f unverified Download test_sources.py from aim4composites/AutonomousAgent: direct link, hf CLI and curl.
- Browser
- Download file 25.7 kB
-
https://huggingface.co/spaces/aim4composites/AutonomousAgent/resolve/main/test_sources.py
- Command line
-
hf download hf://spaces/aim4composites/AutonomousAgent/test_sources.py
-
curl -L -o test_sources.py https://huggingface.co/spaces/aim4composites/AutonomousAgent/resolve/main/test_sources.py
25.7 kB
| """Unit tests for the Sep 2026 "more papers" crawler/orchestrator changes. | |
| No database and no network: HTTP is replaced with scripted responses. Run | |
| with: python test_sources.py | |
| Covers: per-host circuit breaker (budget 429s), arXiv query builder, gate | |
| relief without abstract, OpenAlex repository-first URL ranking + landing | |
| page, Semantic Scholar bulk gating, citation_pdf_url fallback, the | |
| orchestrator skipping a closed source, the datasheet-round schedule, the | |
| topic walk cursors and the query grid. | |
| """ | |
| import json | |
| import os | |
| import sys | |
| import tempfile | |
| import time | |
| from pathlib import Path | |
| os.environ["AGENT_USE_S2"] = "0" | |
| os.environ["AGENT_USE_OPENALEX_TOPICS"] = "0" | |
| os.environ["AGENT_USE_DATASHEETS"] = "0" | |
| import pdf_crawler as pc | |
| from agent import agdb, orchestrator, query_grid | |
| fails = [] | |
| def check(name, cond): | |
| print(("PASS " if cond else "FAIL ") + name) | |
| if not cond: | |
| fails.append(name) | |
| # --- fake HTTP --------------------------------------------------------------- | |
| class FakeResp: | |
| def __init__(self, status, headers=None, body=b"", url="", json_body=None): | |
| self.status_code = status | |
| self.headers = headers or {} | |
| self._body = body if isinstance(body, bytes) else body.encode() | |
| self.url = url | |
| self._json = json_body | |
| if json_body is not None: | |
| self._body = json.dumps(json_body).encode() | |
| self.headers.setdefault("Content-Type", "application/json") | |
| def text(self): | |
| return self._body.decode(errors="replace") | |
| def json(self): | |
| if self._json is None: | |
| return json.loads(self._body.decode()) | |
| return self._json | |
| def iter_content(self, chunk_size=65536): | |
| yield self._body | |
| SCRIPT: dict = {} # url prefix -> list of responses (popped in order) or single | |
| CALLS: list = [] | |
| SLEEPS: list = [] | |
| def fake_get(url, timeout=None, **kw): | |
| if kw.get("params"): | |
| import urllib.parse | |
| url = url + ("&" if "?" in url else "?") + urllib.parse.urlencode(kw["params"]) | |
| CALLS.append(url) | |
| for prefix, resp in SCRIPT.items(): | |
| if url.startswith(prefix): | |
| if isinstance(resp, list): | |
| return resp.pop(0) if len(resp) > 1 else resp[0] | |
| return resp | |
| return FakeResp(404) | |
| pc.SESSION.get = fake_get | |
| pc.time.sleep = lambda s: SLEEPS.append(s) | |
| pc.THROTTLE.wait = lambda url: None | |
| # --- 1. circuit breaker -------------------------------------------------------- | |
| pc.reopen_sources() | |
| CALLS.clear(); SLEEPS.clear() | |
| budget_429 = FakeResp(429, {"Retry-After": "65169", "X-RateLimit-Remaining": "0"}, | |
| json_body={"error": "Rate limit exceeded", | |
| "message": "Insufficient budget. This request costs $0.001 " | |
| "but you only have $0 remaining. Resets at midnight UTC."}) | |
| SCRIPT["https://api.openalex.org/"] = budget_429 | |
| r = pc.http_get("https://api.openalex.org/works?search=x") | |
| check("budget 429 returns None without retrying", r is None and len(CALLS) == 1) | |
| check("budget 429 does not sleep", SLEEPS == []) | |
| reason = pc.source_closed("api.openalex.org") | |
| check("host is closed with the server's reason", bool(reason) and "Insufficient budget" in reason) | |
| CALLS.clear() | |
| r = pc.http_get("https://api.openalex.org/works?search=y") | |
| check("closed host short-circuits (no HTTP call)", r is None and CALLS == []) | |
| check("other hosts unaffected", pc.source_closed("export.arxiv.org") is None) | |
| cands = list(pc.search_openalex("thermoplastic composite", 5)) | |
| check("search_openalex yields nothing while closed", cands == [] and CALLS == []) | |
| pc.reopen_sources() | |
| check("reopen_sources clears the closure", pc.source_closed("api.openalex.org") is None) | |
| # short Retry-After is still a normal retry | |
| CALLS.clear(); SLEEPS.clear() | |
| SCRIPT["https://api.openalex.org/"] = [FakeResp(429, {"Retry-After": "5"}), FakeResp(200, json_body={"results": []})] | |
| r = pc.http_get("https://api.openalex.org/works?search=z") | |
| check("short Retry-After: slept then retried", SLEEPS == [5.0] and r is not None and r.status_code == 200) | |
| check("short Retry-After does not close the host", pc.source_closed("api.openalex.org") is None) | |
| # a huge Retry-After closes even without the X-RateLimit header | |
| CALLS.clear(); SLEEPS.clear() | |
| SCRIPT["https://api.semanticscholar.org/"] = FakeResp(429, {"Retry-After": "7200"}) | |
| r = pc.http_get("https://api.semanticscholar.org/graph/v1/paper/search/bulk?query=a") | |
| check("Retry-After above cap closes the host", r is None and pc.source_closed("api.semanticscholar.org") is not None and SLEEPS == []) | |
| pc.reopen_sources() | |
| # --- 2. arXiv query builder ------------------------------------------------------ | |
| q = pc.arxiv_search_query("thermoplastic composite mechanical properties") | |
| check("arXiv: phrase becomes ANDed terms", q == "all:thermoplastic AND all:composite AND all:mechanical") | |
| q = pc.arxiv_search_query("PA66 glass fiber composite mechanical properties") | |
| check("arXiv: matrix and reinforcement beat generic words", q == "all:pa66 AND all:glass AND (all:fiber OR all:fibre)") | |
| q = pc.arxiv_search_query("carbon fiber PEEK composite tensile properties") | |
| check("arXiv: PEEK kept, fibre spelled both ways", "all:peek" in q and "(all:fiber OR all:fibre)" in q and q.count(" AND ") == 2) | |
| q2 = pc.arxiv_search_query("carbon fiber PEEK composite tensile properties", max_terms=2) | |
| check("arXiv: two-term fallback drops the weakest term", q2 == "(all:fiber OR all:fibre) AND all:peek" or q2 == "all:carbon AND all:peek") | |
| check("arXiv: single word passes through", pc.arxiv_search_query("PEEK") == "all:peek") | |
| # search_arxiv retries broader when the specific query is empty | |
| CALLS.clear() | |
| EMPTY = '<?xml version="1.0"?><feed xmlns="http://www.w3.org/2005/Atom"></feed>' | |
| ONE = ('<?xml version="1.0"?><feed xmlns="http://www.w3.org/2005/Atom"><entry>' | |
| '<title>Carbon fibre PEEK laminates: tensile modulus</title><summary>thermoplastic composite ' | |
| 'carbon fiber PEEK tensile strength</summary><published>2024-01-01</published>' | |
| '<link title="pdf" href="https://arxiv.org/pdf/2401.00001"/></entry></feed>') | |
| SCRIPT["https://export.arxiv.org/"] = [FakeResp(200, body=EMPTY), FakeResp(200, body=ONE)] | |
| got = list(pc.search_arxiv("carbon fiber PEEK composite tensile properties", 5)) | |
| check("arXiv: empty result triggers one broader retry", len(CALLS) == 2 and len(got) == 1) | |
| check("arXiv: retry used fewer terms", CALLS[1].count("AND") < CALLS[0].count("AND")) | |
| # --- 3. gate relief ---------------------------------------------------------------- | |
| check("gate: full score with abstract", pc.effective_min_score(4, "some abstract") == 4) | |
| check("gate: relief without abstract", pc.effective_min_score(4, "") == 2) | |
| check("gate: floor at 1", pc.effective_min_score(1, "") == 1) | |
| check("_passes_gate: title-only paper accepted with relief", | |
| pc._passes_gate("Fatigue of thermoplastic laminates", "", "https://x/a.pdf", 4)) | |
| check("_passes_gate: irrelevant title rejected", not pc._passes_gate("Bird migration", "", "https://x/b.pdf", 4)) | |
| # --- 4. OpenAlex URL ranking + candidate ------------------------------------------ | |
| work = { | |
| "display_name": "Interlaminar shear strength of carbon fibre PEEK laminates", | |
| "doi": "https://doi.org/10.1/abc", "publication_year": 2023, | |
| "abstract_inverted_index": {"thermoplastic": [0], "composite": [1], "tensile": [2]}, | |
| "best_oa_location": {"pdf_url": "https://www.mdpi.com/x/pdf", "landing_page_url": "https://www.mdpi.com/x", | |
| "version": "publishedVersion", "source": {"type": "journal"}}, | |
| "locations": [ | |
| {"pdf_url": "https://www.mdpi.com/x/pdf", "source": {"type": "journal"}}, | |
| {"pdf_url": "https://hal.science/hal-1/document", "landing_page_url": "https://hal.science/hal-1", | |
| "version": "acceptedVersion", "source": {"type": "repository"}}, | |
| {"pdf_url": "https://europepmc.org/articles/PMC1?pdf=render", "version": "publishedVersion", | |
| "source": {"type": "repository"}}, | |
| {"pdf_url": "https://x.org/slides.pdf", "source": {"type": "repository"}}, | |
| ], | |
| } | |
| urls, landing = pc._rank_oa_urls(work) | |
| check("ranking: repository publishedVersion first", urls[0].startswith("https://europepmc.org")) | |
| check("ranking: repository acceptedVersion second", urls[1].startswith("https://hal.science")) | |
| check("ranking: publisher URL last", urls[-1] == "https://www.mdpi.com/x/pdf") | |
| check("ranking: junk URL dropped", not any("slides" in u for u in urls)) | |
| check("ranking: landing page comes from the first ranked location", landing == "https://hal.science/hal-1") | |
| cand = pc._openalex_work_to_candidate(work, "q", 4) | |
| check("candidate: pdf_url is the top-ranked URL, alts follow", cand.pdf_url == urls[0] and cand.alt_urls == urls[1:]) | |
| check("candidate: doi stripped of prefix", cand.doi == "10.1/abc") | |
| work_noabs = dict(work, abstract_inverted_index=None, | |
| display_name="Consolidation quality of AFP-made thermoplastic parts") | |
| check("candidate: title-only work passes with relief", | |
| pc._openalex_work_to_candidate(work_noabs, "q", 4) is not None) | |
| check("candidate: title-only work rejected without relief", | |
| pc._openalex_work_to_candidate(dict(work_noabs, display_name="Bird migration"), "q", 4) is None) | |
| # OpenAlex request carries the key and the OA-status filter | |
| pc.OPENALEX_API_KEY = "k-test" | |
| CALLS.clear() | |
| SCRIPT["https://api.openalex.org/"] = FakeResp(200, json_body={"results": [work]}) | |
| got = list(pc.search_openalex("thermoplastic composite", 3)) | |
| check("openalex request has api_key, oa_status filter and per-page>=25", | |
| "api_key=k-test" in CALLS[0] and "oa_status:gold%7Chybrid%7Cgreen" in CALLS[0].replace("|", "%7C") | |
| and "per-page=25" in CALLS[0]) | |
| check("openalex yields the ranked candidate", len(got) == 1 and got[0].pdf_url.startswith("https://europepmc.org")) | |
| pc.OPENALEX_API_KEY = "" | |
| # topic page call | |
| CALLS.clear() | |
| SCRIPT["https://api.openalex.org/"] = FakeResp(200, json_body={"results": [work], "meta": {"next_cursor": "abc"}}) | |
| cands, nxt = pc.list_openalex_topic("T11664", 50, "*", 4, query="topic:x") | |
| check("topic page: filter call (no search=), cursor and sort present", | |
| "search=" not in CALLS[0] and "topics.id:T11664" in CALLS[0] and "cursor=" in CALLS[0] | |
| and "cited_by_count" in CALLS[0]) | |
| check("topic page: candidates + next cursor", len(cands) == 1 and nxt == "abc" and cands[0].query == "topic:x") | |
| SCRIPT["https://api.openalex.org/"] = FakeResp(200, json_body={"results": [], "meta": {"next_cursor": None}}) | |
| cands, nxt = pc.list_openalex_topic("T11664", 50, "abc", 4) | |
| check("topic page: exhausted → next cursor None", cands == [] and nxt is None) | |
| pc.close_source("api.openalex.org", 100, "test") | |
| cands, nxt = pc.list_openalex_topic("T11664", 50, "abc", 4) | |
| check("topic page: closed source keeps the cursor", cands == [] and nxt == "abc") | |
| pc.reopen_sources() | |
| # --- 5. Semantic Scholar bulk ------------------------------------------------------ | |
| papers = [ | |
| {"title": "Glass fiber polypropylene composite tensile modulus", "abstract": "thermoplastic composite tensile", | |
| "year": 2020, "externalIds": {"DOI": "10.2/a"}, "openAccessPdf": {"url": "https://www.mdpi.com/a/pdf"}}, | |
| {"title": "PEEK carbon fibre laminate fatigue", "abstract": "thermoplastic composite laminate", | |
| "year": 2021, "externalIds": {"DOI": "10.2/b"}, "openAccessPdf": {"url": "https://osti.gov/b.pdf"}}, | |
| {"title": "arXiv-only preprint on PEEK composites", "abstract": "thermoplastic composite", | |
| "year": 2022, "externalIds": {"ArXiv": "2201.1"}, "openAccessPdf": {"url": "https://arxiv.org/pdf/2201.1"}}, | |
| {"title": "No pdf thermoplastic composite", "abstract": "", "year": 2019, "externalIds": {}, "openAccessPdf": None}, | |
| {"title": "Bird migration", "abstract": "", "year": 2018, "externalIds": {"DOI": "10.2/c"}, | |
| "openAccessPdf": {"url": "https://x/c.pdf"}}, | |
| ] | |
| CALLS.clear() | |
| SCRIPT["https://api.semanticscholar.org/"] = FakeResp(200, json_body={"total": 5, "data": papers}) | |
| got = list(pc.search_semantic_scholar("thermoplastic composite", 1)) | |
| check("S2: bulk endpoint with openAccessPdf flag", "search/bulk" in CALLS[0] and "openAccessPdf=" in CALLS[0]) | |
| check("S2: limit keeps the highest-scoring gated paper", len(got) == 1 and got[0].title.startswith("PEEK")) | |
| got = list(pc.search_semantic_scholar("thermoplastic composite", 10)) | |
| check("S2: arXiv-only, no-pdf and irrelevant papers dropped", [g.doi for g in got] == ["10.2/b", "10.2/a"]) | |
| # --- 6. citation_pdf_url fallback ------------------------------------------------------ | |
| pc.robots_allows = lambda url: True | |
| HTML1 = '<html><head><meta name="citation_pdf_url" content="/files/paper.pdf"></head></html>' | |
| HTML2 = '<html><head><meta content="https://r.org/p.pdf" name="citation_pdf_url"></head></html>' | |
| HTML3 = '<html><head><link rel="alternate" type="application/pdf" href="alt.pdf"></head></html>' | |
| SCRIPT["https://r.org/land1"] = FakeResp(200, {"Content-Type": "text/html"}, HTML1, url="https://r.org/land1") | |
| SCRIPT["https://r.org/land2"] = FakeResp(200, {"Content-Type": "text/html"}, HTML2, url="https://r.org/land2") | |
| SCRIPT["https://r.org/land3"] = FakeResp(200, {"Content-Type": "text/html"}, HTML3, url="https://r.org/land3") | |
| SCRIPT["https://r.org/none"] = FakeResp(200, {"Content-Type": "text/html"}, "<html></html>", url="https://r.org/none") | |
| check("landing: meta name/content order resolved relative", pc.pdf_url_from_landing_page("https://r.org/land1") == "https://r.org/files/paper.pdf") | |
| check("landing: meta content/name order", pc.pdf_url_from_landing_page("https://r.org/land2") == "https://r.org/p.pdf") | |
| check("landing: <link type=application/pdf>", pc.pdf_url_from_landing_page("https://r.org/land3") == "https://r.org/alt.pdf") | |
| check("landing: nothing found → None", pc.pdf_url_from_landing_page("https://r.org/none") is None) | |
| # download_pdf uses the landing page only after every listed URL failed | |
| attempts = [] | |
| def fake_download_url(cand, url, pdf_dir, state): | |
| attempts.append(url) | |
| if url == "https://r.org/files/paper.pdf": | |
| return {"filename": "ok.pdf", "url": url} | |
| return None | |
| _real_download_url = pc._download_url | |
| _real_unpaywall = pc.unpaywall_pdf_url | |
| pc._download_url = fake_download_url | |
| pc.unpaywall_pdf_url = lambda doi: None | |
| st = pc.CrawlerState(Path(tempfile.mkdtemp()) / "state.json") | |
| c = pc.Candidate(title="t", pdf_url="https://pub/x.pdf", source="openalex", doi="10.1/x", | |
| alt_urls=["https://repo/x.pdf"], landing_url="https://r.org/land1") | |
| row = pc.download_pdf(c, Path(tempfile.mkdtemp()), st) | |
| check("download_pdf: landing fallback after primary, alt and Unpaywall", | |
| row is not None and attempts == ["https://pub/x.pdf", "https://repo/x.pdf", "https://r.org/files/paper.pdf"]) | |
| check("download_pdf: landing page marked final (one look ever)", "https://r.org/land1" in st.seen_urls) | |
| attempts.clear() | |
| row = pc.download_pdf(c, Path(tempfile.mkdtemp()), st) | |
| check("download_pdf: landing page not re-fetched", attempts == ["https://pub/x.pdf", "https://repo/x.pdf"] and row is None) | |
| pc._download_url = _real_download_url | |
| pc.unpaywall_pdf_url = _real_unpaywall | |
| # --- 7. orchestrator: closed source is skipped, one event per cycle ------------------- | |
| EVENTS = [] | |
| STATE_KV: dict = {} | |
| class Conn: | |
| def close(self): | |
| pass | |
| agdb.connect = lambda: Conn() | |
| agdb.log_event = lambda conn, message, level="info", node="", run_id=None: EVENTS.append((level, node, message)) | |
| agdb.get_state = lambda conn, key, default=None: STATE_KV.get(key, default) | |
| agdb.set_state = lambda conn, key, value: STATE_KV.__setitem__(key, value) | |
| agdb.seen_doi = lambda conn, doi: False | |
| agdb.record_download = lambda conn, rec: None | |
| def cand(title, url, doi=""): | |
| return pc.Candidate(title=title, pdf_url=url, source="openalex", doi=doi, | |
| abstract="thermoplastic composite tensile carbon fiber PEEK") | |
| pc.search_openalex = lambda query, limit: iter([cand("oa " + query, f"https://oa/{query}.pdf", "10.9/" + query)]) | |
| pc.search_arxiv = lambda query, limit: iter([cand("ax " + query, f"https://ax/{query}.pdf")]) | |
| pc.search_semantic_scholar = lambda query, limit: iter([cand("s2 " + query, f"https://s2/{query}.pdf")]) | |
| cfg = {"max_per_query": 5, "min_relevance_score": 4, "use_openalex": True, "use_arxiv": True, | |
| "use_semantic_scholar": True, "use_ntrs": False} | |
| state = {"run_id": 1, "cfg": cfg, "metrics": {}} | |
| pc.close_source("api.openalex.org", 600, "HTTP 429: Insufficient budget") | |
| got = orchestrator._discover_round(Conn(), state, [(1, "q1"), (2, "q2")], 1) | |
| check("closed source skipped, other sources still searched", | |
| sorted(c.source for c in got) == ["openalex"] * 4 and all(not c.title.startswith("oa") for c in got)) | |
| warns = [e for e in EVENTS if e[0] == "warn" and "openalex skipped" in e[2]] | |
| check("one warn event for the closed source (not per query)", len(warns) == 1) | |
| orchestrator._discover_round(Conn(), state, [(3, "q3")], 2) | |
| warns = [e for e in EVENTS if e[0] == "warn" and "openalex skipped" in e[2]] | |
| check("still one warn event across rounds of the same cycle", len(warns) == 1) | |
| summary = [e for e in EVENTS if e[1] == "discover" and e[2].startswith("round 1:")][0][2] | |
| check("round summary carries a per-source breakdown", "arxiv 2" in summary and "semantic_scholar 2" in summary) | |
| pc.reopen_sources() | |
| # --- 8. datasheet round schedule ------------------------------------------------------ | |
| EVENTS.clear(); STATE_KV.clear() | |
| crawled = [] | |
| def fake_seed(seed): | |
| crawled.append(seed) | |
| yield pc.Candidate(title="TDS", pdf_url=f"{seed}/tds1.pdf", source="datasheet", query=seed) | |
| yield pc.Candidate(title="TDS2", pdf_url=f"{seed}/tds2.pdf", source="datasheet", query=seed) | |
| pc.crawl_datasheet_seed = fake_seed | |
| downloaded = [] | |
| def fake_download(c, pdf_dir, st): | |
| downloaded.append(c.pdf_url) | |
| st.mark_final(c.pdf_url) | |
| return {"filename": c.pdf_url.rsplit("/", 1)[-1], "url": c.pdf_url, "source": "datasheet", | |
| "title": c.title, "doi": "", "sha256": c.pdf_url, "query": c.query} | |
| pc.download_pdf = fake_download | |
| cs = pc.CrawlerState(Path(tempfile.mkdtemp()) / "state.json") | |
| dcfg = {"use_datasheets": True, "datasheet_seed_urls": "https://a; https://b; https://c", | |
| "datasheet_hours": 24, "datasheet_seeds_per_cycle": 2} | |
| dl = [] | |
| n = orchestrator._datasheet_round(Conn(), {"run_id": 2, "cfg": dcfg, "metrics": {}}, dcfg, cs, set(), dl, 10) | |
| check("datasheet: two due seeds crawled per cycle", crawled == ["https://a", "https://b"] and n == 4) | |
| n = orchestrator._datasheet_round(Conn(), {"run_id": 3, "cfg": dcfg, "metrics": {}}, dcfg, cs, set(), [], 10) | |
| check("datasheet: third seed next cycle, crawled ones parked for 24 h", crawled[-1] == "https://c" and len(crawled) == 3) | |
| n = orchestrator._datasheet_round(Conn(), {"run_id": 4, "cfg": dcfg, "metrics": {}}, dcfg, cs, set(), [], 10) | |
| check("datasheet: nothing due → no crawl", len(crawled) == 3 and n == 0) | |
| STATE_KV["datasheet_seeds"]["https://a"]["last"] -= 25 * 3600 | |
| n = orchestrator._datasheet_round(Conn(), {"run_id": 5, "cfg": dcfg, "metrics": {}}, dcfg, cs, set(), [], 10) | |
| check("datasheet: seed due again after 24 h, already-seen PDFs skipped", crawled[-1] == "https://a" and n == 0) | |
| n = orchestrator._datasheet_round(Conn(), {"run_id": 6, "cfg": dcfg, "metrics": {}}, dcfg, cs, set(), [{}] * 10, 10) | |
| check("datasheet: skipped when the cap is already reached", n == 0 and len(crawled) == 4) | |
| check("datasheet: disabled by config", orchestrator._datasheet_round( | |
| Conn(), {"run_id": 7, "cfg": {}, "metrics": {}}, {"use_datasheets": False}, cs, set(), [], 10) == 0) | |
| # --- 9. topic walk --------------------------------------------------------------------- | |
| EVENTS.clear(); STATE_KV.clear() | |
| pages = [] | |
| def fake_topic(tid, per_page, cursor, min_score, query="", sort=""): | |
| pages.append((tid, cursor)) | |
| if cursor == "*": | |
| return [cand(f"{tid} p1", f"https://t/{tid}/1.pdf")], "c2" | |
| return [cand(f"{tid} p2", f"https://t/{tid}/2.pdf")], None | |
| pc.list_openalex_topic = fake_topic | |
| tcfg = {"use_openalex_topics": True, "use_openalex": True, | |
| "openalex_topic_names": "T11664 Fiber-reinforced polymer composites; T10219 Mechanical Behavior of Composites", | |
| "openalex_topics_per_cycle": 1, "openalex_topic_page": 50, "min_relevance_score": 4} | |
| got = orchestrator._discover_topics(Conn(), {"run_id": 8, "cfg": tcfg, "metrics": {}}, tcfg) | |
| check("topics: ids parsed from config, one topic per cycle", pages == [("T11664", "*")] and len(got) == 1) | |
| check("topics: cursor persisted", STATE_KV.get("openalex_topic_cursor:T11664") == "c2") | |
| got = orchestrator._discover_topics(Conn(), {"run_id": 9, "cfg": tcfg, "metrics": {}}, tcfg) | |
| check("topics: round-robin to the other topic", pages[-1] == ("T10219", "*")) | |
| got = orchestrator._discover_topics(Conn(), {"run_id": 10, "cfg": tcfg, "metrics": {}}, tcfg) | |
| check("topics: back to the first topic at its saved cursor", pages[-1] == ("T11664", "c2")) | |
| check("topics: exhausted topic parked as done", STATE_KV.get("openalex_topic_cursor:T11664") == "done") | |
| got = orchestrator._discover_topics(Conn(), {"run_id": 11, "cfg": tcfg, "metrics": {}}, tcfg) | |
| got = orchestrator._discover_topics(Conn(), {"run_id": 12, "cfg": tcfg, "metrics": {}}, tcfg) | |
| check("topics: a done topic is skipped", ("T11664", "done") not in pages) | |
| pc.close_source("api.openalex.org", 600, "HTTP 429: budget") | |
| st12 = {"run_id": 13, "cfg": tcfg, "metrics": {}} | |
| got = orchestrator._discover_topics(Conn(), st12, tcfg) | |
| check("topics: closed OpenAlex → no call, one warn", got == [] and any("openalex skipped" in e[2] for e in EVENTS)) | |
| pc.reopen_sources() | |
| # --- 10. query grid ------------------------------------------------------------------ | |
| qs = query_grid.build() | |
| check("grid: a few hundred distinct intents", 200 <= len(qs) <= 600 and len(set(qs)) == len(qs)) | |
| check("grid: deterministic", qs == query_grid.build()) | |
| check("grid: covers PEEK, PPS and glass/carbon fibre", any("PEEK" in q for q in qs) and any("PPS" in q for q in qs) | |
| and any("glass fiber" in q for q in qs) and any("carbon fiber" in q for q in qs)) | |
| # --- 11. UDSpace lane (DSpace 7 REST: discover → bundles → bitstreams) ---------------- | |
| SCRIPT.clear(); CALLS.clear() | |
| _item = {"uuid": "11111111-1111-1111-1111-111111111111", "type": "item", | |
| "name": "Void consolidation of thermoplastic composites via non-autoclave processing", | |
| "metadata": {"dc.description.abstract": [{"value": "carbon fiber PEEK laminate tensile modulus"}], | |
| "dc.date.issued": [{"value": "2021-05-01"}], | |
| "dc.identifier.uri": [{"value": "https://udspace.udel.edu/handle/19716/99999"}]}} | |
| _junk = {"uuid": "22222222-2222-2222-2222-222222222222", "type": "item", | |
| "name": "Medieval poetry and its readers", | |
| "metadata": {"dc.description.abstract": [{"value": "verse"}]}} | |
| SCRIPT["https://udspace.udel.edu/server/api/discover/search/objects"] = FakeResp(200, json_body={ | |
| "_embedded": {"searchResult": {"_embedded": {"objects": [ | |
| {"_embedded": {"indexableObject": _item}}, {"_embedded": {"indexableObject": _junk}}]}}}}) | |
| SCRIPT["https://udspace.udel.edu/server/api/core/items/11111111-1111-1111-1111-111111111111/bundles"] = FakeResp(200, json_body={ | |
| "_embedded": {"bundles": [{"uuid": "bbbb", "name": "LICENSE"}, {"uuid": "aaaa", "name": "ORIGINAL"}]}}) | |
| SCRIPT["https://udspace.udel.edu/server/api/core/bundles/aaaa/bitstreams"] = FakeResp(200, json_body={ | |
| "_embedded": {"bitstreams": [ | |
| {"name": "thesis.pdf", "_links": {"content": {"href": "https://udspace.udel.edu/server/api/core/bitstreams/cccc/content"}}}]}}) | |
| got = list(pc.search_udspace("thermoplastic composite consolidation", 5)) | |
| check("udspace: one candidate (the poetry item fails the gate before any bundle call)", | |
| len(got) == 1 and got[0].source == "udspace" and got[0].year == "2021") | |
| check("udspace: PDF is the ORIGINAL bundle's bitstream content link", | |
| got[0].pdf_url.endswith("/bitstreams/cccc/content")) | |
| check("udspace: handle used as the DOI-like key and landing page", | |
| got[0].doi == "udspace.udel.edu/handle/19716/99999" and got[0].landing_url.endswith("/19716/99999")) | |
| check("udspace: no bundle calls for the rejected item", | |
| not any("22222222" in u for u in CALLS)) | |
| # --- 12. CORE lane (key-gated; hosted download URLs) ---------------------------------- | |
| SCRIPT.clear(); CALLS.clear() | |
| pc.CORE_API_KEY = "" | |
| check("core: inactive without a key (no call)", list(pc.search_core("thermoplastic composite", 5)) == [] and CALLS == []) | |
| pc.CORE_API_KEY = "test-key" | |
| SCRIPT["https://api.core.ac.uk/v3/search/works"] = FakeResp(200, json_body={"results": [ | |
| {"id": 1, "title": "Tensile behaviour of carbon fibre PEEK thermoplastic composites", | |
| "abstract": "thermoplastic composite tensile modulus", "yearPublished": 2020, | |
| "doi": "https://doi.org/10.5/core.1", "downloadUrl": "https://core.ac.uk/download/1.pdf"}, | |
| {"id": 2, "title": "Poems", "abstract": "", "downloadUrl": "https://core.ac.uk/download/2.pdf"}, | |
| {"id": 3, "title": "Thermoplastic composite welding review", "abstract": "composite fiber", | |
| "yearPublished": 2019, "downloadUrl": ""}, | |
| ]}) | |
| got = list(pc.search_core("thermoplastic composite", 5)) | |
| check("core: relevant work with a hosted download URL becomes a candidate; junk and URL-less dropped", | |
| [c.doi for c in got] == ["10.5/core.1"] and got[0].pdf_url == "https://core.ac.uk/download/1.pdf" | |
| and got[0].year == "2020" and got[0].source == "core") | |
| check("core: the key travels as a bearer token", any("core.ac.uk" in u for u in CALLS)) | |
| pc.CORE_API_KEY = "" | |
| print() | |
| print("all tests passed" if not fails else f"{len(fails)} FAILED: {fails}") | |
| sys.exit(1 if fails else 0) | |