File size: 8,385 Bytes
fafbad3 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 | """Area-level analytics over the corpus: trends, prominent authors, award standouts.
Built on the deterministic venue-to-area registry (``areas.py``). Authors are parsed
from the stored author string; an author's per-area counts reflect the venues they
publish in. These functions support literature-review work: which areas are growing,
who the prominent authors are, and which papers were recognized with awards.
"""
from __future__ import annotations
import collections
import re
import sqlite3
from pathlib import Path
from src.areas import area_for
from src.awards import AwardMatch, load_award_records, match_awards_to_corpus
from src.tiers import TOP4, TOP4_REGIONAL, TOP_TIER, tier_for, weight_for
_AUTHOR_SPLIT = re.compile(r"\s*,\s*|\s+and\s+")
def _split_authors(raw: str | None) -> list[str]:
if not raw:
return []
return [
# Keep DBLP's numeric suffix. It distinguishes homonyms; removing it
# silently merges different people and invalidates author rankings.
name.strip()
for name in _AUTHOR_SPLIT.split(raw)
if name.strip()
]
def area_year_counts(db_path: Path) -> dict[str, dict[int, int]]:
"""Papers per area per year: ``{area: {year: count}}`` (trend signal)."""
counts: dict[str, dict[int, int]] = collections.defaultdict(
lambda: collections.defaultdict(int)
)
connection = sqlite3.connect(db_path)
try:
for event, year in connection.execute(
"select event, year from papers where year is not null"
):
counts[area_for(event)][int(year)] += 1
finally:
connection.close()
return {area: dict(sorted(years.items())) for area, years in counts.items()}
def top_authors(
db_path: Path, area: str | None = None, limit: int = 15
) -> list[tuple[str, int]]:
"""Most prolific authors, optionally restricted to a single area."""
counter: collections.Counter[str] = collections.Counter()
connection = sqlite3.connect(db_path)
try:
for authors, event in connection.execute("select authors, event from papers"):
if area is not None and area_for(event) != area:
continue
counter.update(_split_authors(authors))
finally:
connection.close()
return counter.most_common(limit)
def reference_authors(
db_path: Path,
topic: str | None = None,
area: str | None = None,
limit: int = 20,
awards_dir: Path | None = None,
) -> list[dict]:
"""Tier-weighted ranking of the reference authors for a topic or area.
A raw publication count treats a workshop paper and an IEEE S&P paper as
equal; this ranking does not. Each paper contributes its venue-tier weight
(``tiers.WEIGHT_BY_TIER``) to every listed author, so recurring top-tier
names dominate the list. The optional ``topic`` restricts scope with a
case-insensitive match over title and abstract; ``area`` restricts by the
venue-to-area registry. Deterministic: ties break by top-4 count, paper
count, then name.
This is a corpus-visibility heuristic, not a citation/quality/authority
metric. Each entry exposes the evidence used by the heuristic so users do
not have to interpret an opaque score.
"""
sql = "select paper_id, authors, event, year from papers where authors is not null"
params: list = []
if topic:
sql += " and (title like ? or abstract like ?)"
pattern = f"%{topic}%"
params.extend([pattern, pattern])
awarded_ids: set[str] = set()
if awards_dir is not None and awards_dir.exists():
matched, _ = match_awards_to_corpus(load_award_records(awards_dir), db_path)
awarded_ids = {match.paper_id for match in matched}
stats: dict[str, dict] = {}
connection = sqlite3.connect(db_path)
try:
for paper_id, authors, event, year in connection.execute(sql, params):
if area is not None and area_for(event) != area:
continue
tier = tier_for(event)
weight = weight_for(tier)
for author in _split_authors(authors):
entry = stats.setdefault(author, {
"author": author,
"score": 0.0,
"papers": 0,
"top4": 0,
"top_tier": 0,
"top4_regional": 0,
"awards": 0,
"first_year": year,
"last_year": year,
"_venues": collections.Counter(),
})
entry["score"] += weight
entry["papers"] += 1
if tier == TOP4:
entry["top4"] += 1
elif tier == TOP_TIER:
entry["top_tier"] += 1
elif tier == TOP4_REGIONAL:
entry["top4_regional"] += 1
if paper_id in awarded_ids:
entry["awards"] += 1
if year is not None:
entry["first_year"] = min(entry["first_year"] or year, year)
entry["last_year"] = max(entry["last_year"] or year, year)
if event:
entry["_venues"][event] += 1
finally:
connection.close()
ranked = sorted(
stats.values(),
key=lambda e: (-e["score"], -e["top4"], -e["papers"], e["author"]),
)[:limit]
for entry in ranked:
entry["venues"] = [venue for venue, _ in entry.pop("_venues").most_common(3)]
entry["score"] = round(entry["score"], 1)
return ranked
def topic_trend(
db_path: Path,
topic: str,
area: str | None = None,
year_start: int | None = None,
) -> dict:
"""Yearly volume and corpus share for a topic, plus its main venues.
Answers two review questions at once: "is this topic growing?" and
"where does it get published?". The topic matches title and abstract
case-insensitively. ``share_pct`` normalizes each year's topic count by
that year's corpus size (within the same ``area`` scope), so a growing
corpus does not masquerade as a growing topic. Abstracts exist only where
the curation policy backfills them, so counts are a lower bound for
venues outside the enriched layers.
"""
pattern = f"%{topic}%"
matched_by_year: dict[int, int] = collections.defaultdict(int)
total_by_year: dict[int, int] = collections.defaultdict(int)
venue_counter: collections.Counter[str] = collections.Counter()
connection = sqlite3.connect(db_path)
try:
rows = connection.execute(
"select event, year, (title like ? or abstract like ?) as matched "
"from papers where year is not null",
(pattern, pattern),
)
for event, year, matched in rows:
if area is not None and area_for(event) != area:
continue
year = int(year)
if year_start is not None and year < year_start:
continue
total_by_year[year] += 1
if matched:
matched_by_year[year] += 1
if event:
venue_counter[event] += 1
finally:
connection.close()
by_year = [
{
"year": year,
"papers": matched_by_year.get(year, 0),
"share_pct": round(100.0 * matched_by_year.get(year, 0) / total, 2),
}
for year, total in sorted(total_by_year.items())
]
return {
"topic": topic,
"total": sum(matched_by_year.values()),
"by_year": by_year,
"top_venues": venue_counter.most_common(5),
}
def awarded_by_area(awards_dir: Path, db_path: Path) -> dict[str, list[AwardMatch]]:
"""Award-winning corpus papers grouped by research area."""
matched, _ = match_awards_to_corpus(load_award_records(awards_dir), db_path)
connection = sqlite3.connect(db_path)
try:
area_by_id = {
paper_id: area_for(event)
for paper_id, event in connection.execute(
"select paper_id, event from papers"
)
}
finally:
connection.close()
by_area: dict[str, list[AwardMatch]] = collections.defaultdict(list)
for match in matched:
by_area[area_by_id.get(match.paper_id, "unknown")].append(match)
return dict(by_area)
|