topvenues-explorer / src /hf_export.py
sidneibarbieri's picture
Upload 46 files
fafbad3 verified
Raw
History Blame Contribute Delete
6.67 kB
"""Export the corpus as a Hugging Face dataset (Parquet + dataset card).
The export is derived from the released SQLite snapshot, so publishing to the
Hub is reproducible: the same snapshot always produces the same Parquet file
and the same dataset card statistics.
"""
from __future__ import annotations
import sqlite3
from pathlib import Path
import pandas as pd
from .areas import area_for
# Columns published on the Hub, in viewer display order. Bookkeeping columns
# (created_at/updated_at) and sparse internal fields (score/access) stay local.
_EXPORT_COLUMNS = [
"paper_id",
"key",
"title",
"authors",
"event",
"venue",
"area",
"year",
"paper_type",
"pages",
"url",
"ee",
"abstract",
"bibtex",
]
_CARD_TEMPLATE = """\
---
pretty_name: TopVenues Cybersecurity Corpus
license: odc-by
language:
- en
task_categories:
- text-retrieval
tags:
- cybersecurity
- security
- bibliography
- literature-review
- dblp
size_categories:
- 100K<n<1M
configs:
- config_name: default
data_files:
- split: train
path: data/train-*.parquet
---
# TopVenues Cybersecurity Corpus
A reproducible bibliographic corpus for cybersecurity literature reviews:
{total:,} papers ({year_min}–{year_max}) across {n_events} canonical venues,
with {n_abstracts:,} abstracts (concentrated on the security core and survey
venues by curation policy) and a BibTeX entry for every record.
- **Foundational paper:** [TopVenues: A Reproducible Corpus and Tooling
Substrate for Cybersecurity Literature Reviews
(arXiv:2606.18320)](https://arxiv.org/abs/2606.18320). That paper reports an
earlier, smaller frozen snapshot; this dataset is the expanded tool snapshot.
- **Code / tool:** <https://github.com/sidneibarbieri/topvenues-tool>
- **Pinned source of truth:** the gzipped SQLite snapshot shipped with the
tool; this dataset is a faithful Parquet export of the same frozen state.
## Usage
```python
from datasets import load_dataset
corpus = load_dataset("{repo_id}", split="train")
security = corpus.filter(lambda paper: paper["area"] == "security")
```
## Curation policy
The corpus has two layers, and the abstract column reflects that on purpose:
1. **Security core** ({security_total:,} papers): the security venues
(USENIX Security, ACM CCS, IEEE S&P, NDSS, ESORICS, RAID, ACSAC, and
others) are enriched with abstracts from open scholarly sources
({security_abstracts:,} abstracts, {security_pct:.1f}% coverage).
2. **Bibliographic context** (AI/ML/NLP, networking, mobile, and systems
venues): metadata and BibTeX only, kept as a stable cross-area
denominator for measurement studies.
## Schema
| Column | Description |
| --- | --- |
| `paper_id` | Stable identifier (DBLP-derived) |
| `key` | DBLP key |
| `title` | Paper title |
| `authors` | Author list as a single string |
| `event` | Canonical venue name (normalized) |
| `venue` | Venue string as recorded by DBLP |
| `area` | Research area (`security`, `ai`, `networks`, `mobile`, `systems`, `cross-area`) |
| `year` | Publication year |
| `paper_type` | Record type (article, proceedings, …) |
| `pages` | Page range |
| `url` / `ee` | DBLP URL / electronic edition (DOI or publisher link) |
| `abstract` | Abstract, where the curation policy backfills it |
| `bibtex` | DBLP-canonical BibTeX entry |
## Venues
{venue_table}
## Sources and licensing
Bibliographic metadata and BibTeX come from [DBLP](https://dblp.org)
(released CC0). Abstracts were collected from open scholarly sources
(Semantic Scholar, OpenAlex, CrossRef) and publisher pages; they remain the
copyright of their respective publishers and are included for research and
indexing purposes. The compilation is released under
[ODC-BY 1.0](https://opendatacommons.org/licenses/by/1-0/).
## Citation
```bibtex
@misc{{barbieri2026topvenues,
title = {{TopVenues: A Reproducible Corpus and Tooling Substrate for
Cybersecurity Literature Reviews}},
author = {{Barbieri, Sidnei and Ferraz, Agney Lopes Roth and
Pereira J{{\\'u}}nior, Louren{{\\c{{c}}}}o Alves}},
year = {{2026}},
eprint = {{2606.18320}},
archivePrefix = {{arXiv}},
primaryClass = {{cs.CR}},
url = {{https://arxiv.org/abs/2606.18320}}
}}
```
"""
def export_hf_dataset(
db_path: Path,
out_dir: Path,
repo_id: str = "sidneibarbieri/topvenues",
) -> dict:
"""Write ``data/train-*.parquet`` and ``README.md`` under ``out_dir``.
Returns the statistics used in the card so callers can display them.
"""
with sqlite3.connect(db_path) as conn:
df = pd.read_sql_query("SELECT * FROM papers", conn)
df["area"] = df["event"].map(area_for)
df = df[_EXPORT_COLUMNS].sort_values(
["area", "event", "year", "title"]
).reset_index(drop=True)
data_dir = out_dir / "data"
data_dir.mkdir(parents=True, exist_ok=True)
# Sharded to stay well under per-file web-upload limits and to let the
# Hub viewer stream row groups independently.
rows_per_shard = 20_000
shard_count = max(1, -(-len(df) // rows_per_shard))
parquet_paths = []
for shard_index in range(shard_count):
shard = df.iloc[shard_index * rows_per_shard:(shard_index + 1) * rows_per_shard]
path = data_dir / f"train-{shard_index:05d}-of-{shard_count:05d}.parquet"
shard.to_parquet(path, index=False)
parquet_paths.append(path)
has_abstract = df["abstract"].notna() & (df["abstract"] != "")
security = df["area"] == "security"
stats = {
"total": len(df),
"n_events": df["event"].nunique(),
"n_abstracts": int(has_abstract.sum()),
"year_min": int(df["year"].min()),
"year_max": int(df["year"].max()),
"security_total": int(security.sum()),
"security_abstracts": int((security & has_abstract).sum()),
"parquet_paths": parquet_paths,
}
stats["security_pct"] = 100.0 * stats["security_abstracts"] / stats["security_total"]
venue_rows = ["| Venue | Area | Papers | Abstracts |", "| --- | --- | ---: | ---: |"]
grouped = df.assign(has_abstract=has_abstract).groupby("event", sort=False)
summary = grouped.agg(
area=("area", "first"), papers=("title", "size"), abstracts=("has_abstract", "sum")
).sort_values("papers", ascending=False)
for event, row in summary.iterrows():
venue_rows.append(
f"| {event} | {row['area']} | {row['papers']:,} | {int(row['abstracts']):,} |"
)
card = _CARD_TEMPLATE.format(venue_table="\n".join(venue_rows), repo_id=repo_id, **stats)
(out_dir / "README.md").write_text(card, encoding="utf-8")
return stats