morphsql / scripts /publish_dataset.py
waghelad's picture
Upload folder using huggingface_hub
12eff8e verified
Raw
History Blame Contribute Delete
2.33 kB
#!/usr/bin/env python3
"""Publish Vertica↔Snowflake pair dataset to the Hugging Face Hub."""
from __future__ import annotations
import argparse
import json
from pathlib import Path
from morphsql.eval.pairs import ensure_pairs_file, load_pairs
def main() -> None:
parser = argparse.ArgumentParser(description="Publish MorphSQL pair dataset")
parser.add_argument(
"--repo",
default="dgvj-work/vertica-snowflake-pairs",
help="Hub dataset repo id (user/name)",
)
parser.add_argument("--private", action="store_true")
args = parser.parse_args()
path = ensure_pairs_file()
pairs = load_pairs()
print(f"Loaded {len(pairs)} pairs from {path}")
try:
from datasets import Dataset
from huggingface_hub import HfApi, login
except ImportError as exc:
raise SystemExit(
"Install: pip install datasets huggingface_hub\n" + str(exc)
) from exc
# Prefer token from env; login() is interactive otherwise
api = HfApi()
try:
api.whoami()
except Exception:
login()
ds = Dataset.from_list(pairs)
ds.push_to_hub(args.repo, private=args.private)
readme = f"""---
license: apache-2.0
task_categories:
- text2text-generation
language:
- en
tags:
- sql
- code
- migration
- snowflake
- vertica
- dbt
- evaluation
size_categories:
- n<1K
---
# Vertica / Oracle / Redshift / BigQuery → Snowflake SQL pairs
Synthetic + curated migration pairs for **MorphSQL** evals and fine-tuning.
- Rows: {len(pairs)}
- Fields: `id`, `category`, `source_dialect`, `target_dialect`, `source_sql`, `target_sql`, `notes`
- Categories: function, date, aggregate, ddl, ml_feature
Space: https://huggingface.co/spaces/dgvj-work/morphsql
Code: https://github.com/dgvj-work/morphsql
Author: Digvijay Waghela
"""
api.upload_file(
path_or_fileobj=readme.encode("utf-8"),
path_in_repo="README.md",
repo_id=args.repo,
repo_type="dataset",
)
# Also upload raw jsonl
api.upload_file(
path_or_fileobj=str(path),
path_in_repo="vertica_snowflake_pairs.jsonl",
repo_id=args.repo,
repo_type="dataset",
)
print(f"Published https://huggingface.co/datasets/{args.repo}")
if __name__ == "__main__":
main()