#!/usr/bin/env python3 """Publish Vertica↔Snowflake pair dataset to the Hugging Face Hub.""" from __future__ import annotations import argparse import json from pathlib import Path from morphsql.eval.pairs import ensure_pairs_file, load_pairs def main() -> None: parser = argparse.ArgumentParser(description="Publish MorphSQL pair dataset") parser.add_argument( "--repo", default="dgvj-work/vertica-snowflake-pairs", help="Hub dataset repo id (user/name)", ) parser.add_argument("--private", action="store_true") args = parser.parse_args() path = ensure_pairs_file() pairs = load_pairs() print(f"Loaded {len(pairs)} pairs from {path}") try: from datasets import Dataset from huggingface_hub import HfApi, login except ImportError as exc: raise SystemExit( "Install: pip install datasets huggingface_hub\n" + str(exc) ) from exc # Prefer token from env; login() is interactive otherwise api = HfApi() try: api.whoami() except Exception: login() ds = Dataset.from_list(pairs) ds.push_to_hub(args.repo, private=args.private) readme = f"""--- license: apache-2.0 task_categories: - text2text-generation language: - en tags: - sql - code - migration - snowflake - vertica - dbt - evaluation size_categories: - n<1K --- # Vertica / Oracle / Redshift / BigQuery → Snowflake SQL pairs Synthetic + curated migration pairs for **MorphSQL** evals and fine-tuning. - Rows: {len(pairs)} - Fields: `id`, `category`, `source_dialect`, `target_dialect`, `source_sql`, `target_sql`, `notes` - Categories: function, date, aggregate, ddl, ml_feature Space: https://huggingface.co/spaces/dgvj-work/morphsql Code: https://github.com/dgvj-work/morphsql Author: Digvijay Waghela """ api.upload_file( path_or_fileobj=readme.encode("utf-8"), path_in_repo="README.md", repo_id=args.repo, repo_type="dataset", ) # Also upload raw jsonl api.upload_file( path_or_fileobj=str(path), path_in_repo="vertica_snowflake_pairs.jsonl", repo_id=args.repo, repo_type="dataset", ) print(f"Published https://huggingface.co/datasets/{args.repo}") if __name__ == "__main__": main()