#!/usr/bin/env python3
"""Auto-sync script: push new/updated products from BEP40 products DB
into V.AISTUDIO & vai-avatar2 HF Spaces."""
import json
import logging
import os
import re
import sys
import time
from pathlib import Path
logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(message)s")
logger = logging.getLogger("sync")
# ── Config ──────────────────────────────────────────────
DB_PATH = "/tmp/products.json" # will be populated from HF Dataset
USER = "bep40"
SOURCE_SPACE = f"{USER}/V.AISTUDIO"
TARGET_SPACE = f"{USER}/vai-avatar2"
CACHE_DIR = Path(__file__).parent / ".sync_cache"
MAX_PRODUCTS = int(os.getenv("SYNC_MAX", "99999"))
# URLs used by the frontend templates
SITE_URL_VAISTUDIO = "https://bep40.github.io/V.AISTUDIO/"
SITE_URL_VAI_AVATAR = "https://bep40.github.io/vai-avatar2/"
HF_TOKEN = os.getenv("HF_TOKEN")
if not HF_TOKEN:
raise SystemExit("ERROR: HF_TOKEN env var is required")
# ── helpers ─────────────────────────────────────────────
def get_products_from_dataset():
"""Pull latest product records from the HF dataset."""
try:
from datasets import load_dataset # type: ignore
logger.info("Loading dataset …")
ds = load_dataset("bep40/grob-products-updated", split="train")
except Exception:
try:
from datasets import load_dataset # type: ignore
ds = load_dataset("bep40/grob-products-updated", split="train[1:]")
except Exception:
logger.error("Cannot load dataset — falling back to local JSON")
ds = None
products = []
if ds is not None:
for row in ds:
prod = _row_to_product(row)
if prod:
products.append(prod)
if len(products) >= MAX_PRODUCTS:
break
else:
logger.warning("No dataset loaded — checking %s", DB_PATH)
if Path(DB_PATH).exists():
with open(DB_PATH) as f:
products = json.load(f)
logger.info("Collected %d products from source", len(products))
return products
def _row_to_product(row):
"""Convert a dataset row dict → product dict."""
slug = ""
for key in ("slug", "_source_alias", "sku", "mod"):
v = str(row.get(key, "")).strip()
if v:
slug = v.lower().replace(" ", "-")
break
if not slug:
return None
return {
"id": slug,
"name": str(row.get("pn", "")),
"price": str(row.get("price", "")),
"category": str(row.get("_brand", row.get("_category", ""))),
"slug": slug,
# image hint
"img_key": str(row.get("img_key", "")),
"image_url": str(row.get("image_url", "")),
}
# ── template rendering ─────────────────────────────────
PRODUCT_HTML_TEMPLATE = '''
{name}
'''
def make_product_html(product):
"""Render minimal product page HTML."""
name = product["name"] or "Sản phẩm không tên"
price = product["price"] or "Liên hệ"
cat = product["category"] or "Khác"
esc_name = name.replace("&","&").replace("<","<").replace(">",">")
esc_cat = cat.replace("&","&").replace("<","<").replace(">",">")
esc_price = price.replace("&","&").replace("<","<").replace(">",">")
img_url = product.get("image_url") or ""
return PRODUCT_HTML_TEMPLATE.format(
name=name, esc_name=esc_name, price=price, esc_price=esc_price,
slug=product["slug"], cat=cat, esc_cat=esc_cat,
img_url=img_url,
)
# ── diff engine ─────────────────────────────────────────
def load_existing_ids(space_dir: Path) -> set:
"""Return set of known IDs from an existing products.json cache."""
cache_file = space_dir / "products.json"
if cache_file.exists():
with open(cache_file) as f:
data = json.load(f)
return {p["id"] for p in data}
return set()
def compute_new_products(all_products: list, existing_ids: set) -> list:
ids_seen = set()
result = []
for p in all_products:
pid = p["id"]
if pid in existing_ids or pid in ids_seen:
continue
ids_seen.add(pid)
result.append(p)
return result
# ── push to Space ───────────────────────────────────────
def upload_to_space(space_id: str, products: list, base_url: str, site_label: str):
"""Upload new product HTML + image thumbnails to a HF Space."""
from huggingface_hub import HfApi
api = HfApi(token=HF_TOKEN)
local_img_dir = CACHE_DIR / "images" / space_id
local_img_dir.mkdir(parents=True, exist_ok=True)
# Build index manifest for incremental update
existing_ids = load_existing_ids(CACHE_DIR / space_id)
new_prods = compute_new_products(products, existing_ids)
if not new_prods:
logger.info("[%s] No new products to upload", site_label)
return False
logger.info("[%s] Uploading %d new products …", site_label, len(new_prods))
uploaded = 0
for i, prod in enumerate(new_prods):
slug = prod["slug"]
sub_path = f"san-pham/{slug}"
html_path = f"{sub_path}/index.html"
# Upload product page
html_content = make_product_html(prod)
api.upload_file(
path_or_fileobj=html_content.encode(),
path_in_repo=html_path,
repo_id=f"{USER}/{space_id}",
repo_type="space",
commit_message=f"[auto-sync] Add/update product: {prod['name'][:60]}",
)
uploaded += 1
# Upload image thumbnail if available
img_src = prod.get("image_url", "")
if img_src:
img_dst = f"{sub_path}/thumb.jpg"
local_img = local_img_dir / f"{slug}.jpg"
if not local_img.exists() and img_src:
try:
_download_and_save(img_src, local_img)
except Exception as exc:
logger.warning("[%s] Image download failed for %s: %s", site_label, slug, exc)
local_img = None
if local_img and local_img.exists():
try:
api.upload_file(
path_or_fileobj=str(local_img),
path_in_repo=img_dst,
repo_id=f"{USER}/{space_id}",
repo_type="space",
commit_message=f"[auto-sync] Product image: {slug}",
)
except Exception as exc:
logger.warning("[%s] Image upload failed for %s: %s", site_label, slug, exc)
if i % 50 == 0:
logger.info("[%s] Progress %d / %d", site_label, uploaded, len(new_prods))
# Save updated IDs cache
cache_dir = CACHE_DIR / space_id
cache_dir.mkdir(parents=True, exist_ok=True)
cache_file = cache_dir / "products.json"
try:
old = json.loads(cache_file.read_text()) if cache_file.exists() else []
except Exception:
old = []
for p in new_prods:
old.append(p)
cache_file.write_text(json.dumps(old, ensure_ascii=False, indent=2))
logger.info("[%s] ✅ Uploaded %d products", site_label, uploaded)
return True
def _download_and_save(url: str, dest: Path):
import urllib.request
req = urllib.request.Request(url, headers={"User-Agent": "Mozilla/5.0"})
with urllib.request.urlopen(req, timeout=30) as resp, open(dest, "wb") as f:
f.write(resp.read())
def trigger_rebuild(space_id: str, filename: str):
"""Upload a trigger file to force Space rebuild."""
from huggingface_hub import HfApi
api = HfApi(token=HF_TOKEN)
api.upload_file(
path_or_fileobj=b"triggered\n",
path_in_repo=filename,
repo_id=f"{USER}/{space_id}",
repo_type="space",
commit_message=f"[auto-sync] Rebuild trigger: {filename}",
)
logger.info("🔄 Rebuild triggered for %s (%s)", space_id, filename)
# ── Main ────────────────────────────────────────────────
def main():
logger.info("=" * 60)
logger.info("Sync started at %s", time.strftime("%Y-%m-%d %H:%M:%S"))
logger.info("=" * 60)
products = get_products_from_dataset()
if not products:
logger.error("No products — nothing to do.")
sys.exit(1)
# ── Push to V.AISTUDIO ──
logger.info("\n━━━ Pushing to V.AISTUDIO … ━━━")
ok1 = upload_to_space(SOURCE_SPACE, products, SITE_URL_VAISTUDIO, "V.AISTUDIO")
if ok1:
trigger_rebuild(SOURCE_SPACE, ".gitignore") # any tiny file triggers rebuild
time.sleep(3)
# ── Push to vai-avatar2 ──
logger.info("\n━━━ Pushing to vai-avatar2 … ━━━")
ok2 = upload_to_space(TARGET_SPACE, products, SITE_URL_VAI_AVATAR, "vai-avatar2")
if ok2:
trigger_rebuild(TARGET_SPACE, ".gitignore")
total = sum([ok1, ok2])
logger.info("\n✅ Done — successful spaces: %d / 2", total)
if total == 0:
logger.info("No changes needed — both spaces are up-to-date.")
elif total < 2:
logger.warning("Only %d/2 spaces updated successfully.", total)
if __name__ == "__main__":
main()