File size: 3,389 Bytes
8c3e275
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
"""Check for model updates against Hugging Face.

Usage: python scripts/check_model_updates.py [--download]
"""

from __future__ import annotations

import argparse
import json
import urllib.request
from pathlib import Path

MODELS_DIR = Path(__file__).resolve().parent.parent / "models"

MODEL_SOURCES = {
    "qwen2.5-1.5b-instruct-q4_k_m.gguf": {
        "url": "https://huggingface.co/Qwen/Qwen2.5-1.5B-Instruct-GGUF/resolve/main/qwen2.5-1.5b-instruct-q4_k_m.gguf",
        "version_url": "https://huggingface.co/api/models/Qwen/Qwen2.5-1.5B-Instruct-GGUF",
    },
    "tr_ocr_base_handwritten.onnx": {
        "url": "https://huggingface.co/Xenova/trocr-base-handwritten/resolve/main/onnx/model.onnx",
        "version_url": "https://huggingface.co/api/models/Xenova/trocr-base-handwritten",
    },
    "all-MiniLM-L6-v2.onnx": {
        "url": "https://huggingface.co/sentence-transformers/all-MiniLM-L6-v2/resolve/main/onnx/model.onnx",
        "version_url": "https://huggingface.co/api/models/sentence-transformers/all-MiniLM-L6-v2",
    },
    "ggml-tiny.en-q5_0.bin": {
        "url": "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-tiny.en-q5_0.bin",
        "version_url": "https://huggingface.co/api/models/ggerganov/whisper.cpp",
    },
}


def check_model(name: str, info: dict) -> dict:
    model_path = MODELS_DIR / name
    result = {
        "name": name,
        "exists": model_path.exists(),
        "size_mb": round(model_path.stat().st_size / 1024 / 1024, 1) if model_path.exists() else 0,
    }

    try:
        req = urllib.request.Request(
            info["version_url"],
            headers={"User-Agent": "Mozilla/5.0"},
        )
        with urllib.request.urlopen(req, timeout=15) as resp:
            data = json.loads(resp.read().decode())
            last_modified = data.get("lastModified", data.get("createdAt", "unknown"))
            downloads = data.get("downloads", 0)
            result["remote_last_modified"] = last_modified
            result["remote_downloads"] = downloads
            result["update_available"] = False
    except Exception as e:
        result["check_error"] = str(e)

    return result


def main():
    parser = argparse.ArgumentParser(description="Check PageParse model updates")
    parser.add_argument("--download", action="store_true", help="Download missing/updated models")
    args = parser.parse_args()

    print("Checking model versions...")
    results = []
    for name, info in MODEL_SOURCES.items():
        r = check_model(name, info)
        results.append(r)
        status = "OK" if r["exists"] else "MISSING"
        print(f"  {name}: {status} ({r.get('size_mb', 0)} MB)")

    outdated = [r for r in results if r.get("update_available")]
    missing = [r for r in results if not r["exists"]]

    if outdated:
        print(f"\nUpdates available for: {', '.join(r['name'] for r in outdated)}")
    if missing:
        print(f"\nMissing models: {', '.join(r['name'] for r in missing)}")

    if args.download:
        from fetch_models import download
        print("\nDownloading missing/updated models...")
        for r in missing + outdated:
            info = MODEL_SOURCES[r["name"]]
            download(info["url"], MODELS_DIR / r["name"])

    if not outdated and not missing:
        print("\nAll models are up to date and present.")


if __name__ == "__main__":
    main()