agent-api / push_ollama_model_to_hf.py
github-actions[bot]
Sync GitHub snapshot to Hugging Face
9a1014e
Raw
History Blame Contribute Delete
4.64 kB
"""Descarga un modelo Ollama, exporta su GGUF, lo sube a HF y lo elimina localmente."""
from __future__ import annotations
import argparse
import os
import re
import shutil
import subprocess
from pathlib import Path
from huggingface_hub import HfApi
from huggingface_hub.errors import RepositoryNotFoundError
DEFAULT_MODEL = "qwen3:4b-instruct"
DEFAULT_REPO = "AlbertiTechnology/qwen3-4b-instruct-gguf"
MAX_MODEL_BYTES = 3 * 1024**3
def find_ollama() -> str:
executable = shutil.which("ollama")
if executable:
return executable
candidates = [
Path(os.getenv("LOCALAPPDATA", "")) / "Programs" / "Ollama" / "ollama.exe",
Path(os.getenv("LOCALAPPDATA", "")) / "Ollama" / "ollama.exe",
Path(os.getenv("ProgramFiles", "")) / "Ollama" / "ollama.exe",
]
for candidate in candidates:
if candidate.is_file():
return str(candidate)
raise RuntimeError(
"Ollama no esta instalado. Instale Ollama, cierre y abra la terminal, "
"y vuelva a ejecutar este script."
)
def run_ollama(*args: str, capture: bool = False) -> str:
executable = find_ollama()
result = subprocess.run(
[executable, *args],
check=True,
text=True,
capture_output=capture,
encoding="utf-8",
errors="replace",
)
return result.stdout if capture else ""
def model_blob_from_modelfile(modelfile: str) -> Path:
match = re.search(r"^FROM\s+(.+?)\s*$", modelfile, flags=re.MULTILINE)
if not match:
raise RuntimeError("No se encontro la capa GGUF en 'ollama show --modelfile'.")
raw_path = match.group(1).strip().strip('"')
blob_path = Path(raw_path)
if not blob_path.is_file():
raise FileNotFoundError(f"Ollama informo un blob inexistente: {blob_path}")
return blob_path
def exported_modelfile(original: str) -> str:
return re.sub(
r"^FROM\s+.+?$",
"FROM ./model.gguf",
original,
count=1,
flags=re.MULTILINE,
)
def upload_model(model: str, repo_id: str, token: str) -> None:
print(f"Descargando {model} con Ollama...", flush=True)
run_ollama("pull", model)
modelfile = run_ollama("show", "--modelfile", model, capture=True)
blob_path = model_blob_from_modelfile(modelfile)
blob_size = blob_path.stat().st_size
if blob_size > MAX_MODEL_BYTES:
raise RuntimeError(
f"El GGUF ocupa {blob_size / 1024**3:.2f} GiB y supera el limite de 3 GiB."
)
readme = "\n".join(
[
"---",
"library_name: llama.cpp",
"tags:",
"- gguf",
"- ollama",
"- qwen3",
"---",
"",
f"# {model}",
"",
"Modelo GGUF exportado desde Ollama para inferencia local.",
]
)
api = HfApi(token=token)
try:
api.repo_info(repo_id=repo_id, repo_type="model")
print(f"Repositorio existente encontrado: {repo_id}", flush=True)
except RepositoryNotFoundError:
api.create_repo(repo_id=repo_id, repo_type="model")
print(
f"Subiendo directamente el blob de {blob_size / 1024**3:.2f} GiB; "
"no se creara una copia local...",
flush=True,
)
api.upload_file(
path_or_fileobj=blob_path,
path_in_repo="model.gguf",
repo_id=repo_id,
repo_type="model",
commit_message=f"Upload {model} GGUF from Ollama",
)
api.upload_file(
path_or_fileobj=exported_modelfile(modelfile).encode("utf-8"),
path_in_repo="Modelfile",
repo_id=repo_id,
repo_type="model",
)
api.upload_file(
path_or_fileobj=readme.encode("utf-8"),
path_in_repo="README.md",
repo_id=repo_id,
repo_type="model",
)
print(f"Subida completada: https://huggingface.co/{repo_id}", flush=True)
print(f"Eliminando {model} del almacenamiento local de Ollama...", flush=True)
run_ollama("rm", model)
print("Modelo local y archivos temporales eliminados.", flush=True)
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser()
parser.add_argument("--model", default=DEFAULT_MODEL)
parser.add_argument("--repo", default=DEFAULT_REPO)
return parser.parse_args()
def main() -> None:
args = parse_args()
token = os.getenv("HF_TOKEN") or os.getenv("HUGGINGFACEHUB_API_TOKEN")
if not token:
raise RuntimeError("Defina HF_TOKEN con permiso de escritura antes de ejecutar.")
upload_model(args.model, args.repo, token)
if __name__ == "__main__":
main()