agentic-model-selector / src /agents /deployment_agent.py
honourjesus's picture
Agentic Model Selector
af2c3f6
Raw
History Blame Contribute Delete
14.7 kB
"""
Deployment Agent - Generates deployment code and configuration
"""
import os
import yaml
import torch
from datetime import datetime
from typing import Dict, Optional
from jinja2 import Template
from src.models.schemas import (
TaskType, DeploymentType, BenchmarkResult, UserRequirements
)
class DeploymentAgent:
"""
Generates production-ready deployment code for selected models.
Supports FastAPI, Gradio, and Docker deployments with
appropriate configuration files.
"""
def __init__(self, output_dir: str = "deployments"):
self.output_dir = output_dir
self._load_templates()
def _load_templates(self):
"""Load deployment templates"""
self.templates = {}
# FastAPI template
fastapi_template = '''"""
FastAPI deployment for {{ model_id }}
Generated by HuggingFace Model Selector
"""
from fastapi import FastAPI, HTTPException
from pydantic import BaseModel
from transformers import pipeline
import torch
import uvicorn
import time
from typing import Dict, Any
app = FastAPI(
title="{{ model_id }} API",
description="Model deployment for {{ task_type }}",
version="1.0.0"
)
# Load model
print("Loading model {{ model_id }}...")
model = pipeline("{{ task_type }}", model="{{ model_id }}")
print("Model loaded successfully!")
class InferenceRequest(BaseModel):
text: str
parameters: Dict[str, Any] = {}
class InferenceResponse(BaseModel):
result: Any
inference_time: float
@app.post("/predict", response_model=InferenceResponse)
async def predict(request: InferenceRequest):
"""Run inference on input text"""
try:
start_time = time.time()
result = model(request.text, **request.parameters)
inference_time = time.time() - start_time
return InferenceResponse(
result=result,
inference_time=inference_time
)
except Exception as e:
raise HTTPException(status_code=500, detail=str(e))
@app.get("/health")
async def health():
"""Health check endpoint"""
return {"status": "healthy", "model": "{{ model_id }}"}
@app.get("/info")
async def info():
"""Model information"""
return {
"model_id": "{{ model_id }}",
"task": "{{ task_type }}",
"device": "cuda" if torch.cuda.is_available() else "cpu"
}
if __name__ == "__main__":
uvicorn.run(app, host="0.0.0.0", port=8000)
'''
self.templates[DeploymentType.FASTAPI] = Template(fastapi_template)
# Gradio template
gradio_template = '''"""
Gradio deployment for {{ model_id }}
Generated by HuggingFace Model Selector
"""
import gradio as gr
from transformers import pipeline
import torch
# Load model
print("Loading model {{ model_id }}...")
model = pipeline("{{ task_type }}", model="{{ model_id }}")
print("Model loaded successfully!")
def predict(text):
"""Run inference on input text"""
try:
result = model(text)
return result
except Exception as e:
return f"Error: {str(e)}"
# Create interface
interface = gr.Interface(
fn=predict,
inputs=gr.Textbox(label="Input Text", lines=3),
outputs=gr.Textbox(label="Result", lines=5),
title="{{ model_id }}",
description="Model deployment for {{ task_type }}",
examples=[["This is a sample input"]]
)
if __name__ == "__main__":
interface.launch(server_name="0.0.0.0", server_port=7860)
'''
self.templates[DeploymentType.GRADIO] = Template(gradio_template)
# Dockerfile template
docker_template = '''FROM python:3.9-slim
WORKDIR /app
# Install system dependencies
RUN apt-get update && apt-get install -y \\
gcc \\
g++ \\
&& rm -rf /var/lib/apt/lists/*
# Copy requirements
COPY requirements.txt .
RUN pip install --no-cache-dir -r requirements.txt
# Copy application code
COPY . .
# Expose port
EXPOSE 8000
# Run the application
CMD ["python", "app.py"]
'''
self.templates["dockerfile"] = Template(docker_template)
def generate_deployment(self,
model_id: str,
task_type: TaskType,
deployment_type: DeploymentType,
benchmark_results: Optional[BenchmarkResult] = None,
requirements: Optional[UserRequirements] = None) -> Dict[str, str]:
"""
Generate deployment code and configuration.
Returns:
Dictionary mapping filenames to file contents
"""
deployment_files = {}
# Create timestamp for unique folder
timestamp = datetime.now().strftime("%Y%m%d_%H%M%S")
safe_model_id = model_id.replace("/", "_")
deploy_folder = f"{self.output_dir}/{safe_model_id}_{timestamp}"
# Generate main application file
if deployment_type == DeploymentType.FASTAPI:
deployment_files["app.py"] = self.templates[DeploymentType.FASTAPI].render(
model_id=model_id,
task_type=task_type.value
)
elif deployment_type == DeploymentType.GRADIO:
deployment_files["app.py"] = self.templates[DeploymentType.GRADIO].render(
model_id=model_id,
task_type=task_type.value
)
# Generate requirements.txt
deployment_files["requirements.txt"] = self._generate_requirements(
task_type, deployment_type
)
# Generate Dockerfile if requested
if deployment_type == DeploymentType.DOCKER:
deployment_files["Dockerfile"] = self.templates["dockerfile"].render()
# Generate docker-compose.yml
deployment_files["docker-compose.yml"] = self._generate_docker_compose(
model_id, requirements
)
# Generate configuration file
deployment_files["config.yaml"] = self._generate_config(
model_id, task_type, benchmark_results, requirements
)
# Generate README
deployment_files["README.md"] = self._generate_readme(
model_id, task_type, deployment_type, benchmark_results
)
# Add folder info
deployment_files["_folder"] = deploy_folder
return deployment_files
def _generate_requirements(self, task_type: TaskType, deployment_type: DeploymentType) -> str:
"""Generate requirements.txt"""
requirements = "# Generated requirements\n"
requirements += "transformers>=4.35.0\n"
requirements += "torch>=2.0.0\n"
requirements += "huggingface-hub>=0.19.0\n"
if deployment_type == DeploymentType.FASTAPI:
requirements += "fastapi>=0.104.0\n"
requirements += "uvicorn>=0.24.0\n"
requirements += "pydantic>=2.0.0\n"
elif deployment_type == DeploymentType.GRADIO:
requirements += "gradio>=4.0.0\n"
# Task-specific requirements - using correct TaskType names
if task_type in [TaskType.IMAGE_CLASSIFICATION, TaskType.OBJECT_DETECTION]:
requirements += "pillow>=10.0.0\n"
elif task_type == TaskType.SPEECH_TO_TEXT: # Changed from SPEECH_RECOGNITION
requirements += "librosa>=0.10.0\n"
elif task_type == TaskType.TEXT_TO_SPEECH:
requirements += "librosa>=0.10.0\n"
elif task_type == TaskType.OCR:
requirements += "pillow>=10.0.0\n"
requirements += "pytesseract>=0.3.10\n"
return requirements
def _generate_docker_compose(self, model_id: str, requirements: Optional[UserRequirements]) -> str:
"""Generate docker-compose.yml"""
compose = "version: '3.8'\n\n"
compose += "services:\n"
compose += " model-service:\n"
compose += " build: .\n"
compose += " ports:\n"
compose += " - \"8000:8000\"\n"
compose += " environment:\n"
compose += f" - MODEL_ID={model_id}\n"
compose += " restart: unless-stopped\n"
# Add GPU support if needed
if requirements and any("gpu" in c.value for c in requirements.hardware_constraints):
compose += " runtime: nvidia\n"
compose += " environment:\n"
compose += " - NVIDIA_VISIBLE_DEVICES=all\n"
return compose
def _generate_config(self, model_id: str, task_type: TaskType,
benchmark_results: Optional[BenchmarkResult],
requirements: Optional[UserRequirements]) -> str:
"""Generate YAML configuration"""
config = {
"model": {
"id": model_id,
"task": task_type.value,
},
"deployment": {
"batch_size": 1,
"max_length": 512,
"device": "cuda" if torch.cuda.is_available() else "cpu"
}
}
if benchmark_results and not benchmark_results.error:
config["performance"] = {
"latency_ms": benchmark_results.latency_ms,
"memory_mb": benchmark_results.memory_usage_mb,
"throughput_sps": benchmark_results.throughput
}
if requirements:
config["requirements"] = {}
# Add hardware constraints
if requirements.hardware_constraints:
config["requirements"]["hardware_constraints"] = [c.value for c in requirements.hardware_constraints]
# Add max model size
if requirements.max_model_size_gb:
config["requirements"]["max_model_size_gb"] = requirements.max_model_size_gb
# Add task-specific requirements based on task type
if requirements.task_type == TaskType.TRANSLATION and requirements.translation_reqs:
req = requirements.translation_reqs
config["requirements"]["source_language"] = req.source_language.value
config["requirements"]["target_language"] = req.target_language.value
if req.domain:
config["requirements"]["domain"] = req.domain
elif requirements.task_type == TaskType.TEXT_TO_SPEECH and requirements.tts_reqs:
req = requirements.tts_reqs
config["requirements"]["language"] = req.language.value
config["requirements"]["voice_type"] = req.voice_type.value
elif requirements.task_type == TaskType.SPEECH_TO_TEXT and requirements.stt_reqs:
req = requirements.stt_reqs
config["requirements"]["language"] = req.language.value
if req.domain:
config["requirements"]["domain"] = req.domain
elif requirements.llm_reqs:
req = requirements.llm_reqs
config["requirements"]["model_size"] = req.model_size.value
config["requirements"]["context_length"] = req.context_length
elif requirements.ocr_reqs:
req = requirements.ocr_reqs
config["requirements"]["languages"] = [lang.value for lang in req.languages]
config["requirements"]["handwritten"] = req.handwritten
return yaml.dump(config, default_flow_style=False)
def _generate_readme(self, model_id: str, task_type: TaskType,
deployment_type: DeploymentType,
benchmark_results: Optional[BenchmarkResult]) -> str:
"""Generate README.md"""
readme = f"# {model_id} Deployment\n\n"
readme += "This deployment was automatically generated by the HuggingFace Model Selector.\n\n"
readme += "## Model Information\n\n"
readme += f"- **Model ID**: {model_id}\n"
readme += f"- **Task**: {task_type.value}\n"
readme += f"- **Deployment Type**: {deployment_type.value}\n\n"
if benchmark_results and not benchmark_results.error:
readme += "## Performance Metrics\n\n"
readme += f"- **Average Latency**: {benchmark_results.latency_ms:.2f} ms\n"
readme += f"- **Memory Usage**: {benchmark_results.memory_usage_mb:.2f} MB\n"
readme += f"- **Throughput**: {benchmark_results.throughput_samples_per_second:.2f} samples/second\n\n"
readme += "## Quick Start\n\n"
readme += "### 1. Install dependencies\n"
readme += "```bash\n"
readme += "pip install -r requirements.txt\n"
readme += "```\n\n"
readme += "### 2. Run the application\n"
readme += "```bash\n"
readme += "python app.py\n"
readme += "```\n\n"
readme += "### 3. Test the API\n\n"
if deployment_type == DeploymentType.FASTAPI:
readme += "```bash\n"
readme += "# Health check\n"
readme += "curl http://localhost:8000/health\n\n"
readme += "# Run inference\n"
readme += 'curl -X POST http://localhost:8000/predict \\\n'
readme += ' -H "Content-Type: application/json" \\\n'
readme += ' -d \'{"text": "Your input text here"}\'\n'
readme += "```\n"
elif deployment_type == DeploymentType.GRADIO:
readme += "Open http://localhost:7860 in your browser to use the Gradio interface.\n"
if deployment_type == DeploymentType.DOCKER:
readme += "\n## Docker Deployment\n\n"
readme += "```bash\n"
readme += "# Build the image\n"
readme += "docker build -t model-service .\n\n"
readme += "# Run the container\n"
readme += "docker run -p 8000:8000 model-service\n\n"
readme += "# Or use docker-compose\n"
readme += "docker-compose up\n"
readme += "```\n"
return readme
def save_deployment_files(self, deployment_files: Dict[str, str]) -> str:
"""Save generated files to disk"""
folder = deployment_files.pop("_folder")
os.makedirs(folder, exist_ok=True)
for filename, content in deployment_files.items():
filepath = os.path.join(folder, filename)
with open(filepath, "w", encoding="utf-8") as f:
f.write(content)
print(f" Created {filepath}")
return folder