Spaces:
Sleeping
Sleeping
File size: 3,040 Bytes
0e2c396 0bb352f 0e2c396 cd2a36b 0e2c396 cd2a36b 0e2c396 cd2a36b 0e2c396 7d42d5c | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 | # Use an official PyTorch image as a base
FROM pytorch/pytorch:2.4.0-cuda12.1-cudnn9-runtime
# Set the working directory inside the container
WORKDIR /app
# Install system dependencies
RUN apt-get update && apt-get install -y \
build-essential \
git \
&& rm -rf /var/lib/apt/lists/*
# Create a non-root user
RUN useradd -m -u 1000 appuser
# Create cache directories and set permissions
RUN mkdir -p /app/models/sentence_transformer && \
mkdir -p /app/models/qwen && \
mkdir -p /.cache/huggingface && \
mkdir -p /.cache/torch && \
mkdir -p /.cache/sentence_transformers
# Set environment variables for cache and model locations
ENV HF_HOME="/.cache/huggingface"
ENV TORCH_HOME="/.cache/torch"
ENV SENTENCE_TRANSFORMERS_HOME="/app/models/sentence_transformer"
# Install Python dependencies
RUN pip install --no-cache-dir \
pandas \
torch \
sentence-transformers \
transformers \
numpy \
faiss-cpu \
fastapi \
uvicorn[standard] \
pydantic \
python-multipart \
huggingface_hub \
accelerate>=0.26.0
# Create a script to download models
COPY <<EOF /app/download_models.py
import os
from sentence_transformers import SentenceTransformer
from transformers import AutoTokenizer, AutoModelForCausalLM
import torch
print("Downloading sentence transformer model...")
model = SentenceTransformer("sentence-transformers/all-mpnet-base-v2")
model.save("/app/models/sentence_transformer")
print("Sentence transformer model saved!")
print("Downloading Qwen model...")
model_name = "Qwen/Qwen2.5-0.5B-Instruct" # Fixed: was Qwen3.5 (doesn't exist)
tokenizer = AutoTokenizer.from_pretrained(model_name, trust_remote_code=True)
model = AutoModelForCausalLM.from_pretrained(
model_name,
trust_remote_code=True,
torch_dtype=torch.float16,
device_map=None
)
tokenizer.save_pretrained("/app/models/qwen")
model.save_pretrained("/app/models/qwen")
print("Qwen model saved!")
EOF
# Download models during build
RUN python /app/download_models.py
# Only set TRANSFORMERS_OFFLINE after downloading models
ENV TRANSFORMERS_OFFLINE=1
# Copy the main.py and modify it to use local paths
COPY main.py /app/main.py
RUN sed -i 's|"sentence-transformers/all-mpnet-base-v2"|"/app/models/sentence_transformer"|g' main.py && \
sed -i 's|"Qwen/Qwen2.5-0.5B-Instruct"|"/app/models/qwen"|g' main.py
# Copy the data file
COPY updated_plant_data_chunks_and_embeddings.csv /app/updated_plant_data_chunks_and_embeddings.csv
# Set proper permissions
RUN chown -R appuser:appuser /app && \
chown -R appuser:appuser /.cache && \
chmod -R 755 /app/models
# Switch to non-root user
USER appuser
# Set the environment variable to point to the app directory
ENV PYTHONPATH=/app
# Expose the port
EXPOSE 5000
# Reduce model loading time by setting specific environment variables
ENV OMP_NUM_THREADS=1
ENV MKL_NUM_THREADS=1
ENV TORCH_NUM_THREADS=1
# Command to run the app
CMD ["uvicorn", "main:app", "--host", "0.0.0.0", "--port", "7860", "--timeout-keep-alive", "300"] |