File size: 3,040 Bytes
0e2c396
0bb352f
0e2c396
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
cd2a36b
0e2c396
cd2a36b
 
0e2c396
 
 
 
 
 
 
 
 
cd2a36b
0e2c396
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
7d42d5c
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
# Use an official PyTorch image as a base
FROM pytorch/pytorch:2.4.0-cuda12.1-cudnn9-runtime

# Set the working directory inside the container
WORKDIR /app

# Install system dependencies
RUN apt-get update && apt-get install -y \
    build-essential \
    git \
    && rm -rf /var/lib/apt/lists/*

# Create a non-root user
RUN useradd -m -u 1000 appuser

# Create cache directories and set permissions
RUN mkdir -p /app/models/sentence_transformer && \
    mkdir -p /app/models/qwen && \
    mkdir -p /.cache/huggingface && \
    mkdir -p /.cache/torch && \
    mkdir -p /.cache/sentence_transformers

# Set environment variables for cache and model locations
ENV HF_HOME="/.cache/huggingface"
ENV TORCH_HOME="/.cache/torch"
ENV SENTENCE_TRANSFORMERS_HOME="/app/models/sentence_transformer"

# Install Python dependencies
RUN pip install --no-cache-dir \
    pandas \
    torch \
    sentence-transformers \
    transformers \
    numpy \
    faiss-cpu \
    fastapi \
    uvicorn[standard] \
    pydantic \
    python-multipart \
    huggingface_hub \
    accelerate>=0.26.0

# Create a script to download models
COPY <<EOF /app/download_models.py
import os
from sentence_transformers import SentenceTransformer
from transformers import AutoTokenizer, AutoModelForCausalLM
import torch

print("Downloading sentence transformer model...")
model = SentenceTransformer("sentence-transformers/all-mpnet-base-v2")
model.save("/app/models/sentence_transformer")
print("Sentence transformer model saved!")

print("Downloading Qwen model...")
model_name = "Qwen/Qwen2.5-0.5B-Instruct"  # Fixed: was Qwen3.5 (doesn't exist)
tokenizer = AutoTokenizer.from_pretrained(model_name, trust_remote_code=True)
model = AutoModelForCausalLM.from_pretrained(
    model_name,
    trust_remote_code=True,
    torch_dtype=torch.float16,
    device_map=None
)
tokenizer.save_pretrained("/app/models/qwen")
model.save_pretrained("/app/models/qwen")
print("Qwen model saved!")
EOF

# Download models during build
RUN python /app/download_models.py

# Only set TRANSFORMERS_OFFLINE after downloading models
ENV TRANSFORMERS_OFFLINE=1

# Copy the main.py and modify it to use local paths
COPY main.py /app/main.py
RUN sed -i 's|"sentence-transformers/all-mpnet-base-v2"|"/app/models/sentence_transformer"|g' main.py && \
    sed -i 's|"Qwen/Qwen2.5-0.5B-Instruct"|"/app/models/qwen"|g' main.py

# Copy the data file
COPY updated_plant_data_chunks_and_embeddings.csv /app/updated_plant_data_chunks_and_embeddings.csv

# Set proper permissions
RUN chown -R appuser:appuser /app && \
    chown -R appuser:appuser /.cache && \
    chmod -R 755 /app/models

# Switch to non-root user
USER appuser

# Set the environment variable to point to the app directory
ENV PYTHONPATH=/app

# Expose the port
EXPOSE 5000

# Reduce model loading time by setting specific environment variables
ENV OMP_NUM_THREADS=1
ENV MKL_NUM_THREADS=1
ENV TORCH_NUM_THREADS=1

# Command to run the app
CMD ["uvicorn", "main:app", "--host", "0.0.0.0", "--port", "7860", "--timeout-keep-alive", "300"]