bugtrace / app.py
M4rc0s
Deploy BugTraceAI Fast 7B with CUDA, fallback llama-server, ZeroGPU A10G
b997567
Raw
History Blame Contribute Delete
5.71 kB
import os
import sys
import site
import subprocess
import json
# ── CUDA Runtime Detection & LD_LIBRARY_PATH Patch ──
# ZeroGPU provides NVIDIA drivers but not CUDA toolkit.
# torch==2.11.0 bundles its own CUDA runtime at site-packages/nvidia/*/lib/
_CUDA_FOUND = False
for sp in site.getsitepackages():
for subdir in ['nvidia/cuda_runtime/lib', 'nvidia/cublas/lib', 'nvidia/cudnn/lib']:
path = os.path.join(sp, subdir)
if os.path.isdir(path) and os.path.exists(os.path.join(path, 'libcudart.so.12')):
os.environ['LD_LIBRARY_PATH'] = path + ':' + os.environ.get('LD_LIBRARY_PATH', '')
_CUDA_FOUND = True
print(f"[cuda] Found runtime at: {path}")
break
if _CUDA_FOUND:
break
if not _CUDA_FOUND:
print("[cuda] WARNING: No bundled CUDA runtime found. Will try torch's lib path.")
try:
import torch
torch_lib = os.path.join(os.path.dirname(torch.__file__), 'lib')
os.environ['LD_LIBRARY_PATH'] = torch_lib + ':' + os.environ.get('LD_LIBRARY_PATH', '')
print(f"[cuda] Fallback to torch lib: {torch_lib}")
except Exception as e:
print(f"[cuda] ERROR: {e}. GPU acceleration may not work!")
# ── Imports (must be AFTER LD_LIBRARY_PATH patch) ──
from huggingface_hub import hf_hub_download
import gradio as gr
# Try llama-cpp-python with CUDA, fallback to CPU
MODEL_REPO = "BugTraceAI/BugTraceAI-CORE-Fast"
MODEL_FILE = "bugtraceai-core-fast.gguf"
print(f"[model] Downloading {MODEL_REPO}/{MODEL_FILE}...")
model_path = hf_hub_download(repo_id=MODEL_REPO, filename=MODEL_FILE)
model_size_gb = os.path.getsize(model_path) / 1e9
print(f"[model] Path: {model_path}")
print(f"[model] Size: {model_size_gb:.2f} GB")
print("[llama] Loading model...")
try:
from llama_cpp import Llama
llm = Llama(
model_path=model_path,
n_ctx=8192,
n_batch=512,
n_gpu_layers=-1 if _CUDA_FOUND else 0,
flash_attn=True,
use_mmap=True,
use_mlock=False,
chat_format="chatml",
verbose=False,
)
n_gpu = llm.n_gpu_layers
print(f"[llama] Loaded OK. GPU layers: {n_gpu}")
except Exception as e:
print(f"[llama] ERROR loading with llama-cpp-python: {e}")
print(f"[llama] Falling back to llama.cpp server binary...")
# Download llama-server binary from GitHub releases
import urllib.request
LLAMA_BIN_URL = "https://github.com/ggml-org/llama.cpp/releases/download/b9946/llama-b9946-bin-ubuntu-x64.tar.gz"
bin_tarball = "/tmp/llama-server.tar.gz"
print(f"[llama-bin] Downloading {LLAMA_BIN_URL}...")
urllib.request.urlretrieve(LLAMA_BIN_URL, bin_tarball)
import tarfile
with tarfile.open(bin_tarball) as tf:
tf.extractall("/tmp/llama-bin")
llama_server = "/tmp/llama-bin/bin/llama-server"
os.chmod(llama_server, 0o755)
# Start llama-server as subprocess
server_args = [
llama_server,
"-m", model_path,
"-c", "8192",
"-b", "512",
"--port", "8081",
"--host", "0.0.0.0",
"-ngl", "99", # all layers to GPU
"-fa", # flash attention
"--metrics",
"--no-webui", # API only
]
print(f"[llama-bin] Starting server: {' '.join(server_args)}")
proc = subprocess.Popen(server_args, stdout=subprocess.PIPE, stderr=subprocess.STDOUT)
# Wait for server to be ready
import time
for i in range(30):
try:
r = urllib.request.urlopen("http://localhost:8081/health")
if r.status == 200:
print("[llama-bin] Server ready!")
break
except:
pass
time.sleep(1)
else:
print("[llama-bin] WARNING: Server may not be ready after 30s")
# Wrapper class that calls llama-server API
class LlamaServerClient:
def __init__(self, base_url="http://localhost:8081"):
self.base_url = base_url
def create_chat_completion(self, messages, max_tokens=512, temperature=0.7, top_p=0.9):
body = json.dumps({
"messages": messages,
"max_tokens": max_tokens,
"temperature": temperature,
"top_p": top_p,
"stream": False,
}).encode()
req = urllib.request.Request(
f"{self.base_url}/v1/chat/completions",
data=body,
headers={"Content-Type": "application/json"}
)
resp = urllib.request.urlopen(req, timeout=120)
return json.loads(resp.read())
llm = LlamaServerClient()
n_gpu = "all (via llama-server subprocess)"
print(f"[llama-bin] Client ready. GPU layers: {n_gpu}")
# ── Chat Function ──
def chat(prompt, max_tokens=512, temperature=0.7):
response = llm.create_chat_completion(
messages=[{"role": "user", "content": prompt}],
max_tokens=max_tokens,
temperature=temperature,
top_p=0.9,
)
return response["choices"][0]["message"]["content"]
# ── Gradio UI ──
demo = gr.Interface(
fn=chat,
inputs=[
gr.Textbox(label="Prompt", lines=5, placeholder="Bug bounty analysis prompt..."),
gr.Slider(64, 1024, value=512, label="Max tokens"),
gr.Slider(0.0, 1.5, value=0.7, label="Temperature"),
],
outputs=gr.Textbox(label="Response", lines=12),
title="BugTraceAI Fast 7B β€” Bug Bounty Copilot",
description="Security-focused 7B model for vulnerability triage, exploit reasoning, and finding analysis. Runs on ZeroGPU A10G (24 GB).",
)
demo.launch(server_name="0.0.0.0", server_port=7860)