| import os |
| import sys |
| import site |
| import subprocess |
| import json |
|
|
| |
| |
| |
| _CUDA_FOUND = False |
| for sp in site.getsitepackages(): |
| for subdir in ['nvidia/cuda_runtime/lib', 'nvidia/cublas/lib', 'nvidia/cudnn/lib']: |
| path = os.path.join(sp, subdir) |
| if os.path.isdir(path) and os.path.exists(os.path.join(path, 'libcudart.so.12')): |
| os.environ['LD_LIBRARY_PATH'] = path + ':' + os.environ.get('LD_LIBRARY_PATH', '') |
| _CUDA_FOUND = True |
| print(f"[cuda] Found runtime at: {path}") |
| break |
| if _CUDA_FOUND: |
| break |
|
|
| if not _CUDA_FOUND: |
| print("[cuda] WARNING: No bundled CUDA runtime found. Will try torch's lib path.") |
| try: |
| import torch |
| torch_lib = os.path.join(os.path.dirname(torch.__file__), 'lib') |
| os.environ['LD_LIBRARY_PATH'] = torch_lib + ':' + os.environ.get('LD_LIBRARY_PATH', '') |
| print(f"[cuda] Fallback to torch lib: {torch_lib}") |
| except Exception as e: |
| print(f"[cuda] ERROR: {e}. GPU acceleration may not work!") |
|
|
| |
| from huggingface_hub import hf_hub_download |
| import gradio as gr |
|
|
| |
| MODEL_REPO = "BugTraceAI/BugTraceAI-CORE-Fast" |
| MODEL_FILE = "bugtraceai-core-fast.gguf" |
|
|
| print(f"[model] Downloading {MODEL_REPO}/{MODEL_FILE}...") |
| model_path = hf_hub_download(repo_id=MODEL_REPO, filename=MODEL_FILE) |
| model_size_gb = os.path.getsize(model_path) / 1e9 |
| print(f"[model] Path: {model_path}") |
| print(f"[model] Size: {model_size_gb:.2f} GB") |
|
|
| print("[llama] Loading model...") |
|
|
| try: |
| from llama_cpp import Llama |
| llm = Llama( |
| model_path=model_path, |
| n_ctx=8192, |
| n_batch=512, |
| n_gpu_layers=-1 if _CUDA_FOUND else 0, |
| flash_attn=True, |
| use_mmap=True, |
| use_mlock=False, |
| chat_format="chatml", |
| verbose=False, |
| ) |
| n_gpu = llm.n_gpu_layers |
| print(f"[llama] Loaded OK. GPU layers: {n_gpu}") |
| except Exception as e: |
| print(f"[llama] ERROR loading with llama-cpp-python: {e}") |
| print(f"[llama] Falling back to llama.cpp server binary...") |
| |
| |
| import urllib.request |
| LLAMA_BIN_URL = "https://github.com/ggml-org/llama.cpp/releases/download/b9946/llama-b9946-bin-ubuntu-x64.tar.gz" |
| bin_tarball = "/tmp/llama-server.tar.gz" |
| print(f"[llama-bin] Downloading {LLAMA_BIN_URL}...") |
| urllib.request.urlretrieve(LLAMA_BIN_URL, bin_tarball) |
| |
| import tarfile |
| with tarfile.open(bin_tarball) as tf: |
| tf.extractall("/tmp/llama-bin") |
| |
| llama_server = "/tmp/llama-bin/bin/llama-server" |
| os.chmod(llama_server, 0o755) |
| |
| |
| server_args = [ |
| llama_server, |
| "-m", model_path, |
| "-c", "8192", |
| "-b", "512", |
| "--port", "8081", |
| "--host", "0.0.0.0", |
| "-ngl", "99", |
| "-fa", |
| "--metrics", |
| "--no-webui", |
| ] |
| print(f"[llama-bin] Starting server: {' '.join(server_args)}") |
| proc = subprocess.Popen(server_args, stdout=subprocess.PIPE, stderr=subprocess.STDOUT) |
| |
| |
| import time |
| for i in range(30): |
| try: |
| r = urllib.request.urlopen("http://localhost:8081/health") |
| if r.status == 200: |
| print("[llama-bin] Server ready!") |
| break |
| except: |
| pass |
| time.sleep(1) |
| else: |
| print("[llama-bin] WARNING: Server may not be ready after 30s") |
| |
| |
| class LlamaServerClient: |
| def __init__(self, base_url="http://localhost:8081"): |
| self.base_url = base_url |
| |
| def create_chat_completion(self, messages, max_tokens=512, temperature=0.7, top_p=0.9): |
| body = json.dumps({ |
| "messages": messages, |
| "max_tokens": max_tokens, |
| "temperature": temperature, |
| "top_p": top_p, |
| "stream": False, |
| }).encode() |
| req = urllib.request.Request( |
| f"{self.base_url}/v1/chat/completions", |
| data=body, |
| headers={"Content-Type": "application/json"} |
| ) |
| resp = urllib.request.urlopen(req, timeout=120) |
| return json.loads(resp.read()) |
| |
| llm = LlamaServerClient() |
| n_gpu = "all (via llama-server subprocess)" |
| print(f"[llama-bin] Client ready. GPU layers: {n_gpu}") |
|
|
| |
| def chat(prompt, max_tokens=512, temperature=0.7): |
| response = llm.create_chat_completion( |
| messages=[{"role": "user", "content": prompt}], |
| max_tokens=max_tokens, |
| temperature=temperature, |
| top_p=0.9, |
| ) |
| return response["choices"][0]["message"]["content"] |
|
|
| |
| demo = gr.Interface( |
| fn=chat, |
| inputs=[ |
| gr.Textbox(label="Prompt", lines=5, placeholder="Bug bounty analysis prompt..."), |
| gr.Slider(64, 1024, value=512, label="Max tokens"), |
| gr.Slider(0.0, 1.5, value=0.7, label="Temperature"), |
| ], |
| outputs=gr.Textbox(label="Response", lines=12), |
| title="BugTraceAI Fast 7B β Bug Bounty Copilot", |
| description="Security-focused 7B model for vulnerability triage, exploit reasoning, and finding analysis. Runs on ZeroGPU A10G (24 GB).", |
| ) |
|
|
| demo.launch(server_name="0.0.0.0", server_port=7860) |