auralynq-rag / web /lib /modelfit.ts
asdfasdfqrqwer's picture
sync: bring the Space up to the current GitHub tree
656439d
Raw
History Blame Contribute Delete
12.3 kB
/**
* Auralynq ModelFit Index — API client
*/
import { consumeSSE } from "@/lib/sse";
const API_BASE = process.env.NEXT_PUBLIC_API_BASE ?? "http://localhost:8000";
export interface GPUInfo {
vendor: string;
name: string;
vram_gb: number;
backend: string;
device_index: number;
vram_free_gb?: number | null;
vram_used_gb?: number | null;
integrated?: boolean;
}
export interface HardwareProfile {
os: { name: string; version: string };
python_version: string;
cpu: {
model: string;
cores_physical: number;
cores_logical: number;
arch?: string;
avx2?: boolean;
avx512?: boolean;
};
ram_gb: number;
gpus: GPUInfo[];
total_vram_gb: number;
total_vram_free_gb?: number | null;
disk_free_gb: number;
best_backend: string;
cuda_available: boolean;
cuda_version: string | null;
metal_available: boolean;
rocm_available: boolean;
avx2?: boolean;
avx512?: boolean;
ollama_available: boolean;
ollama_version: string | null;
hf_available: boolean;
hf_cache_path: string | null;
in_container: boolean;
warnings: string[];
}
/** UI-facing verdict from the resource estimator (see resource_estimator.py). */
export type Verdict =
| "runs_great"
| "runs_ok"
| "runs_offload"
| "runs_cpu"
| "runs_cpu_tight"
| "too_big";
export interface ModelMeta {
model_id: string;
source: string;
display_name: string;
family: string;
parameter_count_b: number | null;
context_length: number | null;
license: string;
gated: boolean;
tasks: string[];
available_quantizations: string[];
embedding: boolean;
reranker: boolean;
supports_adapters: boolean;
vision: boolean;
tool_calling: boolean;
multilingual: boolean;
ollama_tag: string | null;
hf_repo: string | null;
local_path: string | null;
notes: string[];
}
export interface ResourceEstimate {
model_id: string;
quantization: string;
context_tokens: number;
estimated_vram_gb: number;
estimated_ram_gb: number;
estimated_disk_gb: number;
fit_level: "comfortable" | "tight" | "not_recommended" | "impossible";
fits: boolean;
recommended_context: number;
peak_vram_at_max_ctx_gb: number;
verdict?: Verdict;
fits_in_vram?: boolean;
requires_cpu_offload?: boolean;
headroom_gb?: number;
vram_free_gb?: number | null;
warnings: string[];
is_estimate: true;
}
export interface ModelFitScore {
model_id: string;
overall_score: number;
hardware_fit: number;
speed_fit: number;
rag_fit: number;
task_fit: number;
deployment_fit: number;
label: string;
best_quantization: string;
reason: string;
resource_estimate: ResourceEstimate | null;
benchmark: BenchmarkResult | null;
estimate_used: boolean;
warnings: string[];
}
export interface BenchmarkPlan {
model_id: string;
quantization: string;
task: string;
num_examples: number;
estimated_duration_min: number;
sample_prompts: string[];
requires_ollama: boolean;
requires_model_download: false;
warnings: string[];
note: string;
}
export interface BenchmarkResult {
run_id: string;
model_id: string;
quantization: string;
task: string;
status: "pending" | "running" | "completed" | "failed" | "cancelled";
hardware: Partial<HardwareProfile>;
avg_tok_per_sec: number | null;
p50_latency_ms: number | null;
p95_latency_ms: number | null;
time_to_first_token_ms: number | null;
peak_memory_gb: number | null;
num_examples: number;
completed_examples: number;
rag_metrics: {
citation_coverage: number | null;
groundedness: number | null;
abstention_accuracy: number | null;
};
error: string | null;
started_at: string;
completed_at: string | null;
warnings: string[];
is_measured: boolean;
}
async function apiFetch<T>(path: string, init?: RequestInit): Promise<T> {
const res = await fetch(`${API_BASE}${path}`, {
headers: { "Content-Type": "application/json" },
...init,
});
if (!res.ok) {
const body = await res.text().catch(() => "");
throw new Error(extractApiError(body) ?? `API ${path}${res.status}: ${body}`);
}
return res.json();
}
/**
* Pull the human sentence out of the API's error envelope.
*
* Without this the UI printed the raw `{"error":{"code":"http_error",...}}` JSON
* at the user, which is how a missing binary surfaced as an unreadable blob.
*/
export function extractApiError(body: string): string | null {
try {
const parsed = JSON.parse(body);
const msg = parsed?.error?.message ?? parsed?.detail ?? parsed?.message;
return typeof msg === "string" && msg.trim() ? msg : null;
} catch {
return null;
}
}
// Hardware
export async function fetchHardware(): Promise<HardwareProfile> {
return apiFetch("/api/modelfit/hardware");
}
// Models
export async function fetchModels(params?: {
source?: string;
family?: string;
task?: string;
embedding_only?: boolean;
open_license?: boolean;
q?: string;
limit?: number;
}): Promise<{ models: ModelMeta[]; total: number }> {
const qs = new URLSearchParams();
if (params?.source) qs.set("source", params.source);
if (params?.family) qs.set("family", params.family);
if (params?.task) qs.set("task", params.task);
if (params?.embedding_only) qs.set("embedding_only", "true");
if (params?.open_license) qs.set("open_license", "true");
if (params?.q) qs.set("q", params.q);
if (params?.limit) qs.set("limit", String(params.limit));
return apiFetch(`/api/modelfit/models?${qs}`);
}
export async function fetchInstalledModels(): Promise<{ models: ModelMeta[]; total: number; warnings: string[] }> {
return apiFetch("/api/modelfit/models/installed");
}
// Estimate
export async function fetchEstimate(
model_id: string,
params_b: number,
quantization: string,
context_tokens = 4096,
): Promise<ResourceEstimate> {
return apiFetch("/api/modelfit/estimate", {
method: "POST",
body: JSON.stringify({ model_id, params_b, quantization, context_tokens }),
});
}
// Score
export async function fetchScore(
model_id: string,
quantization?: string,
requested_tasks?: string[],
): Promise<ModelFitScore> {
return apiFetch("/api/modelfit/score", {
method: "POST",
body: JSON.stringify({ model_id, quantization, requested_tasks: requested_tasks ?? [] }),
});
}
// Recommendations
export async function fetchRecommendations(task?: string, limit = 5): Promise<{
recommendations: ModelFitScore[];
hardware_summary: { ram_gb: number; total_vram_gb: number; best_backend: string };
task: string | null;
}> {
const qs = new URLSearchParams();
if (task) qs.set("task", task);
qs.set("limit", String(limit));
return apiFetch(`/api/modelfit/recommendations?${qs}`);
}
// Benchmark
export async function fetchBenchmarkPreview(
model_id: string,
quantization: string,
task: string,
num_examples: number,
): Promise<BenchmarkPlan> {
return apiFetch("/api/modelfit/benchmark/preview", {
method: "POST",
body: JSON.stringify({ model_id, quantization, task, num_examples, confirmed: false }),
});
}
export async function runBenchmark(
model_id: string,
quantization: string,
task: string,
num_examples: number,
): Promise<BenchmarkResult> {
return apiFetch("/api/modelfit/benchmark/run", {
method: "POST",
body: JSON.stringify({ model_id, quantization, task, num_examples, confirmed: true }),
});
}
export async function fetchBenchmarkRuns(): Promise<{ runs: BenchmarkResult[]; total: number }> {
return apiFetch("/api/modelfit/benchmark/runs");
}
export async function fetchBenchmarkRun(run_id: string): Promise<BenchmarkResult> {
return apiFetch(`/api/modelfit/benchmark/${run_id}`);
}
// ── Discover & Setup ──────────────────────────────────────────────────────────
export interface DiscoverHardware {
os: string;
cpu: string;
cpu_cores_physical?: number;
cpu_cores_logical?: number;
arch?: string;
avx2?: boolean;
avx512?: boolean;
ram_gb: number;
total_vram_gb: number;
total_vram_free_gb?: number | null;
gpus: GPUInfo[];
best_backend: string;
ollama_available: boolean;
in_container?: boolean;
}
export interface DiscoverEntry {
model_id: string;
overall_score: number;
hardware_fit: number;
speed_fit: number;
rag_fit: number;
task_fit: number;
deployment_fit: number;
label: string;
best_quantization: string;
reason: string;
resource_estimate: ResourceEstimate | null;
benchmark: BenchmarkResult | null;
estimate_used: boolean;
warnings: string[];
model_meta: ModelMeta;
already_installed: boolean;
pull_command: string | null;
source: string;
}
export interface DiscoverResult {
hardware: DiscoverHardware;
task: string | null;
total_candidates: number;
recommendations: DiscoverEntry[];
note: string;
}
export interface PullResult {
status: string;
model_id: string;
message: string;
local_path?: string;
/** Ollama pulls only — the background job to stream progress from. */
job_id?: string;
stream_url?: string;
}
export type PullPhase = "queued" | "manifest" | "downloading" | "verifying" | "success" | "error";
export interface PullProgress {
job_id: string;
model_id: string;
tag: string;
phase: PullPhase;
status_text: string;
completed_bytes: number;
total_bytes: number;
percent: number;
speed_bps: number;
eta_s: number | null;
layers_done: number;
layers_total: number;
message: string;
error: string;
error_status: number | null;
elapsed_s: number;
}
export async function discoverModels(
task: string | null,
includeHf: boolean,
refresh: boolean,
limit = 30,
): Promise<DiscoverResult> {
return apiFetch("/api/modelfit/discover", {
method: "POST",
body: JSON.stringify({ task, include_hf: includeHf, refresh, limit }),
});
}
export async function pullModel(modelId: string): Promise<PullResult> {
return apiFetch("/api/modelfit/pull", {
method: "POST",
body: JSON.stringify({ model_id: modelId, confirmed: true }),
});
}
export async function fetchPullJob(jobId: string): Promise<PullProgress> {
return apiFetch(`/api/modelfit/pull/${jobId}`);
}
/**
* Stream a pull job's progress until it terminates.
*
* The download belongs to the job, not to this request — closing the stream (or
* the modal) does not cancel it, and re-subscribing resumes from current state.
*/
export async function streamPullProgress(
jobId: string,
onProgress: (p: PullProgress) => void,
signal?: AbortSignal,
): Promise<PullProgress> {
const res = await fetch(`${API_BASE}/api/modelfit/pull/${jobId}/stream`, {
headers: { Accept: "text/event-stream" },
signal,
});
if (!res.ok || !res.body) {
const body = await res.text().catch(() => "");
throw new Error(extractApiError(body) ?? `Could not follow pull ${jobId}.`);
}
const reader = res.body.getReader();
const decoder = new TextDecoder();
let buffer = "";
let last: PullProgress | null = null;
for (;;) {
const { done, value } = await reader.read();
if (done) break;
buffer += decoder.decode(value, { stream: true });
const { events, rest } = consumeSSE<PullProgress>(buffer);
buffer = rest;
for (const ev of events) {
last = ev;
onProgress(ev);
if (ev.phase === "success" || ev.phase === "error") {
void reader.cancel().catch(() => {});
return ev;
}
}
}
// Stream closed without a terminal frame — ask the job directly.
return last?.phase === "success" || last?.phase === "error" ? last : fetchPullJob(jobId);
}
export function formatBytes(bytes: number): string {
if (!bytes) return "0 B";
const units = ["B", "KB", "MB", "GB", "TB"];
const i = Math.min(units.length - 1, Math.floor(Math.log(bytes) / Math.log(1024)));
return `${(bytes / 1024 ** i).toFixed(i >= 2 ? 1 : 0)} ${units[i]}`;
}
export function formatEta(seconds: number | null): string {
if (seconds == null || !Number.isFinite(seconds) || seconds <= 0) return "—";
if (seconds < 60) return `${Math.round(seconds)}s`;
const m = Math.floor(seconds / 60);
const s = Math.round(seconds % 60);
return m < 60 ? `${m}m ${s}s` : `${Math.floor(m / 60)}h ${m % 60}m`;
}