Michael Arana
Add ZeroGPU backend so any HF model runs on Space GPU (default XHToken/Spark-X2.5-4B)
042d836
Raw History Blame Contribute Delete
6.64 kB
import gradio as gr
import os
from src.llm import LLMClient
from src.compiler import CppCompiler
from src.pipeline import Pipeline
DEFAULT_MODEL = "XHToken/Spark-X2.5-4B"
POPULAR_INSTRUCT_MODELS = [
"XHToken/Spark-X2.5-4B",
"Qwen/Qwen2.5-7B-Instruct",
"Qwen/Qwen2.5-14B-Instruct",
"Qwen/Qwen2.5-32B-Instruct",
"meta-llama/Meta-Llama-3-8B-Instruct",
"meta-llama/Meta-Llama-3-70B-Instruct",
"google/gemma-2-9b-it",
"google/gemma-2-27b-it",
"mistralai/Mistral-7B-Instruct-v0.3",
"mistralai/Mixtral-8x7B-Instruct-v0.1",
"mistralai/Mixtral-8x22B-Instruct-v0.1",
"HuggingFaceH4/zephyr-7b-beta",
"HuggingFaceH4/mistral-7b-anchor",
"microsoft/Phi-3.5-mini-instruct",
"microsoft/Phi-3-medium-128k-instruct",
"tiiuae/falcon-7b-instruct",
"tiiuae/falcon-40b-instruct",
"codellama/CodeLlama-7b-Instruct-hf",
"codellama/CodeLlama-34b-Instruct-hf",
"stabilityai/stablelm-2-1_6b-chat",
"stabilityai/stablelm-2-zephyr-16b-chat",
]
def run_finder(problem, objective, user_metric, model_id, hf_token, hf_provider, n_algorithms, n_scenarios, max_rounds):
if not problem or not problem.strip() or not objective or not objective.strip():
return "Please provide both a problem and an objective.", "", ""
token = (hf_token or "").strip() or os.getenv("HF_TOKEN") or os.getenv("HUGGINGFACE_HUB_TOKEN")
if (hf_token or "").strip():
os.environ["HF_TOKEN"] = token
provider = (hf_provider or "").strip() or os.getenv("HF_PROVIDER") or None
model = (model_id or "").strip() or DEFAULT_MODEL
llm = LLMClient(model_id=model, token=token, provider=provider)
compiler = CppCompiler()
progress_log = ""
summary_md = ""
log_output = ""
def progress_callback(message: str):
nonlocal progress_log
progress_log += message + "\n"
return progress_log
pipeline = Pipeline(
llm_client=llm,
compiler=compiler,
n_algorithms=int(n_algorithms),
n_scenarios=int(n_scenarios),
max_validation_rounds=int(max_rounds),
progress_callback=progress_callback,
)
try:
result = pipeline.run(problem.strip(), objective.strip(), (user_metric or "").strip() or "overall performance")
except Exception as e:
err = str(e)
hint = ""
if "g++" in err:
hint = "\nFix: add 'g++' to packages.txt (Gradio Spaces) or install a C++ toolchain locally."
elif "HF_TOKEN" in err or "auth" in err.lower() or "401" in err or "403" in err:
hint = "\nFix: add an HF_TOKEN Space secret with access to the selected model, or use Qwen/Qwen2.5-7B-Instruct."
elif "enabled Inference Providers" in err or "not supported" in err.lower():
hint = "\nFix: use a model your providers serve (e.g. Qwen/Qwen2.5-7B-Instruct), set an HF Provider override, or enable that model/provider."
log_output = f"ERROR: {err}{hint}"
return log_output, "", f"Pipeline failed with error: {err}{hint}"
finally:
compiler.cleanup()
if result["status"] != "success":
log_output = f"Status: {result['status']}\nStage: {result['stage']}\nMessage: {result['message']}"
return log_output, "", result.get("summary", "No summary available.")
log_output = f"Status: {result['status']}\nWinner: Algorithm {result['winner_index']}\n{validation_metrics_table(result)}"
summary_md = result["summary"]
winner_code = result["winner"]["code"]
return log_output, winner_code, summary_md
def validation_metrics_table(result: dict) -> str:
lines = []
lines.append("Baseline:")
b = result.get("baseline", {})
rr = b.get("run_result") or {}
lines.append(f" Compiled: {b.get('compile_result', {}).get('compiled')}")
lines.append(f" Time: {rr.get('execution_time_s', 'N/A')}s")
lines.append(f" Memory: {rr.get('memory_kb', 'N/A')} KB")
lines.append("")
lines.append(f"Winner (Algorithm {result.get('winner_index', 'N/A')}):")
w = result.get("winner", {})
wr = w.get("run_result") or {}
lines.append(f" Compiled: {w.get('compile_result', {}).get('compiled')}")
lines.append(f" Time: {wr.get('execution_time_s', 'N/A')}s")
lines.append(f" Memory: {wr.get('memory_kb', 'N/A')} KB")
return "\n".join(lines)
with gr.Blocks(title="Algorithm Finder") as demo:
gr.Markdown("# Algorithm Finder\nFind the best algorithm for any problem using C++ and LLM-powered agents.")
with gr.Row():
with gr.Column(scale=2):
problem_input = gr.Textbox(label="Problem Description", placeholder="e.g., What is the best way to optimize fluid simulation?")
objective_input = gr.Textbox(label="Optimization Objective", placeholder="e.g., Minimize execution time for large grids")
user_metric_input = gr.Textbox(label="User-defined Priority Metric", placeholder="e.g., Speed vs Accuracy tradeoff")
with gr.Column(scale=2):
model_input = gr.Dropdown(
label="Hugging Face Model ID (ZeroGPU: any model runs on the Space GPU)",
choices=POPULAR_INSTRUCT_MODELS,
value=DEFAULT_MODEL,
allow_custom_value=True,
)
hf_token_input = gr.Textbox(label="HF Token (for gated/private models)", type="password", placeholder="hf_...")
hf_provider_input = gr.Textbox(label="HF Provider override (API mode only)", placeholder="together")
n_algorithms_input = gr.Slider(label="Number of candidate algorithms", minimum=2, maximum=4, value=2, step=1)
n_scenarios_input = gr.Slider(label="Number of real-world scenarios", minimum=1, maximum=3, value=2, step=1)
max_rounds_input = gr.Slider(label="Max validation rounds", minimum=1, maximum=3, value=2, step=1)
run_btn = gr.Button("Find Best Algorithm", variant="primary")
with gr.Row():
with gr.Column(scale=1):
log_output = gr.Textbox(label="Pipeline Logs", lines=15, interactive=False)
with gr.Column(scale=1):
winner_code_output = gr.Code(label="Winning Algorithm (C++)", language="cpp")
summary_output = gr.Markdown(label="Algorithm Discovery Summary")
run_btn.click(
fn=run_finder,
inputs=[problem_input, objective_input, user_metric_input, model_input, hf_token_input, hf_provider_input, n_algorithms_input, n_scenarios_input, max_rounds_input],
outputs=[log_output, winner_code_output, summary_output],
)
if __name__ == "__main__":
demo.queue()
demo.launch(server_name="0.0.0.0", server_port=7860)