import gradio as gr import os from src.llm import LLMClient from src.compiler import CppCompiler from src.pipeline import Pipeline DEFAULT_MODEL = "XHToken/Spark-X2.5-4B" POPULAR_INSTRUCT_MODELS = [ "XHToken/Spark-X2.5-4B", "Qwen/Qwen2.5-7B-Instruct", "Qwen/Qwen2.5-14B-Instruct", "Qwen/Qwen2.5-32B-Instruct", "meta-llama/Meta-Llama-3-8B-Instruct", "meta-llama/Meta-Llama-3-70B-Instruct", "google/gemma-2-9b-it", "google/gemma-2-27b-it", "mistralai/Mistral-7B-Instruct-v0.3", "mistralai/Mixtral-8x7B-Instruct-v0.1", "mistralai/Mixtral-8x22B-Instruct-v0.1", "HuggingFaceH4/zephyr-7b-beta", "HuggingFaceH4/mistral-7b-anchor", "microsoft/Phi-3.5-mini-instruct", "microsoft/Phi-3-medium-128k-instruct", "tiiuae/falcon-7b-instruct", "tiiuae/falcon-40b-instruct", "codellama/CodeLlama-7b-Instruct-hf", "codellama/CodeLlama-34b-Instruct-hf", "stabilityai/stablelm-2-1_6b-chat", "stabilityai/stablelm-2-zephyr-16b-chat", ] def run_finder(problem, objective, user_metric, model_id, hf_token, hf_provider, n_algorithms, n_scenarios, max_rounds): if not problem or not problem.strip() or not objective or not objective.strip(): return "Please provide both a problem and an objective.", "", "" token = (hf_token or "").strip() or os.getenv("HF_TOKEN") or os.getenv("HUGGINGFACE_HUB_TOKEN") if (hf_token or "").strip(): os.environ["HF_TOKEN"] = token provider = (hf_provider or "").strip() or os.getenv("HF_PROVIDER") or None model = (model_id or "").strip() or DEFAULT_MODEL llm = LLMClient(model_id=model, token=token, provider=provider) compiler = CppCompiler() progress_log = "" summary_md = "" log_output = "" def progress_callback(message: str): nonlocal progress_log progress_log += message + "\n" return progress_log pipeline = Pipeline( llm_client=llm, compiler=compiler, n_algorithms=int(n_algorithms), n_scenarios=int(n_scenarios), max_validation_rounds=int(max_rounds), progress_callback=progress_callback, ) try: result = pipeline.run(problem.strip(), objective.strip(), (user_metric or "").strip() or "overall performance") except Exception as e: err = str(e) hint = "" if "g++" in err: hint = "\nFix: add 'g++' to packages.txt (Gradio Spaces) or install a C++ toolchain locally." elif "HF_TOKEN" in err or "auth" in err.lower() or "401" in err or "403" in err: hint = "\nFix: add an HF_TOKEN Space secret with access to the selected model, or use Qwen/Qwen2.5-7B-Instruct." elif "enabled Inference Providers" in err or "not supported" in err.lower(): hint = "\nFix: use a model your providers serve (e.g. Qwen/Qwen2.5-7B-Instruct), set an HF Provider override, or enable that model/provider." log_output = f"ERROR: {err}{hint}" return log_output, "", f"Pipeline failed with error: {err}{hint}" finally: compiler.cleanup() if result["status"] != "success": log_output = f"Status: {result['status']}\nStage: {result['stage']}\nMessage: {result['message']}" return log_output, "", result.get("summary", "No summary available.") log_output = f"Status: {result['status']}\nWinner: Algorithm {result['winner_index']}\n{validation_metrics_table(result)}" summary_md = result["summary"] winner_code = result["winner"]["code"] return log_output, winner_code, summary_md def validation_metrics_table(result: dict) -> str: lines = [] lines.append("Baseline:") b = result.get("baseline", {}) rr = b.get("run_result") or {} lines.append(f" Compiled: {b.get('compile_result', {}).get('compiled')}") lines.append(f" Time: {rr.get('execution_time_s', 'N/A')}s") lines.append(f" Memory: {rr.get('memory_kb', 'N/A')} KB") lines.append("") lines.append(f"Winner (Algorithm {result.get('winner_index', 'N/A')}):") w = result.get("winner", {}) wr = w.get("run_result") or {} lines.append(f" Compiled: {w.get('compile_result', {}).get('compiled')}") lines.append(f" Time: {wr.get('execution_time_s', 'N/A')}s") lines.append(f" Memory: {wr.get('memory_kb', 'N/A')} KB") return "\n".join(lines) with gr.Blocks(title="Algorithm Finder") as demo: gr.Markdown("# Algorithm Finder\nFind the best algorithm for any problem using C++ and LLM-powered agents.") with gr.Row(): with gr.Column(scale=2): problem_input = gr.Textbox(label="Problem Description", placeholder="e.g., What is the best way to optimize fluid simulation?") objective_input = gr.Textbox(label="Optimization Objective", placeholder="e.g., Minimize execution time for large grids") user_metric_input = gr.Textbox(label="User-defined Priority Metric", placeholder="e.g., Speed vs Accuracy tradeoff") with gr.Column(scale=2): model_input = gr.Dropdown( label="Hugging Face Model ID (ZeroGPU: any model runs on the Space GPU)", choices=POPULAR_INSTRUCT_MODELS, value=DEFAULT_MODEL, allow_custom_value=True, ) hf_token_input = gr.Textbox(label="HF Token (for gated/private models)", type="password", placeholder="hf_...") hf_provider_input = gr.Textbox(label="HF Provider override (API mode only)", placeholder="together") n_algorithms_input = gr.Slider(label="Number of candidate algorithms", minimum=2, maximum=4, value=2, step=1) n_scenarios_input = gr.Slider(label="Number of real-world scenarios", minimum=1, maximum=3, value=2, step=1) max_rounds_input = gr.Slider(label="Max validation rounds", minimum=1, maximum=3, value=2, step=1) run_btn = gr.Button("Find Best Algorithm", variant="primary") with gr.Row(): with gr.Column(scale=1): log_output = gr.Textbox(label="Pipeline Logs", lines=15, interactive=False) with gr.Column(scale=1): winner_code_output = gr.Code(label="Winning Algorithm (C++)", language="cpp") summary_output = gr.Markdown(label="Algorithm Discovery Summary") run_btn.click( fn=run_finder, inputs=[problem_input, objective_input, user_metric_input, model_input, hf_token_input, hf_provider_input, n_algorithms_input, n_scenarios_input, max_rounds_input], outputs=[log_output, winner_code_output, summary_output], ) if __name__ == "__main__": demo.queue() demo.launch(server_name="0.0.0.0", server_port=7860)