Spaces:
Running
Running
File size: 2,703 Bytes
c7928dd | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 | from __future__ import annotations
from pathlib import Path
import gradio as gr
import numpy as np
from game import is_terminal, legal_actions, play, winner
from mcts import NeuralMCTS
from model import MicroZeroNet
from safetensors.torch import load_file
ARTIFACT = Path(__file__).resolve().parent / "artifacts" / "microzero"
MODEL = MicroZeroNet()
MODEL.load_state_dict(load_file(ARTIFACT / "model.safetensors"))
MODEL.eval()
SYMBOLS = {0: ".", 1: "X", -1: "O"}
def neural_action(
model: MicroZeroNet,
board: np.ndarray,
player: int,
simulations: int,
seed: int,
) -> int:
search = NeuralMCTS(model, simulations=simulations)
policy = search.policy(
board,
player,
temperature=0.0,
add_noise=False,
rng=np.random.default_rng(seed),
)
return int(np.argmax(policy))
def render(board: list[int], message: str) -> str:
cells = [SYMBOLS[int(value)] for value in board]
rows = [" | ".join(cells[index : index + 3]) for index in range(0, 9, 3)]
return "```\n" + "\n---------\n".join(rows) + f"\n```\n{message}"
def new_game() -> tuple[list[int], str]:
board = [0] * 9
return board, render(board, "You are X. Choose a cell from 0 to 8.")
def move(board: list[int], action: int) -> tuple[list[int], str]:
array = np.asarray(board, dtype=np.int8)
action = int(action)
if action not in legal_actions(array):
return board, render(board, "That cell is occupied.")
array = play(array, action, 1)
if is_terminal(array):
message = "You win." if winner(array) == 1 else "Draw."
return array.tolist(), render(array.tolist(), message)
response = neural_action(MODEL, array, -1, simulations=96, seed=7000 + action)
array = play(array, response, -1)
result = winner(array)
if result == -1:
message = f"MicroZero played {response} and won."
elif is_terminal(array):
message = f"MicroZero played {response}. Draw."
else:
message = f"MicroZero played {response}. Your turn."
return array.tolist(), render(array.tolist(), message)
with gr.Blocks(title="MicroZero") as demo:
gr.Markdown("# MicroZero\nPlay against the neural MCTS agent.")
state = gr.State([0] * 9)
board = gr.Markdown(render([0] * 9, "You are X. Choose a cell from 0 to 8."))
action = gr.Slider(0, 8, value=4, step=1, label="Your move")
with gr.Row():
play_button = gr.Button("Play", variant="primary")
reset_button = gr.Button("New game")
play_button.click(move, inputs=[state, action], outputs=[state, board])
reset_button.click(new_game, outputs=[state, board])
if __name__ == "__main__":
demo.launch()
|