File size: 2,703 Bytes
c80bb82
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
from __future__ import annotations

from pathlib import Path

import gradio as gr
import numpy as np
from game import is_terminal, legal_actions, play, winner
from mcts import NeuralMCTS
from model import MicroZeroNet
from safetensors.torch import load_file

ARTIFACT = Path(__file__).resolve().parent / "artifacts" / "microzero"
MODEL = MicroZeroNet()
MODEL.load_state_dict(load_file(ARTIFACT / "model.safetensors"))
MODEL.eval()
SYMBOLS = {0: ".", 1: "X", -1: "O"}


def neural_action(
    model: MicroZeroNet,
    board: np.ndarray,
    player: int,
    simulations: int,
    seed: int,
) -> int:
    search = NeuralMCTS(model, simulations=simulations)
    policy = search.policy(
        board,
        player,
        temperature=0.0,
        add_noise=False,
        rng=np.random.default_rng(seed),
    )
    return int(np.argmax(policy))


def render(board: list[int], message: str) -> str:
    cells = [SYMBOLS[int(value)] for value in board]
    rows = [" | ".join(cells[index : index + 3]) for index in range(0, 9, 3)]
    return "```\n" + "\n---------\n".join(rows) + f"\n```\n{message}"


def new_game() -> tuple[list[int], str]:
    board = [0] * 9
    return board, render(board, "You are X. Choose a cell from 0 to 8.")


def move(board: list[int], action: int) -> tuple[list[int], str]:
    array = np.asarray(board, dtype=np.int8)
    action = int(action)
    if action not in legal_actions(array):
        return board, render(board, "That cell is occupied.")
    array = play(array, action, 1)
    if is_terminal(array):
        message = "You win." if winner(array) == 1 else "Draw."
        return array.tolist(), render(array.tolist(), message)
    response = neural_action(MODEL, array, -1, simulations=96, seed=7000 + action)
    array = play(array, response, -1)
    result = winner(array)
    if result == -1:
        message = f"MicroZero played {response} and won."
    elif is_terminal(array):
        message = f"MicroZero played {response}. Draw."
    else:
        message = f"MicroZero played {response}. Your turn."
    return array.tolist(), render(array.tolist(), message)


with gr.Blocks(title="MicroZero") as demo:
    gr.Markdown("# MicroZero\nPlay against the neural MCTS agent.")
    state = gr.State([0] * 9)
    board = gr.Markdown(render([0] * 9, "You are X. Choose a cell from 0 to 8."))
    action = gr.Slider(0, 8, value=4, step=1, label="Your move")
    with gr.Row():
        play_button = gr.Button("Play", variant="primary")
        reset_button = gr.Button("New game")
    play_button.click(move, inputs=[state, action], outputs=[state, board])
    reset_button.click(new_game, outputs=[state, board])


if __name__ == "__main__":
    demo.launch()