Files
tic_tac_toe/tic_tac_toe.py
T
jensandClaude Sonnet 5 2f56608c54 Make --num control the number of play games, not just training
--num was declared as "Number of games to play" but the play() call
used a hardcoded 10. Also give --num an int type so it works when
passed on the CLI (range() would otherwise crash on the raw string),
and make --verbose a store_true flag.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01PQS7ZHqxQfy5DhKcz2gHau
2026-07-15 21:14:26 +02:00

109 lines
3.1 KiB
Python

import numpy as np
from state import create_empty_state, to_state_string
from machine_player import MachinePlayer
from human_player import HumanPlayer
from player_provider import PlayerProvider
import json
import argparse
float_formatter = "{:.3f}".format
np.set_printoptions(formatter={'float_kind': float_formatter})
REPORT_INTERVAL = 1000
report_counter = REPORT_INTERVAL
def play(player_provider: PlayerProvider, k_max=10000, with_debug=False):
player_provider.set_debug(with_debug)
for k in range(0, k_max):
state = create_empty_state()
players = player_provider.choose(state)
move = 1
run = True
other_player = players[-1]
while run:
do_stop = False
this_reward = 0
other_reward = 0
for this_player in players:
if with_debug:
print(f"---------------------------------------------------")
print(f"- Game {k:06d}, Move {move} -----------------------------")
print(f"---------------------------------------------------")
last_state = state
state, has_moved = this_player.move(state)
if with_debug:
print(to_state_string(last_state, state))
if not has_moved:
this_reward = 0
other_reward = 0
do_stop = True
if with_debug:
print(f"{this_player.mark}: No more moves")
if this_player.has_won(state):
this_reward = 1
other_reward = 0
do_stop = True
if with_debug:
print(f"{this_player.mark}: Has won the game")
if do_stop:
this_player.end_game(state, this_reward)
other_player.end_game(state, other_reward)
run = False
break
other_player = this_player
move += 1
if __name__ == '__main__':
parser = argparse.ArgumentParser()
parser.add_argument("--verbose", default=False, action='store_true', help="Be verbose")
parser.add_argument("--train", default=False, help="Train before play")
parser.add_argument("--num", default=5, type=int, help="Number of games to play")
parser.add_argument("--human", default=False, help="X-player is human")
parser.add_argument("--load", default=True, help="Load experience from previous trainings")
args = parser.parse_args()
if args.load:
try:
with open("values_x.json", "r") as fp:
x_values = json.load(fp)
with open("values_o.json", "r") as fp:
o_values = json.load(fp)
except Exception:
x_values = {}
o_values = {}
else:
x_values = {}
o_values = {}
p_mx = MachinePlayer(mark='X', params=MachinePlayer.Params(p_explore=0.1, alpha=0.1), values=x_values)
p_mo = MachinePlayer(mark='O', params=MachinePlayer.Params(p_explore=0.1, alpha=0.1), values=o_values)
N = args.num
if args.train:
for n in range(0, N):
play(PlayerProvider(p_mx, p_mo), 1000, False)
# Values after training
# Convert and write JSON object to file
with open("values_x.json", "w") as fp:
json.dump(p_mx.values, fp, indent=0)
with open("values_o.json", "w") as fp:
json.dump(p_mo.values, fp, indent=0)
print(f"Iteration: {n*N:8d}")
if args.human:
p_hx = HumanPlayer(mark='X', name="Jens")
pp = PlayerProvider(p_hx, p_mo)
else:
pp = PlayerProvider(p_mx, p_mo)
play(pp, N, args.verbose)