moving towards json engine

This commit is contained in:
Sander Land committed 2020-01-23 16:20:08 +01:00
1 parent def9ff372a
commit dcc80c6ad3
9 files changed
+402 -436

No files matched your search

@@ -1,151 +1,72 @@
# Example config for C++ (non-python) gtp bot
# RUNNING ON AN ONLINE SERVER OR IN A REAL TOURNAMENT OR MATCH:
# If you plan to do so, you may want to read through the "Rules" section
# below carefully for proper handling of komi and handicap games and end-of-game cleanup
# and various other details.
# NOTES ABOUT PERFORMANCE AND MEMORY USAGE:
# You will likely want to tune one or more the following:
#
# numSearchThreads:
# The number of CPU threads to use. If your GPU is powerful, it can actually be much higher than
# the number of cores on your processor because you will need many threads to feed large enough
# batches to make good use of the GPU.
#
# nnMaxBatchSize:
# The maximum GPU batch size. Should often be at least as large as numSearchThreads.
# Larger won't do anything, but also won't hurt except use a little bit more GPU memory.
# Smaller can be fine if you have more than one GPU, since the GPUs will be sharing the work
# of servicing the CPU threads.
#
# cudaUseFP16 and cudaUseNHWC:
# These have a good chance of improving peformance at larger threads/batch sizes if
# you are using the CUDA implementation with an NVIDIA GPU with FP16 tensor cores.
#
# nnCacheSizePowerOfTwo:
# This controls the NN Cache size, which is the primary RAM/memory use.
# Each neural net entry takes very approximately 1.5KB, except when using whole-board
# ownership/territory visualizations, each entry will take very approximately 3KB.
# The number of entries is (2 ** nnCacheSizePowerOfTwo), for example 2 ** 18 = 262144.
# Increase this if you don't mind the memory use and want better performance
# for searches with tens of thousands of visits or more (due to birthday paradox
# it can start mattering well before cache actually fills entirely up).
# Decrease this if you want to limit memory usage.
#
# OTHER NOTES:
# If you have more than one GPU, take a look at "OpenCL GPU settings" or "CUDA GPU settings" below.
#
# If using OpenCL, you will want to verify that KataGo is picking up the correct device!
# (e.g. some systems may have both an Intel CPU OpenCL and GPU OpenCL, if KataGo appears to pick
# the wrong one, you correct this by specifying "openclGpuToUse" below).
#
# You may also want to adjust "maxVisits", "ponderingEnabled", "resignThreshold", and possibly
# other parameters depending on your intended usage.
# SEE NOTES ABOUT PERFORMANCE AND MEMORY USAGE IN gtp_example.cfg
# Logs------------------------------------------------------------------------------------
# Where to output log?
logFile = gtp.log
# Logging options
logAllGTPCommunication = true
logSearchInfo = true
logToStderr = false
# KataGo will display some info to stderr on GTP startup
# Uncomment this to suppress that and remain silent
# startupPrintMessageToStderr = false
# Chat some stuff to stderr, for use in things like malkovich chat to OGS.
# ogsChatToStderr = true
# Configure the maximum length of analysis printed out by lz-analyze and other places.
# Controls the number of moves after the first move in a variation.
# analysisPVLen = 9
# analysisPVLen = 15
# Report winrates for chat and analysis as (BLACK|WHITE|SIDETOMOVE).
# Default is SIDETOMOVE, which is what tools that use LZ probably also expect
# reportAnalysisWinratesAs = SIDETOMOVE
# Report winrates for analysis as (BLACK|WHITE|SIDETOMOVE).
reportAnalysisWinratesAs = SIDETOMOVE
# Rules------------------------------------------------------------------------------------
# Bot behavior---------------------------------------------------------------------------------------
# koRule = SIMPLE #Simple ko rules (triple ko = no result)
koRule = POSITIONAL #Positional superko
# koRule = SITUATIONAL #Situational superko
# koRule = SPIGHT #Spight superko - https://senseis.xmp.net/?SpightRules
scoringRule = AREA #Area scoring
# scoringRule = TERRITORY #Territory scoring (uses a sort of special computer-friendly territory ruleset)
multiStoneSuicideLegal = false #Is multiple-stone suicide legal? (Single-stone suicide is always illegal).
# Make the bot capture stones that are part of pass-alive territory
# This is necessary to get correct play under tromp-taylor rules since the bot otherwise assumes (and is trained under)
# a ruleset where those stones need not be captured. It obviously should NOT be enabled if playing under territory scoring.
cleanupBeforePass = false
# Uncomment this to make it so that if the game seems to be a handicap game, assume that white gets +1 point per
# black handicap stone. Some Go servers like OGS will silently give white such points without including it in the komi.
# whiteBonusPerHandicapStone = 1
# Resignation occurs if for at least resignConsecTurns in a row,
# the winLossUtility (which is on a [-1,1] scale) is below resignThreshold.
allowResignation = false
resignThreshold = -0.98
resignConsecTurns = 3
# Handicap -------------
# Assume that if black makes many moves in a row right at the start of the game, then the game is a handicap game.
# This is necessary on some servers and for some GUIs and also when initializing from many SGF files, which may
# set up a handicap games using repeated GTP "play" commands for black rather than GTP "place_free_handicap" commands.
# However, it may also lead to incorrect undersanding of komi if whiteBonusPerHandicapStone = 1 and a server does NOT
# have such a practice.
# Defaults to true. Uncomment and set to false to disable this behavior.
# assumeMultipleStartingBlackMovesAreHandicap = false
# Defaults to true! Uncomment and set to false to disable this behavior.
# assumeMultipleStartingBlackMovesAreHandicap = true
# Passing and cleanup -------------
# Make the bot never assume that its pass will end the game, even if passing would end and "win" under Tromp-Taylor rules.
# Usually this is a good idea when using it for analysis or playing on servers where scoring may be implemented non-tromp-taylorly.
# Defaults to true! Uncomment and set to false to disable this.
conservativePass = true
# When using territory scoring, self-play games continue beyond two passes with special cleanup
# rules that may be confusing for human players. This option prevents the special cleanup phases from being
# reachable when using the bot for GTP play.
# Defaults to true! Uncomment and set to false if you want KataGo to be able to enter special cleanup.
# For example, if you are testing it against itself, or against another bot that has precisely implemented the rules
# documented at https://lightvector.github.io/KataGo/rules.html
# preventCleanupPhase = true
# Search limits-----------------------------------------------------------------------------------
# If provided, limit maximum number of root visits per search to this much. (With tree reuse, visits do count earlier search)
maxVisits = 1000
# If provided, limit maximum number of new playouts per search to this much. (With tree reuse, playouts do not count earlier search)
# maxPlayouts = 1000
# If provided, cap search time at this many seconds (search will still try to follow GTP time controls)
# By default, if NOT specified in an individual request, limit maximum number of root visits per search to this much
maxVisits = 500
# If provided, cap search time at this many seconds
# maxTime = 60
# Ponder on the opponent's turn?
ponderingEnabled = false
# Same limits but for ponder searches if pondering is enabled
# maxVisitsPondering = 1000
# maxPlayoutsPondering = 1000
# maxTimePondering = 60
# Number of seconds to buffer for lag for GTP time controls
lagBuffer = 1.0
# Number of threads to use in search
numSearchThreads = 1
# Play a little faster if the opponent is passing, for friendliness
searchFactorAfterOnePass = 0.50
searchFactorAfterTwoPass = 0.25
# Play a little faster if super-winning, for friendliess
searchFactorWhenWinning = 0.40
searchFactorWhenWinningThreshold = 0.95
# Number of threads to use in each search in parallel for any SINGLE position.
# NOTE: Analysis engine can specify number of POSITIONS to be able to search in parallel via command line argument
# so this number does not necessarily need to be larger than 1, although you can still set it larger if you prefer
# to analyze fewer positions in parallel but spend more threads on each position.
# Generally, having more threads on a single position will worsen the quality of search slightly, holding fixed the
# number of visits, and thread contention will reduce efficiency, so cross-position parallelization is preferable
# to numSearchThreads, but numSearchThreads is preferable if you want to reduce latency, and have individual
# searches complete faster by doing fewer of them at a time.
numSearchThreads = 2
# GPU Settings-------------------------------------------------------------------------------
# Maximum number of positions to send to GPU at once. Note that you will also need to increase numSearchThreads
# to make use of this, as every thread in KataGo is synchronous, so with 1 thread max batch will only be 1 anyways.
nnMaxBatchSize = 16
# Maximum number of positions to send to GPU at once.
nnMaxBatchSize = 128
# Cache up to 2 ** this many neural net evaluations in case of transpositions in the tree.
nnCacheSizePowerOfTwo = 18
nnCacheSizePowerOfTwo = 23
# Size of mutex pool for nnCache is 2 ** this
nnMutexPoolSizePowerOfTwo = 14
nnMutexPoolSizePowerOfTwo = 17
# Randomize board orientation when running neural net evals?
nnRandomize = true
# If provided, force usage of a specific seed for nnRandomize instead of randomizing
# nnRandSeed = abcdefg
# How many threads should there be to feed positions to the neural net?
# Server threads are indexed 0,1,...(n-1) for the purposes of the below GPU settings arguments
@@ -185,11 +106,8 @@ numNNServerThreadsPerModel = 1
# Uncomment to tune OpenCL for every board size separately, rather than only the largest possible size
# openclReTunePerBoardSize = true
# Search randomization------------------------------------------------------------------------------
# Note that multithreading can also introduce a significant amount of nondeterminism.
# If provided, force usage of a specific seed for various things in the search instead of randomizing
# searchRandSeed = hijklmn
# Root move selection and biases------------------------------------------------------------------------------
# Not all of these parameters are applicable to analysis, some are only used for actual play
# Temperature for the early game, randomize between chosen moves with this temperature
chosenMoveTemperatureEarly = 0.5
@@ -210,6 +128,9 @@ rootDirichletNoiseTotalConcentration = 10.83
# Proportion of root policy that is noise
rootDirichletNoiseWeight = 0.25
# Number of symmetries to sample (WITH replacement) and average at the root
rootNumSymmetriesToSample = 1
# Using LCB for move selection?
useLcbForSelection = true
# How many stdevs a move needs to be better than another for LCB selection
@@ -220,21 +141,25 @@ minVisitPropForLCB = 0.15
# Internal params------------------------------------------------------------------------------
# Scales the utility of winning/losing
winLossUtilityFactor = 0.0
winLossUtilityFactor = 1.0
# Scales the utility for trying to maximize score
staticScoreUtilityFactor = 0.6
dynamicScoreUtilityFactor = 0.4
staticScoreUtilityFactor = 0.10
dynamicScoreUtilityFactor = 0.30
# Adjust dynamic score center this proportion of the way towards zero, capped at a reasonable amount.
dynamicScoreCenterZeroWeight = 0.20
dynamicScoreCenterScale = 0.75
# The utility of getting a "no result" due to triple ko or other long cycle in non-superko rulesets (-1 to 1)
noResultUtilityForWhite = 0.0
# The number of wins that a draw counts as, for white. (0 to 1)
drawEquivalentWinsForWhite = 0.5
# Exploration constant for mcts
cpuctExploration = 2.0
# adjusted from 0.9 / 0.6 to be more exploratory
cpuctExploration = 2
cpuctExplorationLog = 0.9
# FPU reduction constant for mcts
fpuReductionMax = 0.2
rootFpuReductionMax = 0.1
# Use parent average value for fpu base point instead of point value net estimate
fpuUseParentAverage = true
# Amount to apply a downweighting of children with very bad values relative to good ones
@@ -247,6 +172,6 @@ rootEndingBonusPoints = 0.5
rootPruneUselessMoves = true
# How big to make the mutex pool for search synchronization
mutexPoolSize = 8192
mutexPoolSize = 2048
# How many virtual losses to add when a thread descends through a node
numVirtualLossesPerThread = 1
+80 -95
View File
@@ -1,71 +1,102 @@
import numpy as np
from move import Move
class Board:
_move_id_counter = 0 # used to make a map to all moves across all games
class IllegalMoveException(Exception):
pass
def __init__(self, board_size = 19):
class Board:
_move_id_counter = 0 # used to make a map to all moves across all games
def __init__(self, board_size=19):
self.board_size = board_size
self.root = Move(None, (None, None))
self.root = Move(1, (None, None)) # root is 1=white so black is first
self.root.id = -1
self.current_move = self.root
self.all_moves = {}
self.board = np.empty( (self.board_size,self.board_size ) ) # values are indexes in `chains`
self.board.fill(np.nan)
self.chains = [] # cache of chain id
self._init_chains()
# -- move tree functions --
def _init_chains(self):
self.board = [[-1 for x in range(self.board_size)] for y in range(self.board_size)] # board pos -> chain id
self.chains = [] # chain id -> chain
self.prisoners = []
self.last_capture = []
try:
for m in self.moves:
self._validate_move_and_update_chains(m, True) # ignore ko since we didn't know if it was forced
except IllegalMoveException as e:
raise Exception(f"Unexpected illegal move ({str(e)})")
# -- move tree functions --
def update_board(self,move):
def neighbours_ix(cs):
return {(x + dx, y + dy) for x, y in cs for dy, dx in [(-1, 0), (1, 0), (0, -1), (0, 1)] if x + dx >= 0 and y + dy >= 0 and y + dy < self.board_size and x + dx < self.board_size}
def _validate_move_and_update_chains(self, move, ignore_ko):
def neighbours(moves):
return {
self.board[m.coords[1] + dy][m.coords[0] + dx]
for m in moves
for dy, dx in [(-1, 0), (1, 0), (0, -1), (0, 1)]
if 0 <= m.coords[0] + dx < self.board_size and 0 <= m.coords[1] + dy < self.board_size
}
def neighbours(cs):
return {self.board[y][x] for x, y in neighbours_ix(cs)}
ko_or_snapback = len(self.last_capture) == 1 and self.last_capture[0] == move
self.last_capture = []
nb_chains = list({int(c) for c in neighbours([move.coords]) if not np.isnan(c) and self.chains[int(c)][0].player == move.player})
if move.is_pass:
return
if self.board[move.coords[1]][move.coords[0]] != -1:
raise IllegalMoveException("Space occupied")
nb_chains = list({c for c in neighbours([move]) if c >= 0 and self.chains[c][0].player == move.player})
if nb_chains:
self.board[move.coords[1], move.coords[0]] = nb_chains[0]
self.board[np.isin(self.board, nb_chains)] = nb_chains[0]
this_chain = nb_chains[0]
self.board = [
[nb_chains[0] if sq in nb_chains else sq for sq in line] for line in self.board
] # merge chains connected by this move
for oc in nb_chains[1:]:
self.chains[nb_chains[0]] += self.chains[oc]
self.chains[oc] = []
self.chains[nb_chains[0]].append(move)
else:
self.board[move.coords[1], move.coords[0]] = len(self.chains)
this_chain = len(self.chains)
self.chains.append([move])
opp_nb_chains = {int(c) for c in neighbours([move.coords]) if self.chains[int(c)][0].player != move.player}
capture = False
self.board[move.coords[1]][move.coords[0]] = this_chain
opp_nb_chains = {c for c in neighbours([move]) if c >= 0 and self.chains[c][0].player != move.player}
for c in opp_nb_chains:
if np.nan not in neighbours([m.coords for m in self.chains[c]]):
capture = True
if -1 not in neighbours(self.chains[c]):
self.last_capture += self.chains[c]
for om in self.chains[c]:
self.board[om.coords[1], om.coords[0]] = np.nan
self.board[om.coords[1]][om.coords[0]] = -1
self.chains[c] = []
if not capture:
if np.nan not in neighbours([m.coords for m in self.chains[c]]):
if ko_or_snapback and len(self.last_capture) == 1 and not ignore_ko:
raise IllegalMoveException("Ko")
self.prisoners += self.last_capture
if -1 not in neighbours(self.chains[this_chain]):
raise IllegalMoveException("Suicide")
# Play a Move from the current position, returns false if invalid.
def play(self, move) -> bool:
def play(self, move, ignore_ko=False):
try:
self._validate_move_and_update_chains(move, ignore_ko)
except IllegalMoveException as e:
self._init_chains() # restore
raise
move = self.current_move.play(move) # traverse or append
move = self.current_move.play(move) # traverse or append
if not move.id:
move.id = Board._move_id_counter
Board._move_id_counter += 1
self.all_moves[move.id] = move
self.current_move = move
return True
return move
def undo(self):
if self.current_move != self.root:
self.current_move = self.current_move.parent
self._init_chains()
@property
def moves(self) -> list: # flat list of moves to current
moves = []
p = self.current_move
@@ -74,82 +105,36 @@ class Board:
p = p.parent
return moves[::-1]
def __iter__(self):
return self.moves.__iter__()
def __getitem__(self, ix):
if ix == -1:
return self.current_move
else:
return self.moves[ix]
@property
def current_player(self):
return self.current_move.player
# --analysis
# --analysis
def store_analysis(self,json):
id = int(json["id"])
def store_analysis(self, json):
if json["id"].starts_with("PASS_"):
id = int(json["id"].lstrip("PASS_"))
else:
id = int(json["id"])
move = self.all_moves.get(id)
if move: # else this should be old
move.set
if move: # else this should be old
move.set_analysis(json)
else:
print("WARNING: ORPHANED ANALYSIS FOUND - RECENT NEW GAME?")
# -- board visualization etc
# -- board visualization etc
# ko: single capture and
# other not allowed: suicide
# todo - factor into global state etc? for valid move, cached etc
@property
def stones(self):
board = np.empty( (self.board_size,self.board_size ) )
def neighbours_ix(cs):
return {(x+dx,y+dy) for x,y in cs for dy, dx in [(-1,0),(1,0),(0,-1),(0,1)] if x+dx >=0 and y+dy >=0 and y+dy < self.board_size and x+dx < self.board_size}
def neighbours(cs):
return {board[y][x] for x,y in neighbours_ix(cs) if not np.isnan(board[y][x]) }
board.fill(np.nan)
moves = self.moves()
chains = []
for m in moves:
nb_chains = list({int(c) for c in neighbours([m.coords]) if chains[int(c)][0].player==m.player})
if nb_chains:
board[m.coords[1],m.coords[0]] = nb_chains[0]
board[ np.isin(board,nb_chains) ] = nb_chains[0]
for oc in nb_chains[1:]:
chains[nb_chains[0]] += chains[oc]
chains[oc] = []
chains[nb_chains[0]].append(m)
else:
board[m.coords[1],m.coords[0]] = len(chains)
chains.append([m])
opp_nb_chains = {int(c) for c in neighbours([m.coords]) if chains[int(c)][0].player != m.player}
for c in opp_nb_chains:
if np.nan not in neighbours([m.coords for m in chains[c]]):
for om in chains[c]:
board[om.coords[1],om.coords[0]] = np.nan
chains[c] = []
return chains
return sum(self.chains, [])
def sgf(self):
return "SGF[]"
if __name__ == "__main__":
b=Board(9)
b.play(Move(gtpcoords="A3",player=0))
b.play(Move(gtpcoords="A9",player=0))
b.play(Move(gtpcoords="B9",player=0))
b.play(Move(gtpcoords="A4",player=0))
b.play(Move(gtpcoords="C8",player=0))
b.play(Move(gtpcoords="C9",player=0))
print(b.stones)
b.play(Move(gtpcoords="J9",player=1))
b.play(Move(gtpcoords="J8", player=0))
print(b.stones)
b.play(Move(gtpcoords="H9", player=0))
print(b.stones)
b.play(Move(gtpcoords="J9", player=0))
print(b.stones)
def __str__(self):
return (
"\n".join("".join("BW"[self.chains[c][0].player] if c >= 0 else "-" for c in l) for l in self.board)
+ f"\ncaptures: {self.prisoners}"
)
+1 -1
View File
@@ -27,7 +27,7 @@
"line_color": [0,0,0]
},
"engine": {
"command": "KataGo/katago analysis -model KataGo/b10.gz -config KataGo/japanese_explore_score.cfg"
"command": "KataGo/katago analysis -model KataGo/b10.gz -config KataGo/analysis_config.cfg -analysis-threads 8"
},
"trainer": {
"balance_play_target_score": 2,
+6 -2
View File
@@ -18,6 +18,10 @@ class EngineControls(GridLayout):
def action(self, message, *args):
self.engine.action(message, *args)
@property
def board(self):
return self.engine.board
@property
def ready(self):
return self.engine.ready
@@ -32,11 +36,11 @@ class EngineControls(GridLayout):
@property
def moves(self):
return self.engine.moves
return self.engine.board.moves
@property
def current_player(self):
return self.engine.current_player()
return self.engine.current_player
def redraw(self, include_board=False):
if include_board:
+127 -178
View File
@@ -1,12 +1,14 @@
import re
import json
import random
import re
import shlex
import subprocess
import threading
import json
import time
from queue import Queue
from move import Move, MoveTree
from board import Board, IllegalMoveException
from move import Move
class KataEngine:
@@ -15,7 +17,10 @@ class KataEngine:
self.command = shlex.split(config.get("engine")["command"])
analysis_settings = config.get("analysis")
self.visits = [[analysis_settings["pass_visits"], analysis_settings["visits"]], [analysis_settings["pass_visits_fast"], analysis_settings["visits_fast"]]]
self.visits = [
[analysis_settings["pass_visits"], analysis_settings["visits"]],
[analysis_settings["pass_visits_fast"], analysis_settings["visits_fast"]],
]
self.min_nopass_visits = analysis_settings["nopass_visits"]
self.train_settings = config.get("trainer")
self.debug = config.get("debug")["level"]
@@ -24,19 +29,18 @@ class KataEngine:
self.ready = False
self.stones = []
self.message_queue = None
self.move_tree = MoveTree()
self.board = Board(self.boardsize)
self.kata = None
@property
def current_player(self):
return self.move_tree.current_player
return self.board.current_player
def restart(self, boardsize):
self.ready = False
if not self.message_queue:
self.message_queue = Queue()
self.analysis_semaphore = threading.Semaphore(1)
self.stop_analyzing = True
self.thread = threading.Thread(target=self._engine_thread, daemon=True).start()
else:
with self.message_queue.mutex:
@@ -47,68 +51,11 @@ class KataEngine:
def action(self, message, *args):
self.message_queue.put([message, *args])
def gtpread(self):
lines = []
while self.kata:
lines.append(self.kata.stdout.readline().decode())
if lines[-1].strip() == "":
break
return lines[:-1]
def gtpwrite(self, cmd):
if self.debug:
print("WRITE", cmd)
try:
self.kata.stdin.write((cmd + "\n").encode("utf-8"))
self.kata.stdin.flush()
except Exception:
self.controls.info.text = "Engine died, please restart app"
raise
def gtpcommand(self, cmd):
self.gtpwrite(cmd)
return self.gtpread()
def raw_gtpplaycommand(self, move):
if move == "undo":
output = self.gtpcommand("undo")
else:
output = self.gtpcommand(f"play {Move.PLAYERS[move.player]} {move.gtp()}")
output = "".join(output)
if self.debug and "?" in output:
print(move, output)
return "?" not in output
def update_stones(self):
board_output = self.gtpcommand("showboard")
info = self.gtpread() # new kata
board = [re.sub(r"[^\.ox]", "", l.lower()) for l in board_output[2:]]
self.stones = []
for y, line in enumerate(board[::-1]):
for x, st in enumerate(line):
if st != ".":
self.stones.append(("xo".index(st), x, y))
self.controls.redraw(include_board=False)
def gtpplaycommand(self, move):
self.stop_analyzing = True
self.analysis_semaphore.acquire()
if self.raw_gtpplaycommand(move): # update moves array if engine accepts move
if move == "undo":
self.move_tree.undo()
else:
self.move_tree.play(move)
self.update_stones()
# start analyzing new board position
self.stop_analyzing = False
self.analysis_semaphore.release()
# engine main loop
def _engine_thread(self):
self.kata = subprocess.Popen(self.command, stdin=subprocess.PIPE, stdout=subprocess.PIPE)
print(self.command, self.kata)
analysis_thread = threading.Thread(target=self._analyze_thread, args=(25,), daemon=True).start()
self.stop_analyzing = False
print("STARTING KATAGO", self.command, self.kata)
analysis_thread = threading.Thread(target=self._analyze_thread, daemon=True).start()
msg, *args = self.message_queue.get()
while True:
@@ -121,168 +68,170 @@ class KataEngine:
raise
msg, *args = self.message_queue.get()
def play(self, move):
try:
mr = self.board.play(move)
move_id = mr.id
except IllegalMoveException as e:
print(str(e))
self.controls.info.text = f"Illegal move: {str(e)}"
return
self._request_analysis(mr)
def _request_analysis(self, move):
while not self.kata:
print("waiting for kata to start")
time.sleep(0.05)
move_id = move.id
moves = self.board.moves
fast = self.controls.ai_fast.active
query = {
"id": str(move_id),
"moves": [str(m) for m in moves],
"rules": "japanese",
"komi": self.komi,
"boardXSize": self.boardsize,
"boardYSize": self.boardsize,
"analyzeTurns": [len(moves) - 1],
"includeOwnership": True,
"maxVisits": self.visits[fast][1],
}
print("query", query)
self.kata.stdin.write(json.dumps(query).encode())
query.update({"id": f"PASS_{move_id}", "maxVisits": self.visits[fast][0], "includeOwnership": True})
query["moves"] += ["pass"]
query["analyzeTurns"][0] += 1
print("pass-query", query)
self.kata.stdin.write(json.dumps(query).encode())
# engine action functions
def _do_play(self, *args):
self.gtpplaycommand(Move(player=self.current_player(), coords=args[0]))
move = Move(player=self.current_player, coords=args[0])
self.play(move)
self.controls.undo.disabled = True # undo while waiting for this does weird things
undid = False
self.controls.info.text = ""
if self.controls.auto_undo.active(1 - self.current_player()):
print("undo active", self.current_player(), self.controls.auto_undo.active(self.current_player()))
undid = self._auto_undo()
if self.controls.auto_undo.active(1 - self.current_player):
undid = self._auto_undo(move)
if self.controls.ai_auto.active and not undid:
self._do_aimove(True)
self.controls.undo.disabled = False
def _evaluate_move(self, show=True):
while not self.move_tree[-1].analysis: # ensure analysis has started, otherwise race condition on multi ai move
time.sleep(0.01)
self.analysis_semaphore.acquire() and self.analysis_semaphore.release() # wait for analysis to finish
if self.moves[-1].evaluation and show:
def _evaluate_move(self, move, show=True):
while not move.analysis:
time.sleep(0.01) # wait for analysis
if self.board.current_move.evaluation and show:
self.controls.info.text = f"Your move {self.moves[-1].gtp()} was {100 * self.moves[-1].evaluation:.1f}% efficient and lost {self.moves[-1].points_lost:.1f} point(s).\n"
def _auto_undo(self):
def _auto_undo(self, move):
ts = self.train_settings
self.controls.info.text = "Evaluating..."
self._evaluate_move()
if (
self.moves[-1].evaluation
and self.moves[-1].evaluation < ts["undo_eval_threshold"]
and self.moves[-1].points_lost >= ts["undo_point_threshold"]
move.evaluation
and move.evaluation < ts["undo_eval_threshold"]
and move.points_lost >= ts["undo_point_threshold"]
and ts["num_undo_prompts"] > 0
):
if self.moves[-1].outdated_evaluation:
outdated_points_lost = (1 - self.moves[-1].outdated_evaluation) * self.moves[-1].points_lost / (1 - self.moves[-1].evaluation)
if move.outdated_evaluation:
outdated_points_lost = (1 - move.outdated_evaluation) * move.points_lost / (1 - move.evaluation)
# so if the move was not that far off (>undo_outdated_eval_threshold) and according to last move's analysis it was fine, don't undo.
if (
self.moves[-1].outdated_evaluation
and (self.moves[-1].outdated_evaluation >= ts["undo_eval_threshold"] or outdated_points_lost < ts["undo_point_threshold"])
and (self.moves[-1].evaluation > ts["undo_outdated_eval_threshold"] or outdated_points_lost < ts["undo_point_threshold"])
move.outdated_evaluation
and (
move.outdated_evaluation >= ts["undo_eval_threshold"]
or outdated_points_lost < ts["undo_point_threshold"]
)
and (
move.evaluation > ts["undo_outdated_eval_threshold"]
or outdated_points_lost < ts["undo_point_threshold"]
)
):
self.controls.info.text += f"\nBut according to my previous evaluation it was {self.moves[-1].outdated_evaluation*100:.1f}% effective and lost {outdated_points_lost:.1f} point(s), so let's continue anyway.\n"
self.controls.info.text += f"\nBut according to my previous evaluation it was {move.outdated_evaluation*100:.1f}% effective and lost {outdated_points_lost:.1f} point(s), so let's continue anyway.\n"
else:
if len(self.moves[-2].undos) < ts["num_undo_prompts"]:
if len(self.board.current_move.parent.children) <= ts["num_undo_prompts"]:
self.controls.info.text += f"\nLet's try again.\n"
self.gtpplaycommand("undo")
self.board.undo()
return True
else:
evaled_moves = sorted([m for m in self.moves[-2].undos + [self.moves[-1]] if m.evaluation], key=lambda m: -m.evaluation)
if evaled_moves and evaled_moves[0].coords != self.moves[-1].coords:
self.gtpplaycommand("undo")
self.gtpplaycommand(evaled_moves[0])
evaled_moves = sorted(
[m for m in self.board.current_move.parent.children if m.evaluation], key=lambda m: -m.evaluation
)
if evaled_moves and evaled_moves[0].coords != move.coords:
self.board.undo()
self.board.play(evaled_moves[0])
summary = "\n".join(f"{m.gtp()}: {100*m.evaluation:.1f}% effective" for m in evaled_moves)
self.controls.info.text += f"\nYour moves:\n{summary}.\nLet's continue with {evaled_moves[0].gtp()}.\n"
self.controls.info.text += (
f"\nYour moves:\n{summary}.\nLet's continue with {evaled_moves[0].gtp()}.\n"
)
return False
def _do_aimove(self, auto=False):
def _do_aimove(self, move, auto=False):
ts = self.train_settings
if not auto:
self.controls.info.text = "Thinking..."
self._evaluate_move(auto and not self.controls.auto_undo.active(1 - self.current_player()))
self._evaluate_move(auto and not self.controls.auto_undo.active(1 - self.current_player))
# select move
pos_moves = [(d["move"], float(d["scoreMean"]), d["evaluation"]) for d in self.moves[-1].analysis if int(d["visits"]) >= ts["balance_play_min_visits"]]
pos_moves = [
(d["move"], float(d["scoreMean"]), d["evaluation"])
for d in move.analysis
if int(d["visits"]) >= ts["balance_play_min_visits"]
]
if ts["show_ai_options"]:
self.controls.info.text += "AI Options: " + " ".join([f"{move}({100*eval:.0f}%,{score:.1f}pt)" for move, score, eval in pos_moves])
self.controls.info.text += "AI Options: " + " ".join(
[f"{move}({100*eval:.0f}%,{score:.1f}pt)" for move, score, eval in pos_moves]
)
selmove = pos_moves[0][0]
if self.controls.ai_balance.active and pos_moves[0][0] != "pass": # don't play suicidal to balance score - pass when it's best
if (
self.controls.ai_balance.active and pos_moves[0][0] != "pass"
): # don't play suicidal to balance score - pass when it's best
selmoves = [
move
for move, score, eval in pos_moves
if eval > ts["balance_play_randomize_eval"] or eval > ts["balance_play_min_eval"] and score > ts["balance_play_target_score"]
if eval > ts["balance_play_randomize_eval"]
or eval > ts["balance_play_min_eval"]
and score > ts["balance_play_target_score"]
]
selmove = random.choice(selmoves) # some kind of when further ahead play worse?
self.gtpplaycommand(Move(player=self.current_player(), gtpcoords=selmove, robot=True))
self.board.play(Move(player=self.current_player, gtpcoords=selmove, robot=True))
def _do_undo(self):
if self.controls.ai_auto.active and self.moves[-1].robot:
self.gtpplaycommand("undo")
if self.controls.ai_lock.active and self.controls.auto_undo.active(self.moves[-2].player) and len(self.moves[-2].undos) >= self.train_settings["num_undo_prompts"]:
self.controls.info.text = f"Can't undo more than {self.train_settings['num_undo_prompts']} time(s) when locked"
if self.controls.ai_auto.active and self.board.current_move.robot:
self.board.undo()
if (
self.controls.ai_lock.active
and self.controls.auto_undo.active(self.board.current_move.parent.player)
and len(self.board.current_move.parent.player.children) > self.train_settings["num_undo_prompts"]
):
self.controls.info.text = (
f"Can't undo more than {self.train_settings['num_undo_prompts']} time(s) when locked"
)
return
self.gtpplaycommand("undo")
self.board.undo()
def _do_init(self, boardsize, komi=None):
self.boardsize = boardsize
self.stop_analyzing = True
self.analysis_semaphore.acquire()
self.stones = []
self.move_tree = MoveTree()
self.board = Board(boardsize)
self._request_analysis(self.board.root)
self.controls.redraw(include_board=True)
self.gtpcommand(f"boardsize {boardsize}")
self.gtpcommand(f"komi {komi or self.komi}")
self.gtpcommand("clear_board")
self.ready = True
self.analysis_semaphore.release()
self.stop_analyzing = False
def _do_analyze_sgf(self, sgf):
self._do_init(self.boardsize, self.komi)
sgfmoves = re.findall(r"([BW])\[([a-z]{2})\]", sgf)
for move in [Move(player=Move.PLAYERS.index(p.upper()), sgfcoords=(mv, self.boardsize)) for p, mv in sgfmoves]:
while not self.moves[-1].analysis:
time.sleep(0.01)
self.analysis_semaphore.acquire() and self.analysis_semaphore.release() # wait for analysis to finish
self.gtpplaycommand(move)
self.controls.info.text = f"Analyzing move {move.gtp()}"
self.controls.info.text = "Analysis done!"
moves = [Move(player=Move.PLAYERS.index(p.upper()), sgfcoords=(mv, self.boardsize)) for p, mv in sgfmoves]
for move in moves:
self.board.play(move)
while not all(m.analysis for m in moves):
time.sleep(0.01)
self.controls.info.text = f"{sum([1 if m.analysis else 0 for m in moves])}/{len(moves)} analyzed"
# analysis thread
def _analyze_thread(self, interval):
def _analyze_thread(self):
while True:
num_visits = self.visits[1 if self.controls.ai_fast.active else 0]
while self.stop_analyzing: # TODO: cleaner concurrency?
time.sleep(0.01)
self.analysis_semaphore.acquire()
for mode in [0, 1]: # pass, analyze
if self.stop_analyzing:
break
if mode == 0:
passmove = Move(player=self.current_player(), gtpcoords="pass")
if len(self.moves) >= 2:
undo_mode = 0 # reverse order mode
self.raw_gtpplaycommand("undo")
self.raw_gtpplaycommand("undo")
self.raw_gtpplaycommand(passmove)
if not self.raw_gtpplaycommand(self.moves[-1]): # could not change order -> restore state and fall back
undo_mode = 1
self.raw_gtpplaycommand("undo") # pass
self.raw_gtpplaycommand(self.moves[-2])
self.raw_gtpplaycommand(self.moves[-1])
elif not self.raw_gtpplaycommand(self.moves[-2]): # could not change order -> restore state and fall back
undo_mode = 1
self.raw_gtpplaycommand("undo") # moves[-1]
self.raw_gtpplaycommand("undo") # pass
self.raw_gtpplaycommand(self.moves[-2])
self.raw_gtpplaycommand(self.moves[-1])
else:
undo_mode = 1 # play corner for pass mode
if undo_mode == 1:
for coords in [(0, 0), (0, self.boardsize - 1), (self.boardsize - 1, 0), (self.boardsize - 1, self.boardsize - 1), (None, None)]:
if self.raw_gtpplaycommand(Move(player=self.current_player(), coords=coords)):
break
self.gtpwrite(f"kata-analyze interval {interval} minmoves 2 {'ownership true' if mode==1 else ''}")
self.kata.stdout.readline() # =
tot_visits = tot_nopass_visits = 0
while not self.stop_analyzing and (tot_visits < num_visits[mode] or tot_nopass_visits < self.min_nopass_visits):
line = self.kata.stdout.readline().decode()
line, *ownership = line.split("ownership")
moves = [re.sub("pv .*", "", str).split(" ") for str in line.split("info ")[1:]]
move_dicts = [{move[i]: move[i + 1] for i in range(0, len(move) - 1, 2)} for move in moves]
self.controls.update_analysis(move_dicts, mode, ownership)
tot_visits = sum([int(d["visits"]) for d in move_dicts], 0)
tot_nopass_visits = sum([int(d["visits"]) for d in move_dicts if d["move"] != "pass"], 0)
if self.debug:
print("mode=", mode, "visits=", tot_visits, "nopass=", tot_nopass_visits) # , "stop_analyzing?", stop_analyzing
self.gtpcommand("stop") # reads for analyze empty line
self.gtpread() # for stop line empty line
# for modes loop
if mode == 0: # undo A1
self.raw_gtpplaycommand("undo")
if undo_mode == 0:
self.raw_gtpplaycommand("undo")
self.raw_gtpplaycommand("undo")
self.raw_gtpplaycommand(self.moves[-2])
self.raw_gtpplaycommand(self.moves[-1])
else:
self.stop_analyzing = True # ehh
self.analysis_semaphore.release() # signal other threads waiting for analysis to finish
line = self.kata.stdout.readline()
print("KATA LINE", line)
self.board.store_analysis(json.loads(line))
+2 -2
View File
@@ -93,7 +93,7 @@
id: value
bold: True
<Badukpan>:
<BadukPanWidget>:
size: self.parent.height, self.parent.height
engine: self.parent.controls
@@ -278,7 +278,7 @@
Rectangle:
pos: self.pos
size: self.size
Badukpan:
BadukPanWidget:
id: board
pos_hint: {"x":0, "top":0}
EngineControls:
+18 -17
View File
@@ -17,9 +17,9 @@ COLORS = Config.get("ui")["stones"]
GHOST_ALPHA = Config.get("ui")["ghost_alpha"]
class Badukpan(Widget):
class BadukPanWidget(Widget):
def __init__(self, **kwargs):
super(Badukpan, self).__init__(**kwargs)
super(BadukPanWidget, self).__init__(**kwargs)
self.ghost_stone = []
self.gridpos = []
self.grid_size = 0
@@ -113,20 +113,21 @@ class Badukpan(Widget):
self.canvas.clear()
with self.canvas:
# stones
last_move = self.engine.moves[-1].coords
eval_map = {m.coords: (m.evaluation, m.previous_temperature) for m in self.engine.moves}
moves = self.engine.board.moves
last_move = self.engine.board.current_move
eval_map = {m.coords: (m.evaluation, m.previous_temperature) for m in moves}
eval_on = [self.engine.eval.active(0), self.engine.eval.active(1)]
has_stone = {}
for i, (ci, x, y) in enumerate(self.engine.stones):
has_stone[(x, y)] = ci
eval, evalsize = eval_map.get((x, y), (None, None))
evalcol = self._eval_spectrum(eval) if eval_on[ci] and eval else None
inner = COLORS[1 - ci] if ((x, y) == last_move) else None
self.draw_stone(x, y, COLORS[ci], inner, evalcol, evalsize)
for i, m in enumerate(self.engine.stones):
has_stone[m.coords] = m.player
eval, evalsize = eval_map.get(m.coords, (None, None))
evalcol = self._eval_spectrum(eval) if eval_on[m.player] and eval else None
inner = COLORS[1 - m.player] if (m == last_move) else None
self.draw_stone(m.coords[0], m.coords[1], COLORS[m.player], inner, evalcol, evalsize)
# ownership
ownership = self.engine.moves[-1].ownership
if self.engine.ownership.active and ownership:
if self.engine.ownership.active and last_move.ownership:
ownership = last_move.ownership
rsz = self.grid_size * 0.2
ix = 0
cp = self.engine.current_player
@@ -141,15 +142,15 @@ class Badukpan(Widget):
# undos
undo_coords = set()
alpha = Config.get("ui")["undo_alpha"]
for m in self.engine.moves[-1].undos:
for m in self.engine.board.current_move.children:
if m.evaluation and m.coords[0] is not None:
undo_coords.add(m.coords)
evalcol = (*self._eval_spectrum(m.evaluation), alpha)
self.draw_stone(m.coords[0], m.coords[1], (*COLORS[m.player][:3], alpha), Config.get("ui")["undo_circle_col"], evalcol, self.EVAL_BOUNDS[1])
# hints
if self.engine.moves[-1].analysis and self.engine.hints.active(self.engine.current_player):
for d in self.engine.moves[-1].analysis:
if last_move.analysis and self.engine.hints.active(self.engine.current_player):
for d in last_move.analysis:
move = Move(gtpcoords=d["move"], player=0)
c = [*self._eval_spectrum(d["evaluation"]), 0.5]
if move.coords[0] is not None and move.coords not in undo_coords:
@@ -160,9 +161,9 @@ class Badukpan(Widget):
self.draw_stone(*self.ghost_stone, (*COLORS[self.engine.current_player], GHOST_ALPHA))
# pass circle
passed = len(self.engine.moves) > 1 and self.engine.moves[-1].gtp() == "pass"
passed = len(moves) > 1 and last_move.is_pass
if passed:
if len(self.engine.moves) > 2 and self.engine.moves[-2].gtp() == "pass":
if len(moves) > 2 and moves[-2].is_pass:
text = "game\nend"
else:
text = "pass"
+23 -13
View File
@@ -13,8 +13,8 @@ class Move:
self.parent = None
self.robot = robot
self.analysis = None
self.outdated_evaluation = None
self.pass_analysis = None
self.outdated_evaluation = None
self.evaluation = None
self.ownership = None
self.points_lost = 0
@@ -27,6 +27,9 @@ class Move:
def __eq__(self, other):
return self.coords == other.coords and self.player == other.player
def __hash__(self):
return self.gtp().__hash__()
def play(self, move):
try:
return self.children[self.children.index(move)]
@@ -37,32 +40,35 @@ class Move:
def temperature(self):
if self.analysis:
best_score = float(self.analysis[0]["scoreMean"])
worst_score = -float(self.pass_analysis[0]["scoreMean"])
best_score = float(self.analysis[0]["scoreLead"])
worst_score = -float(self.pass_analysis[0]["scoreLead"])
return best_score - worst_score
else:
return 0
def evaluate(self,analysis):
self.analysis = analysis
def evaluate(self,analysis_blob):
self.analysis = analysis_blob['moveInfos']
self.ownership = analysis_blob['ownership']
previous_move = self.parent
if not self.analysis and self.pass_analysis and previous_move.analysis:
return
# TODO: update children?
best_score = float(previous_move.analysis[0]["scoreMean"])
worst_score = -float(previous_move.pass_analysis[0]["scoreMean"])
last_move_score = -float(self.analysis[0]["scoreMean"])
best_score = float(previous_move.analysis[0]["scoreLead"])
worst_score = -float(previous_move.pass_analysis[0]["scoreLead"])
last_move_score = -float(self.analysis[0]["scoreLead"])
self.previous_temperature = best_score - worst_score
self.points_lost = best_score - last_move_score
prev_analysis_current_move = [d for d in previous_move.analysis if d["move"] == self.gtp()]
if abs(self.previous_temperature) > 0.5:
self.evaluation = (last_move_score - worst_score) / (best_score - worst_score)
self.move_options = [previous_move.analysis[0]["scoreMean"]]
self.move_options = [previous_move.analysis[0]["scoreLead"]]
else:
self.evaluation = None
if self.evaluation:
self.comment = f"Evaluation: {100*self.evaluation:.1f}%{' (AI Move)' if self.robot else ''}\n"
if prev_analysis_current_move:
self.outdated_evaluation = (prev_analysis_current_move[0]["scoreMean"] - worst_score) / (
self.outdated_evaluation = (prev_analysis_current_move[0]["scoreLead"] - worst_score) / (
best_score - worst_score
)
self.comment += f"(Was considered last move as: {100 * self.outdated_evaluation:.1f}%)\n"
@@ -70,9 +76,13 @@ class Move:
self.comment = "Temperature too low for evaluation\n"
self.comment += f"Estimate point loss: {self.points_lost:.1f}\n"
self.comment += f"Last move score was {last_move_score:.1f}\n"
self.comment += f"Score of top move was {previous_move.analysis[0]['scoreMean']:.1f} @ {previous_move.analysis[0]['move']}\n"
self.comment += f"Score of top move was {previous_move.analysis[0]['scoreLead']:.1f} @ {previous_move.analysis[0]['move']}\n"
self.comment += f"Pass score was {worst_score:.1f}\n"
@property
def is_pass(self):
return self.coords[0] is None
def gtp2ix(self, gtpmove):
if "pass" in gtpmove:
return (None, None)
@@ -85,7 +95,7 @@ class Move:
return Move.SGF_COORD.index(sgfmove[0]), boardsize - Move.SGF_COORD.index(sgfmove[1]) - 1
def gtp(self):
if self.coords[0] is None:
if self.is_pass:
return "pass"
return Move.GTP_COORD[self.coords[0]] + str(self.coords[1] + 1)
@@ -93,7 +103,7 @@ class Move:
return f"{Move.SGF_COORD[self.coords[0]]}{Move.SGF_COORD[boardsize - self.coords[1] - 1]}"
def sgf(self, boardsize):
if self.coords[0] is None:
if self.is_pass:
return f"{Move.PLAYERS[self.player]}[]"
else:
return f"{Move.PLAYERS[self.player]}[{self.sgfcoords(boardsize)}]"
+92
View File
@@ -0,0 +1,92 @@
import pytest
from board import Board, IllegalMoveException
from move import Move
class TestBoard:
def nonempty_chains(self, b):
return [c for c in b.chains if c]
def test_merge(self):
b = Board(9)
b.play(Move(gtpcoords="B9", player=0))
b.play(Move(gtpcoords="A3", player=0))
b.play(Move(gtpcoords="A9", player=0))
assert 2 == len(self.nonempty_chains(b))
assert 3 == len(b.stones)
assert 0 == len(b.prisoners)
def test_collide(self):
b = Board(9)
b.play(Move(gtpcoords="B9", player=0))
with pytest.raises(IllegalMoveException):
b.play(Move(gtpcoords="B9", player=1))
assert 1 == len(self.nonempty_chains(b))
assert 1 == len(b.stones)
assert 0 == len(b.prisoners)
def test_capture(self):
b = Board(9)
b.play(Move(gtpcoords="A2", player=0))
b.play(Move(gtpcoords="B1", player=1))
b.play(Move(gtpcoords="A1", player=1))
b.play(Move(gtpcoords="C1", player=0))
assert 3 == len(self.nonempty_chains(b))
assert 4 == len(b.stones)
assert 0 == len(b.prisoners)
b.play(Move(gtpcoords="B2", player=0))
assert 2 == len(self.nonempty_chains(b))
assert 3 == len(b.stones)
assert 2 == len(b.prisoners)
b.play(Move(gtpcoords="B1", player=0))
with pytest.raises(IllegalMoveException) as exc:
b.play(Move(gtpcoords="A1", player=1))
assert "Suicide" in str(exc.value)
assert 1 == len(self.nonempty_chains(b))
assert 4 == len(b.stones)
assert 2 == len(b.prisoners)
def test_snapback(self):
b = Board(9)
for move in ["C1", "D1", "E1", "C2", "D3", "E4", "F2", "F3", "F4"]:
b.play(Move(gtpcoords=move, player=0))
for move in ["D2", "E2", "C3", "D4", "C4"]:
b.play(Move(gtpcoords=move, player=1))
assert 5 == len(self.nonempty_chains(b))
assert 14 == len(b.stones)
assert 0 == len(b.prisoners)
b.play(Move(gtpcoords="E3", player=1))
assert 4 == len(self.nonempty_chains(b))
assert 14 == len(b.stones)
assert 1 == len(b.prisoners)
b.play(Move(gtpcoords="D3", player=0))
assert 4 == len(self.nonempty_chains(b))
assert 12 == len(b.stones)
assert 4 == len(b.prisoners)
def test_ko(self):
b = Board(9)
for move in ["A2", "B1"]:
b.play(Move(gtpcoords=move, player=0))
for move in ["B2", "C1"]:
b.play(Move(gtpcoords=move, player=1))
b.play(Move(gtpcoords="A1", player=1))
assert 4 == len(self.nonempty_chains(b))
assert 4 == len(b.stones)
assert 1 == len(b.prisoners)
with pytest.raises(IllegalMoveException) as exc:
b.play(Move(gtpcoords="B1", player=0))
assert "Ko" in str(exc.value)
b.play(Move(gtpcoords="B1", player=0), ignore_ko=True)
assert 2 == len(b.prisoners)
with pytest.raises(IllegalMoveException) as exc:
b.play(Move(gtpcoords="A1", player=1))
b.play(Move(gtpcoords="F1", player=1))
b.play(Move(coords=(None, None), player=0))
b.play(Move(gtpcoords="A1", player=1))
assert 3 == len(b.prisoners)