From d6141d91d62ce76cf2f8d12ba67dbf0b05783494 Mon Sep 17 00:00:00 2001 From: Sander Land Date: Mon, 20 Apr 2020 23:09:47 +0200 Subject: [PATCH] play mode --- game.py | 51 +++++++------ game_node.py | 14 +++- gui/badukpan.py | 7 +- gui/controls.py | 84 ++++++++-------------- gui/kivyutils.py | 70 ++++++++++++++++-- gui.kv => katrain.kv | 165 +++++++++++++++++++------------------------ katrain.py | 15 ++-- sgf_parser.py | 14 ++-- 8 files changed, 227 insertions(+), 193 deletions(-) rename gui.kv => katrain.kv (82%) diff --git a/game.py b/game.py index 0ce4771..9ba9fc0 100644 --- a/game.py +++ b/game.py @@ -7,6 +7,7 @@ from typing import List from kivy.clock import Clock +from constants import OUTPUT_DEBUG from game_node import GameNode from sgf_parser import SGF, Move @@ -195,26 +196,34 @@ class Game: return f"SGF with analysis written to {file_name}" def ai_move(self, train_settings): - while not self.current_node.analysis_ready: - self.katrain.controls.set_status("Thinking...") + cn = self.current_node + while not cn.analysis_ready: + self.katrain.controls.set_status("Thinking...") # TODO: non blocking somehow? time.sleep(0.05) - # select move - ai_moves = self.current_node.candidate_moves - pos_moves = [ - [d["move"], d["scoreLead"], d["pointsLost"]] for i, d in enumerate(ai_moves) if i == 0 or int(d["visits"]) >= train_settings["balance_play_min_visits"] - ] # TODO: lcb based ? - sel_moves = pos_moves[:1] - # don't play suicidal to balance score - pass when it's best - if self.katrain.controls.ai_balance.active and pos_moves[0][0] != "pass": # TODO: settings where they belong? - sel_moves = [ - (move, score, points_lost) - for move, score, points_lost in pos_moves - if points_lost < train_settings["balance_play_randomize_score"] - or points_lost < train_settings["balance_play_min_eval"] - and -self.current_node.move.player_sign * score > self.config["balance_play_target_score"] - ] or sel_moves - aimove = Move.from_gtp(random.choice(sel_moves)[0], player=self.next_player) + ai_moves = cn.candidate_moves + mode = self.katrain.controls.player_mode(cn.next_player) + + if "policy" in mode and cn.policy: + policy_moves = cn.policy_ranking + self.katrain.log(f"Top 5 policy moves are: {policy_moves[:5]}", OUTPUT_DEBUG) + aimove = policy_moves[0][0] + elif "balance" in mode and ai_moves[0]['move'] != "pass": # don't play suicidal to balance score - pass when it's best + sign = cn.player_sign(cn.next_player) # TODO check + sel_moves = [ # top move, or anything not too bad, or anything that makes you still ahead + move + for i, move in enumerate(ai_moves) + if i == 0 + or move["visits"] >= train_settings["balance_play_min_visits"] + and ( + move["pointsLost"] < train_settings["balance_play_randomize_score"] + or move["pointsLost"] < train_settings["balance_play_min_eval"] + and sign * move["scoreLead"] > train_settings["balance_play_target_score"] + ) + ] + aimove = Move.from_gtp(random.choice(sel_moves)["move"], player=cn.next_player) # TODO: could be weighted towards worse + else: + aimove = Move.from_gtp(ai_moves[0]["move"], player=cn.next_player) self.play(aimove) def analyze_undo(self, node, train_config): @@ -252,7 +261,7 @@ class Game: return if mode == "extra": - visits = cn.analysis['root']['visits'] + self.engine.config["visits"] + visits = cn.analysis["root"]["visits"] + self.engine.config["visits"] self.katrain.controls.set_status(f"Performing additional analysis to {visits} visits") cn.analyze(self.engine, visits=visits, priority=-1_000) return @@ -262,8 +271,8 @@ class Game: self.katrain.controls.set_status(f"Refining analysis of entire board to {visits} visits") priority = -1_000_000_000 else: # mode=='refine': - analyze_moves = [Move.from_gtp(gtp, player=cn.next_player) for gtp,_ in cn.analysis['moves'].items()] - visits = max(d['visits'] for d in cn.analysis['moves'].values()) + self.engine.config["visits_fast"] + analyze_moves = [Move.from_gtp(gtp, player=cn.next_player) for gtp, _ in cn.analysis["moves"].items()] + visits = max(d["visits"] for d in cn.analysis["moves"].values()) + self.engine.config["visits_fast"] self.katrain.controls.set_status(f"Refining analysis of candidate moves to {visits} visits") priority = -1_000 for move in analyze_moves: diff --git a/game_node.py b/game_node.py index d0f4a80..c52c2d3 100644 --- a/game_node.py +++ b/game_node.py @@ -2,7 +2,7 @@ import copy import random from typing import Dict, List, Optional -from sgf_parser import SGFNode +from sgf_parser import SGFNode, Move class GameNode(SGFNode): @@ -110,3 +110,15 @@ class GameNode(SGFNode): [{"pointsLost": self.player_sign(self.next_player) * (self.analysis["root"]["scoreLead"] - d["scoreLead"]), **d} for d in self.analysis["moves"].values()], key=lambda d: (d["order"], d["pointsLost"]), ) + + @property + def policy_ranking(self) -> Optional[List]: # return moves from highest policy value to lowest + if self.policy: + ix = 0 + moves = [] + for y in range(self.board_size - 1, -1, -1): + for x in range(self.board_size): + moves.append((Move((x, y)), self.policy[ix])) + ix += 1 + moves.append((Move(None), self.policy[ix])) + return sorted(moves, key=lambda mp: -mp[1]) diff --git a/gui/badukpan.py b/gui/badukpan.py index 731eda9..ef4fba9 100644 --- a/gui/badukpan.py +++ b/gui/badukpan.py @@ -12,7 +12,7 @@ from game import Move class BadukPanWidget(Widget): - board_color = ListProperty( [0.85, 0.68, 0.40] ) + board_color = ListProperty([0.85, 0.68, 0.40]) def __init__(self, **kwargs): super(BadukPanWidget, self).__init__(**kwargs) @@ -180,10 +180,9 @@ class BadukPanWidget(Widget): Rectangle(pos=(self.gridpos_x[x] - rsz / 2, self.gridpos_y[y] - rsz / 2), size=(rsz, rsz)) ix = ix + 1 - policy = current_node.policy - if not policy and current_node.parent and current_node.parent.policy and set(katrain.controls.ai_auto.active_map.values())=={True}: - policy = current_node.parent.policy # in the case of AI self-play we allow the policy to be one step out of date + if not policy and current_node.parent and current_node.parent.policy and set(katrain.controls.ai_auto.active_map.values()) == {True}: + policy = current_node.parent.policy # in the case of AI self-play we allow the policy to be one step out of date pass_btn = katrain.board_controls.pass_btn pass_btn.canvas.after.clear() if katrain.controls.policy.active and policy and not katrain.controls.ownership.active: diff --git a/gui/controls.py b/gui/controls.py index 6ab3060..0054204 100644 --- a/gui/controls.py +++ b/gui/controls.py @@ -11,7 +11,7 @@ class Controls(BoxLayout): def set_status(self, msg, at_node=None): self.status = msg - self.status_node = at_node or self.parent.game.current_node + self.status_node = at_node or self.parent.game and self.parent.game.current_node self.info.text = msg self.update_evaluation() @@ -38,6 +38,9 @@ class Controls(BoxLayout): else: self.points_lost.text = f"..." + def player_mode(self, player): + return self.player_mode_groups[player].value + def unlock(self): if self.ai_lock.active: self.ai_lock.checkbox.trigger_action(duration=0) @@ -50,65 +53,36 @@ class Controls(BoxLayout): # handles showing completed analysis and score graph def update_evaluation(self): katrain = self.parent - current_node = katrain.game.current_node - move = current_node.single_move - current_player_is_human_or_both_robots = not current_node.player or not self.ai_auto.active(current_node.player) or self.ai_auto.active(current_node.next_player) + current_node = katrain.game and katrain.game.current_node info = "" - if current_node is self.status_node: + if current_node is self.status_node or (self.status is not None and self.status_node is None and current_node.is_root): # startup errors on root info += self.status + "\n" else: self.status_node = None - if current_player_is_human_or_both_robots and not current_node.is_root and move: - info += current_node.comment(eval=True, hints=self.hints.active(move.player)) - if current_player_is_human_or_both_robots: - self.show_evaluation_stats(current_node) + if current_node: + move = current_node.single_move + current_player_is_human_or_both_robots = not current_node.player or not self.ai_auto.active(current_node.player) or self.ai_auto.active(current_node.next_player) + if current_player_is_human_or_both_robots and not current_node.is_root and move: + info += current_node.comment(eval=True, hints=self.hints.active(move.player)) + if current_player_is_human_or_both_robots: + self.show_evaluation_stats(current_node) + + game_node = katrain.game.current_node + scores = [n.score for n in game_node.nodes_from_root] + # TODO: like redo, what is the node to redo / should we append? cache? + self.graph.canvas.clear() + with self.graph.canvas: + pt = [] + nnscores = [s for s in scores if s is not None] + [-5, 5] + scale = max(max(*nnscores), -min(*nnscores)) * 1.05 + xscale = self.graph.width * 0.9 / max(len(scores), 20) + ls = 0 + for i, s in enumerate(scores): + ls = s or ls + pt.extend([self.graph.pos[0] + 0.05 * self.graph.width + i * xscale, self.graph.pos[1] + self.graph.height / 2 * (1 + ls / scale)]) + Color(0, 0, 0) + Line(points=pt, width=1.0) # just set points? self.info.text = info - - game_node = katrain.game.current_node - scores = [n.score for n in game_node.nodes_from_root] - # TODO: like redo, what is the node to redo / should we append? cache? - self.graph.canvas.clear() - with self.graph.canvas: - pt = [] - nnscores = [s for s in scores if s is not None] + [-5, 5] - scale = max(max(*nnscores), -min(*nnscores)) * 1.05 - xscale = self.graph.width * 0.9 / max(len(scores), 20) - ls = 0 - for i, s in enumerate(scores): - ls = s or ls - pt.extend([self.graph.pos[0] + 0.05 * self.graph.width + i * xscale, self.graph.pos[1] + self.graph.height / 2 * (1 + ls / scale)]) - Color(0, 0, 0) - Line(points=pt, width=1.0) # just set points? - - if False: # TODO: UNDO AND AI MOVE - if current_node.analysis_ready and current_node.parent and current_node.parent.analysis_ready and not current_node.children and not current_node.x_comment.get("undo"): - # handle automatic undo - if self.auto_undo.active(move.player) and not self.ai_auto.active(move.player) and not current_node.auto_undid: - ts = self.train_settings - # TODO: is this overly generous wrt low visit outdated evaluations? - evaluation = current_node.evaluation if current_node.evaluation is not None else 1 # assume move is fine if temperature is negative - move_eval = max(evaluation, current_node.outdated_evaluation or 0) - points_lost = (current_node.parent or current_node).temperature_stats[2] * (1 - move_eval) - if move_eval < ts["undo_eval_threshold"] and points_lost >= ts["undo_point_threshold"]: - if self.num_undos(current_node) == 0: - current_node.x_comment["undid"] = f"Move was below threshold, but no undo granted (probability is {ts['num_undo_prompts']:.0%}).\n" - self.update_evaluation() - else: - current_node.auto_undid = True - self.parent.game.undo() - if len(current_node.parent.children) >= ts["num_undo_prompts"] + 1: - best_move = sorted([m for m in current_node.parent.children], key=lambda m: -(m.evaluation_info[0] or 0))[0] - best_move.x_comment["undo_autoplay"] = f"Automatically played as best option after max. {ts['num_undo_prompts']} undo(s).\n" - self.parent.game.play(best_move) - self.update_evaluation() - return - # ai player doesn't technically need parent ready, but don't want to override waiting for undo - current_node = self.parent.game.current_node # this effectively checks undo didn't just happen - if self.ai_auto.active(move.opponent) and not self.parent.game.game_ended: - if current_node.children: - self.info.text = "AI paused since moves were undone. Press 'AI Move' or choose a move for the AI to continue playing." - else: - self._do_aimove() diff --git a/gui/kivyutils.py b/gui/kivyutils.py index 1670dfe..9fbdcbf 100644 --- a/gui/kivyutils.py +++ b/gui/kivyutils.py @@ -1,20 +1,21 @@ +import random + +from kivy.clock import Clock from kivy.core.text import Label as CoreLabel from kivy.graphics import * -from kivy.properties import BooleanProperty, StringProperty, NumericProperty +from kivy.properties import BooleanProperty, StringProperty, NumericProperty, ListProperty, ObjectProperty +from kivy.uix.behaviors import ToggleButtonBehavior from kivy.uix.boxlayout import BoxLayout import re from kivy.uix.button import Button from kivy.uix.checkbox import CheckBox +from kivy.uix.gridlayout import GridLayout from kivy.uix.label import Label from kivy.uix.spinner import Spinner from kivy.uix.textinput import TextInput -class StyledButton(Button): - pass - - class CheckBoxHint(BoxLayout): __events__ = ("on_active",) @@ -30,6 +31,65 @@ class DarkLabel(Label): pass +class StyledButton(Button): + pass + + +class StyledToggleButton(StyledButton, ToggleButtonBehavior): + value = StringProperty("") + + +class ToggleButtonContainer(GridLayout): + __events__ = ("on_selection",) + + options = ListProperty([]) + labels = ListProperty([]) + selected = StringProperty("") + group = StringProperty(None) + autosize = BooleanProperty(True) + button_class = ObjectProperty(StyledToggleButton) + margin = ListProperty([1, 1, 0, 0]) + + def __init__(self, **kwargs): + super().__init__(**kwargs) + Clock.schedule_once(self._build, 0) + + def on_selection(self, *args): + pass + + def _build(self, _dt): + self.rows = 1 + self.cols = len(self.options) + self.group = self.group or str(random.random()) + if not self.selected and self.options: + self.selected = self.options[0] + if len(self.labels) < len(self.options): + self.labels += self.options[len(self.labels) + 1 :] + + def state_handler(btn,*args): + self.dispatch("on_selection") + btn.state = 'down' # no toggle + + for i, opt in enumerate(self.options): + state = "down" if opt == self.selected else "normal" + self.add_widget(self.button_class(group=self.group, text=self.labels[i], value=opt, state=state, + margin=self.margin,on_press=state_handler)) + Clock.schedule_once(self._size, 0) + + def _size(self, _dt): + if self.autosize: + for tb in self.children: + tb.size_hint = (tb.texture_size[0] + 3, 1) + + @property + def value(self): + for tb in self.children: + if tb.state == "down": + return tb.value + if self.options: + return self.options[0] + + class BaseCircleWithText(DarkLabel): radius = NumericProperty(0.48) diff --git a/gui.kv b/katrain.kv similarity index 82% rename from gui.kv rename to katrain.kv index da1248e..03fdd30 100644 --- a/gui.kv +++ b/katrain.kv @@ -1,6 +1,8 @@ #:kivy 1.11.0 #:import ew kivy.uix.effectwidget +# import utils? + # margin left bottom right top : text_color: 0.95,0.95,0.95,1 @@ -24,7 +26,10 @@ pos: (self.pos[0]+self.margin[0],self.pos[1]+self.margin[1]) radius: root.radius or (0.0,) -: + +: + +: : bold: True @@ -32,7 +37,10 @@ radius: (self.size[1]/3,self.size[1]/3,0,0) - + + + +: button_color: 0.71, 0.78, 0.81, 1 icon_margin: 0.15 * self.size[0] icon: 'missing.png' @@ -43,7 +51,7 @@ pos: [root.pos[i] + (root.size[i] - root.icon_size)/2 for i in [0,1]] if root.icon_size else [0,0] source: root.icon - +: background_normal: '' background_color: (0,0,0,0) margin: (1,1,1,1) @@ -61,17 +69,17 @@ : bold: True - +: color: (0.05,0.05,0.05,1) - +: color: (0.95,0.95,0.95,1) : halign: 'center' valign: 'center' - +: orientation: 'vertical' checkbox: checkbox label: label @@ -90,7 +98,7 @@ on_active: root.dispatch('on_active') active: root.default_active - +: black: black white: white label: label @@ -114,7 +122,7 @@ active: root.default_active on_active: root.dispatch('on_active') - +: orientation: 'horizontal' text: '' label: '' @@ -130,7 +138,7 @@ id: value bold: True - +: text: '' color: 0.95,0.95,0.95,1 font_size: min(self.height,self.width) * 0.6 / (0.5 + 0.5 * len(self.text)) @@ -138,7 +146,7 @@ valign: 'center' bold: True - +: canvas.before: Color: rgba: 0.05,0.05,0.05,1 @@ -146,7 +154,7 @@ pos: self.pos[0] + self.width/2 - min(self.height,self.width) * root.radius, self.pos[1] + self.height/2 - min(self.height,self.width) * root.radius size: min(self.height,self.width) * 2 * root.radius, min(self.height,self.width) * 2* root.radius - +: color: 0.05,0.05,0.05,1 outline: True canvas.before: @@ -228,7 +236,7 @@ size_hint: 0.25, 1 - +: orientation: 'vertical' play_tab_button: play_tab_button analyze_tab_button: analyze_tab_button @@ -246,6 +254,7 @@ ai_lock: ai_lock ai_move: ai_move auto_undo: auto_undo + player_mode_groups: {'B':B_player_mode,'W':W_player_mode} graph: graph katrain: self.parent canvas.before: @@ -345,39 +354,6 @@ size_hint: 1,1 opacity: 1 id: play_tab - GridLayout: - cols: 5 - rows: 1 - size_hint: 1, 0.1 - BoxLayout: - orientation: 'vertical' - size_hint: 0.1, 0.5 - Label: - size_hint: 1, 0.2 - BlackCircleWithText: - text: 'B' - WhiteCircleWithText: - text: 'W' - BWCheckBoxHint: - text: 'ai' - id: ai_auto - default_active: False - size_hint: 0.25, 1 - BWCheckBoxHint: - size_hint: 0.2, 0.5 - id: auto_undo - text: 'undo' - on_active: root.katrain.update_state() - CheckBoxHint: - size_hint: 0.25, 1 - text: 'fast' - id: ai_fast - default_active: True - CheckBoxHint: - size_hint: 0.25, 1 - text: 'balance\nscore' - id: ai_balance - default_active: False GridLayout: cols: 4 rows: 1 @@ -393,60 +369,30 @@ id: ai_lock on_active: self.checkbox.disabled = analyze_tab_button.disabled = ai_auto.white = ai_auto.black = ai_move.disabled = True GridLayout: - cols: 6 + cols: 2 rows: 2 - size_hint: 1,0.1 + size_hint: 1,0.125 BlackCircleWithText: text: 'B' - size_hint: 0.4, 1 - StyledToggleButton: - group: 'black' - text: 'Human' - state: 'down' - size_hint: 1, 1 - StyledToggleButton: - group: 'black' - text: ' AI\nAssist' - size_hint: 1, 1 - StyledToggleButton: - group: 'black' - text: 'AI' - size_hint: 0.5, 1 - StyledToggleButton: - group: 'black' - text: ' AI\nBalance' - size_hint: 1, 1 - StyledToggleButton: - group: 'black' - text: ' AI\nPolicy' - size_hint: 1, 1 + size_hint: 0.1, 1 + ToggleButtonContainer: + size_hint: 0.9, 1 + id: B_player_mode + options: ['human','human+undo','ai','ai+balance','ai+policy'] + labels: ['Human', ' AI\nAssist','AI',' AI\nBalance',' AI\nPolicy'] + on_selection: root.katrain.update_state() WhiteCircleWithText: text: 'W' - size_hint: 0.4, 1 - StyledToggleButton: - group: 'white' - text: 'Human' - state: 'down' - size_hint: 1, 1 - StyledToggleButton: - group: 'white' - text: ' AI\nAssist' - size_hint: 1, 1 - StyledToggleButton: - group: 'white' - text: 'AI' - size_hint: 0.5, 1 - StyledToggleButton: - group: 'white' - text: ' AI\nBalance' - size_hint: 1, 1 - StyledToggleButton: - group: 'white' - text: ' AI\nPolicy' - size_hint: 1, 1 + size_hint: 0.1, 1 + ToggleButtonContainer: + size_hint: 0.9, 1 + id: W_player_mode + options: ['human','human+undo','ai','ai+balance','ai+policy'] + labels: ['Human', ' AI\nAssist','AI',' AI\nBalance',' AI\nPolicy'] + on_selection: root.katrain.update_state() LargeLabel: text: 'free real estate' - size_hint: 1,0.1 + size_hint: 1,0.2 CensorableLabel: id: points_lost size_hint: 1, 0.03 @@ -456,6 +402,39 @@ id: info size_hint: 1, 0.25 valign: 'middle' + GridLayout: + cols: 5 + rows: 1 + size_hint: 1, 0.1 + BoxLayout: + orientation: 'vertical' + size_hint: 0.1, 0.5 + Label: + size_hint: 1, 0.2 + BlackCircleWithText: + text: 'B' + WhiteCircleWithText: + text: 'W' + BWCheckBoxHint: + text: 'ai' + id: ai_auto + default_active: False + size_hint: 0.25, 1 + BWCheckBoxHint: + size_hint: 0.2, 0.5 + id: auto_undo + text: 'undo' + on_active: root.katrain.update_state() + CheckBoxHint: + size_hint: 0.25, 1 + text: 'balance\nscore' + id: ai_balance + default_active: False + CheckBoxHint: + size_hint: 0.25, 1 + text: 'fast' + id: ai_fast + default_active: True BoxLayout: orientation: 'horizontal' size_hint: 1, None @@ -545,7 +524,7 @@ color: (0.95, 0.95, 0.95) size_hint: 0.05,1 - +: orientation: 'vertical' rules_spinner: rules_spinner BoxLayout: diff --git a/katrain.py b/katrain.py index ba396ba..2d92e5b 100644 --- a/katrain.py +++ b/katrain.py @@ -40,7 +40,7 @@ class KaTrainGui(BoxLayout): def log(self, message, level=OUTPUT_INFO): if level == OUTPUT_ERROR: - self.controls.set_status(f"ERROR: {message}", self.game.current_node) + self.controls.set_status(f"ERROR: {message}") print(f"ERROR: {message}") elif self.debug_level >= level: print(message) @@ -81,10 +81,11 @@ class KaTrainGui(BoxLayout): def update_state(self, redraw_board=False): # AI and Trainer/auto-undo handlers cn = self.game.current_node - auto_undo = cn.player and self.controls.auto_undo.active(cn.player) - if auto_undo and cn.analysis_ready: + auto_undo = cn.player and "undo" in self.controls.player_mode(cn.player) + if auto_undo and cn.analysis_ready: self.game.analyze_undo(cn, self.config("trainer")) # not via message loop - if cn.analysis_ready and self.controls.ai_auto.active(cn.next_player) and not cn.children and not self.game.game_ended and not (auto_undo and cn.auto_undo is None): + + if cn.analysis_ready and "ai" in self.controls.player_mode(cn.next_player) and not cn.children and not self.game.game_ended and not (auto_undo and cn.auto_undo is None): self("ai-move", cn) # cn mismatch stops this if undo fired # Handle prisoners and next player display @@ -264,7 +265,7 @@ class KaTrainApp(App): self.gui.start() def on_request_close(self, *args): - if getattr(self,'gui',None) and self.gui.engine: + if getattr(self, "gui", None) and self.gui.engine: self.gui.engine.shutdown() def signal_handler(self, signal, frame): @@ -284,8 +285,8 @@ class KaTrainApp(App): if __name__ == "__main__": - with open("gui.kv", encoding="utf-8") as f: # avoid windows using another encoding - Builder.load_string(f.read()) + # with open("katrain.kv", encoding="utf-8") as f: # avoid windows using another encoding + # Builder.load_string(f.read()) app = KaTrainApp() signal.signal(signal.SIGINT, app.signal_handler) try: diff --git a/sgf_parser.py b/sgf_parser.py index b339e50..554716d 100644 --- a/sgf_parser.py +++ b/sgf_parser.py @@ -131,18 +131,18 @@ class SGFNode: # root properties available on any node @property - def board_size(self) -> Union[int,Tuple]: - size = str(self.root.get_first("SZ", '19')) - if ':' in size: - return tuple(map(size.split(':'),int)) + def board_size(self) -> Union[int, Tuple]: + size = str(self.root.get_first("SZ", "19")) + if ":" in size: + return tuple(map(size.split(":"), int)) return int(size) @property - def board_size_xy(self) -> Tuple[int,int]: + def board_size_xy(self) -> Tuple[int, int]: x, y = self.board_size if not y: - y=x - return x,y + y = x + return x, y @property def komi(self) -> float: