play mode

This commit is contained in:
Sander Land committed 2020-04-20 23:09:47 +02:00
1 parent 12b1b1a212
commit d6141d91d6
8 files changed
+227 -193

No files matched your search

+30 -21
View File
@@ -7,6 +7,7 @@ from typing import List
from kivy.clock import Clock
from constants import OUTPUT_DEBUG
from game_node import GameNode
from sgf_parser import SGF, Move
@@ -195,26 +196,34 @@ class Game:
return f"SGF with analysis written to {file_name}"
def ai_move(self, train_settings):
while not self.current_node.analysis_ready:
self.katrain.controls.set_status("Thinking...")
cn = self.current_node
while not cn.analysis_ready:
self.katrain.controls.set_status("Thinking...") # TODO: non blocking somehow?
time.sleep(0.05)
# select move
ai_moves = self.current_node.candidate_moves
pos_moves = [
[d["move"], d["scoreLead"], d["pointsLost"]] for i, d in enumerate(ai_moves) if i == 0 or int(d["visits"]) >= train_settings["balance_play_min_visits"]
] # TODO: lcb based ?
sel_moves = pos_moves[:1]
# don't play suicidal to balance score - pass when it's best
if self.katrain.controls.ai_balance.active and pos_moves[0][0] != "pass": # TODO: settings where they belong?
sel_moves = [
(move, score, points_lost)
for move, score, points_lost in pos_moves
if points_lost < train_settings["balance_play_randomize_score"]
or points_lost < train_settings["balance_play_min_eval"]
and -self.current_node.move.player_sign * score > self.config["balance_play_target_score"]
] or sel_moves
aimove = Move.from_gtp(random.choice(sel_moves)[0], player=self.next_player)
ai_moves = cn.candidate_moves
mode = self.katrain.controls.player_mode(cn.next_player)
if "policy" in mode and cn.policy:
policy_moves = cn.policy_ranking
self.katrain.log(f"Top 5 policy moves are: {policy_moves[:5]}", OUTPUT_DEBUG)
aimove = policy_moves[0][0]
elif "balance" in mode and ai_moves[0]['move'] != "pass": # don't play suicidal to balance score - pass when it's best
sign = cn.player_sign(cn.next_player) # TODO check
sel_moves = [ # top move, or anything not too bad, or anything that makes you still ahead
move
for i, move in enumerate(ai_moves)
if i == 0
or move["visits"] >= train_settings["balance_play_min_visits"]
and (
move["pointsLost"] < train_settings["balance_play_randomize_score"]
or move["pointsLost"] < train_settings["balance_play_min_eval"]
and sign * move["scoreLead"] > train_settings["balance_play_target_score"]
)
]
aimove = Move.from_gtp(random.choice(sel_moves)["move"], player=cn.next_player) # TODO: could be weighted towards worse
else:
aimove = Move.from_gtp(ai_moves[0]["move"], player=cn.next_player)
self.play(aimove)
def analyze_undo(self, node, train_config):
@@ -252,7 +261,7 @@ class Game:
return
if mode == "extra":
visits = cn.analysis['root']['visits'] + self.engine.config["visits"]
visits = cn.analysis["root"]["visits"] + self.engine.config["visits"]
self.katrain.controls.set_status(f"Performing additional analysis to {visits} visits")
cn.analyze(self.engine, visits=visits, priority=-1_000)
return
@@ -262,8 +271,8 @@ class Game:
self.katrain.controls.set_status(f"Refining analysis of entire board to {visits} visits")
priority = -1_000_000_000
else: # mode=='refine':
analyze_moves = [Move.from_gtp(gtp, player=cn.next_player) for gtp,_ in cn.analysis['moves'].items()]
visits = max(d['visits'] for d in cn.analysis['moves'].values()) + self.engine.config["visits_fast"]
analyze_moves = [Move.from_gtp(gtp, player=cn.next_player) for gtp, _ in cn.analysis["moves"].items()]
visits = max(d["visits"] for d in cn.analysis["moves"].values()) + self.engine.config["visits_fast"]
self.katrain.controls.set_status(f"Refining analysis of candidate moves to {visits} visits")
priority = -1_000
for move in analyze_moves:
+13 -1
View File
@@ -2,7 +2,7 @@ import copy
import random
from typing import Dict, List, Optional
from sgf_parser import SGFNode
from sgf_parser import SGFNode, Move
class GameNode(SGFNode):
@@ -110,3 +110,15 @@ class GameNode(SGFNode):
[{"pointsLost": self.player_sign(self.next_player) * (self.analysis["root"]["scoreLead"] - d["scoreLead"]), **d} for d in self.analysis["moves"].values()],
key=lambda d: (d["order"], d["pointsLost"]),
)
@property
def policy_ranking(self) -> Optional[List]: # return moves from highest policy value to lowest
if self.policy:
ix = 0
moves = []
for y in range(self.board_size - 1, -1, -1):
for x in range(self.board_size):
moves.append((Move((x, y)), self.policy[ix]))
ix += 1
moves.append((Move(None), self.policy[ix]))
return sorted(moves, key=lambda mp: -mp[1])
+3 -4
View File
@@ -12,7 +12,7 @@ from game import Move
class BadukPanWidget(Widget):
board_color = ListProperty( [0.85, 0.68, 0.40] )
board_color = ListProperty([0.85, 0.68, 0.40])
def __init__(self, **kwargs):
super(BadukPanWidget, self).__init__(**kwargs)
@@ -180,10 +180,9 @@ class BadukPanWidget(Widget):
Rectangle(pos=(self.gridpos_x[x] - rsz / 2, self.gridpos_y[y] - rsz / 2), size=(rsz, rsz))
ix = ix + 1
policy = current_node.policy
if not policy and current_node.parent and current_node.parent.policy and set(katrain.controls.ai_auto.active_map.values())=={True}:
policy = current_node.parent.policy # in the case of AI self-play we allow the policy to be one step out of date
if not policy and current_node.parent and current_node.parent.policy and set(katrain.controls.ai_auto.active_map.values()) == {True}:
policy = current_node.parent.policy # in the case of AI self-play we allow the policy to be one step out of date
pass_btn = katrain.board_controls.pass_btn
pass_btn.canvas.after.clear()
if katrain.controls.policy.active and policy and not katrain.controls.ownership.active:
+29 -55
View File
@@ -11,7 +11,7 @@ class Controls(BoxLayout):
def set_status(self, msg, at_node=None):
self.status = msg
self.status_node = at_node or self.parent.game.current_node
self.status_node = at_node or self.parent.game and self.parent.game.current_node
self.info.text = msg
self.update_evaluation()
@@ -38,6 +38,9 @@ class Controls(BoxLayout):
else:
self.points_lost.text = f"..."
def player_mode(self, player):
return self.player_mode_groups[player].value
def unlock(self):
if self.ai_lock.active:
self.ai_lock.checkbox.trigger_action(duration=0)
@@ -50,65 +53,36 @@ class Controls(BoxLayout):
# handles showing completed analysis and score graph
def update_evaluation(self):
katrain = self.parent
current_node = katrain.game.current_node
move = current_node.single_move
current_player_is_human_or_both_robots = not current_node.player or not self.ai_auto.active(current_node.player) or self.ai_auto.active(current_node.next_player)
current_node = katrain.game and katrain.game.current_node
info = ""
if current_node is self.status_node:
if current_node is self.status_node or (self.status is not None and self.status_node is None and current_node.is_root): # startup errors on root
info += self.status + "\n"
else:
self.status_node = None
if current_player_is_human_or_both_robots and not current_node.is_root and move:
info += current_node.comment(eval=True, hints=self.hints.active(move.player))
if current_player_is_human_or_both_robots:
self.show_evaluation_stats(current_node)
if current_node:
move = current_node.single_move
current_player_is_human_or_both_robots = not current_node.player or not self.ai_auto.active(current_node.player) or self.ai_auto.active(current_node.next_player)
if current_player_is_human_or_both_robots and not current_node.is_root and move:
info += current_node.comment(eval=True, hints=self.hints.active(move.player))
if current_player_is_human_or_both_robots:
self.show_evaluation_stats(current_node)
game_node = katrain.game.current_node
scores = [n.score for n in game_node.nodes_from_root]
# TODO: like redo, what is the node to redo / should we append? cache?
self.graph.canvas.clear()
with self.graph.canvas:
pt = []
nnscores = [s for s in scores if s is not None] + [-5, 5]
scale = max(max(*nnscores), -min(*nnscores)) * 1.05
xscale = self.graph.width * 0.9 / max(len(scores), 20)
ls = 0
for i, s in enumerate(scores):
ls = s or ls
pt.extend([self.graph.pos[0] + 0.05 * self.graph.width + i * xscale, self.graph.pos[1] + self.graph.height / 2 * (1 + ls / scale)])
Color(0, 0, 0)
Line(points=pt, width=1.0) # just set points?
self.info.text = info
game_node = katrain.game.current_node
scores = [n.score for n in game_node.nodes_from_root]
# TODO: like redo, what is the node to redo / should we append? cache?
self.graph.canvas.clear()
with self.graph.canvas:
pt = []
nnscores = [s for s in scores if s is not None] + [-5, 5]
scale = max(max(*nnscores), -min(*nnscores)) * 1.05
xscale = self.graph.width * 0.9 / max(len(scores), 20)
ls = 0
for i, s in enumerate(scores):
ls = s or ls
pt.extend([self.graph.pos[0] + 0.05 * self.graph.width + i * xscale, self.graph.pos[1] + self.graph.height / 2 * (1 + ls / scale)])
Color(0, 0, 0)
Line(points=pt, width=1.0) # just set points?
if False: # TODO: UNDO AND AI MOVE
if current_node.analysis_ready and current_node.parent and current_node.parent.analysis_ready and not current_node.children and not current_node.x_comment.get("undo"):
# handle automatic undo
if self.auto_undo.active(move.player) and not self.ai_auto.active(move.player) and not current_node.auto_undid:
ts = self.train_settings
# TODO: is this overly generous wrt low visit outdated evaluations?
evaluation = current_node.evaluation if current_node.evaluation is not None else 1 # assume move is fine if temperature is negative
move_eval = max(evaluation, current_node.outdated_evaluation or 0)
points_lost = (current_node.parent or current_node).temperature_stats[2] * (1 - move_eval)
if move_eval < ts["undo_eval_threshold"] and points_lost >= ts["undo_point_threshold"]:
if self.num_undos(current_node) == 0:
current_node.x_comment["undid"] = f"Move was below threshold, but no undo granted (probability is {ts['num_undo_prompts']:.0%}).\n"
self.update_evaluation()
else:
current_node.auto_undid = True
self.parent.game.undo()
if len(current_node.parent.children) >= ts["num_undo_prompts"] + 1:
best_move = sorted([m for m in current_node.parent.children], key=lambda m: -(m.evaluation_info[0] or 0))[0]
best_move.x_comment["undo_autoplay"] = f"Automatically played as best option after max. {ts['num_undo_prompts']} undo(s).\n"
self.parent.game.play(best_move)
self.update_evaluation()
return
# ai player doesn't technically need parent ready, but don't want to override waiting for undo
current_node = self.parent.game.current_node # this effectively checks undo didn't just happen
if self.ai_auto.active(move.opponent) and not self.parent.game.game_ended:
if current_node.children:
self.info.text = "AI paused since moves were undone. Press 'AI Move' or choose a move for the AI to continue playing."
else:
self._do_aimove()
+65 -5
View File
@@ -1,20 +1,21 @@
import random
from kivy.clock import Clock
from kivy.core.text import Label as CoreLabel
from kivy.graphics import *
from kivy.properties import BooleanProperty, StringProperty, NumericProperty
from kivy.properties import BooleanProperty, StringProperty, NumericProperty, ListProperty, ObjectProperty
from kivy.uix.behaviors import ToggleButtonBehavior
from kivy.uix.boxlayout import BoxLayout
import re
from kivy.uix.button import Button
from kivy.uix.checkbox import CheckBox
from kivy.uix.gridlayout import GridLayout
from kivy.uix.label import Label
from kivy.uix.spinner import Spinner
from kivy.uix.textinput import TextInput
class StyledButton(Button):
pass
class CheckBoxHint(BoxLayout):
__events__ = ("on_active",)
@@ -30,6 +31,65 @@ class DarkLabel(Label):
pass
class StyledButton(Button):
pass
class StyledToggleButton(StyledButton, ToggleButtonBehavior):
value = StringProperty("")
class ToggleButtonContainer(GridLayout):
__events__ = ("on_selection",)
options = ListProperty([])
labels = ListProperty([])
selected = StringProperty("")
group = StringProperty(None)
autosize = BooleanProperty(True)
button_class = ObjectProperty(StyledToggleButton)
margin = ListProperty([1, 1, 0, 0])
def __init__(self, **kwargs):
super().__init__(**kwargs)
Clock.schedule_once(self._build, 0)
def on_selection(self, *args):
pass
def _build(self, _dt):
self.rows = 1
self.cols = len(self.options)
self.group = self.group or str(random.random())
if not self.selected and self.options:
self.selected = self.options[0]
if len(self.labels) < len(self.options):
self.labels += self.options[len(self.labels) + 1 :]
def state_handler(btn,*args):
self.dispatch("on_selection")
btn.state = 'down' # no toggle
for i, opt in enumerate(self.options):
state = "down" if opt == self.selected else "normal"
self.add_widget(self.button_class(group=self.group, text=self.labels[i], value=opt, state=state,
margin=self.margin,on_press=state_handler))
Clock.schedule_once(self._size, 0)
def _size(self, _dt):
if self.autosize:
for tb in self.children:
tb.size_hint = (tb.texture_size[0] + 3, 1)
@property
def value(self):
for tb in self.children:
if tb.state == "down":
return tb.value
if self.options:
return self.options[0]
class BaseCircleWithText(DarkLabel):
radius = NumericProperty(0.48)
+72 -93
View File
@@ -1,6 +1,8 @@
#:kivy 1.11.0
#:import ew kivy.uix.effectwidget
# import utils?
# margin left bottom right top
<StyledButton>:
text_color: 0.95,0.95,0.95,1
@@ -24,7 +26,10 @@
pos: (self.pos[0]+self.margin[0],self.pos[1]+self.margin[1])
radius: root.radius or (0.0,)
<StyledToggleButton@StyledButton+ToggleButtonBehavior>:
<StyledToggleButton>:
<ToggleButtonContainer>:
<StyledTabButton@StyledToggleButton>:
bold: True
@@ -32,7 +37,10 @@
radius: (self.size[1]/3,self.size[1]/3,0,0)
<StyledIconButton@StyledButton>
<StyledIconButton@StyledButton>:
button_color: 0.71, 0.78, 0.81, 1
icon_margin: 0.15 * self.size[0]
icon: 'missing.png'
@@ -43,7 +51,7 @@
pos: [root.pos[i] + (root.size[i] - root.icon_size)/2 for i in [0,1]] if root.icon_size else [0,0]
source: root.icon
<TransparentIconButton@Button>
<TransparentIconButton@Button>:
background_normal: ''
background_color: (0,0,0,0)
margin: (1,1,1,1)
@@ -61,17 +69,17 @@
<LargeLabel@DarkLabel>:
bold: True
<CheckBox>
<CheckBox>:
color: (0.05,0.05,0.05,1)
<LabelledCheckBox>
<LabelledCheckBox>:
color: (0.95,0.95,0.95,1)
<CheckBoxHintLabel@ButtonBehavior+DarkLabel>:
halign: 'center'
valign: 'center'
<CheckBoxHint>
<CheckBoxHint>:
orientation: 'vertical'
checkbox: checkbox
label: label
@@ -90,7 +98,7 @@
on_active: root.dispatch('on_active')
active: root.default_active
<BWCheckBoxHint>
<BWCheckBoxHint>:
black: black
white: white
label: label
@@ -114,7 +122,7 @@
active: root.default_active
on_active: root.dispatch('on_active')
<CensorableLabel>
<CensorableLabel>:
orientation: 'horizontal'
text: ''
label: ''
@@ -130,7 +138,7 @@
id: value
bold: True
<BaseCircleWithText>
<BaseCircleWithText>:
text: ''
color: 0.95,0.95,0.95,1
font_size: min(self.height,self.width) * 0.6 / (0.5 + 0.5 * len(self.text))
@@ -138,7 +146,7 @@
valign: 'center'
bold: True
<BlackCircleWithText@BaseCircleWithText>
<BlackCircleWithText@BaseCircleWithText>:
canvas.before:
Color:
rgba: 0.05,0.05,0.05,1
@@ -146,7 +154,7 @@
pos: self.pos[0] + self.width/2 - min(self.height,self.width) * root.radius, self.pos[1] + self.height/2 - min(self.height,self.width) * root.radius
size: min(self.height,self.width) * 2 * root.radius, min(self.height,self.width) * 2* root.radius
<WhiteCircleWithText@BaseCircleWithText>
<WhiteCircleWithText@BaseCircleWithText>:
color: 0.05,0.05,0.05,1
outline: True
canvas.before:
@@ -228,7 +236,7 @@
size_hint: 0.25, 1
<Controls>
<Controls>:
orientation: 'vertical'
play_tab_button: play_tab_button
analyze_tab_button: analyze_tab_button
@@ -246,6 +254,7 @@
ai_lock: ai_lock
ai_move: ai_move
auto_undo: auto_undo
player_mode_groups: {'B':B_player_mode,'W':W_player_mode}
graph: graph
katrain: self.parent
canvas.before:
@@ -345,39 +354,6 @@
size_hint: 1,1
opacity: 1
id: play_tab
GridLayout:
cols: 5
rows: 1
size_hint: 1, 0.1
BoxLayout:
orientation: 'vertical'
size_hint: 0.1, 0.5
Label:
size_hint: 1, 0.2
BlackCircleWithText:
text: 'B'
WhiteCircleWithText:
text: 'W'
BWCheckBoxHint:
text: 'ai'
id: ai_auto
default_active: False
size_hint: 0.25, 1
BWCheckBoxHint:
size_hint: 0.2, 0.5
id: auto_undo
text: 'undo'
on_active: root.katrain.update_state()
CheckBoxHint:
size_hint: 0.25, 1
text: 'fast'
id: ai_fast
default_active: True
CheckBoxHint:
size_hint: 0.25, 1
text: 'balance\nscore'
id: ai_balance
default_active: False
GridLayout:
cols: 4
rows: 1
@@ -393,60 +369,30 @@
id: ai_lock
on_active: self.checkbox.disabled = analyze_tab_button.disabled = ai_auto.white = ai_auto.black = ai_move.disabled = True
GridLayout:
cols: 6
cols: 2
rows: 2
size_hint: 1,0.1
size_hint: 1,0.125
BlackCircleWithText:
text: 'B'
size_hint: 0.4, 1
StyledToggleButton:
group: 'black'
text: 'Human'
state: 'down'
size_hint: 1, 1
StyledToggleButton:
group: 'black'
text: ' AI\nAssist'
size_hint: 1, 1
StyledToggleButton:
group: 'black'
text: 'AI'
size_hint: 0.5, 1
StyledToggleButton:
group: 'black'
text: ' AI\nBalance'
size_hint: 1, 1
StyledToggleButton:
group: 'black'
text: ' AI\nPolicy'
size_hint: 1, 1
size_hint: 0.1, 1
ToggleButtonContainer:
size_hint: 0.9, 1
id: B_player_mode
options: ['human','human+undo','ai','ai+balance','ai+policy']
labels: ['Human', ' AI\nAssist','AI',' AI\nBalance',' AI\nPolicy']
on_selection: root.katrain.update_state()
WhiteCircleWithText:
text: 'W'
size_hint: 0.4, 1
StyledToggleButton:
group: 'white'
text: 'Human'
state: 'down'
size_hint: 1, 1
StyledToggleButton:
group: 'white'
text: ' AI\nAssist'
size_hint: 1, 1
StyledToggleButton:
group: 'white'
text: 'AI'
size_hint: 0.5, 1
StyledToggleButton:
group: 'white'
text: ' AI\nBalance'
size_hint: 1, 1
StyledToggleButton:
group: 'white'
text: ' AI\nPolicy'
size_hint: 1, 1
size_hint: 0.1, 1
ToggleButtonContainer:
size_hint: 0.9, 1
id: W_player_mode
options: ['human','human+undo','ai','ai+balance','ai+policy']
labels: ['Human', ' AI\nAssist','AI',' AI\nBalance',' AI\nPolicy']
on_selection: root.katrain.update_state()
LargeLabel:
text: 'free real estate'
size_hint: 1,0.1
size_hint: 1,0.2
CensorableLabel:
id: points_lost
size_hint: 1, 0.03
@@ -456,6 +402,39 @@
id: info
size_hint: 1, 0.25
valign: 'middle'
GridLayout:
cols: 5
rows: 1
size_hint: 1, 0.1
BoxLayout:
orientation: 'vertical'
size_hint: 0.1, 0.5
Label:
size_hint: 1, 0.2
BlackCircleWithText:
text: 'B'
WhiteCircleWithText:
text: 'W'
BWCheckBoxHint:
text: 'ai'
id: ai_auto
default_active: False
size_hint: 0.25, 1
BWCheckBoxHint:
size_hint: 0.2, 0.5
id: auto_undo
text: 'undo'
on_active: root.katrain.update_state()
CheckBoxHint:
size_hint: 0.25, 1
text: 'balance\nscore'
id: ai_balance
default_active: False
CheckBoxHint:
size_hint: 0.25, 1
text: 'fast'
id: ai_fast
default_active: True
BoxLayout:
orientation: 'horizontal'
size_hint: 1, None
@@ -545,7 +524,7 @@
color: (0.95, 0.95, 0.95)
size_hint: 0.05,1
<NewGamePopup>
<NewGamePopup>:
orientation: 'vertical'
rules_spinner: rules_spinner
BoxLayout:
+8 -7
View File
@@ -40,7 +40,7 @@ class KaTrainGui(BoxLayout):
def log(self, message, level=OUTPUT_INFO):
if level == OUTPUT_ERROR:
self.controls.set_status(f"ERROR: {message}", self.game.current_node)
self.controls.set_status(f"ERROR: {message}")
print(f"ERROR: {message}")
elif self.debug_level >= level:
print(message)
@@ -81,10 +81,11 @@ class KaTrainGui(BoxLayout):
def update_state(self, redraw_board=False):
# AI and Trainer/auto-undo handlers
cn = self.game.current_node
auto_undo = cn.player and self.controls.auto_undo.active(cn.player)
if auto_undo and cn.analysis_ready:
auto_undo = cn.player and "undo" in self.controls.player_mode(cn.player)
if auto_undo and cn.analysis_ready:
self.game.analyze_undo(cn, self.config("trainer")) # not via message loop
if cn.analysis_ready and self.controls.ai_auto.active(cn.next_player) and not cn.children and not self.game.game_ended and not (auto_undo and cn.auto_undo is None):
if cn.analysis_ready and "ai" in self.controls.player_mode(cn.next_player) and not cn.children and not self.game.game_ended and not (auto_undo and cn.auto_undo is None):
self("ai-move", cn) # cn mismatch stops this if undo fired
# Handle prisoners and next player display
@@ -264,7 +265,7 @@ class KaTrainApp(App):
self.gui.start()
def on_request_close(self, *args):
if getattr(self,'gui',None) and self.gui.engine:
if getattr(self, "gui", None) and self.gui.engine:
self.gui.engine.shutdown()
def signal_handler(self, signal, frame):
@@ -284,8 +285,8 @@ class KaTrainApp(App):
if __name__ == "__main__":
with open("gui.kv", encoding="utf-8") as f: # avoid windows using another encoding
Builder.load_string(f.read())
# with open("katrain.kv", encoding="utf-8") as f: # avoid windows using another encoding
# Builder.load_string(f.read())
app = KaTrainApp()
signal.signal(signal.SIGINT, app.signal_handler)
try:
+7 -7
View File
@@ -131,18 +131,18 @@ class SGFNode:
# root properties available on any node
@property
def board_size(self) -> Union[int,Tuple]:
size = str(self.root.get_first("SZ", '19'))
if ':' in size:
return tuple(map(size.split(':'),int))
def board_size(self) -> Union[int, Tuple]:
size = str(self.root.get_first("SZ", "19"))
if ":" in size:
return tuple(map(size.split(":"), int))
return int(size)
@property
def board_size_xy(self) -> Tuple[int,int]:
def board_size_xy(self) -> Tuple[int, int]:
x, y = self.board_size
if not y:
y=x
return x,y
y = x
return x, y
@property
def komi(self) -> float: