This commit is contained in:
Sander Land committed 2020-04-26 19:42:01 +02:00
1 parent edd4eea5b5
commit ba0a1548ee
12 files changed
+127 -76

No files matched your search

+4
View File
@@ -1,3 +1,7 @@
Primary author and project maintainer (https://github.com/sanderland/katrain):
Sander Land
Thanks to:
'Dontbtme' for feedback and testing of v1.0.
+3 -4
View File
@@ -10,15 +10,14 @@
[x] Self-play tournaments in separate script.
[x] ai thoughts in sgf
[x] more AI modes?
[x] pol value override > 0.95 ? policy threshold override p+pick?
[/] README
[] engine status
[] policy threshold override p+pick?
[] sgf review improvements
[] sgf review improvements -- Likewise, in the 0.9 version, better alternatives to the played move were shown with squares, which was also pretty useful when using the sgf outside of Katrain. I mean, having the top move mentioned is all and good, but when you see multiple squares shown on the board as better alternatives to the move played in the game, it makes obvious how far from perfect that move actually was :D
[] Release notes
-- Likewise, in the 0.9 version, better alternatives to the played move were shown with squares, which was also pretty useful when using the sgf outside of Katrain. I mean, having the top move mentioned is all and good, but when you see multiple squares shown on the board as better alternatives to the move played in the game, it makes obvious how far from perfect that move actually was :D
[] pol value override > 0.9 ?
[] clarify score change vs score
[] pv with overlap?
[] config player to sep. row/popups?
+10 -15
View File
@@ -8,7 +8,7 @@ import numpy as np
from common import OUTPUT_INFO, var_to_grid, OUTPUT_DEBUG, OUTPUT_ERROR
from engine import EngineDiedException
from game import Move, Game, IllegalMoveException
from game import Move, Game, IllegalMoveException, GameNode
def weighted_selection_without_replacement(items: List[Tuple[float, float, int, int]], pick_n: int) -> List[Tuple[float, float, int, int]]:
@@ -25,7 +25,7 @@ def fmt_moves(moves: List[Tuple[float, Move]]):
return ", ".join(f"{mv.gtp()} ({p:.2%})" for p, mv in moves)
def ai_move(game: Game, ai_mode: str, ai_settings: Dict):
def ai_move(game: Game, ai_mode: str, ai_settings: Dict) -> Tuple[Move, GameNode]:
cn = game.current_node
while not cn.analysis_ready:
time.sleep(0.01)
@@ -69,11 +69,11 @@ def ai_move(game: Game, ai_mode: str, ai_settings: Dict):
n_moves = int(ai_settings["pick_frac"] * len(legal_policy_moves) + ai_settings["pick_n"])
if "influence" in ai_mode or "territory" in ai_mode:
if "influence" in ai_mode:
weight = lambda x, y: ai_settings["influence_weight"] ** max(0, 3 - min(size[0] - 1 - x, x, y, size[1] - 1 - y))
weight = lambda x, y: ai_settings["line_weight"] ** max(0, 3 - min(size[0] - 1 - x, x, y, size[1] - 1 - y))
else:
weight = lambda x, y: ai_settings["influence_weight"] ** max(0, min(size[0] - 1 - x, x, y, size[1] - 1 - y) - 2)
weight = lambda x, y: ai_settings["line_weight"] ** max(0, min(size[0] - 1 - x, x, y, size[1] - 1 - y) - 2)
weighted_coords = [(policy_grid[y][x] * weight(x, y), weight(x, y), x, y) for x in range(size[0]) for y in range(size[1]) if policy_grid[y][x] > 0]
ai_thoughts += f"Generated weights for {ai_mode} according to weight factor {ai_settings['influence_weight']} and distance from 4th line. "
ai_thoughts += f"Generated weights for {ai_mode} according to weight factor {ai_settings['line_weight']} and distance from 4th line. "
elif "local" in ai_mode or "tenuki" in ai_mode:
var = ai_settings["local_stddev"] ** 2
if not cn.single_move or cn.single_move.coords is None:
@@ -100,8 +100,8 @@ def ai_move(game: Game, ai_mode: str, ai_settings: Dict):
aimove = new_top[0][1]
ai_thoughts += f"Top 5 among these were {fmt_moves(new_top)} and picked top {aimove.gtp()}. "
if new_top[0][0] < pass_policy:
ai_thoughts += f"But found pass ({pass_policy:.2%} to be higher rated than {aimove.gtp()} ({new_top[0][0]:.2%}) so will pass instead."
aimove = Move(None, player=cn.next_player)
ai_thoughts += f"But found pass ({pass_policy:.2%} to be higher rated than {aimove.gtp()} ({new_top[0][0]:.2%}) so will play top policy move instead."
aimove = top_policy_move
else:
aimove = top_policy_move
ai_thoughts += f"Pick policy strategy {ai_mode} failed to find legal moves, so is playing top policy move {aimove.gtp()}."
@@ -113,12 +113,8 @@ def ai_move(game: Game, ai_mode: str, ai_settings: Dict):
move
for i, move in enumerate(candidate_ai_moves)
if i == 0
or move["visits"] >= ai_settings["balance_min_visits"]
and (
move["pointsLost"] < ai_settings["balance_random_loss"]
or move["pointsLost"] < ai_settings["balance_max_loss"]
and sign * move["scoreLead"] > ai_settings["balance_target_score"]
)
or move["visits"] >= ai_settings["min_visits"]
and (move["pointsLost"] < ai_settings["random_loss"] or move["pointsLost"] < ai_settings["max_loss"] and sign * move["scoreLead"] > ai_settings["target_score"])
]
aimove = Move.from_gtp(random.choice(sel_moves)["move"], player=cn.next_player) # TODO: could be weighted towards worse
ai_thoughts += f"Balance strategy selected moves {sel_moves} based on target score and max points lost, and randomly chose {aimove.gtp()}."
@@ -137,7 +133,6 @@ def ai_move(game: Game, ai_mode: str, ai_settings: Dict):
try:
played_node = game.play(aimove)
played_node.ai_thoughts = ai_thoughts
return aimove, played_node
except IllegalMoveException as e:
game.katrain.log(f"AI Strategy {ai_mode} generated illegal move {aimove.gtp()}: {e}", OUTPUT_ERROR)
return aimove, played_node
+13 -10
View File
@@ -35,22 +35,25 @@ ai_settings = {
"noise_strength": 0.8,
"pick_n": 10,
"pick_frac": 0.2,
"local_stddev": 10,
"influence_weight": 0.1,
"stddev": 10,
"line_weight": 0.1,
"pick_override": 0.95,
}
engine = KataGoEngine(logger, ENGINE_SETTINGS)
ai_strategy = "P+Pick"
ai_strategy = "P+Local"
ai_settings["pick_frac"] = 0.0
ai_settings["pick_n"] = 10
ai_settings["local_stddev"] = 1.0
ai_strategy = "P+Influence"
ai_settings["pick_frac"] = 0.5
ai_settings["influence_weight"] = 0.1
ai_settings["line_weight"] = 0.1
ai_strategy = "P+Local"
ai_settings["pick_frac"] = 0.0
ai_settings["pick_n"] = 25
ai_settings["stddev"] = 1.5
ai_strategy = "P+Pick"
ai_settings["pick_frac"] = 0.33 # dropping below 7k
ai_settings["pick_n"] = 5 # dropping below 7k at 5/0.33
logger.log(f"STARTED ENGINE", OUTPUT_ERROR)
@@ -70,6 +73,7 @@ while not game.ended:
logger.log(f"Setting komi {game.root.properties}", OUTPUT_ERROR)
elif "genmove" in line:
game.current_node.analyze(engine)
game.root.add_property(f"P{game.current_node.next_player}", f"KaTrain {ai_strategy}")
move, node = ai_move(game, ai_strategy, ai_settings)
logger.log(f"SENT TO GTP: = {move.gtp()}", OUTPUT_ERROR)
print(f"= {move.gtp()}\n")
@@ -81,11 +85,10 @@ while not game.ended:
moves = sorted(list(cn.analysis["moves"].values()), key=lambda d: d["order"])
if moves:
pv = " ".join(moves[0]["pv"])
# print(cn.analysis,cn.analysis.get('root'),file=sys.stderr)
print(
f"CHAT:Visits {cn.ai_thoughts} Winrate {cn.analysis['root']['winrate']:.2%} ScoreLead {cn.analysis['root']['scoreLead']:.1f} ScoreStdev 0.0 PV {move.gtp()} {pv}",
file=sys.stderr,
) #
)
continue
elif "play" in line:
_, player, move = line.split(" ")
+8 -10
View File
@@ -38,16 +38,14 @@
"eval_show_ai": false
},
"ai": {
"balance_target_score": 2,
"balance_random_loss": 1,
"balance_max_loss": 5,
"balance_min_visits": 20,
"noise_strength": 0.8,
"pick_override": 0.95,
"pick_n": 10,
"pick_frac": 0.2,
"local_stddev": 5,
"influence_weight": 0.1
"Default": {},
"Balance": {"target_score":2, "random_loss": 1, "max_loss": 5, "min_visits": 20},
"P+Noise" : {"noise_strength": 0.8},
"P+Pick": {"pick_override": 0.95, "pick_n":5, "pick_frac": 0.33},
"P+Local": {"pick_override": 0.95, "stddev": 1.5, "pick_n":15, "pick_frac": 0.0},
"P+Tenuki": {"pick_override": 0.95, "stddev": 10, "pick_n":5, "pick_frac": 0.33},
"P+Influence": {"pick_override": 0.95, "pick_n":5, "pick_frac": 0.5,"line_weight": 0.1},
"P+Territory": {"pick_override": 0.95, "pick_n":5, "pick_frac": 0.5,"line_weight": 0.1}
},
"board_ui": {
"starpoint_size": 0.1,
+1 -1
View File
@@ -34,7 +34,7 @@ class Game:
if move_tree:
self.root = move_tree
self.komi = self.root.komi
handicap = int(self.root.get_first("HA",0))
handicap = int(self.root.get_first("HA", 0))
if handicap and not self.root.placements:
self.place_handicap_stones(handicap)
else:
+13 -4
View File
@@ -1,6 +1,9 @@
from kivy.graphics.context_instructions import Color
from kivy.graphics.vertex_instructions import Line, SmoothLine
from kivy.uix.boxlayout import BoxLayout
from kivy.uix.popup import Popup
from gui.popups import ConfigAIPopup
class Controls(BoxLayout):
@@ -31,7 +34,7 @@ class Controls(BoxLayout):
return self.player_mode_groups[player].value
def ai_mode(self, player):
return self.ai_mode_groups[player].text.lower()
return self.ai_mode_groups[player].text
def on_size(self, *args):
self.update_evaluation()
@@ -60,13 +63,19 @@ class Controls(BoxLayout):
self.score.text = current_node.format_score()
self.win_rate.text = current_node.format_win_rate()
if move and next_player_is_human_or_both_robots: # don't immediately hide this when an ai moves comes in
self.score_change.label = f"Score change"
points_lost = current_node.points_lost
self.score_change.text = f"{move.player}{-current_node.points_lost:+.1f}" if points_lost else "..."
self.score_change.label = f"Points lost for {move.player}" if points_lost and points_lost > 0 else f"Points gained for {move.player}"
self.score_change.text = f"{abs(points_lost):.1f}" if points_lost else "..."
elif not current_player_is_ai_playing_human:
self.score_change.label = f"Score change"
self.score_change.label = f"Points lost"
self.score_change.text = ""
self.graph.update_value(current_node)
self.info.text = info
def configure_ais(self):
config_popup = Popup(title="Edit AI Settings", size_hint=(0.9, 0.9))
popup_contents = ConfigAIPopup(self.katrain, config_popup, {self.ai_mode("B"), self.ai_mode("W")})
config_popup.add_widget(popup_contents)
config_popup.open()
+37 -13
View File
@@ -23,6 +23,19 @@ class QuickConfigGui(BoxLayout):
if initial_values:
self.set_properties(self, initial_values)
@staticmethod
def type_to_widget_class(value):
if isinstance(value, float):
return LabelledFloatInput
elif isinstance(value, bool):
return LabelledCheckBox
elif isinstance(value, int):
return LabelledIntInput
if isinstance(value, dict):
return LabelledObjectInputArea
else:
return LabelledTextInput
def collect_properties(self, widget):
if isinstance(widget, (LabelledTextInput, LabelledSpinner, LabelledCheckBox)):
try:
@@ -69,19 +82,6 @@ class NewGamePopup(QuickConfigGui):
class ConfigPopup(QuickConfigGui):
@staticmethod
def type_to_widget_class(value):
if isinstance(value, float):
return LabelledFloatInput
elif isinstance(value, bool):
return LabelledCheckBox
elif isinstance(value, int):
return LabelledIntInput
if isinstance(value, dict):
return LabelledObjectInputArea
else:
return LabelledTextInput
def __init__(self, katrain, popup, config, ignore_cats):
self.config = config
self.ignore_cats = ignore_cats
@@ -157,3 +157,27 @@ class ConfigPopup(QuickConfigGui):
self.katrain.debug_level = self.config["debug"]["level"]
self.katrain.update_state(redraw_board=True)
class ConfigAIPopup(QuickConfigGui):
def __init__(self, katrain, popup, ai_modes, **kwargs):
self.settings = self.katrain.ai_settings
super().__init__(katrain, popup, self.settings, **kwargs)
self.ai_modes = ai_modes
Clock.schedule_once(self._build, 0)
def _build(self):
for mode in self.ai_modes:
mode_settings = self.settings[mode]
column = GridLayout(rows=2 + len(mode_settings), columns=2, size_hint=(0.5, 1))
column.add_widget(ScaledLightLabel(text=f"Settings for AI {mode}", bold=True))
for k, v in mode_settings.items():
column.add_widget(ScaledLightLabel(text=f"{k}"))
column.add_widget(ConfigPopup.type_to_widget_class(v)(text=str(v), input_property=f"{mode}/{k}"))
column.add_widget(Label(text=f"Settings for AI {mode}", bold=True))
self.add_widget(column)
def on_submit(self):
self.popup.dismiss()
+17 -7
View File
@@ -31,14 +31,14 @@
radius: root.radius
<StyledSpinnerOption@SpinnerOption>:
font_size: self.size[1] * 0.4
font_size: self.size[1] * 0.33
background_color: BUTTON_COLOR
background_normal: ''
color: WHITE
<StyledSpinner>:
text: self.values[0] if self.values else ''
font_size: self.size[1] * 0.4
font_size: self.size[1] * 0.35
sync_height: True
background_color: [*[c*255/88 for c in BUTTON_COLOR[:3]], 1] # compensate for texture
option_cls: 'StyledSpinnerOption'
@@ -449,9 +449,9 @@
on_selection: root.katrain.update_state()
StyledSpinner:
id: B_AI_mode
values: ['Default']
size_hint: 0.3, 1
values: AI_MODES
on_text: B_player_mode.children[0].trigger_action(duration=0)
on_text: if B_player_mode.children: B_player_mode.children[0].trigger_action(duration=0)
Label:
size_hint: None,1
width: 3
@@ -467,14 +467,24 @@
StyledSpinner:
id: W_AI_mode
size_hint: 0.3, 1
values: AI_MODES
on_text: W_player_mode.children[0].trigger_action(duration=0)
values: ['Default']
on_text: if W_player_mode.children: W_player_mode.children[0].trigger_action(duration=0)
BoxLayout:
size_hint: 1,0.05
padding: 1
spacing: 1
Label:
size_hint: 0.55,1
StyledButton:
text: 'Configure AIs'
on_press: root.configure_ais()
size_hint: 0.45,1
Label:
size_hint: None,1
width: 3
LargeLabel:
text: ''
size_hint: 1,0.3
size_hint: 1,0.2
CensorableLabel:
id: score_change
size_hint: 1, 0.0225
+9 -3
View File
@@ -36,7 +36,10 @@ class KaTrainGui(BoxLayout):
self.logger = lambda message, level=OUTPUT_INFO: self.log(message, level)
self._load_config()
self.debug_level = self.config("debug/level", OUTPUT_INFO)
self.ai_settings = self.config("ai")
self.controls.ai_mode_groups["W"].values = self.controls.ai_mode_groups["B"].values = self.ai_settings.keys()
self.message_queue = Queue()
self._keyboard = Window.request_keyboard(None, self, "")
@@ -91,8 +94,8 @@ class KaTrainGui(BoxLayout):
self.game.analyze_undo(cn, self.config("trainer")) # not via message loop
if (
cn.analysis_ready
and "ai" in self.controls.player_mode(cn.next_player)
and not "pause" in self.controls.ai_mode(cn.next_player)
and "ai" in self.controls.player_mode(cn.next_player).lower()
and not "pause" in self.controls.ai_mode(cn.next_player).lower()
and not cn.children
and not self.game.ended
and not (auto_undo and cn.auto_undo is None)
@@ -141,7 +144,10 @@ class KaTrainGui(BoxLayout):
def _do_ai_move(self, node=None):
if node is None or self.game.current_node == node:
ai_move(self.game, self.controls.ai_mode(self.game.current_node.next_player), self.config("ai"))
mode = self.controls.ai_mode(self.game.current_node.next_player)
settings = self.ai_settings[mode]
if settings:
ai_move(self.game, mode, settings)
def _do_undo(self, n_times=1):
self.game.undo(n_times)
+7 -7
View File
@@ -37,15 +37,15 @@ class AI:
}
NUM_THREADS = 8
DEFAULT_SETTINGS = {
"balance_target_score": 2,
"balance_random_loss": 1,
"balance_max_loss": 5,
"balance_min_visits": 20,
"target_score": 2,
"random_loss": 1,
"max_loss": 5,
"min_visits": 20,
"noise_strength": 0.8,
"pick_n": 10,
"pick_frac": 0.2,
"local_stddev": 10,
"influence_weight": 0.1,
"line_weight": 0.1,
"pick_override": 0.95,
}
IGNORE_SETTINGS_IN_TAG = {"threads", "enable_ownership", "katago"} # katago for switching from/to bs version
@@ -133,8 +133,8 @@ test_ais = [
AI("P+Pick", {"pick_frac": 0.5, "pick_n": 0}),
AI("P+Influence", {"pick_frac": 0.2, "pick_n": 20}),
AI("P+Territory", {"pick_frac": 0.2, "pick_n": 20}),
AI("P+Influence", {"pick_frac": 0.33, "influence_weight": 0.05}),
AI("P+Territory", {"pick_frac": 0.33, "influence_weight": 0.05}),
AI("P+Influence", {"pick_frac": 0.33, "line_weight": 0.05}),
AI("P+Territory", {"pick_frac": 0.33, "line_weight": 0.05}),
AI("P+Pick", {"pick_frac": 0.0, "pick_n": 1}),
AI("P+Tenuki", {"local_stddev": 20}),
AI("P+Tenuki", {"local_stddev": 10}),
+5 -2
View File
@@ -1,3 +1,6 @@
GREETING="Hello, welcome to an experimental version of KaTrain AIs - These are based on weakened policy nets of KataGo. Current mode is: Play an influential style."
MAXGAMES=3
gtp2ogs --apikey $(cat my/apikey) --username katrain-dev --greeting "$GREETING" --debug --ogspv katago --noclock --maxconnectedgames $MAXGAMES --persist --minrank 15k --noautohandicap --maxhandicap 0 --fakerank 3k --boardsizes 9,13,19 --komis automatic,6.5 -- python ai2gtp.py
GREETING="Hello, welcome to an experimental version of KaTrain AIs - These are based on weakened policy nets of KataGo. Current mode is: Attach to ALL the things."
GREETING="Hello, welcome to an experimental version of KaTrain AIs - These are based on weakened policy nets of KataGo. Current mode is: Balanced style."
BYEMSG="Thank you for playing. If you have any feedback, please message my admin!"
MAXGAMES=5
gtp2ogs --apikey $(cat my/apikey) --username katrain-dev --greeting "$GREETING" --rankedonly --farewell "$BYEMSG" --debug --ogspv katago --noclock --speeds blitz,live --maxconnectedgames $MAXGAMES --persist --minrank 15k --noautohandicap --maxhandicap 0 --fakerank 3k --boardsizes 9,13,19 --komis automatic,6.5 -- python ai2gtp.py