ai improvements etc
This commit is contained in:
1 parent
612aa011b9
commit
461fbb4b64
12 files changed
+111
-70
No files matched your search
@@ -10,6 +10,7 @@ log.txt
|
||||
sgfout
|
||||
sgf_selfplay
|
||||
log*
|
||||
tmp.pickle
|
||||
my
|
||||
|
||||
# debug
|
||||
|
||||
@@ -46,7 +46,9 @@ Some uses include:
|
||||
* Balance is KataGo occasionally making weaker moves, attempting to win by ~2 points.
|
||||
* Jigo is KataGo aggressively making weaker moves, attempting to win by 0.5 points.
|
||||
* Policy is the top move from the policy network (it's 'shape sense' without reading), should be around high dan level depending on the model used.
|
||||
* P+Pick will pick a `pick_n + <number of legal moves> * pick_frac` moves at random, and play the best move among them.
|
||||
* P+Pick will pick a `pick_n + pick_frac * <number of legal moves>` moves at random, and play the best move among them.
|
||||
The setting `pick_override` determines the minimum value at which this process is bypassed to play the best move instead, preventing obvious blunders.
|
||||
This is probably the best choice for kyu players who want a chance of winning. Variants of this strategy include:
|
||||
* P+Local will pick such moves biased towards the last move with probability related to `local_stddev`.
|
||||
* P+Tenuki is biased in the opposite way as P+Local, using the same setting.
|
||||
* P+Influence is biased towards 4th+ line moves, with every line below that dividing both the chance of considering the move and the policy value by `influence_weight`. Consider setting `pick_frac=1.0` to only affect the policy weight.
|
||||
|
||||
@@ -8,16 +8,19 @@
|
||||
[x] Scrolling add a move on the board instead of navigating through the game. This was already the case in the 0.9 version and it's quite annoying as scrolling seemed only natural and I kept forgetting not to do it :p
|
||||
[x] show PV on hint hover? Although Katrain wasn't meant to be like Lizzie to begin with, it would be really neat if we could visualize the expected variations when hovering over the top moves.
|
||||
[x Self-play tournaments in separate script.
|
||||
[x] ai thoughts in sgf
|
||||
[x] more AI modes?
|
||||
|
||||
[] Score instead of game end
|
||||
[/] README
|
||||
[] engine status
|
||||
[] README
|
||||
[] Release notes
|
||||
[] more AI modes?
|
||||
[] policy threshold override p+pick?
|
||||
[] sgf review improvements
|
||||
[] Release notes
|
||||
-- Likewise, in the 0.9 version, better alternatives to the played move were shown with squares, which was also pretty useful when using the sgf outside of Katrain. I mean, having the top move mentioned is all and good, but when you see multiple squares shown on the board as better alternatives to the move played in the game, it makes obvious how far from perfect that move actually was :D
|
||||
[] ai thoughts in sgf
|
||||
|
||||
[] pol value override > 0.9 ?
|
||||
[] clarify score change vs score
|
||||
[] pv with overlap?
|
||||
|
||||
- dots: SPINNER! off last few / white black / >x pt (multi select?)
|
||||
|
||||
@@ -27,6 +30,7 @@ Low priority
|
||||
[] box to label ? split in status and comment?
|
||||
[] dual engine support -- easily possible but has weird effects on win rate etc
|
||||
[] When creating a new game, the 9 buttons on the right side aren't all that useful. Maybe the 9, 13 and 19 ones make sense since these three board sizes are the traditionally used ones, but why 2, 4 and 9 stones buttons? Why 0.5, 6.5 and -40pts komi buttons?
|
||||
[] Score instead of game end
|
||||
|
||||
Wont do for now
|
||||
[] List edit settings/object edit settings?
|
||||
|
||||
@@ -2,7 +2,7 @@ import heapq
|
||||
import math
|
||||
import random
|
||||
import time
|
||||
from typing import Dict
|
||||
from typing import Dict, List, Tuple, Any
|
||||
|
||||
import numpy as np
|
||||
|
||||
@@ -11,15 +11,18 @@ from engine import EngineDiedException
|
||||
from game import Move, Game, IllegalMoveException
|
||||
|
||||
|
||||
def weighted_selection_without_replacement(items, m):
|
||||
"""For a list of arrays where the first element is a weight, returns random items with those weights, without replacement."""
|
||||
elt = [(math.log(random.random()) / item[0], item) for item in items] # magic
|
||||
return [e[1] for e in heapq.nlargest(m, elt)] # NB fine if too small
|
||||
def weighted_selection_without_replacement(items: List[Tuple[float,float,int,int]], pick_n: int) -> List[Tuple[float,float,int,int]]:
|
||||
"""For a list of tuples where the second element is a weight, returns random items with those weights, without replacement."""
|
||||
elt = [(math.log(random.random()) / item[1], item) for item in items] # magic
|
||||
return [e[1] for e in heapq.nlargest(pick_n, elt)] # NB fine if too small
|
||||
|
||||
|
||||
def dirichlet_noise(num, dir_alpha=0.3):
|
||||
return np.random.dirichlet([dir_alpha] * num)
|
||||
|
||||
def fmt_moves(moves: List[Tuple[float,Move]]):
|
||||
return ', '.join(f"{mv.gtp()} ({p:.2%})" for p, mv in moves)
|
||||
|
||||
|
||||
def ai_move(game: Game, ai_mode: str, ai_settings: Dict):
|
||||
cn = game.current_node
|
||||
@@ -29,69 +32,78 @@ def ai_move(game: Game, ai_mode: str, ai_settings: Dict):
|
||||
if engine.katago_process.poll() is not None: # TODO: clean up
|
||||
raise EngineDiedException(f"Engine for {cn.next_player} ({engine.config}) died")
|
||||
ai_mode = ai_mode.lower()
|
||||
ai_thoughts = ''
|
||||
candidate_ai_moves = cn.candidate_moves
|
||||
if ("policy" in ai_mode or "p+" in ai_mode) and cn.policy:
|
||||
policy_moves = cn.policy_ranking
|
||||
pass_policy = cn.policy[-1]
|
||||
top_5_pass = any([polmove[0].is_pass for polmove in policy_moves[:5]]) # dont make it jump around for the last few sensible non pass moves
|
||||
top_5_pass = any([polmove[1].is_pass for polmove in policy_moves[:5]]) # dont make it jump around for the last few sensible non pass moves
|
||||
|
||||
size = game.board_size
|
||||
policy_grid = var_to_grid(cn.policy, size)
|
||||
legal_policy_moves = [(mv, pol) for mv, pol in policy_moves if not mv.is_pass if pol > 0]
|
||||
top_policy_move = policy_moves[0][0]
|
||||
game.katrain.log(f"Policy strategy {ai_mode} found {top_policy_move} as top move", OUTPUT_DEBUG)
|
||||
policy_grid = var_to_grid(cn.policy, size) # type: List[List[float]]
|
||||
legal_policy_moves = [(pol,mv) for pol, mv in policy_moves if not mv.is_pass if pol > 0]
|
||||
top_policy_move = policy_moves[0][1]
|
||||
ai_thoughts += f"Using policy based strategy, base top 5 moves are {fmt_moves(policy_moves[:5])}. "
|
||||
if top_policy_move.is_pass:
|
||||
aimove = top_policy_move
|
||||
elif top_5_pass:
|
||||
weighted_coords = [(policy_grid[y][x], x, y, policy_grid[y][x]) for x in range(size[0]) for y in range(size[1]) if policy_grid[y][x] > 0]
|
||||
aimove = Move(weighted_selection_without_replacement(weighted_coords, 1)[0][1:3], player=cn.next_player) # just take a random move by policy w/o noise
|
||||
game.katrain.log(f"Policy strategy {ai_mode} found pass in top 5 moves so chose {aimove} as weighted-by-policy move", OUTPUT_DEBUG)
|
||||
ai_thoughts += 'Playing top one because it is pass.'
|
||||
elif "policy" in ai_mode:
|
||||
aimove = top_policy_move
|
||||
ai_thoughts += f"Playing top policy move {aimove.gtp()} due to mode chosen."
|
||||
elif policy_moves[0][0] > ai_settings['pick_override']:
|
||||
aimove = top_policy_move
|
||||
ai_thoughts += f"Top policy move has weight > {ai_settings['pick_override']:.1%}, so overriding other strategies."
|
||||
elif top_5_pass:
|
||||
weighted_coords = [(policy_grid[y][x], policy_grid[y][x], x, y) for x in range(size[0]) for y in range(size[1]) if policy_grid[y][x] > 0]
|
||||
aimove = Move(weighted_selection_without_replacement(weighted_coords, 1)[0][2:], player=cn.next_player) # just take a random move by policy w/o noise
|
||||
ai_thoughts += f"Playing policy-weighted random move {aimove.gtp()} because one of them is pass."
|
||||
elif "noise" in ai_mode:
|
||||
noise_str = ai_settings["noise_strength"]
|
||||
d_noise = dirichlet_noise(len(legal_policy_moves))
|
||||
noisy_policy_moves = [(mv, (1 - noise_str) * pol + noise_str * noise) for ((mv, pol), noise) in zip(legal_policy_moves, d_noise)]
|
||||
best = max(noisy_policy_moves, key=lambda mp: mp[1])
|
||||
aimove = best[0]
|
||||
game.katrain.log(f"Noisy policy strategy (strength={noise_str:.2f}) generated move {aimove.gtp()} with value {best[1]}", OUTPUT_DEBUG)
|
||||
noisy_policy_moves = [(((1 - noise_str) * pol + noise_str * noise), mv) for ((pol, mv), noise) in zip(legal_policy_moves, d_noise)]
|
||||
new_top = heapq.nlargest(5,noisy_policy_moves)
|
||||
aimove = new_top[0][1]
|
||||
ai_thoughts +=f"Noisy policy strategy (strength={noise_str:.2f}) generated 5 moves {fmt_moves(new_top)} so picked {aimove.gtp()}. "
|
||||
elif any(keyword in ai_mode for keyword in ["influence", "territory", "local", "tenuki", "pick"]):
|
||||
n_moves = int(ai_settings["pick_frac"] * len(legal_policy_moves) + ai_settings["pick_n"])
|
||||
if "influence" in ai_mode or "territory" in ai_mode:
|
||||
|
||||
if "influence" in ai_mode:
|
||||
weight = lambda x, y: ai_settings["influence_weight"] ** max(0, 3 - min(size[0] - 1 - x, x, y, size[1] - 1 - y))
|
||||
else:
|
||||
weight = lambda x, y: ai_settings["influence_weight"] ** max(0, min(size[0] - 1 - x, x, y, size[1] - 1 - y) - 2)
|
||||
weighted_coords = [(weight(x, y), x, y, policy_grid[y][x] * weight(x, y)) for x in range(size[0]) for y in range(size[1]) if policy_grid[y][x] > 0]
|
||||
weighted_coords = [(policy_grid[y][x] * weight(x, y), weight(x, y), x, y) for x in range(size[0]) for y in range(size[1]) if policy_grid[y][x] > 0]
|
||||
ai_thoughts += f"Generated weights for {ai_mode} according to weight factor {ai_settings['influence_weight']} and distance from 4th line. "
|
||||
elif "local" in ai_mode or "tenuki" in ai_mode:
|
||||
var = ai_settings["local_stddev"] ** 2
|
||||
if not cn.single_move or cn.single_move.coords is None:
|
||||
weighted_coords = [(1, *top_policy_move.coords, 1)] # if "pick" in ai_mode -> even
|
||||
game.katrain.log(f"Local strategy: no previous non-pass move, playing top policy move {top_policy_move}", OUTPUT_DEBUG)
|
||||
weighted_coords = [(1, 1, *top_policy_move.coords)] # if "pick" in ai_mode -> even
|
||||
ai_thoughts += f"No previous non-pass move, faking weights to play top policy move. "
|
||||
else:
|
||||
mx, my = cn.single_move.coords
|
||||
weighted_coords = [
|
||||
(math.exp(-0.5 * ((x - mx) ** 2 + (y - my) ** 2) / var), x, y, policy_grid[y][x]) for x in range(size[0]) for y in range(size[1]) if policy_grid[y][x] > 0
|
||||
(policy_grid[y][x],math.exp(-0.5 * ((x - mx) ** 2 + (y - my) ** 2) / var), x, y) for x in range(size[0]) for y in range(size[1]) if policy_grid[y][x] > 0
|
||||
]
|
||||
game.katrain.log(f"Generated weights based on gaussian with var {var} around {mx},{my}", OUTPUT_DEBUG)
|
||||
if "tenuki" in ai_mode:
|
||||
weighted_coords = [(1 - w, x, y, p) for w, x, y, p in weighted_coords]
|
||||
weighted_coords = [(p,1 - w, x, y) for p, w, x, y in weighted_coords]
|
||||
ai_thoughts += f"Generated weights based on one minus gaussian with variance {var} around coordinates {mx},{my}. "
|
||||
else:
|
||||
ai_thoughts += f"Generated weights based on gaussian with variance {var} around coordinates {mx},{my}. "
|
||||
elif "pick" in ai_mode:
|
||||
weighted_coords = [(1, x, y, policy_grid[y][x]) for x in range(size[0]) for y in range(size[1]) if policy_grid[y][x] > 0]
|
||||
weighted_coords = [(policy_grid[y][x], 1, x, y ) for x in range(size[0]) for y in range(size[1]) if policy_grid[y][x] > 0]
|
||||
else:
|
||||
raise ValueError(f"Unknown AI mode {ai_mode}")
|
||||
pick_moves = weighted_selection_without_replacement(weighted_coords, n_moves)
|
||||
ai_thoughts +=f"Picked {min(n_moves,len(weighted_coords))} random moves according to weights. "
|
||||
if pick_moves:
|
||||
best = max(pick_moves, key=lambda m: m[3])
|
||||
aimove = Move((best[1], best[2]), player=cn.next_player)
|
||||
game.katrain.log(f"Pick policy strategy {ai_mode} (n={n_moves}) generated move {aimove.gtp()} with weight {best[0]} and value {best[3]}", OUTPUT_DEBUG)
|
||||
if best[3] < pass_policy:
|
||||
game.katrain.log(f"Pick policy strategy found pass is better than {aimove} so will pass instead", OUTPUT_DEBUG)
|
||||
new_top = [(p,Move((x,y),player=cn.next_player)) for p,wt,x,y in heapq.nlargest(5,pick_moves)]
|
||||
aimove = new_top[0][1]
|
||||
ai_thoughts += f"Top 5 among these were {fmt_moves(new_top)} and picked top {aimove.gtp()}. "
|
||||
if new_top[0][0] < pass_policy:
|
||||
ai_thoughts += f"But found pass ({pass_policy:.1%} to be higher rated than {aimove.gtp()} ({new_top[0][0]:.1%}) so will pass instead."
|
||||
aimove = Move(None, player=cn.next_player)
|
||||
else:
|
||||
aimove = top_policy_move
|
||||
game.katrain.log(f"Pick policy strategy {ai_mode} failed to find legal moves, so is playing top policy move {aimove}", OUTPUT_DEBUG)
|
||||
ai_thoughts += f"Pick policy strategy {ai_mode} failed to find legal moves, so is playing top policy move {aimove.gtp()}."
|
||||
else:
|
||||
raise ValueError(f"Unknown AI mode {ai_mode}")
|
||||
elif "balance" in ai_mode and candidate_ai_moves[0]["move"] != "pass": # don't play suicidal to balance score - pass when it's best
|
||||
@@ -108,21 +120,23 @@ def ai_move(game: Game, ai_mode: str, ai_settings: Dict):
|
||||
)
|
||||
]
|
||||
aimove = Move.from_gtp(random.choice(sel_moves)["move"], player=cn.next_player) # TODO: could be weighted towards worse
|
||||
game.katrain.log(f"Balance strategy considered {len(sel_moves)} moves and chose {aimove} randomly", OUTPUT_DEBUG)
|
||||
ai_thoughts += f"Balance strategy selected moves {sel_moves} based on target score and max points lost, and randomly chose {aimove.gtp()}."
|
||||
elif "jigo" in ai_mode and candidate_ai_moves[0]["move"] != "pass":
|
||||
sign = cn.player_sign(cn.next_player) # TODO check
|
||||
jigo_move = min(candidate_ai_moves, key=lambda move: abs(sign * move["scoreLead"] - 0.5))
|
||||
aimove = Move.from_gtp(jigo_move["move"], player=cn.next_player)
|
||||
game.katrain.log(f"Jigo strategy found {len(candidate_ai_moves)} moves and chose {aimove} as closest to 0.5 point win", OUTPUT_DEBUG)
|
||||
ai_thoughts += f"Jigo strategy found candidate moves {candidate_ai_moves} moves and chose {aimove.gtp()} as closest to 0.5 point win"
|
||||
else:
|
||||
if "default" not in ai_mode and "katago" not in ai_mode:
|
||||
game.katrain.log(f"Unknown AI mode {ai_mode} or policy missing, using default.", OUTPUT_INFO)
|
||||
ai_thoughts += f"Strategy {ai_mode} not found or unexpected fallback."
|
||||
aimove = Move.from_gtp(candidate_ai_moves[0]["move"], player=cn.next_player)
|
||||
game.katrain.log(f"Default strategy found {len(candidate_ai_moves)} moves and chose {aimove} as top move", OUTPUT_DEBUG)
|
||||
|
||||
ai_thoughts += f"Default strategy found {len(candidate_ai_moves)} moves returned from the engine and chose {aimove.gtp()} as top move"
|
||||
game.katrain.log(f"AI thoughts: {ai_thoughts}",OUTPUT_DEBUG)
|
||||
try:
|
||||
game.play(aimove)
|
||||
played_move = game.play(aimove)
|
||||
played_move.ai_thoughts = ai_thoughts
|
||||
except IllegalMoveException as e:
|
||||
game.katrain.log(f"AI Strategy {ai_mode} generated illegal move {aimove}: {e}", OUTPUT_ERROR)
|
||||
game.katrain.log(f"AI Strategy {ai_mode} generated illegal move {aimove.gtp()}: {e}", OUTPUT_ERROR)
|
||||
|
||||
return aimove
|
||||
Binary file not shown.
Binary file not shown.
@@ -1,10 +1,12 @@
|
||||
from typing import List, Any, Tuple
|
||||
|
||||
OUTPUT_ERROR = -1
|
||||
OUTPUT_INFO = 0
|
||||
OUTPUT_DEBUG = 1
|
||||
OUTPUT_EXTRA_DEBUG = 2
|
||||
|
||||
|
||||
def var_to_grid(array_var, size):
|
||||
def var_to_grid(array_var: List[Any], size: Tuple[int,int]) -> List[List[Any]]:
|
||||
"""convert ownership/policy to grid format such that grid[y][x] is for move with coords x,y"""
|
||||
ix = 0
|
||||
grid = [[]] * size[1]
|
||||
|
||||
@@ -43,6 +43,7 @@
|
||||
"balance_max_loss": 5,
|
||||
"balance_min_visits": 20,
|
||||
"noise_strength": 0.8,
|
||||
"pick_override": 0.95,
|
||||
"pick_n": 10,
|
||||
"pick_frac": 0.2,
|
||||
"local_stddev": 5,
|
||||
|
||||
+12
-12
@@ -2,6 +2,7 @@ import copy
|
||||
import random
|
||||
from typing import Dict, List, Optional, Tuple
|
||||
|
||||
from common import var_to_grid
|
||||
from sgf_parser import Move, SGFNode
|
||||
|
||||
|
||||
@@ -14,6 +15,7 @@ class GameNode(SGFNode):
|
||||
self.ownership = None
|
||||
self.policy = None
|
||||
self.auto_undo = None # None = not analyzed. False: not undone (good move). True: undone (bad move)
|
||||
self.ai_thoughts = ""
|
||||
self.move_number = 0
|
||||
self.undo_threshold = random.random() # for fractional undos, store the random threshold in the move itself for consistency
|
||||
|
||||
@@ -84,13 +86,15 @@ class GameNode(SGFNode):
|
||||
text += f"Move was predicted best move.\n"
|
||||
if sgf or hints or teach:
|
||||
policy_ranking = self.parent.policy_ranking
|
||||
policy_ix = [ix + 1 for (m, p), ix in zip(policy_ranking, range(len(policy_ranking))) if m == single_move]
|
||||
policy_ix = [ix + 1 for (p, m), ix in zip(policy_ranking, range(len(policy_ranking))) if m == single_move]
|
||||
if policy_ix:
|
||||
text += f"Move was #{policy_ix[0]} according to policy.\n"
|
||||
if not policy_ix or policy_ix[0] != 1 and (sgf or hints):
|
||||
text += f"Top policy move was {policy_ranking[0][0].gtp()}.\n"
|
||||
text += f"Top policy move was {policy_ranking[0][1].gtp()} ({policy_ranking[0][0]:.1%}).\n"
|
||||
if self.auto_undo and sgf:
|
||||
text += "Move was automatically undone in teaching mode."
|
||||
if self.ai_thoughts:
|
||||
text += f"\nAI thought process: {self.ai_thoughts}"
|
||||
else:
|
||||
text = "No analysis available" if sgf else "Analyzing move..."
|
||||
return text
|
||||
@@ -118,7 +122,7 @@ class GameNode(SGFNode):
|
||||
return []
|
||||
if not self.analysis["moves"]:
|
||||
polmoves = self.policy_ranking
|
||||
top_polmove = polmoves[0][0] if polmoves else Move(None) # if no info at all, pass
|
||||
top_polmove = polmoves[0][1] if polmoves else Move(None) # if no info at all, pass
|
||||
return [{**self.analysis["root"], "pointsLost": 0, "order": 0, "move": top_polmove.gtp()}] # single visit -> go by policy/root
|
||||
|
||||
return sorted(
|
||||
@@ -127,14 +131,10 @@ class GameNode(SGFNode):
|
||||
)
|
||||
|
||||
@property
|
||||
def policy_ranking(self) -> Optional[List[Tuple[Move, float]]]: # return moves from highest policy value to lowest
|
||||
def policy_ranking(self) -> Optional[List[Tuple[float, Move]]]: # return moves from highest policy value to lowest
|
||||
if self.policy:
|
||||
ix = 0
|
||||
moves = []
|
||||
szx, szy = self.board_size
|
||||
for y in range(szy - 1, -1, -1):
|
||||
for x in range(szx):
|
||||
moves.append((Move((x, y), player=self.next_player), self.policy[ix]))
|
||||
ix += 1
|
||||
moves.append((Move(None, player=self.next_player), self.policy[ix]))
|
||||
return sorted(moves, key=lambda mp: -mp[1])
|
||||
policy_grid = var_to_grid(self.policy, size=[szx, szy])
|
||||
moves = [(policy_grid[y][x], Move((x, y), player=self.next_player)) for x in range(szx) for y in range(szy)]
|
||||
moves.append((self.policy[-1],Move(None, player=self.next_player)))
|
||||
return sorted(moves, key=lambda mp: -mp[0])
|
||||
+2
-2
@@ -194,8 +194,8 @@ class BadukPanWidget(Widget):
|
||||
evalsize = 1
|
||||
for m in node.move_with_placements:
|
||||
if has_stone.get(m.coords) and not drawn_stone.get(m.coords): # skip captures, last only for
|
||||
move_eval_on = full_eval_on or i < show_n_eval
|
||||
if move_eval_on and points_lost is not None and show_dots_for.get(m.player):
|
||||
move_eval_on = full_eval_on or (i < show_n_eval and show_dots_for.get(m.player))
|
||||
if move_eval_on and points_lost is not None:
|
||||
evalcol = self.eval_color(points_lost)
|
||||
else:
|
||||
evalcol = None
|
||||
|
||||
+1
-1
@@ -526,7 +526,7 @@
|
||||
BadukPanControls:
|
||||
id: board_controls
|
||||
size_hint_y: None
|
||||
height: self.width / 13.5
|
||||
height: min(100,self.width / 13.5)
|
||||
BoxLayout: # avoids a weird syntax error in builder .. somehow
|
||||
size_hint_x: None
|
||||
size_hint_y: 1
|
||||
|
||||
+27
-10
@@ -45,7 +45,8 @@ class AI:
|
||||
"pick_n": 10,
|
||||
"pick_frac": 0.2,
|
||||
"local_stddev": 10,
|
||||
"influence_weight": 0.01,
|
||||
"influence_weight": 0.1,
|
||||
"pick_override": 0.95,
|
||||
}
|
||||
IGNORE_SETTINGS_IN_TAG = {"threads", "enable_ownership", "katago"} # katago for switching from/to bs version
|
||||
ENGINES = []
|
||||
@@ -116,13 +117,9 @@ test_ais = [
|
||||
AI("P+Local", {"local_stddev": 10}),
|
||||
AI("P+Local", {"local_stddev": 5}),
|
||||
AI("P+Pick", {"pick_frac": 0.0, "pick_n": 1}),
|
||||
]
|
||||
|
||||
test_ais += [
|
||||
AI("Jigo", {}, {"max_visits": 50}),
|
||||
AI("Policy", {}),
|
||||
AI("Jigo", {}, {"max_visits": 50})
|
||||
# AI("Policy", {},{'model':'models/g170-b40c256x2-s2990766336-d830712531.bin.gz'}),
|
||||
# AI("KataGo", {}, {"max_visits": 50}),
|
||||
]
|
||||
|
||||
test_ais = [
|
||||
@@ -133,20 +130,35 @@ test_ais = [
|
||||
AI("P+Noise", {"noise_strength": 0.9}),
|
||||
AI("P+Pick", {}),
|
||||
AI("P+Pick", {"pick_frac": 0.3, "pick_n": 20}),
|
||||
AI("P+Pick", {"pick_frac": 0.5, "pick_n": 0}),
|
||||
AI("P+Influence", {"pick_frac": 0.2, "pick_n": 20}),
|
||||
AI("P+Territory", {"pick_frac": 0.2, "pick_n": 20}),
|
||||
AI("P+Influence", {"pick_frac": 0.33, "influence_weight": 0.05}),
|
||||
AI("P+Territory", {"pick_frac": 0.33, "influence_weight": 0.05}),
|
||||
AI("P+Local", {"local_stddev": 5}),
|
||||
AI("P+Tenuki", {"local_stddev": 10}),
|
||||
AI("P+Pick", {"pick_frac": 0.0, "pick_n": 1}),
|
||||
AI("P+Tenuki", {"local_stddev": 20}),
|
||||
AI("P+Tenuki", {"local_stddev": 10}),
|
||||
AI("P+Tenuki", {"local_stddev": 5}),
|
||||
AI("P+Local", {"local_stddev": 10}),
|
||||
AI("P+Local", {"local_stddev": 5}),
|
||||
AI("P+Local", {"local_stddev": 1}),
|
||||
AI("P+Local", {"local_stddev": 1,"pick_frac": 0.0, "pick_n": 20}),
|
||||
]
|
||||
|
||||
#test_ais = [
|
||||
# AI("Policy", {}),
|
||||
# AI("P+Noise", {"noise_strength": 0.4}),
|
||||
# AI("P+Noise", {"noise_strength": 0.5}),
|
||||
# AI("P+Noise", {"noise_strength": 0.6}),
|
||||
# AI("P+Noise", {"noise_strength": 0.7}),
|
||||
# AI("P+Noise", {"noise_strength": 0.8}),
|
||||
#]
|
||||
|
||||
# ai_database = [ai for ai in ai_database if "Territory" not in ai.name and "Influence" not in ai.name]
|
||||
for ai in test_ais:
|
||||
add_ai(ai)
|
||||
|
||||
N_GAMES = 10
|
||||
N_GAMES = 5
|
||||
|
||||
ais_to_test = retrieve_ais(test_ais)
|
||||
# ais_to_test = ai_database
|
||||
@@ -182,9 +194,14 @@ def play_games(black: AI, white: AI, n: int = N_GAMES):
|
||||
|
||||
results[tag].append(score)
|
||||
all_results.append((black.name, white.name, score))
|
||||
|
||||
with open('tmp.pickle', "wb") as f:
|
||||
pickle.dump((ai_database, all_results), f)
|
||||
except Exception as e:
|
||||
print(f"Exception in playing {tag}: {e}")
|
||||
print(f"Exception in playing {tag}: {e}", file=sys.stderr)
|
||||
traceback.print_tb(file=sys.stderr)
|
||||
traceback.print_exc()
|
||||
traceback.print_exc(file=sys.stderr)
|
||||
|
||||
|
||||
def fmt_score(score):
|
||||
|
||||
Reference in new issue
Block a user