scoreloss ai fix / p:noise ai removed / tenuki tweaked

This commit is contained in:
Sander Land committed 2020-05-09 22:36:43 +02:00
1 parent e3b1491856
commit abc21943b5
8 files changed
+68 -82

No files matched your search

+4 -6
View File
@@ -84,20 +84,18 @@ while stronger players can pay more attention to smaller mistakes.
Available AIs, with strength indicating an estimate for the default settings, are:
* **[9p+]** **Default** is full KataGo, above professional level.
* **[~1d?]** **ScoreLoss** is KataGo making moves with probability `~ e^(-strength * points lost)`.
* **Balance** is KataGo occasionally making weaker moves, attempting to win by ~2 points.
* **Jigo** is KataGo aggressively making weaker moves, attempting to win by 0.5 points.
* **[~4d]** **Policy** uses the top move from the policy network (it's 'shape sense' without reading), should be around high dan level depending on the model used. There is a setting to increase variety in the opening, but otherwise it plays deterministically.
* **[~5k]**: **P:Weighted** picks a random move weighted by the policy, as long as it's above `lower_bound`. `weaken_fac` uses `policy^(1/weaken_fac)`, increasing the chance for weaker moves.
* **[~2k]**: **P:Weighted** picks a random move weighted by the policy, as long as it's above `lower_bound`. `weaken_fac` uses `policy^(1/weaken_fac)`, increasing the chance for weaker moves.
* **[~5k]**: **P:Pick** picks `pick_n + pick_frac * <number of legal moves>` moves at random, and play the best move among them.
The setting `pick_override` determines the minimum value at which this process is bypassed to play the best move instead, preventing obvious blunders.
This, along with 'Weighted' are probably the best choice for kyu players who want a chance of winning without playing the sillier bots below. Variants of this strategy include:
* **[~5k]**: **P:Local** will pick such moves biased towards the last move with probability related to `local_stddev`.
* **[~10k]**: **~P:Tenuki** is biased in the opposite way as P:Local, using the same setting.
* **[~2k]**: **P:Local** will pick such moves biased towards the last move with probability related to `local_stddev`.
* **[~10k]**: **P:Tenuki** is biased in the opposite way as P:Local, using the same setting.
* **[~10k]**: **P:Influence** is biased towards 4th+ line moves, with every line below that dividing both the chance of considering the move and the policy value by `influence_weight`. Consider setting `pick_frac=1.0` to only affect the policy weight.
* **[~10k]**: **P:Territory** is biased in the opposite way, towards 1-3rd line moves, using the same setting.
* * **[~5k]**: **P:Noise** mixes the policy with `noise_strength` Dirichlet noise. At `noise_strength=0.9` play is near-random, while `noise_strength=0.7` is still quite strong. A threshold setting is included to avoid senseless first-line moves.
Selecting the AI as either white or black opens up the option to configure it under 'Configure AI'.
### Analysis
+4 -8
View File
@@ -40,14 +40,15 @@ ENGINE_SETTINGS = {
"threads": 1,
}
engine = KataGoEngine(logger, ENGINE_SETTINGS)
with open("config.json") as f:
settings = json.load(f)
all_ai_settings = settings["ai"]
all_ai_settings["dev"] = all_ai_settings["P:Noise"]
if bot == "dev":
engine.override_settings["maxVisits"] = 500
all_ai_settings["dev"] = all_ai_settings["ScoreLoss"]
ai_strategy = bot_strategy_names[bot]
ai_settings = all_ai_settings[ai_strategy]
@@ -149,12 +150,7 @@ while True:
move = game.play(Move(None, player=game.next_player)).move
else:
move, node = ai_move(game, ai_strategy, ai_settings)
if node is None:
while node is None:
logger.log(f"ERROR generating move, backing up with weighted.", OUTPUT_ERROR)
move, node = ai_move(game, "p:weighted", {"pick_override": 1.0, "lower_bound": 0.001, "weaken_fac": 1})
else:
logger.log(f"Generated move {move}", OUTPUT_ERROR)
logger.log(f"Generated move {move}", OUTPUT_ERROR)
print(f"= {move.gtp()}\n")
sys.stdout.flush()
malkovich_analysis(game.current_node)
+6 -7
View File
@@ -57,8 +57,7 @@ class AI:
def fix_settings(self):
self.ai_settings = {**DEFAULT_AI_SETTINGS[self.strategy], **self.ai_settings}
self.engine_settings = {**AI.DEFAULT_ENGINE_SETTINGS, **self.engine_settings,
"threads": AI.NUM_THREADS}
self.engine_settings = {**AI.DEFAULT_ENGINE_SETTINGS, **self.engine_settings, "threads": AI.NUM_THREADS}
def get_engine(self): # factory
with AI.LOCK:
@@ -98,8 +97,8 @@ def retrieve_ais(selected_ais):
test_ais = [
# AI("Jigo", {}, {"max_visits": 100}),
AI("Policy", {}, {'model': 'my/model.bin.gz'}),
AI("Policy", {}, {'model': 'KataGo/models/b10-1.3.txt.gz'}),
AI("Policy", {}, {"model": "my/model.bin.gz"}),
AI("Policy", {}, {"model": "KataGo/models/b10-1.3.txt.gz"}),
AI("Policy", {}),
AI("P:Local", {}),
AI("P:Pick", {}),
@@ -128,7 +127,7 @@ def play_games(black: AI, white: AI):
engines = {"B": black.get_engine(), "W": white.get_engine()}
tag = f"{black.name} vs {white.name}"
try:
game = Game(Logger(), engines, {'init_size':BOARDSIZE})
game = Game(Logger(), engines, {"init_size": BOARDSIZE})
game.root.add_list_property("PW", [white.name])
game.root.add_list_property("PB", [black.name])
start_time = time.time()
@@ -163,9 +162,9 @@ print(len(ais_to_test), "ais to test")
global_start = time.time()
for n in range(N_GAMES):
for _,e in AI.ENGINES: # no caching/replays
for _, e in AI.ENGINES: # no caching/replays
e.shutdown()
AI.ENGINES=[]
AI.ENGINES = []
with ThreadPoolExecutor(max_workers=16) as threadpool:
for b in ais_to_test:
+4 -2
View File
@@ -1,5 +1,6 @@
bot_strategy_names = {
"dev": "P:Noise",
# "dev": "P:Noise",
"dev": "ScoreLoss",
"dev-beta": "P:Weighted",
"strong": "Policy",
"influence": "P:Influence",
@@ -12,7 +13,8 @@ bot_strategy_names = {
greetings = {
"dev": "Policy+Dirichlet noise.",
# "dev": "Policy+Dirichlet noise.",
"dev": "Point loss-weighted random move.",
"dev-beta": "Play a policy-weighted move.",
"strong": "Play top policy move.",
"influence": "Play an influential style.",
+4 -10
View File
@@ -67,7 +67,7 @@
"ScoreLoss": {
"strength": 0.5,
"_help_left": "Plays moves weighted inversely by point loss.",
"_help_right": "Also affected by engine settings such as `max_visits`."
"_help_right": "Also affected by engine settings such as `max_visits`, likely to play more varied/weaker with higher visits."
},
"Policy": {
"opening_moves": 0.05,
@@ -81,13 +81,6 @@
"lower_bound": 0.001,
"weaken_fac": 1.25
},
"P:Noise": {
"pick_override": 0.95,
"noise_strength": 0.6,
"lower_bound": 0.001,
"_help_left": "Adds `noise_strength` noise to the policy of all moved > 'lower_bound' and plays the top move.",
"_help_right": "Plays top move if policy value is above `pick_override` to avoid obvious mistakes. Noise above 0.9 is near random, below 0.7 is fairly strong."
},
"P:Pick": {
"pick_override": 0.95,
"pick_n": 5,
@@ -107,9 +100,10 @@
"pick_override": 0.85,
"stddev": 7.5,
"pick_n": 5,
"pick_frac": 0.7,
"pick_frac": 0.5,
"endgame": 0.5,
"_help_left": "Samples `pick_n + pick_frac * <number of legal moves>` away from the last move and plays the best one.",
"_help_right": "Increase `stddev` makes it prefer moves further away."
"_help_right": "Increase `stddev` makes it prefer moves further away. Stops tenukiing after the 'endgame' fraction of the board is filled."
},
"P:Influence": {
"pick_override": 0.95,
+43 -38
View File
@@ -9,7 +9,7 @@ from core.engine import EngineDiedException
from core.game import Game, GameNode, IllegalMoveException, Move
def weighted_selection_without_replacement(items: List[Tuple[float, float, int, int]], pick_n: int) -> List[Tuple[float, float, int, int]]:
def weighted_selection_without_replacement(items: List[Tuple], pick_n: int) -> List[Tuple]:
"""For a list of tuples where the second element is a weight, returns random items with those weights, without replacement."""
elt = [(math.log(random.random()) / item[1], item) for item in items] # magic
return [e[1] for e in heapq.nlargest(pick_n, elt)] # NB fine if too small
@@ -77,7 +77,7 @@ def ai_move(game: Game, ai_mode: str, ai_settings: Dict) -> Tuple[Move, GameNode
ai_thoughts += f"Playing policy-weighted random move {aimove.gtp()} ({policy_value:.1%})" + (
" because no other moves were found." if not top else f" because strategy is weighted (lower bound={lower_bound:.2%}, num moves > lb={len(weighted_coords)})."
)
elif "noise" in ai_mode:
elif "noise" in ai_mode: # DEPRECATED
noise_str = ai_settings["noise_strength"]
lower_bound = max(0, ai_settings["lower_bound"])
selected_policy_moves = [(pol, mv) for pol, mv in policy_moves if not mv.is_pass if pol > lower_bound]
@@ -113,8 +113,12 @@ def ai_move(game: Game, ai_mode: str, ai_settings: Dict) -> Tuple[Move, GameNode
(policy_grid[y][x], math.exp(-0.5 * ((x - mx) ** 2 + (y - my) ** 2) / var), x, y) for x in range(size[0]) for y in range(size[1]) if policy_grid[y][x] > 0
]
if "tenuki" in ai_mode:
weighted_coords = [(p, 1 - w, x, y) for p, w, x, y in weighted_coords]
ai_thoughts += f"Generated weights based on one minus gaussian with variance {var} around coordinates {mx},{my}. "
if cn.depth < ai_settings["endgame"] * size[0] * size[1]:
weighted_coords = [(p, 1 - w, x, y) for p, w, x, y in weighted_coords]
ai_thoughts += f"Generated weights based on one minus gaussian with variance {var} around coordinates {mx},{my}. "
else:
weighted_coords = [(p, 1, x, y) for p, w, x, y in weighted_coords]
ai_thoughts += f"Generated equal weights as move number >= {ai_settings['endgame'] * size[0] * size[1]}. "
else:
ai_thoughts += f"Generated weights based on gaussian with variance {var} around coordinates {mx},{my}. "
elif "pick" in ai_mode:
@@ -137,39 +141,40 @@ def ai_move(game: Game, ai_mode: str, ai_settings: Dict) -> Tuple[Move, GameNode
raise ValueError(f"Unknown AI mode {ai_mode}")
else: # Engine based move
candidate_ai_moves = cn.candidate_moves
if "balance" in ai_mode and candidate_ai_moves[0]["move"] != "pass": # don't play suicidal to balance score - pass when it's best
sign = cn.player_sign(cn.next_player)
sel_moves = [ # top move, or anything not too bad, or anything that makes you still ahead
move
for i, move in enumerate(candidate_ai_moves)
if i == 0
or move["visits"] >= ai_settings["min_visits"]
and (move["pointsLost"] < ai_settings["random_loss"] or move["pointsLost"] < ai_settings["max_loss"] and sign * move["scoreLead"] > ai_settings["target_score"])
]
aimove = Move.from_gtp(random.choice(sel_moves)["move"], player=cn.next_player)
ai_thoughts += f"Balance strategy selected moves {sel_moves} based on target score and max points lost, and randomly chose {aimove.gtp()}."
elif "jigo" in ai_mode and candidate_ai_moves[0]["move"] != "pass":
sign = cn.player_sign(cn.next_player)
jigo_move = min(candidate_ai_moves, key=lambda move: abs(sign * move["scoreLead"] - ai_settings["target_score"]))
aimove = Move.from_gtp(jigo_move["move"], player=cn.next_player)
ai_thoughts += f"Jigo strategy found {len(candidate_ai_moves)} candidate moves and chose {aimove.gtp()} as closest to 0.5 point win"
elif "scoreloss" in ai_mode and candidate_ai_moves[0]["move"] != "pass":
c = ai_settings["strength"]
moves = [[d['pointsLost'],math.exp(-c*d['pointsLost']),Move.from_gtp(d['move'])] for d in candidate_ai_moves]
topmove = weighted_selection_without_replacement(moves,1)[0]
aimove = topmove[2]
ai_thoughts += f"ScoreLoss strategy found {len(candidate_ai_moves)} candidate moves and chose {aimove.gtp()} (weight {topmove[1]}, point loss {topmove[0]}) as based on score weights."
top_cand = Move.from_gtp(candidate_ai_moves[0]["move"], player=cn.next_player)
if top_cand.is_pass: # don't play suicidal to balance score - pass when it's best
aimove = top_cand
ai_thoughts += f"Top move is pass, so passing regardless of strategy."
else:
if "default" not in ai_mode and "katago" not in ai_mode:
game.katrain.log(f"Unknown AI mode {ai_mode} or policy missing, using default.", OUTPUT_INFO)
ai_thoughts += f"Strategy {ai_mode} not found or unexpected fallback."
aimove = Move.from_gtp(candidate_ai_moves[0]["move"], player=cn.next_player)
ai_thoughts += f"Default strategy found {len(candidate_ai_moves)} moves returned from the engine and chose {aimove.gtp()} as top move"
if "balance" in ai_mode:
sign = cn.player_sign(cn.next_player)
sel_moves = [ # top move, or anything not too bad, or anything that makes you still ahead
move
for i, move in enumerate(candidate_ai_moves)
if i == 0
or move["visits"] >= ai_settings["min_visits"]
and (move["pointsLost"] < ai_settings["random_loss"] or move["pointsLost"] < ai_settings["max_loss"] and sign * move["scoreLead"] > ai_settings["target_score"])
]
aimove = Move.from_gtp(random.choice(sel_moves)["move"], player=cn.next_player)
ai_thoughts += f"Balance strategy selected moves {sel_moves} based on target score and max points lost, and randomly chose {aimove.gtp()}."
elif "jigo" in ai_mode:
sign = cn.player_sign(cn.next_player)
jigo_move = min(candidate_ai_moves, key=lambda move: abs(sign * move["scoreLead"] - ai_settings["target_score"]))
aimove = Move.from_gtp(jigo_move["move"], player=cn.next_player)
ai_thoughts += f"Jigo strategy found {len(candidate_ai_moves)} candidate moves (best {top_cand.gtp()}) and chose {aimove.gtp()} as closest to 0.5 point win"
elif "scoreloss" in ai_mode:
c = ai_settings["strength"]
moves = [(d["pointsLost"], math.exp(-c * max(0, d["pointsLost"])), Move.from_gtp(d["move"], player=cn.next_player)) for d in candidate_ai_moves]
topmove = weighted_selection_without_replacement(moves, 1)[0]
aimove = topmove[2]
ai_thoughts += f"ScoreLoss strategy found {len(candidate_ai_moves)} candidate moves (best {top_cand.gtp()}) and chose {aimove.gtp()} (weight {topmove[1]:.3f}, point loss {topmove[0]:.1f}) based on score weights."
else:
if "default" not in ai_mode and "katago" not in ai_mode:
game.katrain.log(f"Unknown AI mode {ai_mode} or policy missing, using default.", OUTPUT_INFO)
ai_thoughts += f"Strategy {ai_mode} not found or unexpected fallback."
aimove = top_cand
ai_thoughts += f"Default strategy found {len(candidate_ai_moves)} moves returned from the engine and chose {aimove.gtp()} as top move"
game.katrain.log(f"AI thoughts: {ai_thoughts}", OUTPUT_DEBUG)
try:
played_node = game.play(aimove)
played_node.ai_thoughts = ai_thoughts
return aimove, played_node
except IllegalMoveException as e:
game.katrain.log(f"AI Strategy {ai_mode} generated illegal move {aimove.gtp()}: {e}", OUTPUT_ERROR)
return None, None
played_node = game.play(aimove)
played_node.ai_thoughts = ai_thoughts
return aimove, played_node
+2 -2
View File
@@ -35,6 +35,7 @@ class KataGoEngine:
self.query_counter = 0
self.katago_process = None
self.base_priority = 0
self.override_settings = {}
self._lock = threading.Lock()
self.start()
self.analysis_thread = threading.Thread(target=self._analysis_read_thread, daemon=True).start()
@@ -162,7 +163,6 @@ class KataGoEngine:
"includeOwnership": ownership,
"includePolicy": not next_move,
"moves": [[m.player, m.gtp()] for m in moves],
"overrideSettings": {"maxTime": self.config["max_time"] if time_limit else 1000.0}
# "overrideSettings": {"playoutDoublingAdvantage": 3.0, "playoutDoublingAdvantagePla": 'BLACK' if not moves or moves[-1].player == 'W' else "WHITE"}
"overrideSettings": {"maxTime": self.config["max_time"] if time_limit else 1000.0, **self.override_settings},
}
self.send_query(query, callback, error_callback, next_move)
+1 -9
View File
@@ -229,7 +229,7 @@
valign: 'bottom'
halign: 'left'
text: '+0'
color: GREY
color: BUTTON_COLOR
size: self.texture_size
<ScoreGraph>:
@@ -271,18 +271,10 @@
id: range_label_top
pos: root.right_edge - self.width-1, root.pos[1]+root.height*(1-root.marginy) - self.font_size
text: 'B+' + str(int(root.y_scale))
# GraphMarkerLabel:
# font_size: 0.1 * root.height
# pos: root.right_edge - self.width-1, root.bhalf - self.font_size + 1
# text: 'B+' + str(int(root.y_scale/2))
GraphMarkerLabel:
font_size: 0.1 * root.height
pos: root.right_edge - self.width-1, root.mid - self.height/2 + 2
text: 'Jigo'
# GraphMarkerLabel:
# font_size: 0.1 * root.height
# pos: root.right_edge - self.width-1, root.whalf - 1
# text: 'W+' + str(int(root.y_scale/2))
GraphMarkerLabel:
font_size: 0.1 * root.height
pos: root.right_edge - self.width-1, root.pos[1]