scoreloss ai fix / p:noise ai removed / tenuki tweaked
This commit is contained in:
1 parent
e3b1491856
commit
abc21943b5
8 files changed
+68
-82
No files matched your search
@@ -84,20 +84,18 @@ while stronger players can pay more attention to smaller mistakes.
|
||||
Available AIs, with strength indicating an estimate for the default settings, are:
|
||||
|
||||
* **[9p+]** **Default** is full KataGo, above professional level.
|
||||
* **[~1d?]** **ScoreLoss** is KataGo making moves with probability `~ e^(-strength * points lost)`.
|
||||
* **Balance** is KataGo occasionally making weaker moves, attempting to win by ~2 points.
|
||||
* **Jigo** is KataGo aggressively making weaker moves, attempting to win by 0.5 points.
|
||||
* **[~4d]** **Policy** uses the top move from the policy network (it's 'shape sense' without reading), should be around high dan level depending on the model used. There is a setting to increase variety in the opening, but otherwise it plays deterministically.
|
||||
* **[~5k]**: **P:Weighted** picks a random move weighted by the policy, as long as it's above `lower_bound`. `weaken_fac` uses `policy^(1/weaken_fac)`, increasing the chance for weaker moves.
|
||||
* **[~2k]**: **P:Weighted** picks a random move weighted by the policy, as long as it's above `lower_bound`. `weaken_fac` uses `policy^(1/weaken_fac)`, increasing the chance for weaker moves.
|
||||
* **[~5k]**: **P:Pick** picks `pick_n + pick_frac * <number of legal moves>` moves at random, and play the best move among them.
|
||||
The setting `pick_override` determines the minimum value at which this process is bypassed to play the best move instead, preventing obvious blunders.
|
||||
This, along with 'Weighted' are probably the best choice for kyu players who want a chance of winning without playing the sillier bots below. Variants of this strategy include:
|
||||
* **[~5k]**: **P:Local** will pick such moves biased towards the last move with probability related to `local_stddev`.
|
||||
* **[~10k]**: **~P:Tenuki** is biased in the opposite way as P:Local, using the same setting.
|
||||
* **[~2k]**: **P:Local** will pick such moves biased towards the last move with probability related to `local_stddev`.
|
||||
* **[~10k]**: **P:Tenuki** is biased in the opposite way as P:Local, using the same setting.
|
||||
* **[~10k]**: **P:Influence** is biased towards 4th+ line moves, with every line below that dividing both the chance of considering the move and the policy value by `influence_weight`. Consider setting `pick_frac=1.0` to only affect the policy weight.
|
||||
* **[~10k]**: **P:Territory** is biased in the opposite way, towards 1-3rd line moves, using the same setting.
|
||||
* * **[~5k]**: **P:Noise** mixes the policy with `noise_strength` Dirichlet noise. At `noise_strength=0.9` play is near-random, while `noise_strength=0.7` is still quite strong. A threshold setting is included to avoid senseless first-line moves.
|
||||
|
||||
Selecting the AI as either white or black opens up the option to configure it under 'Configure AI'.
|
||||
|
||||
### Analysis
|
||||
|
||||
|
||||
+4
-8
@@ -40,14 +40,15 @@ ENGINE_SETTINGS = {
|
||||
"threads": 1,
|
||||
}
|
||||
|
||||
|
||||
engine = KataGoEngine(logger, ENGINE_SETTINGS)
|
||||
|
||||
with open("config.json") as f:
|
||||
settings = json.load(f)
|
||||
all_ai_settings = settings["ai"]
|
||||
|
||||
all_ai_settings["dev"] = all_ai_settings["P:Noise"]
|
||||
if bot == "dev":
|
||||
engine.override_settings["maxVisits"] = 500
|
||||
all_ai_settings["dev"] = all_ai_settings["ScoreLoss"]
|
||||
|
||||
ai_strategy = bot_strategy_names[bot]
|
||||
ai_settings = all_ai_settings[ai_strategy]
|
||||
@@ -149,12 +150,7 @@ while True:
|
||||
move = game.play(Move(None, player=game.next_player)).move
|
||||
else:
|
||||
move, node = ai_move(game, ai_strategy, ai_settings)
|
||||
if node is None:
|
||||
while node is None:
|
||||
logger.log(f"ERROR generating move, backing up with weighted.", OUTPUT_ERROR)
|
||||
move, node = ai_move(game, "p:weighted", {"pick_override": 1.0, "lower_bound": 0.001, "weaken_fac": 1})
|
||||
else:
|
||||
logger.log(f"Generated move {move}", OUTPUT_ERROR)
|
||||
logger.log(f"Generated move {move}", OUTPUT_ERROR)
|
||||
print(f"= {move.gtp()}\n")
|
||||
sys.stdout.flush()
|
||||
malkovich_analysis(game.current_node)
|
||||
|
||||
+6
-7
@@ -57,8 +57,7 @@ class AI:
|
||||
|
||||
def fix_settings(self):
|
||||
self.ai_settings = {**DEFAULT_AI_SETTINGS[self.strategy], **self.ai_settings}
|
||||
self.engine_settings = {**AI.DEFAULT_ENGINE_SETTINGS, **self.engine_settings,
|
||||
"threads": AI.NUM_THREADS}
|
||||
self.engine_settings = {**AI.DEFAULT_ENGINE_SETTINGS, **self.engine_settings, "threads": AI.NUM_THREADS}
|
||||
|
||||
def get_engine(self): # factory
|
||||
with AI.LOCK:
|
||||
@@ -98,8 +97,8 @@ def retrieve_ais(selected_ais):
|
||||
|
||||
test_ais = [
|
||||
# AI("Jigo", {}, {"max_visits": 100}),
|
||||
AI("Policy", {}, {'model': 'my/model.bin.gz'}),
|
||||
AI("Policy", {}, {'model': 'KataGo/models/b10-1.3.txt.gz'}),
|
||||
AI("Policy", {}, {"model": "my/model.bin.gz"}),
|
||||
AI("Policy", {}, {"model": "KataGo/models/b10-1.3.txt.gz"}),
|
||||
AI("Policy", {}),
|
||||
AI("P:Local", {}),
|
||||
AI("P:Pick", {}),
|
||||
@@ -128,7 +127,7 @@ def play_games(black: AI, white: AI):
|
||||
engines = {"B": black.get_engine(), "W": white.get_engine()}
|
||||
tag = f"{black.name} vs {white.name}"
|
||||
try:
|
||||
game = Game(Logger(), engines, {'init_size':BOARDSIZE})
|
||||
game = Game(Logger(), engines, {"init_size": BOARDSIZE})
|
||||
game.root.add_list_property("PW", [white.name])
|
||||
game.root.add_list_property("PB", [black.name])
|
||||
start_time = time.time()
|
||||
@@ -163,9 +162,9 @@ print(len(ais_to_test), "ais to test")
|
||||
global_start = time.time()
|
||||
|
||||
for n in range(N_GAMES):
|
||||
for _,e in AI.ENGINES: # no caching/replays
|
||||
for _, e in AI.ENGINES: # no caching/replays
|
||||
e.shutdown()
|
||||
AI.ENGINES=[]
|
||||
AI.ENGINES = []
|
||||
|
||||
with ThreadPoolExecutor(max_workers=16) as threadpool:
|
||||
for b in ais_to_test:
|
||||
|
||||
+4
-2
@@ -1,5 +1,6 @@
|
||||
bot_strategy_names = {
|
||||
"dev": "P:Noise",
|
||||
# "dev": "P:Noise",
|
||||
"dev": "ScoreLoss",
|
||||
"dev-beta": "P:Weighted",
|
||||
"strong": "Policy",
|
||||
"influence": "P:Influence",
|
||||
@@ -12,7 +13,8 @@ bot_strategy_names = {
|
||||
|
||||
|
||||
greetings = {
|
||||
"dev": "Policy+Dirichlet noise.",
|
||||
# "dev": "Policy+Dirichlet noise.",
|
||||
"dev": "Point loss-weighted random move.",
|
||||
"dev-beta": "Play a policy-weighted move.",
|
||||
"strong": "Play top policy move.",
|
||||
"influence": "Play an influential style.",
|
||||
|
||||
+4
-10
@@ -67,7 +67,7 @@
|
||||
"ScoreLoss": {
|
||||
"strength": 0.5,
|
||||
"_help_left": "Plays moves weighted inversely by point loss.",
|
||||
"_help_right": "Also affected by engine settings such as `max_visits`."
|
||||
"_help_right": "Also affected by engine settings such as `max_visits`, likely to play more varied/weaker with higher visits."
|
||||
},
|
||||
"Policy": {
|
||||
"opening_moves": 0.05,
|
||||
@@ -81,13 +81,6 @@
|
||||
"lower_bound": 0.001,
|
||||
"weaken_fac": 1.25
|
||||
},
|
||||
"P:Noise": {
|
||||
"pick_override": 0.95,
|
||||
"noise_strength": 0.6,
|
||||
"lower_bound": 0.001,
|
||||
"_help_left": "Adds `noise_strength` noise to the policy of all moved > 'lower_bound' and plays the top move.",
|
||||
"_help_right": "Plays top move if policy value is above `pick_override` to avoid obvious mistakes. Noise above 0.9 is near random, below 0.7 is fairly strong."
|
||||
},
|
||||
"P:Pick": {
|
||||
"pick_override": 0.95,
|
||||
"pick_n": 5,
|
||||
@@ -107,9 +100,10 @@
|
||||
"pick_override": 0.85,
|
||||
"stddev": 7.5,
|
||||
"pick_n": 5,
|
||||
"pick_frac": 0.7,
|
||||
"pick_frac": 0.5,
|
||||
"endgame": 0.5,
|
||||
"_help_left": "Samples `pick_n + pick_frac * <number of legal moves>` away from the last move and plays the best one.",
|
||||
"_help_right": "Increase `stddev` makes it prefer moves further away."
|
||||
"_help_right": "Increase `stddev` makes it prefer moves further away. Stops tenukiing after the 'endgame' fraction of the board is filled."
|
||||
},
|
||||
"P:Influence": {
|
||||
"pick_override": 0.95,
|
||||
|
||||
+43
-38
@@ -9,7 +9,7 @@ from core.engine import EngineDiedException
|
||||
from core.game import Game, GameNode, IllegalMoveException, Move
|
||||
|
||||
|
||||
def weighted_selection_without_replacement(items: List[Tuple[float, float, int, int]], pick_n: int) -> List[Tuple[float, float, int, int]]:
|
||||
def weighted_selection_without_replacement(items: List[Tuple], pick_n: int) -> List[Tuple]:
|
||||
"""For a list of tuples where the second element is a weight, returns random items with those weights, without replacement."""
|
||||
elt = [(math.log(random.random()) / item[1], item) for item in items] # magic
|
||||
return [e[1] for e in heapq.nlargest(pick_n, elt)] # NB fine if too small
|
||||
@@ -77,7 +77,7 @@ def ai_move(game: Game, ai_mode: str, ai_settings: Dict) -> Tuple[Move, GameNode
|
||||
ai_thoughts += f"Playing policy-weighted random move {aimove.gtp()} ({policy_value:.1%})" + (
|
||||
" because no other moves were found." if not top else f" because strategy is weighted (lower bound={lower_bound:.2%}, num moves > lb={len(weighted_coords)})."
|
||||
)
|
||||
elif "noise" in ai_mode:
|
||||
elif "noise" in ai_mode: # DEPRECATED
|
||||
noise_str = ai_settings["noise_strength"]
|
||||
lower_bound = max(0, ai_settings["lower_bound"])
|
||||
selected_policy_moves = [(pol, mv) for pol, mv in policy_moves if not mv.is_pass if pol > lower_bound]
|
||||
@@ -113,8 +113,12 @@ def ai_move(game: Game, ai_mode: str, ai_settings: Dict) -> Tuple[Move, GameNode
|
||||
(policy_grid[y][x], math.exp(-0.5 * ((x - mx) ** 2 + (y - my) ** 2) / var), x, y) for x in range(size[0]) for y in range(size[1]) if policy_grid[y][x] > 0
|
||||
]
|
||||
if "tenuki" in ai_mode:
|
||||
weighted_coords = [(p, 1 - w, x, y) for p, w, x, y in weighted_coords]
|
||||
ai_thoughts += f"Generated weights based on one minus gaussian with variance {var} around coordinates {mx},{my}. "
|
||||
if cn.depth < ai_settings["endgame"] * size[0] * size[1]:
|
||||
weighted_coords = [(p, 1 - w, x, y) for p, w, x, y in weighted_coords]
|
||||
ai_thoughts += f"Generated weights based on one minus gaussian with variance {var} around coordinates {mx},{my}. "
|
||||
else:
|
||||
weighted_coords = [(p, 1, x, y) for p, w, x, y in weighted_coords]
|
||||
ai_thoughts += f"Generated equal weights as move number >= {ai_settings['endgame'] * size[0] * size[1]}. "
|
||||
else:
|
||||
ai_thoughts += f"Generated weights based on gaussian with variance {var} around coordinates {mx},{my}. "
|
||||
elif "pick" in ai_mode:
|
||||
@@ -137,39 +141,40 @@ def ai_move(game: Game, ai_mode: str, ai_settings: Dict) -> Tuple[Move, GameNode
|
||||
raise ValueError(f"Unknown AI mode {ai_mode}")
|
||||
else: # Engine based move
|
||||
candidate_ai_moves = cn.candidate_moves
|
||||
if "balance" in ai_mode and candidate_ai_moves[0]["move"] != "pass": # don't play suicidal to balance score - pass when it's best
|
||||
sign = cn.player_sign(cn.next_player)
|
||||
sel_moves = [ # top move, or anything not too bad, or anything that makes you still ahead
|
||||
move
|
||||
for i, move in enumerate(candidate_ai_moves)
|
||||
if i == 0
|
||||
or move["visits"] >= ai_settings["min_visits"]
|
||||
and (move["pointsLost"] < ai_settings["random_loss"] or move["pointsLost"] < ai_settings["max_loss"] and sign * move["scoreLead"] > ai_settings["target_score"])
|
||||
]
|
||||
aimove = Move.from_gtp(random.choice(sel_moves)["move"], player=cn.next_player)
|
||||
ai_thoughts += f"Balance strategy selected moves {sel_moves} based on target score and max points lost, and randomly chose {aimove.gtp()}."
|
||||
elif "jigo" in ai_mode and candidate_ai_moves[0]["move"] != "pass":
|
||||
sign = cn.player_sign(cn.next_player)
|
||||
jigo_move = min(candidate_ai_moves, key=lambda move: abs(sign * move["scoreLead"] - ai_settings["target_score"]))
|
||||
aimove = Move.from_gtp(jigo_move["move"], player=cn.next_player)
|
||||
ai_thoughts += f"Jigo strategy found {len(candidate_ai_moves)} candidate moves and chose {aimove.gtp()} as closest to 0.5 point win"
|
||||
elif "scoreloss" in ai_mode and candidate_ai_moves[0]["move"] != "pass":
|
||||
c = ai_settings["strength"]
|
||||
moves = [[d['pointsLost'],math.exp(-c*d['pointsLost']),Move.from_gtp(d['move'])] for d in candidate_ai_moves]
|
||||
topmove = weighted_selection_without_replacement(moves,1)[0]
|
||||
aimove = topmove[2]
|
||||
ai_thoughts += f"ScoreLoss strategy found {len(candidate_ai_moves)} candidate moves and chose {aimove.gtp()} (weight {topmove[1]}, point loss {topmove[0]}) as based on score weights."
|
||||
top_cand = Move.from_gtp(candidate_ai_moves[0]["move"], player=cn.next_player)
|
||||
if top_cand.is_pass: # don't play suicidal to balance score - pass when it's best
|
||||
aimove = top_cand
|
||||
ai_thoughts += f"Top move is pass, so passing regardless of strategy."
|
||||
else:
|
||||
if "default" not in ai_mode and "katago" not in ai_mode:
|
||||
game.katrain.log(f"Unknown AI mode {ai_mode} or policy missing, using default.", OUTPUT_INFO)
|
||||
ai_thoughts += f"Strategy {ai_mode} not found or unexpected fallback."
|
||||
aimove = Move.from_gtp(candidate_ai_moves[0]["move"], player=cn.next_player)
|
||||
ai_thoughts += f"Default strategy found {len(candidate_ai_moves)} moves returned from the engine and chose {aimove.gtp()} as top move"
|
||||
if "balance" in ai_mode:
|
||||
sign = cn.player_sign(cn.next_player)
|
||||
sel_moves = [ # top move, or anything not too bad, or anything that makes you still ahead
|
||||
move
|
||||
for i, move in enumerate(candidate_ai_moves)
|
||||
if i == 0
|
||||
or move["visits"] >= ai_settings["min_visits"]
|
||||
and (move["pointsLost"] < ai_settings["random_loss"] or move["pointsLost"] < ai_settings["max_loss"] and sign * move["scoreLead"] > ai_settings["target_score"])
|
||||
]
|
||||
aimove = Move.from_gtp(random.choice(sel_moves)["move"], player=cn.next_player)
|
||||
ai_thoughts += f"Balance strategy selected moves {sel_moves} based on target score and max points lost, and randomly chose {aimove.gtp()}."
|
||||
elif "jigo" in ai_mode:
|
||||
sign = cn.player_sign(cn.next_player)
|
||||
jigo_move = min(candidate_ai_moves, key=lambda move: abs(sign * move["scoreLead"] - ai_settings["target_score"]))
|
||||
aimove = Move.from_gtp(jigo_move["move"], player=cn.next_player)
|
||||
ai_thoughts += f"Jigo strategy found {len(candidate_ai_moves)} candidate moves (best {top_cand.gtp()}) and chose {aimove.gtp()} as closest to 0.5 point win"
|
||||
elif "scoreloss" in ai_mode:
|
||||
c = ai_settings["strength"]
|
||||
moves = [(d["pointsLost"], math.exp(-c * max(0, d["pointsLost"])), Move.from_gtp(d["move"], player=cn.next_player)) for d in candidate_ai_moves]
|
||||
topmove = weighted_selection_without_replacement(moves, 1)[0]
|
||||
aimove = topmove[2]
|
||||
ai_thoughts += f"ScoreLoss strategy found {len(candidate_ai_moves)} candidate moves (best {top_cand.gtp()}) and chose {aimove.gtp()} (weight {topmove[1]:.3f}, point loss {topmove[0]:.1f}) based on score weights."
|
||||
else:
|
||||
if "default" not in ai_mode and "katago" not in ai_mode:
|
||||
game.katrain.log(f"Unknown AI mode {ai_mode} or policy missing, using default.", OUTPUT_INFO)
|
||||
ai_thoughts += f"Strategy {ai_mode} not found or unexpected fallback."
|
||||
aimove = top_cand
|
||||
ai_thoughts += f"Default strategy found {len(candidate_ai_moves)} moves returned from the engine and chose {aimove.gtp()} as top move"
|
||||
game.katrain.log(f"AI thoughts: {ai_thoughts}", OUTPUT_DEBUG)
|
||||
try:
|
||||
played_node = game.play(aimove)
|
||||
played_node.ai_thoughts = ai_thoughts
|
||||
return aimove, played_node
|
||||
except IllegalMoveException as e:
|
||||
game.katrain.log(f"AI Strategy {ai_mode} generated illegal move {aimove.gtp()}: {e}", OUTPUT_ERROR)
|
||||
return None, None
|
||||
played_node = game.play(aimove)
|
||||
played_node.ai_thoughts = ai_thoughts
|
||||
return aimove, played_node
|
||||
+2
-2
@@ -35,6 +35,7 @@ class KataGoEngine:
|
||||
self.query_counter = 0
|
||||
self.katago_process = None
|
||||
self.base_priority = 0
|
||||
self.override_settings = {}
|
||||
self._lock = threading.Lock()
|
||||
self.start()
|
||||
self.analysis_thread = threading.Thread(target=self._analysis_read_thread, daemon=True).start()
|
||||
@@ -162,7 +163,6 @@ class KataGoEngine:
|
||||
"includeOwnership": ownership,
|
||||
"includePolicy": not next_move,
|
||||
"moves": [[m.player, m.gtp()] for m in moves],
|
||||
"overrideSettings": {"maxTime": self.config["max_time"] if time_limit else 1000.0}
|
||||
# "overrideSettings": {"playoutDoublingAdvantage": 3.0, "playoutDoublingAdvantagePla": 'BLACK' if not moves or moves[-1].player == 'W' else "WHITE"}
|
||||
"overrideSettings": {"maxTime": self.config["max_time"] if time_limit else 1000.0, **self.override_settings},
|
||||
}
|
||||
self.send_query(query, callback, error_callback, next_move)
|
||||
+1
-9
@@ -229,7 +229,7 @@
|
||||
valign: 'bottom'
|
||||
halign: 'left'
|
||||
text: '+0'
|
||||
color: GREY
|
||||
color: BUTTON_COLOR
|
||||
size: self.texture_size
|
||||
|
||||
<ScoreGraph>:
|
||||
@@ -271,18 +271,10 @@
|
||||
id: range_label_top
|
||||
pos: root.right_edge - self.width-1, root.pos[1]+root.height*(1-root.marginy) - self.font_size
|
||||
text: 'B+' + str(int(root.y_scale))
|
||||
# GraphMarkerLabel:
|
||||
# font_size: 0.1 * root.height
|
||||
# pos: root.right_edge - self.width-1, root.bhalf - self.font_size + 1
|
||||
# text: 'B+' + str(int(root.y_scale/2))
|
||||
GraphMarkerLabel:
|
||||
font_size: 0.1 * root.height
|
||||
pos: root.right_edge - self.width-1, root.mid - self.height/2 + 2
|
||||
text: 'Jigo'
|
||||
# GraphMarkerLabel:
|
||||
# font_size: 0.1 * root.height
|
||||
# pos: root.right_edge - self.width-1, root.whalf - 1
|
||||
# text: 'W+' + str(int(root.y_scale/2))
|
||||
GraphMarkerLabel:
|
||||
font_size: 0.1 * root.height
|
||||
pos: root.right_edge - self.width-1, root.pos[1]
|
||||
|
||||
Reference in new issue
Block a user