diff --git a/README.md b/README.md index 7a318be..1d0649b 100644 --- a/README.md +++ b/README.md @@ -68,9 +68,12 @@ Allowed values: `safe`, `balanced`, `aggressive`. ## PrismBattleSnake_GPT_5_6_Sol `PrismBattleSnake_GPT_5_6_Sol` is a separate snake that keeps Apex's strategy while -accelerating hot spatial operations with a Python-integer bitboard engine. Its -filename, class, and registry key include the model name, while its public -Battlesnake API name remains `PrismBattleSnake`. +accelerating hot spatial operations with a Python-integer bitboard engine. It +also shares duel transpositions across candidate moves, uses principal-variation +ordering and aspiration windows, and runs a compact adversarial multiplayer +rollout with simultaneous enemy responses. Its filename, class, and registry +key include the model name, while its public Battlesnake API name remains +`PrismBattleSnake`. Run it with: ```sh @@ -87,6 +90,17 @@ python scripts/benchmark_snakes_from_db.py \ The benchmark opens SQLite read-only and reports mean, median, p95, and maximum move latency. Increase `--samples` for a broader but slower comparison. +Run the deterministic CI-friendly arena benchmark without a gameplay database: +```sh +just bench-snake-arena positions=100 +``` + +It reports latency, reached minimax depth, and move disagreements between Apex +and Prism. Add `output=data/arena-report.json` to save a machine-readable report. +For representative strategy evaluation, provide recorded positions to +`scripts/benchmark_snake_arena.py --database /path/to/gameplay.sqlite3`, then run +paired seeded games with the local Battlesnake CLI to measure win rate. + ## Compact gameplay database New gameplay turns use normalized storage: the turn row stores food, hazards, move, and thinking data once; snake identity is stored once per game in diff --git a/justfile b/justfile index 95aa2c2..81c4bd5 100644 --- a/justfile +++ b/justfile @@ -62,6 +62,14 @@ bench-best-snake iterations="1000": PYTHONPATH="{{justfile_directory()}}" python "{{justfile_directory()}}/tests/bench_best_battle_snake.py" --iterations "{{iterations}}" +bench-snake-arena positions="100" output="": + #!/usr/bin/env bash + set -euo pipefail + + args=(--positions "{{positions}}") + if [ -n "{{output}}" ]; then args+=(--json-output "{{output}}"); fi + PYTHONPATH="{{justfile_directory()}}" python "{{justfile_directory()}}/scripts/benchmark_snake_arena.py" "${args[@]}" + build-battlesnake-cli: #!/usr/bin/env bash set -euo pipefail diff --git a/scripts/benchmark_snake_arena.py b/scripts/benchmark_snake_arena.py new file mode 100755 index 0000000..09e9a89 --- /dev/null +++ b/scripts/benchmark_snake_arena.py @@ -0,0 +1,117 @@ +#!/usr/bin/env python3 +"""Compare snake decisions and latency on deterministic synthetic positions. + +For outcome/win-rate tournaments use the local Battlesnake CLI. This harness is +fast enough for CI and detects move disagreements, crashes, and latency changes. +""" + +from __future__ import annotations + +import argparse +import json +import sys +from pathlib import Path +from statistics import mean, median +from time import perf_counter + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) + +from scripts.benchmark_snakes_from_db import load_states, percentile +from server.GameBoard import GameBoard +from snakes import SnakeBuilder +from tests.bench_best_battle_snake import build_game_state + +def synthetic_states(count: int) -> list[tuple[dict, dict]]: + states: list[tuple[dict, dict]] = [] + for index in range(count): + payload = build_game_state() + payload["game"]["id"] = f"arena-{index}" + payload["turn"] = 20 + index + # Rotate food deterministically to exercise routing without creating + # invalid bodies or relying on private/real network values. + payload["board"]["food"] = [ + {"x": 1 + index % 3, "y": 9}, + {"x": 9, "y": 1 + (index // 3) % 3}, + ] + states.append((payload["board"], { + "game_id": payload["game"]["id"], + "source": "custom", "map": "standard", + "ruleset": payload["game"]["ruleset"], "turn": payload["turn"], + "you": payload["you"], + })) + return states + +def evaluate(name: str, states: list[tuple[dict, dict]]) -> tuple[list[str], dict]: + moves: list[str] = [] + durations: list[float] = [] + depths: list[int] = [] + for index, (board_data, metadata) in enumerate(states): + snake = SnakeBuilder.build(name) + game_id = f"arena-{name}-{index}-{metadata['game_id']}" + board = GameBoard( + game_id=game_id, width=board_data["width"], height=board_data["height"], + ruleset=metadata["ruleset"], source=metadata["source"], + map=metadata["map"], snake_class=snake, + ) + board.read_game_data({ + "game": { + "id": game_id, "ruleset": metadata["ruleset"], + "source": metadata["source"], "map": metadata["map"], "timeout": 500, + }, + "turn": metadata["turn"], "board": board_data, "you": metadata["you"], + }) + started = perf_counter() + moves.append(snake.choose_move(board)) + durations.append((perf_counter() - started) * 1000.0) + history = snake.get_history() if hasattr(snake, "get_history") else [] + if history: + depths.append(int(history[-1].get("minimax_depth_reached", 0))) + return moves, { + "snake": name, "positions": len(states), + "mean_ms": mean(durations), "median_ms": median(durations), + "p95_ms": percentile(durations, 0.95), "max_ms": max(durations), + "mean_minimax_depth": mean(depths) if depths else 0.0, + } + +def main() -> None: + parser = argparse.ArgumentParser() + parser.add_argument("--snake", action="append", default=[]) + parser.add_argument("--database") + parser.add_argument("--positions", type=int, default=100) + parser.add_argument("--stride", type=int, default=997) + parser.add_argument("--json-output") + args = parser.parse_args() + + states = ( + load_states(args.database, max(1, args.positions), max(1, args.stride)) + if args.database else synthetic_states(max(1, args.positions)) + ) + if not states: + raise SystemExit("No benchmark positions found") + + names = args.snake or ["ApexBattleSnake", "PrismBattleSnake_GPT_5_6_Sol"] + move_sets: dict[str, list[str]] = {} + reports: list[dict] = [] + for name in names: + moves, report = evaluate(name, states) + move_sets[name] = moves + reports.append(report) + print( + f"{name}: mean={report['mean_ms']:.3f} ms, " + f"p95={report['p95_ms']:.3f} ms, max={report['max_ms']:.3f} ms, " + f"depth={report['mean_minimax_depth']:.2f}" + ) + baseline = names[0] + disagreements = { + name: sum(a != b for a, b in zip(move_sets[baseline], move_sets[name])) + for name in names[1:] + } + if disagreements: + print(f"Move disagreements versus {baseline}: {disagreements}") + + payload = {"reports": reports, "baseline": baseline, "disagreements": disagreements} + if args.json_output: + Path(args.json_output).write_text(json.dumps(payload, indent=2) + "\n") + +if __name__ == "__main__": + main() diff --git a/snakes/PrismBattleSnake_GPT_5_6_Sol.py b/snakes/PrismBattleSnake_GPT_5_6_Sol.py index fd8fb71..bbeab60 100644 --- a/snakes/PrismBattleSnake_GPT_5_6_Sol.py +++ b/snakes/PrismBattleSnake_GPT_5_6_Sol.py @@ -26,23 +26,27 @@ Key speedups: S12: Duel minimax uses tuple bodies and bitboard move generation. S13: Iterative deepening reuses a transposition table and move-order hints. S14: Candidate duel moves and enemy replies resolve on the same root turn. + S15: Candidate moves share one duel transposition/search context per turn. + S16: Compact adversarial multiplayer rollout advances plausible enemy replies. + S17: Rollout memoization and adaptive depth spend time on ambiguous positions. """ from __future__ import annotations -from typing import Any + from time import perf_counter +from server.GameBoard import GameBoard from snakes.ApexBattleSnake import ApexBattleSnake from snakes.bitboard import BitBoard from snakes.bitboard_duel_search import BitboardDuelSearch -from server.GameBoard import GameBoard +from snakes.compact_survival_search import CompactSurvivalSearch # Direction offsets for coord-dict → tuple conversion _DIR_DELTAS = ((0, 1), (0, -1), (-1, 0), (1, 0)) _DIR_NAMES = ("up", "down", "left", "right") class PrismBattleSnake_GPT_5_6_Sol(ApexBattleSnake): - VERSION = "1.1.0" + VERSION = "1.2.0" def __init__(self) -> None: super().__init__() @@ -60,6 +64,11 @@ class PrismBattleSnake_GPT_5_6_Sol(ApexBattleSnake): self._enemy_attack_danger: int = 0 # tiles where enemy len >= our len self._enemy_attack_opportunity: int = 0 # tiles where enemy len < our len + # Shared per-turn search contexts. Candidate moves overlap heavily, so + # rebuilding their transposition tables wastes most iterative-deepening work. + self._duel_search_context: BitboardDuelSearch | None = None + self._survival_search_context: CompactSurvivalSearch | None = None + # ── BitBoard accessor ──────────────────────────────────────────────────── def _get_bb(self, width: int, height: int) -> BitBoard: @@ -78,6 +87,8 @@ class PrismBattleSnake_GPT_5_6_Sol(ApexBattleSnake): def choose_move(self, game_data: GameBoard) -> str: bb = self._get_bb(game_data.get_width(), game_data.get_height()) + self._duel_search_context = None + self._survival_search_context = None # S9: precompute enemy body / tail / attack bitboards for survival tree other_snakes = game_data.get_other_snakes() @@ -128,7 +139,17 @@ class PrismBattleSnake_GPT_5_6_Sol(ApexBattleSnake): self._enemy_attack_danger = enemy_attack_danger self._enemy_attack_opportunity = enemy_attack_opportunity - return super().choose_move(game_data) + move = super().choose_move(game_data) + history = self.get_history() + if history: + thinking = history[-1] + if self._duel_search_context is not None: + thinking["prism_duel_nodes"] = self._duel_search_context.nodes + thinking["prism_duel_cache_hits"] = self._duel_search_context.cache_hits + if self._survival_search_context is not None: + thinking["prism_rollout_nodes"] = self._survival_search_context.nodes + thinking["prism_rollout_cache_hits"] = self._survival_search_context.cache_hits + return move # ── S1: Bitboard flood fill ────────────────────────────────────────────── @@ -260,14 +281,16 @@ class PrismBattleSnake_GPT_5_6_Sol(ApexBattleSnake): self, food_set: set, hazard_set: set, hazard_count: dict, hazard_damage: int, width: int, height: int, deadline: float | None, ) -> BitboardDuelSearch: - return BitboardDuelSearch( - board=self._get_bb(width, height), - food=food_set, - hazards=hazard_set, - hazard_count=hazard_count, - hazard_damage=hazard_damage, - deadline=deadline, - ) + if self._duel_search_context is None: + self._duel_search_context = BitboardDuelSearch( + board=self._get_bb(width, height), + food=food_set, + hazards=hazard_set, + hazard_count=hazard_count, + hazard_damage=hazard_damage, + deadline=deadline, + ) + return self._duel_search_context def _minimax_candidate_id( self, my_body: list, enemy_body: list, my_target: tuple[int, int], @@ -281,13 +304,19 @@ class PrismBattleSnake_GPT_5_6_Sol(ApexBattleSnake): food_set, hazard_set, hazard_count, hazard_damage, width, height, deadline, ) + adaptive_depth = max_depth + remaining = self._remaining_ms(deadline) + if remaining > 250: + adaptive_depth = min(7, max_depth + 1) + elif remaining < 120: + adaptive_depth = min(max_depth, 2) return search.search_candidate( my_body=my_body, enemy_body=enemy_body, my_target=my_target, my_health=my_health, enemy_health=enemy_health, - max_depth=max_depth, + max_depth=adaptive_depth, previous_hazards=previous_hazard_set if previous_hazard_set is not None else hazard_set, ) @@ -331,7 +360,51 @@ class PrismBattleSnake_GPT_5_6_Sol(ApexBattleSnake): previous_hazards=previous_hazard_set if previous_hazard_set is not None else hazard_set, ) - # ── S9: Optimised survival tree (bitboard-native) ──────────────────────── + # ── S16/S17: compact adversarial survival rollout ─────────────────────── + + def _future_rollout_bonus( + self, move: str, safe_moves: dict, my_body: list, other_snakes: list, + food_set: set, is_constrictor: bool, width: int, height: int, + enemy_can_grow: dict, deadline: float | None, + ) -> float: + pos = safe_moves.get(move) + if pos is None: + return -250.0 + # Duel minimax already advances the opponent exactly. Keep the much faster + # bitboard-native solo rollout here instead of paying for the same response + # model twice. Constrictor and multiplayer still use adversarial rollouts. + if len(other_snakes) == 1 and not is_constrictor: + return super()._future_rollout_bonus( + move, safe_moves, my_body, other_snakes, food_set, is_constrictor, + width, height, enemy_can_grow, deadline, + ) + if self._survival_search_context is None: + remaining = self._remaining_ms(deadline) + enemy_branch = 2 if len(other_snakes) <= 2 and remaining > 100 else 1 + self._survival_search_context = CompactSurvivalSearch( + board=self._get_bb(width, height), + food=food_set, + is_constrictor=is_constrictor, + deadline=deadline, + branch=self._planning_branch, + enemy_branch=enemy_branch, + response_cap=8 if remaining > 150 else 4, + ) + remaining = self._remaining_ms(deadline) + depth = min(self._planning_depth, 2 if len(other_snakes) > 1 else 3) + if remaining < 90: + depth = min(depth, 2) + elif remaining > 250 and len(other_snakes) <= 2: + depth = min(4, depth + 1) + raw = self._survival_search_context.search_selected( + my_body=my_body, + enemies=other_snakes, + target=(pos["x"], pos["y"]), + depth=depth, + ) + return raw * 0.15 + + # ── S9: Optimised survival tree (compatibility fallback) ──────────────── def _future_position_score( self, my_body: list, other_snakes: list, food_set: set, is_constrictor: bool, @@ -420,7 +493,6 @@ class PrismBattleSnake_GPT_5_6_Sol(ApexBattleSnake): bb = self._bb w = bb.width - h = bb.height head = my_body[0] hx, hy = head["x"], head["y"] head_idx = hy * w + hx diff --git a/snakes/__init__.py b/snakes/__init__.py index f48aef8..34685e4 100644 --- a/snakes/__init__.py +++ b/snakes/__init__.py @@ -11,7 +11,7 @@ SNAKE_REGISTRY = { "UltimateBattleSnake": "4.5.0", "ApexBattleSnake": "1.0.0", "SupremeBattleSnake_ClaudeOpus4_6": "1.0.0", - "PrismBattleSnake_GPT_5_6_Sol": "1.1.0", + "PrismBattleSnake_GPT_5_6_Sol": "1.2.0", } DEFAULT_SNAKE_CONFIG = { diff --git a/snakes/bitboard.py b/snakes/bitboard.py index 8feac3b..b37e85d 100644 --- a/snakes/bitboard.py +++ b/snakes/bitboard.py @@ -326,15 +326,15 @@ class BitBoard: # index when several foods are equally close, which can change contested- # food scoring and therefore the selected move. queue = [start_idx] - distances = [0] seen = start_bit cursor = 0 + layer_end = 1 + dist = 0 w = self.width size = self.size while cursor < len(queue): cell = queue[cursor] - dist = distances[cursor] cursor += 1 x = cell % w candidates = ( @@ -357,6 +357,8 @@ class BitBoard: return dist + 1, neighbor seen |= bit queue.append(neighbor) - distances.append(dist + 1) + if cursor == layer_end: + dist += 1 + layer_end = len(queue) return None, None diff --git a/snakes/bitboard_duel_search.py b/snakes/bitboard_duel_search.py index 2c6d847..1732d65 100644 --- a/snakes/bitboard_duel_search.py +++ b/snakes/bitboard_duel_search.py @@ -2,9 +2,9 @@ from __future__ import annotations +from collections.abc import Iterable from dataclasses import dataclass from time import perf_counter -from typing import Iterable from snakes.bitboard import BitBoard @@ -47,9 +47,10 @@ class BitboardDuelSearch: self.hazard_stacks = { board.idx(x, y): count for (x, y), count in hazard_count.items() } - self.transposition: dict[tuple[DuelState, int], tuple[float, str]] = {} + self.transposition: dict[tuple[DuelState, int], tuple[float, str, int | None]] = {} self.killer_moves: dict[int, int] = {} self.history: dict[int, int] = {} + self._body_bits_cache: dict[Body, int] = {} self.nodes = 0 self.cache_hits = 0 @@ -79,7 +80,11 @@ class BitboardDuelSearch: for depth in range(1, max_depth + 1): if self._out_of_time(5.0): break - value, completed = self._search(state, depth, -float("inf"), float("inf")) + window = 80.0 if completed_depth else float("inf") + alpha, beta = result - window, result + window + value, completed = self._search(state, depth, alpha, beta) + if completed and window != float("inf") and (value <= alpha or value >= beta): + value, completed = self._search(state, depth, -float("inf"), float("inf")) if not completed: break result = value @@ -119,9 +124,13 @@ class BitboardDuelSearch: for depth in range(1, max_depth + 1): if self._out_of_time(5.0): break - value, completed = self._search_selected_move( - state, target_idx, depth, -float("inf"), float("inf") - ) + window = 80.0 if completed_depth else float("inf") + alpha, beta = result - window, result + window + value, completed = self._search_selected_move(state, target_idx, depth, alpha, beta) + if completed and window != float("inf") and (value <= alpha or value >= beta): + value, completed = self._search_selected_move( + state, target_idx, depth, -float("inf"), float("inf") + ) if not completed: break result = value @@ -200,7 +209,7 @@ class BitboardDuelSearch: cached = self.transposition.get(cache_key) if cached is not None: self.cache_hits += 1 - cached_value, bound = cached + cached_value, bound, preferred_move = cached if bound == "exact": return cached_value, True if bound == "lower": @@ -217,9 +226,10 @@ class BitboardDuelSearch: if not enemy_moves: return self.WIN + depth, True - my_moves = self._ordered_moves(my_moves, state, depth, True) + my_moves = self._ordered_moves(my_moves, state, depth, True, preferred_move if cached is not None else None) enemy_moves = self._ordered_moves(enemy_moves, state, depth, False) best = -float("inf") + best_move: int | None = None for my_target in my_moves: worst = float("inf") @@ -240,7 +250,9 @@ class BitboardDuelSearch: self.history[my_target] = self.history.get(my_target, 0) + depth * depth break - best = max(best, worst) + if worst > best: + best = worst + best_move = my_target alpha = max(alpha, best) if alpha >= beta: break @@ -251,7 +263,7 @@ class BitboardDuelSearch: bound = "lower" else: bound = "exact" - self.transposition[cache_key] = (best, bound) + self.transposition[cache_key] = (best, bound, best_move) return best, True def _advance(self, state: DuelState, my_target: int, enemy_target: int) -> tuple[DuelState, float | None]: @@ -308,7 +320,10 @@ class BitboardDuelSearch: """ return list(self._iter_bits(self.board.neighbors_of(body[0]))) - def _ordered_moves(self, moves: list[int], state: DuelState, depth: int, mine: bool) -> list[int]: + def _ordered_moves( + self, moves: list[int], state: DuelState, depth: int, mine: bool, + preferred: int | None = None, + ) -> list[int]: body = state.my_body if mine else state.enemy_body other = state.enemy_body if mine else state.my_body killer = self.killer_moves.get(depth) @@ -320,8 +335,9 @@ class BitboardDuelSearch: food_bonus = 200.0 if (1 << target) & state.food_bits else 0.0 space = self.board.flood_count(target, (self._body_bits(body[1:]) | self._body_bits(other[1:])) & ~(1 << target)) center = -(abs(x - center_x) + abs(y - center_y)) + preferred_bonus = 20_000.0 if target == preferred else 0.0 killer_bonus = 10_000.0 if target == killer else 0.0 - return killer_bonus + self.history.get(target, 0) + food_bonus + space * 2.0 + center, -target + return preferred_bonus + killer_bonus + self.history.get(target, 0) + food_bonus + space * 2.0 + center, -target # Our strongest-looking moves first; enemy ordering uses the same quality # estimate because dangerous enemy replies tend to gain space and food. @@ -329,13 +345,25 @@ class BitboardDuelSearch: def _evaluate(self, state: DuelState) -> float: my_blocked = self._body_bits(state.my_body[1:]) | self._body_bits(state.enemy_body[1:]) - my_space = self.board.flood_count(state.my_body[0], my_blocked) - enemy_space = self.board.flood_count(state.enemy_body[0], my_blocked) - my_liberties = self.board.open_neighbor_count(state.my_body[0], my_blocked) - enemy_liberties = self.board.open_neighbor_count(state.enemy_body[0], my_blocked) - length_score = (len(state.my_body) - len(state.enemy_body)) * 18.0 - health_score = (state.my_health - state.enemy_health) * 0.15 - return (my_space - enemy_space) * 2.0 + (my_liberties - enemy_liberties) * 12.0 + length_score + health_score + my_head, enemy_head = state.my_body[0], state.enemy_body[0] + my_space = self.board.flood_count(my_head, my_blocked) + enemy_space = self.board.flood_count(enemy_head, my_blocked) + my_liberties = self.board.open_neighbor_count(my_head, my_blocked) + enemy_liberties = self.board.open_neighbor_count(enemy_head, my_blocked) + territory = self.board.territory(my_head, [enemy_head], my_blocked) + my_tail_path = self.board.path_distance(my_head, state.my_body[-1], my_blocked) + enemy_tail_path = self.board.path_distance(enemy_head, state.enemy_body[-1], my_blocked) + tail_score = (12.0 if my_tail_path is not None else -24.0) - (12.0 if enemy_tail_path is not None else -24.0) + my_hazard = self._hazard_cost(my_head, state.previous_hazard_bits) + enemy_hazard = self._hazard_cost(enemy_head, state.previous_hazard_bits) + length_score = (len(state.my_body) - len(state.enemy_body)) * 20.0 + health_score = (state.my_health - state.enemy_health) * 0.18 + forced_score = (my_liberties > 1) * 10.0 - (enemy_liberties > 1) * 10.0 + return ( + (my_space - enemy_space) * 1.5 + territory * 1.2 + + (my_liberties - enemy_liberties) * 14.0 + length_score + health_score + + tail_score + forced_score + (enemy_hazard - my_hazard) * 0.8 + ) def _hazard_cost(self, target: int, previous_hazard_bits: int) -> int: bit = 1 << target @@ -351,11 +379,15 @@ class BitboardDuelSearch: def _tail_stacked(body: Body) -> bool: return len(body) >= 2 and body[-1] == body[-2] - @staticmethod - def _body_bits(body: Body) -> int: + def _body_bits(self, body: Body) -> int: + cached = self._body_bits_cache.get(body) + if cached is not None: + return cached bits = 0 for cell in body: bits |= 1 << cell + if len(self._body_bits_cache) < 16_384: + self._body_bits_cache[body] = bits return bits @staticmethod diff --git a/snakes/compact_survival_search.py b/snakes/compact_survival_search.py new file mode 100644 index 0000000..86fbd01 --- /dev/null +++ b/snakes/compact_survival_search.py @@ -0,0 +1,263 @@ +"""Compact adversarial rollout for multiplayer Battlesnake positions.""" + +from __future__ import annotations + +from itertools import product +from time import perf_counter + +from snakes.bitboard import BitBoard + +Body = tuple[int, ...] +EnemyBodies = tuple[Body, ...] +StateKey = tuple[Body, EnemyBodies, int, int] + +class CompactSurvivalSearch: + """Small paranoid beam search with simultaneous enemy responses. + + It is deliberately narrower than full multiplayer minimax: each enemy keeps + only its most dangerous replies and the combined response beam is capped. + This models moving opponents without exhausting the request deadline. + """ + + DEATH = -5000.0 + + def __init__( + self, + board: BitBoard, + food: set[tuple[int, int]], + is_constrictor: bool, + deadline: float | None, + branch: int, + enemy_branch: int = 2, + response_cap: int = 8, + ) -> None: + self.board = board + self.food_bits = board.set_to_bits(food) + self.is_constrictor = is_constrictor + self.deadline = deadline + self.branch = max(1, branch) + self.enemy_branch = max(1, enemy_branch) + self.response_cap = max(1, response_cap) + self.cache: dict[StateKey, float] = {} + self.body_bits_cache: dict[Body, int] = {} + self.nodes = 0 + self.cache_hits = 0 + + def body_from_dicts(self, body: list[dict]) -> Body: + return tuple(self.board.idx(segment["x"], segment["y"]) for segment in body) + + def search_selected( + self, + my_body: list[dict], + enemies: list[dict], + target: tuple[int, int], + depth: int, + ) -> float: + mine = self.body_from_dicts(my_body) + enemy_bodies = tuple(self.body_from_dicts(enemy["body"]) for enemy in enemies) + target_idx = self.board.idx(*target) + if not self.board.neighbors_of(mine[0]) & (1 << target_idx): + return self.DEATH + return self._selected_root(mine, enemy_bodies, self.food_bits, target_idx, depth) + + def _selected_root( + self, mine: Body, enemies: EnemyBodies, food_bits: int, target: int, depth: int, + ) -> float: + replies = self._enemy_responses(enemies, mine, target, food_bits) + if not replies: + replies = [()] + worst = float("inf") + for response in replies: + if self._out_of_time(): + break + child = self._advance(mine, enemies, target, response, food_bits) + if child is None: + value = self.DEATH + else: + next_mine, next_enemies, next_food = child + value = self._evaluate(next_mine, next_enemies) + if depth > 1 and value > self.DEATH: + value += self._search(next_mine, next_enemies, next_food, depth - 1) * 0.72 + worst = min(worst, value) + return self._evaluate(mine, enemies) if worst == float("inf") else worst + + def _search(self, mine: Body, enemies: EnemyBodies, food_bits: int, depth: int) -> float: + self.nodes += 1 + if self._out_of_time() or depth <= 0: + return 0.0 + key = (mine, enemies, food_bits, depth) + cached = self.cache.get(key) + if cached is not None: + self.cache_hits += 1 + return cached + + occupied = self._occupied(mine, enemies) + targets = list(self._iter_bits(self.board.neighbors_of(mine[0]))) + ranked: list[tuple[float, int]] = [] + for target in targets: + # Collision legality is finalized simultaneously because eating controls + # whether tails vacate. + ate = bool((1 << target) & food_bits) + own_tail_blocked = self.is_constrictor or ate + body_blocked = self._body_bits(mine if own_tail_blocked else mine[:-1]) + enemy_blocked = 0 + for enemy in enemies: + enemy_blocked |= self._body_bits(enemy[:-1] if not self.is_constrictor else enemy) + if (1 << target) & (body_blocked | enemy_blocked): + continue + free_space = self.board.flood_count(target, occupied & ~(1 << target)) + ranked.append((free_space + (20 if ate else 0), target)) + ranked.sort(reverse=True) + + if not ranked: + return self.DEATH + + best = self.DEATH + for _, target in ranked[:self.branch]: + replies = self._enemy_responses(enemies, mine, target, food_bits) or [()] + worst = float("inf") + for response in replies: + if self._out_of_time(): + break + child = self._advance(mine, enemies, target, response, food_bits) + if child is None: + value = self.DEATH + else: + next_mine, next_enemies, next_food = child + value = self._evaluate(next_mine, next_enemies) + if depth > 1 and value > self.DEATH: + value += self._search(next_mine, next_enemies, next_food, depth - 1) * 0.72 + worst = min(worst, value) + if worst != float("inf"): + best = max(best, worst) + + if not self._out_of_time() and len(self.cache) < 16_384: + self.cache[key] = best + return best + + def _enemy_responses( + self, enemies: EnemyBodies, mine: Body, my_target: int, food_bits: int, + ) -> list[tuple[int, ...]]: + if not enemies: + return [] + choices: list[list[int]] = [] + my_length_after = len(mine) + int(bool((1 << my_target) & food_bits)) + for enemy in enemies: + ranked: list[tuple[float, int]] = [] + for target in self._iter_bits(self.board.neighbors_of(enemy[0])): + ate = bool((1 << target) & food_bits) + enemy_length_after = len(enemy) + int(ate) + score = 0.0 + if target == my_target: + score += 1000.0 if enemy_length_after >= my_length_after else -1000.0 + tx, ty = self.board.coord(target) + mx, my = self.board.coord(my_target) + score -= abs(tx - mx) + abs(ty - my) + score += self.board.open_neighbor_count(target, self._occupied(mine, enemies)) * 3.0 + score += 20.0 if ate else 0.0 + ranked.append((score, target)) + ranked.sort(reverse=True) + choices.append([target for _, target in ranked[:self.enemy_branch]]) + + responses: list[tuple[int, ...]] = [] + for response in product(*choices): + responses.append(response) + if len(responses) >= self.response_cap: + break + return responses + + def _advance( + self, + mine: Body, + enemies: EnemyBodies, + my_target: int, + enemy_targets: tuple[int, ...], + food_bits: int, + ) -> tuple[Body, EnemyBodies, int] | None: + my_ate = bool((1 << my_target) & food_bits) + next_mine = self._advance_body(mine, my_target, my_ate) + next_enemies = tuple( + self._advance_body(body, target, bool((1 << target) & food_bits)) + for body, target in zip(enemies, enemy_targets) + ) + + # Body and self collisions after all tails have moved. + if my_target in next_mine[1:]: + return None + if any(my_target in enemy[1:] for enemy in next_enemies): + return None + + surviving: list[Body] = [] + for index, enemy in enumerate(next_enemies): + target = enemy[0] + dead = target in enemy[1:] or target in next_mine[1:] + dead = dead or any( + target in other[1:] for other_index, other in enumerate(next_enemies) + if other_index != index + ) + if target == my_target: + if len(enemy) >= len(next_mine): + return None + dead = True + if not dead: + # Enemy/enemy head collisions remove equal-length snakes and the shorter. + for other_index, other in enumerate(next_enemies): + if other_index != index and target == other[0] and len(enemy) <= len(other): + dead = True + break + if not dead: + surviving.append(enemy) + + eaten = (1 << my_target) if my_ate else 0 + for body, target in zip(enemies, enemy_targets): + if (1 << target) & food_bits: + eaten |= 1 << target + return next_mine, tuple(surviving), food_bits & ~eaten + + def _evaluate(self, mine: Body, enemies: EnemyBodies) -> float: + blocked = self._occupied(mine, enemies) & ~(1 << mine[0]) + space = self.board.flood_count(mine[0], blocked) + liberties = self.board.open_neighbor_count(mine[0], blocked) + if liberties == 0 or space < len(mine): + return self.DEATH + enemy_pressure = 0.0 + for enemy in enemies: + enemy_blocked = blocked & ~(1 << enemy[0]) + enemy_space = self.board.flood_count(enemy[0], enemy_blocked) + enemy_liberties = self.board.open_neighbor_count(enemy[0], enemy_blocked) + enemy_pressure += max(0, 3 - enemy_liberties) * 18.0 + if len(mine) > len(enemy): + enemy_pressure += max(0, 8 - enemy_space) * 8.0 + return space * 1.9 + liberties * 32.0 + enemy_pressure - len(enemies) * 4.0 + + def _occupied(self, mine: Body, enemies: EnemyBodies) -> int: + occupied = self._body_bits(mine) + for enemy in enemies: + occupied |= self._body_bits(enemy) + return occupied + + def _body_bits(self, body: Body) -> int: + cached = self.body_bits_cache.get(body) + if cached is not None: + return cached + bits = 0 + for cell in body: + bits |= 1 << cell + if len(self.body_bits_cache) < 16_384: + self.body_bits_cache[body] = bits + return bits + + def _advance_body(self, body: Body, target: int, ate: bool) -> Body: + if self.is_constrictor or ate: + return (target,) + body + return (target,) + body[:-1] + + @staticmethod + def _iter_bits(bits: int): + while bits: + bit = bits & -bits + yield bit.bit_length() - 1 + bits ^= bit + + def _out_of_time(self) -> bool: + return self.deadline is not None and perf_counter() >= self.deadline diff --git a/tests/snakes/test_PrismBattleSnake_GPT_5_6_Sol.py b/tests/snakes/test_PrismBattleSnake_GPT_5_6_Sol.py index 85970a7..4bb28f4 100644 --- a/tests/snakes/test_PrismBattleSnake_GPT_5_6_Sol.py +++ b/tests/snakes/test_PrismBattleSnake_GPT_5_6_Sol.py @@ -3,9 +3,10 @@ from time import perf_counter from snakes import SnakeBuilder, get_snake_version from snakes.ApexBattleSnake import ApexBattleSnake -from snakes.PrismBattleSnake_GPT_5_6_Sol import PrismBattleSnake_GPT_5_6_Sol from snakes.bitboard import BitBoard from snakes.bitboard_duel_search import BitboardDuelSearch +from snakes.compact_survival_search import CompactSurvivalSearch +from snakes.PrismBattleSnake_GPT_5_6_Sol import PrismBattleSnake_GPT_5_6_Sol class TestBitBoard(unittest.TestCase): @@ -51,8 +52,8 @@ class TestPrismBattleSnake_GPT_5_6_Sol(unittest.TestCase): snake = PrismBattleSnake_GPT_5_6_Sol() self.assertEqual(snake.name, "PrismBattleSnake") - self.assertEqual(snake.version, "1.1.0") - self.assertEqual(get_snake_version("PrismBattleSnake_GPT_5_6_Sol"), "1.1.0") + self.assertEqual(snake.version, "1.2.0") + self.assertEqual(get_snake_version("PrismBattleSnake_GPT_5_6_Sol"), "1.2.0") self.assertIsInstance(SnakeBuilder.build("PrismBattleSnake_GPT_5_6_Sol"), PrismBattleSnake_GPT_5_6_Sol) def test_bitboard_primitives_match_apex(self): @@ -156,6 +157,55 @@ class TestPrismBattleSnake_GPT_5_6_Sol(unittest.TestCase): self.assertIn(board.idx(1, 1), advanced) + def test_candidate_duel_searches_share_the_same_context(self): + snake = PrismBattleSnake_GPT_5_6_Sol() + kwargs = { + "my_body": [{"x": 0, "y": 1}, {"x": 0, "y": 0}], + "enemy_body": [{"x": 3, "y": 1}, {"x": 3, "y": 0}], + "food_set": set(), "hazard_set": set(), + "my_health": 100, "enemy_health": 100, + "hazard_damage": 15, "hazard_count": {}, + "width": 4, "height": 3, "max_depth": 2, + "alpha": -1e9, "beta": 1e9, "deadline": perf_counter() + 1.0, + } + + snake._minimax_candidate_id(my_target=(1, 1), **kwargs) + first_context = snake._duel_search_context + snake._minimax_candidate_id(my_target=(0, 2), **kwargs) + + self.assertIs(snake._duel_search_context, first_context) + self.assertGreater(first_context.nodes, 0) + + def test_compact_rollout_models_lethal_enemy_head_response(self): + board = BitBoard(3, 3) + search = CompactSurvivalSearch( + board=board, food=set(), is_constrictor=False, + deadline=perf_counter() + 1.0, branch=2, + ) + mine = [{"x": 0, "y": 1}, {"x": 0, "y": 0}] + enemies = [{ + "body": [{"x": 2, "y": 1}, {"x": 2, "y": 0}, {"x": 1, "y": 0}], + }] + + value = search.search_selected(mine, enemies, (1, 1), depth=1) + + self.assertEqual(value, search.DEATH) + + def test_compact_rollout_reuses_transpositions(self): + board = BitBoard(5, 5) + search = CompactSurvivalSearch( + board=board, food=set(), is_constrictor=False, + deadline=perf_counter() + 1.0, branch=2, + ) + mine = [{"x": 1, "y": 1}, {"x": 1, "y": 0}] + enemies = [{"body": [{"x": 3, "y": 3}, {"x": 3, "y": 4}]}] + + search.search_selected(mine, enemies, (2, 1), depth=3) + hits_before = search.cache_hits + search.search_selected(mine, enemies, (2, 1), depth=3) + + self.assertGreater(search.cache_hits, hits_before) + def test_bitboard_duel_search_reuses_transpositions(self): board = BitBoard(5, 5) search = BitboardDuelSearch(