Imitation training for NeuralNetPlayers and evaluation tournament sweep¶
Author: Rob Hendriks
Last verified: 17 September 2026
This notebook:
- Loops over a grid of
(field_size, comms_size)values. - For each setting:
- Builds a
GameLayoutand majority-teacher players. - Uses the imitation utilities to generate synthetic datasets for
model_aandmodel_b, following the majority strategy. - Trains
NeuralNetPlayers.model_aandNeuralNetPlayers.model_bseparately via imitation. - Runs a tournament with
MajorityPlayersand withNeuralNetPlayers.
- Builds a
- Collects all results in a pandas
DataFrameand prints a compact summary table.
The notebook assumes that:
- The
qsbpackage is importable (or adjust the imports below to your local package layout). - The module
neural_net_imitation_utils.pyis available and implements the functions specified in the design document (especiallygenerate_majority_imitation_datasets).
- The result is summarized in a table
| field_size | comms_size | maj | nn | maj_theory | info_limit |
|---|---|---|---|---|---|
| 4 | 1 | 0.5968 | 0.5924 | 0.5982 | 0.6461 |
| 4 | 2 | 0.6300 | 0.6290 | 0.6367 | 0.7051 |
| 4 | 4 | 0.6816 | 0.6852 | 0.6875 | 0.7855 |
| 4 | 8 | 0.7508 | 0.7526 | 0.7500 | 0.8900 |
| 8 | 1 | 0.5578 | 0.5568 | 0.5497 | 0.5735 |
| 8 | 2 | 0.5716 | 0.5710 | 0.5700 | 0.6037 |
| 8 | 4 | 0.6018 | 0.5884 | 0.5982 | 0.6461 |
| 8 | 8 | 0.6436 | 0.6284 | 0.6367 | 0.7051 |
| 16 | 1 | 0.5198 | 0.5332 | 0.5249 | 0.5368 |
| 16 | 2 | 0.5374 | 0.5414 | 0.5352 | 0.5520 |
| 16 | 4 | 0.5478 | 0.5550 | 0.5497 | 0.5735 |
| 16 | 8 | 0.5708 | 0.5524 | 0.5700 | 0.6037 |
| 32 | 1 | 0.5170 | 0.5160 | 0.5125 | 0.5184 |
| 32 | 2 | 0.5068 | 0.5214 | 0.5176 | 0.5260 |
| 32 | 4 | 0.5272 | 0.5364 | 0.5249 | 0.5368 |
| 32 | 8 | 0.5350 | 0.5318 | 0.5352 | 0.5520 |
| 64 | 1 | 0.5078 | 0.5088 | 0.5062 | 0.5092 |
| 64 | 2 | 0.5022 | 0.5066 | 0.5088 | 0.5130 |
| 64 | 4 | 0.5114 | 0.5160 | 0.5125 | 0.5184 |
In [1]:
import numpy as np
import pandas as pd
import sys
sys.path.append("../src")
import Q_Sea_Battle as qsb
# Core QSeaBattle imports.
# Adjust these imports if your package layout is different.
from Q_Sea_Battle import GameLayout
from Q_Sea_Battle import GameEnv
from Q_Sea_Battle import Tournament
from Q_Sea_Battle import MajorityPlayers, NeuralNetPlayers
from Q_Sea_Battle import neural_net_imitation_utilities as imitation_utils
print("qsb version loaded. Available symbols:", dir(qsb)[:20])
# Debug-friendly settings — use only during development.
import tensorflow as tf
tf.config.run_functions_eagerly(True)
tf.data.experimental.enable_debug_mode()
print("Debug mode enabled: eager execution + eager tf.data")
# Uncomment these lines only for debugging. For large runs or sweeps, disable to restore full performance.
qsb version loaded. Available symbols: ['Any', 'Dict', 'Game', 'GameEnv', 'GameLayout', 'GameplayModelAAdapter', 'GameplayModelBAdapter', 'LinCombineLayerA', 'LinCombineLayerB', 'LinInternalModelA', 'LinInternalModelB', 'LinMeasurementLayerA', 'LinMeasurementLayerB', 'LinTrainableAssistedModelA', 'LinTrainableAssistedModelB', 'MajorityPlayers', 'NeuralNetPlayerA', 'NeuralNetPlayerB', 'NeuralNetPlayers', 'PRAssisted'] Debug mode enabled: eager execution + eager tf.data
In [2]:
def inspect_imitation_datasets(layout, dataset_a, dataset_b, sample_size=50_000):
"""
Inspect dataset_a (field -> comm) and dataset_b (gun+comm -> shoot)
to confirm that the imitation data matches the intended distribution
and the MajorityPlayers logic.
layout: GameLayout
dataset_a: DataFrame with columns ["field", "comm", ...]
dataset_b: DataFrame with columns ["field", "gun", "comm", "shoot", ...]
sample_size: number of rows to subsample for stats (for speed).
"""
import numpy as np
import pandas as pd
import Q_Sea_Battle as qsb
print("=== Imitation Dataset Inspection ===")
# -----------------------------
# Subsample for speed
# -----------------------------
if len(dataset_a) > sample_size:
da = dataset_a.sample(sample_size, random_state=0)
else:
da = dataset_a
if len(dataset_b) > sample_size:
db = dataset_b.sample(sample_size, random_state=0)
else:
db = dataset_b
n2 = layout.field_size ** 2
m = layout.comms_size
# -----------------------------
# A1 — Field distribution p_one
# -----------------------------
fields = np.stack(da["field"].to_numpy(), axis=0).astype(float)
p_emp = fields.mean()
print(f" A1: Field mean p_one (empirical): {p_emp:.4f}")
# -----------------------------
# A2 — Comm bit frequencies
# -----------------------------
comms_a = np.stack(da["comm"].to_numpy(), axis=0)
print(" A2: Comm bit frequencies (dataset A):")
for j in range(m):
print(f" bit {j}: mean={comms_a[:, j].mean():.4f}")
# -----------------------------
# B1 — Gun index uniformity
# -----------------------------
guns_b = np.stack(db["gun"].to_numpy(), axis=0)
gun_idx = guns_b.argmax(axis=1)
print(" B1: Gun index statistics (dataset B):")
print(f" min idx={gun_idx.min()}, max idx={gun_idx.max()}")
print(f" approx std(index) = {gun_idx.std():.1f} (uniform-ish if large)")
# -----------------------------
# B2 — Shoot distribution
# -----------------------------
shoots = np.array(db["shoot"].to_numpy(), dtype=float)
print(f" B2: Shoot distribution: mean(shoot)={shoots.mean():.4f}")
# -----------------------------
# C1 — Cross-check B vs MajorityPlayers
# -----------------------------
print(" C1: Cross-check dataset B vs MajorityPlayers.playerB ...")
maj = qsb.MajorityPlayers(layout)
player_a_maj, player_b_maj = maj.players()
subset_b = db.sample(min(2000, len(db)), random_state=1)
mismatches_b = 0
for _, row in subset_b.iterrows():
gun = row["gun"]
comm = row["comm"]
shoot_ds = int(row["shoot"])
shoot_maj = int(player_b_maj.decide(gun=gun, comm=comm))
if shoot_ds != shoot_maj:
mismatches_b += 1
mismatch_rate_b = mismatches_b / len(subset_b)
print(f" Majority vs Dataset-B shoot mismatch rate: {mismatch_rate_b:.4f}")
# -----------------------------
# C2 — Cross-check A vs MajorityPlayers
# -----------------------------
print(" C2: Cross-check dataset A vs MajorityPlayers.playerA ...")
subset_a = da.sample(min(2000, len(da)), random_state=2)
mismatches_a = 0
for _, row in subset_a.iterrows():
field = row["field"]
comm = row["comm"]
comm_maj = player_a_maj.decide(field)
if not np.array_equal(comm, comm_maj):
mismatches_a += 1
mismatch_rate_a = mismatches_a / len(subset_a)
print(f" Majority vs Dataset-A comm mismatch rate: {mismatch_rate_a:.4f}")
print(" === Inspection Completed ===")
In [3]:
# ----------------------------
# Global configuration
# ----------------------------
FIELD_SIZES = [64,32,16,8,4]
COMMS_SIZES = [8,4,2,1]
# Number of synthetic samples per imitation batch.
# We keep the *total* number of synthetic samples roughly comparable to before,
# but avoid a single huge DataFrame in memory.
NUM_SAMPLES_A = 10_000 # for model_a (field -> comm), per batch
NUM_SAMPLES_B = 10_000 # for model_b (gun + comm -> shoot), per batch
# How many independent imitation batches to generate and train on.
NUM_IM_BATCHES_A = 25
NUM_IM_BATCHES_B = 25
# Training hyper-parameters for imitation
TRAINING_SETTINGS_A = {
"epochs": 25,
"batch_size": 256,
"learning_rate": 1e-3,
"verbose": 0,
"use_sample_weight": False,
}
TRAINING_SETTINGS_B = {
"epochs": 25,
"batch_size": 256,
"learning_rate": 1e-3,
"verbose": 0,
"use_sample_weight": False,
}
# Tournament configuration
NUM_GAMES_TOURNAMENT = 5_000
# Base seed for reproducibility
BASE_SEED = 12345
In [4]:
def run_single_experiment(field_size: int, comms_size: int, seed: int = 0):
"""
Run imitation-training and tournaments for a single (field_size, comms_size).
Returns a dict with summary statistics.
"""
n2 = field_size ** 2
if comms_size > n2:
raise ValueError("comms_size cannot exceed field_size**2 in this setup.")
# 1) Build layout and environment
layout = GameLayout(
field_size=field_size,
comms_size=comms_size,
number_of_games_in_tournament=NUM_GAMES_TOURNAMENT,
)
game_env = GameEnv(layout)
# 2) Evaluate MajorityPlayers teacher
majority_players = MajorityPlayers(layout)
tournament_teacher = Tournament(game_env, majority_players, layout)
log_teacher = tournament_teacher.tournament()
maj_mean, maj_stderr = log_teacher.outcome()
# 3) Build NeuralNetPlayers student
nn_players = NeuralNetPlayers(layout)
# 4) Imitation training with on-the-fly batch generation
#
# Instead of generating one *huge* imitation dataset and training on it multiple
# epochs, we now:
# - generate a fresh synthetic imitation batch for each step, and
# - run a short training phase on that batch.
#
# This keeps the peak memory usage bounded by NUM_SAMPLES_A / NUM_SAMPLES_B,
# while the *total* number of synthetic samples seen during training is
# NUM_IM_BATCHES_* × NUM_SAMPLES_*.
# 4a) Train model_a (field -> comm) via imitation on multiple small batches
for k in range(NUM_IM_BATCHES_A):
dataset_a, _ = imitation_utils.generate_majority_imitation_datasets(
layout=layout,
num_samples_a=NUM_SAMPLES_A,
num_samples_b=NUM_SAMPLES_B,
seed=BASE_SEED + k,
)
nn_players.train_model_a(dataset_a, TRAINING_SETTINGS_A)
# 4b) Train model_b (gun + comm -> shoot) via imitation on multiple small batches
for k in range(NUM_IM_BATCHES_B):
_, dataset_b = imitation_utils.generate_majority_imitation_datasets(
layout=layout,
num_samples_a=NUM_SAMPLES_A,
num_samples_b=NUM_SAMPLES_B,
seed=BASE_SEED + 100 + k,
)
nn_players.train_model_b(dataset_b, TRAINING_SETTINGS_B)
# 5) Evaluate trained NeuralNetPlayers in a fresh tournament
# (re-use the same layout but reset the environment)
game_env_nn = GameEnv(layout)
tournament_student = Tournament(game_env_nn, nn_players, layout)
log_student = tournament_student.tournament()
nn_mean, nn_stderr = log_student.outcome()
return {
"field_size": field_size,
"comms_size": comms_size,
"maj_mean": maj_mean,
"maj_stderr": maj_stderr,
"nn_mean": nn_mean,
"nn_stderr": nn_stderr,
}
In [5]:
results = []
for i, field_size in enumerate(FIELD_SIZES):
for comms_size in COMMS_SIZES:
# Only run configurations where comms_size is not trivially impossible
if comms_size > field_size ** 2:
continue
print(f"\n=== Running experiment: field_size={field_size}, comms_size={comms_size}, sample_size={NUM_SAMPLES_A} ===")
summary = run_single_experiment(field_size, comms_size, seed=i)
results.append(summary)
print(
f"Majority: mean={summary['maj_mean']:.4f}, stderr={summary['maj_stderr']:.4f} | "
f"Neural: mean={summary['nn_mean']:.4f}, stderr={summary['nn_stderr']:.4f}"
)
results_df = pd.DataFrame(results)
results_df = results_df.sort_values(["field_size", "comms_size"]).reset_index(drop=True)
results_df
=== Running experiment: field_size=64, comms_size=8, sample_size=10000 === WARNING:tensorflow:TensorFlow GPU support is not available on native Windows for TensorFlow >= 2.11. Even if CUDA/cuDNN are installed, GPU will not be used. Please use WSL2 or the TensorFlow-DirectML plugin. Majority: mean=0.5118, stderr=0.0071 | Neural: mean=0.5074, stderr=0.0071 === Running experiment: field_size=64, comms_size=4, sample_size=10000 === Majority: mean=0.5158, stderr=0.0071 | Neural: mean=0.5102, stderr=0.0071 === Running experiment: field_size=64, comms_size=2, sample_size=10000 === Majority: mean=0.5042, stderr=0.0071 | Neural: mean=0.5004, stderr=0.0071 === Running experiment: field_size=64, comms_size=1, sample_size=10000 === Majority: mean=0.4984, stderr=0.0071 | Neural: mean=0.5044, stderr=0.0071 === Running experiment: field_size=32, comms_size=8, sample_size=10000 === Majority: mean=0.5242, stderr=0.0071 | Neural: mean=0.5286, stderr=0.0071 === Running experiment: field_size=32, comms_size=4, sample_size=10000 === Majority: mean=0.5202, stderr=0.0071 | Neural: mean=0.5376, stderr=0.0071 === Running experiment: field_size=32, comms_size=2, sample_size=10000 === Majority: mean=0.5216, stderr=0.0071 | Neural: mean=0.5300, stderr=0.0071 === Running experiment: field_size=32, comms_size=1, sample_size=10000 === Majority: mean=0.5130, stderr=0.0071 | Neural: mean=0.5106, stderr=0.0071 === Running experiment: field_size=16, comms_size=8, sample_size=10000 === Majority: mean=0.5656, stderr=0.0070 | Neural: mean=0.5528, stderr=0.0070 === Running experiment: field_size=16, comms_size=4, sample_size=10000 === Majority: mean=0.5508, stderr=0.0070 | Neural: mean=0.5528, stderr=0.0070 === Running experiment: field_size=16, comms_size=2, sample_size=10000 === Majority: mean=0.5364, stderr=0.0071 | Neural: mean=0.5336, stderr=0.0071 === Running experiment: field_size=16, comms_size=1, sample_size=10000 === Majority: mean=0.5340, stderr=0.0071 | Neural: mean=0.5344, stderr=0.0071 === Running experiment: field_size=8, comms_size=8, sample_size=10000 === Majority: mean=0.6480, stderr=0.0068 | Neural: mean=0.6370, stderr=0.0068 === Running experiment: field_size=8, comms_size=4, sample_size=10000 === Majority: mean=0.5870, stderr=0.0070 | Neural: mean=0.5972, stderr=0.0069 === Running experiment: field_size=8, comms_size=2, sample_size=10000 === Majority: mean=0.5692, stderr=0.0070 | Neural: mean=0.5640, stderr=0.0070 === Running experiment: field_size=8, comms_size=1, sample_size=10000 === Majority: mean=0.5374, stderr=0.0071 | Neural: mean=0.5538, stderr=0.0070 === Running experiment: field_size=4, comms_size=8, sample_size=10000 === Majority: mean=0.7500, stderr=0.0061 | Neural: mean=0.7436, stderr=0.0062 === Running experiment: field_size=4, comms_size=4, sample_size=10000 === Majority: mean=0.6888, stderr=0.0065 | Neural: mean=0.6914, stderr=0.0065 === Running experiment: field_size=4, comms_size=2, sample_size=10000 === Majority: mean=0.6344, stderr=0.0068 | Neural: mean=0.6470, stderr=0.0068 === Running experiment: field_size=4, comms_size=1, sample_size=10000 === Majority: mean=0.5994, stderr=0.0069 | Neural: mean=0.6100, stderr=0.0069
Out[5]:
| field_size | comms_size | maj_mean | maj_stderr | nn_mean | nn_stderr | |
|---|---|---|---|---|---|---|
| 0 | 4 | 1 | 0.5994 | 0.006931 | 0.6100 | 0.006899 |
| 1 | 4 | 2 | 0.6344 | 0.006812 | 0.6470 | 0.006759 |
| 2 | 4 | 4 | 0.6888 | 0.006548 | 0.6914 | 0.006533 |
| 3 | 4 | 8 | 0.7500 | 0.006124 | 0.7436 | 0.006176 |
| 4 | 8 | 1 | 0.5374 | 0.007052 | 0.5538 | 0.007031 |
| 5 | 8 | 2 | 0.5692 | 0.007004 | 0.5640 | 0.007014 |
| 6 | 8 | 4 | 0.5870 | 0.006964 | 0.5972 | 0.006937 |
| 7 | 8 | 8 | 0.6480 | 0.006755 | 0.6370 | 0.006801 |
| 8 | 16 | 1 | 0.5340 | 0.007055 | 0.5344 | 0.007055 |
| 9 | 16 | 2 | 0.5364 | 0.007053 | 0.5336 | 0.007056 |
| 10 | 16 | 4 | 0.5508 | 0.007035 | 0.5528 | 0.007032 |
| 11 | 16 | 8 | 0.5656 | 0.007011 | 0.5528 | 0.007032 |
| 12 | 32 | 1 | 0.5130 | 0.007069 | 0.5106 | 0.007070 |
| 13 | 32 | 2 | 0.5216 | 0.007065 | 0.5300 | 0.007059 |
| 14 | 32 | 4 | 0.5202 | 0.007066 | 0.5376 | 0.007052 |
| 15 | 32 | 8 | 0.5242 | 0.007063 | 0.5286 | 0.007060 |
| 16 | 64 | 1 | 0.4984 | 0.007072 | 0.5044 | 0.007072 |
| 17 | 64 | 2 | 0.5042 | 0.007072 | 0.5004 | 0.007072 |
| 18 | 64 | 4 | 0.5158 | 0.007068 | 0.5102 | 0.007070 |
| 19 | 64 | 8 | 0.5118 | 0.007070 | 0.5074 | 0.007071 |
Summary table¶
In [6]:
from Q_Sea_Battle.reference_performance_utilities import (
expected_win_rate_majority,
limit_from_mutual_information,
)
# Add analytic majority prediction (optional)
layout = GameLayout()
results_df["maj_theory"] = results_df.apply(
lambda row: expected_win_rate_majority(
field_size=int(row["field_size"]),
comms_size=int(row["comms_size"]),
enemy_probability=layout.enemy_probability,
channel_noise=0.0,
),
axis=1,
)
# Add information-theoretic limit (mutual information bound)
results_df["info_limit"] = results_df.apply(
lambda row: limit_from_mutual_information(
field_size=row["field_size"],
comms_size=row["comms_size"],
channel_noise=0.0,
),
axis=1,
)
# Format nicely
table = results_df.copy()
table["maj"] = table["maj_mean"].map(lambda x: f"{x:0.4f}")
table["nn"] = table["nn_mean"].map(lambda x: f"{x:0.4f}")
table["maj_theory"] = table["maj_theory"].map(lambda x: f"{x:0.4f}")
table["info_limit"] = table["info_limit"].map(lambda x: f"{x:0.4f}")
display_cols = ["field_size", "comms_size", "maj", "nn", "maj_theory", "info_limit"]
display(table[display_cols])
| field_size | comms_size | maj | nn | maj_theory | info_limit | |
|---|---|---|---|---|---|---|
| 0 | 4 | 1 | 0.5994 | 0.6100 | 0.5982 | 0.6461 |
| 1 | 4 | 2 | 0.6344 | 0.6470 | 0.6367 | 0.7051 |
| 2 | 4 | 4 | 0.6888 | 0.6914 | 0.6875 | 0.7855 |
| 3 | 4 | 8 | 0.7500 | 0.7436 | 0.7500 | 0.8900 |
| 4 | 8 | 1 | 0.5374 | 0.5538 | 0.5497 | 0.5735 |
| 5 | 8 | 2 | 0.5692 | 0.5640 | 0.5700 | 0.6037 |
| 6 | 8 | 4 | 0.5870 | 0.5972 | 0.5982 | 0.6461 |
| 7 | 8 | 8 | 0.6480 | 0.6370 | 0.6367 | 0.7051 |
| 8 | 16 | 1 | 0.5340 | 0.5344 | 0.5249 | 0.5368 |
| 9 | 16 | 2 | 0.5364 | 0.5336 | 0.5352 | 0.5520 |
| 10 | 16 | 4 | 0.5508 | 0.5528 | 0.5497 | 0.5735 |
| 11 | 16 | 8 | 0.5656 | 0.5528 | 0.5700 | 0.6037 |
| 12 | 32 | 1 | 0.5130 | 0.5106 | 0.5125 | 0.5184 |
| 13 | 32 | 2 | 0.5216 | 0.5300 | 0.5176 | 0.5260 |
| 14 | 32 | 4 | 0.5202 | 0.5376 | 0.5249 | 0.5368 |
| 15 | 32 | 8 | 0.5242 | 0.5286 | 0.5352 | 0.5520 |
| 16 | 64 | 1 | 0.4984 | 0.5044 | 0.5062 | 0.5092 |
| 17 | 64 | 2 | 0.5042 | 0.5004 | 0.5088 | 0.5130 |
| 18 | 64 | 4 | 0.5158 | 0.5102 | 0.5125 | 0.5184 |
| 19 | 64 | 8 | 0.5118 | 0.5074 | 0.5176 | 0.5260 |
| field_size | comms_size | maj | nn | maj_theory | info_limit |
|---|---|---|---|---|---|
| 4 | 1 | 0.5968 | 0.5924 | 0.5982 | 0.6461 |
| 4 | 2 | 0.6300 | 0.6290 | 0.6367 | 0.7051 |
| 4 | 4 | 0.6816 | 0.6852 | 0.6875 | 0.7855 |
| 4 | 8 | 0.7508 | 0.7526 | 0.7500 | 0.8900 |
| 8 | 1 | 0.5578 | 0.5568 | 0.5497 | 0.5735 |
| 8 | 2 | 0.5716 | 0.5710 | 0.5700 | 0.6037 |
| 8 | 4 | 0.6018 | 0.5884 | 0.5982 | 0.6461 |
| 8 | 8 | 0.6436 | 0.6284 | 0.6367 | 0.7051 |
| 16 | 1 | 0.5198 | 0.5332 | 0.5249 | 0.5368 |
| 16 | 2 | 0.5374 | 0.5414 | 0.5352 | 0.5520 |
| 16 | 4 | 0.5478 | 0.5550 | 0.5497 | 0.5735 |
| 16 | 8 | 0.5708 | 0.5524 | 0.5700 | 0.6037 |
| 32 | 1 | 0.5170 | 0.5160 | 0.5125 | 0.5184 |
| 32 | 2 | 0.5068 | 0.5214 | 0.5176 | 0.5260 |
| 32 | 4 | 0.5272 | 0.5364 | 0.5249 | 0.5368 |
| 32 | 8 | 0.5350 | 0.5318 | 0.5352 | 0.5520 |
| 64 | 1 | 0.5078 | 0.5088 | 0.5062 | 0.5092 |
| 64 | 2 | 0.5022 | 0.5066 | 0.5088 | 0.5130 |
| 64 | 4 | 0.5114 | 0.5160 | 0.5125 | 0.5184 |
| 64 | 8 | 0.5122 | 0.5172 | 0.5176 | 0.5260 |