From a7a2e278bc37bed27dbe8b6eda7602ec6f877027 Mon Sep 17 00:00:00 2001 From: Spencer Cheng Date: Fri, 11 Sep 2026 16:46:48 +0000 Subject: [PATCH 1/2] Make bot-ladder eval parallelism configurable The ladder eval hardcoded 8192 envs. That suits bots that are just code, but an env whose bot is an external process spawns one child per env, so thousands is not viable. Add [selfplay] eval_bot_envs to set the count and eval_bot_threads to raise vec.num_threads for the eval only -- a bot that blocks on IPC is paced by how many envs step concurrently, not by how many exist. Both default to the previous behaviour. Replace the SELFPLAY_LADDER_ENVS define with the config key, which now carries the value and the rationale. Also: eval_loop hard-exited when [sweep] metric named a trainer-level overlay such as selfplay/bot_ladder_perf, because the env log cannot contain a key the ladder computes from those very evals. Fall back to env/score; the ladder scores off perf and ignores this value. Drop a stray double blank line in constellation.c. --- config/default.ini | 8 ++++++++ src/constellation.c | 1 - src/pufferl.cu | 21 +++++++++++---------- 3 files changed, 19 insertions(+), 11 deletions(-) diff --git a/config/default.ini b/config/default.ini index f7333e4f18..e33857a332 100644 --- a/config/default.ini +++ b/config/default.ini @@ -59,6 +59,14 @@ eval_bot_games = 0 # Ladder rungs, weakest first: the [env] bot_policy ids the env's header # defines. Required when eval_bot_games > 0, e.g. eval_bots = 3,4,5,6. eval_bots = 0 +# Envs the ladder runs in parallel, fixed so rung scores stay comparable across +# trials that vary vec.total_agents. eval_bot_games / this is the games per env; +# bots keep their kNN across episodes, so one game per env only measures cold +# bots. Drop to tens for an out-of-process bot (one child per env). +eval_bot_envs = 8192 +# 0 = leave vec.num_threads alone. A bot that blocks on IPC is paced by how many +# envs step at once, not by how many exist, so raise this to speed such a ladder. +eval_bot_threads = 0 # Per-rung [env] overrides for the bot ladder, as full section.key = value # lines (e.g. env.dr = 0). Env-owned so core needs no per-env knowledge. diff --git a/src/constellation.c b/src/constellation.c index 81e52678e6..a65a22520c 100644 --- a/src/constellation.c +++ b/src/constellation.c @@ -654,7 +654,6 @@ PlotArgs DEFAULT_PLOT_ARGS = { .z_label = "Train/Learning Rate", }; - Table* dataset_table(Dataset *data, char *env) { for (int i = 0; i < data->n; i++) { if (strcmp(data->tables[i].name, env) == 0) { diff --git a/src/pufferl.cu b/src/pufferl.cu index 48247e879f..745bf78ae5 100644 --- a/src/pufferl.cu +++ b/src/pufferl.cu @@ -2513,12 +2513,6 @@ typedef struct { #define SELFPLAY_MAX_HIST 8 #define SELFPLAY_MAX_LADDER 16 #define SELFPLAY_PATH_MAX 4096 -// Bot-ladder parallelism, held fixed so rung scores stay comparable across -// sweep trials that vary vec.total_agents. selfplay.eval_bot_games / this is -// the games each env plays; scripted bots keep their kNN across episodes, so -// one game per env measures only cold bots. -#define SELFPLAY_LADDER_ENVS 8192 - // One historical opponent ↔ policies[policy_idx] (env tag == policy_idx). typedef struct { int policy_idx; @@ -2899,8 +2893,9 @@ static EvalResult eval_loop(Ini* ini, PuffeRL* p, int mode, int verbose, if (verbose) { puf_dashboard_print(ini, p, show, board ? epoch : 0); } + DictItem* m = dict_find(&el, metric_key); result.score = match ? dict_get(&el, "env/policy_0_score") - : dict_get(&el, metric_key); + : (m ? m->value : dict_get(&el, "env/score")); result.perf = dict_get(&el, "env/perf"); if (match) { result.draw = dict_get(&el, "env/draw_rate"); @@ -3304,11 +3299,17 @@ TrainResult run_train(Ini* ini, TrainContext* ctx) { puf_ini_put(ini, ek, over->str); } // Fixed parallelism; ignore swept train total_agents. Each env plays - // bot_games / SELFPLAY_LADDER_ENVS games, so bots face a warmed-up kNN - // rather than being re-measured cold once per env. + // bot_games / ladder_envs games, so bots face a warmed-up kNN rather + // than being re-measured cold once per env. char nbuf[32]; - snprintf(nbuf, sizeof(nbuf), "%d", SELFPLAY_LADDER_ENVS); + long envs = puf_ini_get(ini, "selfplay", "eval_bot_envs"); + snprintf(nbuf, sizeof(nbuf), "%ld", envs); puf_ini_put(ini, "vec.total_agents", nbuf); + long threads = puf_ini_get(ini, "selfplay", "eval_bot_threads"); + if (threads > 0) { + snprintf(nbuf, sizeof(nbuf), "%ld", threads); + puf_ini_put(ini, "vec.num_threads", nbuf); + } double ladder[SELFPLAY_MAX_LADDER]; int rungs = puf_ini_get_list(ini, "selfplay", "eval_bots", ladder, SELFPLAY_MAX_LADDER); From 98c873126e07da1a5bfb59e17433e1f5a51be3d8 Mon Sep 17 00:00:00 2001 From: Spencer Cheng Date: Fri, 11 Sep 2026 18:23:41 +0000 Subject: [PATCH 2/2] remove bad comments --- config/default.ini | 6 ------ 1 file changed, 6 deletions(-) diff --git a/config/default.ini b/config/default.ini index e33857a332..4e0ccf61a2 100644 --- a/config/default.ini +++ b/config/default.ini @@ -59,13 +59,7 @@ eval_bot_games = 0 # Ladder rungs, weakest first: the [env] bot_policy ids the env's header # defines. Required when eval_bot_games > 0, e.g. eval_bots = 3,4,5,6. eval_bots = 0 -# Envs the ladder runs in parallel, fixed so rung scores stay comparable across -# trials that vary vec.total_agents. eval_bot_games / this is the games per env; -# bots keep their kNN across episodes, so one game per env only measures cold -# bots. Drop to tens for an out-of-process bot (one child per env). eval_bot_envs = 8192 -# 0 = leave vec.num_threads alone. A bot that blocks on IPC is paced by how many -# envs step at once, not by how many exist, so raise this to speed such a ladder. eval_bot_threads = 0 # Per-rung [env] overrides for the bot ladder, as full section.key = value