From 560a8c1f997db2d987aef4caf617e3ddb4abb30c Mon Sep 17 00:00:00 2001 From: Richard Higgins Date: Thu, 1 Oct 2026 02:05:00 -0700 Subject: [PATCH 01/11] Match GoTA lockstep drafts and battle clocks to ordinary games Complete scripted drafting before learner decisions, apply the ordinary battle-duration limit, and report draft ticks separately. Compare complete frozen-action trajectories and terminal state on both seats. Co-authored-by: GPT-6 --- examples/gods_of_the_arena/lockstep.nim | 15 ++++--- tests/test_gota_lockstep.nim | 57 ++++++++++++++++++++++++- 2 files changed, 65 insertions(+), 7 deletions(-) diff --git a/examples/gods_of_the_arena/lockstep.nim b/examples/gods_of_the_arena/lockstep.nim index 352959c0..a42fb710 100644 --- a/examples/gods_of_the_arena/lockstep.nim +++ b/examples/gods_of_the_arena/lockstep.nim @@ -37,7 +37,7 @@ type LaneStats* {.bycopy.} = object ## Summary of one finished match, from the first policy team's side. - finished*, outcome*, tick*, selfPlay*: int32 + finished*, outcome*, tick*, selfPlay*, draftTicks*: int32 potential*: int64 ## Leaderboard score numerator at the end; points = potential / 7200. @@ -119,7 +119,7 @@ proc startLane(batch: StepBatch, lane: ptr StepLane, seed: int) = config = loadConfig(batch.configPath) gameMap = generateMap(int32(seed), config.mapPreset) game = newGame( - gameMap, config.spawnIntervalTicks, 10, false, ReplayData(), false + gameMap, config.spawnIntervalTicks, 10, false, ReplayData() ) team = Team((seed mod 10) div 5) opponent = batch.opponents[(seed div 10) mod batch.opponents.len] @@ -148,6 +148,9 @@ proc startLane(batch: StepBatch, lane: ptr StepLane, seed: int) = if hero.team != team: batch.installPolicy(lane, index) doAssert lane.agents.len == batch.agentsPerLane + while game.world.phase == Drafting: + activeGame = game + tickWorld(game, proc() = runBotDecisions(game)) for side in Team: lane.potentials[side] = case batch.rewardMode @@ -223,15 +226,14 @@ proc step*( lane.actions[i] = actions[agent + i] let game = lane.game var ticks = 0 - while ticks < batch.actionTicks and not game.world.gameOver and - game.world.tick < batch.maxTicks: + while ticks < batch.actionTicks and not game.finished(): activeGame = game tickWorld(game, proc() = runBotDecisions(game)) inc ticks for vm in game.heroVms: doAssert not vm.failed, vm.lastError let - done = game.world.gameOver or game.world.tick >= batch.maxTicks + done = game.finished() world = game.world var reward: array[Team, float32] for side in Team: @@ -256,8 +258,9 @@ proc step*( stats[laneIndex] = LaneStats( finished: 1, outcome: outcome(world, lane.team, true), - tick: world.tick, + tick: world.battleTick(), selfPlay: int32(batch.selfPlay), + draftTicks: world.draftTicks, potential: if outcome(world, lane.team, true) == 1: teamPotential(world, lane.team) diff --git a/tests/test_gota_lockstep.nim b/tests/test_gota_lockstep.nim index 3457c2a7..7fb10563 100644 --- a/tests/test_gota_lockstep.nim +++ b/tests/test_gota_lockstep.nim @@ -1,6 +1,16 @@ include ../examples/gods_of_the_arena/lockstep +import std/[os, tempfiles] let policy = "dim f(" & $GotaFeatureCount & ")\n" & """ +if drafting then + for candidate = 0 to 9 + if heroAvailable(candidate) then + draftHero(candidate) + end + end if + next candidate + end +end if f(0) = 100 f(1) = selfTeam * 100 f(2) = selfClass * 10 @@ -54,6 +64,51 @@ echo "Testing agent counts" doAssert newStepBatch(config, bot, bot, policy, 3, 240, 24, false).agentCount == 15 doAssert newStepBatch(config, bot, bot, policy, 3, 240, 24, true).agentCount == 30 +echo "Testing ordinary draft, battle clock and frozen-action trajectories" +block: + let directory = createTempDir("gota-lockstep-", "") + let player = directory / "policy.bas" + defer: + removeFile(player) + removeDir(directory) + writeFile(player, policy.replace("' METTA_DECISION", "decision = 0")) + for seed in [0, 7]: + let batch = newStepBatch(config, bot, bot, policy, 1, 240, 24, false) + batch.reset(seed) + let trained = batch.lanes[0].game + let preset = loadConfig(config) + let ordinary = newGame(generateMap(int32(seed), preset.mapPreset), + preset.spawnIntervalTicks, 10, false, ReplayData()) + let groups = + if batch.lanes[0].team == RedTeam: + @[BotGroup(path: player, count: 5), BotGroup(path: bot, count: 5)] + else: + @[BotGroup(path: bot, count: 5), BotGroup(path: player, count: 5)] + loadBots(ordinary, groups) + ordinary.replayData = initReplayData(currentSetup(ordinary, 240), preset.mapPreset) + while ordinary.world.phase == Drafting: + activeGame = ordinary + tickWorld(ordinary, proc() = runBotDecisions(ordinary)) + doAssert trained.world.drafting + doAssert trained.world.draftTicks > 0 + doAssert trained.world.battleTick() == 0 + doAssert trained.stateHash() == ordinary.stateHash() + var + actions = newSeq[int32](batch.agentCount) + rewards = newSeq[float32](batch.agentCount) + terminals = newSeq[uint8](batch.agentCount) + stats = newSeq[LaneStats](1) + for frame in 0 ..< 10: + batch.step(actions, rewards, terminals, stats) + for tick in 0 ..< 24: + activeGame = ordinary + tickWorld(ordinary, proc() = runBotDecisions(ordinary)) + doAssert trained.stateHash() == ordinary.stateHash() + doAssert stats[0].finished == int32(frame == 9) + doAssert stats[0].tick == 240 + doAssert stats[0].draftTicks == trained.world.draftTicks + doAssert trained.finished() and ordinary.finished() + echo "Testing observations and seats" let played = run(true, XpReward, 2400, 20) for agent in 0 ..< played.seats.len: @@ -87,7 +142,7 @@ for (ended, draw, winner) in [ batch = newStepBatch(config, bot, bot, policy, 1, 28800, 24, true, LeaderboardReward) world = batch.lanes[0].game.world - world.tick = if ended: 15120 else: 28800 + world.tick = if ended: 15120 else: 28920 world.draftTicks = 120 world.gameOver = ended world.draw = draw From f3c9e4c924a1dd2f91623b982bc4613bbfe36b5e Mon Sep 17 00:00:00 2001 From: Richard Higgins Date: Fri, 2 Oct 2026 12:59:20 -0700 Subject: [PATCH 02/11] Use read-only BASIC queries while preserving GoTA draft parity Port the object queries and borrowed visibility getters from upstream 806082d. Pin its Bassy compiler while retaining the scripted draft, battle clock, and native ABI from 560a8c1. Co-authored-by: Treeform Co-authored-by: Codex GPT-6 --- coworld/dependencies.lock | 2 +- examples/gods_of_the_arena/bots.nim | 105 ++++++++++++---------------- examples/gods_of_the_arena/sim.nim | 18 +++++ 3 files changed, 65 insertions(+), 60 deletions(-) diff --git a/coworld/dependencies.lock b/coworld/dependencies.lock index ff4264d8..f82f862e 100644 --- a/coworld/dependencies.lock +++ b/coworld/dependencies.lock @@ -1,4 +1,4 @@ -bassy 0.1.0 https://github.com/treeform/bassy 198781a731849826537446dea6c612de93582806 +bassy 0.1.0 https://github.com/treeform/bassy b1b4f7db1d811b60195c1aa6eddc98392bd07b04 curly 1.1.1 https://github.com/guzba/curly 4f36f01ac6ca881fff4bbf0f8daa362d4b9f0668 libcurl 1.0.0 https://github.com/Araq/libcurl 7a420498f60a31d99fc8513886ce36c4e8c3a4ae fixxy 0.1.0 https://github.com/treeform/fixxy 05e5446dffb70093056cebb0c57721a60deaf52a diff --git a/examples/gods_of_the_arena/bots.nim b/examples/gods_of_the_arena/bots.nim index 94095510..2040b49f 100644 --- a/examples/gods_of_the_arena/bots.nim +++ b/examples/gods_of_the_arena/bots.nim @@ -178,8 +178,9 @@ proc objectProc(heroId: int32, field: ObjectField): HostProc = result = proc(arguments: openArray[int32]): int32 = ## Reads a visible object's field without exposing hidden targets. let world = activeGame.world - var value: WorldObject - if not world.worldObjectAt(heroId, int(arguments[0]), value): + var team: Team + let value = world.scriptObject(heroId, int(arguments[0]), team) + if value == nil: return (if field == ObjectCamp: -1 else: 0) case field of ObjectCamp: int32(value.camp - 1) @@ -210,13 +211,13 @@ proc objectProc(heroId: int32, field: ObjectField): HostProc = WorldScale ) of ObjectTarget: - if value.targetId == 0: + let targetId = value.targetId + if targetId == 0: return 0 - var target: WorldObject for i in 0 ..< world.worldObjectCount(heroId): - if world.worldObjectAt(heroId, i, target) and - target.id == value.targetId: - return target.id + let target = world.scriptObject(heroId, i, team) + if target != nil and target.id == targetId: + return target.id 0 of ObjectVelX: value.velocity.x @@ -332,8 +333,10 @@ proc infoFunctions(host: var Host, heroId: int32) = else: toValue(0) let objectInfo: NumericHostProc = proc(args: openArray[Value]): Value = ## Reads only objects in the same visibility-filtered decision snapshot. - var value: WorldObject - if not activeGame.world.worldObjectAt(heroId, int(args[0].asInt), value): + var team: Team + let value = + activeGame.world.scriptObject(heroId, int(args[0].asInt), team) + if value == nil: return toValue(0) case args[1].asInt of 0: toValue(worldToTiles(value.position.x, WorldScale)) @@ -561,67 +564,51 @@ proc initHeroHost( let objectIdProc: HostProc = proc( arguments: openArray[int32] ): int32 = - var value: WorldObject - if worldObjectAt(activeGame.world, heroId, int(arguments[0]), value): - value.id - else: - 0 + var team: Team + let value = scriptObject(activeGame.world, heroId, int(arguments[0]), team) + if value == nil: 0 else: value.id let objectKindProc: HostProc = proc( arguments: openArray[int32] ): int32 = - var value: WorldObject - if worldObjectAt(activeGame.world, heroId, int(arguments[0]), value): - value.kind - else: - 0 + var team: Team + let value = scriptObject(activeGame.world, heroId, int(arguments[0]), team) + if value == nil: 0 else: value.kind let objectTeamProc: HostProc = proc( arguments: openArray[int32] ): int32 = - var value: WorldObject - if worldObjectAt(activeGame.world, heroId, int(arguments[0]), value): - value.faction - else: - 0 + var team: Team + let value = scriptObject(activeGame.world, heroId, int(arguments[0]), team) + if value == nil: 0 else: value[].faction let objectClassProc: HostProc = proc( arguments: openArray[int32] ): int32 = - var value: WorldObject - if worldObjectAt(activeGame.world, heroId, int(arguments[0]), value): - value.class - else: - -1 + var team: Team + let value = scriptObject(activeGame.world, heroId, int(arguments[0]), team) + if value == nil: -1 else: value.class let objectXProc: HostProc = proc( arguments: openArray[int32] ): int32 = - var value: WorldObject - if worldObjectAt(activeGame.world, heroId, int(arguments[0]), value): - mapCoordinate(value.position.x, activeGame.world.heroById(heroId).team) - else: - 0 + var team: Team + let value = scriptObject(activeGame.world, heroId, int(arguments[0]), team) + if value == nil: 0 else: mapCoordinate(value.position.x, team) let objectYProc: HostProc = proc( arguments: openArray[int32] ): int32 = - var value: WorldObject - if worldObjectAt(activeGame.world, heroId, int(arguments[0]), value): - mapCoordinate(value.position.z, activeGame.world.heroById(heroId).team) - else: - 0 + var team: Team + let value = scriptObject(activeGame.world, heroId, int(arguments[0]), team) + if value == nil: 0 else: mapCoordinate(value.position.z, team) let objectHpProc: HostProc = proc( arguments: openArray[int32] ): int32 = - var value: WorldObject - if worldObjectAt(activeGame.world, heroId, int(arguments[0]), value): - max(value.hp, 0'i32) - else: - 0 + var team: Team + let value = scriptObject(activeGame.world, heroId, int(arguments[0]), team) + if value == nil: 0 else: max(value.hp, 0'i32) let objectAliveProc: HostProc = proc( arguments: openArray[int32] ): int32 = - var value: WorldObject - int32( - worldObjectAt(activeGame.world, heroId, int(arguments[0]), value) and - value.alive - ) + var team: Team + let value = scriptObject(activeGame.world, heroId, int(arguments[0]), team) + int32(value != nil and value.alive) let walkToProc: NumericHostProc = proc( arguments: openArray[Value] ): Value = @@ -945,7 +932,7 @@ proc initHeroHost( ]: let arity = if field in {ObjectItemId, ObjectItemCount}: 2 else: 1 - discard result.addFunction(name, arity, objectProc(heroId, field), 16) + discard result.addQuery(name, arity, objectProc(heroId, field), 16) let spellCountProc: HostProc = proc(arguments: openArray[int32]): int32 = ## Counts warnings and projectiles visible to this hero's team. int32(activeGame.world.visibleSpellCount(heroId)) @@ -959,8 +946,8 @@ proc initHeroHost( ]: discard result.addFunction(name, 1, spellProc(heroId, field), 16) - discard result.addFunction("objectCount", 0, objectCountProc, 2) - discard result.addFunction("objectId", 1, objectIdProc, 4) + discard result.addQuery("objectCount", 0, objectCountProc, 2) + discard result.addQuery("objectId", 1, objectIdProc, 4) let campCountProc: HostProc = proc(arguments: openArray[int32]): int32 = ## Counts generated camp clearings regardless of fog or spawn state. int32(activeGame.world.camps.len) @@ -968,13 +955,13 @@ proc initHeroHost( for (field, name) in [(CampX, "campX"), (CampY, "campY"), (CampTier, "campTier")]: discard result.addFunction(name, 1, campProc(heroId, field), 4) - discard result.addFunction("objectKind", 1, objectKindProc, 4) - discard result.addFunction("objectTeam", 1, objectTeamProc, 4) - discard result.addFunction("objectClass", 1, objectClassProc, 4) - discard result.addFunction("objectX", 1, objectXProc, 4) - discard result.addFunction("objectY", 1, objectYProc, 4) - discard result.addFunction("objectHp", 1, objectHpProc, 4) - discard result.addFunction("objectAlive", 1, objectAliveProc, 4) + discard result.addQuery("objectKind", 1, objectKindProc, 4) + discard result.addQuery("objectTeam", 1, objectTeamProc, 4) + discard result.addQuery("objectClass", 1, objectClassProc, 4) + discard result.addQuery("objectX", 1, objectXProc, 4) + discard result.addQuery("objectY", 1, objectYProc, 4) + discard result.addQuery("objectHp", 1, objectHpProc, 4) + discard result.addQuery("objectAlive", 1, objectAliveProc, 4) discard result.addFunction("walkTo", 2, walkToProc, 800) discard result.addFunction("attackMove", 2, attackMoveProc, 800) discard result.addFunction("attackTarget", 1, attackTargetProc, 20) diff --git a/examples/gods_of_the_arena/sim.nim b/examples/gods_of_the_arena/sim.nim index 73ca5c44..de3ad3dd 100644 --- a/examples/gods_of_the_arena/sim.nim +++ b/examples/gods_of_the_arena/sim.nim @@ -2330,6 +2330,24 @@ proc worldObjectCount*(world: World, heroId: int32): int = let team = world.ensureScriptObjects(heroId) world.scriptObjectCount[team] +proc scriptObject*( + world: World, + heroId: int32, + index: int, + team: var Team +): ptr WorldObject = + ## Returns one object of a hero's visibility-filtered enumeration where + ## it sits, along with the hero's team, or nil. Nothing is copied, and + ## the object stays put until the next decision frame rebuilds the + ## enumeration, so read what is needed from it straight away. + if world.heroIndex(heroId) < 0: + return nil + team = world.ensureScriptObjects(heroId) + if index < 0 or index >= world.scriptObjectCount[team]: + return nil + world.scriptObjects[team][index].addr + + proc worldObjectAt*( world: World, heroId: int32, From f445a6e3a81df1449e49678e0420e82dd5577675 Mon Sep 17 00:00:00 2001 From: Richard Higgins Date: Fri, 2 Oct 2026 14:18:38 -0700 Subject: [PATCH 03/11] Cull distant footman targets before visibility queries Co-authored-by: Codex GPT-6 --- examples/gods_of_the_arena/sim.nim | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/examples/gods_of_the_arena/sim.nim b/examples/gods_of_the_arena/sim.nim index de3ad3dd..b36520eb 100644 --- a/examples/gods_of_the_arena/sim.nim +++ b/examples/gods_of_the_arena/sim.nim @@ -3629,9 +3629,11 @@ proc updateFootman(world: World, footman: var Footman) = let other {.cursor.} = world.footmen[i] if not world.hostile(other, footman.team): continue + let distance = distanceSquared(footman.position, other.position) + if distance > bestSquared: + continue if not visible(world, footman.team, other.position): continue - let distance = distanceSquared(footman.position, other.position) if distance < bestSquared or (distance == bestSquared and bestId != 0 and targetBefore(other.position, other.id, bestPosition, bestId, footman.team)): @@ -3643,9 +3645,11 @@ proc updateFootman(world: World, footman: var Footman) = let hero = world.heroes[i] if hero.team == footman.team or hero.state == Dying or hero.hp <= 0: continue + let distance = distanceSquared(footman.position, hero.position) + if distance > bestSquared: + continue if not visible(world, footman.team, hero.position): continue - let distance = distanceSquared(footman.position, hero.position) if distance < bestSquared or (distance == bestSquared and targetFootman < 0 and bestId != 0 and targetBefore(hero.position, hero.id, bestPosition, bestId, footman.team)): From ddb308452d67b08ffa7b4a6bb31eb2fefc97da11 Mon Sep 17 00:00:00 2001 From: Richard Higgins Date: Sat, 3 Oct 2026 11:15:06 -0700 Subject: [PATCH 04/11] Add bounded opt-in GoTA command attribution Preserve command-handler results and outgoing reward state before lockstep reset. Isolated CPU diagnostic candidate; build and runtime validation delegated to executor. Co-authored-by: Codex (GPT-6) --- examples/gods_of_the_arena/lockstep.nim | 84 +++++++++++++++++++++++++ examples/gods_of_the_arena/sim.nim | 15 +++++ 2 files changed, 99 insertions(+) diff --git a/examples/gods_of_the_arena/lockstep.nim b/examples/gods_of_the_arena/lockstep.nim index a42fb710..6da07d7c 100644 --- a/examples/gods_of_the_arena/lockstep.nim +++ b/examples/gods_of_the_arena/lockstep.nim @@ -41,6 +41,14 @@ type potential*: int64 ## Leaderboard score numerator at the end; points = potential / 7200. + CommandTrace {.bycopy.} = object + agent, ordinal: int32 + command: CommandAttribution + + StepTrace {.bycopy.} = object + heroId, seat, tickBefore, tickAfter: int32 + potentialBefore, potentialAfter: int64 + StepLane = object game: Game seed: int @@ -60,6 +68,9 @@ type selfPlay: bool rewardMode: RewardMode agentsPerLane*: int + commandTraceLimit: int + commandTrace: seq[CommandTrace] + stepTrace: seq[StepTrace] proc teamPotential(world: World, team: Team): int64 = ## Scales whole XP per minute into the training API's score numerator. @@ -218,6 +229,9 @@ proc step*( ) = ## Runs every lane for actionTicks ticks with the given actions. doAssert actions.len == batch.agentCount + if batch.commandTraceLimit > 0: + batch.commandTrace.setLen(0) + batch.stepTrace.setLen(batch.agentCount) var agent = 0 for laneIndex in 0 ..< batch.lanes.len: let lane = batch.lanes[laneIndex].addr @@ -225,6 +239,18 @@ proc step*( doAssert actions[agent + i] in 0 ..< GotaActionCount lane.actions[i] = actions[agent + i] let game = lane.game + if batch.commandTraceLimit > 0: + game.world.commandAttribution.setLen(0) + game.world.commandAttributionHeroes.setLen(0) + game.world.commandAttributionLimit = batch.commandTraceLimit + for i, index in lane.agents: + let hero = game.world.heroes[index] + game.world.commandAttributionHeroes.add hero.id + batch.stepTrace[agent + i] = StepTrace( + heroId: hero.id, seat: int32(index), + tickBefore: game.world.tick, + potentialBefore: lane.potentials[hero.team] + ) var ticks = 0 while ticks < batch.actionTicks and not game.finished(): activeGame = game @@ -270,6 +296,22 @@ proc step*( for i, index in lane.agents: rewards[agent + i] = reward[world.heroes[index].team] terminals[agent + i] = uint8(done) + if batch.commandTraceLimit > 0: + for command in world.commandAttribution: + for i, index in lane.agents: + if world.heroes[index].id == command.heroId: + doAssert batch.commandTrace.len < batch.commandTraceLimit, + "batch command attribution capacity exceeded" + batch.commandTrace.add CommandTrace( + agent: int32(agent + i), ordinal: int32(batch.commandTrace.len), + command: command + ) + break + for i, index in lane.agents: + batch.stepTrace[agent + i].tickAfter = world.tick + batch.stepTrace[agent + i].potentialAfter = lane.potentials[world.heroes[index].team] + # Preserve the completed world's command/reward facts before automatic reset. + world.commandAttributionLimit = 0 if done: batch.startLane(lane, lane.seed + batch.lanes.len) agent += lane.agents.len @@ -329,5 +371,47 @@ proc gota_step( seats.toOpenArray(0, count - 1) ) +proc gota_attribution_version(): cint {.cdecl, exportc, dynlib.} = + 1 + +proc gota_command_trace_size(): cint {.cdecl, exportc, dynlib.} = + cint(sizeof(CommandTrace)) + +proc gota_step_trace_size(): cint {.cdecl, exportc, dynlib.} = + cint(sizeof(StepTrace)) + +proc gota_trace_enable(handle: pointer, maxCommands: cint) {.cdecl, exportc, dynlib.} = + doAssert maxCommands > 0 + cast[StepBatch](handle).commandTraceLimit = int(maxCommands) + +proc gota_trace_identity( + handle: pointer, heroIds, gameSeats: ptr UncheckedArray[int32] +) {.cdecl, exportc, dynlib.} = + let batch = cast[StepBatch](handle) + doAssert batch.commandTraceLimit > 0 + var agent = 0 + for lane in batch.lanes: + for index in lane.agents: + heroIds[agent] = lane.game.world.heroes[index].id + gameSeats[agent] = int32(index) + inc agent + +proc gota_trace_count(handle: pointer): cint {.cdecl, exportc, dynlib.} = + cint(cast[StepBatch](handle).commandTrace.len) + +proc gota_trace_copy( + handle: pointer, + commands: ptr UncheckedArray[CommandTrace], capacity: cint, + steps: ptr UncheckedArray[StepTrace] +) {.cdecl, exportc, dynlib.} = + let batch = cast[StepBatch](handle) + doAssert batch.commandTraceLimit > 0 + doAssert int(capacity) >= batch.commandTrace.len + doAssert batch.stepTrace.len == batch.agentCount + for i, command in batch.commandTrace: + commands[i] = command + for i, step in batch.stepTrace: + steps[i] = step + proc gota_close(handle: pointer) {.cdecl, exportc, dynlib.} = GC_unref(cast[StepBatch](handle)) diff --git a/examples/gods_of_the_arena/sim.nim b/examples/gods_of_the_arena/sim.nim index b36520eb..6c7c4b74 100644 --- a/examples/gods_of_the_arena/sim.nim +++ b/examples/gods_of_the_arena/sim.nim @@ -267,7 +267,14 @@ type team: Team hero, fixed: bool + CommandAttribution* {.bycopy.} = object + ## Result of an actual command-handler call, not proof of later damage/movement. + tick*, heroId*, action*, slot*, first*, second*, error*, offsetX*, offsetY*: int32 + World* = ref object + commandAttributionLimit*: int + commandAttributionHeroes*: seq[int32] + commandAttribution*: seq[CommandAttribution] when defined(replayEvents): events*: seq[GameEvent] eventTick: int32 @@ -1093,6 +1100,14 @@ proc finishAction( offset = FixedVec2Zero ): bool = ## Updates only the submitting hero's diagnostic and records failed commands. + if world.commandAttributionLimit > 0 and heroId in world.commandAttributionHeroes: + doAssert world.commandAttribution.len < world.commandAttributionLimit, + "command attribution capacity exceeded" + world.commandAttribution.add CommandAttribution( + tick: world.tick, heroId: heroId, action: int32(action), slot: slot, + first: first, second: second, error: int32(error.ord), + offsetX: int32(offset.x), offsetY: int32(offset.y) + ) let index = world.heroIndex(heroId) if index >= 0: world.heroes[index].lastActionError = error From 91483bb480b5a867b17f0caafc27e1f7927b9fdc Mon Sep 17 00:00:00 2001 From: Richard Higgins Date: Sat, 3 Oct 2026 11:16:47 -0700 Subject: [PATCH 05/11] Record actual simulator team in attribution identity Distinguish game team from policy-local controlled group before terminal reset. Co-authored-by: Codex (GPT-6) --- examples/gods_of_the_arena/lockstep.nim | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/examples/gods_of_the_arena/lockstep.nim b/examples/gods_of_the_arena/lockstep.nim index 6da07d7c..f7785140 100644 --- a/examples/gods_of_the_arena/lockstep.nim +++ b/examples/gods_of_the_arena/lockstep.nim @@ -46,7 +46,7 @@ type command: CommandAttribution StepTrace {.bycopy.} = object - heroId, seat, tickBefore, tickAfter: int32 + heroId, seat, team, tickBefore, tickAfter: int32 potentialBefore, potentialAfter: int64 StepLane = object @@ -247,7 +247,7 @@ proc step*( let hero = game.world.heroes[index] game.world.commandAttributionHeroes.add hero.id batch.stepTrace[agent + i] = StepTrace( - heroId: hero.id, seat: int32(index), + heroId: hero.id, seat: int32(index), team: int32(hero.team.ord), tickBefore: game.world.tick, potentialBefore: lane.potentials[hero.team] ) @@ -385,7 +385,7 @@ proc gota_trace_enable(handle: pointer, maxCommands: cint) {.cdecl, exportc, dyn cast[StepBatch](handle).commandTraceLimit = int(maxCommands) proc gota_trace_identity( - handle: pointer, heroIds, gameSeats: ptr UncheckedArray[int32] + handle: pointer, heroIds, gameSeats, gameTeams: ptr UncheckedArray[int32] ) {.cdecl, exportc, dynlib.} = let batch = cast[StepBatch](handle) doAssert batch.commandTraceLimit > 0 @@ -394,6 +394,7 @@ proc gota_trace_identity( for index in lane.agents: heroIds[agent] = lane.game.world.heroes[index].id gameSeats[agent] = int32(index) + gameTeams[agent] = int32(lane.game.world.heroes[index].team.ord) inc agent proc gota_trace_count(handle: pointer): cint {.cdecl, exportc, dynlib.} = From a132f2d4ef5f3e4c44bc770b03ccd200278dcf03 Mon Sep 17 00:00:00 2001 From: Richard Higgins Date: Sun, 4 Oct 2026 00:33:05 -0700 Subject: [PATCH 06/11] Remove duplicate scriptObject introduced by main integration Keep the canonical main declaration; the isolated integration compile failed before tests on this duplicate. Preserve frozen 91483 qualification and source. Co-authored-by: Codex (GPT-6) --- examples/gods_of_the_arena/sim.nim | 18 ------------------ 1 file changed, 18 deletions(-) diff --git a/examples/gods_of_the_arena/sim.nim b/examples/gods_of_the_arena/sim.nim index 23df0d30..b2f0d147 100644 --- a/examples/gods_of_the_arena/sim.nim +++ b/examples/gods_of_the_arena/sim.nim @@ -2352,24 +2352,6 @@ proc worldObjectCount*(world: World, heroId: int32): int = let team = world.ensureScriptObjects(heroId) world.scriptObjectCount[team] -proc scriptObject*( - world: World, - heroId: int32, - index: int, - team: var Team -): ptr WorldObject = - ## Returns one object of a hero's visibility-filtered enumeration where - ## it sits, along with the hero's team, or nil. Nothing is copied, and - ## the object stays put until the next decision frame rebuilds the - ## enumeration, so read what is needed from it straight away. - if world.heroIndex(heroId) < 0: - return nil - team = world.ensureScriptObjects(heroId) - if index < 0 or index >= world.scriptObjectCount[team]: - return nil - world.scriptObjects[team][index].addr - - proc worldObjectAt*( world: World, heroId: int32, From 95b5f2178b0578a7f9b1a510afaab6d58f778a1a Mon Sep 17 00:00:00 2001 From: Richard Higgins Date: Sun, 4 Oct 2026 01:37:57 -0700 Subject: [PATCH 07/11] Initialize scalar hero data for lockstep policy VMs installPolicy replaces the ordinary VM with an unstructured policy. Mark that replacement as using scalar hero data so runHeroScript refreshes drafting and hero inputs, matching loadBots. Preserve the ordinary draft stateHash assertion. Co-authored-by: Codex (GPT-6) --- examples/gods_of_the_arena/lockstep.nim | 1 + 1 file changed, 1 insertion(+) diff --git a/examples/gods_of_the_arena/lockstep.nim b/examples/gods_of_the_arena/lockstep.nim index f7785140..9ecdf0a5 100644 --- a/examples/gods_of_the_arena/lockstep.nim +++ b/examples/gods_of_the_arena/lockstep.nim @@ -121,6 +121,7 @@ proc installPolicy(batch: StepBatch, lane: ptr StepLane, index: int) = let program = compile(source, host, limits) bindHeroData(program) game.heroVms[index] = HeroVm( + legacyHeroData: true, runtime: initRuntime(program, host, limits), limits: limits, ready: true ) From cc4f70e01ce1a26a3eeab222fc59aedc8a451bba Mon Sep 17 00:00:00 2001 From: Richard Higgins Date: Sun, 4 Oct 2026 01:42:42 -0700 Subject: [PATCH 08/11] Guard actual team and outgoing terminal attribution against ordinary simulation The complete lockstep suite passes with this exact test source and candidate95b5 under CPU1/RAM2GiB/no swap. Cover both controlled teams, trace-on/off canonical-state parity, potential-delta rewards and command identity across auto-reset. Existing ordinary draft stateHash assertion remains intact. Co-authored-by: Codex (GPT-6) --- tests/test_gota_lockstep.nim | 39 ++++++++++++++++++++++++++++++++++++ 1 file changed, 39 insertions(+) diff --git a/tests/test_gota_lockstep.nim b/tests/test_gota_lockstep.nim index 7fb10563..73d19fa8 100644 --- a/tests/test_gota_lockstep.nim +++ b/tests/test_gota_lockstep.nim @@ -109,6 +109,45 @@ block: doAssert stats[0].draftTicks == trained.world.draftTicks doAssert trained.finished() and ordinary.finished() +echo "Testing attribution preserves actual team and outgoing terminal identity" +for seed in [0, 7]: + let + traced = newStepBatch(config, bot, bot, policy, 1, 48, 24, false) + plain = newStepBatch(config, bot, bot, policy, 1, 48, 24, false) + traced.reset(seed) + plain.reset(seed) + traced.commandTraceLimit = 4096 + var + actions = newSeq[int32](5) + rewards, plainRewards = newSeq[float32](5) + terminals, plainTerminals = newSeq[uint8](5) + stats, plainStats = newSeq[LaneStats](1) + for frame in 0 ..< 2: + let outgoing = traced.lanes[0].game + let tickBefore = outgoing.world.tick + traced.step(actions, rewards, terminals, stats) + plain.step(actions, plainRewards, plainTerminals, plainStats) + doAssert rewards == plainRewards and terminals == plainTerminals + doAssert traced.lanes[0].game.stateHash() == plain.lanes[0].game.stateHash() + doAssert traced.commandTrace.len > 0 + for agent, trace in traced.stepTrace: + let hero = outgoing.world.heroes[trace.seat] + doAssert trace.heroId == hero.id + doAssert trace.team == int32(hero.team.ord) + doAssert trace.team == int32((seed mod 10) div 5) + doAssert trace.tickBefore == tickBefore + doAssert trace.tickAfter == outgoing.world.tick + doAssert rewards[agent] == float32(trace.potentialAfter - trace.potentialBefore) / + float32(ScoreDenominator) + for ordinal, trace in traced.commandTrace: + doAssert trace.ordinal == int32(ordinal) + doAssert trace.command.heroId == traced.stepTrace[trace.agent].heroId + doAssert stats[0].finished == int32(frame == 1) + if frame == 1: + doAssert outgoing.finished() + doAssert traced.lanes[0].game != outgoing + doAssert outgoing.world.commandAttributionLimit == 0 + echo "Testing observations and seats" let played = run(true, XpReward, 2400, 20) for agent in 0 ..< played.seats.len: From bbefc8614d560f2f8e6e4ac3e5e87778eb819968 Mon Sep 17 00:00:00 2001 From: Richard Higgins Date: Mon, 5 Oct 2026 17:51:53 -0700 Subject: [PATCH 09/11] Limit lockstep fix to scalar hero input initialization --- examples/gods_of_the_arena/lockstep.nim | 100 ++---------------------- examples/gods_of_the_arena/sim.nim | 23 +----- tests/test_gota_lockstep.nim | 96 +---------------------- 3 files changed, 9 insertions(+), 210 deletions(-) diff --git a/examples/gods_of_the_arena/lockstep.nim b/examples/gods_of_the_arena/lockstep.nim index 9ecdf0a5..689fe5e8 100644 --- a/examples/gods_of_the_arena/lockstep.nim +++ b/examples/gods_of_the_arena/lockstep.nim @@ -37,18 +37,10 @@ type LaneStats* {.bycopy.} = object ## Summary of one finished match, from the first policy team's side. - finished*, outcome*, tick*, selfPlay*, draftTicks*: int32 + finished*, outcome*, tick*, selfPlay*: int32 potential*: int64 ## Leaderboard score numerator at the end; points = potential / 7200. - CommandTrace {.bycopy.} = object - agent, ordinal: int32 - command: CommandAttribution - - StepTrace {.bycopy.} = object - heroId, seat, team, tickBefore, tickAfter: int32 - potentialBefore, potentialAfter: int64 - StepLane = object game: Game seed: int @@ -68,9 +60,6 @@ type selfPlay: bool rewardMode: RewardMode agentsPerLane*: int - commandTraceLimit: int - commandTrace: seq[CommandTrace] - stepTrace: seq[StepTrace] proc teamPotential(world: World, team: Team): int64 = ## Scales whole XP per minute into the training API's score numerator. @@ -131,7 +120,7 @@ proc startLane(batch: StepBatch, lane: ptr StepLane, seed: int) = config = loadConfig(batch.configPath) gameMap = generateMap(int32(seed), config.mapPreset) game = newGame( - gameMap, config.spawnIntervalTicks, 10, false, ReplayData() + gameMap, config.spawnIntervalTicks, 10, false, ReplayData(), false ) team = Team((seed mod 10) div 5) opponent = batch.opponents[(seed div 10) mod batch.opponents.len] @@ -160,9 +149,6 @@ proc startLane(batch: StepBatch, lane: ptr StepLane, seed: int) = if hero.team != team: batch.installPolicy(lane, index) doAssert lane.agents.len == batch.agentsPerLane - while game.world.phase == Drafting: - activeGame = game - tickWorld(game, proc() = runBotDecisions(game)) for side in Team: lane.potentials[side] = case batch.rewardMode @@ -230,9 +216,6 @@ proc step*( ) = ## Runs every lane for actionTicks ticks with the given actions. doAssert actions.len == batch.agentCount - if batch.commandTraceLimit > 0: - batch.commandTrace.setLen(0) - batch.stepTrace.setLen(batch.agentCount) var agent = 0 for laneIndex in 0 ..< batch.lanes.len: let lane = batch.lanes[laneIndex].addr @@ -240,27 +223,16 @@ proc step*( doAssert actions[agent + i] in 0 ..< GotaActionCount lane.actions[i] = actions[agent + i] let game = lane.game - if batch.commandTraceLimit > 0: - game.world.commandAttribution.setLen(0) - game.world.commandAttributionHeroes.setLen(0) - game.world.commandAttributionLimit = batch.commandTraceLimit - for i, index in lane.agents: - let hero = game.world.heroes[index] - game.world.commandAttributionHeroes.add hero.id - batch.stepTrace[agent + i] = StepTrace( - heroId: hero.id, seat: int32(index), team: int32(hero.team.ord), - tickBefore: game.world.tick, - potentialBefore: lane.potentials[hero.team] - ) var ticks = 0 - while ticks < batch.actionTicks and not game.finished(): + while ticks < batch.actionTicks and not game.world.gameOver and + game.world.tick < batch.maxTicks: activeGame = game tickWorld(game, proc() = runBotDecisions(game)) inc ticks for vm in game.heroVms: doAssert not vm.failed, vm.lastError let - done = game.finished() + done = game.world.gameOver or game.world.tick >= batch.maxTicks world = game.world var reward: array[Team, float32] for side in Team: @@ -285,9 +257,8 @@ proc step*( stats[laneIndex] = LaneStats( finished: 1, outcome: outcome(world, lane.team, true), - tick: world.battleTick(), + tick: world.tick, selfPlay: int32(batch.selfPlay), - draftTicks: world.draftTicks, potential: if outcome(world, lane.team, true) == 1: teamPotential(world, lane.team) @@ -297,22 +268,6 @@ proc step*( for i, index in lane.agents: rewards[agent + i] = reward[world.heroes[index].team] terminals[agent + i] = uint8(done) - if batch.commandTraceLimit > 0: - for command in world.commandAttribution: - for i, index in lane.agents: - if world.heroes[index].id == command.heroId: - doAssert batch.commandTrace.len < batch.commandTraceLimit, - "batch command attribution capacity exceeded" - batch.commandTrace.add CommandTrace( - agent: int32(agent + i), ordinal: int32(batch.commandTrace.len), - command: command - ) - break - for i, index in lane.agents: - batch.stepTrace[agent + i].tickAfter = world.tick - batch.stepTrace[agent + i].potentialAfter = lane.potentials[world.heroes[index].team] - # Preserve the completed world's command/reward facts before automatic reset. - world.commandAttributionLimit = 0 if done: batch.startLane(lane, lane.seed + batch.lanes.len) agent += lane.agents.len @@ -372,48 +327,5 @@ proc gota_step( seats.toOpenArray(0, count - 1) ) -proc gota_attribution_version(): cint {.cdecl, exportc, dynlib.} = - 1 - -proc gota_command_trace_size(): cint {.cdecl, exportc, dynlib.} = - cint(sizeof(CommandTrace)) - -proc gota_step_trace_size(): cint {.cdecl, exportc, dynlib.} = - cint(sizeof(StepTrace)) - -proc gota_trace_enable(handle: pointer, maxCommands: cint) {.cdecl, exportc, dynlib.} = - doAssert maxCommands > 0 - cast[StepBatch](handle).commandTraceLimit = int(maxCommands) - -proc gota_trace_identity( - handle: pointer, heroIds, gameSeats, gameTeams: ptr UncheckedArray[int32] -) {.cdecl, exportc, dynlib.} = - let batch = cast[StepBatch](handle) - doAssert batch.commandTraceLimit > 0 - var agent = 0 - for lane in batch.lanes: - for index in lane.agents: - heroIds[agent] = lane.game.world.heroes[index].id - gameSeats[agent] = int32(index) - gameTeams[agent] = int32(lane.game.world.heroes[index].team.ord) - inc agent - -proc gota_trace_count(handle: pointer): cint {.cdecl, exportc, dynlib.} = - cint(cast[StepBatch](handle).commandTrace.len) - -proc gota_trace_copy( - handle: pointer, - commands: ptr UncheckedArray[CommandTrace], capacity: cint, - steps: ptr UncheckedArray[StepTrace] -) {.cdecl, exportc, dynlib.} = - let batch = cast[StepBatch](handle) - doAssert batch.commandTraceLimit > 0 - doAssert int(capacity) >= batch.commandTrace.len - doAssert batch.stepTrace.len == batch.agentCount - for i, command in batch.commandTrace: - commands[i] = command - for i, step in batch.stepTrace: - steps[i] = step - proc gota_close(handle: pointer) {.cdecl, exportc, dynlib.} = GC_unref(cast[StepBatch](handle)) diff --git a/examples/gods_of_the_arena/sim.nim b/examples/gods_of_the_arena/sim.nim index d9a06aba..28b5de3a 100644 --- a/examples/gods_of_the_arena/sim.nim +++ b/examples/gods_of_the_arena/sim.nim @@ -272,14 +272,7 @@ type team: Team hero, fixed: bool - CommandAttribution* {.bycopy.} = object - ## Result of an actual command-handler call, not proof of later damage/movement. - tick*, heroId*, action*, slot*, first*, second*, error*, offsetX*, offsetY*: int32 - World* = ref object - commandAttributionLimit*: int - commandAttributionHeroes*: seq[int32] - commandAttribution*: seq[CommandAttribution] when defined(replayEvents): events*: seq[GameEvent] eventTick: int32 @@ -1106,14 +1099,6 @@ proc finishAction( offset = FixedVec2Zero ): bool = ## Updates only the submitting hero's diagnostic and records failed commands. - if world.commandAttributionLimit > 0 and heroId in world.commandAttributionHeroes: - doAssert world.commandAttribution.len < world.commandAttributionLimit, - "command attribution capacity exceeded" - world.commandAttribution.add CommandAttribution( - tick: world.tick, heroId: heroId, action: int32(action), slot: slot, - first: first, second: second, error: int32(error.ord), - offsetX: int32(offset.x), offsetY: int32(offset.y) - ) let index = world.heroIndex(heroId) if index >= 0: world.heroes[index].lastActionError = error @@ -3656,11 +3641,9 @@ proc updateFootman(world: World, footman: var Footman) = let other {.cursor.} = world.footmen[i] if not world.hostile(other, footman.team): continue - let distance = distanceSquared(footman.position, other.position) - if distance > bestSquared: - continue if not visible(world, footman.team, other.position): continue + let distance = distanceSquared(footman.position, other.position) if distance < bestSquared or (distance == bestSquared and bestId != 0 and targetBefore(other.position, other.id, bestPosition, bestId, footman.team)): @@ -3672,11 +3655,9 @@ proc updateFootman(world: World, footman: var Footman) = let hero = world.heroes[i] if hero.team == footman.team or hero.state == Dying or hero.hp <= 0: continue - let distance = distanceSquared(footman.position, hero.position) - if distance > bestSquared: - continue if not visible(world, footman.team, hero.position): continue + let distance = distanceSquared(footman.position, hero.position) if distance < bestSquared or (distance == bestSquared and targetFootman < 0 and bestId != 0 and targetBefore(hero.position, hero.id, bestPosition, bestId, footman.team)): diff --git a/tests/test_gota_lockstep.nim b/tests/test_gota_lockstep.nim index 73d19fa8..3457c2a7 100644 --- a/tests/test_gota_lockstep.nim +++ b/tests/test_gota_lockstep.nim @@ -1,16 +1,6 @@ include ../examples/gods_of_the_arena/lockstep -import std/[os, tempfiles] let policy = "dim f(" & $GotaFeatureCount & ")\n" & """ -if drafting then - for candidate = 0 to 9 - if heroAvailable(candidate) then - draftHero(candidate) - end - end if - next candidate - end -end if f(0) = 100 f(1) = selfTeam * 100 f(2) = selfClass * 10 @@ -64,90 +54,6 @@ echo "Testing agent counts" doAssert newStepBatch(config, bot, bot, policy, 3, 240, 24, false).agentCount == 15 doAssert newStepBatch(config, bot, bot, policy, 3, 240, 24, true).agentCount == 30 -echo "Testing ordinary draft, battle clock and frozen-action trajectories" -block: - let directory = createTempDir("gota-lockstep-", "") - let player = directory / "policy.bas" - defer: - removeFile(player) - removeDir(directory) - writeFile(player, policy.replace("' METTA_DECISION", "decision = 0")) - for seed in [0, 7]: - let batch = newStepBatch(config, bot, bot, policy, 1, 240, 24, false) - batch.reset(seed) - let trained = batch.lanes[0].game - let preset = loadConfig(config) - let ordinary = newGame(generateMap(int32(seed), preset.mapPreset), - preset.spawnIntervalTicks, 10, false, ReplayData()) - let groups = - if batch.lanes[0].team == RedTeam: - @[BotGroup(path: player, count: 5), BotGroup(path: bot, count: 5)] - else: - @[BotGroup(path: bot, count: 5), BotGroup(path: player, count: 5)] - loadBots(ordinary, groups) - ordinary.replayData = initReplayData(currentSetup(ordinary, 240), preset.mapPreset) - while ordinary.world.phase == Drafting: - activeGame = ordinary - tickWorld(ordinary, proc() = runBotDecisions(ordinary)) - doAssert trained.world.drafting - doAssert trained.world.draftTicks > 0 - doAssert trained.world.battleTick() == 0 - doAssert trained.stateHash() == ordinary.stateHash() - var - actions = newSeq[int32](batch.agentCount) - rewards = newSeq[float32](batch.agentCount) - terminals = newSeq[uint8](batch.agentCount) - stats = newSeq[LaneStats](1) - for frame in 0 ..< 10: - batch.step(actions, rewards, terminals, stats) - for tick in 0 ..< 24: - activeGame = ordinary - tickWorld(ordinary, proc() = runBotDecisions(ordinary)) - doAssert trained.stateHash() == ordinary.stateHash() - doAssert stats[0].finished == int32(frame == 9) - doAssert stats[0].tick == 240 - doAssert stats[0].draftTicks == trained.world.draftTicks - doAssert trained.finished() and ordinary.finished() - -echo "Testing attribution preserves actual team and outgoing terminal identity" -for seed in [0, 7]: - let - traced = newStepBatch(config, bot, bot, policy, 1, 48, 24, false) - plain = newStepBatch(config, bot, bot, policy, 1, 48, 24, false) - traced.reset(seed) - plain.reset(seed) - traced.commandTraceLimit = 4096 - var - actions = newSeq[int32](5) - rewards, plainRewards = newSeq[float32](5) - terminals, plainTerminals = newSeq[uint8](5) - stats, plainStats = newSeq[LaneStats](1) - for frame in 0 ..< 2: - let outgoing = traced.lanes[0].game - let tickBefore = outgoing.world.tick - traced.step(actions, rewards, terminals, stats) - plain.step(actions, plainRewards, plainTerminals, plainStats) - doAssert rewards == plainRewards and terminals == plainTerminals - doAssert traced.lanes[0].game.stateHash() == plain.lanes[0].game.stateHash() - doAssert traced.commandTrace.len > 0 - for agent, trace in traced.stepTrace: - let hero = outgoing.world.heroes[trace.seat] - doAssert trace.heroId == hero.id - doAssert trace.team == int32(hero.team.ord) - doAssert trace.team == int32((seed mod 10) div 5) - doAssert trace.tickBefore == tickBefore - doAssert trace.tickAfter == outgoing.world.tick - doAssert rewards[agent] == float32(trace.potentialAfter - trace.potentialBefore) / - float32(ScoreDenominator) - for ordinal, trace in traced.commandTrace: - doAssert trace.ordinal == int32(ordinal) - doAssert trace.command.heroId == traced.stepTrace[trace.agent].heroId - doAssert stats[0].finished == int32(frame == 1) - if frame == 1: - doAssert outgoing.finished() - doAssert traced.lanes[0].game != outgoing - doAssert outgoing.world.commandAttributionLimit == 0 - echo "Testing observations and seats" let played = run(true, XpReward, 2400, 20) for agent in 0 ..< played.seats.len: @@ -181,7 +87,7 @@ for (ended, draw, winner) in [ batch = newStepBatch(config, bot, bot, policy, 1, 28800, 24, true, LeaderboardReward) world = batch.lanes[0].game.world - world.tick = if ended: 15120 else: 28920 + world.tick = if ended: 15120 else: 28800 world.draftTicks = 120 world.gameOver = ended world.draw = draw From 331873c0b2bc5af5ce4321ab5cb7ebc2b3368b75 Mon Sep 17 00:00:00 2001 From: Richard Higgins Date: Mon, 5 Oct 2026 18:11:48 -0700 Subject: [PATCH 10/11] Use structured observations in GoTA training and export --- examples/gods_of_the_arena/lockstep.nim | 27 ++++++++++++------- .../neural/policies/andre.bas | 25 ++++++++--------- .../gods_of_the_arena/tools/convert_andre.py | 12 +++++---- .../tools/tensor_packages.py | 3 +++ examples/gods_of_the_arena/training.nim | 23 ++++++++++------ tests/test_gota_lockstep.nim | 2 +- 6 files changed, 57 insertions(+), 35 deletions(-) diff --git a/examples/gods_of_the_arena/lockstep.nim b/examples/gods_of_the_arena/lockstep.nim index 689fe5e8..adf9bbc4 100644 --- a/examples/gods_of_the_arena/lockstep.nim +++ b/examples/gods_of_the_arena/lockstep.nim @@ -37,7 +37,7 @@ type LaneStats* {.bycopy.} = object ## Summary of one finished match, from the first policy team's side. - finished*, outcome*, tick*, selfPlay*: int32 + finished*, outcome*, tick*, selfPlay*, draftTicks*: int32 potential*: int64 ## Leaderboard score numerator at the end; points = potential / 7200. @@ -86,13 +86,16 @@ proc installPolicy(batch: StepBatch, lane: ptr StepLane, index: int) = game = lane.game hero = game.world.heroes[index] agent = lane.agents.len + structured = usesStructures(batch.policy) lane.agents.add index lane.features.add default(array[GotaFeatureCount, int32]) lane.actions.add 0 var host = initHeroHost(hero.id) limits = heroVmLimits() - limits.disableFixed = true + limits.disableFixed = not structured + if structured: + limits = structureLimits(limits) limits.maxParameters = GotaFeatureCount discard host.addFunction("chooseAction", GotaFeatureCount, proc(values: openArray[int32]): int32 = @@ -107,12 +110,15 @@ proc installPolicy(batch: StepBatch, lane: ptr StepLane, index: int) = arguments.add "f(" & $i & ")" let source = batch.policy.replace("' METTA_DECISION", "decision = chooseAction(" & arguments.join(",") & ")") - let program = compile(source, host, limits) + let program = compile( + if structured: StructureSource & "\n" & source else: source, host, limits) bindHeroData(program) game.heroVms[index] = HeroVm( - legacyHeroData: true, + structured: structured, legacyHeroData: program.usesHeroData(), runtime: initRuntime(program, host, limits), limits: limits, ready: true ) + game.structuredBots = game.structuredBots or structured + game.heroVms[index].bindStructures(program, game.world, hero.id) proc startLane(batch: StepBatch, lane: ptr StepLane, seed: int) = ## Builds a fresh match for one lane and installs the policy scripts. @@ -120,7 +126,7 @@ proc startLane(batch: StepBatch, lane: ptr StepLane, seed: int) = config = loadConfig(batch.configPath) gameMap = generateMap(int32(seed), config.mapPreset) game = newGame( - gameMap, config.spawnIntervalTicks, 10, false, ReplayData(), false + gameMap, config.spawnIntervalTicks, 10, false, ReplayData() ) team = Team((seed mod 10) div 5) opponent = batch.opponents[(seed div 10) mod batch.opponents.len] @@ -149,6 +155,9 @@ proc startLane(batch: StepBatch, lane: ptr StepLane, seed: int) = if hero.team != team: batch.installPolicy(lane, index) doAssert lane.agents.len == batch.agentsPerLane + while game.world.phase == Drafting: + activeGame = game + tickWorld(game, proc() = runBotDecisions(game)) for side in Team: lane.potentials[side] = case batch.rewardMode @@ -224,15 +233,14 @@ proc step*( lane.actions[i] = actions[agent + i] let game = lane.game var ticks = 0 - while ticks < batch.actionTicks and not game.world.gameOver and - game.world.tick < batch.maxTicks: + while ticks < batch.actionTicks and not game.finished(): activeGame = game tickWorld(game, proc() = runBotDecisions(game)) inc ticks for vm in game.heroVms: doAssert not vm.failed, vm.lastError let - done = game.world.gameOver or game.world.tick >= batch.maxTicks + done = game.finished() world = game.world var reward: array[Team, float32] for side in Team: @@ -257,8 +265,9 @@ proc step*( stats[laneIndex] = LaneStats( finished: 1, outcome: outcome(world, lane.team, true), - tick: world.tick, + tick: world.battleTick(), selfPlay: int32(batch.selfPlay), + draftTicks: world.draftTicks, potential: if outcome(world, lane.team, true) == 1: teamPotential(world, lane.team) diff --git a/examples/gods_of_the_arena/neural/policies/andre.bas b/examples/gods_of_the_arena/neural/policies/andre.bas index b0852913..a69afe3c 100644 --- a/examples/gods_of_the_arena/neural/policies/andre.bas +++ b/examples/gods_of_the_arena/neural/policies/andre.bas @@ -1,3 +1,4 @@ +' @gota-structures ' Andre PufferNet helpers. BASIC owns observations, timing and sampling. ' Reusable glue and a synthetic example written for this API. ' This is not a submitted player policy and contains no learned coefficients. @@ -39,25 +40,25 @@ end sub sub andreAdvance() ' Advance from the previous captured features before the hero thinks. - if drafting then + if draft.active then exit sub end if if andreInitialized = 0 then andreState = blobCreate() andrePeriod = 24 andreTemperature = 1.0 - andreSeed = matchInfo(2) + selfId * 7919 - for andreI = 0 to draftPlayerCount() - 1 - if draftPlayerId(andreI) = selfId then + andreSeed = match.seed + self.id * 7919 + for andreI = 0 to draft.playerCount - 1 + if players(andreI).id = self.id then andreData(40 + andreI mod 5) = 1.0 end if next andreI andreInitialized = 1 end if - if worldTick < andreNextTick then + if match.tick < andreNextTick then exit sub end if - andreNextTick = worldTick + andrePeriod + andreNextTick = match.tick + andrePeriod andreResult = andre_nn("weights.bin", andreState, andreData) andreChoose() end sub @@ -79,25 +80,25 @@ end sub ' Example policy. dim f(40) andreAdvance() -if drafting then +if draft.active then for candidate = 0 to 9 - if heroAvailable(candidate) then + if heroChoices(candidate).available then draftHero(candidate) end end if next candidate end end if -if selfHp <= 0 then +if self.hp <= 0 then end end if -f(0) = selfHp * 100 \ selfMaxHp +f(0) = self.hp * 100 \ self.maxHp f(39) = decision * 9 andreCapture() decision = andreAction ' The real converted hero keeps all eleven macro behaviours in its BASIC. if decision = 1 then - walkTo(selfX, selfY) + walkTo(self.position.x, self.position.y) else - attackMove(selfX, selfY) + attackMove(self.position.x, self.position.y) end if diff --git a/examples/gods_of_the_arena/tools/convert_andre.py b/examples/gods_of_the_arena/tools/convert_andre.py index c5b25dfc..beeaa85c 100644 --- a/examples/gods_of_the_arena/tools/convert_andre.py +++ b/examples/gods_of_the_arena/tools/convert_andre.py @@ -52,11 +52,13 @@ def convert(source, weights, hidden=None, layers=None, action_ticks=24, raise ValueError("Temperature must be zero (argmax) or 0.01..10") if 0 < temperature < 0.01: raise ValueError("Positive temperature must be at least 0.01") - # The source was trained with integer-only BASIC; retain that arithmetic. - for line in source.splitlines(): - code = line.split("'", 1)[0] - if "/" in code or re.search(r"\d+\.\d+", code): - raise ValueError("Expected integer-only hero arithmetic") + structured = source.lstrip().splitlines()[0].lower() == "' @gota-structures" + if not structured: + # Unstructured lockstep policies use integer-only BASIC. + for line in source.splitlines(): + code = line.split("'", 1)[0] + if "/" in code or re.search(r"\d+\.\d+", code): + raise ValueError("Expected integer-only hero arithmetic") library = (Path(__file__).parent.parent / "neural/policies/andre.bas").read_text() library = library.split("' Example policy.")[0] library = library.replace("andrePeriod = 24", f"andrePeriod = {action_ticks}") diff --git a/examples/gods_of_the_arena/tools/tensor_packages.py b/examples/gods_of_the_arena/tools/tensor_packages.py index ff54d947..c33f9568 100644 --- a/examples/gods_of_the_arena/tools/tensor_packages.py +++ b/examples/gods_of_the_arena/tools/tensor_packages.py @@ -140,7 +140,10 @@ def replacement(match): # DIM declarations are unique even when two Richard networks are present. library = "\n".join(libraries) library = library.replace("dim tensorShape(0)\n", "") + structured = source.lstrip().splitlines()[0].lower() == "' @gota-structures" source = "dim tensorShape(0)\n" + library + "\n" + source + if structured: + source = "' @gota-structures\n" + source if len(source.encode()) > 64 * 1024: raise ValueError("BASIC source with tensor architecture exceeds 64 KiB") return (source, diff --git a/examples/gods_of_the_arena/training.nim b/examples/gods_of_the_arena/training.nim index 0a9a6f74..bc4837fe 100644 --- a/examples/gods_of_the_arena/training.nim +++ b/examples/gods_of_the_arena/training.nim @@ -97,7 +97,7 @@ proc runLane(args: LaneWorkerArgs) {.thread.} = config = loadConfig(args.configPath) gameMap = generateMap(int32(args.seed), config.mapPreset) game = newGame( - gameMap, config.spawnIntervalTicks, 10, false, ReplayData(), false + gameMap, config.spawnIntervalTicks, 10, false, ReplayData() ) team = Team((args.seed mod 10) div 5) groups = @@ -117,7 +117,7 @@ proc runLane(args: LaneWorkerArgs) {.thread.} = observed, stopped: bool proc snapshot(terminal: bool, features: openArray[int32] = []): Transition = - result.tick = game.world.tick + result.tick = game.world.battleTick() result.seat = int32(reportingSeat) result.terminal = int32(terminal) result.outcome = @@ -158,15 +158,19 @@ proc runLane(args: LaneWorkerArgs) {.thread.} = previousScore = score proc installPolicy(index: int) = - let hero = game.world.heroes[index] + let + hero = game.world.heroes[index] + structured = usesStructures(args.policy) var host = initHeroHost(hero.id) limits = heroVmLimits() - limits.disableFixed = true + limits.disableFixed = not structured + if structured: + limits = structureLimits(limits) limits.maxParameters = GotaFeatureCount discard host.addFunction("chooseAction", GotaFeatureCount, proc(values: openArray[int32]): int32 = - if stopped: + if game.world.phase == Drafting or stopped: return 0 reportingSeat = index args.bridge.sendTransition(snapshot(false, values)) @@ -194,17 +198,20 @@ proc runLane(args: LaneWorkerArgs) {.thread.} = " neuralActionCountdown = " & $GotaActionRepeat & "\n" & " decision = chooseAction(" & featureArguments.join(",") & ")\n" & "end if") - let program = compile(source, host, limits) + let program = compile( + if structured: StructureSource & "\n" & source else: source, host, limits) bindHeroData(program) game.heroVms[index] = HeroVm( + structured: structured, legacyHeroData: program.usesHeroData(), runtime: initRuntime(program, host, limits), limits: limits, ready: true ) + game.structuredBots = game.structuredBots or structured + game.heroVms[index].bindStructures(program, game.world, hero.id) for index, hero in game.world.heroes: if hero.team == team: installPolicy(index) - while not stopped and not game.world.gameOver and - game.world.tick < args.maxTicks: + while not stopped and not game.finished(): tickWorld(game, proc() = runBotDecisions(game)) for vm in game.heroVms: doAssert not vm.failed, vm.lastError diff --git a/tests/test_gota_lockstep.nim b/tests/test_gota_lockstep.nim index 3457c2a7..fd6761be 100644 --- a/tests/test_gota_lockstep.nim +++ b/tests/test_gota_lockstep.nim @@ -87,7 +87,7 @@ for (ended, draw, winner) in [ batch = newStepBatch(config, bot, bot, policy, 1, 28800, 24, true, LeaderboardReward) world = batch.lanes[0].game.world - world.tick = if ended: 15120 else: 28800 + world.tick = if ended: 15120 else: 28920 world.draftTicks = 120 world.gameOver = ended world.draw = draw From 5f33aa7b950e28d7f17684cf17ef90362cdf1e3f Mon Sep 17 00:00:00 2001 From: Richard Higgins Date: Tue, 6 Oct 2026 15:13:17 -0700 Subject: [PATCH 11/11] Preserve native command attribution with structured observations Restore bounded opt-in command traces, actual simulator team identity, and outgoing reward state before automatic reset. Keep the structured VM and ordinary draft API intact. Co-authored-by: Codex (GPT-6) --- examples/gods_of_the_arena/lockstep.nim | 85 +++++++++++++++++++++++++ examples/gods_of_the_arena/sim.nim | 15 +++++ tests/test_gota_lockstep.nim | 39 ++++++++++++ 3 files changed, 139 insertions(+) diff --git a/examples/gods_of_the_arena/lockstep.nim b/examples/gods_of_the_arena/lockstep.nim index adf9bbc4..b6421ddb 100644 --- a/examples/gods_of_the_arena/lockstep.nim +++ b/examples/gods_of_the_arena/lockstep.nim @@ -41,6 +41,14 @@ type potential*: int64 ## Leaderboard score numerator at the end; points = potential / 7200. + CommandTrace {.bycopy.} = object + agent, ordinal: int32 + command: CommandAttribution + + StepTrace {.bycopy.} = object + heroId, seat, team, tickBefore, tickAfter: int32 + potentialBefore, potentialAfter: int64 + StepLane = object game: Game seed: int @@ -60,6 +68,9 @@ type selfPlay: bool rewardMode: RewardMode agentsPerLane*: int + commandTraceLimit: int + commandTrace: seq[CommandTrace] + stepTrace: seq[StepTrace] proc teamPotential(world: World, team: Team): int64 = ## Scales whole XP per minute into the training API's score numerator. @@ -225,6 +236,9 @@ proc step*( ) = ## Runs every lane for actionTicks ticks with the given actions. doAssert actions.len == batch.agentCount + if batch.commandTraceLimit > 0: + batch.commandTrace.setLen(0) + batch.stepTrace.setLen(batch.agentCount) var agent = 0 for laneIndex in 0 ..< batch.lanes.len: let lane = batch.lanes[laneIndex].addr @@ -232,6 +246,18 @@ proc step*( doAssert actions[agent + i] in 0 ..< GotaActionCount lane.actions[i] = actions[agent + i] let game = lane.game + if batch.commandTraceLimit > 0: + game.world.commandAttribution.setLen(0) + game.world.commandAttributionHeroes.setLen(0) + game.world.commandAttributionLimit = batch.commandTraceLimit + for i, index in lane.agents: + let hero = game.world.heroes[index] + game.world.commandAttributionHeroes.add hero.id + batch.stepTrace[agent + i] = StepTrace( + heroId: hero.id, seat: int32(index), team: int32(hero.team.ord), + tickBefore: game.world.tick, + potentialBefore: lane.potentials[hero.team] + ) var ticks = 0 while ticks < batch.actionTicks and not game.finished(): activeGame = game @@ -277,6 +303,22 @@ proc step*( for i, index in lane.agents: rewards[agent + i] = reward[world.heroes[index].team] terminals[agent + i] = uint8(done) + if batch.commandTraceLimit > 0: + for command in world.commandAttribution: + for i, index in lane.agents: + if world.heroes[index].id == command.heroId: + doAssert batch.commandTrace.len < batch.commandTraceLimit, + "batch command attribution capacity exceeded" + batch.commandTrace.add CommandTrace( + agent: int32(agent + i), ordinal: int32(batch.commandTrace.len), + command: command + ) + break + for i, index in lane.agents: + batch.stepTrace[agent + i].tickAfter = world.tick + batch.stepTrace[agent + i].potentialAfter = lane.potentials[world.heroes[index].team] + # Preserve the completed world's command/reward facts before automatic reset. + world.commandAttributionLimit = 0 if done: batch.startLane(lane, lane.seed + batch.lanes.len) agent += lane.agents.len @@ -336,5 +378,48 @@ proc gota_step( seats.toOpenArray(0, count - 1) ) +proc gota_attribution_version(): cint {.cdecl, exportc, dynlib.} = + 1 + +proc gota_command_trace_size(): cint {.cdecl, exportc, dynlib.} = + cint(sizeof(CommandTrace)) + +proc gota_step_trace_size(): cint {.cdecl, exportc, dynlib.} = + cint(sizeof(StepTrace)) + +proc gota_trace_enable(handle: pointer, maxCommands: cint) {.cdecl, exportc, dynlib.} = + doAssert maxCommands > 0 + cast[StepBatch](handle).commandTraceLimit = int(maxCommands) + +proc gota_trace_identity( + handle: pointer, heroIds, gameSeats, gameTeams: ptr UncheckedArray[int32] +) {.cdecl, exportc, dynlib.} = + let batch = cast[StepBatch](handle) + doAssert batch.commandTraceLimit > 0 + var agent = 0 + for lane in batch.lanes: + for index in lane.agents: + heroIds[agent] = lane.game.world.heroes[index].id + gameSeats[agent] = int32(index) + gameTeams[agent] = int32(lane.game.world.heroes[index].team.ord) + inc agent + +proc gota_trace_count(handle: pointer): cint {.cdecl, exportc, dynlib.} = + cint(cast[StepBatch](handle).commandTrace.len) + +proc gota_trace_copy( + handle: pointer, + commands: ptr UncheckedArray[CommandTrace], capacity: cint, + steps: ptr UncheckedArray[StepTrace] +) {.cdecl, exportc, dynlib.} = + let batch = cast[StepBatch](handle) + doAssert batch.commandTraceLimit > 0 + doAssert int(capacity) >= batch.commandTrace.len + doAssert batch.stepTrace.len == batch.agentCount + for i, command in batch.commandTrace: + commands[i] = command + for i, step in batch.stepTrace: + steps[i] = step + proc gota_close(handle: pointer) {.cdecl, exportc, dynlib.} = GC_unref(cast[StepBatch](handle)) diff --git a/examples/gods_of_the_arena/sim.nim b/examples/gods_of_the_arena/sim.nim index 28b5de3a..3aed319b 100644 --- a/examples/gods_of_the_arena/sim.nim +++ b/examples/gods_of_the_arena/sim.nim @@ -272,7 +272,14 @@ type team: Team hero, fixed: bool + CommandAttribution* {.bycopy.} = object + ## Result of an actual command-handler call, not proof of later damage/movement. + tick*, heroId*, action*, slot*, first*, second*, error*, offsetX*, offsetY*: int32 + World* = ref object + commandAttributionLimit*: int + commandAttributionHeroes*: seq[int32] + commandAttribution*: seq[CommandAttribution] when defined(replayEvents): events*: seq[GameEvent] eventTick: int32 @@ -1099,6 +1106,14 @@ proc finishAction( offset = FixedVec2Zero ): bool = ## Updates only the submitting hero's diagnostic and records failed commands. + if world.commandAttributionLimit > 0 and heroId in world.commandAttributionHeroes: + doAssert world.commandAttribution.len < world.commandAttributionLimit, + "command attribution capacity exceeded" + world.commandAttribution.add CommandAttribution( + tick: world.tick, heroId: heroId, action: int32(action), slot: slot, + first: first, second: second, error: int32(error.ord), + offsetX: int32(offset.x), offsetY: int32(offset.y) + ) let index = world.heroIndex(heroId) if index >= 0: world.heroes[index].lastActionError = error diff --git a/tests/test_gota_lockstep.nim b/tests/test_gota_lockstep.nim index fd6761be..a75c9edf 100644 --- a/tests/test_gota_lockstep.nim +++ b/tests/test_gota_lockstep.nim @@ -54,6 +54,45 @@ echo "Testing agent counts" doAssert newStepBatch(config, bot, bot, policy, 3, 240, 24, false).agentCount == 15 doAssert newStepBatch(config, bot, bot, policy, 3, 240, 24, true).agentCount == 30 +echo "Testing attribution preserves actual team and outgoing terminal identity" +for seed in [0, 7]: + let + traced = newStepBatch(config, bot, bot, policy, 1, 48, 24, false) + plain = newStepBatch(config, bot, bot, policy, 1, 48, 24, false) + traced.reset(seed) + plain.reset(seed) + traced.commandTraceLimit = 4096 + var + actions = newSeq[int32](5) + rewards, plainRewards = newSeq[float32](5) + terminals, plainTerminals = newSeq[uint8](5) + stats, plainStats = newSeq[LaneStats](1) + for frame in 0 ..< 2: + let outgoing = traced.lanes[0].game + let tickBefore = outgoing.world.tick + traced.step(actions, rewards, terminals, stats) + plain.step(actions, plainRewards, plainTerminals, plainStats) + doAssert rewards == plainRewards and terminals == plainTerminals + doAssert traced.lanes[0].game.stateHash() == plain.lanes[0].game.stateHash() + doAssert traced.commandTrace.len > 0 + for agent, trace in traced.stepTrace: + let hero = outgoing.world.heroes[trace.seat] + doAssert trace.heroId == hero.id + doAssert trace.team == int32(hero.team.ord) + doAssert trace.team == int32((seed mod 10) div 5) + doAssert trace.tickBefore == tickBefore + doAssert trace.tickAfter == outgoing.world.tick + doAssert rewards[agent] == float32(trace.potentialAfter - trace.potentialBefore) / + float32(ScoreDenominator) + for ordinal, trace in traced.commandTrace: + doAssert trace.ordinal == int32(ordinal) + doAssert trace.command.heroId == traced.stepTrace[trace.agent].heroId + doAssert stats[0].finished == int32(frame == 1) + if frame == 1: + doAssert outgoing.finished() + doAssert traced.lanes[0].game != outgoing + doAssert outgoing.world.commandAttributionLimit == 0 + echo "Testing observations and seats" let played = run(true, XpReward, 2400, 20) for agent in 0 ..< played.seats.len: