diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index 0ccbcf58..a904ee75 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -33,6 +33,7 @@ jobs: run: nim r tests/test_neural_cases.nim tmp/neural_cases - name: coworld.nim glue unit tests (readPlayerSource, failPlayer) run: nim r coworld/tools/test_coworld_glue.nim + - run: nim r -d:nimAllocStats tests/test_mailbox_allocations.nim - run: nim r tests/test_gota_events.nim - run: nim r -d:replayEvents tests/test_gota_controls.nim - run: nim r -d:replayEvents tests/test_gota_camps.nim @@ -60,3 +61,13 @@ jobs: - run: nim r -d:headless -d:recordGota tests/test_recordings.nim --bot examples/gods_of_the_arena/players/base.bas:10 - run: nim r -d:headless -d:recordHlf tests/test_recordings.nim --bot examples/heartleaf/players/base.bas:9 - run: nim r -d:headless -d:recordLvd tests/test_recordings.nim --bot examples/light_vs_dark/players/base.bas:2 + - run: nim r -d:headless tests/test_gota_neural.nim + - name: No-neural parity with main (golden hashes) + shell: bash + run: bash tests/gota_golden.sh + - name: Native training library compiles + if: runner.os == 'Linux' + run: >- + nim c --app:lib -d:headless -d:gotaTrainingStats --mm:atomicArc + --threads:on -d:useMalloc -u:nimTypeNames -o:tmp/libgota_env.so + examples/gods_of_the_arena/native_env.nim diff --git a/coworld/dependencies.lock b/coworld/dependencies.lock index 29ff8316..95e9cb58 100644 --- a/coworld/dependencies.lock +++ b/coworld/dependencies.lock @@ -1,4 +1,4 @@ -bassy 0.1.0 https://github.com/treeform/bassy 2c54d822f775ce92d8b109f2f8aedb4c4b56b29d +bassy 0.1.0 https://github.com/treeform/bassy 669a7c4b94e3d5b0a38dc9557608a7e58c2a764d fixxy 0.1.0 https://github.com/treeform/fixxy 05e5446dffb70093056cebb0c57721a60deaf52a silky 0.2.0 https://github.com/treeform/silky fb9b13910edd66cf1751056784c2f7d2932a59fc pixie 6.1.0 https://github.com/treeform/pixie 87cecced5c4c6f311c658a5f3ca0c9b43edb6aa7 diff --git a/coworld/gota/coworld_manifest_template.json b/coworld/gota/coworld_manifest_template.json index 6d4fd53f..a2fa9f3b 100644 --- a/coworld/gota/coworld_manifest_template.json +++ b/coworld/gota/coworld_manifest_template.json @@ -231,7 +231,7 @@ "docs": { "readme": { "type": "text", - "value": "# Gods of the Arena\n\n**New GotA week: everyone needs to update their bot.** Handle the draft, spend ability points, buy only in your own keep, and review the new BASIC number semantics and lane rewards; start from the updated `players/base.bas`.\n\nTwo teams of five BASIC heroes battle to slay the enemy god. Each hero's ladder\nscore is lifetime XP minus 200 per simulated minute. Fractional minutes count,\nincluding drafting time. Losses and timeouts retain their time-adjusted XP,\nwith each hero's score rounded down to whole points and clamped to zero before\naveraging. Victory is recorded separately in the outcome.\n\nDestroying the enemy god grants every hero on your team a flat 1,000 XP,\nincluding dead heroes and heroes elsewhere on the map, regardless of who lands\nthe last hit. This is awarded once on the final tick and included in lifetime XP\nbefore calculating scores. It offsets 5 minutes of the 200-XP-per-minute time\npenalty. Timeouts grant no god reward.\nDestroying a tower or barracks grants its killer 200 XP and 75 gold.\n\nThe gods are the objectives: Hades for Red and Zeus for Blue. Each god has two level-3 guard towers. Clearing all three towers in any one lane exposes the guards. The god cannot take damage from attacks or spells until both of its guards are destroyed. Guards have the same 3900 HP and 60 damage as level-3 lane towers.\n\nSlots 0\u20134 are Red and slots 5\u20139 are Blue. Platform slots are zero-based. Upload a `.bas` file containing BASIC source. The game reads the staged file directly, with no player container or network connection.\n\nStart with the bundled `players/base.bas`. The [game documentation](https://github.com/Metta-AI/polyworld/blob/main/examples/gods_of_the_arena/docs/index.html) describes observations and available BASIC commands. The same source is available under `examples/gods_of_the_arena/bots.nim` and `content.nim`.\n\nThe baseline is a playable reference for all GotA host functions. It drafts\nmissing roles, farms lanes, prioritizes last hits, pushes exposed buildings and\nthe enemy god, upgrades and explicitly casts spells, leads area shots, dodges\nvisible warnings, and uses enemy stats and equipment to judge fights. It also\nshops, stacks and uses recovery items, returns to spawn, uses portal scrolls,\nand buys back when affordable. Actions are conditional on a useful opportunity,\nso one match need not exercise every mechanic. Its observation scans are bounded\nand its main decisions run every six ticks. Abilities and items require\nexplicit policy commands; the engine never casts or uses them automatically.\nSpell ranges and shapes are not queryable, so their small policy table must\nfollow balance changes in `content.nim`; health, damage, costs, and ranks use\nlive observations. This is an editable starting point, not an optimal policy.\n\nEvery hero has a free single-target melee or ranged basic attack in addition\nto four abilities. Basic damage grows each level and includes equipment bonuses.\nIdle heroes automatically acquire the nearest visible, attackable enemy,\nincluding enemy creeps, heroes, and exposed buildings. `attackTarget(objectId)`\ntakes priority; melee and ranged heroes both move into their own attack range\nand repeat basic attacks. `walkTo(x, y)` cancels the attack and suppresses\nautomatic acquisition while walking. Basic attacks do not spend mana or spell\ncharges.\n\n`attackMove(x, y)` uses the same attack-move order as the player controls. It follows a path toward that tile, stops for enemies in the hero's normal acquisition range, and resumes afterward. Like other actions, it returns 1 when accepted and 0 when rejected, and is recorded in replays.\n\nThe bundled `players/rusher.bas` sends all five heroes down mid together. It regroups toward the living team's center when any pair is more than 10 tiles apart, closing to 8 tiles before resuming. It attacks visible, vulnerable enemies within 20 tiles, favoring the enemy closest to the group's center. Otherwise it attack-moves through the middle and toward the opposing god. Dead allies are ignored until they respawn. This policy uses basic attacks only; add explicit casting commands to use abilities.\n\n## Drafting\n\nEvery live match starts with a shared pool of ten heroes. A seeded random\nteam picks first. Teams alternate, and each team's players pick in spawn\norder. Each hero can be selected once across both teams. Combat, waves,\nand the battle clock wait until all ten players have drafted.\n\nOnly the active player's BASIC script runs during drafting, once every\nhalf second. Every player, including humans, has ten simulation seconds to\npick. At the deadline, the game picks a random available hero if the player\nhas not chosen one. The bundled base and rusher scripts try to fill\nfive roles: frontline, carry, mage, support, and fighter. This is a policy\npreference, not a draft rule. Any combination of available heroes is allowed.\nTheir normal movement and combat logic runs after drafting.\n\n| Data or command | Meaning |\n| --- | --- |\n| `drafting` | 1 during drafting, 0 during battle. |\n| `draftTurnId` | ID of the player picking now, or 0 after drafting. |\n| `draftPlayerCount()` | Number of players in the public roster. |\n| `draftPlayerId(i)` | Player ID at zero-based spawn index `i`, or 0 if invalid. |\n| `draftPlayerTeam(i)` | Team at spawn index `i`: Red 0, Blue 1, invalid -1. |\n| `draftedClass(id)` | Hero class picked by player ID, or -1 if unpicked or invalid. Both teams' picks are public. |\n| `heroAvailable(class)` | 1 if the class is valid and unpicked, otherwise 0. |\n| `heroRole(class)` | Frontline 0, carry 1, mage 2, support 3, fighter 4, invalid -1. |\n| `draftHero(class)` | Selects a hero on your turn. Returns 1 on success, 0 on rejection. |\n\n`selfClass` is -1 until your pick. Class constants are `VanguardKnight`,\n`Ranger`, `Arcanist`, `DruidWarden`, `DemonHunter`, `DeathKnight`,\n`Crossbowman`, `Lich`, `Warlock`, and `Berserker`, with IDs 0 through 9.\nDraft queries update immediately. Other sampled self data updates next\nturn. `worldTick` includes draft ticks for action replay timing.\n\n```basic\nif drafting then\n for candidate = 0 to 9\n if heroAvailable(candidate) then\n draftHero(candidate)\n exit for\n end if\n next\nelse\n attackMove(mapWidth / 2, mapHeight / 2)\nend if\n```\n\n`lastActionError()` reports `ActionNotDrafting`, `ActionNotDraftTurn`,\n`ActionUnknownHero`, or `ActionHeroTaken` for rejected picks. Normal game\ncommands are rejected with `ActionDrafting` while players are picking.\nIn player mode, select an available hero in the drafting screen and click\n**Lock in hero** when it is your turn. The current picker appears above\nthe hero grid. Picked heroes turn gray and show their player and team.\nThe countdown shows the active pick's remaining time. Space pauses or\nresumes drafting, including the countdown. Drafting has a separate budget\nof up to 100 simulation seconds for ten players. The configured `maxTicks`\nand CLI duration flags limit battle time only, starting after the last pick.\n\n## Ability progression\n\nHeroes start at level 1 with one ability point and all four abilities locked.\nEach hero level grants another point. Points remain banked until a command\nspends them. `levelAbility(slot)` spends one point to unlock rank 1 or upgrade\nan already learned ability. Slots 0, 1, 2, and 3 correspond to Q, W, E, and R.\n\nQ, W, and E have four ranks requiring hero levels 1, 3, 5, and 7.\nR has three ranks requiring hero levels 6, 12, and 18. Each additional rank\nadds 50% of the rank-1 damage, healing, or mana restoration, rounded down.\nMana costs, range, charge capacity, and timing stay the same. An upgrade\npreserves spent charges and running cooldowns. Pending spells retain the\nrank they had when cast. Respawning preserves learned ranks and banked\npoints and refills only learned abilities.\n\nHero stats, including basic-attack damage, still grow automatically with\nhero level. Ability ranks never increase automatically. The bundled base\nand rusher policies explicitly spend points, prioritizing R, W, E, then Q.\nAbility use requires explicit `castTarget` or `castPoint` commands, including\nslot 0 and ultimates. Items require `useItem` or `useItemAt`. Custom policies\ncan bank points, choose another upgrade order, and reserve any ability. At\nlevel 20, fully ranking all four abilities leaves five banked points.\n\n| BASIC function | Meaning |\n| --- | --- |\n| `levelAbility(slot)` | Unlock or upgrade. Returns 1 on success, 0 on rejection. |\n| `abilityPoints()` | Current unspent points. |\n| `abilityLevel(slot)` | Current rank, with 0 meaning locked. |\n| `abilityMaxLevel(slot)` | Rank limit: 4 for slots 0-2, 3 for slot 3. |\n| `abilityRequiredLevel(slot)` | Hero level required for the next rank, or 0 at maximum rank. |\n| `canLevelAbility(slot)` | 1 when alive with a point and the required level, otherwise 0. |\n| `abilityDamage(slot)` | Damage per target at the learned rank, or 0 when locked. |\n| `abilityHeal(slot)` | Healing per target at the learned rank, or 0 when locked. |\n| `abilityRestore(slot)` | Mana restoration at the learned rank, or 0 when locked. |\n| `abilityManaCost(slot)` | Mana cost of a cast, including while locked. |\n\nThese queries update immediately after commands and return 0 for invalid\nslots. `abilityCharges(slot)`, `abilityCooldown(slot)`, and\n`abilityRecharge(slot)` remain available. Locked abilities have no charges.\nUpgrade actions and rejected attempts are recorded for deterministic replays.\n\n```basic\nif canLevelAbility(3) then\n levelAbility(3)\nelseif canLevelAbility(1) then\n levelAbility(1)\nend if\n```\n\nPlayer controls use Shift+Q/W/E/R, Shift-click on an ability, or its gold\n\"+\" button to spend a point. The HUD shows current ranks, locked abilities,\nand available points.\n\n## BASIC observations\n\nAll bots in one decision phase observe the same starting objects and spell\nwarnings. Earlier bots' actions do not change later bots' observations in\nthat phase. New casts appear in the next observation frame. Own inventory,\nability costs, and command results still update immediately when a command\nis accepted. Object and spell indices are zero-based and may change next\ndecision. Keep `objectId(i)` when tracking an object across decisions or\ncalling `attackTarget`, rather than keeping its list index.\n\nObservations are integers. `worldScale = 60000` is the number of world units per tile, and `tickRate = 24` is the number of simulation ticks per second. `selfX`, `selfY`, `objectX(i)`, `objectY(i)`, `spellX(i)`, and `spellY(i)` use whole global tiles. Facing, speed, range, and velocity retain sub-tile precision in world units. The Y component of these APIs is the second horizontal map axis, not height.\n\nAt an exact tile boundary, Red observers select the higher cell and Blue\nobservers select the lower cell. This makes observed cells rotate exactly\nwhen the arena and teams are swapped. The bundled policies convert global\ncoordinates into their own team's frame before rounding spatial decisions.\n\nObjects appear in groups: gods, buildings, heroes, then creeps. Within each\ngroup, allies precede enemies, followed by position in the observer's team\nframe and stable ID. This order does not depend on simulation storage order.\nSpell warnings are ordered by impact tick, hostile before allied casts,\nthen team-relative position, ability, and stable caster/target identities.\n\nUnits plan movement and attacks from the same starting actor state. Movement\nis published together, then spell impacts and collected damage resolve before\ndeaths and rewards. Opponents can kill each other in the same tick. Collision\ncorrections are also accumulated before moving any participant. Simultaneous\nlast-hit credit uses the match seed, tick, and team-relative actor geometry\nand role, so corresponding fights do not depend on faction-specific IDs.\n\n### Your hero\n\nThe existing `selfId`, `selfTeam`, `selfClass`, `selfX`, `selfY`, `selfHp`, `selfMaxHp`, `selfMana`, `selfMaxMana`, `selfGold`, `selfLevel`, `selfLayer`, and `worldTick` remain available. These additional read-only values describe the current hero:\n\n| Value | Meaning |\n| --- | --- |\n| `selfMoveSpeed` | Unblocked movement speed in world units per tick, including level and equipment bonuses. |\n| `selfAttackRange` | Basic-attack range in world units, measured by planar Euclidean distance between centers. Towers and barracks allow at least 105000 units measured from their occupied footprint; gods use 255000 units. |\n| `selfAttackDamage` | Current basic-attack damage, including level and equipment bonuses. |\n| `selfTarget` | Current ordered or automatically acquired attack target's stable object ID, or zero for none. |\n| `selfAttackCooldown` | Ticks until the next basic hit could land if the target stays in range. Includes remaining recovery and the next windup, or the remainder of a current windup. An idle hero reports a full windup. Excludes chasing and is separate from ability cooldowns. Movement can cancel a swing. |\n| `selfAttacksLanded` | Lifetime count of successful basic hits, preserved across respawns. Spells do not increment it. |\n| `selfPortalCooldown` | Ticks before another Portal Scroll can be used, shared across all inventory stacks and preserved through death. |\n| `selfChannelTicks` | Ticks remaining in the current teleport channel, or zero. |\n| `selfStunTicks`, `selfRootTicks`, `selfSilenceTicks` | Ticks remaining in these control effects, or zero. |\n\n### Shop, potions, and spawn recovery\n\nPurchases only work inside your own keep, including its spawn room. Elsewhere,\n`buyItem(id)` returns 0 with `ActionOutsideKeep`. `canShop()` returns 1 when\npurchases are allowed, and `inOwnSpawn()` returns 1 inside your living hero's\nown spawn room. These queries read live state.\n\nInside that spawn room, health and mana each recover at **20% of maximum per\nsecond**, capped at maximum. The larger keep and the enemy spawn give no such\nrecovery. Normal passive mana regeneration still applies. Damage does not\nturn off spawn recovery, and dead heroes cannot recover until they respawn.\n\n| ID | Item | Gold | Effect |\n| --- | --- | --- | --- |\n| 1 | Health Potion | 30 | 120 health over 10 seconds. |\n| 2 | Vitality Elixir | 75 | 90 health immediately. |\n| 22 | Mana Potion | 45 | 90 mana over 10 seconds. |\n| 3 | Mana Elixir | 90 | 60 mana immediately. |\n\nEach stacks to **8** per slot. `useItem(slot)` spends one dose. Any positive\nincoming damage interrupts both active potion regeneration effects; movement\nand attacks do not. Health items share a **10-second** cooldown, and mana items\nshare a separate **10-second** cooldown. Cooldowns begin on use and survive\ninterruption and death. Full health/mana or a cooldown rejects use without\nspending a dose. `itemCooldown(slot)` returns live remaining ticks (24 per\nsecond), or 0 for empty/invalid slots. Inventory icons show stack counts and\nremaining cooldowns.\n\n### Portal Scrolls\n\nBuy item **21** for **100 gold**. Scrolls stack to eight per slot. Call\n`useItemAt(slot, x, y)` with whole or fractional map coordinates to consume\none scroll and begin a **3-second** channel. The destination is the nearest\nvisible, walkable point inside a living allied tower's sight radius:\n7 tiles for outer/inner towers, 8 for gate/guard towers. Barracks are not\nanchors. A distant requested point is clamped into this area; there is no\ntravel-distance limit. The selected tower must survive until arrival.\n\nThe hero cannot move, attack, or cast while channeling, and still takes\ndamage. Stuns, roots, death, or loss of the anchor interrupt the channel.\nThe scroll is spent when the channel begins. Completion or interruption\nstarts a **60-second** cooldown shared by every scroll the hero holds.\nDamage alone and silence do not interrupt it. Blazing Blade, Golem Seed,\nand Bone Marionette can interrupt the channel on impact.\n`useItem(slot)` rejects scrolls because they require a destination.\n\nFor human play, click the inventory scroll (or press F/G for the first two\nslots), then right-click the map or minimap. Purple circles show tower range, and a\nchannel bar shows the time remaining. Esc cancels destination selection.\n\n### Visible objects\n\nLoop over indices `0` through `objectCount() - 1`. Object kinds are 1 = god, 2 = hero, 3 = creep, 4 = tower, 5 = barracks, and 6 = neutral mob. Barracks have 950 HP and become exposed after their lane towers fall. Each barracks spawns three melee creeps and one ranged creep per wave, giving six melee creeps and two ranged creeps per lane for each team. Ranged creeps carry a staff and cast magic bolts from up to four tiles away. Destroying a barracks stops its four creeps from spawning. For creeps, `objectClass(i)` is 0 for melee and 1 for ranged. Destroyed buildings leave the object list and release their occupied tiles. New queries respect the same visibility filter:\n\n| Function | Meaning |\n| --- | --- |\n| `objectLevel(i)` | Hero level. |\n| `objectMana(i)` | Hero's current mana. |\n| `objectStunTicks(i)`, `objectSilenceTicks(i)`, `objectRootTicks(i)` | Remaining control ticks for a visible unit, or zero. Uses the same frozen, LOS-filtered object list. |\n| `objectItemId(i, slot)` | Hero's held item ID, using the same IDs as `itemId` and `buyItem`. Zero means no item. Slots are 0 through 5. |\n| `objectItemCount(i, slot)` | Stack count in that hero's inventory slot. |\n| `objectFacingX(i)`, `objectFacingY(i)` | Normalized horizontal facing, scaled by `worldScale`. A unit facing positive X reports `(60000, 0)`. |\n| `objectTarget(i)` | Current attack target's stable object ID, or zero if absent or not visible to your team. |\n| `objectVelX(i)`, `objectVelY(i)` | Actual displacement over the last simulation tick in world units, including collision adjustments. Stationary objects report zero. |\n\nThese new object queries return zero for invalid indices or fields that do not apply to that object. Invalid inventory slots also return zero. Hero level, mana, and inventory queries return zero for non-heroes. An object's ID is not a valid substitute for its list index.\n\nEach slain enemy creep provides a shared pool of 15 XP to living heroes within six tiles on the same navigation floor, regardless of starting lane. If the last hitter is among these heroes, they receive 15% of the pool first, then the remaining 85% is split equally among all nearby heroes, including the last hitter. With three heroes, this gives 6.5 XP to the last hitter and 4.25 XP to each teammate. Fractional XP carries forward between kills. If no eligible hero lands the last hit, the full pool is shared equally. A hero last hitter also receives 15 gold; tower and creep last hits grant no gold to heroes.\n\n### Pending spells and warnings\n\nLoop over `0` through `spellCount() - 1`. This list contains unresolved casts from their start through impact, including projectiles and area warnings. Allied casts are observable; enemy casts require their aim position to be visible, matching the viewer's warning visibility. Completed effects are omitted.\n\n| Function | Meaning |\n| --- | --- |\n| `spellAbility(i)` | Ability enum ID from `content.nim`, beginning at zero. Invalid indices return -1. |\n| `spellCasterId(i)` | Caster's stable object ID, or zero if the enemy caster is hidden. |\n| `spellX(i)`, `spellY(i)` | Aim/impact position or area center in whole global tiles. This is not the projectile's interpolated flight position. |\n| `spellImpactTick(i)` | Absolute simulation tick at impact. Subtract `worldTick` to obtain the remaining ticks. |\n\nOther invalid spell queries return zero. Visibility of an enemy warning does not reveal its hidden caster's identity.\n\n## Death and buyback\n\nThe first death takes 9 seconds to respawn, including the 1-second death\nanimation. Each subsequent death adds 5 seconds, up to a total of 60 seconds.\nDeath counts belong to each hero and persist after respawning.\n\n`selfDeaths` is the hero's death count. `selfRespawnTicks` is the remaining\nrespawn delay in ticks, or zero while alive. BASIC decisions continue while\ndead so a policy can request buyback.\n\n`buybackPrice()` returns the dead hero's price: 100 gold times their death\ncount. It returns zero while alive or after the match ends. The price stays\nfixed during a death, even as the respawn timer counts down.\n\n`buyback()` returns 1 when accepted and 0 when rejected. It spends the\nhero's gold and immediately respawns them with full health, mana, and spell\ncharges. It preserves inventory, level, XP, and death count. A living hero,\nan ended match, or insufficient gold causes rejection without spending gold.\nThe HUD shows the countdown, buyback price, and any rejection reason while\ndead. Buyback attempts are recorded for replay and seeking.\n\nThe bundled `players/base.bas` buys back as soon as it can afford the price.\nWhile dead it skips normal commands, resuming them on the decision after buyback.\n\n```basic\nif selfRespawnTicks > 0 then\n price = buybackPrice()\n if price > 0 and selfGold >= price then\n accepted = buyback()\n end if\nend if\n```\n\n## Crowd control\n\n| Hero | Ability | Effect | Rank-one damage |\n| --- | --- | --- | --- |\n| Vanguard | R: Blazing Blade | Stun for 1 second | 72 |\n| Warlock | E: Dread Totem | Silence for 2 seconds | 70 |\n| Druid | R: Golem Seed | Root for 2 seconds | 68 |\n| Lich | E: Bone Marionette | Root for 1 second | 53 |\n\nThese abilities trade about 20% of their damage for control. Durations stay\nfixed at every rank. Effects apply on impact to enemy heroes and creeps;\nbuildings and gods are immune. A stun stops movement, basic attacks,\nabilities, and items, and cancels a pending basic swing. Silence prevents\nall four abilities but allows movement, basic attacks, and items. Root\nprevents movement and teleporting, while allowing in-range attacks,\nabilities, and other items. Stun and root interrupt teleport channels.\nAlready released spells still resolve.\n\nDifferent effects coexist. Reapplying one keeps the later expiration,\nwithout adding durations. All effects clear on death and respawn. Visible\naffected units show icons with countdown rings above their health bars.\n\n## Action feedback\n\n`lastActionError()` returns the reason for your hero's latest submitted\ncommand. A successful action clears it to `NoActionError` (0). A failed\naction returns 0 as before, and sets the first failing validation reason.\nRead-only queries and automatic basic attacks do not change it.\nUnlike the sampled self data, this query updates immediately after commands.\n\n```basic\naccepted = castTarget(1, targetId)\nif accepted = 0 and lastActionError() = ActionInsufficientMana then\n print \"Need more mana\"\nend if\n```\n\nRead-only reason constants are `NoActionError`, `ActionNotAlive`,\n`ActionInvalidSlot`, `ActionUnknownItem`, `ActionInsufficientGold`,\n`ActionAlreadyEquipped`, `ActionStackFull`, `ActionInventoryFull`,\n`ActionEmptySlot`, `ActionNotConsumable`, `ActionFullHealth`, `ActionFullMana`,\n`ActionTargetUnavailable`, `ActionOutOfRange`, `ActionNoRoute`,\n`ActionInvalidPoint`, `ActionCooldown`, `ActionNoCharges`,\n`ActionInsufficientMana`, `ActionSpellLimit`, `ActionChanneling`,\n`ActionStunned`, `ActionRooted`, `ActionOutsideKeep`, `ActionAbilityLocked`,\n`ActionNoAbilityPoints`, `ActionAbilityMaxLevel`, `ActionHeroLevelRequired`,\n`ActionNotDead`, `ActionMatchEnded`, `ActionDrafting`, `ActionNotDrafting`,\n`ActionNotDraftTurn`, `ActionUnknownHero`, `ActionHeroTaken`, and `ActionSilenced`\n(values 0 through 35).\nUnavailable targets share a generic error without exposing hidden state.\nThis feedback is recorded deterministically through submitted replay actions.\n\nFor post-match analysis, the [replay extractor](../../docs/stats.md#gota-replay-events)\nresimulates an exact-version replay and exposes typed damage, healing,\ndeath, reward, and rejection events for all players. Its omniscient buffer\nis not available to live BASIC policies.\n\n## Terrain and execution\n\nBASIC can inspect the complete static terrain with `terrainKind(x, y)`, `terrainWalkable(x, y)`, `terrainHeight(x, y)`, and `terrainWaterDepth(x, y)`. These use global tile coordinates on `selfLayer`. Each has an explicit `At(x, y, layer)` version, such as `terrainKindAt(x, y, GroundLayer)`. Read-only constants expose `mapWidth`, `mapHeight`, `mapLayers`, the layer names, and terrain kinds. Height and water depth use eighths of a tile; invalid or absent tiles return zero. The Terrain API section of the game documentation lists all constants and edge cases. Static terrain is available through fog, while enemy objects remain visibility-filtered. Walkability also includes team-known building footprints; unseen enemy destruction does not reveal newly open tiles.\n\nBASIC `PRINT` output, compiler diagnostics, runtime errors, and VM lifecycle messages go to the owning player's private log. Each log is limited to 10 MiB. Runtime limit errors disable that VM; other seats continue. Invalid BASIC syntax fails the episode with a player failure diagnostic. Public game logs and action replays contain no BASIC source or private print output.\n\nBattles run up to 28,800 deterministic ticks (20 simulated minutes), plus drafting time, without real-time pacing. Replays run entirely in the browser with playback, seeking, speed, and loop controls. The server exposes `/healthz`; legacy clients are static stubs.\n\nThe Competition league schedules 24 games per round with random matchups,\non a 32-minute interval. Each match uses ten distinct policies when at least\nten are eligible: five different policies on Red and five on Blue, with\none hero per policy. The scheduler uses `team_n`, `team_count: 2`,\n`team_layout: \"blocks\"`, `matchmaking: \"random\"`, and\n`distinct_teammates: true`. Preserve these settings when updating the league.\nSeparate baseline filler policies complete short rosters and are not ranked\nentrants. A policy controlling multiple heroes in a short-roster game receives\ntheir average score, so extra seats do not multiply it.\n\nEach player's round score is the arithmetic average of their game scores.\nStandings use an exponential moving average: 15% of the new round score plus\n85% of the previous standing. The first scored round sets the initial standing.\nHigher standings rank first. Opponent ratings and win/loss Elo do not affect\neither standings or matchmaking. For example, 3,000 lifetime XP after 10.5\nsimulated minutes gives a score of 900. A previous standing of 800 followed\nby a round average of 1,000 becomes 830.\n\n## BASIC numbers and coordinates\n\nBASIC uses [Bassy](https://github.com/treeform/bassy) with [Fixxy](https://github.com/treeform/fixxy) Q16.16 decimals enabled. Globals and arrays retain fractional values across decisions. `/` performs decimal division; `\\` performs integer division. Decimal operands must fit -32768 through 32767.99998. Integer-only calculations retain the full signed 32-bit range. When converting large world-unit observations, divide them as integers first, for example `(selfAttackRange \\ 100) / (worldScale \\ 100)` in GotA.\n\n`and`, `or`, `xor`, and `not` are bitwise. Comparisons produce -1 for true and 0 for false; conditions accept any nonzero number. Host flags and action results remain 1 or 0, so use `flag = 0` instead of `not flag` to negate a host flag.\n\n`walkTo(x, y)`, `attackMove(x, y)`, and `castPoint(slot, x, y)` accept fractional tile coordinates. For example, `walkTo(selfX + 0.25, selfY - 0.25)` selects a point a quarter tile from the current tile center. Integers continue to name tile centers. IDs, slots, indices, and terrain queries require exact integers. Passing a fractional value to an integer argument raises a BASIC error instead of truncating it. Accepted fractional destinations are preserved in action replays.\n\n### Neutral camps\n\nThe 14 jungle clearings contain seven mirrored pairs. Each half has three\nlow, two medium, and two high camps. The map preset seed fixes each group's\nappearance and size. Members share one chargen appearance, with one larger\nleader whenever a group has at least two mobs. All neutrals use melee attacks.\n\n| Tier | Mobs | Normal HP | Damage | XP | Last-hit gold |\n| --- | --- | --- | --- | --- | --- |\n| Low | 1\u20132 | 100 | 8 | 20 | 10 |\n| Medium | 2\u20133 | 180 | 14 | 35 | 20 |\n| High | 3\u20135 | 300 | 20 | 50 | 30 |\n\nLeaders are 35% larger, have twice the health, deal 50% more damage, and give\ntwice the XP and gold. Camps attack when a hero or lane creep from either team\ncomes within two tiles of a living mob and is in its line of sight. Damaging a\nmob also engages the whole group. Idle auto-attacks ignore resting camps,\nwhile explicit attacks, attack-move, and damaging spells can engage them.\nLane creeps can pull camps by approaching, and fight back once the camp engages.\nPulled neutrals can also be attacked by towers.\n\nThe group returns home if a member or the aggressor goes beyond 12 tiles from\nthe camp center, or the aggressor dies or remains out of sight for three seconds.\nReturning survivors cannot be damaged or controlled. They heal fully when all\nsurvivors reach home. Dead members stay dead until the whole camp is cleared.\nThe full group respawns 60 seconds after the final death, waiting longer if any\nliving hero from either team is within ten tiles, including the boundary.\nCreeps and dead heroes do not block respawns. Respawns receive fresh object IDs.\n\nOnly the last-hitting unit's team receives neutral XP. Eligible living heroes\nwithin six tiles on the same navigation floor split the pool. An eligible hero\nlast hitter receives 15% first; the other 85% is shared among all eligible\nheroes, including that hero. Otherwise the full pool is shared. Only a hero\nlast hitter receives gold. XP contributes to the existing lifetime-XP score.\n\nNeutral objects use `objectKind(i) = 6`, `objectTeam(i) = 2`, and\n`objectClass(i) = 1`, `2`, or `3` for difficulty. Existing health, facing,\nvelocity, target, control observations and attack/cast commands apply. Units\nare LOS-filtered through terrain and brush and grant neither team vision.\n\n| Function | Result |\n| --- | --- |\n| `campCount()` | Number of public, static camp clearings. |\n| `campX(i)`, `campY(i)` | Camp center in the usual team-relative map coordinates. |\n| `campTier(i)` | Difficulty 1\u20133, or 0 for an invalid camp index. |\n| `objectCamp(i)` | Zero-based camp index for a visible neutral, otherwise -1. |\n| `objectLeader(i)` | 1 for a visible camp leader, otherwise 0. |\n| `objectReturning(i)` | 1 for a visible neutral returning home, otherwise 0. |\n\nStatic camp queries never expose hidden living counts or respawn timers.\nThe reference policy farms nearby visible camps between lane fights, beginning\nlow camps at level 1, medium at level 4, and high at level 7. It starts with at\nleast 70% health, withdraws below 40%, and prioritizes enemy heroes and lanes.\n\n`players/puller.bas` copies the reference policy and adds camp pulling. When\nhealthy and a nearby allied wave is available, it walks into neutral aggro,\nthen leads the camp through the wave without using offensive spells or items.\nIt resumes normal play when creeps take over, danger appears, or the attempt\ntimes out. Attempts last at most 15 seconds, with 20 seconds between attempts.\n\nTo test five base bots against five pullers from the repository root:\n\n```sh\nnim r examples/gods_of_the_arena/gota.nim --bot examples/gods_of_the_arena/players/base.bas:5 --bot examples/gods_of_the_arena/players/puller.bas:5\n```\n" + "value": "# Gods of the Arena\n\n**New GotA week: everyone needs to update their bot.** Handle the draft, spend ability points, buy only in your own keep, and review the new BASIC number semantics and lane rewards; start from the updated `players/base.bas`.\n\nTwo teams of five BASIC heroes battle to slay the enemy god. Each hero's ladder\nscore is lifetime XP minus 200 per simulated minute. Fractional minutes count,\nincluding drafting time. Losses and timeouts retain their time-adjusted XP,\nwith each hero's score rounded down to whole points and clamped to zero before\naveraging. Victory is recorded separately in the outcome.\n\nDestroying the enemy god grants every hero on your team a flat 1,000 XP,\nincluding dead heroes and heroes elsewhere on the map, regardless of who lands\nthe last hit. This is awarded once on the final tick and included in lifetime XP\nbefore calculating scores. It offsets 5 minutes of the 200-XP-per-minute time\npenalty. Timeouts grant no god reward.\nDestroying a tower or barracks grants its killer 200 XP and 75 gold.\nTowers fire once per second, and reload continues when they lose or switch\ntargets. Their homing fireballs deal damage on arrival and follow the original\ntarget beyond range and into fog, even if the tower is destroyed. Shots disappear\nif their target dies before impact.\n\nThe gods are the objectives: Hades for Red and Zeus for Blue. Each god has two level-3 guard towers. Clearing all three towers in any one lane exposes the guards. The god cannot take damage from attacks or spells until both of its guards are destroyed. Guards have the same 3900 HP and 60 damage as level-3 lane towers.\n\nSlots 0\u20134 are Red and slots 5\u20139 are Blue. Platform slots are zero-based. Upload a `.bas` file containing BASIC source. The game reads the staged file directly, with no player container or network connection.\n\nStart with the bundled `players/base.bas`. The [game documentation](https://github.com/Metta-AI/polyworld/blob/main/examples/gods_of_the_arena/docs/index.html) describes observations and available BASIC commands. The same source is available under `examples/gods_of_the_arena/bots.nim` and `content.nim`.\n\nThe baseline is a playable reference for all GotA host functions. It drafts\nmissing roles, farms lanes, prioritizes last hits, pushes exposed buildings and\nthe enemy god, upgrades and explicitly casts spells, leads area shots, dodges\nvisible warnings, and uses enemy stats and equipment to judge fights. It also\nshops, stacks and uses recovery items, returns to spawn, uses portal scrolls,\nand buys back when affordable. Actions are conditional on a useful opportunity,\nso one match need not exercise every mechanic. Its observation scans are bounded\nand its main decisions run every six ticks. Abilities and items require\nexplicit policy commands; the engine never casts or uses them automatically.\nSpell ranges and shapes are not queryable, so their small policy table must\nfollow balance changes in `content.nim`; health, damage, costs, and ranks use\nlive observations. This is an editable starting point, not an optimal policy.\n\nEvery hero has a free single-target melee or ranged basic attack in addition\nto four abilities. Basic damage grows each level and includes equipment bonuses.\nIdle heroes automatically acquire the nearest visible, attackable enemy,\nincluding enemy creeps, heroes, and exposed buildings. `attackTarget(objectId)`\ntakes priority; melee and ranged heroes both move into their own attack range\nand repeat basic attacks. `walkTo(x, y)` cancels the attack and suppresses\nautomatic acquisition while walking. Basic attacks do not spend mana or spell\ncharges.\n\n`attackMove(x, y)` uses the same attack-move order as the player controls. It follows a path toward that tile, stops for enemies in the hero's normal acquisition range, and resumes afterward. Like other actions, it returns 1 when accepted and 0 when rejected, and is recorded in replays.\n\nThe bundled `players/rusher.bas` sends all five heroes down mid together. It regroups toward the living team's center when any pair is more than 10 tiles apart, closing to 8 tiles before resuming. It attacks visible, vulnerable enemies within 20 tiles, favoring the enemy closest to the group's center. Otherwise it attack-moves through the middle and toward the opposing god. Dead allies are ignored until they respawn. This policy uses basic attacks only; add explicit casting commands to use abilities.\n\n## Drafting\n\nEvery live match starts with a shared pool of ten heroes. A seeded random\nteam picks first. Teams alternate, and each team's players pick in spawn\norder. Each hero can be selected once across both teams. Combat, waves,\nand the battle clock wait until all ten players have drafted.\n\nOnly the active player's BASIC script runs during drafting, once every\nhalf second. Every player, including humans, has ten simulation seconds to\npick. At the deadline, the game picks a random available hero if the player\nhas not chosen one. The bundled base and rusher scripts try to fill\nfive roles: frontline, carry, mage, support, and fighter. This is a policy\npreference, not a draft rule. Any combination of available heroes is allowed.\nTheir normal movement and combat logic runs after drafting.\n\n| Data or command | Meaning |\n| --- | --- |\n| `drafting` | 1 during drafting, 0 during battle. |\n| `draftTurnId` | ID of the player picking now, or 0 after drafting. |\n| `draftPlayerCount()` | Number of players in the public roster. |\n| `draftPlayerId(i)` | Player ID at zero-based spawn index `i`, or 0 if invalid. |\n| `draftPlayerTeam(i)` | Team at spawn index `i`: Red 0, Blue 1, invalid -1. |\n| `draftedClass(id)` | Hero class picked by player ID, or -1 if unpicked or invalid. Both teams' picks are public. |\n| `heroAvailable(class)` | 1 if the class is valid and unpicked, otherwise 0. |\n| `heroRole(class)` | Frontline 0, carry 1, mage 2, support 3, fighter 4, invalid -1. |\n| `draftHero(class)` | Selects a hero on your turn. Returns 1 on success, 0 on rejection. |\n\n`selfClass` is -1 until your pick. Class constants are `VanguardKnight`,\n`Ranger`, `Arcanist`, `DruidWarden`, `DemonHunter`, `DeathKnight`,\n`Crossbowman`, `Lich`, `Warlock`, and `Berserker`, with IDs 0 through 9.\nDraft queries update immediately. Other sampled self data updates next\nturn. `worldTick` includes draft ticks for action replay timing.\n\n```basic\nif drafting then\n for candidate = 0 to 9\n if heroAvailable(candidate) then\n draftHero(candidate)\n exit for\n end if\n next\nelse\n attackMove(mapWidth / 2, mapHeight / 2)\nend if\n```\n\n`lastActionError()` reports `ActionNotDrafting`, `ActionNotDraftTurn`,\n`ActionUnknownHero`, or `ActionHeroTaken` for rejected picks. Normal game\ncommands are rejected with `ActionDrafting` while players are picking.\nIn player mode, select an available hero in the drafting screen and click\n**Lock in hero** when it is your turn. The current picker appears above\nthe hero grid. Picked heroes turn gray and show their player and team.\nThe countdown shows the active pick's remaining time. Space pauses or\nresumes drafting, including the countdown. Drafting has a separate budget\nof up to 100 simulation seconds for ten players. The configured `maxTicks`\nand CLI duration flags limit battle time only, starting after the last pick.\n\n## Ability progression\n\nHeroes start at level 1 with one ability point and all four abilities locked.\nEach hero level grants another point. Points remain banked until a command\nspends them. `levelAbility(slot)` spends one point to unlock rank 1 or upgrade\nan already learned ability. Slots 0, 1, 2, and 3 correspond to Q, W, E, and R.\n\nQ, W, and E have four ranks requiring hero levels 1, 3, 5, and 7.\nR has three ranks requiring hero levels 6, 12, and 18. Each additional rank\nadds 50% of the rank-1 damage, healing, or mana restoration, rounded down.\nMana costs, range, charge capacity, and timing stay the same. An upgrade\npreserves spent charges and running cooldowns. Pending spells retain the\nrank they had when cast. Respawning preserves learned ranks and banked\npoints and refills only learned abilities.\n\nHero stats, including basic-attack damage, still grow automatically with\nhero level. Ability ranks never increase automatically. The bundled base\nand rusher policies explicitly spend points, prioritizing R, W, E, then Q.\nAbility use requires explicit `castTarget` or `castPoint` commands, including\nslot 0 and ultimates. Items require `useItem` or `useItemAt`. Custom policies\ncan bank points, choose another upgrade order, and reserve any ability. At\nlevel 20, fully ranking all four abilities leaves five banked points.\n\n| BASIC function | Meaning |\n| --- | --- |\n| `levelAbility(slot)` | Unlock or upgrade. Returns 1 on success, 0 on rejection. |\n| `abilityPoints()` | Current unspent points. |\n| `abilityLevel(slot)` | Current rank, with 0 meaning locked. |\n| `abilityMaxLevel(slot)` | Rank limit: 4 for slots 0-2, 3 for slot 3. |\n| `abilityRequiredLevel(slot)` | Hero level required for the next rank, or 0 at maximum rank. |\n| `canLevelAbility(slot)` | 1 when alive with a point and the required level, otherwise 0. |\n| `abilityDamage(slot)` | Damage per target at the learned rank, or 0 when locked. |\n| `abilityHeal(slot)` | Healing per target at the learned rank, or 0 when locked. |\n| `abilityRestore(slot)` | Mana restoration at the learned rank, or 0 when locked. |\n| `abilityManaCost(slot)` | Mana cost of a cast, including while locked. |\n\nThese queries update immediately after commands and return 0 for invalid\nslots. `abilityCharges(slot)`, `abilityCooldown(slot)`, and\n`abilityRecharge(slot)` remain available. Locked abilities have no charges.\nUpgrade actions and rejected attempts are recorded for deterministic replays.\n\n```basic\nif canLevelAbility(3) then\n levelAbility(3)\nelseif canLevelAbility(1) then\n levelAbility(1)\nend if\n```\n\nPlayer controls use Shift+Q/W/E/R, Shift-click on an ability, or its gold\n\"+\" button to spend a point. The HUD shows current ranks, locked abilities,\nand available points.\n\n## BASIC observations\n\nAll bots in one decision phase observe the same starting objects and spell\nwarnings. Earlier bots' actions do not change later bots' observations in\nthat phase. New casts appear in the next observation frame. Own inventory,\nability costs, and command results still update immediately when a command\nis accepted. Object and spell indices are zero-based and may change next\ndecision. Keep `objectId(i)` when tracking an object across decisions or\ncalling `attackTarget`, rather than keeping its list index.\n\nObservations are integers. `worldScale = 60000` is the number of world units per tile, and `tickRate = 24` is the number of simulation ticks per second. `selfX`, `selfY`, `objectX(i)`, `objectY(i)`, `spellX(i)`, and `spellY(i)` use whole global tiles. Facing, speed, range, and velocity retain sub-tile precision in world units. The Y component of these APIs is the second horizontal map axis, not height.\n\nAt an exact tile boundary, Red observers select the higher cell and Blue\nobservers select the lower cell. This makes observed cells rotate exactly\nwhen the arena and teams are swapped. The bundled policies convert global\ncoordinates into their own team's frame before rounding spatial decisions.\n\nObjects appear in groups: gods, buildings, heroes, then creeps. Within each\ngroup, allies precede enemies, followed by position in the observer's team\nframe and stable ID. This order does not depend on simulation storage order.\nSpell warnings are ordered by impact tick, hostile before allied casts,\nthen team-relative position, ability, and stable caster/target identities.\n\nUnits plan movement and attacks from the same starting actor state. Movement\nis published together, then spell impacts and collected damage resolve before\ndeaths and rewards. Opponents can kill each other in the same tick. Collision\ncorrections are also accumulated before moving any participant. Simultaneous\nlast-hit credit uses the match seed, tick, and team-relative actor geometry\nand role, so corresponding fights do not depend on faction-specific IDs.\n\n### Your hero\n\nThe existing `selfId`, `selfTeam`, `selfClass`, `selfX`, `selfY`, `selfHp`, `selfMaxHp`, `selfMana`, `selfMaxMana`, `selfGold`, `selfLevel`, `selfLayer`, and `worldTick` remain available. These additional read-only values describe the current hero:\n\n| Value | Meaning |\n| --- | --- |\n| `selfMoveSpeed` | Unblocked movement speed in world units per tick, including level and equipment bonuses. |\n| `selfAttackRange` | Basic-attack range in world units, measured by planar Euclidean distance between centers. Towers and barracks allow at least 105000 units measured from their occupied footprint; gods use 255000 units. |\n| `selfAttackDamage` | Current basic-attack damage, including level and equipment bonuses. |\n| `selfTarget` | Current ordered or automatically acquired attack target's stable object ID, or zero for none. |\n| `selfAttackCooldown` | Ticks until the next basic hit could land if the target stays in range. Includes remaining recovery and the next windup, or the remainder of a current windup. An idle hero reports a full windup. Excludes chasing and is separate from ability cooldowns. Movement can cancel a swing. |\n| `selfAttacksLanded` | Lifetime count of successful basic hits, preserved across respawns. Spells do not increment it. |\n| `selfPortalCooldown` | Ticks before another Portal Scroll can be used, shared across all inventory stacks and preserved through death. |\n| `selfChannelTicks` | Ticks remaining in the current teleport channel, or zero. |\n| `selfStunTicks`, `selfRootTicks`, `selfSilenceTicks` | Ticks remaining in these control effects, or zero. |\n\n### Shop, potions, and spawn recovery\n\nPurchases only work inside your own keep, including its spawn room. Elsewhere,\n`buyItem(id)` returns 0 with `ActionOutsideKeep`. `canShop()` returns 1 when\npurchases are allowed, and `inOwnSpawn()` returns 1 inside your living hero's\nown spawn room. These queries read live state.\n\nInside that spawn room, health and mana each recover at **20% of maximum per\nsecond**, capped at maximum. The larger keep and the enemy spawn give no such\nrecovery. Normal passive mana regeneration still applies. Damage does not\nturn off spawn recovery, and dead heroes cannot recover until they respawn.\n\n| ID | Item | Gold | Effect |\n| --- | --- | --- | --- |\n| 1 | Health Potion | 30 | 120 health over 10 seconds. |\n| 2 | Vitality Elixir | 75 | 90 health immediately. |\n| 22 | Mana Potion | 45 | 90 mana over 10 seconds. |\n| 3 | Mana Elixir | 90 | 60 mana immediately. |\n\nEach stacks to **8** per slot. `useItem(slot)` spends one dose. Any positive\nincoming damage interrupts both active potion regeneration effects; movement\nand attacks do not. Health items share a **10-second** cooldown, and mana items\nshare a separate **10-second** cooldown. Cooldowns begin on use and survive\ninterruption and death. Full health/mana or a cooldown rejects use without\nspending a dose. `itemCooldown(slot)` returns live remaining ticks (24 per\nsecond), or 0 for empty/invalid slots. Inventory icons show stack counts and\nremaining cooldowns.\n\n### Portal Scrolls\n\nBuy item **21** for **100 gold**. Scrolls stack to eight per slot. Call\n`useItemAt(slot, x, y)` with whole or fractional map coordinates to consume\none scroll and begin a **3-second** channel. The destination is the nearest\nvisible, walkable point inside a living allied tower's sight radius:\n9 tiles for outer towers, 9.5 for inner towers, and 10 for gate/guard towers.\nAttack range, vision, and portal landing range share this same radius. Barracks are not\nanchors. A distant requested point is clamped into this area; there is no\ntravel-distance limit. The selected tower must survive until arrival.\n\nThe hero cannot move, attack, or cast while channeling, and still takes\ndamage. Stuns, roots, death, or loss of the anchor interrupt the channel.\nThe scroll is spent when the channel begins. Completion or interruption\nstarts a **60-second** cooldown shared by every scroll the hero holds.\nDamage alone and silence do not interrupt it. Blazing Blade, Golem Seed,\nand Bone Marionette can interrupt the channel on impact.\n`useItem(slot)` rejects scrolls because they require a destination.\n\nFor human play, click the inventory scroll (or press F/G for the first two\nslots), then right-click the map or minimap. Purple circles show tower range, and a\nchannel bar shows the time remaining. Esc cancels destination selection.\n\n### Visible objects\n\nLoop over indices `0` through `objectCount() - 1`. Object kinds are 1 = god, 2 = hero, 3 = creep, 4 = tower, 5 = barracks, and 6 = neutral mob. Barracks have 950 HP and become exposed after their lane towers fall. Each barracks spawns three melee creeps and one ranged creep per wave, giving six melee creeps and two ranged creeps per lane for each team. Ranged creeps carry a staff and cast magic bolts from up to four tiles away. Destroying a barracks stops its four creeps from spawning. For creeps, `objectClass(i)` is 0 for melee and 1 for ranged. Destroyed buildings leave the object list and release their occupied tiles. New queries respect the same visibility filter:\n\n| Function | Meaning |\n| --- | --- |\n| `objectLevel(i)` | Hero level. |\n| `objectMana(i)` | Hero's current mana. |\n| `objectStunTicks(i)`, `objectSilenceTicks(i)`, `objectRootTicks(i)` | Remaining control ticks for a visible unit, or zero. Uses the same frozen, LOS-filtered object list. |\n| `objectItemId(i, slot)` | Hero's held item ID, using the same IDs as `itemId` and `buyItem`. Zero means no item. Slots are 0 through 5. |\n| `objectItemCount(i, slot)` | Stack count in that hero's inventory slot. |\n| `objectFacingX(i)`, `objectFacingY(i)` | Normalized horizontal facing, scaled by `worldScale`. A unit facing positive X reports `(60000, 0)`. |\n| `objectTarget(i)` | Current attack target's stable object ID, or zero if absent or not visible to your team. |\n| `objectVelX(i)`, `objectVelY(i)` | Actual displacement over the last simulation tick in world units, including collision adjustments. Stationary objects report zero. |\n\nThese new object queries return zero for invalid indices or fields that do not apply to that object. Invalid inventory slots also return zero. Hero level, mana, and inventory queries return zero for non-heroes. An object's ID is not a valid substitute for its list index.\n\nEach slain enemy creep provides a shared pool of 15 XP to living heroes within six tiles on the same navigation floor, regardless of starting lane. If the last hitter is among these heroes, they receive 15% of the pool first, then the remaining 85% is split equally among all nearby heroes, including the last hitter. With three heroes, this gives 6.5 XP to the last hitter and 4.25 XP to each teammate. Fractional XP carries forward between kills. If no eligible hero lands the last hit, the full pool is shared equally. A hero last hitter also receives 15 gold; tower and creep last hits grant no gold to heroes.\n\n### Pending spells and warnings\n\nLoop over `0` through `spellCount() - 1`. This list contains unresolved casts from their start through impact, including projectiles and area warnings. Allied casts are observable; enemy casts require their aim position to be visible, matching the viewer's warning visibility. Completed effects are omitted.\n\n| Function | Meaning |\n| --- | --- |\n| `spellAbility(i)` | Ability enum ID from `content.nim`, beginning at zero. Invalid indices return -1. |\n| `spellCasterId(i)` | Caster's stable object ID, or zero if the enemy caster is hidden. |\n| `spellX(i)`, `spellY(i)` | Aim/impact position or area center in whole global tiles. This is not the projectile's interpolated flight position. |\n| `spellImpactTick(i)` | Absolute simulation tick at impact. Subtract `worldTick` to obtain the remaining ticks. |\n\nOther invalid spell queries return zero. Visibility of an enemy warning does not reveal its hidden caster's identity.\n\n## Death and buyback\n\nThe first death takes 9 seconds to respawn, including the 1-second death\nanimation. Each subsequent death adds 5 seconds, up to a total of 60 seconds.\nDeath counts belong to each hero and persist after respawning.\n\n`selfDeaths` is the hero's death count. `selfRespawnTicks` is the remaining\nrespawn delay in ticks, or zero while alive. BASIC decisions continue while\ndead so a policy can request buyback.\n\n`buybackPrice()` returns the dead hero's price: 100 gold times their death\ncount. It returns zero while alive or after the match ends. The price stays\nfixed during a death, even as the respawn timer counts down.\n\n`buyback()` returns 1 when accepted and 0 when rejected. It spends the\nhero's gold and immediately respawns them with full health, mana, and spell\ncharges. It preserves inventory, level, XP, and death count. A living hero,\nan ended match, or insufficient gold causes rejection without spending gold.\nThe HUD shows the countdown, buyback price, and any rejection reason while\ndead. Buyback attempts are recorded for replay and seeking.\n\nThe bundled `players/base.bas` buys back as soon as it can afford the price.\nWhile dead it skips normal commands, resuming them on the decision after buyback.\n\n```basic\nif selfRespawnTicks > 0 then\n price = buybackPrice()\n if price > 0 and selfGold >= price then\n accepted = buyback()\n end if\nend if\n```\n\n## Crowd control\n\n| Hero | Ability | Effect | Rank-one damage |\n| --- | --- | --- | --- |\n| Vanguard | R: Blazing Blade | Stun for 1 second | 72 |\n| Warlock | E: Dread Totem | Silence for 2 seconds | 70 |\n| Druid | R: Golem Seed | Root for 2 seconds | 68 |\n| Lich | E: Bone Marionette | Root for 1 second | 53 |\n\nThese abilities trade about 20% of their damage for control. Durations stay\nfixed at every rank. Effects apply on impact to enemy heroes and creeps;\nbuildings and gods are immune. A stun stops movement, basic attacks,\nabilities, and items, and cancels a pending basic swing. Silence prevents\nall four abilities but allows movement, basic attacks, and items. Root\nprevents movement and teleporting, while allowing in-range attacks,\nabilities, and other items. Stun and root interrupt teleport channels.\nAlready released spells still resolve.\n\nDifferent effects coexist. Reapplying one keeps the later expiration,\nwithout adding durations. All effects clear on death and respawn. Visible\naffected units show icons with countdown rings above their health bars.\n\n## Action feedback\n\n`lastActionError()` returns the reason for your hero's latest submitted\ncommand. A successful action clears it to `NoActionError` (0). A failed\naction returns 0 as before, and sets the first failing validation reason.\nRead-only queries and automatic basic attacks do not change it.\nUnlike the sampled self data, this query updates immediately after commands.\n\n```basic\naccepted = castTarget(1, targetId)\nif accepted = 0 and lastActionError() = ActionInsufficientMana then\n print \"Need more mana\"\nend if\n```\n\nRead-only reason constants are `NoActionError`, `ActionNotAlive`,\n`ActionInvalidSlot`, `ActionUnknownItem`, `ActionInsufficientGold`,\n`ActionAlreadyEquipped`, `ActionStackFull`, `ActionInventoryFull`,\n`ActionEmptySlot`, `ActionNotConsumable`, `ActionFullHealth`, `ActionFullMana`,\n`ActionTargetUnavailable`, `ActionOutOfRange`, `ActionNoRoute`,\n`ActionInvalidPoint`, `ActionCooldown`, `ActionNoCharges`,\n`ActionInsufficientMana`, `ActionSpellLimit`, `ActionChanneling`,\n`ActionStunned`, `ActionRooted`, `ActionOutsideKeep`, `ActionAbilityLocked`,\n`ActionNoAbilityPoints`, `ActionAbilityMaxLevel`, `ActionHeroLevelRequired`,\n`ActionNotDead`, `ActionMatchEnded`, `ActionDrafting`, `ActionNotDrafting`,\n`ActionNotDraftTurn`, `ActionUnknownHero`, `ActionHeroTaken`, and `ActionSilenced`\n(values 0 through 35).\nUnavailable targets share a generic error without exposing hidden state.\nThis feedback is recorded deterministically through submitted replay actions.\n\nFor post-match analysis, the [replay extractor](../../docs/stats.md#gota-replay-events)\nresimulates an exact-version replay and exposes typed damage, healing,\ndeath, reward, and rejection events for all players. Its omniscient buffer\nis not available to live BASIC policies.\n\n## Terrain and execution\n\nBASIC can inspect the complete static terrain with `terrainKind(x, y)`, `terrainWalkable(x, y)`, `terrainHeight(x, y)`, and `terrainWaterDepth(x, y)`. These use global tile coordinates on `selfLayer`. Each has an explicit `At(x, y, layer)` version, such as `terrainKindAt(x, y, GroundLayer)`. Read-only constants expose `mapWidth`, `mapHeight`, `mapLayers`, the layer names, and terrain kinds. Height and water depth use eighths of a tile; invalid or absent tiles return zero. The Terrain API section of the game documentation lists all constants and edge cases. Static terrain is available through fog, while enemy objects remain visibility-filtered. Walkability also includes team-known building footprints; unseen enemy destruction does not reveal newly open tiles.\n\nBASIC `PRINT` output, compiler diagnostics, runtime errors, and VM lifecycle messages go to the owning player's private log. Each log is limited to 10 MiB. Runtime limit errors disable that VM; other seats continue. Invalid BASIC syntax fails the episode with a player failure diagnostic. Public game logs and action replays contain no BASIC source or private print output.\n\nBattles run up to 28,800 deterministic ticks (20 simulated minutes), plus drafting time, without real-time pacing. Replays run entirely in the browser with playback, seeking, speed, and loop controls. The server exposes `/healthz`; legacy clients are static stubs.\n\nThe Competition league schedules 24 games per round with random matchups,\non a 32-minute interval. Each match uses ten distinct policies when at least\nten are eligible: five different policies on Red and five on Blue, with\none hero per policy. The scheduler uses `team_n`, `team_count: 2`,\n`team_layout: \"blocks\"`, `matchmaking: \"random\"`, and\n`distinct_teammates: true`. Preserve these settings when updating the league.\nSeparate baseline filler policies complete short rosters and are not ranked\nentrants. A policy controlling multiple heroes in a short-roster game receives\ntheir average score, so extra seats do not multiply it.\n\nEach player's round score is the arithmetic average of their game scores.\nStandings use an exponential moving average: 15% of the new round score plus\n85% of the previous standing. The first scored round sets the initial standing.\nHigher standings rank first. Opponent ratings and win/loss Elo do not affect\neither standings or matchmaking. For example, 3,000 lifetime XP after 10.5\nsimulated minutes gives a score of 900. A previous standing of 800 followed\nby a round average of 1,000 becomes 830.\n\n## BASIC numbers and coordinates\n\nBASIC uses [Bassy](https://github.com/treeform/bassy) with [Fixxy](https://github.com/treeform/fixxy) Q16.16 decimals enabled. Globals and arrays retain fractional values across decisions. `/` performs decimal division; `\\` performs integer division. Decimal operands must fit -32768 through 32767.99998. Integer-only calculations retain the full signed 32-bit range. When converting large world-unit observations, divide them as integers first, for example `(selfAttackRange \\ 100) / (worldScale \\ 100)` in GotA.\n\n`and`, `or`, `xor`, and `not` are bitwise. Comparisons produce -1 for true and 0 for false; conditions accept any nonzero number. Host flags and action results remain 1 or 0, so use `flag = 0` instead of `not flag` to negate a host flag.\n\n`walkTo(x, y)`, `attackMove(x, y)`, and `castPoint(slot, x, y)` accept fractional tile coordinates. For example, `walkTo(selfX + 0.25, selfY - 0.25)` selects a point a quarter tile from the current tile center. Integers continue to name tile centers. IDs, slots, indices, and terrain queries require exact integers. Passing a fractional value to an integer argument raises a BASIC error instead of truncating it. Accepted fractional destinations are preserved in action replays.\n\n### Neutral camps\n\nThe 14 jungle clearings contain seven mirrored pairs. Each half has three\nlow, two medium, and two high camps. The map preset seed fixes each group's\nappearance and size. Members share one chargen appearance, with one larger\nleader whenever a group has at least two mobs. All neutrals use melee attacks.\n\n| Tier | Mobs | Normal HP | Damage | XP | Last-hit gold |\n| --- | --- | --- | --- | --- | --- |\n| Low | 1\u20132 | 100 | 8 | 20 | 10 |\n| Medium | 2\u20133 | 180 | 14 | 35 | 20 |\n| High | 3\u20135 | 300 | 20 | 50 | 30 |\n\nLeaders are 35% larger, have twice the health, deal 50% more damage, and give\ntwice the XP and gold. Camps attack when a hero or lane creep from either team\ncomes within two tiles of a living mob and is in its line of sight. Damaging a\nmob also engages the whole group. Idle auto-attacks ignore resting camps,\nwhile explicit attacks, attack-move, and damaging spells can engage them.\nLane creeps can pull camps by approaching, and fight back once the camp engages.\nPulled neutrals can also be attacked by towers.\n\nThe group returns home if a member or the aggressor goes beyond 12 tiles from\nthe camp center, or the aggressor dies or remains out of sight for three seconds.\nReturning survivors cannot be damaged or controlled. They heal fully when all\nsurvivors reach home. Dead members stay dead until the whole camp is cleared.\nThe full group respawns 60 seconds after the final death, waiting longer if any\nliving hero from either team is within ten tiles, including the boundary.\nCreeps and dead heroes do not block respawns. Respawns receive fresh object IDs.\n\nOnly the last-hitting unit's team receives neutral XP. Eligible living heroes\nwithin six tiles on the same navigation floor split the pool. An eligible hero\nlast hitter receives 15% first; the other 85% is shared among all eligible\nheroes, including that hero. Otherwise the full pool is shared. Only a hero\nlast hitter receives gold. XP contributes to the existing lifetime-XP score.\n\nNeutral objects use `objectKind(i) = 6`, `objectTeam(i) = 2`, and\n`objectClass(i) = 1`, `2`, or `3` for difficulty. Existing health, facing,\nvelocity, target, control observations and attack/cast commands apply. Units\nare LOS-filtered through terrain and brush and grant neither team vision.\n\n| Function | Result |\n| --- | --- |\n| `campCount()` | Number of public, static camp clearings. |\n| `campX(i)`, `campY(i)` | Camp center in the usual team-relative map coordinates. |\n| `campTier(i)` | Difficulty 1\u20133, or 0 for an invalid camp index. |\n| `objectCamp(i)` | Zero-based camp index for a visible neutral, otherwise -1. |\n| `objectLeader(i)` | 1 for a visible camp leader, otherwise 0. |\n| `objectReturning(i)` | 1 for a visible neutral returning home, otherwise 0. |\n\nStatic camp queries never expose hidden living counts or respawn timers.\nThe reference policy farms nearby visible camps between lane fights, beginning\nlow camps at level 1, medium at level 4, and high at level 7. It starts with at\nleast 70% health, withdraws below 40%, and prioritizes enemy heroes and lanes.\n\n`players/puller.bas` copies the reference policy and adds camp pulling. When\nhealthy and a nearby allied wave is available, it walks into neutral aggro,\nthen leads the camp through the wave without using offensive spells or items.\nIt resumes normal play when creeps take over, danger appears, or the attempt\ntimes out. Attempts last at most 15 seconds, with 20 seconds between attempts.\n\nTo test five base bots against five pullers from the repository root:\n\n```sh\nnim r examples/gods_of_the_arena/gota.nim --bot examples/gods_of_the_arena/players/base.bas:5 --bot examples/gods_of_the_arena/players/puller.bas:5\n```\n" } } }, diff --git a/coworld/gota/guide.md b/coworld/gota/guide.md index e66999a6..5608887f 100644 --- a/coworld/gota/guide.md +++ b/coworld/gota/guide.md @@ -14,6 +14,10 @@ the last hit. This is awarded once on the final tick and included in lifetime XP before calculating scores. It offsets 5 minutes of the 200-XP-per-minute time penalty. Timeouts grant no god reward. Destroying a tower or barracks grants its killer 200 XP and 75 gold. +Towers fire once per second, and reload continues when they lose or switch +targets. Their homing fireballs deal damage on arrival and follow the original +target beyond range and into fog, even if the tower is destroyed. Shots disappear +if their target dies before impact. The gods are the objectives: Hades for Red and Zeus for Blue. Each god has two level-3 guard towers. Clearing all three towers in any one lane exposes the guards. The god cannot take damage from attacks or spells until both of its guards are destroyed. Guards have the same 3900 HP and 60 damage as level-3 lane towers. @@ -237,7 +241,8 @@ Buy item **21** for **100 gold**. Scrolls stack to eight per slot. Call `useItemAt(slot, x, y)` with whole or fractional map coordinates to consume one scroll and begin a **3-second** channel. The destination is the nearest visible, walkable point inside a living allied tower's sight radius: -7 tiles for outer/inner towers, 8 for gate/guard towers. Barracks are not +9 tiles for outer towers, 9.5 for inner towers, and 10 for gate/guard towers. +Attack range, vision, and portal landing range share this same radius. Barracks are not anchors. A distant requested point is clamped into this area; there is no travel-distance limit. The selected tower must survive until arrival. diff --git a/coworld/gota/runtime/neural_package.py b/coworld/gota/runtime/neural_package.py new file mode 100644 index 00000000..7b5a6c40 --- /dev/null +++ b/coworld/gota/runtime/neural_package.py @@ -0,0 +1,156 @@ +"""Staging validator and builder for GotA neural packages (gota-neural-basic/1). + +GotA's contract on top of the shared polyworld validator (coworld/runtime/neural_package.py), +which mirrors src/polyworld/neural_package.nim; the GotA extension keys (goal, decoder.defer_script, +decoder.mask_empty_targets, decoder.mask_mode) mirror parseGotaOptions in +examples/gods_of_the_arena/neural_contract.nim. Plain .bas submissions are not packages and are +untouched. + + python3 neural_package.py validate PACKAGE.zip [--obs-hash H --action-hash H] + python3 neural_package.py build OUT.zip --policy policy.bas --model model.bin + [--period 4] [--decoder argmax|sample --temperature T] [--goal-red 16 floats] [--goal-blue ...] +""" +import argparse, dataclasses, importlib.util, os, sys + +_SHARED = os.path.join(os.path.dirname(os.path.abspath(__file__)), "../../runtime/neural_package.py") +_spec = importlib.util.spec_from_file_location("polyworld_neural_package", _SHARED) +tier = importlib.util.module_from_spec(_spec) +_spec.loader.exec_module(tier) + +PackageError = tier.PackageError +is_package = tier.is_package +SCHEMA = "gota-neural-basic/1" +FILES = tier.FILES +MAX_PACKAGE = tier.MAX_PACKAGE +MAX_POLICY = tier.MAX_POLICY +HEADS = [8, 25, 49, 4, 6] +OBS_SIZE = 1407 +MASK_SIZE = 187 +GOAL_SIZE = 16 +WIDTHS = tier.WIDTHS +MAGIC = tier.MAGIC +OP_BUDGET = tier.DEFAULT_OP_BUDGET +MAX_PARAMS = tier.MAX_PARAMS +# Contract v1 hashes (examples/gods_of_the_arena/neural_contract.nim; gota_*_contract_hash). +OBS_HASH = "ae4046e83cc02e861f9c8cc32550c6cc4d6f9c161c9225a9b34a314d310ea991" +ACTION_HASH = "ecc7d53c11a9db0912467c66ecb3e65b60b3e71ef14dad1442ba3b4f6ac14697" + + +def _goal(node, where): + if not isinstance(node, list) or len(node) != GOAL_SIZE or not all(tier.is_number(v) for v in node): + raise PackageError(f"{where} must be 16 numbers") + if any(v < -1 or v > 1 for v in node): + raise PackageError(f"{where} values must be in [-1, 1]") + if node[15] != 0: + raise PackageError(f"{where} w_reserved must be 0") + + +def parse_options(m): + """GotA manifest keys: goal {red, blue}; decoder.defer_script / mask_empty_targets / mask_mode.""" + options = {"defer_script": False, "mask_mode": None} + if "goal" in m: + tier.require_keys(m["goal"], {"red", "blue"}, "goal") + if set(m["goal"]) != {"red", "blue"}: + raise PackageError("goal needs red and blue") + _goal(m["goal"]["red"], "goal.red") + _goal(m["goal"]["blue"], "goal.blue") + if "decoder" in m: + d = m["decoder"] + masked = tier.require_bool(d, "mask_empty_targets", "decoder") + if masked: + options["mask_mode"] = "conditional" + if "mask_mode" in d: + if not masked: + raise PackageError("decoder.mask_mode needs mask_empty_targets true") + if d["mask_mode"] not in ("conditional", "static"): + raise PackageError("decoder.mask_mode must be conditional or static") + options["mask_mode"] = d["mask_mode"] + options["defer_script"] = tier.require_bool(d, "defer_script", "decoder") + return options + + +GOTA = tier.Contract(schema=SCHEMA, obs_hash=OBS_HASH, action_hash=ACTION_HASH, obs_size=OBS_SIZE, + heads=HEADS, max_period=24, op_budget=OP_BUDGET, manifest_keys=("goal",), + decoder_keys=("defer_script", "mask_empty_targets", "mask_mode"), + parse_options=parse_options) + + +def contract(obs_hash=OBS_HASH, action_hash=ACTION_HASH): + if obs_hash == OBS_HASH and action_hash == ACTION_HASH: + return GOTA + return dataclasses.replace(GOTA, obs_hash=obs_hash, action_hash=action_hash) + + +def check_model(data: bytes, obs_hash=OBS_HASH, action_hash=ACTION_HASH): + """Validates a GOTANET1 model.bin; returns (hidden, operations).""" + return tier.check_model(data, contract(obs_hash, action_hash)) + + +def validate(data: bytes, obs_hash=OBS_HASH, action_hash=ACTION_HASH) -> dict: + """Raises PackageError with the reason; returns the manifest on success.""" + manifest, _ = tier.validate(data, contract(obs_hash, action_hash)) + return manifest + + +def encode_model(weights, hidden, obs_hash=OBS_HASH, action_hash=ACTION_HASH, inputs=OBS_SIZE, heads=HEADS): + """GOTANET1 bytes from a flat float32 list: W_enc[H][I], W_rec[3H][H], W_dec[O][H].""" + return tier.encode_model(weights, hidden, contract(obs_hash, action_hash), inputs, heads) + + +def build(policy: bytes, model: bytes, period=4, decoder=None, goal=None) -> bytes: + return tier.build(policy, model, GOTA, period, decoder, {"goal": goal} if goal else None) + + +def main(): + ap = argparse.ArgumentParser() + sub = ap.add_subparsers(dest="cmd", required=True) + v = sub.add_parser("validate") + v.add_argument("package") + v.add_argument("--obs-hash", default=OBS_HASH) + v.add_argument("--action-hash", default=ACTION_HASH) + b = sub.add_parser("build") + b.add_argument("out") + b.add_argument("--policy", required=True) + b.add_argument("--model", required=True) + b.add_argument("--period", type=int, default=4) + b.add_argument("--decoder", choices=["argmax", "sample"]) + b.add_argument("--temperature", type=float) + b.add_argument("--defer-script", action="store_true", + help="decoder.defer_script: verb 0 defers to policy.bas (e.g. base.bas verbatim)") + b.add_argument("--mask-empty-targets", action="store_true", + help="decoder.mask_empty_targets: host applies the gota_action_mask before decode") + b.add_argument("--mask-mode", choices=["conditional", "static"]) + b.add_argument("--goal-red", type=float, nargs=16) + b.add_argument("--goal-blue", type=float, nargs=16) + a = ap.parse_args() + if a.cmd == "validate": + data = open(a.package, "rb").read() + if not is_package(data): + print("plain BASIC submission (not a package): unchanged") + return + try: + m = validate(data, a.obs_hash, a.action_hash) + except PackageError as e: + print(f"REJECTED: {e}") + sys.exit(1) + print(f"OK {SCHEMA} hidden={m['model']['hidden']} period={m['decision_period']}") + else: + decoder = None + if a.decoder: + decoder = {"mode": a.decoder} + if a.temperature is not None: + decoder["temperature"] = a.temperature + if a.defer_script: + decoder = dict(decoder or {}, defer_script=True) + if a.mask_empty_targets: + decoder = dict(decoder or {}, mask_empty_targets=True) + if a.mask_mode: + decoder["mask_mode"] = a.mask_mode + goal = {"red": a.goal_red, "blue": a.goal_blue} if a.goal_red and a.goal_blue else None + data = build(open(a.policy, "rb").read(), open(a.model, "rb").read(), a.period, decoder, goal) + open(a.out, "wb").write(data) + print(f"wrote {a.out} ({len(data)} bytes)") + + +if __name__ == "__main__": + main() diff --git a/coworld/releases/2026-09-24-gota-towers.json b/coworld/releases/2026-09-24-gota-towers.json new file mode 100644 index 00000000..ca722956 --- /dev/null +++ b/coworld/releases/2026-09-24-gota-towers.json @@ -0,0 +1,74 @@ +{ + "date": "2026-09-24", + "version": "2026.9.24.2", + "url": "https://softmax.com/gods-of-the-arena", + "coworld_id": "cow_e282a46f-31c4-43b1-a9e2-aaa31d3aaed4", + "canonical": true, + "source_commit": "f4456bea13d967cb3f188df357656e1410cb0a40", + "asset_commit": "32946b35b92e2b33f38945e96b293eb6e2c98ac9", + "replay_version": 64, + "changes": [ + "Increase tower attack, vision, and portal range together to 9, 9.5, and 10 tiles.", + "Keep tower reload progress when targets leave range or change.", + "Make tower fireballs track their original target beyond range and into fog, dealing damage on arrival.", + "Restore in-flight projectiles during replay seeking and support only gameplay replay version 64.", + "Include the mailbox changes already merged on main." + ], + "checks": [ + "Nim checks and the complete test suite passed after rebasing onto current main.", + "Tower range, reload, homing, target death, and mid-flight snapshot regressions passed.", + "Native and WASM replay verification matched, including four backward seeks and a mid-flight snapshot.", + "All ten local certification checks and hosted certification passed.", + "All five hosted smoke games completed with verified integer scores.", + "Served JavaScript, WASM, and data SHA-256 hashes match the production build.", + "The Linux certification replay and a replay from the forced league round reproduced exactly on the native client." + ], + "browser_verification": { + "episode_id": "08e742c8-c109-448d-b580-0ca8edfb3e19", + "replay_version": 64, + "end_tick": 30049, + "visible_tick_label": "Tick 30049 / 30049", + "browser_errors": [], + "method": "Deployed viewer rendered combat, sought through the final tick, and displayed the final scoreboard.", + "verified_at": "2026-09-24T22:58:15.877536+00:00" + }, + "league_settings_preserved": true, + "known_issues": [], + "native_league_replay": { + "round_number": 742, + "id": "ereq_64ec48a7-6c9b-4816-a536-f3ab3fea3aff", + "episode_id": "08e742c8-c109-448d-b580-0ca8edfb3e19", + "replay_version": 64, + "sha256": "70e00c59901c88151b72806e2560ab29b3c3553c16c7ac981e02769a995ae906", + "ticks": 30049, + "outcome": "time_limit" + }, + "verification_round": { + "round_id": "round_d745770e-56e3-4c76-bcfc-6c70df21ffc8", + "round_number": 742, + "status": "completed", + "games": 24, + "completed_games": 24, + "failed_games": 0, + "failures": [], + "coworld_id": "cow_e282a46f-31c4-43b1-a9e2-aaa31d3aaed4", + "version": "2026.9.24.2", + "integer_scores_verified": 240, + "outcomes": { + "time_limit": 22, + "BlueTeam": 1, + "RedTeam": 1 + }, + "replay_version": 64, + "watch_url": "https://softmax.com/gods-of-the-arena?e=90f3176a-9090-40ff-b1c2-9430401fb6a6", + "verified_at": "2026-09-24T23:00:37.226208+00:00" + }, + "manifest_hash": "sha256:d9be78bff6cc9c74d3de3f9e55ccb5bd20b3f181523b6ec88d90a38d68435a78", + "viewer_sha256": { + "wasm": "de36c8625be15bd101dc82ea4e1f4dedef0b65981dcd3e8420ba3cb35a362000", + "data": "6a1d07f38179cd0ce3498dbaefb725c634d74a7705426ded9a119a83dd98e3f0", + "js": "12e3b4c6d0d346db5689253a768a9c6ba4f8c68d7743311272439875577c97ca" + }, + "ci_url": "https://github.com/Metta-AI/polyworld/actions/runs/36069192354", + "verified_at": "2026-09-24T23:00:55.692025+00:00" +} diff --git a/coworld/tools/test_runtime.nim b/coworld/tools/test_runtime.nim index a8b2ef45..24f1acc2 100644 --- a/coworld/tools/test_runtime.nim +++ b/coworld/tools/test_runtime.nim @@ -3,7 +3,10 @@ import std/[json, monotimes, net, os, osproc, strtabs, strutils, tempfiles, times, uri], - crunchy, jsony + crunchy, jsony, + polyworld/neural_host, + ../../examples/gods_of_the_arena/neural_contract, + ../../tests/neural_toy as zip const Root = currentSourcePath().parentDir.parentDir.parentDir @@ -11,6 +14,40 @@ const SocketTimeout = 5000 Games = [("gota", 10), ("lvd", 2), ("cta", 4)] +proc gotaModelBytes(): string = + ## A minimal valid w64 GOTANET1 model matching the live GotA contract. + var weights = newSeq[float32]( + zip.weightCount(64, ObservationSize, ActionOutputs)) + encodeActor(ObservationSize, 64, HeadSizes, ObservationContractHash, + ActionContractHash, weights) + +proc gotaBrokenPolicyPackage(): string = + ## A neural package with a valid manifest and model, but a policy.bas + ## that fails to compile: exercises `installPackageSeat`'s own compile + ## error path (a valid package whose embedded script is broken), as + ## opposed to a corrupt/unparsable package. + let + model = gotaModelBytes() + policy = "THIS IS NOT BASIC\n" + manifest = %*{ + "schema": PackageSchema, + "observation_contract": ObservationContractHash, + "action_contract": ActionContractHash, + "decision_period": 4, + "files": { + "policy.bas": sha256Hex(policy), "model.bin": sha256Hex(model) + }, + "model": { + "format": "GOTANET1", "inputs": ObservationSize, "hidden": 64, + "heads": HeadSizes + } + } + zip.writeZip([ + zip.entry("manifest.json", $manifest), + zip.entry("policy.bas", policy), + zip.entry("model.bin", model) + ]) + proc fileUri(path: string): string = ## Encodes one absolute path for the runner's local file handoff. "file://" & encodeUrl(path, usePlus = false).replace("%2F", "/") @@ -101,7 +138,10 @@ proc episode( count: int, scripts: seq[string], failure = false, - ticks = 240 + ticks = 240, + expectedOutput = "", + failureLog = "BASIC error:", + failureMessage = "" ) = ## Runs one local roster and inspects outputs at the completion marker. doAssert scripts.len == count @@ -178,7 +218,8 @@ proc episode( marker = directory / (if failure: "failure.json" else: "results.json") deadline = getMonoTime() + initDuration(seconds = 120) while not fileExists(marker): - doAssert process.running(), readFile(logPath) + doAssert process.running(), "exit " & $process.peekExitCode() & + ": " & readFile(logPath) doAssert getMonoTime() < deadline, "episode timed out" sleep(20) let output = readFile(marker).fromJson(JsonNode) @@ -186,6 +227,9 @@ proc episode( var logs: seq[string] for slot in 0 ..< count: logs.add readFile(directory / ("player-" & $slot & ".log")) + if expectedOutput.len > 0: + for private in logs: + doAssert private.contains(expectedOutput), private for slot, private in logs: doAssert private.len <= LogLimit let marker = "PRIVATE-" & $slot @@ -201,10 +245,11 @@ proc episode( if failure: doAssert not fileExists(directory / "results.json") doAssert output["failed_policy_index"].getInt() == 0 - doAssert output["message"].getStr() == - "BASIC compilation failed for player slot 0", - "failPlayer must report the failing slot and reason: " & $output - doAssert logs[0].contains("BASIC error:") + doAssert logs[0].contains(failureLog), logs[0] + if failureMessage.len > 0: + doAssert output["message"].getStr().startsWith(failureMessage), + "failPlayer must report the failing slot and reason: " & + output["message"].getStr() let status = readFile(directory / "status.json").fromJson(JsonNode) doAssert status["players"][0]["state"].getStr() == "exited" doAssert status["players"][0]["exit_code"].getInt() == 1 @@ -264,10 +309,34 @@ for (game, count) in Games: scripts[slot] = "END\n" scripts[0] = "THIS IS NOT BASIC\n" episode(game, count, scripts, failure = true) + if game == "gota": + # A staged file that starts like a ZIP is a neural package; a broken + # one ends the episode with the package reason, not a compile error. + scripts[0] = "PK\x03\x04 not a neural package\n" + episode(game, count, scripts, failure = true, + failureLog = "neural package rejected:", + failureMessage = "neural package rejected for player slot 0") + # A valid package whose embedded policy.bas fails to compile is its + # own failure, reported through the same failPlayer path (not the + # generic "BASIC compilation failed" message). + scripts[0] = gotaBrokenPolicyPackage() + episode(game, count, scripts, failure = true, + failureLog = "neural package rejected for player slot 0", + failureMessage = "neural package rejected for player slot 0: " & + "policy.bas failed to compile:") scripts[0] = "WHILE 1\nWEND\n" episode(game, count, scripts) echo game, ": runtime contracts passed" + for slot in 0 ..< scripts.len: + scripts[slot] = """ +sendChat(-2, "CHAT") +print pullMailbox$(), mailboxId() +""" + episode(game, count, scripts, ticks = 3, + expectedOutput = "CHAT") + echo game, ": hosted mailbox integration passed" + episode( "lvd", 2, diff --git a/docs/mailboxes.md b/docs/mailboxes.md new file mode 100644 index 00000000..894066fd --- /dev/null +++ b/docs/mailboxes.md @@ -0,0 +1,58 @@ +# Player mailboxes + +Each player has one inbox with 100 message slots. A message is an integer ID +and a string, limited to 1024 bytes. Sending to a full inbox does nothing; +unread messages remain until the player pulls them. + +Each game defines `sendChat` and its BASIC callbacks in its own `bots.nim`. +The examples use these routing rules: + +| Game | Global (`-2`) | Team (`-1`) | DM (player ID) | +| --- | --- | --- | --- | +| Gods of the Arena | Everyone. | Heroes on the sender's red/blue team. | One player. | +| Light vs Dark | Everyone. | Unsupported. | One player. | +| Call to Adventure | Players within 16 tiles on the same level. | Unsupported. | Unsupported. | + +Broadcasts include the sender. CTA uses its existing Chebyshev tile distance: +the difference along each tile axis must be at most 16. Range and level are +checked when sending; moving afterward does not remove queued messages. +Each game owns this routing loop. There is no shared routing policy or +routing callback framework. + +`sendChat` returns the number of inboxes that accepted a copy. Empty or +oversized messages and unsupported or invalid destinations return zero. +A broadcast can reach some players even when another player's inbox is full. + +`pullMailbox$()` consumes the oldest message, or returns an empty string if +there is none. `mailboxId()` identifies the last message pulled: `-2` for +global, `-1` for team, or the sender's player ID for a DM. An empty pull sets +that ID to `-3`. There is no separate sender, channel name, or timestamp. + +`mailboxCount()` counts unread messages, `mailboxSelf()` returns the caller's +roster ID, and `mailboxPlayers()` gives the roster size. GotA uses IDs 0 to 9, +CTA uses 0 to 3, and Light vs Dark uses 0 to 1. + +```basic +sendChat(-2, "Hello everyone.") +' GotA only: +sendChat(-1, "Meet at the checkpoint.") +' GotA and Light vs Dark: +sendChat(0, "A direct message to player zero.") +message$ = pullMailbox$() +while message$ <> "" + print mailboxId(), message$ + message$ = pullMailbox$() +wend +``` + +Delivery follows script execution order. A later player can read a message +in the same tick. Players can send and pull multiple times per decision, +within their normal BASIC instruction and string limits. Inboxes persist +across decisions and are replaced when policies are loaded for a new run. +Chat text is not recorded in replays and has no graphical chat panel. + +`mailboxes.nim` is only a bounded inbox: an array of IDs, an array of reserved +strings, and read/count fields. It has no BASIC dependency. Each game owns +one inbox per player and copies message text through Bassy's public API. +Bassy owns its string storage and reclaims temporary strings on `restart()`. +Polyworld does not inspect or manage BASIC's private storage. diff --git a/examples/call_to_adventure/bots.nim b/examples/call_to_adventure/bots.nim index 8bf4b9de..78a75139 100644 --- a/examples/call_to_adventure/bots.nim +++ b/examples/call_to_adventure/bots.nim @@ -7,7 +7,8 @@ import polyworld/neural, bassy, - polyworld/[bodies, metrics, cli, controllers, pathing, profiles], + polyworld/[mailboxes, bodies, metrics, cli, controllers, + pathing, profiles], content, sim, replays @@ -94,13 +95,14 @@ proc issueHeroAction(action: ReplayAction): int32 = proc heroLimits(): Limits = ## Defines one isolated hero VM's source, memory, and decision budgets. result = defaultLimits() + result.maxStringBytes = 256 * 1024 result.maxSourceBytes = 128 * 1024 result.maxCodeInstructions = 50_000 result.maxArrays = 32 result.maxArrayElements = 16_384 result.maxGlobals = 512 result.maxHostData = 32 - result.maxHostFunctions = 32 + result.maxHostFunctions = 64 result.maxRoutines = 64 result.maxParameters = 16 result.maxRegisters = 256 @@ -112,10 +114,56 @@ proc heroLimits(): Limits = result.maxPrintBytes = 4 * 1024 result.maxPrintEvents = 256 +proc sendChat*( + game: Game, sender, target: int, text: openArray[char] +): int32 = + ## Broadcasts to players within 16 tiles on the sender's level. + if sender notin 0 ..< game.inboxes.len or target != -2: + return 0 + let origin = game.world.actors[sender].home + for recipient in 0 ..< game.inboxes.len: + let distance = tileDistance(origin, game.world.actors[recipient].home) + if distance in 0 .. 16 and game.inboxes[recipient].push(-2, text): + inc result + proc buildHeroHost(heroId: int32): Host = ## Builds the world-query and high-level action API for one hero. result = initHost() result.addNeuralFunctions() + let sendChatProc: NumericHostProc = proc(args: openArray[Value]): Value = + ## Sends script text through the game's routing rules. + let player = int(heroId - 100) + activeGame.heroVms[player].runtime.withString(args[1], text): + result = activeGame.sendChat(player, int(args[0].asInt), text) + let pullMailboxProc: NumericHostProc = proc(args: openArray[Value]): Value = + ## Copies the oldest message into BASIC and consumes it on success. + let + player = int(heroId - 100) + inbox = activeGame.inboxes[player] + var runtime = activeGame.heroVms[player].runtime + if inbox.count == 0: + result = runtime.putString("") + else: + result = runtime.putString(inbox.messages[inbox.first]) + discard inbox.pop() + let mailboxIdProc: HostProc = proc(args: openArray[int32]): int32 = + ## Returns the channel or DM sender of the last pulled message. + activeGame.inboxes[int(heroId - 100)].lastId + let mailboxCountProc: HostProc = proc(args: openArray[int32]): int32 = + ## Counts this player's unread messages. + int32(activeGame.inboxes[int(heroId - 100)].count) + let mailboxSelfProc: HostProc = proc(args: openArray[int32]): int32 = + ## Returns this player's zero-based mailbox address. + int32(int(heroId - 100)) + let mailboxPlayersProc: HostProc = proc(args: openArray[int32]): int32 = + ## Returns the number of player mailboxes in this game. + int32(activeGame.inboxes.len) + discard result.addFunction("sendChat", 2, sendChatProc, 256) + discard result.addFunction("pullMailbox$", 0, pullMailboxProc, 256) + discard result.addFunction("mailboxId", 0, mailboxIdProc, 4) + discard result.addFunction("mailboxCount", 0, mailboxCountProc, 4) + discard result.addFunction("mailboxSelf", 0, mailboxSelfProc, 4) + discard result.addFunction("mailboxPlayers", 0, mailboxPlayersProc, 4) for name in HeroDataNames: discard result.addData(name) @@ -227,15 +275,18 @@ proc loadBots*( schema = buildHeroHost(100) kinds = controllerKinds(PartySize, playerSlot) sources = groups.expandBotSources(kinds) + for inbox in game.inboxes.mitems: + inbox = newMailbox() var bound = false for slot in 0 ..< PartySize: if kinds[slot] == PlayerController: continue + let source = sources[slot] let program = when defined(coworld): - compilePlayer(sources[slot], schema, limits, int(slot)) + compilePlayer(source, schema, limits, int(slot)) else: - compile(sources[slot], schema, limits) + compile(source, schema, limits) if not bound: bindHeroData(program) bound = true @@ -245,7 +296,7 @@ proc loadBots*( buildHeroHost(int32(100 + slot)), limits ), - ready: true + ready: true, ) when defined(coworld): game.heroVms[slot].output = playerPrinter(int(slot)) @@ -262,8 +313,8 @@ proc runBotDecisions*(game: Game, slot: int32) {.measure.} = activeGame = game activeHeroSlot = slot let objective = game.objectiveTile(slot) - game.heroVms[slot].runtime.restart() try: + game.heroVms[slot].runtime.restart() game.heroVms[slot].runtime.setData(heroDataIds[DataSelfId], actor.id) game.heroVms[slot].runtime.setData( heroDataIds[DataSelfClass], @@ -331,4 +382,3 @@ proc runBotDecisions*(game: Game, slot: int32) {.measure.} = ) activeGame = nil activeHeroSlot = -1 - diff --git a/examples/call_to_adventure/sim.nim b/examples/call_to_adventure/sim.nim index 5d9fe1e2..9e7b55c5 100644 --- a/examples/call_to_adventure/sim.nim +++ b/examples/call_to_adventure/sim.nim @@ -11,7 +11,7 @@ import bassy, fixxy, polyworld/[bodies, hashes, metrics, pathing, profiles, rngs, tapes, - visions], + visions, mailboxes], content, maps, replays @@ -41,6 +41,7 @@ type historyPlayback*: bool replayMode*: bool heroVms*: array[PartySize, HeroVm] + inboxes*: array[PartySize, Mailbox] const AggroTiles* = 9'i32 diff --git a/examples/gods_of_the_arena/bots.nim b/examples/gods_of_the_arena/bots.nim index 29be517e..e4c0013e 100644 --- a/examples/gods_of_the_arena/bots.nim +++ b/examples/gods_of_the_arena/bots.nim @@ -2,12 +2,15 @@ ## on the simulation. import + std/math, polyworld/neural, bassy, fixxy, - polyworld/[metrics, bodies, cli, controllers, pathing, profiles, tapes], + polyworld/[mailboxes, metrics, bodies, cli, controllers, neural_host, + pathing, profiles, tapes], content, maps, motions, + neural_contract, observations, sim, replays, @@ -91,18 +94,26 @@ const ] var - activeGame: Game + activeGame {.threadvar.}: Game + ## Per thread: separate games may run their heroes on separate threads. heroDataIds: array[HeroDataSlot, int32] + heroDataBound: bool proc bindHeroData(program: Program) = ## Resolves host data slots once so think ticks do not allocate names. + ## Every hero program shares the host's data order, so one binding serves + ## all of them (and is never rewritten while other threads read it). + if heroDataBound: + return for slot, name in HeroDataNames: heroDataIds[slot] = program.hostDataIndex(name) doAssert heroDataIds[slot] >= 0, "missing host data " & name + heroDataBound = true proc heroVmLimits(): Limits = ## Returns independent structural and per-decision limits for a hero VM. result = defaultLimits() + result.maxStringBytes = 256 * 1024 result.maxSourceBytes = 64 * 1024 result.maxCodeInstructions = 20_000 result.maxArrays = 32 @@ -260,10 +271,147 @@ proc abilityProc(heroId: int32, field: AbilityField): HostProc = of AbilityRestore: spec.restore of AbilityManaCost: spec.manaCost +proc sendChat*( + game: Game, sender, target: int, text: openArray[char] +): int32 = + ## Routes chat according to this game's player and team rules. + if sender notin 0 ..< game.inboxes.len or + target < -2 or target >= game.inboxes.len: + return 0 + let id = int32(if target < 0: target else: sender) + for recipient in 0 ..< game.inboxes.len: + case target + of -2: + discard + of -1: + if game.world.heroes[recipient].team != game.world.heroes[sender].team: + continue + else: + if recipient != target: + continue + if game.inboxes[recipient].push(id, text): + inc result + + +proc issueCommand(game: Game, heroId: int32, accepted: bool): bool = + ## Credits an accepted order to the hero's command metrics. + if accepted: + game.metrics.command(heroIndex(game.world, heroId), game.world.tick) + accepted + +proc recordingFailed(game: Game, error: ref ReplayError) {.noreturn.} = + game.recordingError = error.msg + raise newException(BasicError, "replay recording failed: " & error.msg) + +proc issueWalkTo*(game: Game, heroId, x, y: int32, offset: FixedVec2): bool = + ## Records and applies a walk order exactly as BASIC's walkTo does. + try: + if game.recorder != nil: + game.recorder.recordWalkTo(uint32(game.world.tick), heroId, x, y, offset) + except ReplayError as error: + game.recordingFailed(error) + game.issueCommand(heroId, applyWalkTo(game.world, heroId, x, y, offset)) + +proc issueAttackMove*(game: Game, heroId, x, y: int32, offset: FixedVec2): bool = + ## Records and applies the same attack-move order used by human players. + try: + if game.recorder != nil: + game.recorder.record ReplayAction( + tick: uint32(game.world.tick), heroId: heroId, + kind: ActionAttackMove, first: x, second: y, offset: offset + ) + except ReplayError as error: + game.recordingFailed(error) + game.issueCommand(heroId, game.world.applyAttackMove(heroId, x, y, offset)) + +proc issueAttackTarget*(game: Game, heroId, targetId: int32): bool = + ## Records and applies an attack order. + try: + if game.recorder != nil: + game.recorder.recordAttackTarget(uint32(game.world.tick), heroId, targetId) + except ReplayError as error: + game.recordingFailed(error) + game.issueCommand(heroId, applyAttackTarget(game.world, heroId, targetId)) + +proc issueUseItem*(game: Game, heroId, slot: int32): bool = + ## Records and applies an item use. + try: + if game.recorder != nil: + game.recorder.recordUseItem(uint32(game.world.tick), heroId, slot) + except ReplayError as error: + game.recordingFailed(error) + game.issueCommand(heroId, applyUseItem(game.world, heroId, slot)) + +proc issueUseItemAt*(game: Game, heroId, slot, x, y: int32, + offset: FixedVec2): bool = + ## Records and attempts a scroll channel at fractional map coordinates. + try: + game.recorder.recordUseItemAt(uint32(game.world.tick), heroId, slot, x, y, offset) + except ReplayError as error: + game.recordingFailed(error) + game.issueCommand(heroId, game.world.applyUseItemAt(heroId, slot, x, y, offset)) + +proc issueCastTarget*(game: Game, heroId, slot, targetId: int32): bool = + ## Records and attempts an explicit object-targeted spell. + try: + game.recorder.recordCast(uint32(game.world.tick), heroId, slot, targetId, 0, false) + except ReplayError as error: + game.recordingFailed(error) + game.issueCommand(heroId, game.world.applyCastTarget(heroId, slot, targetId)) + +proc issueCastPoint*(game: Game, heroId, slot, x, y: int32, + offset: FixedVec2): bool = + ## Records and attempts a ground-aimed spell. + try: + game.recorder.recordCast(uint32(game.world.tick), heroId, slot, x, y, true, offset) + except ReplayError as error: + game.recordingFailed(error) + game.issueCommand(heroId, game.world.applyCastPoint(heroId, slot, x, y, offset)) + +include neural_host_hooks + proc initHeroHost(heroId: int32): Host = ## Builds the bounded world-query and action interface for one hero. result = initHost() result.addNeuralFunctions() + let sendChatProc: NumericHostProc = proc(args: openArray[Value]): Value = + ## Sends script text through the game's routing rules. + if shadowRunning: + return toValue(0'i32) + let player = activeGame.world.heroIndex(heroId) + activeGame.heroVms[player].runtime.withString(args[1], text): + result = activeGame.sendChat(player, int(args[0].asInt), text) + let pullMailboxProc: NumericHostProc = proc(args: openArray[Value]): Value = + ## Copies the oldest message into BASIC and consumes it on success. + if shadowRunning: + return activeGame.neuralSeat(activeGame.world.heroIndex(heroId)).shadow.runtime.putString("") + let + player = activeGame.world.heroIndex(heroId) + inbox = activeGame.inboxes[player] + var runtime = activeGame.heroVms[player].runtime + if inbox.count == 0: + result = runtime.putString("") + else: + result = runtime.putString(inbox.messages[inbox.first]) + discard inbox.pop() + let mailboxIdProc: HostProc = proc(args: openArray[int32]): int32 = + ## Returns the channel or DM sender of the last pulled message. + activeGame.inboxes[activeGame.world.heroIndex(heroId)].lastId + let mailboxCountProc: HostProc = proc(args: openArray[int32]): int32 = + ## Counts this player's unread messages. + int32(activeGame.inboxes[activeGame.world.heroIndex(heroId)].count) + let mailboxSelfProc: HostProc = proc(args: openArray[int32]): int32 = + ## Returns this player's zero-based mailbox address. + int32(activeGame.world.heroIndex(heroId)) + let mailboxPlayersProc: HostProc = proc(args: openArray[int32]): int32 = + ## Returns the number of player mailboxes in this game. + int32(activeGame.inboxes.len) + discard result.addFunction("sendChat", 2, sendChatProc, 256) + discard result.addFunction("pullMailbox$", 0, pullMailboxProc, 256) + discard result.addFunction("mailboxId", 0, mailboxIdProc, 4) + discard result.addFunction("mailboxCount", 0, mailboxCountProc, 4) + discard result.addFunction("mailboxSelf", 0, mailboxSelfProc, 4) + discard result.addFunction("mailboxPlayers", 0, mailboxPlayersProc, 4) for error in ActionError: discard result.addData($error, error.ord.int32) for class in HeroClass: @@ -287,6 +435,8 @@ proc initHeroHost(heroId: int32): Host = let draftHeroProc: HostProc = proc(arguments: openArray[int32]): int32 = ## Records and validates the active player's hero choice. + if shadowRunning: + return 1 try: activeGame.recorder.record ReplayAction( tick: uint32(activeGame.world.tick), heroId: heroId, @@ -412,80 +562,29 @@ proc initHeroHost(heroId: int32): Host = ): Value = let (x, y, offset) = splitTilePoint(fixedVec2( arguments[0].asFixed, arguments[1].asFixed)) - try: - if activeGame.recorder != nil: - activeGame.recorder.recordWalkTo( - uint32(activeGame.world.tick), - heroId, - x, - y, - offset - ) - except ReplayError as error: - activeGame.recordingError = error.msg - raise newException( - BasicError, - "replay recording failed: " & error.msg - ) - let accepted = applyWalkTo( - activeGame.world, heroId, x, y, offset - ) - if accepted: - activeGame.metrics.command( - heroIndex(activeGame.world, heroId), activeGame.world.tick - ) - int32(accepted) + let command = NeuralCommand(kind: WalkCommand, point: fixedVec2( + arguments[0].asFixed, arguments[1].asFixed)) + if activeGame.interceptCommand(heroId, command): + return 1'i32 + int32(activeGame.issueWalkTo(heroId, x, y, offset)) let attackMoveProc: NumericHostProc = proc( arguments: openArray[Value] ): Value = ## Records and applies the same attack-move order used by human players. let (x, y, offset) = splitTilePoint(fixedVec2( arguments[0].asFixed, arguments[1].asFixed)) - try: - if activeGame.recorder != nil: - activeGame.recorder.record ReplayAction( - tick: uint32(activeGame.world.tick), - heroId: heroId, - kind: ActionAttackMove, - first: x, - second: y, - offset: offset - ) - except ReplayError as error: - activeGame.recordingError = error.msg - raise newException(BasicError, "replay recording failed: " & error.msg) - let accepted = activeGame.world.applyAttackMove( - heroId, x, y, offset - ) - if accepted: - activeGame.metrics.command( - heroIndex(activeGame.world, heroId), activeGame.world.tick - ) - int32(accepted) + let command = NeuralCommand(kind: AttackMoveCommand, point: fixedVec2( + arguments[0].asFixed, arguments[1].asFixed)) + if activeGame.interceptCommand(heroId, command): + return 1'i32 + int32(activeGame.issueAttackMove(heroId, x, y, offset)) let attackTargetProc: HostProc = proc( arguments: openArray[int32] ): int32 = - try: - if activeGame.recorder != nil: - activeGame.recorder.recordAttackTarget( - uint32(activeGame.world.tick), - heroId, - arguments[0] - ) - except ReplayError as error: - activeGame.recordingError = error.msg - raise newException( - BasicError, - "replay recording failed: " & error.msg - ) - let accepted = applyAttackTarget( - activeGame.world, heroId, arguments[0] - ) - if accepted: - activeGame.metrics.command( - heroIndex(activeGame.world, heroId), activeGame.world.tick - ) - int32(accepted) + if activeGame.interceptCommand(heroId, + NeuralCommand(kind: AttackTargetCommand, objectId: arguments[0])): + return 1 + int32(activeGame.issueAttackTarget(heroId, arguments[0])) let itemIdProc: HostProc = proc( arguments: openArray[int32] ): int32 = @@ -518,6 +617,8 @@ proc initHeroHost(heroId: int32): Host = let buyItemProc: HostProc = proc( arguments: openArray[int32] ): int32 = + if shadowRunning: + return 1 try: if activeGame.recorder != nil: activeGame.recorder.recordBuyItem( @@ -544,6 +645,8 @@ proc initHeroHost(heroId: int32): Host = activeGame.world.buybackPrice(heroId) let buybackProc: HostProc = proc(arguments: openArray[int32]): int32 = ## Records and attempts a buyback using only this hero's gold. + if shadowRunning: + return 1 try: activeGame.recorder.recordBuyback( uint32(activeGame.world.tick), heroId @@ -560,50 +663,26 @@ proc initHeroHost(heroId: int32): Host = let useItemProc: HostProc = proc( arguments: openArray[int32] ): int32 = - try: - if activeGame.recorder != nil: - activeGame.recorder.recordUseItem( - uint32(activeGame.world.tick), - heroId, - arguments[0] - ) - except ReplayError as error: - activeGame.recordingError = error.msg - raise newException( - BasicError, - "replay recording failed: " & error.msg - ) - let accepted = applyUseItem( - activeGame.world, heroId, arguments[0] - ) - if accepted: - activeGame.metrics.command( - heroIndex(activeGame.world, heroId), activeGame.world.tick - ) - int32(accepted) + if activeGame.interceptCommand(heroId, + NeuralCommand(kind: UseItemCommand, item: arguments[0])): + return 1 + int32(activeGame.issueUseItem(heroId, arguments[0])) let useItemAtProc: NumericHostProc = proc(arguments: openArray[Value]): Value = ## Records and attempts a scroll channel at fractional map coordinates. let - (x, y, offset) = splitTilePoint(fixedVec2( - arguments[1].asFixed, arguments[2].asFixed)) + point = fixedVec2(arguments[1].asFixed, arguments[2].asFixed) + (x, y, offset) = splitTilePoint(point) slot = arguments[0].asInt - try: - activeGame.recorder.recordUseItemAt( - uint32(activeGame.world.tick), heroId, slot, x, y, offset - ) - except ReplayError as error: - activeGame.recordingError = error.msg - raise newException(BasicError, "replay recording failed: " & error.msg) - let accepted = activeGame.world.applyUseItemAt(heroId, slot, x, y, offset) - if accepted: - activeGame.metrics.command( - heroIndex(activeGame.world, heroId), activeGame.world.tick - ) - int32(accepted) + if activeGame.interceptCommand(heroId, + NeuralCommand(kind: UseItemAtCommand, item: slot, point: point)): + return 1'i32 + int32(activeGame.issueUseItemAt(heroId, slot, x, y, offset)) let levelAbilityProc: HostProc = proc(arguments: openArray[int32]): int32 = ## Records and spends a point through the shared upgrade validator. + if shadowRunning: + return 1 try: activeGame.recorder.recordLevelAbility( uint32(activeGame.world.tick), heroId, arguments[0] @@ -639,46 +718,20 @@ proc initHeroHost(heroId: int32): Host = let castTargetProc: HostProc = proc(arguments: openArray[int32]): int32 = ## Records and attempts an explicit object-targeted spell. - let slot = arguments[0] - try: - activeGame.recorder.recordCast( - uint32(activeGame.world.tick), heroId, slot, arguments[1], 0, false - ) - except ReplayError as error: - activeGame.recordingError = error.msg - raise newException(BasicError, "replay recording failed: " & error.msg) - let accepted = activeGame.world.applyCastTarget(heroId, slot, arguments[1]) - if accepted: - activeGame.metrics.command( - heroIndex(activeGame.world, heroId), activeGame.world.tick - ) - int32(accepted) + if activeGame.interceptCommand(heroId, NeuralCommand(kind: CastTargetCommand, + ability: arguments[0], objectId: arguments[1])): + return 1 + int32(activeGame.issueCastTarget(heroId, arguments[0], arguments[1])) let castPointProc: NumericHostProc = proc(arguments: openArray[Value]): Value = ## Records and attempts a ground-aimed spell. - let (x, y, offset) = splitTilePoint(fixedVec2( - arguments[1].asFixed, arguments[2].asFixed)) - let slot = arguments[0].asInt - try: - activeGame.recorder.recordCast( - uint32(activeGame.world.tick), - heroId, - slot, - x, - y, - true, - offset - ) - except ReplayError as error: - activeGame.recordingError = error.msg - raise newException(BasicError, "replay recording failed: " & error.msg) - let accepted = activeGame.world.applyCastPoint( - heroId, slot, x, y, offset - ) - if accepted: - activeGame.metrics.command( - heroIndex(activeGame.world, heroId), activeGame.world.tick - ) - int32(accepted) + let + point = fixedVec2(arguments[1].asFixed, arguments[2].asFixed) + (x, y, offset) = splitTilePoint(point) + slot = arguments[0].asInt + if activeGame.interceptCommand(heroId, + NeuralCommand(kind: CastPointCommand, ability: slot, point: point)): + return 1'i32 + int32(activeGame.issueCastPoint(heroId, slot, x, y, offset)) let abilityChargesProc: HostProc = proc(arguments: openArray[int32]): int32 = ## Reads remaining charges for one of this hero's four ability slots. let index = heroIndex(activeGame.world, heroId) @@ -787,6 +840,55 @@ proc initHeroHost(heroId: int32): Host = 32 ) +proc installPackageSeat*(game: Game, i: int, bytes: string) = + ## Loads a neural package into one seat: policy.bas runs as the seat's + ## BASIC program and the network runs natively at each decision tick. + activeGame = game + var package: NeuralPackage + try: + package = parseGotaPackage(bytes) + except CatchableError: + let message = "neural package rejected: " & getCurrentExceptionMsg() + when defined(coworld): + failPlayer(i, message, "neural package rejected for player slot " & $i) + raise newException(BasicError, message) + let + options = package.gotaOptions + limits = if options.deferScript: deferVmLimits() else: neuralVmLimits() + heroId = game.world.heroes[i].id + var schema = initHeroHost(0) + schema.addNeuralSeatFunctions(0) + let program = + try: + compile(package.policy, schema, limits) + except BasicError as error: + let message = "neural package rejected for player slot " & $i & + ": policy.bas failed to compile: " & error.msg + when defined(coworld): + failPlayer(i, message, message) + raise newException(BasicError, message) + bindHeroData(program) + var host = initHeroHost(heroId) + host.addNeuralSeatFunctions(heroId) + let seat = newNeuralSeat(NeuralHosted, package.decisionPeriod, + game.config.maxTicks) + seat.actor = package.actor + seat.goal = options.goals[game.world.heroes[i].team.ord] + seat.sampling = package.decoder.sampling + seat.temperature = package.decoder.temperatureOf + seat.telemetry = true + seat.deferEnabled = options.deferScript + seat.maskMode = options.maskMode + seat.resetEpisode(game.world.matchSeed, i) + game.heroVms[i] = HeroVm( + runtime: initRuntime(program, host, limits), + limits: limits, + ready: true, + neural: seat + ) + when defined(coworld): + game.heroVms[i].output = playerPrinter(i) + proc loadBots*( game: Game, groups: openArray[BotGroup], @@ -800,15 +902,22 @@ proc loadBots*( kinds = controllerKinds(game.world.heroes.len, playerSlot) sources = groups.expandBotSources(kinds) game.heroVms.setLen(game.world.heroes.len) + game.inboxes.setLen(game.world.heroes.len) + for inbox in game.inboxes.mitems: + inbox = newMailbox() var bound = false for i in 0 ..< game.world.heroes.len: if kinds[i] == PlayerController: continue + let source = sources[i] + if source.isPackage: + game.installPackageSeat(i, source) + continue let program = when defined(coworld): - compilePlayer(sources[i], schema, limits, int(i)) + compilePlayer(source, schema, limits, int(i)) else: - compile(sources[i], schema, limits) + compile(source, schema, limits) if not bound: bindHeroData(program) bound = true @@ -824,17 +933,16 @@ proc loadBots*( when defined(coworld): game.heroVms[i].output = playerPrinter(int(i)) -proc runHeroScript(game: Game, index: int) = +proc runHeroVm(game: Game, index: int, vm: HeroVm, primary: bool) = ## Runs one bounded BASIC decision, including while awaiting respawn. - if index < 0 or index >= game.heroVms.len: - return - let - hero = game.world.heroes[index] - vm = game.heroVms[index] + ## `primary` is false only for a learner seat's shadow expert script. + let hero = game.world.heroes[index] if vm == nil or vm.failed: return - vm.runtime.restart() try: + if primary and vm.neural != nil and NeuralSeat(vm.neural).deferEnabled: + game.deferConsult(index) + vm.runtime.restart() discard game.world.worldObjectCount(hero.id) vm.runtime.setData(heroDataIds[DataSelfId], hero.id) vm.runtime.setData(heroDataIds[DataSelfTeam], int32(hero.team.ord)) @@ -890,6 +998,8 @@ proc runHeroScript(game: Game, index: int) = vm.runtime.setData(heroDataIds[DataSelfRespawnTicks], hero.respawnTicks()) discard vm.runtime.run(vm.output) inc vm.decisions + if primary and vm.neural != nil and NeuralSeat(vm.neural).mode == NeuralOverride: + game.runOverride(index, NeuralSeat(vm.neural)) except BasicError as error: vm.failed = true vm.lastError = error.msg @@ -899,10 +1009,26 @@ proc runHeroScript(game: Game, index: int) = echo "hero ", hero.id, " BASIC error: ", error.msg vm.lastWork = vm.runtime.workUsed vm.lastInstructions = vm.runtime.instructionsUsed - game.metrics.decision( - index, game.world.tick, vm.lastInstructions, - heroVmLimits().maxInstructions - ) + if primary: + game.metrics.decision( + index, game.world.tick, vm.lastInstructions, + vm.limits.maxInstructions + ) + +proc runHeroScript(game: Game, index: int) = + ## Runs one hero's BASIC decision (and a learner's shadow expert, if any). + if index < 0 or index >= game.heroVms.len: + return + let vm = game.heroVms[index] + game.runHeroVm(index, vm, true) + if vm != nil and vm.neural != nil: + let shadow = NeuralSeat(vm.neural).shadow + if shadow != nil and not shadow.failed: + shadowRunning = true + try: + game.runHeroVm(index, shadow, false) + finally: + shadowRunning = false proc runBotDecisions*(game: Game) {.measure.} = ## Runs every VM in seeded cyclic order and advances the first slot. @@ -914,6 +1040,10 @@ proc runBotDecisions*(game: Game) {.measure.} = defer: if ownsFrame: game.world.thawObservations() + for vm in game.heroVms: + if vm != nil and vm.neural != nil: + game.neuralPrelude() + break for offset in 0 ..< game.world.heroes.len: let index = (game.world.heroTurnStart + offset) mod game.world.heroes.len runHeroScript(game, index) diff --git a/examples/gods_of_the_arena/docs/index.html b/examples/gods_of_the_arena/docs/index.html index 1c7f0df7..99bfd4c5 100644 --- a/examples/gods_of_the_arena/docs/index.html +++ b/examples/gods_of_the_arena/docs/index.html @@ -2235,7 +2235,7 @@

Poison Potion

Inside your own spawn room, health and mana each recover at 20% of maximum per second. This is separate from potions and continues under attack. It does not apply elsewhere in the keep or in the enemy spawn, and cannot revive a dead hero. inOwnSpawn() reports whether your living hero is in this recovery area.

Portal Scroll

-

Portal Scroll Item 21 costs 100 gold and stacks to eight. Click the scroll, then right-click the map or minimap. It consumes one scroll and channels for 3 seconds, teleporting to visible, walkable ground within a living allied tower's sight (7 tiles for outer/inner towers, 8 for gate/guard towers). Distant destinations are clamped into tower range. Barracks cannot anchor a teleport.

+

Portal Scroll Item 21 costs 100 gold and stacks to eight. Click the scroll, then right-click the map or minimap. It consumes one scroll and channels for 3 seconds, teleporting to visible, walkable ground within a living allied tower's sight (9 tiles for outer towers, 9.5 for inner towers, 10 for gate/guard towers, matching attack range). Distant destinations are clamped into tower range. Barracks cannot anchor a teleport.

You cannot move, attack, or cast during the channel, but can still take damage and die. Stun, root, death, or destruction of the selected tower cancels it. Damage alone does not cancel it. Completion or interruption starts a shared 60-second cooldown, preserved through death. Blazing Blade, Golem Seed, and Bone Marionette can interrupt the channel on impact. Silence does not interrupt it.

Control effects: stun blocks movement, basic attacks, abilities, and items, cancelling a pending swing. Silence blocks all four abilities but allows moving, attacking, and items. Root blocks moving and teleporting but allows in-range attacks, abilities, and other items. Stun and root interrupt teleport channels. Already released spells still resolve. Effects apply to enemy heroes and creeps; buildings and gods are immune. Different effects coexist, with icons and countdown rings above visible units. Reapplying one keeps the later expiration instead of adding durations. Death clears all effects. Silenced casts return ActionSilenced (35).

BASIC: useItemAt(slot, x, y) accepts fractional coordinates and returns 1 on success, 0 on rejection. selfPortalCooldown, selfChannelTicks, selfStunTicks, selfRootTicks, and selfSilenceTicks report remaining ticks. useItem(slot) rejects scrolls because they require a destination.

diff --git a/examples/gods_of_the_arena/graphics.nim b/examples/gods_of_the_arena/graphics.nim index 861231a7..cfc12a41 100644 --- a/examples/gods_of_the_arena/graphics.nim +++ b/examples/gods_of_the_arena/graphics.nim @@ -1818,8 +1818,7 @@ proc runGraphics*() = proc emitTickParticles( oldHeroLanded: seq[bool], - oldFootmanLanded: Table[int32, bool], - oldTowerTicks: seq[int32] + oldFootmanLanded: Table[int32, bool] ) = ## Emits each authoritative attack transition exactly once. for i, hero in run.world.heroes: @@ -1860,27 +1859,6 @@ proc runGraphics*() = renderPoint(footman.position) + vec3(0, 1.4'f, 0), target.position ) - for i, tower in run.world.buildings: - if i >= oldTowerTicks.len or tower.targetId == 0 or - oldTowerTicks[i] != TowerAttackTicks - 1 or - tower.attackTicks != 0: - continue - let target = particleTargetPosition(tower.targetId) - if not target.found: - continue - let origin = renderPoint(tower.position) + - vec3(0, buildingScale(tower) * 0.72'f32, 0) - particles.emitParticleProjectile( - Fireball, - FireBurst, - origin, - target.position, - clamp( - (target.position - origin).length / 14.0'f32, - 0.14'f32, - 0.5'f32 - ) - ) proc advanceRenderedSimulation() = ## Advances one simulation tick and starts any new god animation. @@ -1889,14 +1867,11 @@ proc runGraphics*() = oldHeroLanded = newSeq[bool](run.world.heroes.len) oldPortalEnds = newSeq[int32](run.world.heroes.len) oldFootmanLanded: Table[int32, bool] - oldTowerTicks = newSeq[int32](run.world.buildings.len) for i, hero in run.world.heroes: oldHeroLanded[i] = hero.damageLanded oldPortalEnds[i] = hero.portalEnds for footman in run.world.footmen: oldFootmanLanded[footman.id] = footman.damageLanded - for i, tower in run.world.buildings: - oldTowerTicks[i] = tower.attackTicks captureUnitPositions() advanceGame() for i, hero in run.world.heroes: @@ -1905,8 +1880,7 @@ proc runGraphics*() = feedGotaActions(observeTick = true) emitTickParticles( oldHeroLanded, - oldFootmanLanded, - oldTowerTicks + oldFootmanLanded ) if int(run.world.tick) mod SeekCheckpointTicks == 0 or int32(run.world.tick) == transport.timelineEnd: @@ -2268,7 +2242,7 @@ proc runGraphics*() = var ring: array[65, Vec3] let center = renderPoint(tower.position) - radius = tower.tier.towerSightTiles.float32 + radius = TowerAttackRanges[tower.tier].float32 / WorldScale.float32 for i in 0 .. ring.high: let angle = i.float32 * 2 * PI.float32 / ring.high.float32 ring[i] = center + vec3(cos(angle) * radius, 0, diff --git a/examples/gods_of_the_arena/native_env.h b/examples/gods_of_the_arena/native_env.h new file mode 100644 index 00000000..edcc20ad --- /dev/null +++ b/examples/gods_of_the_arena/native_env.h @@ -0,0 +1,369 @@ +#ifndef GOTA_NATIVE_ENV_H +#define GOTA_NATIVE_ENV_H +/* Gods of the Arena native training environment, ABI v1. + * + * Built from examples/gods_of_the_arena/native_env.nim as a shared library + * (build command: the header comment of native_env.nim). One handle owns one + * ten-seat match. + * Threads (built with --mm:atomicArc --threads:on -d:useMalloc, the documented + * build): DIFFERENT handles may be called concurrently from different threads; + * one handle must be used by one thread at a time, but that thread may change + * between calls, and gota_destroy may run on any thread. Per-world scratch is + * thread-local; the map, navigation graph and lane globals are built once + * (first gota_reset, under a process lock) and only read afterwards. + * Acceptance: tools/test_native_concurrency.py (N threads x M handles give the + * serial per-step hashes), and tools/test_thread_isolation.py (several + * worlds per thread over full matches == each world alone; concurrent + * gota_create). gota_create is thread-safe (it runs under the process lock). + * World state that one world's tick reads later (vision blockers included) + * lives in the World, never in per-thread scratch. A --mm:orc build is ~10% + * faster but single-thread. + * All handles of a process must use the same map preset (gota_create refuses + * a second preset). + * + * Seats are hero slots 0..9: seats 0..4 are red, 5..9 blue. Every seat is + * either a LEARNER seat (the caller supplies its actions) or a SCRIPTED seat + * (any .bas; default players/base.bas). Any mix is allowed on either team: + * one learner + nine scripts (league-like), ten learners (self-play), or e.g. + * two learners per team with scripted teammates. Learner seats run the policy + * glue script (default neural/policy.bas: draft, shopping, ability leveling + * and buyback from base.bas's routines, then gota_act) exactly as a hosted + * neural-package seat does; only the network is replaced by the caller. + * + * Time. gota_reset plays the draft (BASIC picks) and pauses at the first + * battle decision. gota_step executes the learner actions at the paused + * decision tick, then advances decision_period ticks in total (default 4, + * config "decision_period"; the manifest key of the same name must match) and + * pauses at the next decision tick, after that tick's cooldowns and vision + * are updated and its observation frame is frozen, before any seat's BASIC + * runs (the same point where a hosted neural seat observes). The decoded + * command is issued once, on the decision tick, when the seat's policy.bas + * calls gota_act; the engine keeps executing it (paths, attack targets) + * until the next decision ("held command"); noop issues nothing. + * + * Observation contract v1 (GOTA_OBS_SIZE floats per seat, ego-centric, team + * frame: blue seats see the 180-degree rotated map so both teams share one + * frame): self 48 | abilities 4x16 | items 6x25 | objects 25x40 | spell + * warnings 4x8 | summary 16 | terrain 9x9 | goal 16. Exact layout and + * normalizers: neural_basic.md. The LAST 16 floats are the goal vector w + * (order below) set by gota_set_seat_goal. + * + * Action contract v1: five int32 heads per seat, in this order: + * verb 8 {0 noop, 1 walk, 2 attackMove, 3 attackTarget, 4 castTarget, + * 5 castPoint, 6 useItem, 7 useItemAt} + * target 25 object slot (0 self, 1-4 allies, 5-9 enemy heroes, 10-16 lane + * creeps, 17-20 structures {own god, enemy god, own forward + * tower, nearest enemy tower/barracks}, 21-24 neutrals) + * point 49 0 = centre; 1 + ring*16 + dir, dir k at k*22.5 degrees + * counter-clockwise from team-frame +x. walk/attackMove: centre + * self, rings {2, 5, 12} tiles. castPoint/useItemAt: centre the + * target slot's object, rings {0.5, 1.25, 2.5} tiles. + * ability 4 ability slot (castTarget/castPoint) + * item 6 inventory slot (useItem/useItemAt) + * Out-of-range indices, empty target slots and dead seats decode to noop. + * + * Every entry point returns >= 0 on success and -1 for bad arguments (NULL + * handle, seat out of range, short buffer) unless stated otherwise. Worlds in + * which nobody calls an optional setter are byte-identical (state hash and + * replay) to plain BASIC matches with the same scripts. */ +#include +#ifdef __cplusplus +extern "C" { +#endif + +/* Opaque handles. A destroyed or foreign handle is rejected (-1 / NULL) + * when the library can tell; never pass freed handles. */ +typedef struct GotaEnv GotaEnv; +typedef struct GotaNet GotaNet; + +#define GOTA_ENV_VERSION 1 +#define GOTA_SEATS 10 +#define GOTA_OBS_SIZE 1407 +#define GOTA_GOAL_SIZE 16 +#define GOTA_HEADS 5 +#define GOTA_ACTION_OUTPUTS 92 /* 8 + 25 + 49 + 4 + 6 logits */ +#define GOTA_STAT_COUNT 24 +#define GOTA_ORDER_SIZE 16 +#define GOTA_DEFAULT_DECISION_PERIOD 4 + +/* Goal vector order; w_reserved must be 0. */ +enum { + GOTA_W_SCORE = 0, GOTA_W_WIN, GOTA_W_XP, GOTA_W_GOLD, GOTA_W_HERO_KILL, + GOTA_W_ASSIST, GOTA_W_DEATH, GOTA_W_LAST_HIT, GOTA_W_NEUTRAL_KILL, + GOTA_W_TOWER_DAMAGE, GOTA_W_STRUCTURE_KILL, GOTA_W_HERO_DAMAGE, + GOTA_W_DAMAGE_TAKEN, GOTA_W_PUSH_DEPTH, GOTA_W_GOD_DAMAGE, GOTA_W_RESERVED +}; + +/* Static contract facts, for load-time checks. */ +int gota_env_version(void); /* GOTA_ENV_VERSION */ +int gota_observation_size(void); /* GOTA_OBS_SIZE */ +int gota_goal_size(void); /* GOTA_GOAL_SIZE */ +int gota_action_heads(int32_t *sizes); /* writes {8,25,49,4,6}; returns 5 */ +int gota_stat_count(void); /* GOTA_STAT_COUNT */ +int gota_mask_size(void); /* GOTA_MASK_SIZE */ +/* 64 lowercase hex chars + NUL (capacity >= 65). Returns 0. */ +int gota_observation_contract_hash(char *out, int32_t capacity); +int gota_action_contract_hash(char *out, int32_t capacity); + +/* config_json (UTF-8, NUL-terminated; every key optional): + * "config_path": GotA match config JSON (map preset, spawn interval); + * default = the league preset (presets/ default). + * "seed": match seed of the first episode (default 0). + * "max_ticks": battle ticks per episode, 1..28800 (default 28800). + * "decision_period": ticks per gota_step, 1..24 (default 4). + * "learner_seats": array of seat indices (default [0]). + * "script_path": .bas for every scripted seat (default players/base.bas). + * "policy_path": glue .bas for learner seats (default neural/policy.bas). + * "data_root": directory the relative defaults resolve against + * (default: the examples/gods_of_the_arena directory + * compiled into the library). + * "record": record a replay (default false; gota_save_replay). + * "capture": label scripted seats' commands (default true). + * Returns NULL on failure with the reason in error (may be NULL). The match is + * not started: call gota_reset. */ +GotaEnv *gota_create(const char *config_json, char *error, int32_t capacity); +void gota_destroy(GotaEnv *handle); + +/* Starts a new episode with this match seed (map preset fixed; seed drives + * draft order, turn order and combat randomness), re-instantiates every + * seat's BASIC program (persistent variables cleared), zeroes seat stats, + * plays the draft and pauses at battle tick 1's decision. 0, or -2 when a + * script failed to compile (see gota_seat_script_status; that seat idles). */ +int gota_reset(GotaEnv *handle, int64_t seed); + +/* Learner seat mask (bit s = seat s). Takes effect at the next gota_reset. */ +int gota_set_learner_seats(GotaEnv *handle, uint32_t mask); +uint32_t gota_learner_seats(GotaEnv *handle); +int gota_seat_team(GotaEnv *handle, int seat); /* 0 red, 1 blue */ +int gota_seat_class(GotaEnv *handle, int seat); /* drafted HeroClass, -1 */ +int gota_decision_period(GotaEnv *handle); + +/* Writes the paused decision tick's observation for every seat whose bit is + * set (learner or scripted: scripted rows are the BC inputs), at row seat of + * obs[GOTA_SEATS][GOTA_OBS_SIZE]; unselected rows untouched. resets[seat] = + * 1 when a recurrent actor must zero its state before this inference (first + * decision of the episode; first alive decision after a death), else 0. + * acting[seat] = 1 when the seat is alive at this decision frame and the + * episode is running: for a learner the step will execute its action (0 = + * dead/respawning, action ignored; a hosted seat does not run its net then); + * for a scripted seat it marks a valid BC row. Read only. */ +int gota_observe_seats(GotaEnv *handle, uint32_t seats, float *obs, float *resets, + float *acting); + +/* actions[GOTA_SEATS][GOTA_HEADS]; only learner rows are read. Executes one + * decision and advances decision_period ticks (fewer when the match ends). + * rewards[seat] = change in the seat's league score / 1000 over the step + * (score = lifetime XP - 200 per battle minute, floored at 0, incl. the god + * kill bonus); terminals[seat] = 1 on the step that ends the episode (god + * destroyed or max_ticks). Both buffers are float[GOTA_SEATS] and may be NULL. + * Returns 0 paused at the next decision, 1 episode over (call gota_reset; + * gota_observe_seats then reports the final frame with acting = 0), + * -2 already over, -1 bad args, -3 internal error (gota_last_error): the + * episode is then in an undefined state and the caller must gota_reset + * before stepping again. */ +int gota_step(GotaEnv *handle, const int32_t *actions, float *rewards, + float *terminals); + +/* float[8] = {over, winner (0 red, 1 blue, -1 none), draw, time_limit, + * battle_ticks, red_god_hp, blue_god_hp, total_ticks}. */ +int gota_results(GotaEnv *handle, float *eight); +uint64_t gota_state_hash(GotaEnv *handle); /* the replay hash of the current tick */ +int32_t gota_battle_tick(GotaEnv *handle); + +/* Training-only per-seat counters, cumulative since gota_reset; reading them + * never changes the world. int64[GOTA_STAT_COUNT] in this order: + * 0 score (league score now) 1 outcome (+1 won, -1 lost, 0 running/draw) + * 2 xp (lifetime XP) 3 gold (gold earned, all sources) + * 4 hero_kills 5 assists + * 6 deaths 7 last_hits (lane creep killing blows) + * 8 neutral_kills (killing blows) 9 tower_damage (HP removed from enemy + * towers and barracks) + * 10 structure_kills (towers + barracks destroyed, killing blow) + * 11 hero_damage (HP removed from enemy heroes) + * 12 damage_taken (HP lost, all sources) + * 13 push_depth (milli-tiles: deepest team-frame advance past the map centre + * line while alive, max(0, .); non-decreasing) + * 14 god_damage (HP removed from the enemy god) + * 15 level 16 gold_now 17 team 18 class 19 alive + * 20 decisions (acting learner decisions) 21 invalid_actions (decoded noop + * for an invalid choice) 22 basic_max_instructions 23 reserved (0) + * Terms 0..14 are goal terms 0..14 (w_score..w_god_damage). */ +int gota_seat_stats(GotaEnv *handle, int seat, int64_t *out); + +/* Goal vector w[16] in [-1, 1], w[15] must be 0; appended as the last 16 obs + * floats of the seat from the next observation on. Kept across gota_reset. + * Default: w_score = 1, others 0; a package seat (gota_set_seat_package) + * keeps its manifest goal for its team unless this was called for the seat. + * -3 = out of range. */ +int gota_set_seat_goal(GotaEnv *handle, int seat, const float *w); + +/* Makes the seat SCRIPTED with this BASIC source (removes it from the learner + * mask). Compiled now under the production limits and re-instantiated on + * every gota_reset. length 0 restores the default script. Takes effect at the + * next gota_reset. 0 ok, 1 compile failed (seat idles, as hosted). + * gota_set_policy_script replaces the learner glue script for all learner + * seats (length 0 = default neural/policy.bas); 0 ok, 1 compile failed. */ +int gota_set_seat_script(GotaEnv *handle, int seat, const char *source, int32_t length); +int gota_set_policy_script(GotaEnv *handle, const char *source, int32_t length); +/* 0 learner, 1 scripted and running, 2 compile failed, 3 disabled by a runtime + * error. Copies the error text when message/capacity given. */ +int gota_seat_script_status(GotaEnv *handle, int seat, char *message, int32_t capacity); + +/* BC labels. For a scripted seat: the contract encoding of the commands its + * script issued during the last gota_step's ticks (the window that started at + * the frame gota_observe_seats reported before that step; available after the + * step returns), int32[GOTA_ORDER_SIZE]: + * 0 labeled (1 if a contract command was issued, else 0 = noop label) + * 1..5 head indices {verb, target, point, ability, item} (noop: 0s) + * 6 exact (1 when the decoder reproduces the script's command exactly: same + * verb, object id, ability, slot and destination tile; 0 approximate) + * 7 raw kind (0 none, 1 walkTo, 2 attackMove, 3 attackTarget, 4 castTarget, + * 5 castPoint, 6 useItem, 7 useItemAt) + * 8 raw object id 9 raw ability 10 raw item slot + * 11 raw x milli-tiles, 12 raw y milli-tiles (team frame) + * 13 commands issued during the step (contract kinds only) + * 14 decode error in milli-tiles (point verbs), 15 tick of the chosen command + * Selection when several commands were issued in one step: the first cast or + * item use (they are instantaneous), else the last movement/attack order. + * For a learner seat: its own decoded action (labeled = verb != 0). + * The encoding is computed against the decision frame of the step (the frame + * gota_observe_seats reported before it). */ +int gota_seat_orders(GotaEnv *handle, int seat, int32_t *out); + +/* Mapping ceiling (tools/mapping_ceiling.py). With enabled = 1 a SCRIPTED + * seat's contract commands (walkTo .. useItemAt) are not executed: they are + * encoded as above and the DECODED command is executed on the next decision + * tick instead, exactly as a learner seat would issue it. Non-contract calls + * (draft, buyItem, levelAbility, buyback, chat) run normally. Intercepted + * calls return 1 and leave lastActionError unchanged. Kept across reset. + * 0 = off (default, byte-identical). */ +int gota_set_seat_override(GotaEnv *handle, int seat, int32_t enabled); + +/* DAgger shadow expert. On a LEARNER seat, runs this .bas every tick on the + * learner's own hero and frames, with every host call that would change the + * world absorbed (contract commands, buyItem, levelAbility, buyback, draft, + * chat, mailbox reads): nothing it does executes; the caller's action still + * does. Its contract commands are labeled exactly as a scripted seat's and + * gota_seat_orders(seat) then returns those labels instead of the learner's + * own action. Compiled now (0 ok, 1 compile failed), instantiated at every + * gota_reset; length 0 = off (default; byte-identical). */ +int gota_set_seat_shadow(GotaEnv *handle, int seat, const char *source, int32_t length); + +/* Defer script: verb 0 becomes DEFER on this LEARNER seat. + * The seat's BASIC program becomes the script at script_path (relative paths + * resolve against data_root; NULL or "" = off, the default, byte-identical), + * compiled under the neural-seat structure limits with the plain-seat + * per-tick budget (20k instructions; the consult costs none), and run every tick on the seat's + * own hero as its real program: it observes the true world, drafts, shops, + * levels abilities and buys back itself (these non-contract calls always + * execute), and it keeps its own persistent variables. policy.bas glue is + * not used on this seat. + * Decision windows. At each decision tick, before the seat's BASIC runs, the + * learner's action is consulted once: + * verb head == 0 (DEFER): every contract command (walkTo .. useItemAt) the + * script issues during this window (the decision tick and the following + * decision_period - 1 ticks) executes live, at the tick the script issues + * it, exactly as on a plain scripted seat. If it issues none, nothing is + * issued and the engine keeps the held order. + * verb head != 0 (OVERRIDE): the decoded learner command is issued on the + * decision tick (invalid choices issue nothing, counted in stats[21]) and + * every contract command the script issues during the window is absorbed + * shadow-style: not executed, returns 1, lastActionError unchanged. + * Shadow state under overrides: the script is never paused or re-run; it + * reads the true world every tick, so after an override window it sees the + * hero where the learner's command put it, but its own variables may still + * assume its absorbed orders ran (e.g. "already sent attackTarget(x)") and + * base.bas mostly issues orders only when they change, so after an override + * the learner's order stays held until the script issues a new one. + * A seat that is dead / not acting at the decision frame defers (its action + * is ignored, as always). An always-defer learner plays byte-identical (state + * hash, replay commands) to a plain seat running the same script. + * gota_seat_orders(seat) reports the script's contract commands of the last + * window (issued or absorbed) as labels, like a shadow expert; setting a + * defer script clears the seat's gota_set_seat_shadow expert and vice versa. + * Hosted equivalent: package manifest "decoder": {"defer_script": true}, with + * policy.bas = the script (e.g. base.bas verbatim); identical semantics, the + * host consults the network at the same point. Seats without the option keep + * verb 0 = noop. Takes effect at the next gota_reset. 0 ok, 1 compile failed, + * -3 unreadable file (gota_last_error), -1 bad args. */ +int gota_set_seat_defer_script(GotaEnv *handle, int seat, const char *script_path); +/* int64[2] = {defer decisions, override decisions} since gota_reset (acting + * decisions only; zeros for seats without the defer option). */ +int gota_seat_defer_stats(GotaEnv *handle, int seat, int64_t *out); + +/* Action validity mask for the paused decision frame of any seat (learner, + * scripted or package), uint8[GOTA_MASK_SIZE], 1 = allowed: + * [GOTA_MASK_VERB + v] v in 0..7. Verb 0 always 1; 1, 2, 6 are 1 while + * alive; 3, 4, 5, 7 are 1 iff their target row + * (for 4: some ability row) has an allowed slot. + * [GOTA_MASK_ABILITY + a] castTarget with ability a has an allowed target. + * [GOTA_MASK_TARGET + r*25 + slot] target rows r: + * 0 attackTarget: occupied, not self, a living enemy the hero may attack + * (the engine's attack validator: enemy hero/creep/structure/god, a + * fighting or resting neutral camp); + * 1..4 castTarget with ability 0..3: occupied, and the ability is + * self-cast or the object is a living, visible spell target of the right + * faction (Strike: not own; heal/buff: own); + * 5 castPoint, 6 useItemAt: occupied (the point anchor). + * Verbs 0, 1, 2, 6 ignore the target head (all slots allowed). Dead / not + * acting: only verb 0 is allowed. Range, cooldown, mana and charges are NOT + * masked (the engine rejects those as action errors, not decode noops). + * Conditional use (trainer and host): mask verb; then ability by + * [GOTA_MASK_ABILITY] if verb == 4, else unmasked; then target by the row + * of (verb, ability); point and item heads are never masked. A decision that + * follows the mask never decodes to an invalid noop (stats[21] stays 0). + * Static mode (for trainers that sample heads independently with one mask + * per head): verb mask as above, ability unmasked, target mask = the union of + * the rows of the allowed target-reading verbs (3; 4 all abilities; 5; 7). + * It removes "no valid target at all" invalids but not cross-verb ones (a + * slot valid for castTarget sampled with attackTarget). Package key + * "mask_mode": "static" (default "conditional"; needs mask_empty_targets). + * Hosted equivalent: manifest "decoder": {"mask_empty_targets": true}: the + * host computes this mask from the same frame and applies it before argmax or + * sampling, in the order above (sampling still draws one uniform per head in + * head order, verb..item, so RNG use is unchanged). Default off, + * byte-identical. For a package seat with the option this returns the mask + * the host applied. 0 ok, -1 bad args / not paused. */ +#define GOTA_MASK_VERB 0 +#define GOTA_MASK_ABILITY 8 +#define GOTA_MASK_TARGET 12 +#define GOTA_MASK_TARGET_ROWS 7 +#define GOTA_MASK_SIZE 187 /* 8 + 4 + 7 * 25 */ +int gota_action_mask(GotaEnv *handle, int seat, uint8_t *out); + +/* Replays and diagnostics (config "record": true records every tick's hash + * and command; "capture": false turns BC labeling of scripted seats off for + * speed). gota_save_replay writes the replay (0, -1 not recording). + * gota_last_error copies the calling thread's last -3 error text: the error + * is thread-local, so call it on the thread whose call failed, before that + * thread makes another call. */ +int gota_save_replay(GotaEnv *handle, const char *path); +int gota_last_error(char *message, int32_t capacity); + +/* Hosted-seat parity: installs a neural package (ZIP bytes: manifest.json, + * policy.bas, model.bin) on the seat; the seat then runs exactly as on the + * hosted server (network inside the library, argmax or manifest sampling). + * Makes the seat scripted from the caller's point of view (its actions are + * ignored). 0 ok, 1 compile failed, 2 package rejected (reason via + * gota_seat_script_status). Takes effect at the next gota_reset. */ +int gota_set_seat_package(GotaEnv *handle, int seat, const void *zip, int64_t length); + +/* Neural actor (neural_actor.nim), bit-identical to the hosted seat, so a + * trainer can check its FP32 actor against the deployed one bit for bit. + * model.bin format GOTANET1 (docs/neural-policies.md). gota_net_load refuses + * models over 4,000,000 operations per inference and models whose header + * contract hashes, input size or head sizes are not GotA contract v1; NULL + * with the reason on failure. + * gota_net_info: int64[8] = {format, inputs, hidden, outputs, heads, + * state_floats, parameters, operations}. gota_net_infer: observation[inputs], + * state[state_floats] updated in place, logits[outputs]; 0, -2 nonfinite + * (reason via gota_last_error). */ +GotaNet *gota_net_load(const void *data, int64_t length, char *error, int32_t capacity); +void gota_net_destroy(GotaNet *net); +int gota_net_info(GotaNet *net, int64_t *eight); +int gota_net_infer(GotaNet *net, const float *observation, float *state, float *logits); + +#ifdef __cplusplus +} +#endif +#endif diff --git a/examples/gods_of_the_arena/native_env.nim b/examples/gods_of_the_arena/native_env.nim new file mode 100644 index 00000000..3d50c201 --- /dev/null +++ b/examples/gods_of_the_arena/native_env.nim @@ -0,0 +1,812 @@ +## Gods of the Arena native training environment (native_env.h, ABI v1). +## +## Build (shared library, training counters on, handles usable from any +## thread): +## nim c --app:lib -d:release -d:headless -d:gotaTrainingStats \ +## --mm:atomicArc --threads:on -d:useMalloc -u:nimTypeNames \ +## -o:libgota_env.so examples/gods_of_the_arena/native_env.nim +## (--mm:orc instead of atomicArc is ~10% faster single-threaded but then +## every handle of the process must stay on one thread.) +## One handle = one ten-seat match. See native_env.h and neural_basic.md. + +import std/[json, locks, os], jsony, scores, polyworld/visions +include bots + +const + GotaEnvVersion = 1 + EnvMagic = 0x474F5441454E5631'u64 ## "GOTAENV1" + NetMagic = 0x474F54414E455431'u64 ## "GOTANET1" + StatCount = 24 + OrderSize = 16 + DataRoot = currentSourcePath().parentDir + +type + SeatSource = enum SourceDefault, SourceScript, SourcePackage + SeatStatus = object + code: int ## 0 learner, 1 running, 2 compile failed, 3 runtime disabled + message: string + Env = ref object + magic: uint64 + ## EnvMagic while the handle is live (toEnv rejects anything else). + config: GotaConfig + maxTicks, period: int32 + learners: uint32 + record, capture: bool + defaultScript, policyScript: string + sources: array[10, SeatSource] + scripts: array[10, string] + packages: array[10, string] + overrides: array[10, bool] + shadows: array[10, string] + defers: array[10, string] + ## The learner seat's defer script (empty = off). + root: string + goals: array[10, array[GoalSize, float32]] + goalSet: array[10, bool] + ## gota_set_seat_goal was called: overrides a package seat's manifest goal. + status: array[10, SeatStatus] + notes: array[10, string] + ## The last rejected configuration of each seat (kept across resets, + ## cleared when the seat is configured successfully). + game: Game + seeds: int64 + paused, over, started: bool + prevScore: array[10, int64] + pushDepth: array[10, int64] + scratch: seq[float32] + +var + presetKey: string + lastError {.threadvar.}: string + processLock: Lock + ## Serializes create/reset: the shared map, navigation graph and lane + ## globals are built once under it; stepping never writes them. + mapTemplate: MapData + mapReady: bool +initLock(processLock) + +proc copyText(text: string, buffer: ptr char, capacity: int32) = + if buffer == nil or capacity <= 0: + return + let n = min(text.len, int(capacity) - 1) + let dest = cast[ptr UncheckedArray[char]](buffer) + for i in 0 ..< n: + dest[i] = text[i] + dest[n] = '\0' + +proc seatScore(game: Game, index: int): int64 = + int64(score(game.world.heroes[index].totalXp, + int(max(0'i32, game.world.battleTick())))) + +proc compileSeat(env: Env, game: Game, index: int, source: string, + neural: bool, deferring = false): bool = + ## Compiles one seat's program; a failure leaves the seat idle. + let heroId = game.world.heroes[index].id + let limits = if deferring: deferVmLimits() + elif neural: neuralVmLimits() else: heroVmLimits() + var schema = initHeroHost(0) + var host = initHeroHost(heroId) + if neural: + schema.addNeuralSeatFunctions(0) + host.addNeuralSeatFunctions(heroId) + try: + let program = compile(source, schema, limits) + bindHeroData(program) + game.heroVms[index] = HeroVm(runtime: initRuntime(program, host, limits), + limits: limits, ready: true) + result = true + except BasicError as error: + env.status[index] = SeatStatus(code: 2, message: error.msg) + game.heroVms[index] = nil + +proc updatePush(env: Env) = + ## Deepest advance past the centre line along own god -> enemy god. + let world = env.game.world + for i, hero in world.heroes: + if hero.hp <= 0 or hero.state == Dying: + continue + let + own = world.forts[hero.team.ord].center + enemy = world.forts[1 - hero.team.ord].center + ax = float64(enemy.x - own.x) + az = float64(enemy.z - own.z) + length = sqrt(ax*ax + az*az) + mx = float64(own.x + enemy.x) / 2 + mz = float64(own.z + enemy.z) / 2 + if length <= 0: + continue + let depth = ((float64(hero.position.x) - mx) * ax + + (float64(hero.position.z) - mz) * az) / length + let milli = int64(depth * 1000 / float64(WorldScale)) + env.pushDepth[i] = max(env.pushDepth[i], milli) + +proc tickDone(env: Env) = + ## Per-tick replay telemetry, exactly as the headless runner samples it. + if env.record: + let game = env.game + game.sampleMetrics(game.finished()) + game.metrics.finishTick(game.world.tick) + +proc pauseOrFinish(env: Env) = + ## Runs ticks until the next decision tick's heroes' turn (paused, with the + ## observation frame frozen) or the end of the match. + let game = env.game + activeGame = game + env.paused = false + while true: + if game.finished(): + env.over = true + return + let stage = game.tickWorldBegin(proc() = runBotDecisions(game)) + case stage + of TickSkipped: + env.over = true + return + of TickDone: + env.tickDone() + of TickNoTurn: + game.tickWorldFinish() + env.tickDone() + of TickHeroTurn: + if game.world.isDecisionTick(env.period): + # Freeze the common observation frame BEFORE the neural seats observe, + # exactly as runBotDecisions does on the hosted server (the frozen + # frame orders spell warnings by observation key; the live list does + # not). gota_step thaws it after the heroes' turn. + discard game.world.freezeObservations() + game.neuralPrelude() + env.paused = true + env.updatePush() + return + runBotDecisions(game) + game.tickWorldFinish() + env.tickDone() + +proc resetEnv(env: Env, seed: int64): int = + activeGame = nil + neuralTelemetryEnabled = false + var gameMap: MapData + var game: Game + withLock processLock: + if not mapReady: + mapTemplate = generateMap(int32(seed), env.config.mapPreset) + warmEdgeLinks() + initVisionKernel() + mapReady = true + gameMap = mapTemplate + gameMap.seed = int32(seed) + game = newGame(gameMap, env.config.spawnIntervalTicks, 10, false, + ReplayData(), true) + game.replayData = initReplayData(currentSetup(game, uint32(env.maxTicks)), + gameMap.preset) + if env.record: + game.recorder = initReplayRecorder(currentSetup(game, uint32(env.maxTicks)), + gameMap.preset) + game.recorder.data.config = game.replayData.config + activeGame = game + env.game = game + game.heroVms.setLen(game.world.heroes.len) + game.inboxes.setLen(game.world.heroes.len) + for inbox in game.inboxes.mitems: + inbox = newMailbox() + result = 0 + for i in 0 ..< 10: + env.status[i] = SeatStatus(code: 1, message: env.notes[i]) + env.prevScore[i] = 0 + env.pushDepth[i] = 0 + let learner = (env.learners and (1'u32 shl i)) != 0 + if env.sources[i] == SourcePackage: + try: + game.installPackageSeat(i, env.packages[i]) + let seat = game.neuralSeat(i) + seat.telemetry = false + if env.goalSet[i]: + seat.goal = env.goals[i] # else the manifest goal of the seat's team + except BasicError as error: + env.status[i] = SeatStatus(code: 2, message: error.msg) + game.heroVms[i] = nil + result = -2 + continue + if learner: + env.status[i] = SeatStatus(code: 0) + let deferring = env.defers[i].len > 0 + let program = if deferring: env.defers[i] else: env.policyScript + if env.compileSeat(game, i, program, true, deferring): + let seat = newNeuralSeat(NeuralLearner, env.period, env.maxTicks) + seat.deferEnabled = deferring + seat.goal = env.goals[i] + seat.resetEpisode(int32(seed), i) + game.heroVms[i].neural = seat + if env.shadows[i].len > 0 and not deferring: + let heroId = game.world.heroes[i].id + try: + let program = compile(env.shadows[i], initHeroHost(0), heroVmLimits()) + seat.shadow = HeroVm(runtime: initRuntime(program, + initHeroHost(heroId), heroVmLimits()), limits: heroVmLimits(), + ready: true) + except BasicError as error: + env.status[i] = SeatStatus(code: 2, message: "shadow: " & error.msg) + result = -2 + else: + result = -2 + continue + let source = + if env.sources[i] == SourceScript: env.scripts[i] else: env.defaultScript + if env.compileSeat(game, i, source, false): + if env.overrides[i] or env.capture: + let seat = newNeuralSeat( + if env.overrides[i]: NeuralOverride else: NeuralCapture, + env.period, env.maxTicks) + seat.goal = env.goals[i] + seat.resetEpisode(int32(seed), i) + game.heroVms[i].neural = seat + else: + result = -2 + if result == -2: + var failures: string + for i in 0 ..< 10: + if env.status[i].code == 2: + if failures.len > 0: + failures.add "; " + failures.add "seat " & $i & ": " & env.status[i].message + lastError = "seat compilation failed: " & failures + env.over = false + env.started = true + # Draft (BASIC picks), then the first battle decision. + while game.world.phase == Drafting and not game.finished(): + tickWorld(game, proc() = runBotDecisions(game)) + env.tickDone() + env.pauseOrFinish() + +type NetHandle = ref object + magic: uint64 + actor: Actor + +proc toEnv(handle: pointer): Env = + ## The live env behind a handle; nil for nil, destroyed or foreign handles. + if handle == nil: + return nil + result = cast[Env](handle) + if result.magic != EnvMagic: + lastError = "not a live gota env handle" + return nil + if result.game != nil: + activeGame = result.game + +# --------------------------------------------------------------------------- +# C ABI + +proc gota_env_version(): cint {.exportc, dynlib, cdecl.} = GotaEnvVersion +proc gota_observation_size(): cint {.exportc, dynlib, cdecl.} = ObservationSize +proc gota_goal_size(): cint {.exportc, dynlib, cdecl.} = GoalSize +proc gota_stat_count(): cint {.exportc, dynlib, cdecl.} = StatCount +proc gota_mask_size(): cint {.exportc, dynlib, cdecl.} = MaskSize + +proc gota_action_heads(sizes: ptr UncheckedArray[int32]): cint {.exportc, dynlib, cdecl.} = + if sizes != nil: + for i, size in HeadSizes: + sizes[i] = int32(size) + ActionHeads + +proc gota_observation_contract_hash(output: ptr char, capacity: int32): cint {.exportc, dynlib, cdecl.} = + if output == nil or capacity < 65: return -1 + copyText(ObservationContractHash, output, capacity) + +proc gota_action_contract_hash(output: ptr char, capacity: int32): cint {.exportc, dynlib, cdecl.} = + if output == nil or capacity < 65: return -1 + copyText(ActionContractHash, output, capacity) + +proc resolve(root, path: string): string = + if path.isAbsolute: path else: root / path + +proc gota_create(configJson: cstring, error: ptr char, capacity: int32): pointer {.exportc, dynlib, cdecl.} = + ## Thread-safe: the whole create (config parse, file reads, the process-wide + ## preset check) runs under processLock, so concurrent creates serialize. + try: + withLock processLock: + let node = if configJson == nil or configJson[0] == '\0': newJObject() + else: parseJson($configJson) + for key in node.keys: + if key notin ["config_path", "seed", "max_ticks", "decision_period", + "learner_seats", "script_path", "policy_path", "data_root", + "record", "capture"]: + raise newException(ValueError, "unknown config key " & key) + let root = if node.hasKey("data_root"): node["data_root"].getStr else: DataRoot + let env = Env(period: DefaultDecisionPeriod, capture: true, root: root) + env.config = + if node.hasKey("config_path"): loadConfig(resolve(root, node["config_path"].getStr)) + else: parseConfig("{}") + env.maxTicks = int32(if node.hasKey("max_ticks"): node["max_ticks"].getInt + else: 28_800) + if env.maxTicks notin 1'i32 .. 28_800'i32: + raise newException(ValueError, "max_ticks must be 1..28800") + if node.hasKey("decision_period"): + env.period = int32(node["decision_period"].getInt) + if env.period notin 1'i32 .. 24'i32: + raise newException(ValueError, "decision_period must be 1..24") + env.learners = 1 + if node.hasKey("learner_seats"): + env.learners = 0 + for seat in node["learner_seats"]: + let s = seat.getInt + if s notin 0..9: + raise newException(ValueError, "learner seat out of range") + env.learners = env.learners or (1'u32 shl s) + env.record = node.hasKey("record") and node["record"].getBool + if node.hasKey("capture"): + env.capture = node["capture"].getBool + env.defaultScript = readFile(resolve(root, + if node.hasKey("script_path"): node["script_path"].getStr else: "players/base.bas")) + env.policyScript = readFile(resolve(root, + if node.hasKey("policy_path"): node["policy_path"].getStr else: "neural/policy.bas")) + env.seeds = if node.hasKey("seed"): node["seed"].getInt else: 0 + for i in 0 ..< 10: + env.goals[i] = defaultGoal() + let key = env.config.mapPreset.toJson() + if presetKey.len > 0 and presetKey != key: + raise newException(ValueError, "every handle in a process must use the same map preset") + presetKey = key + env.magic = EnvMagic + GC_ref(env) + result = cast[pointer](env) + except CatchableError as e: + copyText(e.msg, error, capacity) + result = nil + +proc gota_destroy(handle: pointer) {.exportc, dynlib, cdecl.} = + let env = toEnv(handle) + if env == nil: return + env.magic = 0 + if activeGame == env.game: + activeGame = nil + env.game = nil + GC_unref(env) + +proc gota_reset(handle: pointer, seed: int64): cint {.exportc, dynlib, cdecl.} = + let env = toEnv(handle) + if env == nil: return -1 + try: + env.seeds = seed + cint(env.resetEnv(seed)) + except CatchableError as e: + lastError = e.msg + -3 + +proc gota_set_learner_seats(handle: pointer, mask: uint32): cint {.exportc, dynlib, cdecl.} = + let env = toEnv(handle) + if env == nil or mask >= 1024: return -1 + env.learners = mask + for i in 0 ..< 10: + if (mask and (1'u32 shl i)) != 0 and env.sources[i] != SourceDefault: + env.sources[i] = SourceDefault + 0 + +proc gota_learner_seats(handle: pointer): uint32 {.exportc, dynlib, cdecl.} = + let env = toEnv(handle) + if env == nil: 0'u32 else: env.learners + +proc gota_seat_team(handle: pointer, seat: cint): cint {.exportc, dynlib, cdecl.} = + if seat notin 0..9: return -1 + if seat < 5: 0 else: 1 + +proc gota_seat_class(handle: pointer, seat: cint): cint {.exportc, dynlib, cdecl.} = + let env = toEnv(handle) + if env == nil or env.game == nil or seat notin 0..9: return -1 + env.game.world.draftedClass(env.game.world.heroes[seat].id) + +proc gota_decision_period(handle: pointer): cint {.exportc, dynlib, cdecl.} = + let env = toEnv(handle) + if env == nil: -1 else: env.period + +proc gota_battle_tick(handle: pointer): int32 {.exportc, dynlib, cdecl.} = + let env = toEnv(handle) + if env == nil or env.game == nil: -1 else: env.game.world.battleTick() + +proc gota_observe_seats(handle: pointer, seats: uint32, + obs, resets, acting: ptr UncheckedArray[float32]): cint {.exportc, dynlib, cdecl.} = + let env = toEnv(handle) + if env == nil or env.game == nil or obs == nil: return -1 + let game = env.game + for i in 0 ..< 10: + if (seats and (1'u32 shl i)) == 0: + continue + let row = cast[ptr UncheckedArray[float32]](addr obs[i * ObservationSize]) + let seat = game.neuralSeat(i) + if seat != nil and env.paused and seat.frameTick == game.world.tick: + for k in 0 ..< ObservationSize: + row[k] = seat.obs[k] + if resets != nil: resets[i] = float32(seat.resetState and seat.acting) + if acting != nil: + acting[i] = float32(seat.acting and not env.over) + else: + if env.scratch.len != ObservationSize: + env.scratch = newSeq[float32](ObservationSize) + var frame: DecisionFrame + let goal = if seat != nil: seat.goal else: env.goals[i] + buildObservation(game.world, i, goal, env.maxTicks, + game.world.stats, env.scratch, frame) + for k in 0 ..< ObservationSize: + row[k] = env.scratch[k] + if resets != nil: resets[i] = 0 + if acting != nil: acting[i] = 0 + 0 + +proc gota_step(handle: pointer, actions: ptr UncheckedArray[int32], + rewards, terminals: ptr UncheckedArray[float32]): cint {.exportc, dynlib, cdecl.} = + let env = toEnv(handle) + if env == nil or env.game == nil: return -1 + if env.over: return -2 + if not env.paused: return -1 + let game = env.game + try: + for i in 0 ..< 10: + env.prevScore[i] = game.seatScore(i) + let seat = game.neuralSeat(i) + if seat != nil and seat.mode == NeuralLearner: + var heads: Heads + if actions != nil: + for h in 0 ..< ActionHeads: + heads[h] = actions[i * ActionHeads + h] + seat.setLearnerHeads(heads) + runBotDecisions(game) # the frame is already frozen: it does not thaw it + game.world.thawObservations() + game.tickWorldFinish() + env.tickDone() + env.pauseOrFinish() + except CatchableError as e: + lastError = e.msg + return -3 + for i in 0 ..< 10: + let now = game.seatScore(i) + if rewards != nil: + rewards[i] = float32(now - env.prevScore[i]) / 1000 + if terminals != nil: + terminals[i] = float32(env.over) + if env.over: 1 else: 0 + +proc gota_results(handle: pointer, eight: ptr UncheckedArray[float32]): cint {.exportc, dynlib, cdecl.} = + let env = toEnv(handle) + if env == nil or env.game == nil or eight == nil: return -1 + let world = env.game.world + eight[0] = float32(env.over) + eight[1] = if world.gameOver and not world.draw: float32(world.winner.ord) else: -1 + eight[2] = float32(world.draw) + eight[3] = float32(env.over and not world.gameOver) + eight[4] = float32(world.battleTick()) + eight[5] = float32(max(world.forts[0].hp, 0)) + eight[6] = float32(max(world.forts[1].hp, 0)) + eight[7] = float32(world.tick) + 0 + +proc gota_state_hash(handle: pointer): uint64 {.exportc, dynlib, cdecl.} = + let env = toEnv(handle) + if env == nil or env.game == nil: 0'u64 else: env.game.stateHash() + +proc gota_seat_stats(handle: pointer, seat: cint, output: ptr UncheckedArray[int64]): cint {.exportc, dynlib, cdecl.} = + let env = toEnv(handle) + if env == nil or env.game == nil or seat notin 0..9 or output == nil: return -1 + let + game = env.game + world = game.world + hero = world.heroes[seat] + values = world.stats.values[seat] + for i in 0 ..< StatCount: + output[i] = 0 + output[0] = game.seatScore(seat) + if world.gameOver and not world.draw: + output[1] = if world.winner == hero.team: 1 else: -1 + output[2] = int64(hero.totalXp) + output[4] = values[KillsMetric] + output[5] = values[AssistsMetric] + output[6] = int64(hero.deaths) + when defined(gotaTrainingStats): + if world.training.len > seat: + let t = world.training[seat] + output[3] = t[TrainGold] + output[7] = t[TrainLastHits] + output[8] = t[TrainNeutralKills] + output[9] = t[TrainTowerDamage] + output[10] = t[TrainStructureKills] + output[11] = t[TrainHeroDamage] + output[12] = t[TrainDamageTaken] + output[14] = t[TrainGodDamage] + else: + output[3] = values[GoldMetric] + output[13] = env.pushDepth[seat] + output[15] = int64(hero.level) + output[16] = int64(hero.gold) + output[17] = int64(hero.team.ord) + output[18] = int64(world.draftedClass(hero.id)) + output[19] = int64(hero.hp > 0 and hero.state != Dying) + let s = game.neuralSeat(seat) + if s != nil: + output[20] = int64(s.decisions) + output[21] = int64(s.invalid) + let vm = game.heroVms[seat] + if vm != nil: + output[22] = vm.lastInstructions + 0 + +proc gota_set_seat_goal(handle: pointer, seat: cint, w: ptr UncheckedArray[float32]): cint {.exportc, dynlib, cdecl.} = + let env = toEnv(handle) + if env == nil or seat notin 0..9 or w == nil: return -1 + for i in 0 ..< GoalSize: + if not (w[i] >= -1 and w[i] <= 1): return -3 + if w[GoalSize - 1] != 0: return -3 + for i in 0 ..< GoalSize: + env.goals[seat][i] = w[i] + env.goalSet[seat] = true + if env.game != nil: + let s = env.game.neuralSeat(seat) + if s != nil: + s.goal = env.goals[seat] + for i in 0 ..< GoalSize: + s.obs[ObsGoalOffset + i] = w[i] + 0 + +proc checkCompile(source: string, neural: bool): (bool, string) = + var schema = initHeroHost(0) + if neural: + schema.addNeuralSeatFunctions(0) + try: + discard compile(source, schema, if neural: neuralVmLimits() else: heroVmLimits()) + (true, "") + except BasicError as error: + (false, error.msg) + +proc gota_set_seat_script(handle: pointer, seat: cint, source: cstring, length: int32): cint {.exportc, dynlib, cdecl.} = + let env = toEnv(handle) + if env == nil or seat notin 0..9 or length < 0 or (length > 0 and source == nil): return -1 + env.learners = env.learners and not (1'u32 shl seat) + if length == 0: + env.sources[seat] = SourceDefault + return 0 + var text = newString(length) + copyMem(addr text[0], source, length) + env.scripts[seat] = text + env.sources[seat] = SourceScript + let (ok, message) = checkCompile(text, false) + env.status[seat] = SeatStatus(code: (if ok: 1 else: 2), message: message) + env.notes[seat] = message + if ok: 0 else: 1 + +proc gota_set_policy_script(handle: pointer, source: cstring, length: int32): cint {.exportc, dynlib, cdecl.} = + let env = toEnv(handle) + if env == nil or length < 0 or (length > 0 and source == nil): return -1 + if length == 0: + env.policyScript = readFile(DataRoot / "neural/policy.bas") + return 0 + var text = newString(length) + copyMem(addr text[0], source, length) + let (ok, message) = checkCompile(text, true) + if not ok: + lastError = message + return 1 + env.policyScript = text + 0 + +proc gota_seat_script_status(handle: pointer, seat: cint, message: ptr char, capacity: int32): cint {.exportc, dynlib, cdecl.} = + let env = toEnv(handle) + if env == nil or seat notin 0..9: return -1 + var status = env.status[seat] + if env.game != nil: + let vm = env.game.heroVms[seat] + if vm != nil and vm.failed: + status = SeatStatus(code: 3, message: vm.lastError) + elif (env.learners and (1'u32 shl seat)) != 0 and status.code != 2: + status.code = 0 + copyText(status.message, message, capacity) + cint(status.code) + +proc gota_seat_orders(handle: pointer, seat: cint, output: ptr UncheckedArray[int32]): cint {.exportc, dynlib, cdecl.} = + let env = toEnv(handle) + if env == nil or env.game == nil or seat notin 0..9 or output == nil: return -1 + for i in 0 ..< OrderSize: + output[i] = 0 + let s = env.game.neuralSeat(seat) + if s == nil: + return 0 + case s.mode + of NeuralCapture, NeuralOverride: + for i in 0 ..< OrderSize: + output[i] = s.lastLabel[i] + of NeuralLearner, NeuralHosted: + if s.shadow != nil or (s.deferEnabled and s.mode == NeuralLearner): + for i in 0 ..< OrderSize: + output[i] = s.lastLabel[i] + return 0 + output[0] = int32(s.headsReady and s.acting and s.heads[0] != 0) + for h in 0 ..< ActionHeads: + output[1 + h] = s.heads[h] + output[6] = 1 + output[7] = int32(s.command.kind.ord) + output[8] = s.command.objectId + output[9] = s.command.ability + output[10] = s.command.item + output[15] = s.frameTick + 0 + +proc gota_set_seat_shadow(handle: pointer, seat: cint, source: cstring, length: int32): cint {.exportc, dynlib, cdecl.} = + ## DAgger: on a learner seat, runs `source` on the same frames without + ## executing anything; gota_seat_orders then reports its labels. + let env = toEnv(handle) + if env == nil or seat notin 0..9 or length < 0 or (length > 0 and source == nil): return -1 + if length == 0: + env.shadows[seat] = "" + return 0 + var text = newString(length) + copyMem(addr text[0], source, length) + let (ok, _) = checkCompile(text, false) + if not ok: return 1 + env.shadows[seat] = text + env.defers[seat] = "" + 0 + +proc gota_set_seat_defer_script(handle: pointer, seat: cint, path: cstring): cint {.exportc, dynlib, cdecl.} = + ## Defer script: verb 0 defers to this script on a learner seat + ## (native_env.h). NULL or "" = off. + let env = toEnv(handle) + if env == nil or seat notin 0..9: return -1 + if path == nil or path[0] == '\0': + env.defers[seat] = "" + return 0 + var text: string + try: + text = readFile(resolve(env.root, $path)) + except CatchableError as e: + lastError = e.msg + return -3 + if text.len == 0: + lastError = "defer script is empty: " & $path + return -3 + let (ok, message) = checkCompile(text, true) + if not ok: + lastError = message + return 1 + env.defers[seat] = text + env.shadows[seat] = "" + 0 + +proc gota_action_mask(handle: pointer, seat: cint, output: ptr UncheckedArray[uint8]): cint {.exportc, dynlib, cdecl.} = + ## Validity mask of the paused decision frame (native_env.h). + let env = toEnv(handle) + if env == nil or env.game == nil or seat notin 0..9 or output == nil or + not env.paused: return -1 + let game = env.game + let s = game.neuralSeat(seat) + var mask: ActionMask + if s != nil and s.frameTick == game.world.tick: + mask = if s.mode == NeuralHosted and s.maskMode != NoMask: s.mask + else: actionMask(game.world, seat, s.frame) + if not s.acting or env.over: + mask = default(ActionMask) + mask[MaskVerb] = 1 + else: + if env.scratch.len != ObservationSize: + env.scratch = newSeq[float32](ObservationSize) + var frame: DecisionFrame + buildObservation(game.world, seat, env.goals[seat], env.maxTicks, + game.world.stats, env.scratch, frame) + mask = actionMask(game.world, seat, frame) + for i in 0 ..< MaskSize: + output[i] = mask[i] + 0 + +proc gota_seat_defer_stats(handle: pointer, seat: cint, output: ptr UncheckedArray[int64]): cint {.exportc, dynlib, cdecl.} = + let env = toEnv(handle) + if env == nil or seat notin 0..9 or output == nil: return -1 + output[0] = 0 + output[1] = 0 + if env.game != nil: + let s = env.game.neuralSeat(seat) + if s != nil and s.deferEnabled: + output[0] = int64(s.deferDecisions) + output[1] = int64(s.overrideDecisions) + 0 + +proc gota_set_seat_override(handle: pointer, seat: cint, enabled: int32): cint {.exportc, dynlib, cdecl.} = + let env = toEnv(handle) + if env == nil or seat notin 0..9 or enabled notin 0'i32..1'i32: return -1 + env.overrides[seat] = enabled == 1 + 0 + +proc gota_set_seat_package(handle: pointer, seat: cint, zip: pointer, length: int64): cint {.exportc, dynlib, cdecl.} = + let env = toEnv(handle) + if env == nil or seat notin 0..9 or zip == nil or length <= 0 or + length > MaxPackageBytes: return -1 + var bytes = newString(length) + copyMem(addr bytes[0], zip, length) + try: + let package = parseGotaPackage(bytes) + let (ok, message) = checkCompile(package.policy, true) + if not ok: + env.status[seat] = SeatStatus(code: 2, message: message) + env.notes[seat] = message + return 1 + except CatchableError as e: + env.status[seat] = SeatStatus(code: 2, message: "neural package rejected: " & e.msg) + env.notes[seat] = env.status[seat].message + return 2 + env.notes[seat] = "" + env.packages[seat] = bytes + env.sources[seat] = SourcePackage + env.learners = env.learners and not (1'u32 shl seat) + 0 + +proc gota_last_error(output: ptr char, capacity: int32): cint {.exportc, dynlib, cdecl.} = + copyText(lastError, output, capacity) + 0 + +proc gota_save_replay(handle: pointer, path: cstring): cint {.exportc, dynlib, cdecl.} = + ## Writes the recorded replay (config "record": true). 0 ok, -1 not recording. + let env = toEnv(handle) + if env == nil or env.game == nil or env.game.recorder == nil or path == nil: return -1 + try: + if env.game.world.tick == env.game.recorder.data.hashes.len: + env.game.sampleMetrics(true) + env.game.recorder.data.metrics = env.game.history.replayMetrics() + saveReplay($path, env.game.recorder.data) + 0 + except CatchableError as e: + lastError = e.msg + -3 + +# Neural actor. + +proc gota_net_load(data: pointer, length: int64, error: ptr char, capacity: int32): pointer {.exportc, dynlib, cdecl.} = + if data == nil or length <= 0: + copyText("empty model", error, capacity) + return nil + var bytes = newString(length) + copyMem(addr bytes[0], data, length) + try: + let handle = NetHandle(magic: NetMagic, actor: loadActor(bytes, GotaContract)) + GC_ref(handle) + cast[pointer](handle) + except CatchableError as e: + copyText(e.msg, error, capacity) + nil + +proc toNet(net: pointer): Actor = + ## The actor behind a live net handle; nil otherwise. + if net == nil: + return nil + let handle = cast[NetHandle](net) + if handle.magic != NetMagic: + lastError = "not a live gota net handle" + return nil + handle.actor + +proc gota_net_destroy(net: pointer) {.exportc, dynlib, cdecl.} = + if toNet(net) != nil: + let handle = cast[NetHandle](net) + handle.magic = 0 + GC_unref(handle) + +proc gota_net_info(net: pointer, eight: ptr UncheckedArray[int64]): cint {.exportc, dynlib, cdecl.} = + let actor = toNet(net) + if actor == nil or eight == nil: return -1 + eight[0] = 1 + eight[1] = actor.inputSize + eight[2] = actor.hiddenSize + eight[3] = actor.outputSize + eight[4] = actor.headSizes.len + eight[5] = actor.hiddenSize + eight[6] = actor.parameterCount + eight[7] = actor.operationCount + 0 + +proc gota_net_infer(net: pointer, observation, state, logits: ptr UncheckedArray[float32]): cint {.exportc, dynlib, cdecl.} = + let actor = toNet(net) + if actor == nil or observation == nil or state == nil or logits == nil: return -1 + try: + var nextState = newSeq[float32](actor.hiddenSize) + var output = newSeq[float32](actor.outputSize) + for i in 0 ..< actor.hiddenSize: nextState[i] = state[i] + actor.infer(observation.toOpenArray(0, actor.inputSize - 1), nextState, output) + for i in 0 ..< actor.hiddenSize: state[i] = nextState[i] + for i in 0 ..< actor.outputSize: logits[i] = output[i] + 0 + except CatchableError as e: + lastError = e.msg + -2 diff --git a/examples/gods_of_the_arena/neural/policy.bas b/examples/gods_of_the_arena/neural/policy.bas new file mode 100644 index 00000000..eb3f82af --- /dev/null +++ b/examples/gods_of_the_arena/neural/policy.bas @@ -0,0 +1,157 @@ +' GotA neural policy glue (gota-neural-basic/1). +' BASIC keeps the draft, shopping, ability leveling and buyback, using +' base.bas's own routines verbatim; every in-battle hero command comes from +' the network through gota_act(), which issues the decoded command once on +' each decision tick (every decision_period ticks) and does nothing otherwise. + +dim owned(22) +dim inventorySlot(22) + +sub chooseHero() + if draftTurnId <> selfId then + exit sub + end if + bestClass = -1 + bestScore = -10000 + for candidate = 0 to 9 + if heroAvailable(candidate) then + role = heroRole(candidate) + score = 100 + for player = 0 to draftPlayerCount() - 1 + if draftPlayerTeam(player) = selfTeam then + picked = draftedClass(draftPlayerId(player)) + if picked >= 0 then + if heroRole(picked) = role then + score = score - 100 + end if + end if + end if + next player + if score > bestScore then + bestScore = score + bestClass = candidate + end if + end if + next candidate + if bestClass >= 0 then + accepted = draftHero(bestClass) + actionError = lastActionError() + end if +end sub + +sub learnAbilities() + ' Rank requirements and effects come from the host, not a stat table. + for upgrade = 1 to 4 + if abilityPoints() = 0 then + exit sub + end if + upgradeSlot = -1 + upgradeScore = -1 + for spellSlot = 0 to 3 + rank = abilityLevel(spellSlot) + if rank < abilityMaxLevel(spellSlot) then + if selfLevel >= abilityRequiredLevel(spellSlot) then + if canLevelAbility(spellSlot) then + ' Prefer R, W, E, Q whenever the next rank is legal. + score = spellSlot + if spellSlot = 1 then + score = 2 + elseif spellSlot = 2 then + score = 1 + end if + if score > upgradeScore then + upgradeScore = score + upgradeSlot = spellSlot + end if + end if + end if + end if + next spellSlot + if upgradeSlot < 0 then + exit sub + end if + accepted = levelAbility(upgradeSlot) + actionError = lastActionError() + next upgrade +end sub + +sub buy(id, price, quantity) + if owned(id) >= quantity or budget < price then + exit sub + end if + if owned(id) = 0 and emptySlots = 0 then + exit sub + end if + accepted = buyItem(id) + actionError = lastActionError() + if accepted then + if owned(id) = 0 then + emptySlots = emptySlots - 1 + end if + owned(id) = owned(id) + 1 + budget = budget - price + for boughtSlot = 0 to 5 + if itemId(boughtSlot) = id then + inventorySlot(id) = boughtSlot + end if + next boughtSlot + end if +end sub + +sub shop() + for id = 0 to 22 + owned(id) = 0 + inventorySlot(id) = -1 + next id + emptySlots = 0 + for itemSlot = 0 to 5 + id = itemId(itemSlot) + if id = 0 then + emptySlots = emptySlots + 1 + else + owned(id) = itemCount(itemSlot) + inventorySlot(id) = itemSlot + end if + next itemSlot + if canShop() = 0 then + exit sub + end if + ' base.bas's purchase plan: recovery, travel, equipment, role consumable. + budget = selfGold + buy(8, 100, 1) + buy(1, 30, 2) + buy(21, 100, 2) + buy(22, 45, 2) + if role = 0 or role = 4 then + buy(16, 160, 1) + buy(2, 75, 2) + elseif role = 1 then + buy(19, 180, 1) + buy(4, 40, 2) + else + buy(20, 190, 1) + buy(3, 90, 2) + end if +end sub + +if drafting then + chooseHero() + end +end if + +if selfHp <= 0 then + price = buybackPrice() + if price > 0 and selfGold >= price then + accepted = buyback() + actionError = lastActionError() + end if + end +end if + +role = heroRole(selfClass) +learnAbilities() +if worldTick >= nextShop then + nextShop = worldTick + 6 + shop() +end if +acted = gota_act() diff --git a/examples/gods_of_the_arena/neural_basic.md b/examples/gods_of_the_arena/neural_basic.md new file mode 100644 index 00000000..66b63fea --- /dev/null +++ b/examples/gods_of_the_arena/neural_basic.md @@ -0,0 +1,343 @@ +# GotA neural seats (`gota-neural-basic/1`) + +This document covers the GotA neural package tier and the native training environment that shares its +code. PR #72 added metered `linear/relu/argmax` and `DATA` arrays to BASIC but deferred external +weights. That leaves a pure-BASIC network at about 4k parameters (the 4,096 array elements) and about 9k +multiply-accumulates per decision. A neural package lifts both limits for one seat. It adds a native FP32 +actor, a separate per-seat operation budget, and hash-pinned observation and action contracts. The seat +still plays through the same BASIC host and the same recorded commands, so replays and the simulation do +not change. Plain `.bas` submissions are untouched: without a neural seat, a match is byte-identical to +upstream (see "Evidence"). + +GotA is a client of the shared neural tier: the package parser, the GOTANET1 actor, the operation budget, +the telemetry line, the recurrent-state lifecycle and the neural BASIC functions live in +`src/polyworld/neural_{package,actor,host}.nim`. How a game plugs in, the package and model formats and the +budget are described in [docs/neural-policies.md](../../docs/neural-policies.md). This page covers GotA's +contract and seats. + +Code map (in `examples/gods_of_the_arena/` unless noted): + +| file | role | +|---|---| +| `neural_contract.nim` | observation v1 builder, action v1 decoder, demonstration encoder, contract texts + SHA-256, GotA package options (goal, defer, mask) | +| `neural_host_hooks.nim` (included by `bots.nim`) | `GotaContract` (the tier's NeuralContract for GotA) and the GotA seats: `gota_act`, defer, mask, label capture, override, shadow | +| `neural/policy.bas` | default glue: draft, shopping, ability leveling, buyback (base.bas routines) + `gota_act()` | +| `native_env.nim` / `native_env.h` | the training C ABI | +| `tools/native_env.py` | ctypes binding | +| `tools/test_native_env.py`, `tools/test_native_concurrency.py`, `tools/mapping_ceiling.py`, `tools/parity_upstream.sh`, `tools/canary.py` | acceptance tools | +| `../../coworld/gota/runtime/neural_package.py` | GotA staging validator + builder, a thin wrapper over the shared `coworld/runtime/neural_package.py` | + +## Package + +A package is a ZIP containing exactly three files: `manifest.json`, `policy.bas` and `model.bin`. Stored +and deflate entries are allowed; encrypted entries are not. The whole package is at most 16 MiB, and +`policy.bas` at most 256 KiB. The runtime recognises a package by its `PK\x03\x04` prefix. Anything else +is plain BASIC, and plain BASIC follows exactly the old path, including its 256 KiB bounded read. + +```json +{ + "schema": "gota-neural-basic/1", + "observation_contract": "ae4046e83cc02e861f9c8cc32550c6cc4d6f9c161c9225a9b34a314d310ea991", + "action_contract": "ecc7d53c11a9db0912467c66ecb3e65b60b3e71ef14dad1442ba3b4f6ac14697", + "decision_period": 4, + "files": {"policy.bas": "", "model.bin": ""}, + "model": {"format": "GOTANET1", "inputs": 1407, "hidden": 128, "heads": [8, 25, 49, 4, 6]}, + "goal": {"red": [1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0], "blue": [ ...16... ]}, + "decoder": {"mode": "argmax"} +} +``` + +Every key at every level is checked, and an unknown key rejects the package. `goal` and `decoder` are +optional. The default goal is w_score, and the default decoder is argmax. +`decoder.mode = "sample"` takes an optional `temperature` in 0.01..10; `temperature` without `sample` is +rejected. Each goal vector has 16 values in [-1, 1], and w_reserved (index 15) must be 0. The host +appends the vector of the seat's own team to that seat's observation. Both contract hashes must match +the game and the hashes stored inside `model.bin`. `decision_period` (1..24) must equal the period the +network was trained with; the native env's `decision_period` config key is the same number. + +Build and validate with `python3 coworld/gota/runtime/neural_package.py build|validate`. The validator +accepts exactly the packages the game accepts. The shared corruption suite (`tests/neural_cases.py` and +`tests/test_neural_cases.nim`) runs both validators in CI. `tools/test_native_env.py` also checks that the +GotA library and the GotA wrapper agree on the GotA-specific corruptions (goal, defer and mask keys, float +integers, booleans as numbers, non-finite numbers, zip bombs). + +## Defer seats: `decoder.defer_script` + +`"decoder": {"defer_script": true}` (a boolean; combines with `mode`/`temperature`) turns verb 0 into +DEFER. `policy.bas` is then the script to defer to, for example `players/base.bas` verbatim, with no glue: +it is the seat's real program. It runs every tick on the true world state, and it drafts, shops, levels +abilities and buys back itself; those non-contract calls always execute. It needs no `gota_act`, and if it +calls one, the call does nothing. The network runs on each decision tick as usual. Before the script's BASIC +turn on that tick, the host consults the decoded heads once: + +- **verb 0, defer.** Every contract command (walkTo, attackMove, attackTarget, castTarget, castPoint, + useItem, useItemAt) the script issues in this decision window executes live, on the tick it is issued. + The window is the decision tick plus the next `decision_period - 1` ticks. +- **Any other verb, override.** The decoded command is issued on the decision tick. For the rest of the + window, the script's contract commands are absorbed: they are not executed, the call returns 1, and + `lastActionError` is unchanged. +- **An override that decodes to nothing** (an invalid choice, e.g. an empty target slot) counts as invalid + and falls back to defer for that window, so the script keeps acting instead of the hero idling. + +The script is never paused or re-run, and it keeps its persistent variables. After an override it sees the +hero where the network's command put it, but its variables may still assume that its absorbed orders ran. +base.bas re-issues an order only when it changes (or on its periodic refresh), so after an override the +network's order stays held until the script issues a new one. A seat that is dead at the decision frame +defers. + +An always-defer network is byte-identical to a plain seat running the script (state hash every tick and the +replay bytes; `tools/test_defer.py` a). The VM has the neural-seat limits, which use the plain-seat budget +(see below). The network consult is host-side and costs no BASIC instructions, so the script runs out of +budget exactly where a plain seat would. Seats without the key keep verb 0 = noop. + +The native equivalent is `gota_set_seat_defer_script(h, seat, path)` on a learner seat (see native_env.h). +It uses the same code path, `deferConsult` in neural_host_hooks.nim, at the same point in the tick. The +native call adds `gota_seat_orders` labels of the script's commands, issued or absorbed. +`gota_seat_defer_stats` gives the {defer, override} decision counts for both kinds of seat (an invalid +override that fell back counts as a defer). Build a +package with `neural_package.py build --defer-script --policy players/base.bas --model model.bin`. + +## Action mask: `decoder.mask_empty_targets` + +`"decoder": {"mask_empty_targets": true}` makes the host apply a validity mask before argmax or sampling. An +optional `"mask_mode"` of `"conditional"` (the default) or `"static"` chooses how. The default is off and +byte-identical. The mask comes from the decision frame. `gota_action_mask(h, seat, uint8[187])` in +native_env.h returns the same bytes to a trainer: verb[8], castTarget ability[4], then 7 target rows x 25: + +| row | verb | allowed slot | +|---|---|---| +| 0 | attackTarget | occupied, not self, and a living enemy the engine lets the hero attack (`isEnemyTarget`) | +| 1-4 | castTarget with ability 0-3 | occupied, and the ability is self-cast, or the object is a living, visible spell target of the right faction (Strike: not own; others: own) | +| 5 | castPoint | occupied (point anchor) | +| 6 | useItemAt | occupied (point anchor) | + +Verb 0 is always allowed. Verbs 1, 2 and 6 are allowed while the seat is alive. Verbs 3, 5 and 7 are allowed +when their row has a slot, and verb 4 when some ability row does. A dead seat allows only verb 0. Range, +cooldown, mana and charges are not masked: the engine reports those as action errors, not decode noops. + +- **conditional:** mask the verb, then the ability (only for castTarget), then the target by the + (verb, ability) row. A decision decoded this way never becomes an invalid noop (`stats[21]` = 0). +- **static:** for trainers that sample heads independently with one mask per head (PufferLib). Mask the verb, + leave the ability unmasked, and mask the target by the union of the rows of the allowed target-reading + verbs. Cross-verb invalids remain. + +Point and item heads are never masked. When sampling, the host draws one uniform per head in head order +first, so it uses the RNG exactly as unmasked sampling does. `tools/native_env.py` has `masked_argmax` and +`static_masks`, which reproduce the host decoder. + +## model.bin (GOTANET1) + +GOTANET1 is the Polyworld neural model format (docs/neural-policies.md); GotA introduced it. The layout: +`magic[8] | u32 version=1, inputs I, hidden H, outputs O, heads n, parameters P | obs sha256 hex[64] | +action sha256 hex[64] | u32 head sizes[n] | f32 LE weights`. The weights come in this order: `W_enc[H][I]` +(x = W_enc·obs, no bias, no activation), `W_rec[3H][H]`, then `W_dec[O][H]`. H must be 64, 128, 256, 384 or 512, and +P must equal `I·H + 3H² + O·H` ≤ 2,000,000. + +One MinGRU step, the same as PufferLib's `mingru_gate`: + +``` +c = W_rec · x (split into hidden, gate, proj, H each) +h~ = hidden >= 0 ? hidden + 0.5 : sigmoid(hidden) +state' = lerp(state, h~, sigmoid(gate)) (PufferLib's two-branch lerp) +y = sigmoid(proj) * state' + (1 - sigmoid(proj)) * x +logits = W_dec · y +``` + +The published cost is `2·P + 32·H` operations per inference: 218,496 for w64, 486,144 for w128, +1,168,896 for w256, 2,048,256 for w384 and 3,124,224 for w512 (a w512 model.bin is 6.1 MB, inside +the 16 MiB package limit). A seat runs at most one inference per tick, against its own **4,000,000 +operations per tick** budget. A model over budget is rejected at load. This budget is separate from +BASIC's instruction budget. Nonfinite weights, inputs, state or outputs are errors. An inference error +disables the seat the same way a BASIC runtime error does. + +## Decisions, state and telemetry + +Decision ticks are battle ticks 1, 1 + p, 1 + 2p, … where p is `decision_period`. At the start of the +heroes' turn on a decision tick, before any seat's BASIC runs, every neural seat freezes its **decision +frame**. The frame is the observation plus the 25 object slots, taken from the same frozen object frame +BASIC reads. The native env freezes the object frame at the same point, so native and hosted observations +are identical. The time feature divides by the match length as the match actually runs it; a package +seat reads that length at each decision. `tools/test_package_goal.py` checks that a native package seat and +the headless runner's package seat produce identical hash and action streams. If the seat is alive, the host then runs the network (package seat) or takes the trainer's +heads (native learner seat) and decodes one command. `policy.bas` issues that command by calling +`gota_act()` during the seat's normal turn. The command goes through the same recorded host path as +`walkTo`/`castTarget`/…, so replays contain ordinary actions and re-simulate without the network. Between +decisions nothing is issued. The engine keeps executing the last order (its path or attack target); that +is the "held command". A decoded noop issues nothing. + +The recurrent state is zeroed before the first decision of a match, and before the first alive decision +after a death (death, respawn and buyback included). No inference runs while the seat is dead. The +native env's `resets[]` flag marks exactly these points. + +Sampling (`decoder.mode = "sample"`) keeps one SplitMix64 stream per seat, seeded with +`uint32(match seed) * 1000003 + seat + 1`. Each head takes one 53-bit draw in head order from +softmax(logits / T), computed in float64. + +Package seats write to their private seat log: +`neural: peak_ops= budget=4000000 model=w ticks= inferences= decisions= +invalid=`, plus ` defer= override=` for defer seats. The line is written at the first inference, +then every 1,800 inferences, then at the last decision of a full-length match. A package that fails +validation ends a hosted episode before play with the reason in the seat log, and the platform reports +"neural package rejected for player slot N". + +BASIC surface for `policy.bas`, registered only for neural seats: +- `gota_act()`: issues the decoded command. It returns 1 if the command was accepted and 0 otherwise, + and does nothing on non-decision ticks or a second call in the same tick. +- `run_neural_net()`: 1 when this tick has a fresh decision. +- `neuralObservation(i)`, `neuralLogits(i)`, `neuralState(i)`: Q16.16 reads. +- `neuralModel(k)`: k = 0 width, 1 inputs, 2 outputs, 3 period, 5..9 the chosen heads. + +Package seats, learner seats and defer seats get `neuralVmLimits`. That is the plain-seat per-tick budget, +the same as every `.bas` seat (20,000 instructions and 50,000 work units per tick), with 160 host functions +instead of 128 to make room for the neural host functions. The per-tick budget is kept equal for fairness: +the network runs on its own separate 4,000,000-op budget. The shipped `policy.bas` peaks at 1,065 +instructions per tick (10 learner seats, a full match). Replay metrics report each seat's instructions +against that seat's own limit. + +## Observation contract v1 (1407 float32) + +The observation is ego-centric in the **team frame**. Blue seats see the map rotated 180°: x_t = −x and +y_t = −z. Both teams therefore read the same geometry. Distances are in tiles (1 tile = 60,000 world +units), and time is in ticks (24 per second). Only information a BASIC script of that seat can read is +used: fog-gated objects, team vision and known building state. + +| offset | size | block | +|---|---|---| +| 0 | 48 | self | +| 48 | 4×16 | abilities (HeroAbilitySlot order: passive, primary, secondary, ultimate) | +| 112 | 6×25 | inventory slots | +| 262 | 25×40 | object slots | +| 1262 | 4×8 | spell warnings | +| 1294 | 16 | summary | +| 1310 | 81 | terrain | +| 1391 | 16 | goal w | + +**Self (48):** +- 0 alive, 1 hp/maxHp, 2 mana/maxMana, 3 maxHp/2000, 4 maxMana/1000, 5 gold/1000, 6 level/20, + 7 xp/xpForNextLevel, 8 totalXp/20000, 9-10 position/64 (team frame), 11 team. +- 12 battleTick/max_ticks, 13 attack cooldown/48, 14 attack range tiles/10, 15 attack damage/200, + 16 move tiles per second/5, 17 has attack target. +- 18-21 stun/root/silence/channel ticks/72, 22 portal cooldown/1440, 23 respawn/1440. +- 24 ability points/4, 25 in own spawn, 26 can shop, 27 deaths/10, 28 last action error, 29 buyback + affordable, 30-39 class one-hot, 40-44 role one-hot, 45-46 velocity ×10, 47 has move target. + +Time features are clipped to 2. + +**Ability (16):** rank/max, learned, cooldown/240, charges/3, recharge/240, mana cost/200, castable now, +damage/300, heal/300, restore/200, range tiles/10, cast kind one-hot (self, melee, projectile, area), +area radius/3. + +**Item slot (25):** one-hot of the 23 items (NoItem … ManaPotion), count/4, cooldown/240. + +**Object slots (25).** These are also the action's target slots: +- 0: self. +- 1-4: allies in seat order (present while alive). +- 5-9: enemy heroes in seat order (present while visible). +- 10-16: the 7 nearest visible living lane creeps of either team. +- 17: own god. +- 18: enemy god (visible). +- 19: own forward tower (the living allied tower nearest the enemy god). +- 20: nearest visible enemy tower or barracks. +- 21-24: the 4 nearest visible living neutrals. + +Ties sort by id. Self is a slot so that `castTarget(self)` exists. Structures are role-named rather than +"nearest 4" so a portal scroll always has its two anchors, home and the front. + +**Object features (40):** +- 0 present, 1-2 d/16 (±4), 3 dist/16 (≤8), 4 hp fraction, 5 hp/2000, 6 team (+1 ally, −1 enemy, + 0 neutral), 7-12 kind one-hot (god, hero, creep, tower, barracks, neutral), 13 alive/attackable, + 14 level/20, 15 mana/1000. +- 16 targets me, 17 is my target, 18 targets an allied hero, 19-21 stun/root/silence/72, 22-23 velocity + ×10, 24-25 facing (unit, team frame), 26 inside my attack range, 27 hp ≤ my attack damage, + 28 returning, 29 creep kind or neutral tier/3, 30-39 hero class one-hot. + +**Spell warnings (4 × 8):** the visible unresolved casts sorted by impact tick, then distance. Features: +present, d/16, dist/16, ticks to impact/72, hostile, harmless (heal/restore), ability/40. + +**Summary (16):** +- 0 own god hp, 1 own god exposed, 2-3 own towers and barracks alive fraction, 4-5 enemy towers and + barracks known alive, 6 enemy god hp if visible else −1, 7 own heroes alive/5, 8 visible enemy + heroes/5, 9 wave timer fraction. +- 10-11 own and visible-enemy level sums/50, 12 my league score/5000, 13-14 my kills and assists/10, + 15 zero. + +**Terrain (81):** a 9×9 known-walkable patch at a 2-tile stride centred on the hero's tile, in the team +frame and row-major. Offset (−8, −8) comes first. + +**Goal (16), in order:** +- w_score, w_win, w_xp, w_gold, w_hero_kill, w_assist, w_death, w_last_hit, w_neutral_kill, +- w_tower_damage, w_structure_kill, w_hero_damage, w_damage_taken, w_push_depth, w_god_damage, + w_reserved. + +## Action contract v1: heads {verb 8, target 25, point 49, ability 4, item 6} + +| verb | command | +|---|---| +| 0 noop | nothing (hold) | +| 1 walk / 2 attackMove | to self + point (walk rings 2, 5, 12 tiles), rounded to the tile centre; point 0 = own tile (stop) | +| 3 attackTarget | the target slot's object (not self) | +| 4 castTarget | ability head on the target slot's object (self allowed) | +| 5 castPoint | ability head at target object + point (cast rings 0.5, 1.25, 2.5 tiles, fractional) | +| 6 useItem | item head's inventory slot | +| 7 useItemAt | item head's slot at target object + point (cast rings) | + +The point is `0` for the centre, or `1 + ring·16 + dir`, where dir k is at k·22.5° counter-clockwise from +team-frame +x. The direction table is integer Q16, so decoding is identical on every platform. An empty +target slot, an out-of-range head, or a dead seat decodes to noop and is counted in +`invalid_actions`. Draft, shopping, ability leveling and buyback stay in BASIC (`policy.bas`). + +**Demonstration encoder** (BC labels, the mapping ceiling, DAgger): +- walk/attackMove follow the route the order would take (`planRoute` on a copy of the hero). The ring is + chosen by route length (<1 centre, <3.5 ring 0, <8.5 ring 1, else ring 2), and the direction is the + one nearest the route point at that ring's radius. +- attackTarget and castTarget look up the slot of the object id. An attack target outside the slots + becomes an attackMove toward it (`exact = 0`). +- castPoint and useItemAt take the best (anchor slot, point) pair of the 25 × 49. + +In one decision window, the first cast or item use wins; otherwise the last movement/attack order is +the label. + +## Native training environment + +See `native_env.h`. The build is: + +``` +POLYWORLD_DEPS= nim c --app:lib -d:release -d:headless \ + -d:gotaTrainingStats --mm:atomicArc --threads:on -d:useMalloc -u:nimTypeNames \ + -o:libgota_env.so examples/gods_of_the_arena/native_env.nim +``` + +A handle is one ten-seat match. Any mix of learner seats (which run `policy.bas` exactly as a package +seat does, with the caller's heads) and scripted seats is allowed on either team. Handles may be +stepped concurrently from different threads, and a handle may migrate between threads. +`-d:gotaTrainingStats` compiles the per-seat goal counters (gold earned, last hits, neutral kills, tower +and structure damage/kills, hero damage dealt/taken, god damage) into the library only. Live games never +compile them, and they are never hashed. + +The trainer, the hosted seat and the mapping ceiling share one code path: +- A scripted seat in **capture** mode (default) has its commands encoded into `gota_seat_orders` labels. +- In **override** mode (`gota_set_seat_override`) the script's commands are intercepted, encoded, decoded + and executed exactly as a learner's would be. This is the mapping ceiling. +- A learner in **shadow** mode (`gota_set_seat_shadow`) runs an expert script whose calls change + nothing. Its commands become labels (DAgger). + +## Evidence + +- **No-neural parity with upstream** (`tools/parity_upstream.sh`, 20 seeds, full 28,800 ticks, mixed + base/puller/rusher lineup): the headless binary with neural support writes replays byte-identical to + upstream/main's. The native library (capture on, no neural seats) ends on upstream's final hash, and + upstream's binary re-simulates its replay with zero mismatches. +- **Mapping ceiling** (`tools/mapping_ceiling.py`, 100 seeds, full length): base.bas routed through the + contract loses nothing against plain base.bas. + - Red paired XP delta: +187 ± 70. + - Blue paired XP delta: +45 ± 77. + - Score: 260 vs 129 and 211 vs 99. + - Wins: 2–3% vs 0%. +- **Package vs ABI parity** (`tools/test_native_env.py`, full length): a package-hosted seat and a + learner seat driven through `gota_net_infer` with the same weights produce identical worlds at w64, + w128, w256, w384 and w512. The same test checks that both loaders accept exactly those five widths. +- **Concurrency** (`tools/test_native_concurrency.py`): N threads × M handles give the serial per-step + hashes. +- **Hosted canary** (`tools/canary.py`): the `-d:coworld` server runs 5 random-weight packages and 5 + base.bas seats for a full match. Every seat exits 0, the telemetry lines are present, and the replay + re-simulates. diff --git a/examples/gods_of_the_arena/neural_contract.nim b/examples/gods_of_the_arena/neural_contract.nim new file mode 100644 index 00000000..59ccf5bc --- /dev/null +++ b/examples/gods_of_the_arena/neural_contract.nim @@ -0,0 +1,899 @@ +## GotA neural observation and action contract v1 (hash-pinned). +## +## GotA is a client of the shared neural tier (src/polyworld/neural_*.nim, +## docs/neural-policies.md); this module is the GotA half. It owns the +## fixed-size ego-centric observation, the five-head action decoder, the +## demonstration encoder and the GotA package options (goal, defer_script, +## mask_empty_targets), so the hosted neural seat, +## the native training library and the mapping-ceiling tool can never drift +## apart. Layout, normalizers and rationale: neural_basic.md. Any change to a +## feature list, a normalizer or a decode rule must change the contract text +## below (and therefore its SHA-256). +## +## Decoding is integer-only (a Q16 direction table) so a decoded command is +## the same on every platform; the observation is FP32. + +import + std/[algorithm, json, math, strutils], + crunchy, fixxy, + polyworld/[bodies, metrics, neural_host, pathing], + content, maps, observations, scores, sim, terrains + +const + ObsSelf* = 48 + ObsAbilityFeatures* = 16 + ObsAbilities* = 4 * ObsAbilityFeatures + ObsItemFeatures* = 25 + ObsItems* = InventorySlots * ObsItemFeatures + ObjectSlots* = 25 + ObjectFeatures* = 40 + ObsObjects* = ObjectSlots * ObjectFeatures + SpellSlots* = 4 + SpellFeatures* = 8 + ObsSpells* = SpellSlots * SpellFeatures + ObsSummary* = 16 + TerrainSide* = 9 + TerrainStride* = 2 + ObsTerrain* = TerrainSide * TerrainSide + GoalSize* = 16 + ObsSelfOffset* = 0 + ObsAbilityOffset* = ObsSelfOffset + ObsSelf + ObsItemOffset* = ObsAbilityOffset + ObsAbilities + ObsObjectOffset* = ObsItemOffset + ObsItems + ObsSpellOffset* = ObsObjectOffset + ObsObjects + ObsSummaryOffset* = ObsSpellOffset + ObsSpells + ObsTerrainOffset* = ObsSummaryOffset + ObsSummary + ObsGoalOffset* = ObsTerrainOffset + ObsTerrain + ObservationSize* = ObsGoalOffset + GoalSize + + HeadSizes* = [8, 25, 49, 4, 6] + ActionHeads* = HeadSizes.len + ActionOutputs* = 8 + 25 + 49 + 4 + 6 + + SlotSelf* = 0 + SlotAllies* = 1 ## four + SlotEnemies* = 5 ## five + SlotCreeps* = 10 ## seven + SlotStructures* = 17 ## own god, enemy god, own forward tower, enemy structure + SlotNeutrals* = 21 ## four + CreepSlots = 7 + NeutralSlots = 4 + + WalkRings*: array[3, int32] = [2 * WorldScale, 5 * WorldScale, 12 * WorldScale] + CastRings*: array[3, int32] = [WorldScale div 2, WorldScale * 5 div 4, + WorldScale * 5 div 2] + Directions*: array[16, (int32, int32)] = [ + (65536'i32, 0'i32), (60547'i32, 25080'i32), (46341'i32, 46341'i32), + (25080'i32, 60547'i32), (0'i32, 65536'i32), (-25080'i32, 60547'i32), + (-46341'i32, 46341'i32), (-60547'i32, 25080'i32), (-65536'i32, 0'i32), + (-60547'i32, -25080'i32), (-46341'i32, -46341'i32), (-25080'i32, -60547'i32), + (0'i32, -65536'i32), (25080'i32, -60547'i32), (46341'i32, -46341'i32), + (60547'i32, -25080'i32)] + + DefaultDecisionPeriod* = 4 + GoalNames* = ["w_score", "w_win", "w_xp", "w_gold", "w_hero_kill", + "w_assist", "w_death", "w_last_hit", "w_neutral_kill", "w_tower_damage", + "w_structure_kill", "w_hero_damage", "w_damage_taken", "w_push_depth", + "w_god_damage", "w_reserved"] + + SelfFeatureNames = ["alive", "hp_frac", "mana_frac", "max_hp/2000", + "max_mana/1000", "gold/1000", "level/20", "xp_to_next_frac", + "total_xp/20000", "x/64(team)", "y/64(team)", "team", "battle_frac", + "attack_cooldown/48", "attack_range_tiles/10", "attack_damage/200", + "move_tiles_per_s/5", "has_target", "stun/72", "root/72", "silence/72", + "channel/72", "portal_cd/1440", "respawn/1440", "ability_points/4", + "in_own_spawn", "can_shop", "deaths/10", "last_action_error", + "buyback_affordable", "class0", "class1", "class2", "class3", "class4", + "class5", "class6", "class7", "class8", "class9", "role0", "role1", + "role2", "role3", "role4", "vel_x*10", "vel_y*10", "has_move_target"] + AbilityFeatureNames = ["level/max", "learned", "cooldown/240", + "charges/3", "recharge/240", "mana_cost/200", "castable", "damage/300", + "heal/300", "restore/200", "range_tiles/10", "cast_self", "cast_melee", + "cast_projectile", "cast_area", "area_radius_tiles/3"] + ObjectFeatureNames = ["present", "dx/16(clip4)", "dy/16(clip4)", + "dist/16(clip8)", "hp_frac", "hp/2000", "team(+1 ally,-1 enemy)", + "kind_god", "kind_hero", "kind_creep", "kind_tower", "kind_barracks", + "kind_neutral", "alive", "level/20", "mana/1000", "targets_me", + "is_my_target", "targets_ally_hero", "stun/72", "root/72", "silence/72", + "vel_x*10", "vel_y*10", "facing_x", "facing_y", "in_my_range", + "last_hittable", "returning", "subclass", "class0", "class1", "class2", + "class3", "class4", "class5", "class6", "class7", "class8", "class9"] + SpellFeatureNames = ["present", "dx/16", "dy/16", "dist/16", + "impact/72", "hostile", "harmless", "ability/40"] + SummaryNames = ["own_god_hp", "own_god_exposed", "own_towers", + "own_barracks", "enemy_towers_known", "enemy_barracks_known", + "enemy_god_hp_visible_or_-1", "own_heroes_alive", "enemy_heroes_visible", + "wave_timer", "own_levels/50", "visible_enemy_levels/50", "score/5000", + "kills/10", "assists/10", "zero"] + +static: + doAssert SelfFeatureNames.len == ObsSelf + doAssert AbilityFeatureNames.len == ObsAbilityFeatures + doAssert ObjectFeatureNames.len == ObjectFeatures + doAssert SpellFeatureNames.len == SpellFeatures + doAssert SummaryNames.len == ObsSummary + doAssert ObservationSize == 1407 + +proc observationContractText*(): string = + ## The canonical text the observation hash pins. + result = "gota-neural-basic/1 observation v1 float32[" & $ObservationSize & + "] team-frame (blue rotated 180);\nself:" & SelfFeatureNames.join(",") & + "\nability x4:" & AbilityFeatureNames.join(",") & + "\nitem x6: onehot23(NoItem..ManaPotion),count/4,cooldown/240" & + "\nobject slots 25 (0 self,1-4 allies seat order,5-9 enemies seat order " & + "visible,10-16 nearest visible lane creeps,17 own god,18 enemy god " & + "visible,19 own forward tower,20 nearest visible enemy tower/barracks," & + "21-24 nearest visible neutrals):" & ObjectFeatureNames.join(",") & + "\nspells x4 (visible, unresolved, sorted impact,dist):" & + SpellFeatureNames.join(",") & + "\nsummary:" & SummaryNames.join(",") & + "\nterrain 9x9 stride 2 tiles known-walkable, row-major team-frame" & + "\ngoal:" & GoalNames.join(",") + +proc actionContractText*(): string = + ## The canonical text the action hash pins. + "gota-neural-basic/1 action v1 heads verb8,target25,point49,ability4,item6;" & + "verbs noop,walk,attackMove,attackTarget,castTarget,castPoint,useItem," & + "useItemAt; point 0=centre, 1+ring*16+dir, dir k at k*22.5deg ccw team " & + "frame (Q16 table), walk rings 2,5,12 tiles about self rounded to tile " & + "centre, cast rings 0.5,1.25,2.5 tiles about target object exact; " & + "invalid->noop; issued once on the decision tick; slots as observation v1" + +proc hexHash(text: string): string = + for value in sha256(cast[pointer](text.cstring), text.len): + result.add value.toHex(2).toLowerAscii() + +let + ObservationContractHash* = hexHash(observationContractText()) + ActionContractHash* = hexHash(actionContractText()) + +type + CommandKind* = enum + NoCommand, WalkCommand, AttackMoveCommand, AttackTargetCommand, + CastTargetCommand, CastPointCommand, UseItemCommand, UseItemAtCommand + NeuralCommand* = object + ## A concrete hero order, in the host's own argument space. + kind*: CommandKind + objectId*: int32 + ability*, item*: int32 + point*: FixedVec2 ## map tile-centre coordinates (as BASIC passes them) + tick*: int32 + DecisionFrame* = object + ## What the slots named on one decision tick. + tick*: int32 + heroIndex*: int + team*: Team + alive*: bool + selfPos*: WorldPoint + ids*: array[ObjectSlots, int32] + positions*: array[ObjectSlots, WorldPoint] + Heads* = array[ActionHeads, int32] + +template side(team: Team): int32 = (if team == RedTeam: 1'i32 else: -1'i32) + +proc clampf(x, lo, hi: float32): float32 {.inline.} = max(lo, min(hi, x)) + +proc tiles(value: int32): float32 {.inline.} = float32(value) / float32(WorldScale) + +proc mapPoint*(p: WorldPoint): FixedVec2 = + ## World position to BASIC map coordinates (tile centres at integers). + let half = int64(mapTiles() div 2) + proc one(v: int32): Fixed = + Fixed(int32((int64(v) * FixedScale) div int64(WorldScale) + + (half * FixedScale) - FixedScale div 2)) + fixedVec2(one(p.x), one(p.z)) + +proc worldPointOf*(point: FixedVec2): WorldPoint = + ## BASIC map coordinates to a world position (inverse of `mapPoint`). + let half = int64(mapTiles() div 2) + proc one(v: Fixed): int32 = + int32(((int64(int32(v)) - half * FixedScale + FixedScale div 2) * + int64(WorldScale)) div FixedScale) + WorldPoint(x: one(point.x), z: one(point.y)) + +proc clampToMap(p: WorldPoint): WorldPoint = + let edge = int32(mapTiles() div 2) * WorldScale - WorldScale div 2 + WorldPoint(x: clamp(p.x, -edge, edge - 1), y: p.y, z: clamp(p.z, -edge, edge - 1)) + +proc pointCandidate*(anchor: WorldPoint, team: Team, index: int, + rings: array[3, int32]): WorldPoint = + ## The world point a point-head index names about an anchor. + if index <= 0 or index >= 49: + return anchor + let + ring = (index - 1) div 16 + (cx, cy) = Directions[(index - 1) mod 16] + r = int64(rings[ring]) + s = int64(team.side) + clampToMap(WorldPoint( + x: anchor.x + int32((s * int64(cx) * r) div 65536), + y: anchor.y, + z: anchor.z + int32((s * int64(cy) * r) div 65536))) + +# --------------------------------------------------------------------------- +# Observation + +proc objectKindIndex(kind: int32): int = + ## Object kinds 1..6 (god, hero, creep, tower, barracks, neutral) -> 0..5. + clamp(int(kind) - 1, 0, 5) + +proc heroAlive(hero: Hero): bool = hero.hp > 0 and hero.state != Dying + +proc terrainOpen(world: World, team: Team, layer: int32, mx, my: int): bool = + ## Known walkability at one map tile, as terrainWalkable reports it. + if mx < 0 or my < 0 or mx >= mapTiles() or my >= mapTiles(): + return false + if terrainValue(int32(mx), int32(my), layer, TerrainWalkableField) == 0: + return false + let floor = layers[int(layer)] + world.knownWalkable(team, int(layer), mx + mapOrigin() - floor.originX, + my + mapOrigin() - floor.originZ) + +proc writeObject(o: var openArray[float32], base: int, world: World, + hero: Hero, value: WorldObject, damage: int32, rangeUnits: int32, + alliedHeroIds: openArray[int32]) = + let + s = float32(hero.team.side) + dx = tiles(value.position.x - hero.position.x) * s + dy = tiles(value.position.z - hero.position.z) * s + dist = sqrt(dx*dx + dy*dy) + o[base + 0] = 1 + o[base + 1] = clampf(dx / 16, -4, 4) + o[base + 2] = clampf(dy / 16, -4, 4) + o[base + 3] = clampf(dist / 16, 0, 8) + o[base + 4] = if value.maxHp > 0: clampf(float32(max(value.hp, 0)) / float32(value.maxHp), 0, 1) else: 0 + o[base + 5] = float32(max(value.hp, 0)) / 2000 + let faction = value.faction + o[base + 6] = if faction == 2: 0'f32 elif faction == hero.team.ord.int32: 1 else: -1 + o[base + 7 + objectKindIndex(value.kind)] = 1 + o[base + 13] = float32(value.alive) + o[base + 14] = float32(value.level) / 20 + o[base + 15] = float32(value.mana) / 1000 + o[base + 16] = float32(value.targetId != 0 and value.targetId == hero.id) + o[base + 17] = float32(value.id != 0 and value.id == hero.attackObjectId) + var targetsAlly = false + if value.targetId != 0 and value.targetId != hero.id: + for id in alliedHeroIds: + if id == value.targetId: + targetsAlly = true + o[base + 18] = float32(targetsAlly) + o[base + 19] = clampf(float32(value.controlTicks[StunControl]) / 72, 0, 2) + o[base + 20] = clampf(float32(value.controlTicks[RootControl]) / 72, 0, 2) + o[base + 21] = clampf(float32(value.controlTicks[SilenceControl]) / 72, 0, 2) + o[base + 22] = clampf(tiles(value.velocity.x) * s * 10, -4, 4) + o[base + 23] = clampf(tiles(value.velocity.z) * s * 10, -4, 4) + let flen = sqrt(float32(value.facing.x)*float32(value.facing.x) + + float32(value.facing.z)*float32(value.facing.z)) + if flen > 0: + o[base + 24] = float32(value.facing.x) / flen * s + o[base + 25] = float32(value.facing.z) / flen * s + o[base + 26] = float32(dist * float32(WorldScale) <= float32(rangeUnits)) + o[base + 27] = float32(value.hp > 0 and value.hp <= damage) + o[base + 28] = float32(value.returning) + if value.kind == 3: + o[base + 29] = float32(value.class) + elif value.kind == 6: + o[base + 29] = float32(value.class) / 3 + if value.kind == 2 and value.class in 0'i32 .. 9'i32: + o[base + 30 + int(value.class)] = 1 + +proc heroSelfObject(world: World, hero: Hero): WorldObject = + ## The seat's own hero in WorldObject form (it is always in its own frame). + result = WorldObject(id: hero.id, kind: 2, class: int32(hero.class.ord), + team: hero.team, position: hero.position, hp: hero.hp, maxHp: hero.maxHp, + alive: hero.heroAlive, level: int32(hero.level), mana: hero.mana, + facing: hero.facing, velocity: hero.velocity, + targetId: (if hero.heroAlive: hero.attackObjectId else: 0)) + if result.alive: + for effect in ControlEffect: + result.controlTicks[effect] = max(0'i32, hero.controls[effect].ends - world.tick) + +proc buildObservation*(world: World, heroIndex: int, goal: openArray[float32], + maxTicks: int32, stats: CombatStats, o: var openArray[float32], + frame: var DecisionFrame) = + ## Writes observation v1 for one hero from the current decision frame. + doAssert o.len == ObservationSize and goal.len == GoalSize + for i in 0 ..< o.len: + o[i] = 0 + frame = DecisionFrame(tick: world.tick, heroIndex: heroIndex) + let + hero = world.heroes[heroIndex] + team = hero.team + s = float32(team.side) + alive = hero.heroAlive + frame.team = team + frame.alive = alive + frame.selfPos = hero.position + # Self. + var p = ObsSelfOffset + o[p+0] = float32(alive) + o[p+1] = if hero.maxHp > 0: clampf(float32(max(hero.hp, 0)) / float32(hero.maxHp), 0, 1) else: 0 + o[p+2] = if hero.maxMana > 0: clampf(float32(hero.mana) / float32(hero.maxMana), 0, 1) else: 0 + o[p+3] = float32(hero.maxHp) / 2000 + o[p+4] = float32(hero.maxMana) / 1000 + o[p+5] = float32(hero.gold) / 1000 + o[p+6] = float32(hero.level) / 20 + o[p+7] = clampf(float32(hero.xp) / float32(max(1, xpForNextLevel(hero.level))), 0, 1) + o[p+8] = float32(hero.totalXp) / 20000 + o[p+9] = tiles(hero.position.x) * s / 64 + o[p+10] = tiles(hero.position.z) * s / 64 + o[p+11] = float32(team.ord) + o[p+12] = clampf(float32(world.battleTick()) / float32(max(1'i32, maxTicks)), 0, 1) + o[p+13] = clampf(float32(world.heroAttackCooldown(hero)) / 48, 0, 2) + o[p+14] = tiles(hero.class.heroAttackRange()) / 10 + o[p+15] = float32(hero.heroAttackDamage()) / 200 + o[p+16] = tiles(hero.heroMoveSpeed()) * float32(TickRate) / 5 + o[p+17] = float32(hero.attackObjectId != 0) + o[p+18] = clampf(float32(max(0'i32, hero.controls[StunControl].ends - world.tick)) / 72, 0, 2) + o[p+19] = clampf(float32(max(0'i32, hero.controls[RootControl].ends - world.tick)) / 72, 0, 2) + o[p+20] = clampf(float32(max(0'i32, hero.controls[SilenceControl].ends - world.tick)) / 72, 0, 2) + o[p+21] = clampf(float32(max(0'i32, hero.portalEnds - world.tick)) / 72, 0, 2) + o[p+22] = clampf(float32(max(0'i32, hero.portalCooldownEnds - world.tick)) / 1440, 0, 2) + o[p+23] = clampf(float32(hero.respawnTicks()) / 1440, 0, 2) + o[p+24] = float32(hero.abilityPoints()) / 4 + o[p+25] = float32(hero.inOwnSpawn) + o[p+26] = float32(world.phase != Drafting and hero.canShop) + o[p+27] = float32(hero.deaths) / 10 + o[p+28] = float32(hero.lastActionError != NoActionError) + let price = world.buybackPrice(hero.id) + o[p+29] = float32(price > 0 and hero.gold >= price) + o[p+30 + hero.class.ord] = 1 + o[p+40 + hero.class.heroRole.ord] = 1 + o[p+45] = clampf(tiles(hero.velocity.x) * s * 10, -4, 4) + o[p+46] = clampf(tiles(hero.velocity.z) * s * 10, -4, 4) + o[p+47] = float32(hero.hasMoveTarget) + # Abilities. + for slot in HeroAbilitySlot: + let + b = ObsAbilityOffset + slot.ord * ObsAbilityFeatures + rank = hero.abilityLevels[slot] + spec = heroAbility(hero.class, slot).abilitySpec(rank) + cooldown = hero.cooldowns[slot] + charges = hero.charges[slot] + o[b+0] = float32(rank) / float32(slot.abilityMaxLevel) + o[b+1] = float32(rank > 0) + o[b+2] = clampf(float32(cooldown) / 240, 0, 2) + o[b+3] = float32(charges) / 3 + o[b+4] = clampf(float32(hero.recharges[slot]) / 240, 0, 2) + o[b+5] = float32(spec.manaCost) / 200 + o[b+6] = float32(alive and rank > 0 and cooldown == 0 and charges > 0 and + hero.mana >= spec.manaCost and + hero.controls[SilenceControl].ends <= world.tick) + o[b+7] = float32(spec.damage) / 300 + o[b+8] = float32(spec.heal) / 300 + o[b+9] = float32(spec.restore) / 200 + o[b+10] = tiles(spec.range) / 10 + o[b+11 + spec.casting.ord] = 1 + o[b+15] = tiles(spec.area.radius) / 3 + # Items. + for slot in 0 ..< InventorySlots: + let b = ObsItemOffset + slot * ObsItemFeatures + o[b + hero.inventory[slot].ord] = 1 + o[b + 23] = float32(hero.itemCounts[slot]) / 4 + o[b + 24] = clampf(float32(hero.itemCooldown(slot, world.tick)) / 240, 0, 2) + # Objects. + var alliedHeroIds: seq[int32] + for other in world.heroes: + if other.team == team: + alliedHeroIds.add other.id + let + damage = hero.heroAttackDamage() + rangeUnits = hero.class.heroAttackRange() + var + creeps, neutrals, enemyStructures: seq[(int64, WorldObject)] + enemyGod, ownGod: WorldObject + haveEnemyGod, haveOwnGod = false + proc distance2(value: WorldObject): int64 = + let + dx = int64(value.position.x - hero.position.x) + dz = int64(value.position.z - hero.position.z) + dx*dx + dz*dz + var value: WorldObject + let count = world.worldObjectCount(hero.id) + var enemySeen: array[10, bool] + var enemyValues: array[10, WorldObject] + for i in 0 ..< count: + if not world.worldObjectAt(hero.id, i, value): + continue + case value.kind + of 1: + if value.team == team: + ownGod = value + haveOwnGod = true + else: + enemyGod = value + haveEnemyGod = true + of 2: + if value.team != team: + let index = world.heroIndex(value.id) + if index in 0 ..< 10: + enemySeen[index] = true + enemyValues[index] = value + of 3: + if value.alive: + creeps.add((value.distance2, value)) + of 4, 5: + if value.team != team and value.hp > 0: + enemyStructures.add((value.distance2, value)) + of 6: + if value.alive: + neutrals.add((value.distance2, value)) + else: + discard + proc place(o: var openArray[float32], frame: var DecisionFrame, slot: int, + value: WorldObject) = + writeObject(o, ObsObjectOffset + slot * ObjectFeatures, world, hero, + value, damage, rangeUnits, alliedHeroIds) + frame.ids[slot] = value.id + frame.positions[slot] = value.position + place(o, frame, SlotSelf, heroSelfObject(world, hero)) + var slot = SlotAllies + for i, other in world.heroes: + if other.team == team and i != heroIndex: + if other.heroAlive: + place(o, frame, slot, heroSelfObject(world, other)) + inc slot + slot = SlotEnemies + for i, other in world.heroes: + if other.team != team: + if enemySeen[i]: + place(o, frame, slot, enemyValues[i]) + inc slot + proc byDistance(a, b: (int64, WorldObject)): int = + result = cmp(a[0], b[0]) + if result == 0: + result = cmp(a[1].id, b[1].id) + creeps.sort(byDistance) + neutrals.sort(byDistance) + enemyStructures.sort(byDistance) + for i in 0 ..< min(CreepSlots, creeps.len): + place(o, frame, SlotCreeps + i, creeps[i][1]) + if haveOwnGod: + place(o, frame, SlotStructures, ownGod) + if haveEnemyGod: + place(o, frame, SlotStructures + 1, enemyGod) + # Own forward tower: the living allied tower nearest the enemy god. + block: + let enemyCenter = world.forts[1 - team.ord].center + var best = int64.high + var bestIndex = -1 + for i, building in world.buildings: + if building.kind == TowerBuilding and building.team == team and building.hp > 0: + let + dx = int64(building.position.x - enemyCenter.x) + dz = int64(building.position.z - enemyCenter.z) + d = dx*dx + dz*dz + if d < best: + best = d + bestIndex = i + if bestIndex >= 0: + let building = world.buildings[bestIndex] + place(o, frame, SlotStructures + 2, WorldObject(id: building.id, kind: 4, + class: -1, team: building.team, position: building.position, + hp: building.hp, maxHp: building.maxHp, + alive: world.buildingExposed(building), facing: building.facing, + targetId: building.targetId)) + if enemyStructures.len > 0: + place(o, frame, SlotStructures + 3, enemyStructures[0][1]) + for i in 0 ..< min(NeutralSlots, neutrals.len): + place(o, frame, SlotNeutrals + i, neutrals[i][1]) + # Spell warnings. + var warnings: seq[(int32, int64, SpellCast)] + let spellCount = world.visibleSpellCount(hero.id) + var spell: SpellCast + for i in 0 ..< spellCount: + if world.visibleSpellAt(hero.id, i, spell) and spell.impact >= world.tick: + let + dx = int64(spell.position.x - hero.position.x) + dz = int64(spell.position.z - hero.position.z) + warnings.add((spell.impact, dx*dx + dz*dz, spell)) + warnings.sort(proc(a, b: (int32, int64, SpellCast)): int = + result = cmp(a[0], b[0]) + if result == 0: result = cmp(a[1], b[1])) + for i in 0 ..< min(SpellSlots, warnings.len): + let + b = ObsSpellOffset + i * SpellFeatures + w = warnings[i][2] + dx = tiles(w.position.x - hero.position.x) * s + dy = tiles(w.position.z - hero.position.z) * s + caster = world.heroIndex(w.heroId) + allied = caster >= 0 and world.heroes[caster].team == team + spec = w.ability.abilitySpec + o[b+0] = 1 + o[b+1] = clampf(dx / 16, -4, 4) + o[b+2] = clampf(dy / 16, -4, 4) + o[b+3] = clampf(sqrt(dx*dx + dy*dy) / 16, 0, 8) + o[b+4] = clampf(float32(w.impact - world.tick) / 72, 0, 4) + o[b+5] = float32(not allied) + o[b+6] = float32(spec.kind != Strike) + o[b+7] = float32(w.ability.ord) / 40 + # Summary. + p = ObsSummaryOffset + let + ownFort = world.forts[team.ord] + enemyFort = world.forts[1 - team.ord] + o[p+0] = float32(max(ownFort.hp, 0)) / float32(FortHp) + o[p+1] = float32(world.fortExposed(team)) + var ownT, ownTAll, ownB, ownBAll, enT, enTAll, enB, enBAll = 0 + for building in world.buildings: + let mine = building.team == team + if building.kind == TowerBuilding: + if mine: + inc ownTAll + if building.hp > 0: inc ownT + else: + inc enTAll + if building.knownAlive[team]: inc enT + else: + if mine: + inc ownBAll + if building.hp > 0: inc ownB + else: + inc enBAll + if building.knownAlive[team]: inc enB + o[p+2] = float32(ownT) / float32(max(1, ownTAll)) + o[p+3] = float32(ownB) / float32(max(1, ownBAll)) + o[p+4] = float32(enT) / float32(max(1, enTAll)) + o[p+5] = float32(enB) / float32(max(1, enBAll)) + o[p+6] = if haveEnemyGod: float32(max(enemyFort.hp, 0)) / float32(FortHp) else: -1 + var ownAlive, enemyVisible, ownLevels, enemyLevels = 0 + for i, other in world.heroes: + if other.team == team: + ownLevels += other.level + if other.heroAlive: inc ownAlive + elif enemySeen[i]: + inc enemyVisible + enemyLevels += int(enemyValues[i].level) + o[p+7] = float32(ownAlive) / 5 + o[p+8] = float32(enemyVisible) / 5 + o[p+9] = clampf(float32(world.spawnTimerTicks) / float32(max(1'i32, world.spawnIntervalTicks)), 0, 1) + o[p+10] = float32(ownLevels) / 50 + o[p+11] = float32(enemyLevels) / 50 + o[p+12] = float32(score(hero.totalXp, int(max(0'i32, world.battleTick())))) / 5000 + if stats != nil and heroIndex < stats.values.len: + o[p+13] = float32(stats.values[heroIndex][KillsMetric]) / 10 + o[p+14] = float32(stats.values[heroIndex][AssistsMetric]) / 10 + # Terrain: 9x9 known-walkable patch, stride 2 tiles, team frame, row-major. + let + center = mapPoint(hero.position) + cx = int((int64(int32(center.x)) + FixedScale div 2) shr 16) + cy = int((int64(int32(center.y)) + FixedScale div 2) shr 16) + layer = hero.navLayer + var t = ObsTerrainOffset + for row in 0 ..< TerrainSide: + for col in 0 ..< TerrainSide: + let + ox = (col - TerrainSide div 2) * TerrainStride * int(team.side) + oy = (row - TerrainSide div 2) * TerrainStride * int(team.side) + o[t] = float32(world.terrainOpen(team, layer, cx + ox, cy + oy)) + inc t + # Goal. + for i in 0 ..< GoalSize: + o[ObsGoalOffset + i] = goal[i] + +# --------------------------------------------------------------------------- +# Decoder + +proc decodeAction*(frame: DecisionFrame, heads: Heads): NeuralCommand = + ## Maps five head indices to a concrete order; invalid choices -> noop. + result = NeuralCommand(kind: NoCommand, tick: frame.tick) + if not frame.alive: + return + for h in 0 ..< ActionHeads: + if heads[h] < 0 or heads[h] >= HeadSizes[h]: + return + let + verb = heads[0] + target = int(heads[1]) + point = int(heads[2]) + ability = heads[3] + item = heads[4] + case verb + of 1, 2: + let + dest = pointCandidate(frame.selfPos, frame.team, point, WalkRings) + mp = mapPoint(dest) + (x, y, _) = splitTilePoint(mp) + result.kind = if verb == 1: WalkCommand else: AttackMoveCommand + result.point = fixedVec2(fixed(x), fixed(y)) + of 3: + if frame.ids[target] == 0 or target == SlotSelf: + return + result.kind = AttackTargetCommand + result.objectId = frame.ids[target] + of 4: + if frame.ids[target] == 0: + return + result.kind = CastTargetCommand + result.objectId = frame.ids[target] + result.ability = ability + of 5, 7: + if frame.ids[target] == 0: + return + let dest = pointCandidate(frame.positions[target], frame.team, point, CastRings) + result.kind = if verb == 5: CastPointCommand else: UseItemAtCommand + result.point = mapPoint(dest) + result.ability = ability + result.item = item + of 6: + result.kind = UseItemCommand + result.item = item + else: + discard + +# --------------------------------------------------------------------------- +# Action validity mask (native_env.h gota_action_mask; manifest +# decoder.mask_empty_targets). Optional decoder aid: not part of the contract +# text, the decode rules above are unchanged. + +const + MaskVerb* = 0 + MaskAbility* = 8 + MaskTarget* = 12 + MaskTargetRows* = 7 + MaskSize* = MaskTarget + MaskTargetRows * ObjectSlots + +type ActionMask* = array[MaskSize, uint8] + +proc maskTargetRow*(verb, ability: int32): int = + ## Target row of (verb, ability); -1 = the verb ignores the target head. + case verb + of 3: 0 + of 4: (if ability in 0'i32 .. 3'i32: 1 + int(ability) else: -1) + of 5: 5 + of 7: 6 + else: -1 + +proc actionMask*(world: World, heroIndex: int, frame: DecisionFrame): ActionMask = + ## Which choices decode to a real order on this frame (native_env.h). + result[MaskVerb] = 1 + if not frame.alive or world.gameOver: + return + let hero = world.heroes[heroIndex] + var specs: array[4, AbilitySpec] + for a in 0 ..< 4: + specs[a] = heroAbility(hero.class, HeroAbilitySlot(a)).abilitySpec() + for s in 0 ..< ObjectSlots: + let id = frame.ids[s] + if id == 0: + continue + result[MaskTarget + 5 * ObjectSlots + s] = 1 + result[MaskTarget + 6 * ObjectSlots + s] = 1 + if s != SlotSelf and world.isEnemyTarget(hero, id): + result[MaskTarget + s] = 1 + var target: WorldObject + let usable = world.spellTarget(id, target) and target.alive and + world.visible(hero.team, target.position) + for a in 0 ..< 4: + let ok = specs[a].casting == SelfCast or (usable and + (if specs[a].kind == Strike: target.faction != hero.team.ord.int32 + else: target.faction == hero.team.ord.int32)) + if ok: + result[MaskTarget + (1 + a) * ObjectSlots + s] = 1 + for v in [1, 2, 6]: + result[MaskVerb + v] = 1 + proc anyRow(mask: ActionMask, row: int): bool = + for s in 0 ..< ObjectSlots: + if mask[MaskTarget + row * ObjectSlots + s] != 0: + return true + for a in 0 ..< 4: + result[MaskAbility + a] = uint8(result.anyRow(1 + a)) + result[MaskVerb + 3] = uint8(result.anyRow(0)) + result[MaskVerb + 4] = uint8(result[MaskAbility] != 0 or + result[MaskAbility + 1] != 0 or result[MaskAbility + 2] != 0 or + result[MaskAbility + 3] != 0) + result[MaskVerb + 5] = uint8(result.anyRow(5)) + result[MaskVerb + 7] = uint8(result.anyRow(6)) + +proc isInvalid*(frame: DecisionFrame, heads: Heads): bool = + ## A non-noop verb that decoded to noop. + frame.alive and heads[0] != 0 and decodeAction(frame, heads).kind == NoCommand + +# --------------------------------------------------------------------------- +# Demonstration encoder (BC labels and the mapping ceiling) + +type Encoded* = object + heads*: Heads + exact*: bool + errorMilli*: int32 + represented*: bool + +proc slotOf(frame: DecisionFrame, id: int32): int = + if id == 0: + return -1 + for i in 0 ..< ObjectSlots: + if frame.ids[i] == id: + return i + -1 + +proc pathPointAt(start: WorldPoint, path: openArray[WorldPoint], + arc: float64): (WorldPoint, float64) = + ## The point `arc` world units along the route (or its end) and the route length. + var + prev = start + walked = 0.0 + for p in path: + let + dx = float64(p.x - prev.x) + dz = float64(p.z - prev.z) + seg = sqrt(dx*dx + dz*dz) + if walked + seg >= arc and seg > 0: + let f = (arc - walked) / seg + return (WorldPoint(x: prev.x + int32(dx*f), z: prev.z + int32(dz*f)), -1.0) + walked += seg + prev = p + (prev, walked) + +proc encodeMove(frame: DecisionFrame, world: World, verb: int32, + goal: FixedVec2): Encoded = + ## Walk/attack-move: follow the route the order produces, not the chord. + let hero = world.heroes[frame.heroIndex] + let (gx, gy, goff) = splitTilePoint(goal) + let route = world.planRoute(hero, gx, gy, goff) + var target = worldPointOf(goal) + var length: float64 + if route.len > 0: + let (_, total) = pathPointAt(frame.selfPos, route, 1e18) + length = total + else: + let + dx = float64(target.x - frame.selfPos.x) + dz = float64(target.z - frame.selfPos.z) + length = sqrt(dx*dx + dz*dz) + let ws = float64(WorldScale) + var ring = -1 + if length >= 1.0 * ws: + ring = (if length < 3.5 * ws: 0 elif length < 8.5 * ws: 1 else: 2) + var aim = target + if ring >= 0 and route.len > 0: + let (point, _) = pathPointAt(frame.selfPos, route, float64(WalkRings[ring])) + aim = point + var best = (int64.high, 0) + var candidates: seq[int] + if ring < 0: + candidates.add 0 + else: + for d in 0 ..< 16: + candidates.add 1 + ring*16 + d + for index in candidates: + let + c = pointCandidate(frame.selfPos, frame.team, index, WalkRings) + dx = int64(c.x - aim.x) + dz = int64(c.z - aim.z) + e = dx*dx + dz*dz + if e < best[0]: + best = (e, index) + result.heads = [verb, 0, int32(best[1]), 0, 0] + result.represented = true + let decoded = decodeAction(frame, result.heads) + result.exact = decoded.point == fixedVec2(fixed(gx), fixed(gy)) and goff == FixedVec2Zero + result.errorMilli = int32(sqrt(float64(best[0])) * 1000 / ws) + +proc encodeAnchored(frame: DecisionFrame, verb, ability, item: int32, + goal: FixedVec2): Encoded = + ## castPoint/useItemAt: best (anchor slot, offset) pair for the aim point. + let target = worldPointOf(goal) + var best = (int64.high, 0, 0) + for slot in 0 ..< ObjectSlots: + if frame.ids[slot] == 0: + continue + for index in 0 ..< 49: + let + c = pointCandidate(frame.positions[slot], frame.team, index, CastRings) + dx = int64(c.x - target.x) + dz = int64(c.z - target.z) + e = dx*dx + dz*dz + if e < best[0]: + best = (e, slot, index) + if best[0] == int64.high: + return + result.heads = [verb, int32(best[1]), int32(best[2]), ability, item] + result.represented = true + result.errorMilli = int32(sqrt(float64(best[0])) * 1000 / float64(WorldScale)) + result.exact = decodeAction(frame, result.heads).point == goal + +proc encodeCommand*(frame: DecisionFrame, world: World, command: NeuralCommand): Encoded = + ## Nearest action-contract encoding of a script's command on this frame. + case command.kind + of NoCommand: + result.represented = true + result.exact = true + of WalkCommand: + result = encodeMove(frame, world, 1, command.point) + of AttackMoveCommand: + result = encodeMove(frame, world, 2, command.point) + of AttackTargetCommand: + let slot = frame.slotOf(command.objectId) + if slot > 0: + result.heads = [3'i32, int32(slot), 0, 0, 0] + result.represented = true + result.exact = true + else: + # Not in the slots: approach it with an attack-move instead. + var value: WorldObject + let hero = world.heroes[frame.heroIndex] + if world.worldObjectById(hero.id, command.objectId, value): + result = encodeMove(frame, world, 2, mapPoint(value.position)) + result.exact = false + of CastTargetCommand: + let slot = frame.slotOf(command.objectId) + if slot >= 0: + result.heads = [4'i32, int32(slot), 0, clamp(command.ability, 0, 3), 0] + result.represented = true + result.exact = command.ability in 0'i32 .. 3'i32 + of CastPointCommand: + result = encodeAnchored(frame, 5, clamp(command.ability, 0, 3), 0, command.point) + of UseItemCommand: + result.heads = [6'i32, 0, 0, 0, clamp(command.item, 0, 5)] + result.represented = true + result.exact = command.item in 0'i32 .. 5'i32 + of UseItemAtCommand: + result = encodeAnchored(frame, 7, 0, clamp(command.item, 0, 5), command.point) + +proc argmaxHeads*(logits: openArray[float32]): Heads = + ## Deterministic argmax per head (first maximum wins). + argmaxHeads(logits, HeadSizes, result) + +# Package options (the GotA extension keys of a gota-neural-basic/1 manifest). + +const + PackageSchema* = "gota-neural-basic/1" + MaxDecisionPeriod* = 24 + GotaManifestKeys* = ["goal"] + GotaDecoderKeys* = ["defer_script", "mask_empty_targets", "mask_mode"] + +type + MaskMode* = enum + NoMask ## decoder.mask_empty_targets absent or false + ConditionalMask ## mask_mode "conditional" (the default) + StaticMask ## mask_mode "static" + GotaPackageOptions* = ref object of RootObj + goals*: array[2, array[GoalSize, float32]] ## red, blue + deferScript*: bool + ## decoder.defer_script: verb 0 defers to policy.bas, which then acts + ## for that decision window. + maskMode*: MaskMode + ## decoder.mask_empty_targets + mask_mode: action validity mask + ## applied before decode. + +proc defaultGoal*(): array[GoalSize, float32] = + result[0] = 1 + +proc goalVector(node: JsonNode, where: string): array[GoalSize, float32] = + if node.kind != JArray or node.len != GoalSize: + raise newException(ValueError, where & " must be 16 numbers") + for i in 0 ..< GoalSize: + if node[i].kind notin {JInt, JFloat}: + raise newException(ValueError, where & " must be 16 numbers") + let v = node[i].getFloat + if v < -1 or v > 1: + raise newException(ValueError, where & " values must be in [-1, 1]") + result[i] = float32(v) + if result[GoalSize - 1] != 0: + raise newException(ValueError, where & " w_reserved must be 0") + +proc parseGotaOptions*(manifest: JsonNode): RootRef {.nimcall.} = + ## The GotA manifest keys: goal {red, blue} and the decoder options. + let options = GotaPackageOptions(goals: [defaultGoal(), defaultGoal()]) + if manifest.hasKey("goal"): + let goal = manifest["goal"] + goal.requireManifestKeys(["red", "blue"], "goal") + if not goal.hasKey("red") or not goal.hasKey("blue"): + raise newException(ValueError, "goal needs red and blue") + options.goals[0] = goalVector(goal["red"], "goal.red") + options.goals[1] = goalVector(goal["blue"], "goal.blue") + if manifest.hasKey("decoder"): + let decoder = manifest["decoder"] + let masked = decoder.requireBool("mask_empty_targets", "decoder") + if masked: + options.maskMode = ConditionalMask + if decoder.hasKey("mask_mode"): + if not masked: + raise newException(ValueError, "decoder.mask_mode needs mask_empty_targets true") + case (if decoder["mask_mode"].kind == JString: decoder["mask_mode"].getStr else: "") + of "conditional": discard + of "static": options.maskMode = StaticMask + else: + raise newException(ValueError, "decoder.mask_mode must be conditional or static") + options.deferScript = decoder.requireBool("defer_script", "decoder") + options diff --git a/examples/gods_of_the_arena/neural_host_hooks.nim b/examples/gods_of_the_arena/neural_host_hooks.nim new file mode 100644 index 00000000..d981d958 --- /dev/null +++ b/examples/gods_of_the_arena/neural_host_hooks.nim @@ -0,0 +1,441 @@ +## GotA neural seats: the hosted package seat, the native trainer's learner +## seat, and the scripted seats whose commands are captured (BC labels) or +## routed through the contract (mapping ceiling). Included by bots.nim. +## +## The generic parts (package loading, actor, recurrent-state lifecycle, +## budget, telemetry, decoders, the neural BASIC functions) are the shared +## tier in src/polyworld/neural_host.nim; this file is GotA's NeuralContract +## (observation builder, decoder, mask and defer options) and its seat modes. +## +## A seat without a NeuralSeat is untouched by everything here, so plain +## BASIC matches stay byte-identical. + +type + NeuralMode* = enum + NeuralLearner ## native trainer supplies the heads + NeuralHosted ## hosted neural package, network inside + NeuralOverride ## scripted seat, contract commands re-routed (ceiling) + NeuralCapture ## scripted seat, contract commands captured as labels + NeuralSeat* = ref object of NeuralBrain + ## A GotA seat: the shared brain plus the GotA frame, heads and modes. + mode*: NeuralMode + goal*: array[GoalSize, float32] + frame*: DecisionFrame + heads*: Heads + command*: NeuralCommand + issuedTick*: int32 + captured*: seq[NeuralCommand] + label*: array[16, int32] + lastLabel*: array[16, int32] + ## The label of the previous decision window (what gota_seat_orders reports). + labelInstant: bool + decisions*, invalid*: int + shadow*: HeroVm + ## Learner seats only: an expert script run on the same frame whose + ## commands become labels and are never executed (DAgger). + deferEnabled*: bool + ## Defer seat: verb 0 = defer to the seat's own script (its BASIC + ## program); any other verb overrides it for the decision window. + absorbing*: bool + ## Inside an override window: the script's contract commands are absorbed. + deferDecisions*, overrideDecisions*: int + maskMode*: MaskMode + ## decoder.mask_empty_targets / mask_mode (package seats): the mask + ## applied before decode. + mask*: ActionMask + ## The mask of the current decision frame (package seats with a mask). + +var shadowRunning {.threadvar.}: bool + ## True while a shadow expert script runs: its host calls change nothing. + +var neuralTelemetryEnabled* = true + +proc neuralSeat*(game: Game, index: int): NeuralSeat = + if index >= 0 and index < game.heroVms.len and game.heroVms[index] != nil and + game.heroVms[index].neural != nil: + result = NeuralSeat(game.heroVms[index].neural) + +proc isDecisionTick*(world: World, period: int32): bool = + ## Battle ticks 1, 1 + period, ... (the heroes' turn of that tick). + world.phase == Playing and decisionDue(world.battleTick(), period) + +proc seatLog(game: Game, index: int, text: string) = + when defined(coworld): + playerLog(index, text & "\n") + else: + if neuralTelemetryEnabled: + echo "seat ", index, " ", text + +proc sampleHeads*(logits: openArray[float32], temperature: float32, + state: var uint64): Heads = + ## One categorical draw per head, in head order, from softmax(logits / T), + ## using 53 random bits per draw (neural_basic.md, "Decisions, state and + ## telemetry"). + sampleHeads(logits, HeadSizes, temperature, state, result) + +proc staticTargets*(mask: ActionMask): array[ObjectSlots, bool] = + ## mask_mode "static": the union of the target rows of every allowed + ## target-reading verb (independent per-head masks, as PufferLib samples). + for s in 0 ..< ObjectSlots: + if mask[MaskVerb + 3] != 0 and mask[MaskTarget + s] != 0: result[s] = true + for a in 0 ..< 4: + if mask[MaskVerb + 4] != 0 and mask[MaskTarget + (1 + a) * ObjectSlots + s] != 0: + result[s] = true + if mask[MaskVerb + 5] != 0 and mask[MaskTarget + 5 * ObjectSlots + s] != 0: result[s] = true + if mask[MaskVerb + 7] != 0 and mask[MaskTarget + 6 * ObjectSlots + s] != 0: result[s] = true + +proc maskedHeads*(logits: openArray[float32], mask: ActionMask, + sampling: bool, temperature: float32, state: var uint64, + staticMode = false): Heads = + ## decoder.mask_empty_targets. Conditional (default): verb, then ability + ## (masked when verb is castTarget), then target by the (verb, ability) + ## row. Static: verb, and target by the union of the allowed verbs' rows. + ## Point and item are never masked. Sampling draws one uniform per head in + ## head order first (the same RNG use as sampleHeads), then resolves heads + ## in dependency order. Argmax: first maximum among allowed choices. + var offsets: array[ActionHeads, int] + var o = 0 + for h in 0 ..< ActionHeads: + offsets[h] = o + o += HeadSizes[h] + var draws: array[ActionHeads, float64] + if sampling: + for h in 0 ..< ActionHeads: + draws[h] = unitDraw(state) + var allowed: array[64, bool] + template pick(h: int): int32 = + pickHead(logits, offsets[h], HeadSizes[h], allowed, sampling, temperature, + draws[h]) + for i in 0 ..< 8: allowed[i] = mask[MaskVerb + i] != 0 + result[0] = pick(0) + let verb = result[0] + for i in 0 ..< 64: allowed[i] = true + if staticMode: + result[3] = pick(3) + let targets = staticTargets(mask) + for i in 0 ..< ObjectSlots: allowed[i] = targets[i] + result[1] = pick(1) + else: + if verb == 4: + for i in 0 ..< 4: allowed[i] = mask[MaskAbility + i] != 0 + result[3] = pick(3) + for i in 0 ..< 64: allowed[i] = true + let row = maskTargetRow(verb, result[3]) + if row >= 0: + for i in 0 ..< ObjectSlots: + allowed[i] = mask[MaskTarget + row * ObjectSlots + i] != 0 + result[1] = pick(1) + for i in 0 ..< 64: allowed[i] = true + result[2] = pick(2) + result[4] = pick(4) + +# GotA's NeuralContract: the callbacks the shared tier calls on GotA seats. +# They run inside a tick of activeGame (neuralPrelude, BASIC host calls). + +proc gotaObserve(brain: NeuralBrain): bool {.nimcall.} = + ## Observation contract v1 plus the decision frame (slot ids, positions). + let + seat = NeuralSeat(brain) + world = activeGame.world + buildObservation(world, seat.seat, seat.goal, seat.maxTicks, world.stats, + seat.obs, seat.frame) + seat.frame.alive and not world.gameOver + +proc gotaDecode(brain: NeuralBrain) {.nimcall.} = + ## Package seats: heads from the logits (mask, sample or argmax), then the + ## decoded command gota_act or the defer consult issues. + let seat = NeuralSeat(brain) + if seat.maskMode != NoMask: + seat.mask = actionMask(activeGame.world, seat.seat, seat.frame) + seat.heads = maskedHeads(seat.logits, seat.mask, seat.sampling, + seat.temperature, seat.rng, seat.maskMode == StaticMask) + else: + seat.heads = + if seat.sampling: sampleHeads(seat.logits, seat.temperature, seat.rng) + else: argmaxHeads(seat.logits) + seat.command = decodeAction(seat.frame, seat.heads) + +proc gotaHead(brain: NeuralBrain, index: int): int32 {.nimcall.} = + NeuralSeat(brain).heads[index] + +proc gotaTick(brain: NeuralBrain): int32 {.nimcall.} = + activeGame.world.tick + +proc gotaLog(brain: NeuralBrain, text: string) {.nimcall.} = + activeGame.seatLog(brain.seat, text) + +proc gotaTelemetry(brain: NeuralBrain): string {.nimcall.} = + ## GotA counters after the tier's telemetry line. + let seat = NeuralSeat(brain) + result = " decisions=" & $seat.decisions & " invalid=" & $seat.invalid + if seat.deferEnabled: + result.add " defer=" & $seat.deferDecisions & " override=" & + $seat.overrideDecisions + +proc gotaActionMask(brain: NeuralBrain, mask: var openArray[uint8]) {.nimcall.} = + ## The current frame's validity mask (layout: neural_basic.md, Action mask). + let seat = NeuralSeat(brain) + let computed = actionMask(activeGame.world, seat.seat, seat.frame) + for i in 0 ..< MaskSize: + mask[i] = computed[i] + +let GotaContract* = initNeuralContract("gota", PackageSchema, + observationContractText(), actionContractText(), ObservationSize, HeadSizes, + ActionOutputs, gotaObserve, gotaDecode, gotaHead, gotaTick, log = gotaLog, + telemetryExtra = gotaTelemetry, + maskSize = MaskSize, actionMask = gotaActionMask, + maxDecisionPeriod = MaxDecisionPeriod, opBudget = DefaultNeuralOpBudget, + manifestKeys = GotaManifestKeys, decoderKeys = GotaDecoderKeys, + parseOptions = parseGotaOptions) + ## gota-neural-basic/1: observation v1 (1407 floats), action v1 heads + ## [8, 25, 49, 4, 6], decision_period 1..24, 4,000,000 ops per inference. +doAssert GotaContract.observationHash == ObservationContractHash and + GotaContract.actionHash == ActionContractHash + +proc parseGotaPackage*(bytes: string): NeuralPackage = + ## Validates a gota-neural-basic/1 package; raises ValueError. + parsePackage(bytes, GotaContract) + +proc gotaOptions*(package: NeuralPackage): GotaPackageOptions = + GotaPackageOptions(package.options) + +proc newNeuralSeat*(mode: NeuralMode, period: int32, maxTicks: int32): NeuralSeat = + result = NeuralSeat(mode: mode, issuedTick: -1) + result.initBrain(GotaContract, period, maxTicks) + result.goal[0] = 1 + +proc resetEpisode*(seat: NeuralSeat, matchSeed: int32, index: int) = + ## Clears per-match state (a new match or a native reset). + seat.resetBrain(matchSeed, index) + seat.issuedTick = -1 + seat.captured.setLen(0) + seat.label = default(typeof(seat.label)) + seat.lastLabel = default(typeof(seat.lastLabel)) + seat.command = NeuralCommand() + seat.decisions = 0 + seat.invalid = 0 + seat.absorbing = false + seat.deferDecisions = 0 + seat.overrideDecisions = 0 + +proc beginDecision*(game: Game, index: int, seat: NeuralSeat) = + ## Captures the decision frame (observation, slots, resets) once per tick; + ## a package seat also infers and decodes (neural_host.think). + let world = game.world + if seat.mode == NeuralHosted: + # The match length as the match actually runs it. Read here, not at + # install: the hosted runner installs bots before it records the match + # config (game.config.maxTicks was 0 at install, so hosted seats saw a + # zero match length in their time features while the native env saw + # 28,800). + seat.maxTicks = game.config.maxTicks + if not seat.beginFrame(world.tick): + return + seat.command = NeuralCommand(tick: world.tick) + case seat.mode + of NeuralHosted: + if seat.acting: + seat.think(world.battleTick()) + of NeuralCapture, NeuralOverride: + seat.lastLabel = seat.label + seat.label = default(typeof(seat.label)) + seat.labelInstant = false + if seat.mode == NeuralCapture: + seat.captured.setLen(0) + of NeuralLearner: + if seat.shadow != nil or seat.deferEnabled: + seat.lastLabel = seat.label + seat.label = default(typeof(seat.label)) + seat.labelInstant = false + +proc setLearnerHeads*(seat: NeuralSeat, heads: Heads) = + ## The native trainer's action for the paused decision. + seat.heads = heads + seat.headsReady = true + seat.command = if seat.acting: decodeAction(seat.frame, heads) + else: NeuralCommand(tick: seat.frameTick) + +proc writeLabel(seat: NeuralSeat, world: World, command: NeuralCommand) = + ## Folds one script command into the window's label (first instant wins, + ## otherwise the last movement/attack order). + let instant = command.kind in {CastTargetCommand, CastPointCommand, + UseItemCommand, UseItemAtCommand} + if seat.labelInstant: + inc seat.label[13] + return + let encoded = encodeCommand(seat.frame, world, command) + let count = seat.label[13] + 1 + seat.label = default(typeof(seat.label)) + seat.label[13] = count + seat.labelInstant = instant + seat.label[0] = int32(encoded.represented and encoded.heads[0] != 0) + if encoded.represented: + for h in 0 ..< ActionHeads: + seat.label[1 + h] = encoded.heads[h] + seat.label[6] = int32(encoded.exact) + seat.label[7] = int32(command.kind.ord) + seat.label[8] = command.objectId + seat.label[9] = command.ability + seat.label[10] = command.item + let + wp = worldPointOf(command.point) + s = (if seat.frame.team == RedTeam: 1'i64 else: -1'i64) + if command.kind in {WalkCommand, AttackMoveCommand, CastPointCommand, UseItemAtCommand}: + seat.label[11] = int32(s * int64(wp.x) * 1000 div WorldScale) + seat.label[12] = int32(s * int64(wp.z) * 1000 div WorldScale) + seat.label[14] = encoded.errorMilli + seat.label[15] = command.tick + +proc interceptCommand*(game: Game, heroId: int32, command: NeuralCommand): bool = + ## Called by every contract host function before it applies an order. + ## True = the order was absorbed (override mode) and must not run. + if game.heroVms.len == 0: + return false + let index = game.world.heroIndex(heroId) + let seat = game.neuralSeat(index) + if seat == nil: + return false + var tagged = command + tagged.tick = game.world.tick + if shadowRunning: + if seat.frameTick >= 0: + seat.writeLabel(game.world, tagged) + return true + if seat.deferEnabled: + # Defer seat: the script is the seat's program. Its contract commands + # run live in a defer window and are absorbed in an override window. + if seat.mode == NeuralLearner and seat.frameTick >= 0: + seat.writeLabel(game.world, tagged) + return seat.absorbing + case seat.mode + of NeuralOverride: + seat.captured.add tagged + true + of NeuralCapture: + if seat.frameTick >= 0: + seat.writeLabel(game.world, tagged) + false + else: + false + +proc issueDecoded*(game: Game, heroId: int32, command: NeuralCommand): bool = + ## Executes a decoded contract command through the recorded host path. + let (x, y, offset) = splitTilePoint(command.point) + case command.kind + of NoCommand: false + of WalkCommand: game.issueWalkTo(heroId, x, y, offset) + of AttackMoveCommand: game.issueAttackMove(heroId, x, y, offset) + of AttackTargetCommand: game.issueAttackTarget(heroId, command.objectId) + of CastTargetCommand: game.issueCastTarget(heroId, command.ability, command.objectId) + of CastPointCommand: game.issueCastPoint(heroId, command.ability, x, y, offset) + of UseItemCommand: game.issueUseItem(heroId, command.item) + of UseItemAtCommand: game.issueUseItemAt(heroId, command.item, x, y, offset) + +proc actNow*(game: Game, index: int): int32 = + ## gota_act: issues the seat's decoded command once, on its decision tick. + let seat = game.neuralSeat(index) + if seat == nil or seat.mode notin {NeuralLearner, NeuralHosted}: + return 0 + let world = game.world + if seat.frameTick != world.tick or seat.issuedTick == world.tick or + not seat.acting or not seat.headsReady: + return 0 + seat.issuedTick = world.tick + inc seat.decisions + if isInvalid(seat.frame, seat.heads): + inc seat.invalid + int32(game.issueDecoded(world.heroes[index].id, seat.command)) + +proc deferConsult*(game: Game, index: int) = + ## Defer seats: once per decision tick, before the seat's script runs, + ## verb 0 opens a defer window (the script's contract commands execute) and + ## any other verb issues the decoded command and opens an override window + ## (the script's contract commands are absorbed until the next decision). + ## Shared by the native learner seat and the hosted package seat. + let seat = game.neuralSeat(index) + if seat == nil or not seat.deferEnabled: + return + let world = game.world + if seat.frameTick != world.tick or seat.issuedTick == world.tick: + return + seat.issuedTick = world.tick + if not seat.acting or not seat.headsReady: + seat.absorbing = false + return + inc seat.decisions + if seat.heads[0] != 0 and isInvalid(seat.frame, seat.heads): + # An override that decodes to nothing would absorb the script for the + # whole window and idle the hero: fall back to the script instead. + inc seat.invalid + seat.absorbing = false + inc seat.deferDecisions + return + if seat.heads[0] == 0: + seat.absorbing = false + inc seat.deferDecisions + return + seat.absorbing = true + inc seat.overrideDecisions + discard game.issueDecoded(world.heroes[index].id, seat.command) + +proc runOverride*(game: Game, index: int, seat: NeuralSeat) = + ## Mapping ceiling: route the script's queued orders through the contract. + let world = game.world + if seat.frameTick != world.tick: + return + if seat.captured.len == 0 or not seat.acting: + seat.captured.setLen(0) + return + for command in seat.captured: + seat.writeLabel(world, command) + seat.captured.setLen(0) + if seat.label[0] == 0: + return + var heads: Heads + for h in 0 ..< ActionHeads: + heads[h] = seat.label[1 + h] + seat.heads = heads + let decoded = decodeAction(seat.frame, heads) + inc seat.decisions + discard game.issueDecoded(world.heroes[index].id, decoded) + +proc neuralPrelude*(game: Game) = + ## Start of the heroes' turn: every neural seat observes the same frame. + let world = game.world + for index in 0 ..< game.heroVms.len: + let seat = game.neuralSeat(index) + if seat != nil and world.isDecisionTick(seat.period): + let vm = game.heroVms[index] + if vm.failed: + continue + try: + game.beginDecision(index, seat) + except BasicError as error: + vm.failed = true + vm.lastError = error.msg + when defined(coworld): + playerError(index, error.msg) + else: + echo "hero ", world.heroes[index].id, " neural error: ", error.msg + +proc addNeuralSeatFunctions*(host: var Host, heroId: int32) = + ## The policy.bas surface of a neural seat (package or learner): gota_act + ## plus the shared tier's run_neural_net / neuralObservation / + ## neuralLogits / neuralState / neuralModel. + let actProc: HostProc = proc(arguments: openArray[int32]): int32 = + ## Issues the network's decoded command on a decision tick. + activeGame.actNow(activeGame.world.heroIndex(heroId)) + discard host.addFunction("gota_act", 0, actProc, 800) + host.addNeuralHostFunctions(proc(): NeuralBrain = + activeGame.neuralSeat(activeGame.world.heroIndex(heroId))) + +proc neuralVmLimits*(): Limits = + ## Neural seats (package, learner, defer): the plain hero per-tick budget + ## (instructions, work units: fairness with .bas seats; the network runs on + ## its own separate op budget) with room for the neural host functions. + result = heroVmLimits() + result.maxHostFunctions = 160 + +proc deferVmLimits*(): Limits = + ## Defer-script seats: the same limits as every neural seat. + neuralVmLimits() diff --git a/examples/gods_of_the_arena/replays.nim b/examples/gods_of_the_arena/replays.nim index 69d1c212..2d9e2abf 100644 --- a/examples/gods_of_the_arena/replays.nim +++ b/examples/gods_of_the_arena/replays.nim @@ -15,7 +15,7 @@ const ReplayFormatVersion* = 6'u16 ## This client supports only this gameplay version. Bump it when rules change. ## Older replays use their archived client; never add compatibility branches. - ReplayGameVersion* = 63'u16 + ReplayGameVersion* = 64'u16 ActionWalkTo* = 1'u8 ActionAttackTarget* = 2'u8 ActionBuyItem* = 3'u8 diff --git a/examples/gods_of_the_arena/sim.nim b/examples/gods_of_the_arena/sim.nim index e9578b22..e4a79fd8 100644 --- a/examples/gods_of_the_arena/sim.nim +++ b/examples/gods_of_the_arena/sim.nim @@ -12,7 +12,7 @@ import std/algorithm, bassy, fixxy, polyworld/[bodies, hashes, metrics, noises, pathing, profiles, rngs, tapes, - visions], + visions, mailboxes], content, events, motions, maps, replays @@ -36,19 +36,29 @@ var lanePathPoints*: array[3, seq[PathPoint]] lanePathTiles: array[3, seq[PathTile]] laneWorldLayers: array[3, seq[int32]] - visionBlockers: seq[int16] - visionSources: seq[VisionSource] - visionSkipNow: seq[int32] - heroPathPoints: seq[PathPoint] - heroPathTiles: seq[PathTile] - gotaWalkLayer: int - gotaWalkDestLayer: int - gotaWalkOrigin: FixedVec2 + mapGlobalsHash: uint64 + mapGlobalsReady: bool + ## Map-derived globals (lane paths, sight terrain) are rebuilt only when + ## the map changes, so worlds on one map never write them concurrently. + +# Per-thread scratch: separate worlds may tick concurrently on separate threads. +var + visionSources {.threadvar.}: seq[VisionSource] + visionSkipNow {.threadvar.}: seq[int32] + heroPathPoints {.threadvar.}: seq[PathPoint] + heroPathTiles {.threadvar.}: seq[PathTile] + gotaWalkLayer {.threadvar.}: int + gotaWalkDestLayer {.threadvar.}: int + gotaWalkOrigin {.threadvar.}: FixedVec2 ## Simulation type Team* = enum RedTeam, BlueTeam + TrainingCounter* = enum + ## Training-library-only per-hero counters (-d:gotaTrainingStats). + TrainGold, TrainLastHits, TrainNeutralKills, TrainTowerDamage, + TrainStructureKills, TrainHeroDamage, TrainDamageTaken, TrainGodDamage MatchPhase* = enum Playing, Drafting FootmanState* = enum Marching, Fighting, Dying CampState* = enum RestingCamp, FightingCamp, ReturningCamp, EmptyCamp @@ -74,6 +84,10 @@ type lastError*: string decisions*: int lastWork*, lastInstructions*: int64 + neural*: RootRef + ## Neural seat state (a bots.nim NeuralSeat, which derives from + ## polyworld/neural_host.NeuralBrain); nil for plain BASIC seats. Kept + ## untyped here because this module must not import float code. Footman* = object id*: int32 @@ -237,6 +251,12 @@ type ends*: int32 resolved*: bool + TowerShot* = object + sourceId*, targetId*, damage*: int32 + team*: Team + previous*, position*: WorldPoint + started*, impact*: int32 + HitKind = enum UnitHit, StructureHit, GodHit HitKey = tuple[kind, class, x, y, z, spawnX, spawnY, spawnZ: int32] CombatHit = object @@ -259,8 +279,12 @@ type when defined(replayEvents): events*: seq[GameEvent] eventTick: int32 + when defined(gotaTrainingStats): + training*: seq[array[TrainingCounter, int64]] + ## Indexed like `heroes`; never hashed and never read by the sim. heroSpawns*: array[2, WorldPoint] casts*: seq[SpellCast] + towerShots*: seq[TowerShot] stats*: CombatStats ## One match. A ref so `a = b` aliases and a second world is `clone()`. footmen*: seq[Footman] @@ -298,6 +322,10 @@ type teamExplored*: array[2, seq[uint8]] visionCache: array[2, VisionCache] visionSkipKeys: seq[int32] + visionBlockers: seq[int16] + ## Occluder heights of the last rebuild of THIS world (camp sight reads + ## them); per world, not per thread, so worlds sharing a thread never + ## see each other's towers. scriptObjects: seq[WorldObject] scriptObjectCount: int scriptObjectsHeroId: int32 @@ -320,6 +348,7 @@ type replayMode*: bool recordingError*: string heroVms*: seq[HeroVm] + inboxes*: seq[Mailbox] nextFootmen: seq[Footman] nextHeroes: seq[Hero] collisionUnits: seq[CollisionUnit] @@ -327,8 +356,8 @@ type collisionOffsets: seq[FixedVec2] var - navigationWorld: World - gotaWalkTeam: Team + navigationWorld {.threadvar.}: World + gotaWalkTeam {.threadvar.}: Team const WorldScale* = 60_000'i32 @@ -343,17 +372,15 @@ const TowerDamages*: array[TowerTier, int32] = [36'i32, 48, 60] BarracksHitPoints* = 950'i32 TowerAttackRanges*: array[TowerTier, int32] = [ - 300_000'i32, - 330_000, - 360_000 + 540_000'i32, + 570_000, + 600_000 ] TowerAttackTicks* = TickRate + TowerShotStep* = 18 * WorldScale div TickRate + TowerImpactTicks* = TickRate div 2 TowerSiegeRange* = 105_000'i32 -proc towerSightTiles*(tier: TowerTier): int32 = - ## Shares a tower's reveal radius with portal destinations. - TowerAttackRanges[tier] div WorldScale + 2 - proc config*(game: Game): GotaConfig = ## Reads the match configuration owned by the live or loaded replay. if game.recorder != nil: @@ -672,11 +699,11 @@ proc addVisionSource(position: WorldPoint, radius: int32, eyeHeight: int16) = x: tile.x, z: tile.z, radius: radius, eyeHeight: eyeHeight ) -proc addVisionBlocker(position: WorldPoint, height: int16) = +proc addVisionBlocker(world: World, position: WorldPoint, height: int16) = ## Raises the occluder height in each cell touched by a structure's center. for tile in sightTiles(position): let index = tile.z * mapTiles() + tile.x - visionBlockers[index] = max(visionBlockers[index], height) + world.visionBlockers[index] = max(world.visionBlockers[index], height) proc fillVisionKeys(world: World, dest: var seq[int32]) = ## Records living observers and the towers or forts that occlude them. @@ -714,15 +741,15 @@ proc rebuildVision*(world: World) {.measure.} = world.fillVisionKeys(visionSkipNow) if sameVisionKeys(visionSkipNow, world.visionSkipKeys): return - visionBlockers.setLen(sightTerrain.blockerHeights.len) + world.visionBlockers.setLen(sightTerrain.blockerHeights.len) for i, value in sightTerrain.blockerHeights: - visionBlockers[i] = value + world.visionBlockers[i] = value for tower in world.buildings: if tower.hp > 0: - addVisionBlocker(tower.position, 28) + world.addVisionBlocker(tower.position, 28) for fort in world.forts: if fort.hp > 0: - addVisionBlocker(fort.center, 32) + world.addVisionBlocker(fort.center, 32) for team in Team: visionSources.setLen(0) for i in 0 ..< world.heroes.len: @@ -735,7 +762,17 @@ proc rebuildVision*(world: World) {.measure.} = addVisionSource(footman.position, FootmanSightRadius div WorldScale, 12) for tower in world.buildings: if tower.team == team and tower.hp > 0: - addVisionSource(tower.position, tower.tier.towerSightTiles, 24) + let range = TowerAttackRanges[tower.tier] + for tile in sightTiles(tower.position): + visionSources.add VisionSource( + x: tile.x, z: tile.z, + radius: (range + WorldScale - 1) div WorldScale + 1, + eyeHeight: 24, units: WorldScale, range: range, + offsetX: tower.position.x - + (tile.x - mapTiles().int32 div 2) * WorldScale - WorldScale div 2, + offsetZ: tower.position.z - + (tile.z - mapTiles().int32 div 2) * WorldScale - WorldScale div 2 + ) for fort in world.forts: if fort.team == team and fort.hp > 0: addVisionSource(fort.center, FortSightRadius, 28) @@ -745,7 +782,7 @@ proc rebuildVision*(world: World) {.measure.} = mapTiles().int32, mapTiles().int32, sightTerrain.terrainHeights, - visionBlockers, + world.visionBlockers, visionSources ) for i, value in world.teamVisible[team.ord]: @@ -783,6 +820,18 @@ proc heroIndex*(world: World, id: int32): int = return i -1 +proc trainCount(world: World, heroId: int32, counter: TrainingCounter, + amount: int64) {.inline.} = + ## Credits a training counter to a hero; compiled out of live builds. + when defined(gotaTrainingStats): + if amount == 0: + return + let index = world.heroIndex(heroId) + if index >= 0: + if world.training.len < world.heroes.len: + world.training.setLen(world.heroes.len) + world.training[index][counter] += amount + proc buildingIndex*(world: World, id: int32): int = ## Returns the tower slot for one id, or -1. if id == 0: @@ -985,9 +1034,20 @@ proc applyDamage[T: Hero | Footman]( detail = 0'i32 ) = ## Applies a unit hit and records its actual health change. - when T is Footman or defined(replayEvents): + when T is Footman or defined(replayEvents) or defined(gotaTrainingStats): let before = target.hp target.hp -= amount + when defined(gotaTrainingStats): + let removed = int64(max(before, 0'i32) - max(target.hp, 0'i32)) + when T is Hero: + world.trainCount(target.id, TrainDamageTaken, removed) + let attacker = world.heroIndex(source) + if attacker >= 0 and world.heroes[attacker].team != target.team: + world.trainCount(source, TrainHeroDamage, removed) + else: + if before > 0 and target.hp <= 0: + world.trainCount(source, + (if target.camp > 0: TrainNeutralKills else: TrainLastHits), 1) when T is Hero: if amount > 0: for kind in RecoveryKind: @@ -1142,9 +1202,14 @@ proc damageFort( ) = ## Applies damage only after the god's two guards have been destroyed. if damage > 0 and world.fortExposed(world.forts[index].team): - when defined(replayEvents): + when defined(replayEvents) or defined(gotaTrainingStats): let before = world.forts[index].hp world.forts[index].hp = max(0'i32, world.forts[index].hp - damage) + when defined(gotaTrainingStats): + let attacker = world.heroIndex(source) + if attacker >= 0 and world.heroes[attacker].team != world.forts[index].team: + world.trainCount(source, TrainGodDamage, + int64(max(before, 0'i32) - world.forts[index].hp)) when defined(replayEvents): world.damageEvent(source, world.forts[index].id, damage, before, world.forts[index].hp, cause, detail) @@ -1295,6 +1360,8 @@ proc gainRewards( hero.xp += xp hero.totalXp += xp hero.gold += gold + when defined(gotaTrainingStats): + world.trainCount(hero.id, TrainGold, int64(gold)) while hero.level < HeroMaxLevel and hero.xp >= xpForNextLevel(hero.level): hero.xp -= xpForNextLevel(hero.level) @@ -1654,9 +1721,16 @@ proc damageBuilding( cause = BasicAttack, detail = 0'i32 ) = ## Applies building damage and releases occupied tiles on the killing hit. - when defined(replayEvents): + when defined(replayEvents) or defined(gotaTrainingStats): let before = world.buildings[index].hp world.buildings[index].hp -= damage + when defined(gotaTrainingStats): + let attacker = world.heroIndex(source) + if attacker >= 0 and world.heroes[attacker].team != world.buildings[index].team: + world.trainCount(source, TrainTowerDamage, + int64(max(before, 0'i32) - max(world.buildings[index].hp, 0'i32))) + if before > 0 and world.buildings[index].hp <= 0: + world.trainCount(source, TrainStructureKills, 1) when defined(replayEvents): world.damageEvent(source, world.buildings[index].id, damage, before, world.buildings[index].hp, cause, detail) @@ -2541,6 +2615,17 @@ proc setHeroDestination( hero.hasMoveTarget = true true +proc planRoute*(world: World, hero: Hero, mapX, mapY: int32, + offset = FixedVec2Zero): seq[WorldPoint] = + ## Returns the path a walk order to this tile would give the hero now, + ## without changing the hero (planning runs on a copy). Empty = no route. + navigationWorld = world + var probe = Hero() + probe[] = hero[] + probe.hasMoveTarget = false + if probe.setHeroDestination(int(mapX), int(mapY), hero.position.y, offset): + result = probe.movePath[probe.movePathIndex .. ^1] + proc stopHeroPath(hero: Hero) = ## Drops the finished chase so a hero can acquire nearby creeps again. hero.hasMoveTarget = false @@ -2672,7 +2757,7 @@ proc portalLanding*( towerId: var int32, anchorId = 0'i32 ): bool = - ## Finds the closest visible, open landing within a living allied tower's sight. + ## Finds the closest visible, open landing within an allied tower's range. navigationWorld = world let orientation = if team == RedTeam: -1'i64 else: 1'i64 var best = (int64.high, int64.high, int64.high, @@ -2682,7 +2767,8 @@ proc portalLanding*( (anchorId != 0 and tower.id != anchorId): continue let - radius = tower.tier.towerSightTiles.int + range = TowerAttackRanges[tower.tier] + radius = int((range + WorldScale - 1) div WorldScale) centerX = floorWorldTile(tower.position.x, team) + GridTiles div 2 centerZ = floorWorldTile(tower.position.z, team) + GridTiles div 2 for layerIndex, layer in layers: @@ -2700,7 +2786,7 @@ proc portalLanding*( point.x = aim.x point.z = aim.z discard layerFixedHeight(layerIndex, point, point.y, team) - if not within(point, tower.position, radius.int32 * WorldScale) or + if not within(point, tower.position, range) or not world.visible(team, point) or not inWalkMargin(toPlanar(point)): continue let rank = ( @@ -2815,7 +2901,7 @@ proc applyAttackMove*(world: World, heroId, mapX, mapY: int32, offset ) -proc isEnemyTarget(world: World, hero: Hero, targetId: int32): bool = +proc isEnemyTarget*(world: World, hero: Hero, targetId: int32): bool = ## Returns whether `targetId` is a living enemy the hero can chase. if targetId == 0: return false @@ -3034,11 +3120,12 @@ proc startSwing(world: World, hero: Hero) = hero.damageLanded = false proc updateTower*(world: World, tower: var Building) = - ## Acquires one nearby enemy and applies a deterministic periodic attack. + ## Reloads independently of acquisition and fires a homing shot when ready. if tower.kind == BarracksBuilding or tower.hp <= 0: tower.targetId = 0 tower.attackTicks = 0 return + tower.attackTicks = min(TowerAttackTicks, tower.attackTicks + 1) let attackRange = TowerAttackRanges[tower.tier] var targetFootman = footmanIndex(world, tower.targetId) @@ -3096,21 +3183,57 @@ proc updateTower*(world: World, tower: var Building) = if targetFootman >= 0: world.footmen[targetFootman].id elif targetHero >= 0: world.heroes[targetHero].id else: 0'i32 - if tower.targetId != targetId: - tower.targetId = targetId - tower.attackTicks = 0 - if targetId == 0: + tower.targetId = targetId + if targetId == 0 or tower.attackTicks < TowerAttackTicks: return - inc tower.attackTicks - if tower.attackTicks < TowerAttackTicks: - return - tower.attackTicks -= TowerAttackTicks - let damage = TowerDamages[tower.tier] - if targetFootman >= 0: - world.hitTarget(UnitHit, world.footmen[targetFootman].id, - damage, tower.id) - else: - world.hitTarget(UnitHit, world.heroes[targetHero].id, damage, tower.id) + tower.attackTicks = 0 + var origin = tower.position + origin.y += 2 * WorldScale + world.towerShots.add TowerShot( + sourceId: tower.id, targetId: targetId, team: tower.team, + damage: TowerDamages[tower.tier], previous: origin, position: origin, + started: world.tick + ) + +proc advanceTowerShots*(world: World) = + ## Tracks the original living target, ignoring range and vision after firing. + var write = 0 + for original in world.towerShots: + var shot = original + if shot.impact > 0: + if world.tick - shot.impact >= TowerImpactTicks: + continue + elif shot.started < world.tick: + var target: WorldPoint + let + hero = world.heroIndex(shot.targetId) + creep = world.footmanIndex(shot.targetId) + if hero >= 0: + if world.heroes[hero].hp <= 0 or world.heroes[hero].state == Dying: + continue + target = world.heroes[hero].position + elif creep >= 0: + if world.footmen[creep].hp <= 0 or world.footmen[creep].state == Dying: + continue + target = world.footmen[creep].position + else: + continue + target.y += WorldScale + shot.previous = shot.position + if within(shot.position, target, TowerShotStep): + shot.position = target + shot.impact = world.tick + world.hitTarget(UnitHit, shot.targetId, shot.damage, shot.sourceId) + else: + let + offset = target - shot.position + distance = integerSqrt(distanceSquared(shot.position, target)) + shot.position.x += int32(int64(offset.x) * TowerShotStep div distance) + shot.position.y += int32(int64(offset.y) * TowerShotStep div distance) + shot.position.z += int32(int64(offset.z) * TowerShotStep div distance) + world.towerShots[write] = shot + inc write + world.towerShots.setLen(write) proc spellTarget*(world: World, id: int32, value: var WorldObject): bool ## Resolves one live unit or structure for spells and camp leashes. @@ -3199,12 +3322,12 @@ proc spawnCamp(world: World, index: int) = world.lifecycleEvent(EntitySpawned, 0, unit.id, if camp.started: Respawn else: Initialization) -proc campCanSee(unit: Footman, point: WorldPoint, radius: int32): bool = +proc campCanSee(world: World, unit: Footman, point: WorldPoint, radius: int32): bool = ## Tests the mob's own terrain-occluded sight without granting team vision. # Exact world range is checked before allowing for rounded tile centers. within(unit.position, point, radius * WorldScale) and lineVisible( mapTiles().int32, mapTiles().int32, - sightTerrain.terrainHeights, visionBlockers, + sightTerrain.terrainHeights, world.visionBlockers, mapCoordinate(unit.position.x, unit.team), mapCoordinate(unit.position.z, unit.team), mapCoordinate(point.x, unit.team), mapCoordinate(point.z, unit.team), @@ -3242,7 +3365,7 @@ proc provokeCamp(world: World, index: int) = continue let distance = distanceSquared(unit.position, point) if distance > best or - not unit.campCanSee(point, NeutralAggroTiles): + not world.campCanSee(unit, point, NeutralAggroTiles): continue if distance < best or targetBefore(point, candidateId, targetPoint, targetId, unit.team) or @@ -3329,7 +3452,7 @@ proc updateCamps(world: World) = var seen = false for unit in world.footmen: if unit.camp == index + 1 and unit.hp > 0 and - unit.campCanSee(target.position, 12): + world.campCanSee(unit, target.position, 12): seen = true break if seen: @@ -3363,7 +3486,7 @@ proc updateNeutral(world: World, unit: var Footman) = template consider(id: int32, at: WorldPoint) = ## Chooses the nearest visible hero or lane creep while defending a camp. let radius = if id == camp.targetId: 12'i32 else: 5'i32 - if within(at, camp.center, NeutralLeash) and unit.campCanSee(at, radius): + if within(at, camp.center, NeutralLeash) and world.campCanSee(unit, at, radius): let distance = distanceSquared(unit.position, at) if distance < best or (distance == best and targetBefore(at, id, point, targetId, unit.team)): @@ -5335,6 +5458,16 @@ proc stateHash*(game: Game): uint64 = hash.addHashy(spell.impact) hash.addHashy(spell.ends) hash.addHashy(spell.resolved) + hash.addHashy(world.towerShots.len) + for shot in world.towerShots: + hash.addHashy(shot.sourceId) + hash.addHashy(shot.targetId) + hash.addHashy(shot.damage) + hash.addHashy(shot.team.ord) + hash.addHashy(shot.previous) + hash.addHashy(shot.position) + hash.addHashy(shot.started) + hash.addHashy(shot.impact) hash.addHashy(world.stats) uint64(hash) @@ -5363,18 +5496,28 @@ proc finishTick(game: Game) = except ReplayError as error: game.recordingError = error.msg -proc tickWorld*(game: Game, onHeroTurn: proc() {.closure.}) {.measure.} = - ## Advances exactly one authoritative integer simulation tick. +type TickStage* = enum + ## How far `tickWorldBegin` took the current tick. + TickSkipped, ## Nothing ran: the match is over or the replay is exhausted. + TickDone, ## A draft tick ran to completion. + TickHeroTurn, ## Paused before the heroes' turn; call `tickWorldFinish`. + TickNoTurn ## Paused on a tick without a heroes' turn; call `tickWorldFinish`. + +proc tickWorldFinish*(game: Game) + +proc tickWorldBegin*(game: Game, onDraftTurn: proc() {.closure.}): TickStage = + ## Runs a tick up to (not including) the heroes' decisions. Battle ticks + ## stop with observations frozen; `tickWorldFinish` completes them. let world = game.world when defined(replayEvents): world.events.setLen(0) world.eventTick = world.tick + 1 world.syncBuildings() if game.finished(): - return + return TickSkipped if game.replayMode and world.tick >= game.replayData.hashes.len: - return + return TickSkipped if world.phase == Drafting: inc world.tick @@ -5391,8 +5534,8 @@ proc tickWorld*(game: Game, onHeroTurn: proc() {.closure.}) {.measure.} = while game.replayPlayer.takeActionAt(uint32(world.tick), action): if world.applyReplayAction(action): game.metrics.command(world.heroIndex(action.heroId), world.tick) - elif decide and onHeroTurn != nil: - onHeroTurn() + elif decide and onDraftTurn != nil: + onDraftTurn() if world.phase == Drafting and world.draftTicksLeft() == 0: var available: seq[HeroClass] for class in HeroClass: @@ -5401,7 +5544,7 @@ proc tickWorld*(game: Game, onHeroTurn: proc() {.closure.}) {.measure.} = let class = available[world.rng.below(available.len.int32)] discard world.applyDraft(world.draftHeroId(), class.ord.int32, TimeLimit) game.finishTick() - return + return TickDone dec world.spawnTimerTicks if world.spawnTimerTicks <= 0: @@ -5417,27 +5560,29 @@ proc tickWorld*(game: Game, onHeroTurn: proc() {.closure.}) {.measure.} = profileBlock "vision": rebuildVision(world) world.updateKnownBuildings() - block: - discard world.freezeObservations() - defer: - world.thawObservations() + discard world.freezeObservations() + if game.historyPlayback: + if game.recorder != nil: + game.replayPlayer.data = game.recorder.data + var action: ReplayAction + while game.replayPlayer.takeActionAt(uint32(world.tick), action): + if applyReplayAction(world, action): + game.metrics.command(heroIndex(world, action.heroId), world.tick) + dec world.heroTurnTicks + if world.heroTurnTicks <= 0: + world.heroTurnTicks += DecisionTicks if game.historyPlayback: - if game.recorder != nil: - game.replayPlayer.data = game.recorder.data - var action: ReplayAction - while game.replayPlayer.takeActionAt(uint32(world.tick), action): - if applyReplayAction(world, action): - game.metrics.command(heroIndex(world, action.heroId), world.tick) - dec world.heroTurnTicks - if world.heroTurnTicks <= 0: - world.heroTurnTicks += DecisionTicks - if game.historyPlayback: - world.heroTurnStart = (world.heroTurnStart + 1) mod world.heroes.len - else: - profileBlock "decisions": - if onHeroTurn != nil: - onHeroTurn() + world.heroTurnStart = (world.heroTurnStart + 1) mod world.heroes.len + else: + return TickHeroTurn + TickNoTurn +proc tickWorldFinish*(game: Game) = + ## Completes a battle tick that `tickWorldBegin` paused. + let world = game.world + # Another world may have ticked on this thread while this one was paused. + navigationWorld = world + world.thawObservations() world.updateCamps() # Plan every unit against the same actor state, then publish together. @@ -5467,6 +5612,7 @@ proc tickWorld*(game: Game, onHeroTurn: proc() {.closure.}) {.measure.} = world.heroes[i][] = hero[] world.advanceSpells() + world.advanceTowerShots() world.resolveCombat() world.updateCamps() @@ -5546,6 +5692,23 @@ proc tickWorld*(game: Game, onHeroTurn: proc() {.closure.}) {.measure.} = game.finishTick() +proc tickWorld*(game: Game, onHeroTurn: proc() {.closure.}) {.measure.} = + ## Advances exactly one authoritative integer simulation tick. + case game.tickWorldBegin(onHeroTurn) + of TickSkipped, TickDone: + discard + of TickHeroTurn: + try: + profileBlock "decisions": + if onHeroTurn != nil: + onHeroTurn() + except CatchableError as error: + game.world.thawObservations() + raise error + game.tickWorldFinish() + of TickNoTurn: + game.tickWorldFinish() + proc initLanePaths(map: MapData) = ## Samples symmetric lane goals without baking live buildings into roads. proc gate(lane: int, team: Team): ArenaStop = @@ -5673,7 +5836,9 @@ proc newGame*( world.rng = initRng(map.seed) world.matchSeed = map.seed initTowers(world, map) - initLanePaths(map) + let rebuildMapGlobals = not mapGlobalsReady or mapGlobalsHash != map.hash + if rebuildMapGlobals: + initLanePaths(map) for team in Team: var point = worldPoint(map.layout.spawns[team.ord]) point.y = fixedSurfaceHeight(point, team) @@ -5714,21 +5879,25 @@ proc newGame*( ) world.initOccupancy() world.initCamps(map) - sightTerrain = buildSightTerrain() + if rebuildMapGlobals: + sightTerrain = buildSightTerrain() let visionCells = mapTiles() * mapTiles() for team in Team: world.teamVisible[team.ord] = newSeq[uint8](visionCells) world.teamExplored[team.ord] = newSeq[uint8](visionCells) - for lane in 0 .. 2: - laneWorldPaths[lane].setLen(0) - laneWorldLayers[lane].setLen(0) - for tile in lanePathTiles[lane]: - laneWorldPaths[lane].add worldPoint(pathPoint( - int(tile.layer), - int(tile.x), - int(tile.z) - )) - laneWorldLayers[lane].add tile.layer + if rebuildMapGlobals: + for lane in 0 .. 2: + laneWorldPaths[lane].setLen(0) + laneWorldLayers[lane].setLen(0) + for tile in lanePathTiles[lane]: + laneWorldPaths[lane].add worldPoint(pathPoint( + int(tile.layer), + int(tile.x), + int(tile.z) + )) + laneWorldLayers[lane].add tile.layer + mapGlobalsHash = map.hash + mapGlobalsReady = true for fort in world.forts.mitems: fort.center.y = fixedSurfaceHeight(fort.center, fort.team) let heroSetup = diff --git a/examples/gods_of_the_arena/spelleffects.nim b/examples/gods_of_the_arena/spelleffects.nim index f7b75630..d62d100e 100644 --- a/examples/gods_of_the_arena/spelleffects.nim +++ b/examples/gods_of_the_arena/spelleffects.nim @@ -20,6 +20,7 @@ type meshes: array[Ability, SpellMesh] glow: SpellMesh trail: SpellMesh + fireball: SpellMesh proc initSpellRenderer*(): SpellRenderer = ## Creates the shared spell mesh shader and reusable geometry buffer. @@ -267,6 +268,59 @@ proc matches(area: AreaMesh, spell: SpellCast): bool = area.ability == spell.ability and area.position == spell.position and area.direction == spell.direction +proc drawTowerShots( + effects: var SpellRenderer, + world: World, + viewProjection: Mat4, + alpha: float32, + viewMode: int32 +) = + ## Draws homing fireballs and impacts directly from replay simulation state. + var settings = defaultFxSettings() + settings.shape = SphereShape + settings.pivot = CenterPivot + settings.radius = 0.22 + settings.radialSegments = 16 + settings.heightSegments = 8 + settings.texture = NoiseTexture + settings.blendMode = AdditiveBlend + settings.startColor = vec4(1.0, 0.85, 0.25, 1.0) + settings.endColor = vec4(1.0, 0.16, 0.02, 0.6) + settings.scroll = vec2(0.7, -1.8) + settings.waveAmp = 0.04 + settings.waveFreq = 5 + settings.waveSpeed = 6 + if effects.fireball.vertices.len == 0: + effects.fireball = buildFxMesh(settings) + for shot in world.towerShots: + if viewMode > 0 and shot.team != Team(viewMode - 1) and + not world.visible(Team(viewMode - 1), shot.position): + continue + let + age = (world.tick.float32 + alpha - shot.started.float32) / + TickRate.float32 + impact = shot.impact > 0 + fade = + if impact: + clamp(1 - (world.tick.float32 + alpha - shot.impact.float32) / + TowerImpactTicks.float32, 0, 1) + else: + 1.0'f + position = + if impact: + shot.position.spellPoint() + else: + mix(shot.previous.spellPoint(), shot.position.spellPoint(), alpha) + size = if impact: 1 + (1 - fade) * 3 else: 1.0'f + base = translate(position) + settings.startColor.w = fade + settings.endColor.w = fade * 0.6 + effects.renderer.uploadFxMesh(effects.fireball) + effects.renderer.drawFxMesh( + settings, viewProjection, base, age, 0.2, sizeScale = size + ) + effects.drawGlow(viewProjection, base, vec3(1.0, 0.5, 0.04), age, fade) + proc drawSpells*( effects: var SpellRenderer, world: World, @@ -278,6 +332,7 @@ proc drawSpells*( let time = world.tick.float32 + alpha seconds = time / TickRate.float32 + effects.drawTowerShots(world, viewProjection, alpha, viewMode) var write = 0 for area in effects.areas: for spell in world.casts: @@ -476,3 +531,4 @@ proc closeSpellRenderer*(effects: var SpellRenderer) = effects.meshes = default(typeof(effects.meshes)) effects.glow = default(SpellMesh) effects.trail = default(SpellMesh) + effects.fireball = default(SpellMesh) diff --git a/examples/gods_of_the_arena/tools/canary.py b/examples/gods_of_the_arena/tools/canary.py new file mode 100644 index 00000000..55c93e7f --- /dev/null +++ b/examples/gods_of_the_arena/tools/canary.py @@ -0,0 +1,92 @@ +"""Local hosted canary: the -d:coworld GotA server plays a full match with five +neural-package seats (random GOTANET1 weights) and five base.bas seats through the platform's +file handoff (COGAME_* URIs), then the recorded replay is re-simulated by the headless binary. +Pass = results.json written, every seat's player status exit_code 0, every neural seat log has +the `neural: peak_ops=... budget=... model=w... ticks=... inferences=... decisions=... invalid=...` +telemetry line (plus `defer=... override=...` in defer mode) on every line, and the replay re-simulates with zero hash mismatches. +Usage: python3 canary.py COWORLD_BIN HEADLESS_BIN WORKDIR [HIDDEN] [MAX_TICKS] [plain|defer] + defer: the neural seats are defer seats (decoder.defer_script, policy.bas = base.bas). +""" +import hashlib, json, os, re, socket, subprocess, sys, time, urllib.parse +import numpy as np +HERE = os.path.dirname(os.path.abspath(__file__)) +ROOT = os.path.abspath(os.path.join(HERE, "../../..")) +sys.path.insert(0, os.path.join(ROOT, "coworld/gota/runtime")) +import neural_package as npk + + +def uri(path): + return "file://" + urllib.parse.quote(path) + + +def main(): + coworld, headless, work = sys.argv[1:4] + hidden = int(sys.argv[4]) if len(sys.argv) > 4 else 128 + max_ticks = int(sys.argv[5]) if len(sys.argv) > 5 else 28800 + mode = sys.argv[6] if len(sys.argv) > 6 else "plain" + os.makedirs(work, exist_ok=True) + rng = np.random.default_rng(7) + n = 1407 * hidden + 3 * hidden * hidden + 92 * hidden + model = npk.encode_model((rng.standard_normal(n) * 0.05).astype(np.float32).tolist(), hidden) + policy = open(os.path.join(ROOT, "examples/gods_of_the_arena/neural/policy.bas"), "rb").read() + base = open(os.path.join(ROOT, "examples/gods_of_the_arena/players/base.bas"), "rb").read() + pkg = npk.build(base, model, decoder={"defer_script": True}) if mode == "defer" else npk.build(policy, model) + sources = [pkg if slot in (0, 2, 4, 6, 8) else base for slot in range(10)] # 5 neural, both teams + config = {"tokens": [], "players": [], "seed": 2026, "max_ticks": max_ticks} + seats = [] + for slot, src in enumerate(sources): + path = os.path.join(work, f"player-{slot}") + open(path, "wb").write(src) + config["tokens"].append(f"token-{slot}") + config["players"].append({"name": f"CANARY-{slot}"}) + seats.append({"slot": slot, "file_uri": uri(path), "content_hash": "sha256:" + hashlib.sha256(src).hexdigest(), + "size_bytes": len(src), "log_uri": uri(os.path.join(work, f"player-{slot}.log")), + "artifact_uri": uri(os.path.join(work, f"player-{slot}.zip"))}) + doc = {"schema": "coworld-player-seats/1", "seats": seats, "player_status_uri": uri(os.path.join(work, "status.json"))} + env = dict(os.environ) + s = socket.socket(); s.bind(("127.0.0.1", 0)); port = s.getsockname()[1]; s.close() + env.update(COGAME_HOST="127.0.0.1", COGAME_PORT=str(port)) + for key, value in (("CONFIG", config), ("PLAYER_SEATS", doc)): + p = os.path.join(work, key + ".json"); json.dump(value, open(p, "w")); env[f"COGAME_{key}_URI"] = uri(p) + for key, name in (("RESULTS", "results.json"), ("SAVE_REPLAY", "replay"), ("PLAYER_FAILURE", "failure.json")): + env[f"COGAME_{key}_URI"] = uri(os.path.join(work, name)) + for f in ("results.json", "failure.json", "status.json", "replay"): + if os.path.exists(os.path.join(work, f)): + os.remove(os.path.join(work, f)) + t = time.time() + proc = subprocess.Popen([coworld], cwd=ROOT, env=env, stdout=open(os.path.join(work, "game.log"), "w"), stderr=subprocess.STDOUT) + ok = True + try: + while not os.path.exists(os.path.join(work, "results.json")): + if os.path.exists(os.path.join(work, "failure.json")) or proc.poll() is not None: + print("FAIL game ended without results:", open(os.path.join(work, "game.log")).read()[-2000:]) + sys.exit(1) + if time.time() - t > 3600: + print("FAIL timeout"); sys.exit(1) + time.sleep(0.5) + elapsed = time.time() - t + results = json.load(open(os.path.join(work, "results.json"))) + status = json.load(open(os.path.join(work, "status.json"))) + finally: + proc.terminate(); proc.wait(10) + codes = [p["exit_code"] for p in status["players"]] + print("results", json.dumps({k: results[k] for k in results if k != "total_xp"}), "total_xp", results.get("total_xp")) + print("player exit codes", codes, f"match seconds {elapsed:.0f}") + ok &= all(c == 0 for c in codes) and len(codes) == 10 + for slot in (0, 2, 4, 6, 8): + log = open(os.path.join(work, f"player-{slot}.log")).read() + lines = [l for l in log.splitlines() if l.startswith("neural: ")] + print(f"seat {slot} telemetry:", lines[-1] if lines else "MISSING") + want = (r"^neural: peak_ops=\d+ budget=4000000 model=w%d ticks=\d+ inferences=\d+ decisions=\d+ invalid=\d+" + % hidden) + (r" defer=\d+ override=\d+$" if mode == "defer" else "$") + ok &= bool(lines) and all(re.match(want, l) for l in lines) and "BASIC error" not in log + out = subprocess.run([headless, "--replay", os.path.join(work, "replay")], cwd=ROOT, capture_output=True, text=True) + tail = [l for l in out.stdout.splitlines() if l.startswith(("hash:", "replay", "result"))] + print("replay re-simulation:", tail, "rc", out.returncode) + ok &= out.returncode == 0 and not any("mismatches" in l for l in tail) + print("CANARY PASS" if ok else "CANARY FAIL") + sys.exit(0 if ok else 1) + + +if __name__ == "__main__": + main() diff --git a/examples/gods_of_the_arena/tools/mapping_ceiling.py b/examples/gods_of_the_arena/tools/mapping_ceiling.py new file mode 100644 index 00000000..e8a23033 --- /dev/null +++ b/examples/gods_of_the_arena/tools/mapping_ceiling.py @@ -0,0 +1,100 @@ +"""Mapping ceiling: play base.bas through the action contract and compare with plain base.bas. + +For every seed three matches run to max_ticks: A = ten plain base.bas seats; R = red seats 0-4 +under gota_set_seat_override (their contract commands are encoded to the 5 heads, decoded and +executed exactly as a learner seat's would be); B = blue seats 5-9 under override. The ceiling +holds when override teams keep plain base.bas's results: win rate and per-seat XP/score within +noise of A. Usage: python3 mapping_ceiling.py LIB SEEDS [MAX_TICKS] [PROCS] [OUT.json] +""" +import json, os, sys, time +from multiprocessing import Pool +import numpy as np +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +from native_env import Env, Lib + +LIB = None + + +def play(job): + seed, mode, max_ticks = job + lib = Lib(LIB) + env = Env(lib, learner_seats=[], max_ticks=max_ticks) + seats = {"A": [], "R": range(0, 5), "B": range(5, 10)}[mode] + for s in seats: + env.set_override(s, True) + env.reset(seed) + kinds, exact, errs, labeled = np.zeros(8, np.int64), np.zeros(8, np.int64), [], 0 + noop = np.zeros((10, 5), np.int32) + while True: + r = env.step(noop) + for s in seats: + o = env.orders(s) + if o[13] > 0: + labeled += int(o[0]) + kinds[o[7]] += 1 + exact[o[7]] += int(o[6]) + if o[7] in (1, 2, 5, 7): + errs.append(int(o[14])) + if r == 1: + break + res = env.results() + stats = [env.stats(s) for s in range(10)] + env.close() + return dict(seed=seed, mode=mode, winner=int(res[1]), ticks=int(res[4]), + xp=[int(x[2]) for x in stats], score=[int(x[0]) for x in stats], + kills=[int(x[4]) for x in stats], deaths=[int(x[6]) for x in stats], + kinds=kinds.tolist(), exact=exact.tolist(), labeled=labeled, + err_median=float(np.median(errs)) if errs else 0.0, + err_p90=float(np.percentile(errs, 90)) if errs else 0.0) + + +def init(lib): + global LIB + LIB = lib + + +def ci(x): + x = np.asarray(x, float) + return float(x.mean()), float(1.96 * x.std(ddof=1) / np.sqrt(len(x))) if len(x) > 1 else 0.0 + + +def main(): + lib, nseeds = sys.argv[1], int(sys.argv[2]) + max_ticks = int(sys.argv[3]) if len(sys.argv) > 3 else 28800 + procs = int(sys.argv[4]) if len(sys.argv) > 4 else os.cpu_count() + out = sys.argv[5] if len(sys.argv) > 5 else None + jobs = [(seed, m, max_ticks) for seed in range(1, nseeds + 1) for m in "ARB"] + t = time.time() + with Pool(procs, initializer=init, initargs=(lib,)) as pool: + rows = pool.map(play, jobs, chunksize=1) + by = {m: [r for r in rows if r["mode"] == m] for m in "ARB"} + def team(r, t): + return sum(r["xp"][5 * t:5 * t + 5]) / 5.0 + def tscore(r, t): + return sum(r["score"][5 * t:5 * t + 5]) / 5.0 + summary = {"seeds": nseeds, "max_ticks": max_ticks, "seconds": round(time.time() - t, 1)} + # Red-side comparison: A red vs R red; blue-side: A blue vs B blue. + for label, mode, t_ in (("red", "R", 0), ("blue", "B", 1)): + base, over = by["A"], by[mode] + summary[label] = { + "plain_win": ci([r["winner"] == t_ for r in base]), + "override_win": ci([r["winner"] == t_ for r in over]), + "plain_xp": ci([team(r, t_) for r in base]), + "override_xp": ci([team(r, t_) for r in over]), + "plain_score": ci([tscore(r, t_) for r in base]), + "override_score": ci([tscore(r, t_) for r in over]), + "paired_xp_delta": ci([team(o, t_) - team(b, t_) for o, b in zip(over, base)]), + } + k = np.sum([r["kinds"] for r in by["R"] + by["B"]], 0) + e = np.sum([r["exact"] for r in by["R"] + by["B"]], 0) + names = ["none", "walk", "attackMove", "attackTarget", "castTarget", "castPoint", "useItem", "useItemAt"] + summary["commands"] = {names[i]: {"n": int(k[i]), "exact": round(float(e[i]) / max(1, k[i]), 3)} for i in range(8)} + summary["point_err_median_milli"] = float(np.median([r["err_median"] for r in by["R"] + by["B"]])) + summary["point_err_p90_milli"] = float(np.median([r["err_p90"] for r in by["R"] + by["B"]])) + print(json.dumps(summary, indent=1)) + if out: + json.dump({"summary": summary, "rows": rows}, open(out, "w")) + + +if __name__ == "__main__": + main() diff --git a/examples/gods_of_the_arena/tools/native_env.py b/examples/gods_of_the_arena/tools/native_env.py new file mode 100644 index 00000000..b4664981 --- /dev/null +++ b/examples/gods_of_the_arena/tools/native_env.py @@ -0,0 +1,250 @@ +"""ctypes binding for libgota_env (native_env.h), used by the GotA neural tools.""" +import ctypes, json, os +import numpy as np + +SEATS, HEADS, STATS, ORDERS = 10, 5, 24, 16 +MASK_VERB, MASK_ABILITY, MASK_TARGET, MASK_SIZE = 0, 8, 12, 187 + + +def mask_target_row(verb, ability): + """Target row of (verb, ability) in the action mask; -1 = target unused.""" + return {3: 0, 5: 5, 7: 6}.get(verb, 1 + ability if verb == 4 and 0 <= ability < 4 else -1) + + +def static_masks(mask): + """mask_mode "static" (independent per-head masks): (verb bool[8], target bool[25]). The target mask is + the union of the rows of the allowed target-reading verbs.""" + verb = mask[0:8] != 0 + rows = mask[12:].reshape(7, 25) != 0 + target = np.zeros(25, bool) + if verb[3]: target |= rows[0] + if verb[4]: target |= rows[1:5].any(0) + if verb[5]: target |= rows[5] + if verb[7]: target |= rows[6] + return verb, target + + +def masked_argmax(logits, mask, static=False): + """The host's decoder.mask_empty_targets argmax (conditional: verb, ability, target by row; static: + verb + union target mask; point, item free).""" + import numpy as _np + def pick(lo, n, allowed): + seg = logits[lo:lo + n] + ok = _np.asarray(allowed, bool) + if not ok.any(): + ok[:] = True + best = -1 + for i in range(n): + if ok[i] and (best < 0 or seg[i] > seg[best]): + best = i + return best + verb = pick(0, 8, mask[0:8] != 0) + if static: + _, tmask = static_masks(mask) + return [verb, pick(8, 25, tmask), pick(33, 49, [True] * 49), pick(82, 4, [True] * 4), pick(86, 6, [True] * 6)] + ability = pick(82, 4, (mask[8:12] != 0) if verb == 4 else [True] * 4) + row = mask_target_row(verb, ability) + target = pick(8, 25, (mask[12 + row * 25:12 + row * 25 + 25] != 0) if row >= 0 else [True] * 25) + point = pick(33, 49, [True] * 49) + item = pick(86, 6, [True] * 6) + return [verb, target, point, ability, item] +HEAD_SIZES = [8, 25, 49, 4, 6] +STAT_NAMES = ["score", "outcome", "xp", "gold", "hero_kills", "assists", "deaths", "last_hits", + "neutral_kills", "tower_damage", "structure_kills", "hero_damage", "damage_taken", + "push_depth", "god_damage", "level", "gold_now", "team", "class", "alive", + "decisions", "invalid_actions", "basic_max_instructions", "reserved"] + + +class NativeEnvError(RuntimeError): + """A negative return code from libgota_env (code = the return value).""" + def __init__(self, call, code, message=""): + super().__init__(f"{call} returned {code}" + (f": {message}" if message else "")) + self.call, self.code, self.message = call, code, message + + +class Lib: + def __init__(self, path=None): + path = path or os.environ.get("GOTA_ENV_LIB", "libgota_env.so") + L = self.L = ctypes.CDLL(path) + vp, f32p, i32p, i64p = ctypes.c_void_p, ctypes.POINTER(ctypes.c_float), ctypes.POINTER(ctypes.c_int32), ctypes.POINTER(ctypes.c_int64) + L.gota_create.restype = vp + L.gota_create.argtypes = [ctypes.c_char_p, ctypes.c_char_p, ctypes.c_int32] + L.gota_destroy.argtypes = [vp] + L.gota_reset.argtypes = [vp, ctypes.c_int64] + L.gota_observe_seats.argtypes = [vp, ctypes.c_uint32, f32p, f32p, f32p] + L.gota_step.argtypes = [vp, i32p, f32p, f32p] + L.gota_results.argtypes = [vp, f32p] + L.gota_state_hash.restype = ctypes.c_uint64 + L.gota_state_hash.argtypes = [vp] + L.gota_seat_stats.argtypes = [vp, ctypes.c_int, i64p] + L.gota_set_seat_goal.argtypes = [vp, ctypes.c_int, f32p] + L.gota_set_seat_script.argtypes = [vp, ctypes.c_int, ctypes.c_char_p, ctypes.c_int32] + L.gota_seat_orders.argtypes = [vp, ctypes.c_int, i32p] + L.gota_set_seat_override.argtypes = [vp, ctypes.c_int, ctypes.c_int32] + L.gota_set_seat_shadow.argtypes = [vp, ctypes.c_int, ctypes.c_char_p, ctypes.c_int32] + L.gota_set_policy_script.argtypes = [vp, ctypes.c_char_p, ctypes.c_int32] + if hasattr(L, "gota_set_seat_defer_script"): # absent in libs built before defer scripts + L.gota_set_seat_defer_script.argtypes = [vp, ctypes.c_int, ctypes.c_char_p] + L.gota_seat_defer_stats.argtypes = [vp, ctypes.c_int, i64p] + if hasattr(L, "gota_action_mask"): + L.gota_action_mask.argtypes = [vp, ctypes.c_int, ctypes.POINTER(ctypes.c_uint8)] + L.gota_set_seat_package.argtypes = [vp, ctypes.c_int, ctypes.c_char_p, ctypes.c_int64] + L.gota_seat_script_status.argtypes = [vp, ctypes.c_int, ctypes.c_char_p, ctypes.c_int32] + L.gota_set_learner_seats.argtypes = [vp, ctypes.c_uint32] + L.gota_battle_tick.argtypes = [vp] + L.gota_save_replay.argtypes = [vp, ctypes.c_char_p] + L.gota_last_error.argtypes = [ctypes.c_char_p, ctypes.c_int32] + L.gota_net_load.restype = vp + L.gota_net_load.argtypes = [ctypes.c_char_p, ctypes.c_int64, ctypes.c_char_p, ctypes.c_int32] + L.gota_net_infer.argtypes = [vp, f32p, f32p, f32p] + L.gota_net_destroy.argtypes = [vp] + L.gota_net_info.argtypes = [vp, i64p] + L.gota_observation_contract_hash.argtypes = [ctypes.c_char_p, ctypes.c_int32] + L.gota_action_contract_hash.argtypes = [ctypes.c_char_p, ctypes.c_int32] + if hasattr(L, "gota_mask_size"): + L.gota_mask_size.restype = ctypes.c_int + self.obs_size = L.gota_observation_size() + # The constants above must match the library (a stale binding or library fails here). + heads = (ctypes.c_int32 * 32)() + n = L.gota_action_heads(heads) + assert self.obs_size == 1407, f"observation size {self.obs_size} != 1407" + assert list(heads[:n]) == HEAD_SIZES and sum(heads[:n]) == 92, f"heads {list(heads[:n])} != {HEAD_SIZES}" + assert L.gota_stat_count() == STATS, "stat count mismatch" + if hasattr(L, "gota_mask_size"): + assert L.gota_mask_size() == MASK_SIZE, f"mask size {L.gota_mask_size()} != {MASK_SIZE}" + + def hashes(self): + a, b = ctypes.create_string_buffer(65), ctypes.create_string_buffer(65) + self.L.gota_observation_contract_hash(a, 65) + self.L.gota_action_contract_hash(b, 65) + return a.value.decode(), b.value.decode() + + def last_error(self): + b = ctypes.create_string_buffer(1024) + self.L.gota_last_error(b, 1024) + return b.value.decode() + + def check(self, call, r): + """Raises NativeEnvError on any negative return code; returns r otherwise.""" + if r < 0: + raise NativeEnvError(call, r, self.lib_error(call, r)) + return r + + def lib_error(self, call, r): + return self.last_error() if r in (-2, -3) or call.startswith("gota_net") else "" + + +def ptr(a, t): + return a.ctypes.data_as(ctypes.POINTER(t)) + + +class Env: + def __init__(self, lib, **config): + self.lib, self.L = lib, lib.L + err = ctypes.create_string_buffer(1024) + self.h = self.L.gota_create(json.dumps(config).encode(), err, 1024) + if not self.h: + raise RuntimeError(err.value.decode()) + self.obs = np.zeros((SEATS, lib.obs_size), np.float32) + self.resets = np.zeros(SEATS, np.float32) + self.acting = np.zeros(SEATS, np.float32) + self.rewards = np.zeros(SEATS, np.float32) + self.terms = np.zeros(SEATS, np.float32) + + def _c(self, call, r): + return self.lib.check(call, r) + + def reset(self, seed): + """0; raises NativeEnvError on -2 (a seat failed to compile: see status) or -3.""" + return self._c("gota_reset", self.L.gota_reset(self.h, seed)) + + def observe(self, mask=0x3FF): + self._c("gota_observe_seats", self.L.gota_observe_seats( + self.h, mask, ptr(self.obs, ctypes.c_float), ptr(self.resets, ctypes.c_float), + ptr(self.acting, ctypes.c_float))) + return self.obs, self.resets, self.acting + + def step(self, actions): + """0 paused, 1 over; raises NativeEnvError on -1/-2/-3 (after -3, reset before stepping).""" + a = np.ascontiguousarray(actions, np.int32) + return self._c("gota_step", self.L.gota_step(self.h, ptr(a, ctypes.c_int32), + ptr(self.rewards, ctypes.c_float), ptr(self.terms, ctypes.c_float))) + + def results(self): + out = np.zeros(8, np.float32) + self._c("gota_results", self.L.gota_results(self.h, ptr(out, ctypes.c_float))) + return out + + def stats(self, seat): + out = np.zeros(STATS, np.int64) + self._c("gota_seat_stats", self.L.gota_seat_stats(self.h, seat, ptr(out, ctypes.c_int64))) + return out + + def orders(self, seat): + out = np.zeros(ORDERS, np.int32) + self._c("gota_seat_orders", self.L.gota_seat_orders(self.h, seat, ptr(out, ctypes.c_int32))) + return out + + def state_hash(self): + return self.L.gota_state_hash(self.h) + + def set_script(self, seat, source, check=True): + """0 ok, 1 compile failed (see status). Always raises NativeEnvError on + a negative code; also raises on a compile failure (code 1) unless + `check=False`, in which case the caller inspects `status(seat)`.""" + b = source.encode() + r = self._c("gota_set_seat_script", + self.L.gota_set_seat_script(self.h, seat, b, len(b))) + if check and r != 0: + raise NativeEnvError("gota_set_seat_script", r, self.status(seat)[1]) + return r + + def set_override(self, seat, on): + return self._c("gota_set_seat_override", self.L.gota_set_seat_override(self.h, seat, int(on))) + + def set_defer_script(self, seat, path): + """Defer script: verb 0 defers to this script on a learner seat (None = off).""" + return self._c("gota_set_seat_defer_script", + self.L.gota_set_seat_defer_script(self.h, seat, path.encode() if path else None)) + + def defer_stats(self, seat): + out = np.zeros(2, np.int64) + self._c("gota_seat_defer_stats", self.L.gota_seat_defer_stats(self.h, seat, ptr(out, ctypes.c_int64))) + return out + + def action_mask(self, seat): + """uint8[187]: verb[8] | castTarget ability[4] | 7 target rows x 25 (native_env.h).""" + out = np.zeros(MASK_SIZE, np.uint8) + self._c("gota_action_mask", self.L.gota_action_mask(self.h, seat, ptr(out, ctypes.c_uint8))) + return out + + def set_package(self, seat, data, check=True): + """0 ok, 1 compile failed, 2 package rejected (see status). Always + raises NativeEnvError on a negative code; also raises on 1/2 unless + `check=False`, in which case the caller inspects `status(seat)`.""" + r = self._c("gota_set_seat_package", + self.L.gota_set_seat_package(self.h, seat, data, len(data))) + if check and r != 0: + raise NativeEnvError("gota_set_seat_package", r, self.status(seat)[1]) + return r + + def set_goal(self, seat, w): + w = np.ascontiguousarray(w, np.float32) + return self._c("gota_set_seat_goal", self.L.gota_set_seat_goal(self.h, seat, ptr(w, ctypes.c_float))) + + def status(self, seat): + b = ctypes.create_string_buffer(2048) + code = self._c("gota_seat_script_status", self.L.gota_seat_script_status(self.h, seat, b, 2048)) + return code, b.value.decode() + + def save_replay(self, path): + return self._c("gota_save_replay", self.L.gota_save_replay(self.h, path.encode())) + + def close(self): + if self.h: + self.L.gota_destroy(self.h) + self.h = None + + +def random_actions(rng, n=SEATS): + return np.stack([rng.integers(0, s, n) for s in HEAD_SIZES], 1).astype(np.int32) diff --git a/examples/gods_of_the_arena/tools/parity_tier.py b/examples/gods_of_the_arena/tools/parity_tier.py new file mode 100644 index 00000000..4571997d --- /dev/null +++ b/examples/gods_of_the_arena/tools/parity_tier.py @@ -0,0 +1,158 @@ +"""Byte-identity battery between two builds of libgota_env. Run it on a build host, not a laptop. + + python3 parity_tier.py OLD_LIB NEW_LIB [--seeds 20] [--ticks 28800] [--jobs 16] [--only L0,L1] + +For every seed and lineup, both libraries play the same full match. The battery compares every +step's state hash, the recorded replay bytes and the per-seat counters (stats; defer and override +counts). Each (library, seed, lineup) runs in its own process. Lineups: + +L0 no neural seats: base / puller / rusher scripts +L1 hosted packages: random w64 and w128 argmax packages, a w64 sampling package, base.bas elsewhere +L2 defer: an always-defer package and a mixed defer/override package (policy.bas = base.bas) +L3 action mask: conditional and static mask packages +L4 trainer seats: two learners on random actions (one with a shadow script), an override seat, + capture on, and a package seat +L5 always-defer only: two always-defer packages (no override is ever chosen) +""" +import argparse, hashlib, json, os, sys, tempfile +from concurrent.futures import ProcessPoolExecutor +import numpy as np + +HERE = os.path.dirname(os.path.abspath(__file__)) +sys.path.insert(0, HERE) +sys.path.insert(0, os.path.join(HERE, "../../../coworld/gota/runtime")) +ROOT = os.path.join(HERE, "..") + + +def read(path, mode="rb"): + with open(os.path.join(ROOT, path), mode) as f: + return f.read() + + +def weights(hidden, seed, scale=0.05, verb0_bias=0.0): + rng = np.random.default_rng(seed) + n_enc, n_rec, n_dec = 1407 * hidden, 3 * hidden * hidden, 92 * hidden + w = (rng.standard_normal(n_enc + n_rec + n_dec) * scale).astype(np.float32) + w[n_enc + n_rec: n_enc + n_rec + hidden] += verb0_bias + return w.tolist() + + +def packages(): + import neural_package as npk + base, policy = read("players/base.bas"), read("neural/policy.bas") + m64, m128 = npk.encode_model(weights(64, 1), 64), npk.encode_model(weights(128, 2), 128) + always = npk.encode_model(weights(64, 0, scale=0.0, verb0_bias=1.0), 64) + mixed = npk.encode_model(weights(128, 7, scale=0.05, verb0_bias=0.02), 128) + return { + "w64": npk.build(policy, m64), + "w128": npk.build(policy, m128), + "sample64": npk.build(policy, m64, decoder={"mode": "sample", "temperature": 0.7}), + "always_defer": npk.build(base, always, decoder={"defer_script": True}), + "mixed_defer": npk.build(base, mixed, decoder={"defer_script": True}), + "mask_conditional": npk.build(policy, m64, decoder={"mask_empty_targets": True}), + "mask_static": npk.build(policy, m128, decoder={"mask_empty_targets": True, "mask_mode": "static"}), + } + + +LINEUPS = ("L0", "L1", "L2", "L3", "L4", "L5") + + +def configure(Env, lib, lineup, seed, ticks, pk): + puller, rusher = read("players/puller.bas", "r"), read("players/rusher.bas", "r") + learners = [1, 6] if lineup == "L4" else [] + env = Env(lib, learner_seats=learners, max_ticks=ticks, record=True, capture=lineup == "L4") + s = seed % 10 + seats = [(s + k) % 10 for k in range(10)] + if lineup == "L0": + env.set_script(seats[0], puller); env.set_script(seats[1], rusher); env.set_script(seats[5], puller) + elif lineup == "L1": + env.set_package(seats[0], pk["w64"]); env.set_package(seats[3], pk["w128"]) + env.set_package(seats[6], pk["sample64"]); env.set_script(seats[8], rusher) + elif lineup == "L2": + env.set_package(seats[0], pk["always_defer"]); env.set_package(seats[5], pk["mixed_defer"]) + elif lineup == "L3": + env.set_package(seats[0], pk["mask_conditional"]); env.set_package(seats[7], pk["mask_static"]) + elif lineup == "L5": + env.set_package(seats[0], pk["always_defer"]); env.set_package(seats[5], pk["always_defer"]) + elif lineup == "L4": + assert env.set_script(2, puller) == 0 and env.set_script(8, rusher) == 0 + assert env.set_override(5, 1) == 0 + env.lib.L.gota_set_seat_shadow(env.h, 6, read("players/base.bas"), len(read("players/base.bas"))) + env.set_package(3, pk["w64"]) + for seat in range(10): + code, message = env.status(seat) + assert code in (0, 1), f"{lineup} seat {seat}: status {code} {message}" + return env + + +def job(args): + lib_path, lineup, seed, ticks = args + from native_env import Env, Lib, random_actions + lib = Lib(lib_path) + env = configure(Env, lib, lineup, seed, ticks, packages()) + env.reset(seed) + rng = np.random.default_rng(seed) + hashes = hashlib.sha256() + steps = 0 + while True: + actions = np.zeros((10, 5), np.int32) + if lineup == "L4": + env.observe() + actions = random_actions(rng) + r = env.step(actions) + hashes.update(int(env.state_hash()).to_bytes(8, "little")) + steps += 1 + if r == 1: + break + fd, path = tempfile.mkstemp(suffix=".replay"); os.close(fd) + env.save_replay(path) + replay = open(path, "rb").read(); os.unlink(path) + stats = [env.stats(s).tolist() for s in range(10)] + defer = [env.defer_stats(s).tolist() for s in range(10)] if hasattr(lib.L, "gota_seat_defer_stats") else [] + final = int(env.state_hash()) + env.close() + return dict(lib=lib_path, lineup=lineup, seed=seed, steps=steps, final=f"{final:016x}", + hashes=hashes.hexdigest(), replay=hashlib.sha256(replay).hexdigest(), replay_bytes=len(replay), + stats=hashlib.sha256(json.dumps(stats).encode()).hexdigest(), defer=defer) + + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("old") + ap.add_argument("new") + ap.add_argument("--seeds", type=int, default=20) + ap.add_argument("--first-seed", type=int, default=1) + ap.add_argument("--ticks", type=int, default=28800) + ap.add_argument("--jobs", type=int, default=16) + ap.add_argument("--only", default=",".join(LINEUPS)) + ap.add_argument("--json", default="") + a = ap.parse_args() + lineups = [x for x in a.only.split(",") if x] + seeds = range(a.first_seed, a.first_seed + a.seeds) + work = [(lib, ln, s, a.ticks) for ln in lineups for s in seeds for lib in (a.old, a.new)] + with ProcessPoolExecutor(a.jobs) as ex: + results = list(ex.map(job, work)) + by = {(r["lineup"], r["seed"], r["lib"]): r for r in results} + rows, same_all = [], True + for ln in lineups: + same_n = 0 + for s in seeds: + o, n = by[(ln, s, a.old)], by[(ln, s, a.new)] + same = all(o[k] == n[k] for k in ("steps", "final", "hashes", "replay", "stats", "defer")) + same_n += same + diff = [k for k in ("steps", "final", "hashes", "replay", "stats", "defer") if o[k] != n[k]] + print(f"{ln} seed {s:3d} steps {n['steps']:5d} final {n['final']} replay {n['replay'][:12]} " + f"{'IDENTICAL' if same else 'DIFFERENT ' + ','.join(diff)} defer={n['defer'] and [d for d in n['defer'] if d != [0, 0]]}") + rows.append((ln, same_n, len(seeds))) + same_all &= same_n == len(seeds) + print("\nlineup identical") + for ln, k, n in rows: + print(f"{ln} {k}/{n}") + if a.json: + json.dump(results, open(a.json, "w"), indent=1) + print("ALL IDENTICAL" if same_all else "DIFFERENCES FOUND") + sys.exit(0 if same_all else 1) + + +if __name__ == "__main__": + main() diff --git a/examples/gods_of_the_arena/tools/parity_upstream.sh b/examples/gods_of_the_arena/tools/parity_upstream.sh new file mode 100755 index 00000000..bbb3c59e --- /dev/null +++ b/examples/gods_of_the_arena/tools/parity_upstream.sh @@ -0,0 +1,55 @@ +#!/usr/bin/env bash +# No-neural-seat parity against upstream: for SEEDS full-length matches with a mixed +# base/puller/rusher lineup, (1) this branch's headless binary and upstream/main's headless binary +# must write byte-identical replays (every tick's state hash plus every action), and (2) the native +# training library with the same lineup and no neural seats (capture on) must record a replay that +# upstream/main's binary verifies tick for tick and that ends on the same final hash. +# Usage: parity_upstream.sh WORKDIR FORK_REV UPSTREAM_REV [SEEDS] [JOBS] (POLYWORLD_DEPS set) +set -euo pipefail +W=$1; FORK=$2; UP=$3; SEEDS=${4:-20}; JOBS=${5:-16} +SRC=$(git rev-parse --show-toplevel) +rm -rf "$W" && mkdir -p "$W"/{fork,up,out} +git -C "$SRC" archive "$FORK" | tar -x -C "$W/fork" +git -C "$SRC" archive "$UP" | tar -x -C "$W/up" +FLAGS="--hints:off -d:release -d:headless" +(cd "$W/up" && nim c $FLAGS --nimcache:"$W/nc-up" -o:"$W/gota_up" examples/gods_of_the_arena/gota.nim) & +(cd "$W/fork" && nim c $FLAGS --nimcache:"$W/nc-fk" -o:"$W/gota_fork" examples/gods_of_the_arena/gota.nim) & +(cd "$W/fork" && nim c $FLAGS --app:lib -d:gotaTrainingStats --mm:atomicArc --threads:on -d:useMalloc -u:nimTypeNames \ + --nimcache:"$W/nc-lib" -o:"$W/libgota_env.so" examples/gods_of_the_arena/native_env.nim) & +wait +P=examples/gods_of_the_arena/players +LINEUP="--bot $P/base.bas:4 --bot $P/puller.bas:3 --bot $P/rusher.bas:3" +run() { + s=$1 + (cd "$W/up" && "$W/gota_up" $LINEUP --seed "$s" --record "$W/out/up-$s.replay" > "$W/out/up-$s.log" 2>&1) + (cd "$W/fork" && "$W/gota_fork" $LINEUP --seed "$s" --record "$W/out/fork-$s.replay" > "$W/out/fork-$s.log" 2>&1) + (cd "$W/fork" && python3 - "$W" "$s" <<'PY' > "$W/out/native-$s.log" 2>&1 +import sys, numpy as np +W, s = sys.argv[1], int(sys.argv[2]) +sys.path.insert(0, W + "/fork/examples/gods_of_the_arena/tools") +from native_env import Env, Lib +P = W + "/fork/examples/gods_of_the_arena/players/" +env = Env(Lib(W + "/libgota_env.so"), learner_seats=[], record=True, capture=True) +for seat in range(10): + name = "base" if seat < 4 else "puller" if seat < 7 else "rusher" + assert env.set_script(seat, open(P + name + ".bas").read()) == 0 +env.reset(s) +while env.step(np.zeros((10, 5), np.int32)) == 0: + pass +print("final_hash %016x" % env.state_hash()) +assert env.save_replay(W + "/out/native-%d.replay" % s) == 0 +PY + ) + vrc=0; (cd "$W/up" && "$W/gota_up" --replay "$W/out/native-$s.replay" > "$W/out/verify-$s.log" 2>&1) || vrc=$? + same=$(cmp -s "$W/out/up-$s.replay" "$W/out/fork-$s.replay" && echo yes || echo no) + uph=$(grep -o 'hash: [0-9A-Fa-f]*' "$W/out/up-$s.log" | awk '{print tolower($2)}') + nh=$(grep -o 'final_hash [0-9a-f]*' "$W/out/native-$s.log" | awk '{print $2}') + vh=$(grep -o 'hash: [0-9A-Fa-f]*' "$W/out/verify-$s.log" | awk '{print tolower($2)}') + mism=$(grep -o 'replay hashes: [0-9]* mismatches' "$W/out/verify-$s.log" || true) + ver=0; [[ $vrc -eq 0 && -z "$mism" && "$vh" == "$uph" && "$nh" == "$uph" ]] && ver=1 + echo "seed=$s fork_replay_identical=$same upstream_final=$uph native_final=$nh replayed_final=$vh native_verified_by_upstream=$ver $mism" +} +export -f run; export W LINEUP +seq 1 "$SEEDS" | xargs -P "$JOBS" -I{} bash -c 'run {}' | sort -t= -k2 -n | tee "$W/parity.txt" +echo "identical_replays=$(grep -c 'fork_replay_identical=yes' "$W/parity.txt")/$SEEDS" +echo "native_verified=$(grep -c 'native_verified_by_upstream=1' "$W/parity.txt")/$SEEDS" diff --git a/examples/gods_of_the_arena/tools/replay_diff.nim b/examples/gods_of_the_arena/tools/replay_diff.nim new file mode 100644 index 00000000..24f09c58 --- /dev/null +++ b/examples/gods_of_the_arena/tools/replay_diff.nim @@ -0,0 +1,27 @@ +## Compares two GotA replays: setup, config, first differing tick hash and +## first differing action. nim r tools/replay_diff.nim A.replay B.replay +import std/[os, strutils], jsony, ../replays + +let a = loadReplay(paramStr(1)) +let b = loadReplay(paramStr(2)) +echo "setup equal: ", a.header.setup == b.header.setup +if a.header.setup != b.header.setup: + echo " A setup: ", a.header.setup.toJson() + echo " B setup: ", b.header.setup.toJson() +echo "config equal: ", a.config.toJson() == b.config.toJson() +if a.config.toJson() != b.config.toJson(): + echo " A config: ", a.config.toJson() + echo " B config: ", b.config.toJson() +echo "hashes: ", a.hashes.len, " vs ", b.hashes.len +for i in 0 ..< min(a.hashes.len, b.hashes.len): + if a.hashes[i] != b.hashes[i]: + echo "first hash mismatch at tick ", i + break +echo "actions: ", a.actions.len, " vs ", b.actions.len +for i in 0 ..< min(a.actions.len, b.actions.len): + if a.actions[i] != b.actions[i]: + echo "first action mismatch #", i + for j in max(0, i - 3) .. min(i + 3, min(a.actions.len, b.actions.len) - 1): + echo " A ", a.actions[j].toJson() + echo " B ", b.actions[j].toJson() + break diff --git a/examples/gods_of_the_arena/tools/test_defer.py b/examples/gods_of_the_arena/tools/test_defer.py new file mode 100644 index 00000000..d46b53df --- /dev/null +++ b/examples/gods_of_the_arena/tools/test_defer.py @@ -0,0 +1,270 @@ +"""Defer-script proofs. Run on a build host, not a laptop. + + python3 test_defer.py LIB [--seeds N] [--ticks T] [--jobs J] [--old-lib OLD] [--only a,b,c] + +a always-defer: a net whose verb-0 logit dominates, in a learner seat with gota_set_seat_defer_script(base.bas) + (ABI) and in a hosted package seat (decoder.defer_script, policy.bas = base.bas verbatim), plays + byte-identical to plain base.bas in that seat: every step's state hash and the recorded replay bytes. +b ABI == package on a net that mixes defer and override: per-step hashes, defer/override counts. +d invalid overrides fall back to defer: a learner defer seat (ABI) that picks attackTarget on an empty + object slot at every decision plays byte-identical to plain base.bas (hashes and + replay bytes), with 0 overrides and every such decision counted invalid. With --old-lib, the same run + on OLD is reported too (before the fix the seat idled through those windows). +c seats without the option (learners, shadow, capture, override, plain package) are byte-identical to + OLD (the lib built from the branch head before the defer change): per-step hashes and replay bytes. +""" +import argparse, ctypes, hashlib, os, sys, tempfile +from concurrent.futures import ProcessPoolExecutor +import numpy as np +HERE = os.path.dirname(os.path.abspath(__file__)) +sys.path.insert(0, HERE) +sys.path.insert(0, os.path.join(HERE, "../../../coworld/gota/runtime")) +from native_env import Env, Lib, random_actions, HEAD_SIZES, MASK_TARGET +import neural_package as npk + +ROOT = os.path.join(HERE, "..") +BASE = open(os.path.join(ROOT, "players/base.bas"), "rb").read() +POLICY = open(os.path.join(ROOT, "neural/policy.bas"), "rb").read() +F32 = ctypes.POINTER(ctypes.c_float) + + +def model_weights(hidden, seed, scale, verb0_bias): + """Encoder/recurrent ~ N(0, scale); decoder row 0 (verb 0) gets +verb0_bias on every unit. + scale 0 + bias 1: x = 0, y > 0 for every unit, so logit 0 = bias * sum(y) > 0 = every other logit.""" + rng = np.random.default_rng(seed) + n_enc, n_rec, n_dec = 1407 * hidden, 3 * hidden * hidden, 92 * hidden + w = (rng.standard_normal(n_enc + n_rec + n_dec) * scale).astype(np.float32) + w[n_enc + n_rec: n_enc + n_rec + hidden] += verb0_bias + return npk.encode_model(w.tolist(), hidden) + + +ALWAYS = dict(hidden=64, seed=0, scale=0.0, verb0_bias=1.0) +MIXED = dict(hidden=128, seed=7, scale=0.05, verb0_bias=0.02) + + +def argmax_heads(logits): + out, o = [], 0 + for s in HEAD_SIZES: + out.append(int(np.argmax(logits[o:o + s]))); o += s + return out + + +class Driver: + """Drives one learner seat through gota_net_infer (the trainer's view).""" + def __init__(self, lib, model, seat): + err = ctypes.create_string_buffer(512) + self.lib, self.seat = lib, seat + self.net = lib.L.gota_net_load(model, len(model), err, 512) + assert self.net, err.value + info = np.zeros(8, np.int64) + lib.L.gota_net_info(self.net, info.ctypes.data_as(ctypes.POINTER(ctypes.c_int64))) + self.state = np.zeros(int(info[5]), np.float32) + self.logits = np.zeros(92, np.float32) + + def act(self, env, actions): + obs, res, act = env.observe(1 << self.seat) + if act[self.seat]: + if res[self.seat]: + self.state[:] = 0 + o = np.ascontiguousarray(obs[self.seat]) + assert self.lib.L.gota_net_infer(self.net, o.ctypes.data_as(F32), self.state.ctypes.data_as(F32), + self.logits.ctypes.data_as(F32)) == 0 + actions[self.seat] = argmax_heads(self.logits) + + +def play(env, seed, driver=None, rng=None, watch=None): + """Full episode: per-step hashes. watch = seat whose BASIC instructions (last tick of each step) are + sampled into env.peak_instr.""" + env.reset(seed) + hashes = [] + env.peak_instr = 0 + while True: + actions = np.zeros((10, 5), np.int32) + if rng is not None: + env.observe() + actions = random_actions(rng) + if driver is not None: + driver.act(env, actions) + r = env.step(actions) + hashes.append(env.state_hash()) + if watch is not None: + env.peak_instr = max(env.peak_instr, int(env.stats(watch)[22])) + if r == 1: + return hashes + + +def replay_bytes(env): + fd, path = tempfile.mkstemp(suffix=".replay"); os.close(fd) + assert env.save_replay(path) == 0 + data = open(path, "rb").read(); os.unlink(path) + return data + + +def job_a(lib_path, seed, ticks): + lib = Lib(lib_path) + seat = seed % 10 + cfg = dict(max_ticks=ticks, record=True, capture=False) + model = model_weights(**ALWAYS) + ref = Env(lib, learner_seats=[], **cfg) + h_ref = play(ref, seed, watch=seat); r_ref = replay_bytes(ref) + instr = ref.peak_instr + abi = Env(lib, learner_seats=[seat], **cfg) + assert abi.set_defer_script(seat, "players/base.bas") == 0 + drv = Driver(lib, model, seat) + h_abi = play(abi, seed, drv); r_abi = replay_bytes(abi) + st_abi = abi.defer_stats(seat).tolist() + pkg = Env(lib, learner_seats=[], **cfg) + assert pkg.set_package(seat, npk.build(BASE, model, decoder={"defer_script": True})) == 0 + h_pkg = play(pkg, seed); r_pkg = replay_bytes(pkg) + st_pkg = pkg.defer_stats(seat).tolist() + return dict(seed=seed, seat=seat, steps=len(h_ref), ticks=int(ref.results()[4]), + final=f"{h_ref[-1]:016x}", + abi_hashes=h_abi == h_ref, abi_replay=r_abi == r_ref, + pkg_hashes=h_pkg == h_ref, pkg_replay=r_pkg == r_ref, + abi_defer=st_abi, pkg_defer=st_pkg, ref_seat_peak_instr_sampled=int(instr), + replay_sha=hashlib.sha256(r_ref).hexdigest()[:12]) + + +def job_b(lib_path, seed, ticks): + lib = Lib(lib_path) + seat = (seed * 3) % 10 + cfg = dict(max_ticks=ticks, capture=False) + model = model_weights(**MIXED) + abi = Env(lib, learner_seats=[seat], **cfg) + assert abi.set_defer_script(seat, "players/base.bas") == 0 + h_abi = play(abi, seed, Driver(lib, model, seat)) + pkg = Env(lib, learner_seats=[], **cfg) + assert pkg.set_package(seat, npk.build(BASE, model, decoder={"defer_script": True})) == 0 + h_pkg = play(pkg, seed) + ref = Env(lib, learner_seats=[], **cfg) + h_ref = play(ref, seed) + sa, sp = abi.defer_stats(seat).tolist(), pkg.defer_stats(seat).tolist() + return dict(seed=seed, seat=seat, steps=len(h_abi), same=h_abi == h_pkg, abi_defer=sa, pkg_defer=sp, + counts_same=sa == sp, invalid=[int(abi.stats(seat)[21]), int(pkg.stats(seat)[21])], + diverges_from_plain=h_abi != h_ref, + xp=[int(abi.stats(seat)[2]), int(ref.stats(seat)[2])]) + + +class InvalidDriver: + """Every acting decision: attackTarget (verb 3) on an empty object slot (its castPoint mask bit, set for + every occupied slot, is 0); verb 0 (a real defer) when no slot is empty.""" + def __init__(self, seat): + self.seat, self.invalid_picks, self.defers = seat, 0, 0 + + def act(self, env, actions): + _, _, act = env.observe(1 << self.seat) + if act[self.seat]: + row = env.action_mask(self.seat)[MASK_TARGET + 5 * 25:MASK_TARGET + 6 * 25] + empty = np.flatnonzero(row == 0) + if len(empty): + actions[self.seat] = [3, int(empty[-1]), 0, 0, 0] + self.invalid_picks += 1 + else: + self.defers += 1 + + +def run_d(lib_path, seed, ticks): + lib = Lib(lib_path) + seat = (seed * 7) % 10 + cfg = dict(max_ticks=ticks, record=True, capture=False) + ref = Env(lib, learner_seats=[], **cfg) + h_ref = play(ref, seed); r_ref = replay_bytes(ref) + abi = Env(lib, learner_seats=[seat], **cfg) + assert abi.set_defer_script(seat, "players/base.bas") == 0 + drv = InvalidDriver(seat) + h_abi = play(abi, seed, drv); r_abi = replay_bytes(abi) + first = next((i for i, (x, y) in enumerate(zip(h_abi, h_ref)) if x != y), -1) + return dict(same=h_abi == h_ref and r_abi == r_ref, first_diff_step=first, + defer=abi.defer_stats(seat).tolist(), invalid=int(abi.stats(seat)[21]), + picks=drv.invalid_picks, real_defers=drv.defers, seat=seat) + + +def job_d(lib_path, old_path, seed, ticks): + new = run_d(lib_path, seed, ticks) + old = run_d(old_path, seed, ticks) if old_path else None + return dict(seed=seed, new=new, old=old) + + +def lineup_c(lib, seed, ticks, package=True): + """Every non-defer feature at once: 2 learners (random actions, one with a shadow), a plain package + seat (random net, no defer_script), an override seat, capture on, puller/rusher scripts.""" + env = Env(lib, learner_seats=[1, 6], max_ticks=ticks, record=True, capture=True) + assert env.set_script(2, open(os.path.join(ROOT, "players/puller.bas")).read()) == 0 + assert env.set_script(8, open(os.path.join(ROOT, "players/rusher.bas")).read()) == 0 + assert env.set_override(5, 1) == 0 + b = BASE + assert env.L.gota_set_seat_shadow(env.h, 6, b, len(b)) == 0 + if package: + assert env.set_package(3, npk.build(POLICY, model_weights(hidden=64, seed=seed, scale=0.05, verb0_bias=0))) == 0 + h = play(env, seed, rng=np.random.default_rng(seed)) + orders = [env.orders(s).tolist() for s in range(10)] + return h, replay_bytes(env), orders + + +def job_c(lib_path, old_path, seed, ticks, package=True): + h_new, r_new, o_new = lineup_c(Lib(lib_path), seed, ticks, package) + h_old, r_old, o_old = lineup_c(Lib(old_path), seed, ticks, package) + # plain all-script lineup (no neural seats, capture off) too + p_new = play(Env(Lib(lib_path), learner_seats=[], max_ticks=ticks, capture=False), seed) + p_old = play(Env(Lib(old_path), learner_seats=[], max_ticks=ticks, capture=False), seed) + return dict(seed=seed, steps=len(h_new), hashes=h_new == h_old, replay=r_new == r_old, + orders=o_new == o_old, plain=p_new == p_old, final=f"{h_new[-1]:016x}") + + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("lib") + ap.add_argument("--old-lib") + ap.add_argument("--seeds", type=int, default=20) + ap.add_argument("--ticks", type=int, default=28800) + ap.add_argument("--jobs", type=int, default=8) + ap.add_argument("--only", default="a,b,c") + ap.add_argument("--c-no-package", action="store_true", + help="c without the package seat (when the reference lib predates a package-seat fix)") + a = ap.parse_args() + fails = [] + with ProcessPoolExecutor(a.jobs) as ex: + if "a" in a.only: + rows = list(ex.map(job_a, [a.lib] * a.seeds, range(1, a.seeds + 1), [a.ticks] * a.seeds)) + for r in rows: print("a", r, flush=True) + ok = sum(r["abi_hashes"] and r["abi_replay"] and r["pkg_hashes"] and r["pkg_replay"] + and r["abi_defer"][1] == 0 and r["pkg_defer"][1] == 0 for r in rows) + print(f"A always-defer == plain base.bas (ABI+package, hashes+replay): {ok}/{len(rows)}; " + f"peak plain-seat BASIC instr/tick (sampled last tick of every step) {max(r['ref_seat_peak_instr_sampled'] for r in rows)}", flush=True) + if ok != len(rows): fails.append("a") + if "b" in a.only: + n = max(8, a.seeds // 2) + rows = list(ex.map(job_b, [a.lib] * n, range(1, n + 1), [a.ticks] * n)) + for r in rows: print("b", r, flush=True) + ok = sum(r["same"] and r["counts_same"] for r in rows) + d = sum(r["abi_defer"][0] for r in rows); o = sum(r["abi_defer"][1] for r in rows) + print(f"B ABI == package on mixed net: {ok}/{n}; defer {d} override {o} " + f"(override share {o / max(1, d + o):.3f}); diverged from plain {sum(r['diverges_from_plain'] for r in rows)}/{n}", + flush=True) + if ok != n or o == 0 or d == 0: fails.append("b") + if "d" in a.only: + n = max(8, a.seeds // 2) + rows = list(ex.map(job_d, [a.lib] * n, [a.old_lib] * n, range(1, n + 1), [a.ticks] * n)) + for r in rows: print("d", r, flush=True) + ok = sum(r["new"]["same"] and r["new"]["defer"][1] == 0 and r["new"]["picks"] > 0 + and r["new"]["invalid"] == r["new"]["picks"] for r in rows) + picks = sum(r["new"]["picks"] for r in rows) + line = f"D invalid overrides == plain base.bas (ABI, hashes+replay, 0 overrides, invalid == picks): {ok}/{n}; {picks} invalid picks" + if a.old_lib: + line += f"; old lib same as plain {sum(r['old']['same'] for r in rows)}/{n}" + print(line, flush=True) + if ok != n: fails.append("d") + if "c" in a.only and a.old_lib: + n = max(8, a.seeds // 2) + rows = list(ex.map(job_c, [a.lib] * n, [a.old_lib] * n, range(1, n + 1), [a.ticks] * n, + [not a.c_no_package] * n)) + for r in rows: print("c", r, flush=True) + ok = sum(r["hashes"] and r["replay"] and r["orders"] and r["plain"] for r in rows) + print(f"C no-option seats == pre-change lib (hashes+replay+orders, plus plain lineup): {ok}/{n}", flush=True) + if ok != n: fails.append("c") + print("ALL PASS" if not fails else f"FAILURES: {fails}") + sys.exit(1 if fails else 0) + + +if __name__ == "__main__": + main() diff --git a/examples/gods_of_the_arena/tools/test_mask.py b/examples/gods_of_the_arena/tools/test_mask.py new file mode 100644 index 00000000..795840a7 --- /dev/null +++ b/examples/gods_of_the_arena/tools/test_mask.py @@ -0,0 +1,97 @@ +"""Action-mask proofs (gota_action_mask / decoder.mask_empty_targets). Run on a build host, not a laptop. + + python3 test_mask.py LIB [--seeds N] [--ticks T] [--jobs J] + +2 ABI mask == host mask: a package seat (random w128 net, mask_empty_targets) and a learner seat driven by + gota_net_infer + the documented masked argmax over gota_action_mask give the same mask bytes, the same heads + and the same per-step state hash every decision (conditional mode; static mode on odd seeds). +3 0 invalid: with the conditional mask the random net makes 0 invalid actions (stats[21]) in both seats; + the same net unmasked and in static mode are reported for comparison. +(Option-absent parity is test_defer.py c against the pre-mask lib.) +""" +import argparse, ctypes, os, sys +from concurrent.futures import ProcessPoolExecutor +import numpy as np +HERE = os.path.dirname(os.path.abspath(__file__)) +sys.path.insert(0, HERE) +sys.path.insert(0, os.path.join(HERE, "../../../coworld/gota/runtime")) +from native_env import Env, Lib, masked_argmax, static_masks, mask_target_row +import neural_package as npk +from test_defer import Driver, model_weights, argmax_heads + +POLICY = open(os.path.join(HERE, "../neural/policy.bas"), "rb").read() +F32 = ctypes.POINTER(ctypes.c_float) + + +def job(lib_path, seed, ticks): + lib = Lib(lib_path) + seat = (seed * 7) % 10 + static = seed % 2 == 1 + model = model_weights(hidden=128, seed=100 + seed, scale=0.05, verb0_bias=0.0) + decoder = {"mask_empty_targets": True, "mask_mode": "static" if static else "conditional"} + cfg = dict(max_ticks=ticks, capture=False) + pkg = Env(lib, learner_seats=[], **cfg) + assert pkg.set_package(seat, npk.build(POLICY, model, decoder=decoder)) == 0 + abi = Env(lib, learner_seats=[seat], **cfg) + raw = Env(lib, learner_seats=[], **cfg) # same net, no mask (comparison only) + assert raw.set_package(seat, npk.build(POLICY, model)) == 0 + drv = Driver(lib, model, seat) + pkg.reset(seed); abi.reset(seed) + same_mask = same_heads = same_hash = True + decisions = 0 + rows_all_valid = True # every allowed attackTarget slot is occupied and not self + while True: + obs, res, act = abi.observe(1 << seat) + pkg.observe(1 << seat) + actions = np.zeros((10, 5), np.int32) + if act[seat]: + if res[seat]: + drv.state[:] = 0 + o = np.ascontiguousarray(obs[seat]) + assert lib.L.gota_net_infer(drv.net, o.ctypes.data_as(F32), drv.state.ctypes.data_as(F32), + drv.logits.ctypes.data_as(F32)) == 0 + ma, mp = abi.action_mask(seat), pkg.action_mask(seat) + same_mask &= bool(np.array_equal(ma, mp)) + actions[seat] = masked_argmax(drv.logits, ma, static) + same_heads &= list(pkg.orders(seat)[1:6]) == list(actions[seat]) + decisions += 1 + r1 = abi.step(actions); r2 = pkg.step(actions) + same_hash &= abi.state_hash() == pkg.state_hash() + if r1 == 1 or r2 == 1: + break + raw.reset(seed) + while raw.step(np.zeros((10, 5), np.int32)) == 0: + pass + return dict(seed=seed, seat=seat, mode="static" if static else "conditional", decisions=decisions, + mask=same_mask, heads=same_heads, hashes=same_hash, + invalid_abi=int(abi.stats(seat)[21]), invalid_pkg=int(pkg.stats(seat)[21]), + invalid_unmasked=int(raw.stats(seat)[21]), unmasked_decisions=int(raw.stats(seat)[20])) + + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("lib") + ap.add_argument("--seeds", type=int, default=12) + ap.add_argument("--ticks", type=int, default=28800) + ap.add_argument("--jobs", type=int, default=12) + a = ap.parse_args() + with ProcessPoolExecutor(a.jobs) as ex: + rows = list(ex.map(job, [a.lib] * a.seeds, range(1, a.seeds + 1), [a.ticks] * a.seeds)) + for r in rows: + print("m", r, flush=True) + parity = sum(r["mask"] and r["heads"] and r["hashes"] for r in rows) + cond = [r for r in rows if r["mode"] == "conditional"] + stat = [r for r in rows if r["mode"] == "static"] + zero = sum(r["invalid_abi"] == 0 and r["invalid_pkg"] == 0 for r in cond) + print(f"2 ABI mask == host mask (mask bytes, heads, hashes every decision): {parity}/{len(rows)} " + f"({sum(r['decisions'] for r in rows)} decisions)") + print(f"3 conditional mask: 0 invalid in {zero}/{len(cond)} seeds; invalid per game: conditional " + f"{sum(r['invalid_pkg'] for r in cond)}/{len(cond)}, static {sum(r['invalid_pkg'] for r in stat)}/{len(stat)}, " + f"unmasked {sum(r['invalid_unmasked'] for r in rows)}/{len(rows)} games") + ok = parity == len(rows) and zero == len(cond) and len(cond) >= 10 - len(stat) // 2 + print("ALL PASS" if parity == len(rows) and zero == len(cond) else "FAILURES") + sys.exit(0 if parity == len(rows) and zero == len(cond) else 1) + + +if __name__ == "__main__": + main() diff --git a/examples/gods_of_the_arena/tools/test_native_concurrency.py b/examples/gods_of_the_arena/tools/test_native_concurrency.py new file mode 100644 index 00000000..aee336f2 --- /dev/null +++ b/examples/gods_of_the_arena/tools/test_native_concurrency.py @@ -0,0 +1,71 @@ +"""Concurrency acceptance: N threads x M handles == serial, bit for bit. + +Every handle gets a fixed seed, learner mask and seeded random actions. The serial pass steps +handles one at a time on the main thread; the threaded pass steps them from a thread pool in +which a handle may move to a different thread on every step (one thread at a time per handle, +as the Puffer native trainer does), then destroys every handle from the main thread. The +per-step state hash, reward, observation checksum and BC-label sequences must be identical. +Usage: python3 test_native_concurrency.py LIB [HANDLES] [THREADS] [STEPS] +""" +import hashlib, os, sys, threading +from concurrent.futures import ThreadPoolExecutor +import numpy as np +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +from native_env import Env, Lib, random_actions + + +def make(lib, i): + env = Env(lib, learner_seats=[i % 10, (i * 3 + 5) % 10], max_ticks=28800) + env.reset(1000 + i) + return env + + +def advance(env, rng, trace): + obs, res, act = env.observe() + r = env.step(random_actions(rng)) + labels = np.stack([env.orders(s) for s in range(10)]) + h = hashlib.sha256(obs.tobytes() + res.tobytes() + act.tobytes() + env.rewards.tobytes() + labels.tobytes()).hexdigest()[:16] + trace.append((env.state_hash(), h)) + if r == 1: + env.reset(99999) + + +def main(): + lib = Lib(sys.argv[1]) + handles = int(sys.argv[2]) if len(sys.argv) > 2 else 12 + threads = int(sys.argv[3]) if len(sys.argv) > 3 else 6 + steps = int(sys.argv[4]) if len(sys.argv) > 4 else 300 + serial = [] + for i in range(handles): + env, rng, trace = make(lib, i), np.random.default_rng(i), [] + for _ in range(steps): + advance(env, rng, trace) + serial.append(trace) + env.close() + envs = [make(lib, i) for i in range(handles)] + rngs = [np.random.default_rng(i) for i in range(handles)] + traces = [[] for _ in range(handles)] + seen_threads = [set() for _ in range(handles)] + + def one(i): + seen_threads[i].add(threading.get_ident()) + advance(envs[i], rngs[i], traces[i]) + + with ThreadPoolExecutor(threads) as pool: + for _ in range(steps): + list(pool.map(one, range(handles))) # every handle once per round, any thread + for env in envs: + env.close() # destroy from the main thread after foreign-thread stepping + bad = [i for i in range(handles) if traces[i] != serial[i]] + migrated = sum(len(s) > 1 for s in seen_threads) + print(f"handles={handles} threads={threads} steps={steps} migrated_handles={migrated} mismatched={bad}") + if bad: + i = bad[0] + k = next(k for k in range(steps) if traces[i][k] != serial[i][k]) + print("first mismatch handle", i, "step", k, traces[i][k], serial[i][k]) + sys.exit(1) + print("CONCURRENCY OK") + + +if __name__ == "__main__": + main() diff --git a/examples/gods_of_the_arena/tools/test_native_env.py b/examples/gods_of_the_arena/tools/test_native_env.py new file mode 100644 index 00000000..da31418f --- /dev/null +++ b/examples/gods_of_the_arena/tools/test_native_env.py @@ -0,0 +1,347 @@ +"""Native env acceptance tests (run on a build host, not a laptop). + + python3 test_native_env.py LIB [quick] + +1 labels: ten scripted base.bas seats, 500 decisions: labeled rows > 50% of acting rows. +2 capture/ceiling-off identity: capture on vs off give identical per-step state hashes. +3 shadow: a learner with a base.bas shadow plays exactly as without one (hashes), and gets labels. +4 goals: a goal set before reset is the last 16 floats of the first observation. +5 rewards: rewards[s] == delta(stats[s][0]) / 1000 every step. +6 package parity: a seat hosting a neural package and a learner seat driven through + gota_net_infer with the same weights produce identical worlds (w64/w128/w256/w384/w512). +7 package validation: the Nim loader and neural_package.py reject the same corrupted packages. +8 ops per tick: gota_net_info operations for w64/w128/w256/w384/w512 against the 4,000,000 budget. +9 error paths: gota_net_infer on a non-finite observation returns -2 with gota_last_error set; an + invalid gota_set_policy_script returns 1 with the compile message in gota_last_error; a package + rejection reason and a script compile message are still reported by gota_seat_script_status after + gota_reset; the Python binding raises NativeEnvError on negative returns. +10 widths: w64/w128/w256/w384/w512 load in both the Nim loader and neural_package.py; every other + width (including ones whose parameter count fits) is rejected by both, with the dimensions error. +""" +import os, sys, json, zipfile, io, hashlib +import numpy as np +HERE = os.path.dirname(os.path.abspath(__file__)) +sys.path.insert(0, HERE) +sys.path.insert(0, os.path.join(HERE, "../../../coworld/gota/runtime")) +from native_env import Env, Lib, NativeEnvError, random_actions, HEAD_SIZES +import neural_package as npk +import ctypes + +POLICY = open(os.path.join(HERE, "../neural/policy.bas"), "rb").read() +BASE = open(os.path.join(HERE, "../players/base.bas")).read() +failures = [] + + +def check(name, ok, detail=""): + print(("PASS " if ok else "FAIL ") + name + (" " + detail if detail else "")) + if not ok: + failures.append(name) + + +def run_hashes(env, steps, seed, rng_seed=0, learners=()): + env.reset(seed) + rng = np.random.default_rng(rng_seed) + out = [] + for _ in range(steps): + env.observe() + r = env.step(random_actions(rng)) + out.append(env.state_hash()) + if r == 1: + break + return out + + +def test_labels(lib, steps): + # base.bas thinks every 6 ticks and re-issues a standing order only when it + # changes (or every 2 s), so most 4-tick windows are genuinely "noop, hold". + # Required: every window in which the script issued a contract command is + # labeled (represented in the contract) >= 95% of the time, and labels are + # not rare (>= 10% of acting rows). + env = Env(lib, learner_seats=[]) + env.reset(3) + acting = labeled = issued = issued_labeled = 0 + for _ in range(steps): + _, _, act = env.observe() + a = act.copy() + env.step(np.zeros((10, 5), np.int32)) + for s in range(10): + if a[s]: + o = env.orders(s) + acting += 1 + labeled += int(o[0]) + if o[13] > 0: + issued += 1 + issued_labeled += int(o[0]) + check("labels cover command windows >=95%", issued_labeled >= 0.95 * issued, f"{issued_labeled}/{issued}") + check("labels >=10% of acting rows", labeled >= 0.10 * acting, f"{labeled}/{acting}={labeled / max(1, acting):.2f}") + env.close() + + +def test_capture_identity(lib, steps): + a = run_hashes(Env(lib, learner_seats=[], capture=True), steps, 5) + b = run_hashes(Env(lib, learner_seats=[], capture=False), steps, 5) + check("capture on == off (hashes)", a == b, f"{len(a)} steps") + + +def test_shadow(lib, steps): + plain = Env(lib, learner_seats=[2]) + shadow = Env(lib, learner_seats=[2]) + code = lib.L.gota_set_seat_shadow(shadow.h, 2, BASE.encode(), len(BASE.encode())) + a = run_hashes(plain, steps, 7, 1) + shadow.reset(7) + rng = np.random.default_rng(1) + b, labeled, acting = [], 0, 0 + for _ in range(steps): + _, _, act = shadow.observe() + alive = act[2] + shadow.step(random_actions(rng)) + b.append(shadow.state_hash()) + if alive: + acting += 1 + labeled += int(shadow.orders(2)[0]) + check("shadow executes nothing (hashes)", code == 0 and a == b) + check("shadow labels present", labeled > 0.05 * acting, f"{labeled}/{acting}") + + +def test_goal(lib): + env = Env(lib, learner_seats=[0, 6]) + w = np.linspace(-0.9, 0.9, 16).astype(np.float32) + w[15] = 0 + check("goal setter ok", env.set_goal(6, w) == 0) + bad = w.copy(); bad[15] = 0.5 + try: + env.set_goal(6, bad); code = 0 + except NativeEnvError as e: + code = e.code + check("goal w_reserved != 0 rejected", code == -3) + env.reset(11) + obs, _, _ = env.observe() + check("goal is last 16 obs floats of first obs", np.array_equal(obs[6, -16:], w) and obs[0, -16] == 1.0) + + +def test_rewards(lib, steps): + env = Env(lib, learner_seats=[0, 5]) + env.reset(13) + rng = np.random.default_rng(2) + prev = np.array([env.stats(s)[0] for s in range(10)]) + ok = True + for _ in range(steps): + env.observe() + env.step(random_actions(rng)) + now = np.array([env.stats(s)[0] for s in range(10)]) + ok &= np.array_equal(env.rewards, ((now - prev) / 1000).astype(np.float32)) + prev = now + check("rewards == delta score / 1000", bool(ok)) + + +def random_model(hidden, seed): + rng = np.random.default_rng(seed) + n = 1407 * hidden + 3 * hidden * hidden + 92 * hidden + w = (rng.standard_normal(n) * 0.05).astype(np.float32) + return npk.encode_model(w.tolist(), hidden) + + +ACCEPT_WIDTHS = [64, 128, 256, 384, 512] +REJECT_WIDTHS = [1, 32, 63, 65, 96, 192, 255, 257, 320, 383, 385, 448, 511, 513, 640] + + +def test_widths(lib): + """Both loaders accept exactly the five widths and reject every neighbor + tried above, even where the parameter count alone would fit.""" + env = Env(lib, learner_seats=[]) + for hidden in ACCEPT_WIDTHS + REJECT_WIDTHS: + model = random_model(hidden, 7) + want = hidden in ACCEPT_WIDTHS + try: + npk.validate(npk.build(POLICY, model)); py = "accepted" + except npk.PackageError as e: + py = "rejected: " + str(e) + err = ctypes.create_string_buffer(512) + net = lib.L.gota_net_load(model, len(model), err, 512) + nim = "accepted" if net else "rejected: " + err.value.decode() + if net: + info = np.zeros(8, np.int64) + lib.L.gota_net_info(net, info.ctypes.data_as(ctypes.POINTER(ctypes.c_int64))) + lib.L.gota_net_destroy(net) + if want: + ok = py == "accepted" and nim == "accepted" + else: + # Both must reject. Most neighbors fail the width check itself + # ("dimensions"); a width far enough out (640) also overflows + # model.bin's own size cap, and the package-level Python + # validator (which checks a ZIP entry's declared size before + # parsing it) surfaces that first, while gota_net_load here is + # given the raw model bytes with no package wrapper, so it only + # ever sees the dimensions check. Both rejections are correct. + def is_rejection(msg): + return "dimensions" in msg or "byte limit" in msg or \ + "parameter count" in msg + ok = is_rejection(py) and is_rejection(nim) + check(f"width {hidden} {'accepted' if want else 'rejected'} by both", ok, + f"py={py[:60]} nim={nim[:60]}") + if want: + rc = env.set_package(hidden % 10, npk.build(POLICY, model)) + check(f"width {hidden} accepted by the hosted seat too", rc == 0, + f"rc={rc} {env.status(hidden % 10)[1][:70]!r}") + + +def argmax_heads(logits): + out, o = [], 0 + for s in HEAD_SIZES: + out.append(int(np.argmax(logits[o:o + s]))); o += s + return out + + +def test_package_parity(lib, steps, widths): + for hidden in widths: + model = random_model(hidden, hidden) + pkg = npk.build(POLICY, model) + err = ctypes.create_string_buffer(512) + net = lib.L.gota_net_load(model, len(model), err, 512) + info = np.zeros(8, np.int64) + lib.L.gota_net_info(net, info.ctypes.data_as(ctypes.POINTER(ctypes.c_int64))) + hosted = Env(lib, learner_seats=[]) + rc = hosted.set_package(4, pkg) + driven = Env(lib, learner_seats=[4]) + hosted.reset(21); driven.reset(21) + state = np.zeros(hidden, np.float32) + logits = np.zeros(92, np.float32) + same, hosted_heads_ok = True, True + for _ in range(steps): + obs, res, act = driven.observe() + hobs, _, _ = hosted.observe() + same &= np.array_equal(obs[4], hobs[4]) + actions = np.zeros((10, 5), np.int32) + if act[4]: + if res[4]: + state[:] = 0 + o = np.ascontiguousarray(obs[4]) + lib.L.gota_net_infer(net, o.ctypes.data_as(ctypes.POINTER(ctypes.c_float)), + state.ctypes.data_as(ctypes.POINTER(ctypes.c_float)), + logits.ctypes.data_as(ctypes.POINTER(ctypes.c_float))) + actions[4] = argmax_heads(logits) + if act[4]: # the hosted seat has already inferred on this paused frame + hosted_heads_ok &= list(hosted.orders(4)[1:6]) == list(actions[4]) + r1 = driven.step(actions); r2 = hosted.step(actions) + same &= driven.state_hash() == hosted.state_hash() + if r1 == 1 or r2 == 1: + break + check(f"package seat == ABI-driven seat w{hidden}", rc == 0 and bool(same) and bool(hosted_heads_ok), + f"ops/inference={info[7]} params={info[6]}") + lib.L.gota_net_destroy(net) + + +def corrupt(pkg, fn): + z = zipfile.ZipFile(io.BytesIO(pkg)) + files = {n: z.read(n) for n in z.namelist()} + fn(files) + out = io.BytesIO() + with zipfile.ZipFile(out, "w", zipfile.ZIP_DEFLATED) as w: + for n, d in files.items(): + w.writestr(n, d) + return out.getvalue() + + +def test_validation(lib): + model = random_model(64, 1) + good = npk.build(POLICY, model) + def edit_manifest(key, value): + def f(files): + m = json.loads(files["manifest.json"]); m[key] = value + files["manifest.json"] = json.dumps(m).encode() + return f + cases = { + "unknown decoder key": edit_manifest("decoder", {"mode": "argmax", "fire_hold": 1}), + "unknown top key": edit_manifest("extra", 1), + "bad period": edit_manifest("decision_period", 0), + "bad goal reserved": edit_manifest("goal", {"red": [1] + [0] * 15, "blue": [0] * 15 + [0.5]}), + "policy hash mismatch": lambda f: f.__setitem__("policy.bas", f["policy.bas"] + b"\n' x\n"), + "extra file": lambda f: f.__setitem__("notes.txt", b"hi"), + "missing model": lambda f: f.pop("model.bin"), + "defer_script not a bool": edit_manifest("decoder", {"defer_script": 1}), + "mask_empty_targets not a bool": edit_manifest("decoder", {"mask_empty_targets": "yes"}), + "mask_mode without mask": edit_manifest("decoder", {"mask_mode": "static"}), + "bad mask_mode": edit_manifest("decoder", {"mask_empty_targets": True, "mask_mode": "greedy"}), + "float integer inputs": lambda f: f.__setitem__("manifest.json", json.dumps( + dict(json.loads(f["manifest.json"]), model=dict(json.loads(f["manifest.json"])["model"], inputs=1407.0))).encode()), + "temperature true": edit_manifest("decoder", {"mode": "sample", "temperature": True}), + "non-finite goal": lambda f: f.__setitem__("manifest.json", f["manifest.json"].replace( + b'"decision_period": 4', b'"decision_period": 4, "goal": {"red": [1e999' + b', 0' * 15 + + b'], "blue": [0' + b', 0' * 15 + b']}')), + "manifest not json": lambda f: f.__setitem__("manifest.json", b"{not json"), + } + env = Env(lib, learner_seats=[]) + check("good package accepted by both", env.set_package(1, good) == 0) + deferring = npk.build(BASE.encode(), model, decoder={"defer_script": True}) + check("defer_script package accepted by both", env.set_package(2, deferring) == 0) + for mode in ("conditional", "static"): + masked = npk.build(POLICY, model, decoder={"mask_empty_targets": True, "mask_mode": mode}) + check(f"mask_empty_targets ({mode}) package accepted by both", env.set_package(3, masked) == 0) + for name, fn in list(cases.items()) + [("truncated package", None)]: + bad = corrupt(good, fn) if fn else good[:len(good) // 2] + try: + npk.validate(bad); py = "accepted" + except npk.PackageError as e: + py = "rejected" + nim = env.set_package(1, bad, check=False) + check(f"reject {name}", py == "rejected" and nim == 2, f"py={py} nim={nim} {env.status(1)[1][:70]}") + + +def test_error_paths(lib): + L = lib.L + rng = np.random.default_rng(3) + n = 1407 * 64 + 3 * 64 * 64 + 92 * 64 + model = npk.encode_model((rng.standard_normal(n) * 0.05).astype(np.float32).tolist(), 64) + err = ctypes.create_string_buffer(512) + net = L.gota_net_load(model, len(model), err, 512) + check("net loads", bool(net), err.value.decode()) + obs = np.full(1407, np.nan, np.float32) + state = np.zeros(64, np.float32) + logits = np.zeros(92, np.float32) + F = ctypes.POINTER(ctypes.c_float) + r = L.gota_net_infer(net, obs.ctypes.data_as(F), state.ctypes.data_as(F), logits.ctypes.data_as(F)) + b = ctypes.create_string_buffer(1024) + L.gota_last_error(b, 1024) + check("gota_net_infer non-finite obs -> -2 with gota_last_error", r == -2 and len(b.value) > 0, + f"r={r} error={b.value.decode()[:80]!r}") + L.gota_net_destroy(net) + env = Env(lib, learner_seats=[]) + bad = b"THIS IS NOT BASIC\n" + r = L.gota_set_policy_script(env.h, bad, len(bad)) + L.gota_last_error(b, 1024) + check("bad policy script -> 1 with the compile message", r == 1 and len(b.value) > 0, + f"r={r} error={b.value.decode()[:80]!r}") + check("rejected package -> 2", env.set_package(4, b"PK\x03\x04 not a package", check=False) == 2) + check("bad script -> 1", env.set_script(7, "THIS IS NOT BASIC\n", check=False) == 1) + raised = False + try: + env.reset(5) + except NativeEnvError: + raised = True # -2: seat 7 failed to compile + check("reset with a failed seat raises NativeEnvError", raised) + code4, msg4 = env.status(4) + code7, msg7 = env.status(7) + check("package rejection reason survives gota_reset", "neural package rejected" in msg4, f"{code4} {msg4[:70]!r}") + check("compile message survives gota_reset", code7 == 2 and len(msg7) > 0, f"{code7} {msg7[:70]!r}") + env.close() + + +def main(): + lib = Lib(sys.argv[1]) + quick = len(sys.argv) > 2 and sys.argv[2] == "quick" + steps = 200 if quick else 700 + test_labels(lib, 500) + test_capture_identity(lib, steps) + test_shadow(lib, steps) + test_goal(lib) + test_rewards(lib, steps) + test_validation(lib) + test_error_paths(lib) + test_widths(lib) + test_package_parity(lib, steps if quick else 7200, [64, 128, 256, 384, 512]) + print("ALL PASS" if not failures else f"FAILURES: {failures}") + sys.exit(1 if failures else 0) + + +if __name__ == "__main__": + main() diff --git a/examples/gods_of_the_arena/tools/test_package_goal.py b/examples/gods_of_the_arena/tools/test_package_goal.py new file mode 100644 index 00000000..bd74f24f --- /dev/null +++ b/examples/gods_of_the_arena/tools/test_package_goal.py @@ -0,0 +1,99 @@ +"""Package manifest goal: native env == hosted-style load (the headless binary's loadBots path). + + python3 test_package_goal.py LIB GOTA_BIN [--seeds 6] [--old-lib OLD] + +A neural package with non-default manifest goals (red and blue differ) sits in seat 0 (odd seeds) or +seat 5 (even seeds), nine base.bas seats around it, full length. Checks per seed: + obs: the package seat's first acting observation ends with the manifest goal of its team; + native: gota_set_seat_package (no gota_set_seat_goal) recorded replay == the binary's replay + (every tick's hash and every action, byte for byte) and final hashes equal. +OLD (optional): the pre-fix lib, expected to differ (it overwrote the manifest goal with the default). +""" +import argparse, os, re, subprocess, sys, tempfile +import numpy as np +HERE = os.path.dirname(os.path.abspath(__file__)) +sys.path.insert(0, HERE) +sys.path.insert(0, os.path.join(HERE, "../../../coworld/gota/runtime")) +from native_env import Env, Lib +import neural_package as npk +from test_defer import model_weights + +REPO = os.path.abspath(os.path.join(HERE, "../../..")) +P = "examples/gods_of_the_arena/players/base.bas" +POLICY = open(os.path.join(HERE, "../neural/policy.bas"), "rb").read() +RED = [0.5, 0, 0.25, 0.1, 0.5, 0.25, -0.5, 0.25, 0.1, 0.5, 0.25, 0.5, -0.25, 0.1, 0.5, 0] +BLUE = [1, 0, -0.1, 0.5, 0, 0.1, -0.25, 0.5, 0.5, 0, 0.1, 0.25, -0.5, 0.25, 0.1, 0] + + +def native(lib_path, seed, seat, pkg_path, replay, manifest_goal=True): + env = Env(Lib(lib_path), learner_seats=[], record=True, capture=False) + assert env.set_package(seat, open(pkg_path, "rb").read()) == 0 + env.reset(seed) + goal_ok = None + while True: + obs, _, act = env.observe(1 << seat) + if goal_ok is None and act[seat]: + want = np.array((RED if seat < 5 else BLUE) if manifest_goal else [1] + [0] * 15, np.float32) + goal_ok = bool(np.array_equal(obs[seat][-16:], want)) + if env.step(np.zeros((10, 5), np.int32)) == 1: + break + assert env.save_replay(replay) == 0 + return goal_ok, "%016x" % env.state_hash() + + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("lib"); ap.add_argument("bin") + ap.add_argument("--seeds", type=int, default=6) + ap.add_argument("--old-lib") + ap.add_argument("--replay-diff", help="tools/replay_diff binary: require identical hash and action streams") + a = ap.parse_args() + tmp = tempfile.mkdtemp() + model = model_weights(hidden=64, seed=11, scale=0.05, verb0_bias=0.0) + goal_pkg = os.path.join(tmp, "goal.zip") + open(goal_pkg, "wb").write(npk.build(POLICY, model, goal={"red": RED, "blue": BLUE})) + plain_pkg = os.path.join(tmp, "plain.zip") + open(plain_pkg, "wb").write(npk.build(POLICY, model)) + ok_all, old_diff, runs = 0, 0, 0 + for seed, variant in [(s, v) for s in range(1, a.seeds + 1) for v in ("goal", "default")]: + runs += 1 + pkg = goal_pkg if variant == "goal" else plain_pkg + seat = 0 if seed % 2 else 5 + lineup = (["--bot", f"{P}:{seat}"] if seat else []) + ["--bot", f"{pkg}:1", "--bot", f"{P}:{9 - seat}"] + hosted = os.path.join(tmp, f"hosted-{seed}-{variant}.replay") + out = subprocess.run([a.bin] + lineup + ["--seed", str(seed), "--record", hosted], cwd=REPO, + capture_output=True, text=True).stdout + hh = re.search(r"hash: ([0-9A-Fa-f]+)", out).group(1).lower().rjust(16, "0") + nat = os.path.join(tmp, f"native-{seed}-{variant}.replay") + goal_ok, nh = native(a.lib, seed, seat, pkg, nat, variant == "goal") + nb, hb = open(nat, "rb").read(), open(hosted, "rb").read() + same_replay = nb == hb + first_diff = next((i for i in range(min(len(nb), len(hb))) if nb[i] != hb[i]), None) if not same_replay else None + v = subprocess.run([a.bin, "--replay", nat], cwd=REPO, capture_output=True, text=True).stdout + vm = re.search(r"replay hashes: ([0-9]+) mismatches", v) + vh = re.search(r"hash: ([0-9A-Fa-f]+)", v) + verified = vm is None and vh is not None and vh.group(1).lower().rjust(16, "0") == hh + streams = None + if a.replay_diff: + d = subprocess.run([a.replay_diff, hosted, nat], capture_output=True, text=True).stdout + streams = "mismatch" not in d and "setup equal: true" in d + m = re.search(r"actions: (\d+) vs (\d+)", d) + streams = streams and m is not None and m.group(1) == m.group(2) + verified = verified and streams + row = dict(seed=seed, variant=variant, seat=seat, streams_identical=streams, obs_goal=goal_ok, native_final=nh, hosted_final=hh, + finals_equal=nh == hh, replay_bytes_equal=same_replay, first_diff=first_diff, + sizes=(len(nb), len(hb)), native_replay_verified_by_binary=verified) + if a.old_lib: + _, oh = native(a.old_lib, seed, seat, pkg, os.path.join(tmp, f"old-{seed}-{variant}.replay"), variant == "goal") + row["old_lib_final"] = oh + old_diff += oh != hh + print(row, flush=True) + ok_all += goal_ok and nh == hh and verified # bytes differ only in config player names + print(f"package seat: native == hosted-style {ok_all}/{runs} (seeds x {{manifest goal, default goal}}: obs goal, " + f"final hash, replay verified)" + (f"; pre-fix lib differs from hosted on {old_diff}/{runs}" if a.old_lib else "")) + print("GOAL OK" if ok_all == runs else "GOAL FAIL") + sys.exit(0 if ok_all == runs else 1) + + +if __name__ == "__main__": + main() diff --git a/examples/gods_of_the_arena/tools/test_thread_isolation.py b/examples/gods_of_the_arena/tools/test_thread_isolation.py new file mode 100644 index 00000000..d426a13e --- /dev/null +++ b/examples/gods_of_the_arena/tools/test_thread_isolation.py @@ -0,0 +1,120 @@ +"""Cross-world isolation with several worlds per thread, and concurrent gota_create. + + python3 test_thread_isolation.py LIB vision [HANDLES] [THREADS] [STEPS] + python3 test_thread_isolation.py LIB create [THREADS] [CREATES_PER_THREAD] + +vision: HANDLES full-length matches (default 16, 4 threads, so >= 4 worlds share each thread and +handles migrate every round) must give the same per-step state hashes as each match run alone. The +per-world tower damage and structure kills show that towers fell differently across worlds, which is +where a shared per-thread vision-blocker grid would leak one world's towers into another's camp sight. +create: THREADS threads create, reset, step and destroy handles concurrently. Every handle must give +the per-step hashes of the same config created serially. +""" +import os, sys, threading +from concurrent.futures import ThreadPoolExecutor +import numpy as np +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +from native_env import Env, Lib, random_actions + + +def make(lib, i, ticks=28800): + env = Env(lib, learner_seats=[i % 10], max_ticks=ticks, capture=False) + env.reset(500 + i) + return env + + +def advance(env, rng, trace): + if env.done: + return + env.observe() + r = env.step(random_actions(rng)) + trace.append(env.state_hash()) + if r == 1: + env.done = True + + +def towers(env): + s = [env.stats(k) for k in range(10)] + return dict(tower_damage=[int(sum(x[9] for x in s[:5])), int(sum(x[9] for x in s[5:]))], + structure_kills=[int(sum(x[10] for x in s[:5])), int(sum(x[10] for x in s[5:]))], + ticks=int(env.results()[4])) + + +def vision(lib, handles, threads, steps): + serial, facts = [], [] + for i in range(handles): + env, rng, trace = make(lib, i), np.random.default_rng(i), [] + env.done = False + for _ in range(steps): + advance(env, rng, trace) + serial.append(trace); facts.append(towers(env)) + env.close() + envs = [make(lib, i) for i in range(handles)] + for e in envs: + e.done = False + rngs = [np.random.default_rng(i) for i in range(handles)] + traces = [[] for _ in range(handles)] + seen = [set() for _ in range(handles)] + + def one(i): + seen[i].add(threading.get_ident()) + advance(envs[i], rngs[i], traces[i]) + + with ThreadPoolExecutor(threads) as pool: + for _ in range(steps): + list(pool.map(one, range(handles))) + for e in envs: + e.close() + bad = [i for i in range(handles) if traces[i] != serial[i]] + for i, f in enumerate(facts): + print("world", i, f) + distinct = len({(tuple(f["tower_damage"]), tuple(f["structure_kills"])) for f in facts}) + kills = sum(sum(f["structure_kills"]) for f in facts) + print(f"vision: handles={handles} threads={threads} steps={steps} migrated={sum(len(s) > 1 for s in seen)} " + f"distinct_tower_histories={distinct}/{handles} structure_kills_total={kills} mismatched={bad}") + if bad: + i = bad[0] + k = next((k for k in range(min(len(traces[i]), len(serial[i]))) if traces[i][k] != serial[i][k]), None) + print("first mismatch world", i, "step", k) + ok = not bad and distinct > 1 and kills > 0 + print("VISION ISOLATION OK" if ok else "VISION ISOLATION FAIL") + return ok + + +def create(lib, threads, per_thread, steps=40, kinds=14): + """Creates race each other; handle i uses config kind i % kinds (seat, capture, seed, action RNG).""" + def run(i): + k = i % kinds + env = Env(lib, learner_seats=[k % 10], max_ticks=28800, capture=bool(k % 2), decision_period=4) + env.reset(900 + k) + env.done = False + rng, trace = np.random.default_rng(k), [] + for _ in range(steps): + advance(env, rng, trace) + env.close() + return trace + reference = {k: run(k) for k in range(kinds)} # serial, one at a time + def worker(t): + return [(t * per_thread + j, run(t * per_thread + j)) for j in range(per_thread)] + with ThreadPoolExecutor(threads) as pool: + results = [r for rows in pool.map(worker, range(threads)) for r in rows] + failures = [i for i, trace in results if trace != reference[i % kinds]] + print(f"create: threads={threads} creates={len(results)} kinds={kinds} steps_each={steps} mismatched={failures}") + ok = not failures and len(results) == threads * per_thread + print("CONCURRENT CREATE OK" if ok else "CONCURRENT CREATE FAIL") + return ok + + +def main(): + lib = Lib(sys.argv[1]) + mode = sys.argv[2] + a = [int(x) for x in sys.argv[3:]] + if mode == "vision": + ok = vision(lib, *(a + [16, 4, 7200][len(a):])) + else: + ok = create(lib, *(a + [16, 25][len(a):])) + sys.exit(0 if ok else 1) + + +if __name__ == "__main__": + main() diff --git a/examples/light_vs_dark/bots.nim b/examples/light_vs_dark/bots.nim index 4bc0456b..05344efc 100644 --- a/examples/light_vs_dark/bots.nim +++ b/examples/light_vs_dark/bots.nim @@ -13,7 +13,7 @@ import polyworld/neural, bassy, - polyworld/[bodies, metrics, profiles], + polyworld/[mailboxes, bodies, metrics, profiles], content, sim @@ -244,13 +244,14 @@ proc overlordLimits*(): Limits = ## unit will exceed it and fail the script, which is the pressure that ## pushes authors onto `nearestEnemy` and friends. result = defaultLimits() + result.maxStringBytes = 256 * 1024 result.maxSourceBytes = 256 * 1024 result.maxCodeInstructions = 100_000 result.maxArrays = 64 result.maxArrayElements = 65_536 result.maxGlobals = 1_024 result.maxHostData = 64 - result.maxHostFunctions = 64 + result.maxHostFunctions = 128 result.maxRoutines = 128 result.maxParameters = 16 result.maxRegisters = 512 @@ -268,6 +269,24 @@ proc observedAt(index: int32): Observed = return Observed(owner: -1) snapshot[index] +proc sendChat*( + game: Game, sender, target: int, text: openArray[char] +): int32 = + ## Routes global broadcasts and direct messages between players. + if sender notin 0 ..< game.inboxes.len or + target < -2 or target == -1 or target >= game.inboxes.len: + return 0 + let id = int32(if target < 0: target else: sender) + for recipient in 0 ..< game.inboxes.len: + case target + of -2: + discard + else: + if recipient != target: + continue + if game.inboxes[recipient].push(id, text): + inc result + proc buildOverlordHost*(playerId: int32): Host = ## Builds the complete world-query and command interface for one player. ## @@ -282,6 +301,40 @@ proc buildOverlordHost*(playerId: int32): Host = ## demand on the simulation rather than only its own arithmetic. result = initHost() result.addNeuralFunctions() + let sendChatProc: NumericHostProc = proc(args: openArray[Value]): Value = + ## Sends script text through the game's routing rules. + let player = int(playerId) + activeGame.brains[player].runtime.withString(args[1], text): + result = activeGame.sendChat(player, int(args[0].asInt), text) + let pullMailboxProc: NumericHostProc = proc(args: openArray[Value]): Value = + ## Copies the oldest message into BASIC and consumes it on success. + let + player = int(playerId) + inbox = activeGame.inboxes[player] + var runtime = activeGame.brains[player].runtime + if inbox.count == 0: + result = runtime.putString("") + else: + result = runtime.putString(inbox.messages[inbox.first]) + discard inbox.pop() + let mailboxIdProc: HostProc = proc(args: openArray[int32]): int32 = + ## Returns the channel or DM sender of the last pulled message. + activeGame.inboxes[int(playerId)].lastId + let mailboxCountProc: HostProc = proc(args: openArray[int32]): int32 = + ## Counts this player's unread messages. + int32(activeGame.inboxes[int(playerId)].count) + let mailboxSelfProc: HostProc = proc(args: openArray[int32]): int32 = + ## Returns this player's zero-based mailbox address. + int32(int(playerId)) + let mailboxPlayersProc: HostProc = proc(args: openArray[int32]): int32 = + ## Returns the number of player mailboxes in this game. + int32(activeGame.inboxes.len) + discard result.addFunction("sendChat", 2, sendChatProc, 256) + discard result.addFunction("pullMailbox$", 0, pullMailboxProc, 256) + discard result.addFunction("mailboxId", 0, mailboxIdProc, 4) + discard result.addFunction("mailboxCount", 0, mailboxCountProc, 4) + discard result.addFunction("mailboxSelf", 0, mailboxSelfProc, 4) + discard result.addFunction("mailboxPlayers", 0, mailboxPlayersProc, 4) for name in OverlordDataNames: discard result.addData(name) @@ -543,23 +596,28 @@ proc buildOverlordHost*(playerId: int32): Host = ## Lifecycle -proc loadBots*(game: Game, sources: array[PlayerCount, string]) = +proc loadBots*( + game: Game, sources: array[PlayerCount, string] +) = ## Compiles one script per player and gives each its own runtime. let limits = overlordLimits() let schema = buildOverlordHost(0) + for inbox in game.inboxes.mitems: + inbox = newMailbox() var bound = false for player in 0'i32 ..< PlayerCount: when not defined(coworld): if sources[player].len == 0: continue + let source = sources[player] let program = when defined(coworld): - compilePlayer(sources[player], schema, limits, int(player)) + compilePlayer(source, schema, limits, int(player)) else: - compile(sources[player], schema, limits) + compile(source, schema, limits) game.brains[player] = OverlordVm( runtime: initRuntime(program, buildOverlordHost(player), limits), - ready: true + ready: true, ) if not bound: bindOverlordData(program) @@ -584,8 +642,8 @@ proc runDecision(game: Game, player: int32) = home = structure.origin break - game.brains[player].runtime.restart() try: + game.brains[player].runtime.restart() let economy = addr game.world.players[player] ids = overlordDataIds diff --git a/examples/light_vs_dark/sim.nim b/examples/light_vs_dark/sim.nim index 21b888c8..cfb3f898 100644 --- a/examples/light_vs_dark/sim.nim +++ b/examples/light_vs_dark/sim.nim @@ -7,7 +7,7 @@ import bassy, fixxy, polyworld/[bodies, hashes, metrics, pathing, profiles, rngs, tapes, - visions], + visions, mailboxes], content, maps, replays @@ -153,6 +153,7 @@ type historyPlayback*: bool replayMode*: bool brains*: array[PlayerCount, OverlordVm] + inboxes*: array[PlayerCount, Mailbox] mapSeed*: int32 maximumTicks*: int32 diff --git a/examples/mailboxes/chat.bas b/examples/mailboxes/chat.bas new file mode 100644 index 00000000..575bec89 --- /dev/null +++ b/examples/mailboxes/chat.bas @@ -0,0 +1,14 @@ +' Drain all unread messages. Empty text means the queue is empty. +message$ = pullMailbox$() +while message$ <> "" + print mailboxId(), message$ + message$ = pullMailbox$() +wend +if announced = 0 then + sendChat(-2, "Hello everyone.") + ' GotA supports team chat. Other games ignore this destination. + sendChat(-1, "Hello teammates.") + ' GotA and Light vs Dark support DMs. CTA ignores this destination. + sendChat(mailboxSelf(), "A private note to myself.") + announced = 1 +end if diff --git a/nimby.lock b/nimby.lock index 9f194070..50ca4a84 100644 --- a/nimby.lock +++ b/nimby.lock @@ -1,4 +1,4 @@ -bassy 0.1.0 https://github.com/treeform/bassy 2c54d822f775ce92d8b109f2f8aedb4c4b56b29d +bassy 0.1.0 https://github.com/treeform/bassy 669a7c4b94e3d5b0a38dc9557608a7e58c2a764d fixxy 0.1.0 https://github.com/treeform/fixxy 05e5446dffb70093056cebb0c57721a60deaf52a silky 0.2.0 https://github.com/treeform/silky fb9b13910edd66cf1751056784c2f7d2932a59fc pixie 6.1.0 https://github.com/treeform/pixie 87cecced5c4c6f311c658a5f3ca0c9b43edb6aa7 diff --git a/src/polyworld/mailboxes.nim b/src/polyworld/mailboxes.nim new file mode 100644 index 00000000..5cd04771 --- /dev/null +++ b/src/polyworld/mailboxes.nim @@ -0,0 +1,38 @@ +const + MaxMailboxMessages* = 100 + MaxChatBytes* = 1024 + NoMailboxId* = -3 + +type Mailbox* = ref object + ids*: array[MaxMailboxMessages, int32] + messages*: array[MaxMailboxMessages, string] + first*, count*: int + lastId*: int32 + +proc newMailbox*(): Mailbox = + ## Reserves one player's bounded inbox before the game starts. + result = Mailbox(lastId: NoMailboxId) + for message in result.messages.mitems: + message = newStringOfCap(MaxChatBytes) + +proc push*(mailbox: Mailbox, id: int32, text: openArray[char]): bool = + ## Appends an ID and text, ignoring empty, oversized, or excess messages. + if mailbox.count == MaxMailboxMessages or + text.len == 0 or text.len > MaxChatBytes: + return false + let index = (mailbox.first + mailbox.count) mod MaxMailboxMessages + mailbox.ids[index] = id + mailbox.messages[index].setLen(text.len) + for i in 0 ..< text.len: + mailbox.messages[index][i] = text[i] + inc mailbox.count + true + +proc pop*(mailbox: Mailbox): int32 = + ## Consumes the first message after the game has read its text. + mailbox.lastId = NoMailboxId + if mailbox.count > 0: + mailbox.lastId = mailbox.ids[mailbox.first] + mailbox.first = (mailbox.first + 1) mod MaxMailboxMessages + dec mailbox.count + mailbox.lastId diff --git a/src/polyworld/pathing.nim b/src/polyworld/pathing.nim index 06324db5..e9e6debb 100644 --- a/src/polyworld/pathing.nim +++ b/src/polyworld/pathing.nim @@ -173,15 +173,19 @@ var nodePathZs: seq[int32] edgeLinks: seq[array[4, EdgeLink]] edgeKnown: seq[array[4, bool]] - pathCosts: seq[int64] - pathCameFrom: seq[int] - pathSeen: seq[uint32] - pathGeneration = 0'u32 - pathResultKeys: seq[int] - pathFrontierEdges: HeapQueue[(int64, int64, int, int)] - pathFrontierEight: HeapQueue[(int32, int, int)] dormantPathingContext: PathingContext +# Search scratch is per thread so separate worlds may search concurrently; +# every search sizes it for the installed graph (see `beginSearch`). +var + pathCosts {.threadvar.}: seq[int64] + pathCameFrom {.threadvar.}: seq[int] + pathSeen {.threadvar.}: seq[uint32] + pathGeneration {.threadvar.}: uint32 + pathResultKeys {.threadvar.}: seq[int] + pathFrontierEdges {.threadvar.}: HeapQueue[(int64, int64, int, int)] + pathFrontierEight {.threadvar.}: HeapQueue[(int32, int, int)] + ## Walkability proc isSteep(firstDelta, secondDelta: int64): bool {.inline.} = @@ -512,6 +516,15 @@ proc edgeLink*(layerIndex, x, z, direction: int): EdgeLink = edgeLinks[index][direction] = result edgeKnown[index][direction] = true +proc warmEdgeLinks*() = + ## Fills the whole edge cache so later searches only read it (required + ## before several threads search the same installed graph). + for layerIndex, layer in layers: + for z in 0 ..< layer.depth: + for x in 0 ..< layer.width: + for direction in 0 .. 3: + discard edgeLink(layerIndex, x, z, direction) + proc edgeMask*( layerIndex, x, z: int, blockers: openArray[seq[int32]] = [], @@ -677,7 +690,12 @@ proc octileCost( orthogonalCost * (max(ax, az) - min(ax, az)) proc beginSearch() = - ## Advances the generation stamp so a new search can reuse scratch. + ## Sizes this thread's search scratch for the installed graph and advances + ## the generation stamp so a new search can reuse it. + if pathCosts.len < nodeLayers.len: + pathCosts.setLen(nodeLayers.len) + pathCameFrom.setLen(nodeLayers.len) + pathSeen.setLen(nodeLayers.len) if pathGeneration == uint32.high: pathSeen = newSeq[uint32](pathSeen.len) pathGeneration = 1 diff --git a/src/polyworld/visions.nim b/src/polyworld/visions.nim index da70fd70..1c41e123 100644 --- a/src/polyworld/visions.nim +++ b/src/polyworld/visions.nim @@ -18,6 +18,8 @@ type x*, z*: int32 radius*: int32 eyeHeight*: int16 + units*, range*, offsetX*, offsetZ*: int32 + ## Optional exact circle in sub-tile units, relative to the source center. VisionRayStep = object ox, oz: int8 VisionOffset* = object @@ -108,6 +110,17 @@ proc copyVisionKeys*(dest: var seq[int32], src: openArray[int32]) = for i, value in src: dest[i] = value +proc inVisionRange(source: VisionSource, x, z: int32): bool = + ## Includes cells intersecting an exact circle when sub-tile units are supplied. + if source.units <= 0: + return true + let + dx = max(0'i64, abs(int64(x - source.x) * source.units - + source.offsetX) - source.units div 2) + dz = max(0'i64, abs(int64(z - source.z) * source.units - + source.offsetZ) - source.units div 2) + dx * dx + dz * dz <= int64(source.range) * source.range + proc rayBlocked( width: int32, terrainHeights, @@ -295,7 +308,7 @@ proc revealVision*( for z in minimumZ .. maximumZ: for x in minimumX .. maximumX: let index = z * width + x - if visible[index] != 0: + if visible[index] != 0 or not source.inVisionRange(x, z): continue if lineVisible( width, @@ -320,6 +333,8 @@ proc revealVision*( z = source.z + int32(offset.dz) if x < 0 or x >= width or z < 0 or z >= height: continue + if not source.inVisionRange(x, z): + continue let index = z * width + x if visible[index] != 0: continue @@ -369,7 +384,7 @@ proc revealVisionCached*( var cells: seq[int32] for z in max(0'i32, source.z - source.radius) .. min(height - 1, source.z + source.radius): for x in max(0'i32, source.x - source.radius) .. min(width - 1, source.x + source.radius): - if source.radius > 0 and lineVisible( + if source.radius > 0 and source.inVisionRange(x, z) and lineVisible( width, height, terrainHeights, blockerHeights, source.x, source.z, x, z, source.radius, source.eyeHeight): cells.add z * width + x diff --git a/tests/gota_golden.sh b/tests/gota_golden.sh new file mode 100755 index 00000000..13950002 --- /dev/null +++ b/tests/gota_golden.sh @@ -0,0 +1,19 @@ +#!/usr/bin/env bash +# No-neural parity against main: a base/puller/rusher lineup must reach the +# final state hash that main's headless binary reached for the same seed +# (tests/gota_golden.txt, recorded from main). Neural support must not move +# a match without neural seats by a single bit. +# bash tests/gota_golden.sh +set -euo pipefail +P=examples/gods_of_the_arena/players +nim c --hints:off -d:release -d:headless -o:tmp/gota_golden examples/gods_of_the_arena/gota.nim +status=0 +while read -r seed ticks want; do + [[ -z "$seed" || "$seed" == \#* ]] && continue + want=${want%$'\r'} # a CRLF checkout (Windows) keeps the \r on the last field + got=$(tmp/gota_golden --bot $P/base.bas:4 --bot $P/puller.bas:3 --bot $P/rusher.bas:3 \ + --seed "$seed" --ticks "$ticks" | grep -o 'hash: [0-9A-Fa-f]*' | tail -1 | awk '{print tolower($2)}') + if [[ "$got" == "$want" ]]; then echo "PASS seed $seed ticks $ticks hash $got" + else echo "FAIL seed $seed ticks $ticks hash $got, main reached $want"; status=1; fi +done < tests/gota_golden.txt +exit $status diff --git a/tests/gota_golden.txt b/tests/gota_golden.txt new file mode 100644 index 00000000..c8c34cac --- /dev/null +++ b/tests/gota_golden.txt @@ -0,0 +1,5 @@ +# seed battle_ticks final_state_hash, recorded with main (23f3384) headless: +# gota --bot base.bas:4 --bot puller.bas:3 --bot rusher.bas:3 --seed S --ticks T +1 7200 000000004bedee39 +2 7200 00000000ddef9e3b +3 7200 0000000067a3e191 diff --git a/tests/test_chats.nim b/tests/test_chats.nim new file mode 100644 index 00000000..65733e0a --- /dev/null +++ b/tests/test_chats.nim @@ -0,0 +1,188 @@ +import + std/[os, strutils, tempfiles], + bassy, + polyworld/[cli, mailboxes] +import ../examples/call_to_adventure/bots as ctaBots +import ../examples/call_to_adventure/content as ctaContent +import ../examples/call_to_adventure/sim as ctaSim +import ../examples/gods_of_the_arena/bots as gotaBots +import ../examples/gods_of_the_arena/maps as gotaMaps +import ../examples/gods_of_the_arena/replays as gotaReplays +import ../examples/gods_of_the_arena/sim as gotaSim +import ../examples/light_vs_dark/bots as lvdBots +import ../examples/light_vs_dark/content as lvdContent +import ../examples/light_vs_dark/maps as lvdMaps +import ../examples/light_vs_dark/sim as lvdSim + +const Program = """ +sent = sendChat(mailboxSelf(), "private hello") +message$ = pullMailbox$() +from = mailboxId() +""" + +const CtaProgram = """ +rejectedTeam = sendChat(-1, "team") +rejectedDm = sendChat(mailboxSelf(), "direct") +sent = sendChat(-2, "nearby hello") +message$ = pullMailbox$() +from = mailboxId() +while mailboxCount() > 0 + ignored$ = pullMailbox$() +wend +""" + +proc checkMailboxes[T](game: T) = + ## Checks each game's chat routing and repeated reads through BASIC. + template send(sender, target, text: untyped): untyped = + ## Uses the game's own routing implementation. + when T is ctaSim.Game: + ctaBots.sendChat(game, sender, target, text) + elif T is gotaSim.Game: + gotaBots.sendChat(game, sender, target, text) + else: + lvdBots.sendChat(game, sender, target, text) + when T is gotaSim.Game: + doAssert send(0, -1, "team") == 5 + for slot, inbox in game.inboxes: + let teammate = game.world.heroes[slot].team == game.world.heroes[0].team + doAssert inbox.count == int(teammate) + if teammate: + doAssert inbox.messages[inbox.first] == "team" + doAssert inbox.pop() == -1 + else: + doAssert send(0, -1, "team") == 0 + for inbox in game.inboxes: + doAssert inbox.count == 0 + doAssert send(0, -2, "global") == game.inboxes.len + for inbox in game.inboxes: + doAssert inbox.messages[inbox.first] == "global" + doAssert inbox.pop() == -2 + when T is ctaSim.Game: + for target in 0 ..< game.inboxes.len: + doAssert send(0, target, "direct") == 0 + for inbox in game.inboxes: + doAssert inbox.count == 0 + else: + doAssert send(0, 1, "direct") == 1 + for slot, inbox in game.inboxes: + doAssert inbox.count == int(slot == 1) + doAssert game.inboxes[1].messages[game.inboxes[1].first] == "direct" + doAssert game.inboxes[1].pop() == 0 + doAssert send(-1, -2, "invalid") == 0 + doAssert send(game.inboxes.len, -2, "invalid") == 0 + doAssert send(0, -3, "invalid") == 0 + doAssert send(0, game.inboxes.len, "invalid") == 0 + for i in 0 ..< MaxMailboxMessages: + doAssert game.inboxes[0].push(-2, "full") + doAssert send(0, -2, "partial") == game.inboxes.len - 1 + doAssert game.inboxes[0].count == MaxMailboxMessages + for inbox in game.inboxes: + while inbox.count > 0: + discard inbox.pop() + for tick in 1 .. 300: + game.world.tick = int32(tick) + when T is ctaSim.Game: + for slot in 0'i32 ..< ctaContent.PartySize: + ctaBots.runBotDecisions(game, slot) + elif T is gotaSim.Game: + gotaBots.runBotDecisions(game) + else: + lvdBots.runBotDecisions(game) + when T is lvdSim.Game: + let vms = game.brains + else: + let vms = game.heroVms + for index, vm in vms: + doAssert vm != nil and not vm.failed, vm.lastError + when T is ctaSim.Game: + doAssert vm.runtime.getGlobal("rejectedTeam") == 0 + doAssert vm.runtime.getGlobal("rejectedDm") == 0 + doAssert vm.runtime.getGlobal("sent") == ctaContent.PartySize + doAssert vm.runtime.getGlobal("from") == -2 + doAssert vm.runtime.getString(vm.runtime.getGlobalValue("message$")) == + "nearby hello" + else: + doAssert vm.runtime.getGlobal("sent") == 1 + doAssert vm.runtime.getGlobal("from") == index + doAssert vm.runtime.getString(vm.runtime.getGlobalValue("message$")) == + "private hello" + let large = vm.runtime.putString(repeat('x', 1024)) + doAssert vm.runtime.getString(large).len == 1024 + when defined(nimAllocStats) and T is gotaSim.Game: + # The GotA decision runner leaves its active game bound for host calls. + # Isolate mailbox callbacks and restart from unrelated world preparation. + let before = getAllocStats() + for decision in 0 ..< 1000: + for vm in game.heroVms: + vm.runtime.restart() + discard vm.runtime.run() + let after = getAllocStats() + doAssert after == before, $(after - before) + +proc checkRange(game: ctaSim.Game) = + ## Checks inclusive tile range, level isolation, and routing at send time. + for inbox in game.inboxes: + while inbox.count > 0: + discard inbox.pop() + for slot in 0 ..< ctaContent.PartySize: + game.world.actors[slot].home.level = 0 + game.world.actors[slot].home.x = 20 + game.world.actors[slot].home.z = 20 + game.world.actors[1].home.x = 36 + game.world.actors[2].home.x = 37 + game.world.actors[3].home.level = 1 + doAssert ctaBots.sendChat(game, 0, -2, "boundary") == 2 + doAssert game.inboxes[0].pop() == -2 + doAssert game.inboxes[2].count == 0 + doAssert game.inboxes[3].count == 0 + game.world.actors[1].home.level = 1 + doAssert game.inboxes[1].messages[game.inboxes[1].first] == "boundary" + doAssert game.inboxes[1].pop() == -2 + doAssert ctaBots.sendChat(game, 0, -2, "different level") == 1 + doAssert game.inboxes[0].pop() == -2 + doAssert game.inboxes[1].count == 0 + game.world.actors[1].home.level = 0 + game.world.actors[1].home.z = 36 + doAssert ctaBots.sendChat(game, 0, -2, "diagonal") == 2 + doAssert game.inboxes[0].pop() == -2 + doAssert game.inboxes[1].pop() == -2 + game.world.actors[1].home.z = 37 + doAssert ctaBots.sendChat(game, 0, -2, "outside") == 1 + doAssert game.inboxes[0].pop() == -2 + doAssert game.inboxes[1].count == 0 + +echo "Testing default mailboxes through all three games' BASIC hosts" +block: + let + directory = createTempDir("polyworld-mailboxes-", "") + path = directory / "player.bas" + defer: + removeDir(directory) + writeFile(path, Program) + + let gota = gotaSim.newGame( + gotaMaps.generateMap(54), + 240, + 10, + false, + gotaReplays.ReplayData(), + drafting = false + ) + gotaBots.loadBots(gota, [BotGroup(path: path, count: 10)]) + echo "Checking GotA" + gota.checkMailboxes() + echo "GotA passed" + + let cta = ctaSim.newGame(2026) + writeFile(path, CtaProgram) + ctaBots.loadBots(cta, [BotGroup(path: path, count: ctaContent.PartySize)]) + echo "Checking CTA" + cta.checkMailboxes() + cta.checkRange() + echo "CTA passed" + + let lvd = lvdSim.newGame(lvdMaps.generateMap(lvdContent.DefaultSeed), 240) + lvdBots.loadBots(lvd, [Program, Program]) + echo "Checking LVD" + lvd.checkMailboxes() + echo "LVD passed" diff --git a/tests/test_gota_base.nim b/tests/test_gota_base.nim index 7cafead4..47eb6691 100644 --- a/tests/test_gota_base.nim +++ b/tests/test_gota_base.nim @@ -79,6 +79,10 @@ block: names.incl(name & "At") doAssert names.len >= 68 for name in names: + # Chat is exercised by the mailbox example instead of the combat policy. + if name in ["sendChat", "pullMailbox$", "mailboxId", "mailboxCount", + "mailboxSelf", "mailboxPlayers"]: + continue doAssert name & "(" in source, "Base policy omits host call " & name echo "Testing base spends ability points in R, W, E, Q order at legal levels" diff --git a/tests/test_gota_camps.nim b/tests/test_gota_camps.nim index 833e8a49..c59d0e01 100644 --- a/tests/test_gota_camps.nim +++ b/tests/test_gota_camps.nim @@ -353,6 +353,8 @@ block: targetId: unit.id, attackTicks: TowerAttackTicks - 1) world.buildings.add tower world.updateTower(tower) + inc world.tick + world.advanceTowerShots() doAssert world.footmen[index].hp <= 0 doAssert hero.totalXp == NeutralXp[unit.campTier - 1] div 2 doAssert ally.totalXp == hero.totalXp diff --git a/tests/test_gota_creeps.nim b/tests/test_gota_creeps.nim index 33dabb21..cb1378f4 100644 --- a/tests/test_gota_creeps.nim +++ b/tests/test_gota_creeps.nim @@ -88,6 +88,8 @@ block: ) game.world.reveal() game.world.updateTower(tower) + inc game.world.tick + game.world.advanceTowerShots() doAssert game.world.footmen[0].hp == 1000 - [36'i32, 48, 60][tier.ord] @@ -174,9 +176,13 @@ block: ) game.world.reveal() game.world.updateTower(tower) + inc game.world.tick + game.world.advanceTowerShots() doAssert hero.totalXp == CreepNearbyXp and hero.level == 2 doAssert hero.gold == gold game.world.updateTower(tower) + inc game.world.tick + game.world.advanceTowerShots() doAssert hero.totalXp == CreepNearbyXp echo "Testing basic attacks award the creep bounty exactly once" diff --git a/tests/test_gota_events.nim b/tests/test_gota_events.nim index 663f0e48..089c2133 100644 --- a/tests/test_gota_events.nim +++ b/tests/test_gota_events.nim @@ -136,6 +136,8 @@ block: world.buildings.add tower world.footmen[0].hp = 1 world.updateTower(world.buildings[^1]) + inc world.tick + world.advanceTowerShots() doAssert world.last(Death).actor.id == 20 doAssert world.last(Death).actor.kind == 4 doAssert world.last(Death).target.kind == 3 diff --git a/tests/test_gota_neural.nim b/tests/test_gota_neural.nim new file mode 100644 index 00000000..0295b1a9 --- /dev/null +++ b/tests/test_gota_neural.nim @@ -0,0 +1,126 @@ +## GotA on the shared neural tier: the GotA contract's identity and its +## package options, with in-memory corruption cases for the GotA keys. +## nim r -d:headless tests/test_gota_neural.nim + +import + std/[json, strutils], + polyworld/neural_host, + ../examples/gods_of_the_arena/[bots, neural_contract], + neural_toy + +proc gotaModel(hidden = 64, obsHash = ObservationContractHash): string = + var weights = newSeq[float32](weightCount(hidden, ObservationSize, ActionOutputs)) + for i, w in weights.mpairs: + w = float32((i * 7919) mod 1000) / 100000'f32 - 0.005'f32 + encodeActor(ObservationSize, hidden, HeadSizes, obsHash, ActionContractHash, + weights) + +let + policy = "' glue\nx = gota_act()\n" + model = gotaModel() + +proc manifest(): JsonNode = + %*{ + "schema": PackageSchema, + "observation_contract": ObservationContractHash, + "action_contract": ActionContractHash, + "decision_period": 4, + "files": {"policy.bas": sha256Hex(policy), "model.bin": sha256Hex(model)}, + "model": {"format": "GOTANET1", "inputs": ObservationSize, "hidden": 64, + "heads": HeadSizes} + } + +proc pack(m: JsonNode, modelBytes = model): string = + writeZip([entry("manifest.json", $m), entry("policy.bas", policy), + entry("model.bin", modelBytes)]) + +proc rejects(bytes, message: string) = + var caught = false + try: + discard parseGotaPackage(bytes) + except ValueError as error: + caught = true + doAssert message in error.msg, "want '" & message & "', got: " & error.msg + doAssert caught, "accepted, expected: " & message + +echo "Testing the GotA contract identity" +block: + doAssert GotaContract.observationHash == ObservationContractHash + doAssert GotaContract.actionHash == ActionContractHash + doAssert GotaContract.observationSize == 1407 + doAssert GotaContract.headSizes == @[8, 25, 49, 4, 6] + doAssert GotaContract.headOutputs == ActionOutputs and ActionOutputs == 92 + doAssert GotaContract.maskSize == MaskSize and MaskSize == 187 + doAssert GotaContract.opBudget == 4_000_000 + doAssert GotaContract.schema == "gota-neural-basic/1" + +echo "Testing GotA package options" +block: + let plain = parseGotaPackage(pack(manifest())) + let options = plain.gotaOptions + doAssert options.goals[0] == defaultGoal() and options.goals[1] == defaultGoal() + doAssert not options.deferScript and options.maskMode == NoMask + var m = manifest() + var red = newSeq[float](16) + red[2] = 0.5 + m["goal"] = %*{"red": red, "blue": red} + m["decoder"] = %*{"mode": "sample", "temperature": 2, "defer_script": true, + "mask_empty_targets": true, "mask_mode": "static"} + let full = parseGotaPackage(pack(m)) + doAssert full.gotaOptions.goals[0][2] == 0.5 + doAssert full.gotaOptions.deferScript + doAssert full.gotaOptions.maskMode == StaticMask + doAssert full.decoder.mode == SampleDecoder and full.decoder.temperature == 2 + m["decoder"] = %*{"mask_empty_targets": true} + doAssert parseGotaPackage(pack(m)).gotaOptions.maskMode == ConditionalMask + +echo "Testing GotA corruption cases" +block: + proc edited(key: string, value: JsonNode): string = + var m = manifest() + m[key] = value + pack(m) + var zeros = newSeq[float](16) + rejects(edited("decoder", %*{"fire_hold": 1}), "unknown manifest key decoder.fire_hold") + rejects(edited("decoder", %*{"defer_script": 1}), "decoder.defer_script must be true or false") + rejects(edited("decoder", %*{"mask_empty_targets": "yes"}), + "decoder.mask_empty_targets must be true or false") + rejects(edited("decoder", %*{"mask_mode": "static"}), "mask_mode needs mask_empty_targets true") + rejects(edited("decoder", %*{"mask_empty_targets": true, "mask_mode": "greedy"}), + "mask_mode must be conditional or static") + rejects(edited("decoder", %*{"mode": "sample", "temperature": true}), + "decoder.temperature must be a number") + var reserved = zeros + reserved[15] = 0.5 + rejects(edited("goal", %*{"red": zeros, "blue": reserved}), "w_reserved must be 0") + rejects(edited("goal", %*{"red": zeros}), "goal needs red and blue") + rejects(edited("goal", %*{"red": zeros, "blue": zeros, "green": zeros}), + "unknown manifest key goal.green") + let boolGoal = %zeros + boolGoal.elems[0] = %true + rejects(edited("goal", %*{"red": zeros, "blue": boolGoal}), "must be 16 numbers") + let infinite = ($manifest()).replace("\"decision_period\":4", + "\"decision_period\":4,\"goal\":{\"red\":[1e999" & ",0".repeat(15) & + "],\"blue\":[0" & ",0".repeat(15) & "]}") + rejects(writeZip([entry("manifest.json", infinite), entry("policy.bas", policy), + entry("model.bin", model)]), "non-finite number") + var m = manifest() + m["model"]["inputs"] = %1407.0 + rejects(pack(m), "model.inputs must be an integer") + m = manifest() + m["decision_period"] = %25 + rejects(pack(m), "decision_period must be an integer 1..24") + let foreign = gotaModel(obsHash = "0".repeat(64)) + var f = manifest() + f["files"]["model.bin"] = %sha256Hex(foreign) + rejects(pack(f, foreign), "model.bin contract hashes do not match") + var bomb = entry("model.bin", "\0".repeat(16 * 1024 * 1024)) + bomb.declaredSize = model.len + rejects(writeZip([entry("manifest.json", $manifest()), entry("policy.bas", policy), + bomb]), "inflated data exceeds its limit") + var caught = false + try: + discard loadActor(foreign, GotaContract) + except ValueError: + caught = true + doAssert caught, "gota_net_load's loader must check the contract hashes" diff --git a/tests/test_gota_neural.nims b/tests/test_gota_neural.nims new file mode 100644 index 00000000..31711863 --- /dev/null +++ b/tests/test_gota_neural.nims @@ -0,0 +1 @@ +switch("define", "headless") diff --git a/tests/test_gota_phases.nim b/tests/test_gota_phases.nim index 364742ea..2bfa19fd 100644 --- a/tests/test_gota_phases.nim +++ b/tests/test_gota_phases.nim @@ -132,7 +132,7 @@ for first in Team: doAssert creep.hp <= 0 and creep.state == Dying doAssert game.world.deaths(creep.id) == 1 -echo "Testing a tower and hero both land their final attacks" +echo "Testing an in-flight tower shot and hero both land their final attacks" for team in Team: let game = arena() @@ -143,6 +143,11 @@ for team in Team: tier: OuterTower, position: middle(), hp: 1, maxHp: 1, targetId: hero.id, attackTicks: TowerAttackTicks - 1) ] + game.world.towerShots.add TowerShot( + sourceId: 10, targetId: hero.id, team: enemy, + damage: TowerDamages[OuterTower], position: hero.position, + previous: hero.position, started: game.world.tick + ) hero.attackObjectId = 10 game.tickWorld(nil) doAssert hero.hp <= 0 and hero.state == Dying diff --git a/tests/test_gota_portals.nim b/tests/test_gota_portals.nim index 2efd92a8..d2986a32 100644 --- a/tests/test_gota_portals.nim +++ b/tests/test_gota_portals.nim @@ -59,7 +59,7 @@ for size in [64, 116, 256]: doAssert hero.itemCounts[0] == 2 doAssert tower.team == team and tower.kind == TowerBuilding doAssert within(destination, tower.position, - tower.tier.towerSightTiles * WorldScale) + TowerAttackRanges[tower.tier]) doAssert not world.applyWalkTo(hero.id, aim, aim) doAssert hero.lastActionError == ActionChanneling doAssert not world.applyAttackMove(hero.id, aim, aim) diff --git a/tests/test_gota_sim.nim b/tests/test_gota_sim.nim index 676430fb..ba1890a2 100644 --- a/tests/test_gota_sim.nim +++ b/tests/test_gota_sim.nim @@ -174,11 +174,15 @@ block: "a tower should protect heroes by targeting a footman first" for _ in 1 ..< TowerAttackTicks: updateTower(run.world, run.world.buildings[0]) + inc run.world.tick + run.world.advanceTowerShots() doAssert run.world.footmen[0].hp == FootmanHp - TowerDamages[run.world.buildings[0].tier] run.world.footmen[0].hp = 0 for _ in 0 ..< TowerAttackTicks: updateTower(run.world, run.world.buildings[0]) + inc run.world.tick + run.world.advanceTowerShots() doAssert run.world.buildings[0].targetId == run.world.heroes[targetHero].id doAssert run.world.heroes[targetHero].hp == run.world.heroes[targetHero].maxHp - @@ -193,6 +197,8 @@ block: let deaths = run.world.stats.values[victim][LossesMetric] for _ in 0 ..< TowerAttackTicks: updateTower(run.world, run.world.buildings[0]) + inc run.world.tick + run.world.advanceTowerShots() doAssert run.world.stats.values[victim][LossesMetric] == deaths + 1 doAssert snapshot.stats.values[victim][LossesMetric] == deaths run.world.restore(snapshot) diff --git a/tests/test_gota_towers.nim b/tests/test_gota_towers.nim new file mode 100644 index 00000000..330924f2 --- /dev/null +++ b/tests/test_gota_towers.nim @@ -0,0 +1,173 @@ +import + polyworld/pathing, + ../examples/gods_of_the_arena/[content, maps, replays, sim] + +proc arena(team = RedTeam, tier = OuterTower): Game = + ## Isolates one tower and an enemy hero for deterministic combat checks. + result = newGame(generateMap(54), 100_000, 10, false, + ReplayData(), drafting = false) + let world = result.world + world.tick = 100 + world.camps.setLen(0) + world.footmen.setLen(0) + for hero in world.heroes: + hero.hp = 0 + hero.state = Dying + hero.deathTicks = -100_000 + world.buildings = @[ + Building(id: 10, kind: TowerBuilding, team: team, tier: tier, + hp: 1000, maxHp: 1000, position: WorldPoint()) + ] + for cells in world.teamVisible.mitems: + for cell in cells.mitems: + cell = 255 + let hero = world.heroes[(1 - team.ord) * 5] + hero.hp = 1000 + hero.state = Marching + hero.place(WorldPoint(x: 5 * WorldScale)) + +proc advance(world: World, ticks = 1) = + ## Advances tower reload and projectile travel without moving the test actors. + for i in 0 ..< ticks: + inc world.tick + for tower in world.buildings.mitems: + world.updateTower(tower) + world.advanceTowerShots() + +proc impact(world: World) = + ## Waits for the first launched shot to arrive, failing on a stuck projectile. + let limit = world.tick + 200 + while world.towerShots.len > 0 and world.towerShots[0].impact == 0: + inc world.tick + world.advanceTowerShots() + doAssert world.tick < limit + +echo "Testing tower range covers the longest hero attack plus its footprint" +block: + let game = newGame(generateMap(54), 100_000, 0, false, ReplayData()) + var longest = 0'i32 + for class in HeroClass: + longest = max(longest, class.heroAttackRange()) + for tower in game.world.buildings: + if tower.kind != TowerBuilding: + continue + for tile in tower.footprint: + let point = pathPoint(tile.layer.int, tile.x.int, tile.z.int) + for dx in [-WorldScale div 2, WorldScale div 2]: + for dz in [-WorldScale div 2, WorldScale div 2]: + let corner = WorldPoint( + x: point.x * (WorldScale div PathUnitsPerTile) + dx, + z: point.z * (WorldScale div PathUnitsPerTile) + dz + ) + doAssert within(tower.position, corner, + TowerAttackRanges[tower.tier] - longest - WorldScale div 4) + +echo "Testing exact range boundaries and delayed damage for both teams" +for team in Team: + for tier in TowerTier: + let + game = arena(team, tier) + world = game.world + hero = world.heroes[(1 - team.ord) * 5] + direction = if team == RedTeam: 1'i32 else: -1'i32 + hero.place(WorldPoint(x: direction * (TowerAttackRanges[tier] + 1))) + world.advance(TowerAttackTicks.int) + doAssert world.towerShots.len == 0 + hero.place(WorldPoint(x: direction * TowerAttackRanges[tier])) + world.advance() + doAssert world.towerShots.len == 1 + doAssert hero.hp == 1000, "damage must wait for projectile arrival" + world.impact() + doAssert hero.hp == 1000 - TowerDamages[tier] + +echo "Testing dipping out of range and changing targets do not reset reload" +block: + let + game = arena() + world = game.world + hero = world.heroes[5] + replacement = world.heroes[6] + world.advance(10) + doAssert world.buildings[0].attackTicks == 10 + hero.place(WorldPoint(x: 20 * WorldScale)) + world.advance(10) + doAssert world.buildings[0].targetId == 0 + doAssert world.buildings[0].attackTicks == 20 + replacement.hp = 1000 + replacement.state = Marching + replacement.place(WorldPoint(x: 5 * WorldScale)) + world.advance(4) + doAssert world.towerShots.len == 1 + doAssert world.towerShots[0].targetId == replacement.id + replacement.place(WorldPoint(x: 20 * WorldScale)) + world.advance(TowerAttackTicks.int) + doAssert world.buildings[0].attackTicks == TowerAttackTicks + hero.place(WorldPoint(x: 5 * WorldScale)) + world.advance() + doAssert world.towerShots[^1].targetId == hero.id + doAssert world.towerShots[^1].started == world.tick + +echo "Testing a fired shot follows a hidden escaping target after its tower dies" +for team in Team: + let + game = arena(team) + world = game.world + hero = world.heroes[(1 - team.ord) * 5] + world.advance(TowerAttackTicks.int) + doAssert world.towerShots.len == 1 + let origin = world.towerShots[0].position + hero.place(WorldPoint(x: -25 * WorldScale, z: 8 * WorldScale)) + world.buildings[0].hp = 0 + for cell in world.teamVisible[team.ord].mitems: + cell = 0 + world.advance() + doAssert world.towerShots[0].position.x < origin.x + doAssert world.towerShots[0].position.z > origin.z + doAssert hero.hp == 1000 + world.impact() + doAssert hero.hp == 1000 - TowerDamages[OuterTower] + for i in 0 ..< TowerImpactTicks: + inc world.tick + world.advanceTowerShots() + doAssert world.towerShots.len == 0 + doAssert hero.hp == 1000 - TowerDamages[OuterTower] + +echo "Testing dead targets cancel shots instead of hitting a replacement or respawn" +block: + let + game = arena() + world = game.world + hero = world.heroes[5] + world.advance(TowerAttackTicks.int) + hero.hp = 0 + hero.state = Dying + inc world.tick + world.advanceTowerShots() + doAssert world.towerShots.len == 0 + hero.hp = 1000 + hero.state = Marching + inc world.tick + world.advanceTowerShots() + doAssert hero.hp == 1000 + +echo "Testing projectile snapshots and hashes restore mid-flight and impact state" +block: + let + game = arena() + world = game.world + world.advance(TowerAttackTicks.int + 2) + let + snapshot = world.clone() + saved = game.stateHash() + inc world.towerShots[0].damage + doAssert game.stateHash() != saved + world.restore(snapshot) + doAssert game.stateHash() == saved + world.impact() + let expected = game.stateHash() + doAssert snapshot.towerShots[0].impact == 0 + world.restore(snapshot) + world.impact() + doAssert game.stateHash() == expected + +echo "test_gota_towers: all checks passed" diff --git a/tests/test_mailbox_allocations.nim b/tests/test_mailbox_allocations.nim new file mode 100644 index 00000000..a3a5b748 --- /dev/null +++ b/tests/test_mailbox_allocations.nim @@ -0,0 +1,24 @@ +import + std/strutils, + polyworld/mailboxes, + test_chats + +when not defined(nimAllocStats): + {.error: "Run this test with -d:nimAllocStats to measure allocations.".} + +echo "Testing inbox storage is reused without heap allocations" +block: + let + inbox = newMailbox() + payload = repeat('x', MaxChatBytes) + before = getAllocStats() + for round in 0 ..< 1000: + for i in 0 ..< MaxMailboxMessages: + doAssert inbox.push(int32(i), payload) + doAssert not inbox.push(0, "overflow") + for i in 0 ..< MaxMailboxMessages: + doAssert inbox.messages[inbox.first] == payload + doAssert inbox.pop() == int32(i) + doAssert inbox.pop() == NoMailboxId + let after = getAllocStats() + doAssert after == before, $(after - before) diff --git a/tests/test_mailboxes.nim b/tests/test_mailboxes.nim new file mode 100644 index 00000000..7a5d81d1 --- /dev/null +++ b/tests/test_mailboxes.nim @@ -0,0 +1,26 @@ +import + std/strutils, + polyworld/mailboxes + +echo "Testing bounded inbox IDs, strings, overflow, and wraparound" +block: + let inbox = newMailbox() + doAssert inbox.pop() == NoMailboxId + doAssert not inbox.push(1, "") + doAssert not inbox.push(1, repeat('x', MaxChatBytes + 1)) + for i in 0 ..< MaxMailboxMessages: + doAssert inbox.push(int32(i), $i) + doAssert not inbox.push(999, "overflow") + for i in 0 ..< 10: + doAssert inbox.messages[inbox.first] == $i + doAssert inbox.pop() == int32(i) + doAssert inbox.push(int32(i + MaxMailboxMessages), $(i + MaxMailboxMessages)) + for i in 10 ..< MaxMailboxMessages + 10: + doAssert inbox.messages[inbox.first] == $i + doAssert inbox.pop() == int32(i) + doAssert inbox.pop() == NoMailboxId + doAssert inbox.count == 0 + for id in [-2'i32, -1, 0, 9]: + doAssert inbox.push(id, "Hello 🌍") + doAssert inbox.messages[inbox.first] == "Hello 🌍" + doAssert inbox.pop() == id diff --git a/tests/test_visions.nim b/tests/test_visions.nim index f7d8fcfd..7d42ace6 100644 --- a/tests/test_visions.nim +++ b/tests/test_visions.nim @@ -1,5 +1,31 @@ import polyworld/visions +echo "Testing fractional vision circles, source offsets and cached occlusion" +block: + const Size = 31'i32 + var + cache: VisionCache + reference, cached: seq[uint8] + terrain = newSeq[int16](Size * Size) + blockers = newSeq[int16](Size * Size) + for offset in [-2500'i32, 0'i32, 2500'i32]: + let source = VisionSource(x: 15, z: 15, radius: 11, eyeHeight: 24, + units: 10_000, range: 95_000, offsetX: offset) + revealVision(reference, Size, Size, terrain, blockers, [source]) + revealVisionCached(cache, cached, Size, Size, terrain, blockers, [source]) + doAssert cached == reference + doAssert cached[15 * Size + 24] == 255 + doAssert cached[15 * Size + 26] == 0 + doAssert (cached[15 * Size + 25] != 0) == (offset >= 0) + doAssert (cached[15 * Size + 5] != 0) == (offset <= 0) + blockers[15 * Size + 18] = 40 + let source = VisionSource(x: 15, z: 15, radius: 11, eyeHeight: 24, + units: 10_000, range: 95_000) + revealVision(reference, Size, Size, terrain, blockers, [source]) + revealVisionCached(cache, cached, Size, Size, terrain, blockers, [source]) + doAssert cached == reference + doAssert cached[15 * Size + 24] == 0 + echo "Testing cached vision against full rebuilds across world changes" block: var diff --git a/tests/tests.nim b/tests/tests.nim index d8ee130b..b330d635 100644 --- a/tests/tests.nim +++ b/tests/tests.nim @@ -56,6 +56,7 @@ import test_gota_symmetry, test_gota_targets, test_gota_terrains, + test_gota_towers, test_gota_walls, test_hashes, test_hlf_content, @@ -70,6 +71,8 @@ import test_lvd_maps, test_lvd_replays, test_lvd_sim, + test_mailboxes, + test_chats, test_metrics, test_stats, test_nav,