From 05755ec478a52c07479ae05ea8b29eb08f6e7377 Mon Sep 17 00:00:00 2001 From: treeform Date: Mon, 5 Oct 2026 11:51:45 -0700 Subject: [PATCH 1/2] add BASIC tensor neural architectures --- coworld/dependencies.lock | 2 +- docs/neural-runners.md | 8 +- docs/neural-tensors.md | 181 ++++ examples/gods_of_the_arena/bots.nim | 6 +- .../neural/examples/tensor-andre.zip | Bin 0 -> 2152 bytes .../neural/examples/tensor-david.zip | Bin 0 -> 6003 bytes .../neural/examples/tensor-fly.zip | Bin 0 -> 2050 bytes .../neural/examples/tensor-richard.zip | Bin 0 -> 1298 bytes .../neural/tensors/andre.bas | 84 ++ .../neural/tensors/david.bas | 67 ++ .../gods_of_the_arena/neural/tensors/fly.bas | 60 ++ .../neural/tensors/richard.bas | 40 + .../gods_of_the_arena/tools/convert_andre.py | 9 +- .../gods_of_the_arena/tools/convert_david.py | 9 +- .../tools/convert_richard.py | 5 + .../tools/tensor_packages.py | 177 ++++ nimby.lock | 2 +- src/polyworld/tensors.nim | 964 ++++++++++++++++++ tests/all.nim | 2 + tests/bench_tensors.nim | 93 ++ tests/tensorfixtures.nim | 185 ++++ tests/test_neural_converters.py | 19 + tests/test_tensor_matches.nim | 47 + tests/test_tensors.nim | 411 ++++++++ 24 files changed, 2363 insertions(+), 8 deletions(-) create mode 100644 docs/neural-tensors.md create mode 100644 examples/gods_of_the_arena/neural/examples/tensor-andre.zip create mode 100644 examples/gods_of_the_arena/neural/examples/tensor-david.zip create mode 100644 examples/gods_of_the_arena/neural/examples/tensor-fly.zip create mode 100644 examples/gods_of_the_arena/neural/examples/tensor-richard.zip create mode 100644 examples/gods_of_the_arena/neural/tensors/andre.bas create mode 100644 examples/gods_of_the_arena/neural/tensors/david.bas create mode 100644 examples/gods_of_the_arena/neural/tensors/fly.bas create mode 100644 examples/gods_of_the_arena/neural/tensors/richard.bas create mode 100644 examples/gods_of_the_arena/tools/tensor_packages.py create mode 100644 src/polyworld/tensors.nim create mode 100644 tests/bench_tensors.nim create mode 100644 tests/tensorfixtures.nim create mode 100644 tests/test_tensor_matches.nim create mode 100644 tests/test_tensors.nim diff --git a/coworld/dependencies.lock b/coworld/dependencies.lock index b96327b2..3978472f 100644 --- a/coworld/dependencies.lock +++ b/coworld/dependencies.lock @@ -1,4 +1,4 @@ -bassy 0.1.0 https://github.com/treeform/bassy 907b69f2d0d4f32186d72947883f5bdac159ce9d +bassy 0.1.0 https://github.com/treeform/bassy b1975841adf49320fb3ae28df062500ac5a2ff47 curly 1.1.1 https://github.com/guzba/curly 29558e2d2b76244a2c65fea08e9f4efff16c1997 libcurl 1.0.0 https://github.com/Araq/libcurl eecbfc31e17ae2f09f0b6d2aaefcfb970c7d2be4 fixxy 0.1.0 https://github.com/treeform/fixxy 05e5446dffb70093056cebb0c57721a60deaf52a diff --git a/docs/neural-runners.md b/docs/neural-runners.md index c62804ac..f50ade28 100644 --- a/docs/neural-runners.md +++ b/docs/neural-runners.md @@ -46,8 +46,9 @@ bytes. The Bassy host interface is documented in its `docs/native-buffers.md`. GOTA allows **100,000 instructions and 250,000 work units per decision** so the BASIC observation and action glue can run. Existing BASIC storage and source -limits remain in force. Each neural invocation costs one host work unit. There -is no inference frequency or multiply/add budget. Package bytes, parsed models, +limits remain in force. Each named-runner invocation costs one host work unit. +Named runners have no inference frequency or multiply/add budget. The generic +tensor API charges work by operation size. Package bytes, parsed models, state, returned arrays and reserved scratch share a separate **32 MiB logical native allowance per hero**. This is not a process RSS cap. Cache reservations survive full VM reset because immutable model caches remain alive. @@ -366,3 +367,6 @@ each small Richard network was below one microsecond. Andre's width-12, one-layer model took about one microsecond; width 64 with three layers took approximately 25 microseconds. A fly model the size of the FlyWire cut (40,619 neurons, 1.06 million edges, four steps) took about 5 ms per call. + +BASIC-defined architectures can also use the [packed tensor API](neural-tensors.md). +This keeps float32 inside neural computation and meters generic operations by size. diff --git a/docs/neural-tensors.md b/docs/neural-tensors.md new file mode 100644 index 00000000..ab71fd04 --- /dev/null +++ b/docs/neural-tensors.md @@ -0,0 +1,181 @@ +# BASIC-defined neural architectures + +GotA policies can describe their forward pass in BASIC using packed tensors. +Weights, intermediate results and recurrent state remain inside each player's VM. +The existing Richard, David, Andre and Fly runners still work. + +BASIC scalar arithmetic remains integer and deterministic Q16.16. Float32 is +available only inside tensor operations, including one-element scalar tensors. +Neither this API nor the named runners automatically issues game actions. +The policy chooses when to run inference, sample outputs and issue commands. + +## Package layout + +A ZIP contains exactly one BASIC entry, `tensors.json` and binary resources. +Resource paths are relative to the ZIP root. For example: + +```json +{"tensors":[ + {"name":"encoder", "dtype":"float32", "shape":[64,45], + "resource":"weights.bin", "offset":0} +]} +``` + +Each entry names a dtype (`int32`, `fixed` or `float32`), shape, resource and byte +offset. All elements are four-byte little-endian words. Fixed values store raw +Q16.16 bits. Float32 uses IEEE-754 binary32. Matrices are row-major. Shapes have +one to eight nonnegative dimensions, at most 4,194,304 elements. Empty sparse +arrays are allowed. Loaded tensors are immutable. Copy them into mutable scratch +tensors when necessary. Names must be unique, and nonfinite weights are rejected. + +```basic +dim shape(0) +dim observations(44) +if initialized = 0 then + encoder = tensorLoad("encoder") + shape(0) = 45 + input = tensorCreate("float32", shape) + shape(0) = tensorDim(encoder, 0) + hidden = tensorCreate("float32", shape) + initialized = 1 +end if +tensorImport(input, observations) +tensorDense(input, encoder, 0, hidden) +tensorRelu(hidden, hidden) +``` + +Keep handles in BASIC globals or DIM arrays while they are needed. Unreferenced +buffers are collected between native calls. Restarting a decision retains state; +resetting a VM invalidates its old handles. Handles cannot cross player VMs. + +## Operations + +Computational operations take reusable destinations and return that destination. +All shapes, dtypes and indices are checked before publishing a result. A failed +operation leaves its destination unchanged. Exact in-place elementwise operations +are supported. Dense and CSR outputs cannot alias their inputs. Slices copy data; +there are no pointer views visible to BASIC. Binary arithmetic requires matching +shapes, except that either operand may be a one-element scalar of the same dtype. + +| Function | Behavior | +| --- | --- | +| `tensorLoad(name$)` | Loads a named immutable tensor from the policy ZIP. | +| `tensorCreate(dtype$, shape)` | Creates zero-filled mutable storage using a BASIC shape array. | +| `tensorScalar(dtype$, decimal$)` | Creates a one-element tensor without quantizing float32 constants through BASIC. | +| `tensorSize(t)`, `tensorDim(t, axis)` | Returns element count or a zero-based dimension. | +| `tensorDtype(t)` | Returns int32=0, fixed=1, float32=2. | +| `tensorImport(dst, array)` | Converts an exact-length BASIC numeric array into the destination dtype. | +| `tensorExport(t)` | Returns a BASIC numeric array, converting float32 to Q16.16. | +| `tensorCopy(src, dst)` | Copies equal counts without converting dtype. | +| `tensorConvert(src, dst)` | Explicitly converts equal element counts. Integer conversion requires integral values in range. | +| `tensorReinterpret(src, dst)` | Preserves raw bits, for formats such as Richard's integer logits. | +| `tensorFill(dst, scalar)` | Fills mutable storage with a same-dtype scalar tensor. | +| `tensorDense(input, weights, bias, dst)` | Ordered matrix-vector multiplication, starting each sum with bias. Pass integer `0` for no bias. | +| `tensorCsr(input, offsets, sources, weights, dst)` | Ordered target-row CSR multiplication with int32 index tensors. | +| `tensorAdd(a,b,dst)`, `tensorSubtract(a,b,dst)` | Elementwise arithmetic with scalar broadcasting. | +| `tensorMultiply(a,b,dst)`, `tensorDivide(a,b,dst)` | Elementwise arithmetic with scalar broadcasting. Division by zero fails. | +| `tensorRelu(src,dst)`, `tensorAbs(src,dst)` | ReLU or absolute value. | +| `tensorExp(src,dst)`, `tensorTanh(src,dst)` | Exponential or hyperbolic tangent, for fixed and float32 tensors. | +| `tensorSigmoid(src,dst)` | Stable sigmoid using branches on the input sign. | +| `tensorSigmoidDirect(src,dst)` | Direct sigmoid `1/(1+exp(-x))`, preserving existing FP32 evaluator order. | +| `tensorCompare(a,b,operation$,dst)` | int32 mask using `lt`, `le`, `eq`, `ne`, `ge` or `gt`. | +| `tensorSelect(mask,a,b,dst)` | Selects `a` where the int32 mask is nonzero, otherwise `b`. | +| `tensorLerp(a,b,weight,dst)` | Stable interpolation with a branch at `abs(weight) < 0.5`. | +| `tensorSlice(src,start,count,dst)` | Copies a flat range into an exact-length vector. | +| `tensorCopySlice(src,start,dst,offset,count)` | Copies a flat range and preserves other destination values. | +| `tensorReshape(src,shape,dst)` | Copies equal counts into a destination with the declared shape. | +| `tensorGather(src,indices,dst)` | Gathers a vector using int32 indices. | +| `tensorScatterAdd(src,indices,dst)` | Adds into the existing destination, preserving order for repeated indices. | +| `tensorArgmax(t)`, `tensorArgmaxMasked(t,mask)` | Returns the first greatest index; returns -1 when empty or fully masked. | + +Float32 operations preserve intermediate precision and ordered products/sums, +with fused multiply-add disabled in the kernels. Nonfinite results fail before +commit. Cross-platform bitwise identity is not promised. Output conversion rounds +ties away from zero and rejects values outside Q16.16. + +Integer arithmetic wraps at 32 bits. Fixed multiplication and division follow +Fixxy's rules, including its negative rounding behavior. Fixed nonlinear functions +use integer range reduction and twelve Taylor terms, with sigmoid/tanh derived +from a stable exponential. Fixed exp accepts -12..10, underflows below -12 and +rejects larger inputs. Sigmoid clamps its magnitude to 12; tanh saturates beyond ++/-6. On the tested -5..5 eighth-step range, maximum errors were 0.000008 +for exp, 0.000022 for sigmoid and 0.000043 for tanh. These are approximations. + +## Resource limits + +GotA retains its 32 MiB native memory allowance. Package storage, parsed manifest, +packed tensors and transactional scratch space are accounted for. The VM permits +256 native buffers independently of its 32 ordinary BASIC arrays. Tensor operations +charge one additional work unit per 256 scalar operations before executing. +Nonlinear functions charge 32 scalar operations per element. This supplements the +ordinary host-call and BASIC instruction charges. Size validation remains enabled +in release builds; proven bounded packed reads do not repeat integer overflow +checks in every multiply. + +The reference David 1407/512/92 network consumes about 12,551 work units per step. +The 40,619-neuron, 1.06-million-edge, four-step Fly benchmark consumes about 80,249. +Both fit GotA's 250,000-work-unit decision budget, including initial loading. + +## Converting current policies + +The three existing converters accept `--tensors`. Their default output remains a +named-runner ZIP. To convert an existing Richard, David, Andre or Fly runner ZIP: + +```sh +python examples/gods_of_the_arena/tools/tensor_packages.py input.zip output.zip +``` + +This retains observations, inference cadence, reset behavior, sampling and action +code. The forward passes live in `examples/gods_of_the_arena/neural/tensors/`: +Richard is affine/ReLU/affine; David and Andre compose recurrent gates and highway +mixtures; Fly composes sparse updates, gather and readout. No native MinGRU, +PufferNet or Fly-specific operator is required for these tensor implementations. +New architectures using this operator set need only a different BASIC policy and +ZIP tensors. Convolution, attention and training are outside this operator set. + +Synthetic runnable packages are `neural/examples/tensor-{author}.zip`. For example: + +```sh +nim r examples/gods_of_the_arena/gota.nim \ + --bot examples/gods_of_the_arena/neural/examples/tensor-fly.zip:10 +``` + +## Validation + +`tests/test_tensors.nim` compares mixed-sign weights, recurrent steps and resets, +checks package failures and sandbox limits, and reports floating differences. +Integer and fixed Richard outputs match exactly. Andre and Fly states matched +exactly in the reference fixtures. David's largest measured state difference was +5.96e-8, with at most one raw Q16.16 output unit of difference. Differences are +reported and bounded, not replaced with updated reference outputs. + +`tests/test_tensor_matches.nim` runs ten-player matches for all four public +synthetic policies, compares every action and simulation hash against the named +runners, and verifies recorded replay playback. Replays retain recorded commands; +playback does not rerun inference or suppress mismatches. + +Run `tests/bench_tensors.nim` for load, memory, work and warmed inference timings. +Performance depends on architecture: bulk dense kernels can improve throughput, +while small models pay additional host-call overhead. The specialized runners +remain available when that overhead matters. + +### Example native timings + +Apple Silicon release build with compiled BASIC, October 5, 2026. These are +synthetic model inference timings, separate from loading and observation code. +Small runner timings include enough iterations to measure microsecond costs. + +| Network | Named runner | BASIC tensors | Tensor/native | +| --- | ---: | ---: | ---: | +| Richard integer 25/16/18 | 0.32 us | 2.23 us | 6.9x | +| Richard fixed 31/8/19 | 0.36 us | 2.72 us | 7.6x | +| David 1407/512/92 | 974 us | 687 us | 0.71x | +| Andre 45/12/12, one layer | 0.99 us | 12.28 us | 12.4x | +| Andre 45/64/12, three layers | 23.83 us | 52.90 us | 2.2x | +| Fly 40,619 neurons, four steps | 7.41 ms | 11.49 ms | 1.55x | + +All reference models stayed within the existing work and memory budgets. Loading +plus BASIC compilation took 0.15..0.46 ms for the small models, 4.22 ms for David +and 17.53 ms for Fly. The ten-player, 2,400-battle-tick synthetic matches had +identical command sequences and simulation hashes for every author. Whole replay +file hashes differ because policy source and names are part of the replay header. diff --git a/examples/gods_of_the_arena/bots.nim b/examples/gods_of_the_arena/bots.nim index 5ad3edeb..0b264d61 100644 --- a/examples/gods_of_the_arena/bots.nim +++ b/examples/gods_of_the_arena/bots.nim @@ -5,7 +5,7 @@ import std/math, bassy, fixxy, polyworld/[policyhosts, llms, mailboxes, metrics, bodies, cli, controllers, - pathing, profiles, tapes], + pathing, profiles, tapes, tensors], neural/[common, richard, david, andre, fly], content, maps, @@ -120,10 +120,11 @@ proc heroVmLimits(): Limits = result.maxSourceBytes = 64 * 1024 result.maxCodeInstructions = 20_000 result.maxArrays = 32 + result.maxNativeBuffers = 256 result.maxArrayElements = 4096 result.maxGlobals = 256 result.maxHostData = 128 - result.maxHostFunctions = 128 + result.maxHostFunctions = 256 result.maxRoutines = 64 result.maxParameters = 16 result.maxRegisters = 256 @@ -1035,6 +1036,7 @@ proc initHeroHost( discard result.addFunction("readTile", 3, readTile, 128) result.infoFunctions(heroId) let context = NeuralContext(policy: policy) + result.addTensorFunctions(policy) result.addNeuralFunctions( richardRunner(context), davidRunner(context), andreRunner(context), flyRunner(context) diff --git a/examples/gods_of_the_arena/neural/examples/tensor-andre.zip b/examples/gods_of_the_arena/neural/examples/tensor-andre.zip new file mode 100644 index 0000000000000000000000000000000000000000..5bf6fda40f52cf1f57527b339f71279dc97caec8 GIT binary patch literal 2152 zcmZ{lX*kq-8^`}>j4@;!TahIdW;E8ahe?W=ERoUJCWI^v84Zn)EY%tNRy>k2wy_L~ z!cZEGr6F4)h2wdi7pK#8J@@_MzOLWz`rg;~{=WIxV+8g{0stTc1h`nb z*KQ}&lLP?(Ap-!yyRRX^LH^#?HM~5DRTstxV+OFOb!QkV7JZkynekM#PqNeKig%%Z zoXUvYT0+igJ^ea0`>f0_@0V9p&dHSGDr1$>I(2XBGIK4uwZ~u(T-)o#yct*$6IGcw zCw2#YP>E6~q*6-YVIBOzzlZuGHR~?CRQAs3&)O0|87K*DWm%dCqI!`F>ZMmh0AJqA ziyDu_B;ryxl3$0qAJiCCDSLkVQjiwqsHRUX^q6+xa`6`wP6k1v^C0c{_i`D7WjVp+ zc^Z|0vRH{|GP^Mib&u&id^vF6+z|;krtK8xzLke}YS$idYP~$kD5z28=&6*wo#N@Y zu6TeFB*=e^bPxa`?lFE@rUsP^pES z%Neb3WR&_ru>}!Ly|tu%e%gpeId%2Nk!^5HlPTi!vu`gfb!ImlV-Ax$&73%2Ne?Dd zGT)FLG2YxuXdKmG0SA4bZOD^a@$)UPSzY<`iBvW@m{=u6tx3zYN86?DV#%kU zGpi|{NJ4ARwOn*gshvWV<)W5WSAgT*s>f!t4dd1F#ecTYup^fLB-=E%e zunSE=uTXJ^3aq9MkLhdsraP?Gk*LyPkX#nbu*hLF$c8+m#3lG+Wcy10Cq znZdc^+T(X6;`1n;$_CNyRk4!dGWIaOx*>|jLte<~Rgg^B=p_yxq4UV-a)HbAKY`E>(Epl}&Xg&`eG_ zcOXmgZP%ZX$IQO4KG;6s#&(#LLH4dL1RKqG>ef?ncD?Jw)TYyo?Gc)1Y%UT@c}(P7 zV9?eLE0y4hf_z)M{s^4{b}vhHaGu?!35qh0u;)5GVV5lb(dr0)z(;R)3h>Dyp(>F`r8c>LtyU3^il4f9#b+nTfd=y(RF;H@g|b% zgHq0`kWaPdCx3?4?Kzu35?vKvZY^^gjndY!k%Hz~72kQgmQHEfMyjYJ&x~zFCfl@& z3bC^LdF|>*`pa{!KTK;uPtW-cD`O(GM6bFETV*C7O2Yjzbw!t!>Qd%Zs-oeLR|~_q zGeI*lwa(vj#>P<2`7U=lsu9wc;$rg?3p$dA14sh%7!Eai46HHmEyZ%Z`tcB7=9)-WZ~YC$$&zEU@`C}%i(r~;sOsYY&TtqTz?t!k8+D9nEHChn8Q*XZ~C z^Txz}_=)Eu4mq@DJNrpl)p}pX$wGa+b5v@)la^zJfHCDwxpZ^4+-w8bioM}>*ps7vuGD~QoX&;s^S~1k%*50mnefijwa~WAcd|O4|+w2rOCvXz}qS^ zX`Q0UArX*38c)&;w_loil${((b#$X^U-dNU<87yayGJK(f0#9^!3>=IbQ5h|^gXaO zVU2P3JKp{ubZR_m=aRfjrEop~5ZRq!J_KTLC{ZJT7)(g+LkTmLU-=@qb@UJyLBtg^&;(=y- zb}5O~y`H-t#;Qp#FM2jKE{#uYYrF$tnLZH=k3F#V|0U1CErE6$J zgyVN{&j0=2TK~nlICp37wb#4ez2E13pS9O>_2{T$<4|B=U=Uz9n<|>mu`)=WKf=H; zdxe2P_~+Kc-NniF4Udhr_n`#DeOV=K^^7Yvh?XOj;p83l8qM?~%EWv0DTGd7!8#1v z^`q-I9e5Vp=2!4+1fTYB#TR*NA8eACUPz5I*uB2;<1 zF1VoZyU;ck7l-3dB=t#G zsVt@R?k{(0jBkD}-HKPK;f8poZ#hM+TP8;xi>se4Yqj;qh2H9=@br-N zs5Ah2oIW>7UWrzz3QV8fnq_TcuX8GeqkIqIqYubVojbT{?3sg@6{R)v4X%p%KI8a7 zjO&Voc{+D!tRHZDG(V7oWIDX0(7Xazm1D3TIDSeci@rHIX*XG8W5w2A>{OECXzge& z*?}=}c$s_@m&=T)s#XHI2@8uDra$!Zjcq@0GqWcRcRziKe}o?-s4*VuE^pKyL?Kh2 zAeHJIV>ng})U6~sOQn!I8#_Lea&I)hvkTW%V?BhD&HVP!FLTMWdJ(&!5j^a5UR8>4j3W8zGvCX)vA?+;KJ+&JyR?h+`udP zr@4qMfi$dS=jmi{-j@iEaFre}QYv|}jWpgnNh=&f(-l=v*6K)>Qm-F;C9NEb=v{mZ zE_l2)O2piPr|Q7X-I^(^teap_m@p3@CeDQGNgUGiV6uYz9IFuro2)g@bwMzt?%tytZ35|MXjRjfj8+7nF^*D5A2De8JL2?+iu7XA?6SATe2og& zKcHH=SE)is5EjSWRA4P(&)*gb%M*=4!i$q>0ZbyU6=LYP3UHnTD_?AO_S z9~sATpN#cBjKw{R|Md<3ojquclS-B!ujB{=Mnkf2)IDBB{9;;qyn$v#6xGMrM?_3R z)tQ5T<8H_)0Gc*fIEO5t!O_4^*?`fy9Sy7$c@gkfv0o|vWSugTu{F~QQLlM@7%H5*5l<(`&LyTVq=7hTm>%Hx&= zee&(y=2l!yMKt3O&<(GmwiO`HRa~#6ooB1D+{B_-QcN{8aan$*&~ocR z1=tT{?Kl`|%bm~G^>e>KDIRYEVEROOFaq;XV!2m}jV5)AYiYHCsc3GviD7>$aN7jw z0*;ijGo937U@4konYY`~7k@XxHbCuIsUpdr#o{3o9s4kph6l+U^r7r46N27LcqwYA zz<;xoG%7z3WrjAY({zUyb`q5X zveZQutoBLk$;71wau#a%Jj#putcRbLC0ZA%o%y-8u`HiNuA~1GY3Rj zd{|K+q^P1p6yMvkMwdcWhZ)6f76jl&@f0g%Vu#9%OAGT)0ac%y=nvp|xae&33)P{w zKg2Tk3y>Y>vZGzuIhO4u13Iy|Lhk(*Eaf14LYrtOt%+{YqkPrAil?qe5M^+w=$AVB zM%u#5kjm=D@HRy*TKrcYJ^R1sgeot;z>Gv&I=U9I zz(KMVd<-@i!?=0^hDV!;g_P94Wcdln1-&-fUS(UIn&VV_i_);)GOhnQ1$$^qt5zWd zX3lN$_fZ`#7{iI1B_|{6ekam1N9S`n^tMaYm{(x*Tny2)6?dpAVFd*)TkX$H1mC=J zc=cGW6)i`=?3C;Eo_*#Zmu-{O0aQ&yR!SSDjp<;D@U#(v4Ni-;Tc=3(>C6sKsAfXa_esPz zDSQbz4>AlEIjZaRjdUyCm$*UqG7J||xN98VizUXyXj9}E8?$#3X4W5XF^_$%?G6Ms zNVL|EW;TuzIi>5e5R7=U?ARWOQXHP!sWexqRz0Ghd8pj9OH6los$k!oPSFL!lEeGh)~<8|VL4{}R;QAOWiQK=_G&Xp-Npl&oCpL)ei#Cu)Sbi>ZM zz1xBC4Y2)62tBdFn>0-?_B?Jz;+uovFj;B<70G~xypB6i#I3?sq6DjRxULpY!gaZA zVmnDl_qkS%mu>q6m%%ueSEZ4y^HBwu&J)&9Jz(j^ka4#)gJnM|2%YP z24c-+nBksJN-m(TR#j=JQn!&`2@(s;S~RE+nRVK7acpD8gwbc>y$zHt+HZJ&0cz9N zb@ku9&=wX=3c3B3QSuGv8O{M`LESDcMXT~rH@+a=Wsn9z9fy^tvLh?Q(&cdO&%}pT z0=Av6bUFYs2)|j@{h#|?H!r)VO7ZqcqNya9OnYGXO!W4aGIW~4NsEUeXTQ3M4m5V0 zklf;#ud~KuA4_nl?4pPq2e>ep%gY0z>V2k^d}twX47-}qK7oz$%#RqY0pIZOsBtqD@y|Ov|-5Xrx#_a z&^SHIYGn5Fi!ya6K;LoY{7Nt7gL9Clf6qG;%m^Fk@V=~k?$$11p!STto_cBEQ08O}QatnZiPheuNb=utm`Ly=CM!$!etVB; zJvRtZ?LlK$H*{k#*O5Me@u_YfPq)0MbD6=_j>+J$TJD>$vP@%7%Dze+XEFPxN=Lr2 zS{w!-FgtHf&zSQE@!$kDuTNfS3x&t z6QAbl4jb&HUuJ#TfUY-5@_gc*rZrmjPY;#=bAAfiXb;562kpPNPs0*p-V#p@lf zdDRr6l_wea3l2%oX{=2@65MjLMy{|_Xl#itFSJc@=(XzfDva-O1=qVdtmQYg@r<`S zUcI{RtJ8mX`s=OH-87*7v8kbrc-e5Nsb{A=FKTVmi>pCehR8f~N7NrZh^&^Y1>~in z4Ow)35S%QdHG+F2oWIZ${;6uOm8|*Iw@w6eb7iK)Gn=$#_ZUXKr(~CWv%l4oBwn5Y z$099AX*azvC^)a^?Lp{u^mcMfr(7n0rD!D8IIAw#>eM`vFusBwkbKJG{c?&8#5_ zQhH4$kzv^Vj{ES5$3DC;e~`gM&o!~CLTYD0BhhlRtVcaKP~&KL&m!J^n8KjX@yVh? za=e}MN|B|8$2N|+nU6#d7zKY5x{3i}52O~@Y#@&izg)s19cj(8x~uO-6=&J5AH+^G6Ed{&EhMpn|y-eCg0>5S_@;(vsc? z9K5fA6RNY7H(9K=J0;>_4^&2R&uwC#^=0xc(+axpWJxTHA3dxQ2JG~ z{CFk4N1K}#M`W=dR_8Ky>DKu|&c;$XyXA@9F{{}V4-r0B$iqe-Q_qWy50`QE#)H%g z%R#^^Fz77%XE5N}See-?gXvvcty&fu_d*ktDpqO2qbA&Y@d(0mtv)v2&936J!{3&o z+kUGNZp9RuSwyF-TA(fhq|R8 z?<9&Zx81j^S+1CeosnT*t;?&Vdf2}Dx4*@@<(dWxbyNI+zNOW;fcle@Ag>wxG%rK{ zMxp2SCax=jf4vQ;*}XNb`aMuIr+#0Fpg$y=cm|yqYFKw~lV^5!Se$!4?ZTB_tZDqw zOHn)>#&+z@-pVRXj={E9!oGI#KNo0oi9LggErJLvR`08NQ+BX)|werz87+qTp^C@x%2ij;IqW?j>& zJL0H;QWYjKTPl(CJDTKvQh~8$IaT?E{e~6n)m^~8%+I~>o2)ELhDzoKOhHixXpHqicZ96z5hj#UmV=&zjuER(thnEgs zNRRh9*p(S<-`YOFMrm?c_DWA95>Dw9NXfdO?|UZaG+RC>zEKT|ap!a)gbt%E8OxB7XO~PWdG@g#JramO=6+T5O}4Pfsh%oL|6-{Ns)WVXNykEvbeF|2g8 z4l3u>IAz1n=UFPs>{(0a;80B?5)|pV@^%n`0KPSfnF1ns{IV7qqXb>Oc=%yBkE!T1 z_gv#uORlK-GZeFe`J>|KQR%{dD>Tsc8mMO&eZq+jwlEx0 zYp!JO{ymatRxvk36)kmtEgO`X8e(yR&b$?S2xB~aZd$z|Sv9uVDu{5x5~wc~RoWz$ zNk|{Iv2|u#Dsn1vG$y*?)%iDXP`H1UDD;mf2*kv|ApR3ygWbH{y}Wsxz1`gslVl0| znaLE7`5_!^q*?x%DN*yORHodepvbkaH9Bg>t}V&Kzd~Z<_<)mETMQ*oGMeI@G$0;k zx671PpCqbeNa=LmmmflbJ$A+t%2APN{^eWk+$L3ip4^B*&j_cK6I(04{&VT?{epyv z%1depwh$8koBQ7$44ZJWeLc&6XNvCxUF-kD5+ffEgXnKeZJgW=C;3P)UJzr~zaaW3 z#~b!LlZ`Gc-RV!w|F8e6A1EyUg8jejt@0ls*MnF(>X=v**#FtH{_k0#e<|Q^4)#BB t|J3upaP)tg|6SexME+A>|3VJ^4f#iHI_kK1e?4OV*)@M+Bh_Eoe*ubUg}nd( literal 0 HcmV?d00001 diff --git a/examples/gods_of_the_arena/neural/examples/tensor-fly.zip b/examples/gods_of_the_arena/neural/examples/tensor-fly.zip new file mode 100644 index 0000000000000000000000000000000000000000..f9967cc6f44aca9e9a71bf2563de5ebce76dc00a GIT binary patch literal 2050 zcmZ{lc{J2*8^?cRml;av(IERY#xqnNLu4CEmaN$l&lo#n#xMpsG+GFWGBMfrC8o)g zHKZ(Il&vgfDT#_awvlDz^}g>PZ*|UlUw>TZzRvkv=lb5)IrrDrlAA{i0003X@PdiE zxovX6dtLyD69s?+`_&KvF2E}aiS-PlySNeFod(DDGY)Q_EVq1gu?f$c0djAmgX(e% z1z$ntO=`ZF-oQ#Sb68vKgr;$3dYzU_Tv2ULD)nq^$*eLCE#`eaZB6{8+)1|lfGQtUe8)h%$VM%2K$t9|q5aKaBnv=TgmWbfDbE=zGDox)Q3VSU zH;TJqj)tx3_sJf1c0Mvx>^Qn*WIhI2l4P{@HaE^uvuWePu<+#v$%JH208)3p!mTbb z%XR5?VYn7+Ro2o0e9@4U5o-7*k~AW8Mz4nQ&eVlnz;Ya5FT&-Z~h5 z)k17mqwK)sgkmD0&?D$dU8V)`)fhZZ5Gty~=`+aveo3SmoQ~<8gaATWba|M*AWh=r z>nH7{&iQgJ+ZCT(pU2N>%8123(T~f{v{VYH4h*TCmspMN5oa|bDKamFS2l;@vU}w0S^pOf>1ejUPpXtqX6{2lx#Ak2e+T2J1W*m`=01Y=Gog%)QCxSpVhI z9CR#>xK1=eP8ENM(kb%|Y~~-Ezl}PBLbL`*lCu*P^23SMS};6LnwU>0@A?K(kGbBCDd*sLwhp=~!6xfI{1#UK{X-eJYV zZ!siSJEel-S0ha4NaD*bh*L|;#}(4+PwesA{^A40!$}{y_W%EOZU6x9XOa&-j1U@z3=AXS=|htTuBeOgFGpyP zDukXAREwr~7&OzO`fdj^U&?nx*qp!~9qUH+>&52m_Fs5-JY|S)ri8bLA#kFiNZlWD zu=`uFjMKHpH#chGTedJ&rGmi<)BN~vnKeDv7DZA<<)+B9=Wdy0cDE&fE4-%ca#-fs zCZ(HdfwL9&-m?_ha+N!q7n7kfHG>uptSL+L%d0!)4QrnFpw*^%dsUMd;i@iVgjHg> zR7`IIpnK$#=} literal 0 HcmV?d00001 diff --git a/examples/gods_of_the_arena/neural/examples/tensor-richard.zip b/examples/gods_of_the_arena/neural/examples/tensor-richard.zip new file mode 100644 index 0000000000000000000000000000000000000000..e5d11eaf631ce594ae920b7e8a275a29b25f2cbd GIT binary patch literal 1298 zcmWIWW@Zs#U|`^2$cl1}4ewsz&&SNb@R5^&feR>Fke`#8T&b6oSiChXH~Y4Mz`u9# zicSl^r8!;VJ$8e;B-j0rRoJ9MW;eu-?ov6PbMi#dPg-U7j8Ui^T_Zz zZ=Qi%?}_`{>P$rTES{X0r|@<5ndP3VwtU>ZlqY)cF<$<8=2JF*W9G6r_jH}WJ4tnm zEq6_8?Hoh4N8L~0`dxE5rh={T-FFuo=G&)Nh=-RK-(SM@y|>Ke>cxGlj2En9o12v} zb>$L{>b0RSB-1ZE+7a~fo&JGR!&MEEDV29hew^KwJ&SK&*#pbj?^R-tCAjQ3V{qe> zN^fJXiIh^3_tKY}W^6pYJM+X6qn_xLq;gj81j^`-sCE(x~_FxeVV<`F#gDc)XnUvovv+QbO2H#_7V?vl9Mi;N#6T|f8X!NH}=k7U|rSiM^kuO?bvJO3QVBHr1nzaMcacHWvDnV;C} zE?FJwczoM}$L6V@1&@V&O#XPZZJXAs2{%kyR^PLn9mX4Z!ysyP&z(7u|0nNW(>X;< zW9Pk%qW&5Vs+`l6q@xU1x2$S5v-wxMzDD~;|Aq^H(*ry|um_;znr9DIbhiO>%M?Zi z1|A??lA2eXUsSA@Rh*yK+waJASV6?){io|fi*j9JzDX%^OV>+@UwvdgMQBg8Qprsd zy>`ia$zdKZG+mhPu77Y=*zvIR-DJNj3*?)Ra`5E*+T-=OW><&p@ve)`%Kw~`R6XgyYkLI{cZZ8M=zO=X7PTi%CWFsm?h#=IJ12Z zBdU9uPOa5<0=mz zEaU 64 * 1024 or args.weights.stat().st_size > MAX_BYTES: parser.error("Input policy or model exceeds the package limits") source, model = convert(args.source.read_text(), args.weights.read_bytes(), args.hidden, args.layers, args.action_ticks, args.temperature) + resources = {"weights.bin": model} + if args.tensors: + from tensor_packages import tensorize + source, resources = tensorize(source, resources) with zipfile.ZipFile(args.destination, "w", zipfile.ZIP_DEFLATED) as package: package.writestr("policy.bas", source) - package.writestr("weights.bin", model) + for name, data in resources.items(): + package.writestr(name, data) if __name__ == "__main__": diff --git a/examples/gods_of_the_arena/tools/convert_david.py b/examples/gods_of_the_arena/tools/convert_david.py index 71b42df2..1ee67326 100644 --- a/examples/gods_of_the_arena/tools/convert_david.py +++ b/examples/gods_of_the_arena/tools/convert_david.py @@ -96,11 +96,18 @@ def main(): parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("source", type=Path) parser.add_argument("destination", type=Path) + parser.add_argument("--tensors", action="store_true", + help="Write a BASIC-defined packed tensor architecture") args = parser.parse_args() source, model = convert(args.source) + resources = {"model.bin": model} + if args.tensors: + from tensor_packages import tensorize + source, resources = tensorize(source, resources) with zipfile.ZipFile(args.destination, "w", zipfile.ZIP_DEFLATED) as package: package.writestr("policy.bas", source) - package.writestr("model.bin", model) + for name, data in resources.items(): + package.writestr(name, data) if __name__ == "__main__": diff --git a/examples/gods_of_the_arena/tools/convert_richard.py b/examples/gods_of_the_arena/tools/convert_richard.py index 0dc4b3ee..9a3556d2 100644 --- a/examples/gods_of_the_arena/tools/convert_richard.py +++ b/examples/gods_of_the_arena/tools/convert_richard.py @@ -101,8 +101,13 @@ def main(): parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("source", type=Path) parser.add_argument("destination", type=Path) + parser.add_argument("--tensors", action="store_true", + help="Write a BASIC-defined packed tensor architecture") args = parser.parse_args() source, resources = convert(args.source.read_text()) + if args.tensors: + from tensor_packages import tensorize + source, resources = tensorize(source, resources) with zipfile.ZipFile(args.destination, "w", zipfile.ZIP_DEFLATED) as package: package.writestr("policy.bas", source) for name, data in resources.items(): diff --git a/examples/gods_of_the_arena/tools/tensor_packages.py b/examples/gods_of_the_arena/tools/tensor_packages.py new file mode 100644 index 00000000..ff54d947 --- /dev/null +++ b/examples/gods_of_the_arena/tools/tensor_packages.py @@ -0,0 +1,177 @@ +"""Export named four-byte tensors and BASIC-defined forward passes.""" +import argparse +import json +import math +from pathlib import Path +import re +import struct +import zipfile + +LIBRARIES = Path(__file__).parent.parent / "neural/tensors" + + +def tensorize(source, resources): + """Replace reviewed runner calls while preserving policy control flow.""" + entries = [] + payload = bytearray() + libraries = [] + + def add(name, dtype, shape, data): + """Append one portable named tensor without duplicating resources.""" + count = math.prod(shape) + if len(data) != 4 * count: + raise ValueError(f"Tensor byte length mismatch: {name}") + entries.append(dict(name=name, dtype=dtype, shape=shape, + resource="tensors.bin", offset=len(payload))) + payload.extend(data) + + def extract(name, dtype, shape, model, offset): + """Copy a row-major range from a validated author checkpoint.""" + count = math.prod(shape) + add(name, dtype, shape, model[offset:offset + count * 4]) + return offset + count * 4 + + for filename, model in resources.items(): + if model.startswith(b"RICHNN01"): + kind, = struct.unpack_from(" expected or expected - (len(model) - 16) // 4 > 7: + raise ValueError("Andre model byte length mismatch") + model += bytes((expected - (len(model) - 16) // 4) * 4) + extract("encoder", "float32", [width, 45], model, 16) + extract("decoder", "float32", [12, width], model, 16 + decoder * 4) + for layer in range(layers): + extract(f"recurrent.{layer}", "float32", [3 * width, width], + model, 16 + (recurrent + layer * stride) * 4) + add("layers", "int32", [1], struct.pack(" 64 * 1024: + raise ValueError("BASIC source with tensor architecture exceeds 64 KiB") + return (source, + {"tensors.json": json.dumps({"tensors": entries}, separators=(",", ":")), + "tensors.bin": bytes(payload)}) + + +def main(): + """Convert an existing runner ZIP without changing its policy settings.""" + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("source", type=Path) + parser.add_argument("destination", type=Path) + args = parser.parse_args() + if args.source.stat().st_size > 16 * 1024 * 1024: + parser.error("Policy exceeds 16 MiB") + with zipfile.ZipFile(args.source) as package: + entries = package.infolist() + if len(entries) > 256 or sum(entry.file_size for entry in entries) > 32 * 1024 * 1024: + parser.error("Package expanded limits exceeded") + sources = [entry.filename for entry in entries if entry.filename.endswith(".bas")] + if len(sources) != 1: + parser.error("Package needs exactly one BASIC entry") + source = package.read(sources[0]).decode() + resources = {entry.filename: package.read(entry.filename) + for entry in entries if entry.filename not in sources} + source, resources = tensorize(source, resources) + with zipfile.ZipFile(args.destination, "w", zipfile.ZIP_DEFLATED) as package: + package.writestr("policy.bas", source) + for name, data in resources.items(): + package.writestr(name, data) + + +if __name__ == "__main__": + main() diff --git a/nimby.lock b/nimby.lock index b96327b2..3978472f 100644 --- a/nimby.lock +++ b/nimby.lock @@ -1,4 +1,4 @@ -bassy 0.1.0 https://github.com/treeform/bassy 907b69f2d0d4f32186d72947883f5bdac159ce9d +bassy 0.1.0 https://github.com/treeform/bassy b1975841adf49320fb3ae28df062500ac5a2ff47 curly 1.1.1 https://github.com/guzba/curly 29558e2d2b76244a2c65fea08e9f4efff16c1997 libcurl 1.0.0 https://github.com/Araq/libcurl eecbfc31e17ae2f09f0b6d2aaefcfb970c7d2be4 fixxy 0.1.0 https://github.com/treeform/fixxy 05e5446dffb70093056cebb0c57721a60deaf52a diff --git a/src/polyworld/tensors.nim b/src/polyworld/tensors.nim new file mode 100644 index 00000000..b759be05 --- /dev/null +++ b/src/polyworld/tensors.nim @@ -0,0 +1,964 @@ +import + std/[math, strutils], + bassy, fixxy, jsony, + policies + +when defined(vcc): + {.localPassC: "/fp:strict".} +else: + {.localPassC: "-ffp-contract=off".} + +const + TensorMagic = "PWTENS01" + TensorRanks = 8 + TensorElements = 4 * 1024 * 1024 + WorkScale = 256'i64 + +type + TensorError* = object of BasicError + TensorDtype* = enum + Int32Tensor, FixedTensor, Float32Tensor + Tensor* = object + dtype*: TensorDtype + shape*: array[TensorRanks, int] + rank*, count*: int + immutable*: bool + data: ptr UncheckedArray[byte] + TensorEntry = object + name, dtype, resource: string + shape: seq[int] + offset: int64 + TensorManifest = object + tensors: seq[TensorEntry] + TensorContext = ref object + policy: Policy + entries: seq[TensorEntry] + loaded, charged: bool + BinaryOperation = enum + AddOperation, SubtractOperation, MultiplyOperation, DivideOperation + UnaryOperation = enum + ReluOperation, SigmoidOperation, SigmoidDirectOperation, ExpOperation, + TanhOperation, AbsOperation + +proc fail(message: string) {.noreturn, raises: [TensorError].} = + ## Reports tensor failures through the ordinary BASIC boundary. + raise newException(TensorError, "Tensor: " & message) + +proc dtype(name: string): TensorDtype = + ## Resolves an explicit four-byte element representation. + case name + of "int32": + Int32Tensor + of "fixed": + FixedTensor + of "float32": + Float32Tensor + else: + fail("dtype must be int32, fixed or float32") + +proc finite(value: float32): bool {.inline.} = + ## Rejects nonfinite values at every floating tensor write boundary. + classify(value) notin {fcNan, fcInf, fcNegInf} + +{.push overflowChecks: off.} + +proc word(data: ptr UncheckedArray[byte], i: int): int32 {.inline, raises: [].} = + ## Reads a little-endian word without assuming pointer alignment. + var raw: uint32 + copyMem(addr raw, addr data[i * 4], 4) + when cpuEndian == bigEndian: + raw = (raw shr 24) or ((raw shr 8) and 0xff00) or + ((raw shl 8) and 0xff0000) or (raw shl 24) + cast[int32](raw) + +{.pop.} + +proc number(tensor: Tensor, i: int): int32 {.inline, raises: [].} = + ## Reads one previously validated packed element. + word(tensor.data, i) + +proc floating(tensor: Tensor, i: int): float32 {.inline, raises: [].} = + ## Reads one floating element without converting it through BASIC. + cast[float32](tensor.number(i)) + +{.push overflowChecks: off.} + +proc writeWord(bytes: var string, offset: int, raw: int32) + {.inline, raises: [].} = + ## Writes a portable word without native object serialization. + let value = cast[uint32](raw) + for i in 0 ..< 4: + bytes[offset + i] = char((value shr (8 * i)) and 255) + +{.pop.} + +proc dimensions(kind: TensorDtype, shape: openArray[int]): Tensor = + ## Bounds rank and element products before allocating any payload. + if shape.len < 1 or shape.len > TensorRanks: + fail("rank must be between 1 and 8") + result.dtype = kind + result.rank = shape.len + result.count = 1 + for i, size in shape: + if size < 0 or size > TensorElements or + (result.count > 0 and size > TensorElements div result.count): + fail("shape must contain nonnegative dimensions within 4194304 elements") + result.shape[i] = size + result.count *= size + +proc tensorView*(runtime: Runtime, value: Value): Tensor = + ## Borrows a checked tensor only until the next buffer mutation. + let view = runtime.borrowBlob(value) + if view.len < 16: + fail("invalid tensor handle") + for i, c in TensorMagic: + if view.data[i] != byte(c): + fail("invalid tensor header") + if view.data[8] > byte(ord(high(TensorDtype))) or + view.data[9] < 1 or view.data[9] > TensorRanks or + view.data[10] > 1 or view.data[11] != 0: + fail("invalid tensor metadata") + let + rank = int(view.data[9]) + header = 12 + rank * 4 + if view.len < header: + fail("truncated tensor shape") + var shape: array[TensorRanks, int] + for i in 0 ..< rank: + shape[i] = int(word(view.data, 3 + i)) + result = dimensions(TensorDtype(view.data[8]), shape.toOpenArray(0, rank - 1)) + if view.len != header + result.count * 4: + fail("tensor byte length does not match its shape") + result.immutable = view.data[10] == 1 + result.data = cast[ptr UncheckedArray[byte]](addr view.data[header]) + +proc writable(tensor: Tensor) = + ## Prevents loaded weights from becoming operation destinations. + if tensor.immutable: + fail("loaded weights are immutable; copy them into a scratch tensor") + +proc sameCount(a, b: Tensor) = + ## Requires two tensors to contain the same number of elements. + if a.count != b.count: + fail("element count mismatch") + +proc sameType(a, b: Tensor) = + ## Prevents implicit conversions between arithmetic representations. + if a.dtype != b.dtype: + fail("dtype mismatch; use tensorConvert explicitly") + +proc sameShape(a, b: Tensor) = + ## Requires matching dtype and complete shape for elementwise operations. + sameType(a, b) + if a.rank != b.rank or a.shape != b.shape: + fail("shape mismatch") + +proc broadcast(tensor, output: Tensor) = + ## Supports a one-element scalar or an exactly matching tensor. + sameType(tensor, output) + if tensor.count != 1: + sameShape(tensor, output) + +proc meter(runtime: Runtime, operations: int64) = + ## Charges bulk arithmetic separately from BASIC instruction execution. + runtime.chargeWork((operations + WorkScale - 1) div WorkScale) + +proc payload(runtime: Runtime, tensor: Tensor): string = + ## Reserves transactional scratch space before allocating an output. + runtime.checkNativeMemory(int64(12 + tensor.rank * 4 + tensor.count * 4)) + result = newString(12 + tensor.rank * 4 + tensor.count * 4) + for i, c in TensorMagic: + result[i] = c + result[8] = char(ord(tensor.dtype)) + result[9] = char(tensor.rank) + result[10] = char(ord(tensor.immutable)) + for i in 0 ..< tensor.rank: + result.writeWord(12 + i * 4, int32(tensor.shape[i])) + +{.push overflowChecks: off.} + +proc store(bytes: var string, tensor: Tensor, i: int, raw: int32) + {.inline, raises: [].} = + ## Writes one element into unpublished transactional scratch storage. + bytes.writeWord(12 + tensor.rank * 4 + i * 4, raw) + +{.pop.} + +proc storeFloat(bytes: var string, tensor: Tensor, i: int, value: float32) + {.inline.} = + ## Rejects a nonfinite intermediate before publishing any destination. + if not value.finite: + fail("nonfinite float32 result") + bytes.store(tensor, i, cast[int32](value)) + +proc create(runtime: Runtime, tensor: Tensor, bytes: string): Value = + ## Creates a new owned tensor after checking its buffer bookkeeping. + runtime.checkNativeMemory(int64(bytes.len) + 128) + result = runtime.createBlob() + runtime.putBlob(result, bytes, "tensor") + +proc shapeValues(runtime: Runtime, value: Value): seq[int] = + ## Reads a small BASIC shape array with exact integral dimensions. + let view = runtime.arrayView(value) + if view.len < 1 or view.len > TensorRanks: + fail("shape array needs 1..8 dimensions") + for i in 0 ..< view.len: + if view[i].kind != IntegerValue: + fail("shape dimensions must be integers") + result.add int(view[i].asInt) + +proc tensorCreate(runtime: Runtime, arguments: openArray[Value]): Value = + ## Allocates a zero-filled mutable tensor with an explicit shape. + let tensor = dimensions( + dtype(runtime.getString(arguments[0])), + shapeValues(runtime, arguments[1]) + ) + runtime.meter(int64(tensor.count)) + let bytes = runtime.payload(tensor) + runtime.create(tensor, bytes) + +proc toFixed(value: float32): int32 = + ## Rounds floating exports to Q16.16 with ties away from zero. + if not value.finite: + fail("nonfinite float32 conversion") + let rounded = round(float64(value) * 65536.0) + if rounded < float64(low(int32)) or rounded > float64(high(int32)): + fail("value is outside Q16.16") + int32(rounded) + +proc scalarRaw(kind: TensorDtype, text: string): int32 = + ## Parses constants without routing float32 through fixed BASIC literals. + try: + case kind + of Int32Tensor: + let value = parseBiggestInt(text) + if value < int64(low(int32)) or value > int64(high(int32)): + fail("scalar is outside int32") + int32(value) + of FixedTensor: + int32(parseFixed(text)) + of Float32Tensor: + let value = float32(parseFloat(text)) + if not value.finite: + fail("nonfinite float32 scalar") + cast[int32](value) + except ValueError as error: + fail("invalid scalar: " & error.msg) + +proc tensorScalar(runtime: Runtime, arguments: openArray[Value]): Value = + ## Allocates one precisely represented constant from decimal text. + let tensor = dimensions(dtype(runtime.getString(arguments[0])), [1]) + let raw = scalarRaw(tensor.dtype, runtime.getString(arguments[1])) + var bytes = runtime.payload(tensor) + bytes.store(tensor, 0, raw) + runtime.create(tensor, bytes) + +proc loadManifest(context: TensorContext, runtime: Runtime) = + ## Validates the complete package tensor directory once per policy VM. + if context.loaded: + return + if context.policy == nil: + fail("tensorLoad requires a ZIP policy") + try: + let bytes = context.policy.resource("tensors.json") + if bytes.len > 64 * 1024: + fail("tensors.json exceeds 64 KiB") + runtime.meter(int64(bytes.len) + int64(context.policy.files.len * 256)) + let manifest = bytes.fromJson(TensorManifest) + if manifest.tensors.len < 1 or manifest.tensors.len > 256: + fail("manifest needs 1..256 named tensors") + var names: seq[string] + for entry in manifest.tensors: + if entry.name.len < 1 or entry.name.len > 128 or entry.name in names: + fail("tensor names must be nonempty and unique") + names.add entry.name + let + tensor = dimensions(dtype(entry.dtype), entry.shape) + resource = context.policy.resource(entry.resource) + if entry.offset < 0 or entry.offset > int64(resource.len) or + int64(tensor.count) * 4 > int64(resource.len) - entry.offset: + fail("tensor resource range is outside " & entry.resource) + runtime.checkNativeMemory(int64(bytes.len * 4 + 4096)) + if not context.charged: + runtime.reserveNativeMemory( + context.policy.memoryBytes + int64(bytes.len * 4 + 4096) + ) + context.charged = true + context.entries = manifest.tensors + context.loaded = true + except PolicyError, JsonError: + fail("invalid tensors.json: " & getCurrentExceptionMsg()) + +proc loadTensor(context: TensorContext): ContextHostProc = + ## Binds immutable named tensor loading to one policy package. + result = proc(runtime: Runtime, arguments: openArray[Value]): Value = + ## Copies a validated resource slice into VM-owned packed storage. + context.loadManifest(runtime) + let name = runtime.getString(arguments[0]) + for entry in context.entries: + if entry.name == name: + var tensor = dimensions(dtype(entry.dtype), entry.shape) + tensor.immutable = true + runtime.meter(int64(tensor.count)) + var bytes = runtime.payload(tensor) + let resource = context.policy.resource(entry.resource) + if tensor.count > 0: + copyMem( + addr bytes[12 + tensor.rank * 4], + unsafeAddr resource[int(entry.offset)], + tensor.count * 4 + ) + if tensor.dtype == Float32Tensor: + let data = cast[ptr UncheckedArray[byte]]( + addr bytes[12 + tensor.rank * 4] + ) + for i in 0 ..< tensor.count: + if not cast[float32](word(data, i)).finite: + fail("nonfinite loaded weight: " & name) + return runtime.create(tensor, bytes) + fail("missing named tensor: " & name) + +proc converted(raw: int32, source, destination: TensorDtype): int32 = + ## Performs an explicit checked conversion between element dtypes. + if source == destination: + return raw + case destination + of Float32Tensor: + let value = + if source == FixedTensor: + float32(raw) / 65536.0'f + else: + float32(raw) + cast[int32](value) + of FixedTensor: + if source == Float32Tensor: + return toFixed(cast[float32](raw)) + if raw < -32768 or raw > 32767: + fail("int32 conversion is outside Q16.16") + raw *% 65536'i32 + of Int32Tensor: + if source == FixedTensor: + if (raw and 65535) != 0: + fail("int32 conversion requires integral values") + return raw div 65536 + let value = cast[float32](raw) + if not value.finite or float64(value) < float64(low(int32)) or + float64(value) > float64(high(int32)) or trunc(value) != value: + fail("int32 conversion requires integral values in range") + int32(value) + +proc tensorImport(runtime: Runtime, arguments: openArray[Value]): Value = + ## Copies observations from an exact BASIC numeric array shape. + let + output = runtime.tensorView(arguments[0]) + input = runtime.arrayView(arguments[1]) + output.writable() + if input.len != output.count: + fail("array import count mismatch") + runtime.meter(int64(output.count)) + var bytes = runtime.payload(output) + for i in 0 ..< output.count: + let value = input[i] + if value.kind notin {IntegerValue, FixedValue}: + fail("tensor imports require numeric BASIC values") + let + source = if value.kind == IntegerValue: Int32Tensor else: FixedTensor + raw = if value.kind == IntegerValue: value.asInt else: int32(value.asFixed) + bytes.store(output, i, converted(raw, source, output.dtype)) + runtime.putBlob(arguments[0], bytes, "tensor") + arguments[0] + +proc tensorExport(runtime: Runtime, arguments: openArray[Value]): Value = + ## Converts only final outputs into ordinary BASIC numeric values. + let input = runtime.tensorView(arguments[0]) + runtime.meter(int64(input.count)) + runtime.checkNativeMemory(int64(input.count) * 32 + 128) + var values = newSeq[Value](input.count) + for i in 0 ..< input.count: + let raw = input.number(i) + values[i] = + case input.dtype + of Int32Tensor: + toValue(raw) + of FixedTensor: + toValue(Fixed(raw)) + of Float32Tensor: + toValue(Fixed(toFixed(cast[float32](raw)))) + runtime.putArray(values) + +proc copyTensor(runtime: Runtime, arguments: openArray[Value], + convert, reinterpret: bool): Value = + ## Copies or explicitly converts a full tensor into a reusable output. + let + input = runtime.tensorView(arguments[0]) + output = runtime.tensorView(arguments[1]) + output.writable() + sameCount(input, output) + if not convert and not reinterpret: + sameType(input, output) + runtime.meter(int64(input.count)) + var bytes = runtime.payload(output) + for i in 0 ..< input.count: + let raw = + if convert: + converted(input.number(i), input.dtype, output.dtype) + else: + input.number(i) + if output.dtype == Float32Tensor and not cast[float32](raw).finite: + fail("nonfinite reinterpretation") + bytes.store(output, i, raw) + runtime.putBlob(arguments[1], bytes, "tensor") + arguments[1] + +proc fillTensor(runtime: Runtime, arguments: openArray[Value]): Value = + ## Fills mutable storage with one scalar of its own dtype. + let + output = runtime.tensorView(arguments[0]) + scalar = runtime.tensorView(arguments[1]) + output.writable() + sameType(output, scalar) + if scalar.count != 1: + fail("fill needs a one-element tensor") + runtime.meter(int64(output.count)) + var bytes = runtime.payload(output) + for i in 0 ..< output.count: + bytes.store(output, i, scalar.number(0)) + runtime.putBlob(arguments[0], bytes, "tensor") + arguments[0] + +proc integerOperation(a, b: int32, kind: TensorDtype, + operation: BinaryOperation): int32 {.inline.} = + ## Preserves wrapping integer and Fixxy arithmetic rules. + case operation + of AddOperation: + a +% b + of SubtractOperation: + a -% b + of MultiplyOperation: + if kind == FixedTensor: + int32(Fixed(a) * Fixed(b)) + else: + a *% b + of DivideOperation: + if b == 0: + fail("division by zero") + if kind == FixedTensor: + int32(Fixed(a) / Fixed(b)) + else: + cast[int32](uint32((int64(a) div int64(b)) and 0xffffffff'i64)) + +proc binaryTensor(operation: BinaryOperation): ContextHostProc = + ## Builds one explicitly selected elementwise arithmetic callback. + result = proc(runtime: Runtime, arguments: openArray[Value]): Value = + ## Executes scalar broadcasting after validating all operand shapes. + let + a = runtime.tensorView(arguments[0]) + b = runtime.tensorView(arguments[1]) + output = runtime.tensorView(arguments[2]) + output.writable() + broadcast(a, output) + broadcast(b, output) + runtime.meter(int64(output.count)) + var bytes = runtime.payload(output) + for i in 0 ..< output.count: + let + ai = if a.count == 1: 0 else: i + bi = if b.count == 1: 0 else: i + if output.dtype == Float32Tensor: + let + av = a.floating(ai) + bv = b.floating(bi) + var value: float32 + case operation + of AddOperation: + value = av + bv + of SubtractOperation: + value = av - bv + of MultiplyOperation: + value = av * bv + of DivideOperation: + if bv == 0: + fail("division by zero") + value = av / bv + bytes.storeFloat(output, i, value) + else: + bytes.store(output, i, + integerOperation(a.number(ai), b.number(bi), output.dtype, operation)) + runtime.putBlob(arguments[2], bytes, "tensor") + arguments[2] + +proc fixedExp(raw: int32): int32 = + ## Approximates exp with integer range reduction and twelve Taylor terms. + if raw < -12 * 65536: + return 0 + if raw > 10 * 65536: + fail("fixed exp result exceeds Q16.16") + # The Q30 residual lies within half of ln(2). + const Ln2 = 744261118'i64 + let + x = int64(raw) * 16384 + exponent = (x + (if x >= 0: Ln2 div 2 else: -Ln2 div 2)) div Ln2 + residual = x - exponent * Ln2 + var + term = 1'i64 shl 30 + sum = term + for i in 1 .. 12: + term = ((term * residual) div (1'i64 shl 30)) div int64(i) + sum += term + if exponent >= 0: + sum = sum shl int(exponent) + else: + sum = sum shr int(-exponent) + let value = (sum + 8192) shr 14 + if value > int64(high(int32)): + fail("fixed exp result exceeds Q16.16") + int32(value) + +proc fixedSigmoid(raw: int32): int32 = + ## Evaluates a stable integer sigmoid without floating point arithmetic. + let magnitude = min(abs(int64(raw)), 12'i64 * 65536) + let exponential = int64(fixedExp(-int32(magnitude))) + let positive = (65536'i64 * 65536) div (65536 + exponential) + int32(if raw >= 0: positive else: 65536 - positive) + +{.push floatChecks: off.} + +proc floatingUnary(value: float32, operation: UnaryOperation): float32 = + ## Keeps nonlinear computation in float32 without fixed scalar conversion. + case operation + of ReluOperation: + max(0.0'f, value) + of AbsOperation: + abs(value) + of ExpOperation: + exp(value) + of SigmoidOperation: + let e = exp(-abs(value)) + if value >= 0: 1.0'f / (1.0'f + e) else: e / (1.0'f + e) + of SigmoidDirectOperation: + 1.0'f / (1.0'f + exp(-value)) + of TanhOperation: + # This formula matches the existing rate network for nonnegative inputs. + let magnitude = abs(value) + let rate = 1.0'f - 2.0'f / (exp(2.0'f * magnitude) + 1.0'f) + if value >= 0: rate else: -rate + +{.pop.} + +proc unaryTensor(operation: UnaryOperation): ContextHostProc = + ## Selects one generic activation or absolute-value kernel. + result = proc(runtime: Runtime, arguments: openArray[Value]): Value = + ## Applies an activation transactionally with exact matching shapes. + let + input = runtime.tensorView(arguments[0]) + output = runtime.tensorView(arguments[1]) + output.writable() + sameShape(input, output) + if input.dtype == Int32Tensor and operation notin {ReluOperation, AbsOperation}: + fail("nonlinear functions require fixed or float32 tensors") + runtime.meter(int64(input.count) * (if operation in {ReluOperation, + AbsOperation}: 1 else: 32)) + var bytes = runtime.payload(output) + for i in 0 ..< input.count: + if input.dtype == Float32Tensor: + bytes.storeFloat(output, i, floatingUnary(input.floating(i), operation)) + else: + let raw = input.number(i) + var value: int32 + case operation + of ReluOperation: + value = max(0'i32, raw) + of AbsOperation: + if raw == low(int32): + fail("absolute value exceeds int32") + value = abs(raw) + of ExpOperation: + value = fixedExp(raw) + of SigmoidOperation, SigmoidDirectOperation: + value = fixedSigmoid(raw) + of TanhOperation: + if raw >= 6 * 65536: + value = 65536 + elif raw <= -6 * 65536: + value = -65536 + else: + value = 2 * fixedSigmoid(raw * 2) - 65536 + bytes.store(output, i, value) + runtime.putBlob(arguments[1], bytes, "tensor") + arguments[1] + +proc sameHandle(a, b: Value): bool = + ## Compares buffer identities without BASIC numeric equality coercion. + a.kind == b.kind and a.bufferOwner == b.bufferOwner and + a.bufferSlot == b.bufferSlot + +{.push overflowChecks: off.} + +proc tensorDense(runtime: Runtime, arguments: openArray[Value]): Value = + ## Computes ordered row-major affine products with optional bias first. + let + input = runtime.tensorView(arguments[0]) + weights = runtime.tensorView(arguments[1]) + output = runtime.tensorView(arguments[3]) + biased = arguments[2].kind == BlobValue + output.writable() + sameType(input, weights) + sameType(input, output) + if input.rank != 1 or weights.rank != 2 or output.rank != 1 or + weights.shape[1] != input.count or weights.shape[0] != output.count: + fail("dense needs vector input, [outputs,inputs] weights and vector output") + if sameHandle(arguments[0], arguments[3]) or + sameHandle(arguments[1], arguments[3]): + fail("dense output must not alias input or weights") + var bias: Tensor + if biased: + bias = runtime.tensorView(arguments[2]) + sameShape(bias, output) + elif arguments[2].kind != IntegerValue or arguments[2].asInt != 0: + fail("dense bias must be a tensor or integer zero") + runtime.meter(int64(weights.count) * 2 + int64(output.count)) + var bytes = runtime.payload(output) + for row in 0 ..< output.count: + if output.dtype == Float32Tensor: + var sum = if biased: bias.floating(row) else: 0.0'f + for i in 0 ..< input.count: + let product = input.floating(i) * weights.floating(row * input.count + i) + sum = sum + product + bytes.storeFloat(output, row, sum) + else: + var sum = if biased: bias.number(row) else: 0'i32 + for i in 0 ..< input.count: + let product = integerOperation(input.number(i), weights.number(row * + input.count + i), output.dtype, MultiplyOperation) + sum = sum +% product + bytes.store(output, row, sum) + runtime.putBlob(arguments[3], bytes, "tensor") + arguments[3] + +{.pop.} + +{.push overflowChecks: off.} + +proc tensorCsr(runtime: Runtime, arguments: openArray[Value]): Value = + ## Computes ordered target-row sparse matrix-vector products. + let + input = runtime.tensorView(arguments[0]) + offsets = runtime.tensorView(arguments[1]) + sources = runtime.tensorView(arguments[2]) + weights = runtime.tensorView(arguments[3]) + output = runtime.tensorView(arguments[4]) + output.writable() + sameType(input, weights) + sameType(input, output) + if input.rank != 1 or output.rank != 1 or offsets.rank != 1 or + sources.rank != 1 or weights.rank != 1 or + offsets.dtype != Int32Tensor or sources.dtype != Int32Tensor or + offsets.count != output.count + 1 or sources.count != weights.count: + fail("invalid CSR tensor shapes or index dtype") + for i in 0 ..< 4: + if sameHandle(arguments[i], arguments[4]): + fail("CSR output must not alias an input") + runtime.meter(int64(weights.count) * 3 + int64(offsets.count)) + var previous = 0'i32 + for i in 0 ..< offsets.count: + let offset = offsets.number(i) + if offset < previous or offset > int32(weights.count): + fail("CSR offsets must be ordered and bounded") + previous = offset + if offsets.number(0) != 0 or previous != weights.count: + fail("CSR offsets must cover all weights") + for i in 0 ..< sources.count: + if sources.number(i) < 0 or sources.number(i) >= input.count: + fail("CSR source index out of bounds") + var bytes = runtime.payload(output) + for row in 0 ..< output.count: + if output.dtype == Float32Tensor: + var sum = 0.0'f + for edge in int(offsets.number(row)) ..< int(offsets.number(row + 1)): + let product = input.floating(int(sources.number(edge))) * + weights.floating(edge) + sum = sum + product + bytes.storeFloat(output, row, sum) + else: + var sum = 0'i32 + for edge in int(offsets.number(row)) ..< int(offsets.number(row + 1)): + let product = integerOperation(input.number(int(sources.number(edge))), + weights.number(edge), output.dtype, MultiplyOperation) + sum = sum +% product + bytes.store(output, row, sum) + runtime.putBlob(arguments[4], bytes, "tensor") + arguments[4] + +{.pop.} + +proc tensorCompare(runtime: Runtime, arguments: openArray[Value]): Value = + ## Produces integer masks from explicitly selected element comparisons. + let + a = runtime.tensorView(arguments[0]) + b = runtime.tensorView(arguments[1]) + operation = runtime.getString(arguments[2]) + output = runtime.tensorView(arguments[3]) + output.writable() + sameType(a, b) + if output.dtype != Int32Tensor: + fail("comparison output must be int32") + if operation notin ["lt", "le", "eq", "ne", "ge", "gt"]: + fail("comparison must be lt, le, eq, ne, ge or gt") + for operand in [a, b]: + if operand.count != 1 and + (operand.rank != output.rank or operand.shape != output.shape): + fail("comparison broadcast shape mismatch") + runtime.meter(int64(output.count)) + var bytes = runtime.payload(output) + for i in 0 ..< output.count: + let + ai = if a.count == 1: 0 else: i + bi = if b.count == 1: 0 else: i + order = + if a.dtype == Float32Tensor: + cmp(a.floating(ai), b.floating(bi)) + else: + cmp(a.number(ai), b.number(bi)) + let matches = + case operation + of "lt": + order < 0 + of "le": + order <= 0 + of "eq": + order == 0 + of "ne": + order != 0 + of "ge": + order >= 0 + else: + order > 0 + bytes.store(output, i, int32(matches)) + runtime.putBlob(arguments[3], bytes, "tensor") + arguments[3] + +proc tensorSelect(runtime: Runtime, arguments: openArray[Value]): Value = + ## Selects values from an integer mask without leaking scalar floats. + let + mask = runtime.tensorView(arguments[0]) + a = runtime.tensorView(arguments[1]) + b = runtime.tensorView(arguments[2]) + output = runtime.tensorView(arguments[3]) + output.writable() + broadcast(a, output) + broadcast(b, output) + if mask.dtype != Int32Tensor or mask.rank != output.rank or mask.shape != output.shape: + fail("selection needs an int32 mask matching the output shape") + runtime.meter(int64(output.count)) + var bytes = runtime.payload(output) + for i in 0 ..< output.count: + let source = if mask.number(i) != 0: a else: b + bytes.store(output, i, source.number(if source.count == 1: 0 else: i)) + runtime.putBlob(arguments[3], bytes, "tensor") + arguments[3] + +proc tensorLerp(runtime: Runtime, arguments: openArray[Value]): Value = + ## Uses a stable two-branch interpolation for generic recurrent updates. + let + a = runtime.tensorView(arguments[0]) + b = runtime.tensorView(arguments[1]) + weight = runtime.tensorView(arguments[2]) + output = runtime.tensorView(arguments[3]) + output.writable() + sameShape(a, output) + sameShape(b, output) + broadcast(weight, output) + runtime.meter(int64(output.count) * 5) + var bytes = runtime.payload(output) + for i in 0 ..< output.count: + let wi = if weight.count == 1: 0 else: i + if output.dtype == Float32Tensor: + let + av = a.floating(i) + bv = b.floating(i) + w = weight.floating(wi) + delta = bv - av + value = if abs(w) < 0.5'f: av + w * delta else: bv - delta * (1.0'f - w) + bytes.storeFloat(output, i, value) + else: + let + av = a.number(i) + bv = b.number(i) + w = weight.number(wi) + delta = bv -% av + one = if output.dtype == FixedTensor: 65536'i32 else: 1'i32 + value = if abs(int64(w)) * 2 < one: + av +% integerOperation(w, delta, output.dtype, MultiplyOperation) + else: + bv -% integerOperation(delta, one -% w, output.dtype, MultiplyOperation) + bytes.store(output, i, value) + runtime.putBlob(arguments[3], bytes, "tensor") + arguments[3] + +proc copySlice(runtime: Runtime, arguments: openArray[Value]): Value = + ## Copies a flat range while preserving every other destination element. + let + input = runtime.tensorView(arguments[0]) + start = int(arguments[1].asInt) + output = runtime.tensorView(arguments[2]) + offset = int(arguments[3].asInt) + count = int(arguments[4].asInt) + output.writable() + sameType(input, output) + if start < 0 or offset < 0 or count < 0 or start > input.count - count or + offset > output.count - count: + fail("slice range out of bounds") + runtime.meter(int64(output.count + count)) + var bytes = runtime.payload(output) + for i in 0 ..< output.count: + bytes.store(output, i, output.number(i)) + for i in 0 ..< count: + bytes.store(output, offset + i, input.number(start + i)) + runtime.putBlob(arguments[2], bytes, "tensor") + arguments[2] + +proc tensorSlice(runtime: Runtime, arguments: openArray[Value]): Value = + ## Copies a flat slice into an exact reusable vector destination. + let output = runtime.tensorView(arguments[3]) + if output.rank != 1 or output.count != int(arguments[2].asInt): + fail("slice destination shape mismatch") + runtime.copySlice([arguments[0], arguments[1], arguments[3], toValue(0), + arguments[2]]) + +proc tensorReshape(runtime: Runtime, arguments: openArray[Value]): Value = + ## Copies values with explicit new dimensions and unchanged dtype. + let + input = runtime.tensorView(arguments[0]) + output = runtime.tensorView(arguments[2]) + shape = dimensions(input.dtype, runtime.shapeValues(arguments[1])) + sameShape(shape, output) + runtime.copyTensor([arguments[0], arguments[2]], false, false) + +proc indexedTensor(scatter: bool): ContextHostProc = + ## Selects gather or ordered scatter-add with checked integer indices. + result = proc(runtime: Runtime, arguments: openArray[Value]): Value = + ## Validates all indices before committing any gathered or scattered value. + let + input = runtime.tensorView(arguments[0]) + indices = runtime.tensorView(arguments[1]) + output = runtime.tensorView(arguments[2]) + output.writable() + sameType(input, output) + if indices.dtype != Int32Tensor or indices.rank != 1 or input.rank != 1 or + output.rank != 1: + fail("gather/scatter requires vectors and int32 indices") + if indices.count != (if scatter: input.count else: output.count): + fail("index count mismatch") + runtime.meter(int64(indices.count * 2 + output.count)) + for i in 0 ..< indices.count: + let index = indices.number(i) + if index < 0 or index >= (if scatter: output.count else: input.count): + fail("gather/scatter index out of bounds") + var bytes = runtime.payload(output) + if scatter: + for i in 0 ..< output.count: + bytes.store(output, i, output.number(i)) + let data = cast[ptr UncheckedArray[byte]](addr bytes[12 + output.rank * 4]) + for i in 0 ..< input.count: + let + index = int(indices.number(i)) + previous = word(data, index) + if output.dtype == Float32Tensor: + bytes.storeFloat(output, index, cast[float32](previous) + + input.floating(i)) + else: + bytes.store(output, index, previous +% input.number(i)) + else: + for i in 0 ..< output.count: + bytes.store(output, i, input.number(int(indices.number(i)))) + runtime.putBlob(arguments[2], bytes, "tensor") + arguments[2] + +proc argmaxTensor(masked: bool): ContextHostProc = + ## Selects the first greatest element and returns minus one for an empty mask. + result = proc(runtime: Runtime, arguments: openArray[Value]): Value = + ## Scans native tensor values without first quantizing float32 logits. + let input = runtime.tensorView(arguments[0]) + var mask: Tensor + if masked: + mask = runtime.tensorView(arguments[1]) + sameCount(input, mask) + if mask.dtype != Int32Tensor: + fail("argmax mask must be int32") + runtime.meter(int64(input.count)) + var best = -1 + for i in 0 ..< input.count: + if masked and mask.number(i) == 0: + continue + if best < 0 or (if input.dtype == Float32Tensor: input.floating(i) > + input.floating(best) else: input.number(i) > input.number(best)): + best = i + toValue(best) + +proc tensorCopy(runtime: Runtime, arguments: openArray[Value]): Value = + ## Copies packed data into explicitly owned mutable storage. + runtime.copyTensor(arguments, false, false) + +proc tensorConvert(runtime: Runtime, arguments: openArray[Value]): Value = + ## Converts numeric values between packed representations. + runtime.copyTensor(arguments, true, false) + +proc tensorReinterpret(runtime: Runtime, arguments: openArray[Value]): Value = + ## Explicitly preserves raw bits for integer logit formats. + runtime.copyTensor(arguments, false, true) + +proc tensorSize(runtime: Runtime, arguments: openArray[Value]): Value = + ## Reads a tensor element count without exposing its bytes. + toValue(runtime.tensorView(arguments[0]).count) + +proc tensorDim(runtime: Runtime, arguments: openArray[Value]): Value = + ## Reads one checked axis size. + let + tensor = runtime.tensorView(arguments[0]) + axis = int(arguments[1].asInt) + if axis < 0 or axis >= tensor.rank: + fail("dimension axis out of bounds") + toValue(tensor.shape[axis]) + +proc tensorDtype(runtime: Runtime, arguments: openArray[Value]): Value = + ## Returns int32=0, fixed=1 or float32=2. + toValue(ord(runtime.tensorView(arguments[0]).dtype)) + +proc addTensorFunctions*(host: var Host, policy: Policy = nil) = + ## Registers architecture-independent tensor math for one policy VM. + let context = TensorContext(policy: policy) + discard host.addFunction("tensorCreate", 2, tensorCreate, 1) + discard host.addFunction("tensorScalar", 2, tensorScalar, 1) + discard host.addFunction("tensorLoad", 1, loadTensor(context), 1) + discard host.addFunction("tensorImport", 2, tensorImport, 1) + discard host.addFunction("tensorExport", 1, tensorExport, 1) + discard host.addFunction("tensorFill", 2, fillTensor, 1) + discard host.addFunction("tensorCopy", 2, tensorCopy, 1) + discard host.addFunction("tensorConvert", 2, tensorConvert, 1) + discard host.addFunction("tensorReinterpret", 2, tensorReinterpret, 1) + for (name, operation) in [("tensorAdd", AddOperation), + ("tensorSubtract", SubtractOperation), ("tensorMultiply", + MultiplyOperation), + ("tensorDivide", DivideOperation)]: + discard host.addFunction(name, 3, binaryTensor(operation), 1) + for (name, operation) in [("tensorRelu", ReluOperation), + ("tensorSigmoid", SigmoidOperation), + ("tensorSigmoidDirect", SigmoidDirectOperation), ("tensorExp", + ExpOperation), + ("tensorTanh", TanhOperation), ("tensorAbs", AbsOperation)]: + discard host.addFunction(name, 2, unaryTensor(operation), 1) + discard host.addFunction("tensorDense", 4, tensorDense, 1) + discard host.addFunction("tensorCsr", 5, tensorCsr, 1) + discard host.addFunction("tensorCompare", 4, tensorCompare, 1) + discard host.addFunction("tensorSelect", 4, tensorSelect, 1) + discard host.addFunction("tensorLerp", 4, tensorLerp, 1) + discard host.addFunction("tensorSlice", 4, tensorSlice, 1) + discard host.addFunction("tensorCopySlice", 5, copySlice, 1) + discard host.addFunction("tensorReshape", 3, tensorReshape, 1) + discard host.addFunction("tensorGather", 3, indexedTensor(false), 1) + discard host.addFunction("tensorScatterAdd", 3, indexedTensor(true), 1) + discard host.addFunction("tensorArgmax", 1, argmaxTensor(false), 1) + discard host.addFunction("tensorArgmaxMasked", 2, argmaxTensor(true), 1) + discard host.addFunction("tensorSize", 1, tensorSize, 1) + discard host.addFunction("tensorDim", 2, tensorDim, 1) + discard host.addFunction("tensorDtype", 1, tensorDtype, 1) diff --git a/tests/all.nim b/tests/all.nim index 9adea5c6..45470bcc 100644 --- a/tests/all.nim +++ b/tests/all.nim @@ -15,6 +15,8 @@ import test_policy_hosts, test_tapes, test_policy_packages, + test_tensors, + test_tensor_matches, test_gota_neural, test_gota_andre, test_gota_fly, diff --git a/tests/bench_tensors.nim b/tests/bench_tensors.nim new file mode 100644 index 00000000..41df839a --- /dev/null +++ b/tests/bench_tensors.nim @@ -0,0 +1,93 @@ +import + std/[monotimes, times], + bassy, + polyworld/tensors, + ../examples/gods_of_the_arena/neural/[common, richard, david, andre, fly], + neuralfixtures, tensorfixtures + +proc measure(name: string, iterations: int, + action: proc() {.closure.}): float = + ## Measures warmed inference separately from compilation and package load. + let start = getMonoTime() + for _ in 0 ..< iterations: + action() + result = float((getMonoTime() - start).inNanoseconds) / + float(iterations) / 1000 + echo name, ": ", result, " us" + +for (author, original, inputs, iterations) in [ + ("richard", richardFixture(), 25, 2000), + ("richard", richardFixture(true), 31, 2000), + ("david", davidFixture(1407, 512), 1407, 100), + ("andre", andreFixture(12, 1), 45, 2000), + ("andre", andreFixture(64, 3), 45, 1000), + ("fly", largeFlyFixture(), 45, 30) +]: + let + package = tensorPolicy(author, original) + source = tensorSource(author, inputs) + label = author & " " & $original.len & " bytes" + var + host = initHost() + limits = defaultLimits() + host.addTensorFunctions(package) + limits.maxNativeMemoryBytes = NativeMemoryBytes + limits.maxInstructions = 100_000 + limits.maxWorkUnits = 250_000 + let start = getMonoTime() + var vm = initRuntime(compile(source, host, limits), host, limits) + for i in 0 ..< inputs: + vm.setArray("data", int32(i), toValue(fixed(1))) + let initial = vm.run() + echo label, " load+compile: ", + float((getMonoTime() - start).inNanoseconds) / 1_000_000, " ms" + echo " native memory: ", vm.nativeMemoryBytes, " bytes" + echo " initial work: ", initial.workUnits + vm.restart() + let warm = vm.run() + echo " inference work: ", warm.workUnits + let basicTime = measure(label & " BASIC", iterations, proc() = + ## Reuses tensors and recurrent state for each measured inference. + vm.restart() + discard vm.run() + ) + var + state: string + data = newSeq[Fixed](inputs) + outputs: seq[Fixed] + for value in data.mitems: + value = fixed(1) + var nativeTime: float + case author + of "richard": + let model = loadRichard(original) + nativeTime = measure(label & " native", iterations, proc() = + ## Measures the existing integer or fixed-point runner. + outputs = inferRichard(model, data) + ) + of "david": + let model = loadDavid(original) + nativeTime = measure(label & " native", iterations, proc() = + ## Measures the existing dense recurrent runner. + let next = inferDavid(model, state, data) + outputs = next.outputs + state = next.state + ) + of "andre": + let model = loadAndre(original) + nativeTime = measure(label & " native", iterations, proc() = + ## Measures the existing stacked recurrent runner. + let next = inferAndre(model, state, data) + outputs = next.outputs + state = next.state + ) + else: + let model = loadFly(original) + nativeTime = measure(label & " native", iterations, proc() = + ## Measures the existing sparse connectome runner. + let next = inferFly(model, state, data) + outputs = next.outputs + state = next.state + ) + doAssert outputs.len > 0 + echo " BASIC/native ratio: ", basicTime / nativeTime diff --git a/tests/tensorfixtures.nim b/tests/tensorfixtures.nim new file mode 100644 index 00000000..16ef2b56 --- /dev/null +++ b/tests/tensorfixtures.nim @@ -0,0 +1,185 @@ +import + std/[os, strutils], + jsony, + polyworld/[policies, tensors], + ../examples/gods_of_the_arena/neural/common + +const TensorLibraries* = currentSourcePath().parentDir.parentDir / + "examples/gods_of_the_arena/neural/tensors" + +type + Entry = object + name, dtype, resource: string + shape: seq[int] + offset: int + Manifest = object + tensors: seq[Entry] + +proc tensorPolicy*(author, model: string): Policy = + ## Splits synthetic model bytes into the documented generic tensor layout. + var + manifest: Manifest + bytes: string + proc add(name, kind: string, shape: seq[int], payload: string) = + ## Adds one named tensor to the synthetic package directory. + manifest.tensors.add Entry(name: name, dtype: kind, shape: shape, + resource: "tensors.bin", offset: bytes.len) + bytes.add payload + proc extract(name, kind: string, shape: seq[int], offset: int) = + ## Extracts a checked row-major tensor from a synthetic author fixture. + var count = 1 + for size in shape: + count *= size + doAssert offset >= 0 and offset + count * 4 <= model.len + add(name, kind, shape, model[offset ..< offset + count * 4]) + var reader = ModelReader(bytes: model, position: 8) + case author + of "richard": + let + residual = reader.readWord() == 1 + inputs = if residual: 31 else: 25 + hidden = if residual: 8 else: 16 + outputs = if residual: 19 else: 18 + kind = if residual: "fixed" else: "int32" + for (name, columns, rows) in [("encoder", inputs, hidden), + ("decoder", hidden, outputs)]: + var bias, weights: string + for _ in 0 ..< rows: + bias.addWord(reader.readWord()) + for _ in 0 ..< columns: + weights.addWord(reader.readWord()) + add("weights." & name, kind, @[rows, columns], weights) + add("weights." & name & "Bias", kind, @[rows], bias) + of "david": + discard reader.readWord() + let + inputs = int(reader.readWord()) + width = int(reader.readWord()) + outputs = int(reader.readWord()) + heads = int(reader.readWord()) + var offset = 160 + heads * 4 + for (name, rows, columns) in [("encoder", width, inputs), + ("recurrent", width * 3, width), ("decoder", outputs, width)]: + extract(name, "float32", @[rows, columns], offset) + offset += rows * columns * 4 + of "andre": + let + width = int(reader.readWord()) + layers = int(reader.readWord()) + decoder = (width * 45 + 7) and not 7 + recurrence = (decoder + width * 12 + 7) and not 7 + stride = (width * width * 3 + 7) and not 7 + extract("encoder", "float32", @[width, 45], 16) + extract("decoder", "float32", @[12, width], 16 + decoder * 4) + for layer in 0 ..< layers: + extract("recurrent." & $layer, "float32", @[width * 3, width], + 16 + (recurrence + stride * layer) * 4) + var count: string + count.addWord(uint32(layers)) + add("layers", "int32", @[1], count) + of "fly": + let + neurons = int(reader.readWord()) + edges = int(reader.readWord()) + inputs = int(reader.readWord()) + driven = int(reader.readWord()) + read = int(reader.readWord()) + steps = reader.readWord() + leak = reader.readWord() + var offset = 36 + for (name, kind, shape) in [ + ("offsets", "int32", @[neurons + 1]), + ("sources", "int32", @[edges]), + ("weights", "float32", @[edges]), + ("bias", "float32", @[neurons]), + ("driveNeurons", "int32", @[driven]), + ("drive", "float32", @[driven, inputs]), + ("readNeurons", "int32", @[read]), + ("readout", "float32", @[12, read]), + ("readoutBias", "float32", @[12])]: + extract(name, kind, shape, offset) + var count = 1 + for size in shape: + count *= size + offset += count * 4 + var stepBytes, leakBytes: string + stepBytes.addWord(steps) + leakBytes.addWord(leak) + add("steps", "int32", @[1], stepBytes) + add("leak", "float32", @[1], leakBytes) + else: + raise newException(TensorError, "Unknown test architecture") + result = Policy(files: @[ + PolicyFile(name: "tensors.json", bytes: manifest.toJson()), + PolicyFile(name: "tensors.bin", bytes: bytes) + ], memoryBytes: int64(bytes.len + manifest.toJson.len + 1024)) + +proc tensorSource*(author: string, inputs: int): string = + ## Wraps a standalone forward pass with reusable input and result globals. + let + prefix = case author + of "richard": + "tr" + of "david": + "td" + of "andre": + "ta" + else: + "tf" + title = author.capitalizeAscii + result = readFile(TensorLibraries / (author & ".bas")) & + "\ndim data(" & $(inputs - 1) & ")\n" & + "if initialized = 0 then\n" + if author == "richard": + result.add " trPrefix$ = \"weights\"\n" + result.add " tensor" & title & "Initialize()\n" & + " initialized = 1\nend if\n" & + prefix & "Data = data\n" & + "tensor" & title & "Step()\nanswer = " & prefix & "Result\n" + +proc mixedWeights*(model: string, start: int, kind: TensorDtype): string = + ## Adds reproducible mixed signs to exercise every dense and recurrent row. + result = model + var seed = 7919'u32 + for offset in countup(start, model.len - 4, 4): + seed = seed * 1664525'u32 + 1013904223'u32 + let raw = + case kind + of Int32Tensor: + cast[uint32](int32(seed mod 17) - 8) + of FixedTensor: + cast[uint32](int32(seed mod 131073) - 65536) + of Float32Tensor: + cast[uint32](float32(int32(seed mod 2001) - 1000) / 32768.0'f) + for i in 0 ..< 4: + result[offset + i] = char((raw shr (i * 8)) and 255) + +proc largeFlyFixture*(): string = + ## Generates the existing benchmark's connectome size without learned data. + const + Neurons = 40_619 + PerNeuron = 26 + Driven = 1627 + Read = 1276 + result = "FLYNN1\0\0" + for value in [Neurons, Neurons * PerNeuron, 45, Driven, Read, 4]: + result.addWord(uint32(value)) + result.addWord(cast[uint32](0.5'f)) + for i in 0 .. Neurons: + result.addWord(uint32(i * PerNeuron)) + for i in 0 ..< Neurons * PerNeuron: + result.addWord(uint32((i * 7919) mod Neurons)) + for i in 0 ..< Neurons * PerNeuron: + result.addWord(cast[uint32](if i mod 5 < 3: 0.05'f else: -0.05'f)) + for _ in 0 ..< Neurons: + result.addWord(cast[uint32](0.01'f)) + for i in 0 ..< Driven: + result.addWord(uint32((i * 23) mod Neurons)) + for _ in 0 ..< Driven * 45: + result.addWord(cast[uint32](0.001'f)) + for i in 0 ..< Read: + result.addWord(uint32((i * 31) mod Neurons)) + for _ in 0 ..< 12 * Read: + result.addWord(cast[uint32](0.001'f)) + for _ in 0 ..< 12: + result.addWord(0) diff --git a/tests/test_neural_converters.py b/tests/test_neural_converters.py index c98dd304..865ecd5f 100644 --- a/tests/test_neural_converters.py +++ b/tests/test_neural_converters.py @@ -12,6 +12,7 @@ import convert_david import convert_richard import convert_andre +import tensor_packages def rejected(call): @@ -41,6 +42,7 @@ def layer(hidden, outputs, fixed=False): source = "bestScore = -2147483647\n" + layer(16, 18) + layer(8, 19, True) converted, resources = convert_richard.convert(source) +richard_source = converted assert converted.count("nn_richard(") == 2 assert "bestScore = -32767.9999847412109375" in converted assert "nnCombatData(nnIndex) = f(nnIndex)" in converted @@ -128,3 +130,20 @@ def write_package(): assert rejected(lambda: convert_andre.convert(source, weights, **options)) print("Synthetic Richard, David and Andre converters passed") + +# Tensor export retains policy decisions and extracts both Richard variants. +tensor_source, tensor_resources = tensor_packages.tensorize(richard_source, resources) +assert "nn_richard(" not in tensor_source +assert "tensorRichardCombatStep()" in tensor_source +assert "tensorRichardResidualStep()" in tensor_source +manifest = json.loads(tensor_resources["tensors.json"]) +assert len(manifest["tensors"]) == 8 +assert len({entry["name"] for entry in manifest["tensors"]}) == 8 +for entry in manifest["tensors"]: + assert entry["resource"] == "tensors.bin" + assert entry["offset"] % 4 == 0 + assert entry["dtype"] in ("int32", "fixed") +assert tensor_source.count("dim tensorShape(0)") == 1 +assert rejected(lambda: tensor_packages.tensorize("end", {"bad.bin": b"bad"})) +assert rejected(lambda: tensor_packages.tensorize("end", {})) +print("Packed tensor export and independent Richard networks passed") diff --git a/tests/test_tensor_matches.nim b/tests/test_tensor_matches.nim new file mode 100644 index 00000000..643d122e --- /dev/null +++ b/tests/test_tensor_matches.nim @@ -0,0 +1,47 @@ +import + std/os, + polyworld/[cli, tapes], + ../examples/gods_of_the_arena/[bots, maps, replays, sim] + +const + Examples = currentSourcePath().parentDir.parentDir / + "examples/gods_of_the_arena/neural/examples" + MatchTicks = 240 + +proc record(path: string): ReplayData = + ## Records ten isolated policy VMs through the production decision pipeline. + let game = newGame(generateMap(2026), 600, 10, false, ReplayData(), + drafting = false) + game.loadBots([BotGroup(path: path, count: 10)]) + game.recorder = initReplayRecorder(game.currentSetup(MatchTicks)) + for _ in 0 ..< MatchTicks: + game.tickWorld(proc() = + ## Issues ordinary commands without replacing simulation behavior. + game.runBotDecisions() + ) + for vm in game.heroVms: + doAssert not vm.failed, vm.lastError + result = decodeReplay(encodeReplay(game.recorder.data)) + doAssert result.actions.len > 0 and result.hashes.len == MatchTicks + +proc verify(data: ReplayData) = + ## Replays recorded commands and requires every simulation hash to match. + let game = newGame( + generateMap(data.config.seed, data.config.mapPreset), + data.config.spawnIntervalTicks, 10, true, data + ) + game.replayPlayer = initReplayPlayer(data) + game.historyPlayback = true + for _ in 0 ..< data.hashes.len: + game.tickWorld(nil) + game.hashCheck.requireReplayComplete(uint32(game.world.tick), data.hashes.len) + doAssert game.replayPlayer.finished + +for author in ["richard", "david", "andre", "fly"]: + let + native = record(Examples / ("synthetic-" & author & ".zip")) + generic = record(Examples / ("tensor-" & author & ".zip")) + doAssert native.actions == generic.actions + doAssert native.hashes == generic.hashes + verify(generic) + echo author, ": ten-player tensor match actions and hashes match" diff --git a/tests/test_tensors.nim b/tests/test_tensors.nim new file mode 100644 index 00000000..b345e16b --- /dev/null +++ b/tests/test_tensors.nim @@ -0,0 +1,411 @@ +import + std/math, + bassy, + polyworld/[policies, tensors], + ../examples/gods_of_the_arena/neural/[common, richard, david, andre, fly], + neuralfixtures, tensorfixtures + +proc rejected(action: proc() {.closure.}): bool = + ## Recognizes controlled tensor and BASIC sandbox failures. + try: + action() + except BasicError: + return true + +proc makeVm(source: string, policy: Policy = nil, + limits = defaultLimits()): Runtime = + ## Creates the same tensor host used by GotA without simulation bindings. + var host = initHost() + host.addBufferFunctions() + host.addTensorFunctions(policy) + initRuntime(compile(source, host, limits), host, limits) + +proc values(runtime: Runtime, name = "answer"): seq[Fixed] = + ## Reads exported final outputs without changing their representations. + let view = runtime.arrayView(runtime.getGlobalValue(name)) + for i in 0 ..< view.len: + result.add view[i].asFixed + +proc stateBytes(runtime: Runtime, name: string): string = + ## Extracts packed state for comparison with an existing native runner. + let bytes = runtime.getBlob(runtime.getGlobalValue(name)) + bytes[12 + ord(bytes[9]) * 4 .. ^1] + +proc stateDifference(actual, expected: string): float = + ## Measures floating differences rather than silently replacing references. + doAssert actual.len == expected.len + var + a = ModelReader(bytes: actual) + b = ModelReader(bytes: expected) + while a.position < actual.len: + let + av = cast[float32](a.readWord()) + bv = cast[float32](b.readWord()) + doAssert av.finite and bv.finite + result = max(result, abs(float(av) - float(bv))) + +echo "Testing tensor arithmetic, broadcasting, slices and float32 boundary" +block: + var vm = makeVm(""" +dim shape(0) +dim data(2) +shape(0) = 3 +a = tensorCreate("float32", shape) +b = tensorCreate("float32", shape) +data(0) = 1.5 +data(1) = -2.0 +data(2) = 0.25 +tensorImport(a, data) +one = tensorScalar("float32", "1") +tensorMultiply(a, one, b) +tensorAdd(b, one, b) +answer = tensorExport(b) +best = tensorArgmax(b) +shape(0) = 1 +small = tensorCreate("float32", shape) +tensorSlice(b, 1, 1, small) +tensorCopySlice(small, 0, b, 2, 1) +copy = tensorExport(b) +precise = tensorScalar("float32", "0.10000000149011612") +""") + discard vm.run() + doAssert vm.values() == @[2.5'fx, -1.0'fx, 1.25'fx] + doAssert vm.values("copy") == @[2.5'fx, -1.0'fx, -1.0'fx] + doAssert vm.getGlobal("best") == 0 + var reader = ModelReader(bytes: vm.stateBytes("precise")) + doAssert reader.readWord() == cast[uint32](0.1'f) + +for residual in [false, true]: + echo "Testing exact Richard BASIC tensor outputs, residual=", residual + let + original = mixedWeights(richardFixture(residual), 12, + if residual: FixedTensor else: Int32Tensor) + model = loadRichard(original) + var vm = makeVm(tensorSource("richard", model.inputs), + tensorPolicy("richard", original)) + for step in 0 ..< 30: + var data = newSeq[Fixed](model.inputs) + for i in 0 ..< data.len: + data[i] = fixed(int32((i * 37 + step * 13) mod 201 - 100)) + vm.setArray("data", int32(i), toValue(data[i])) + vm.restart() + discard vm.run() + doAssert vm.values() == inferRichard(model, data) + +for author in ["david", "andre", "fly"]: + echo "Testing recurrent BASIC tensor states and logits: ", author + for layers in 1 .. (if author == "andre": 3 else: 1): + let + original = case author + of "david": + mixedWeights(davidFixture(45), 180, Float32Tensor) + of "andre": + mixedWeights(andreFixture(12, layers), 16, Float32Tensor) + else: + flyFixture(leak = 0.7'f) + policy = tensorPolicy(author, original) + var + vm = makeVm(tensorSource(author, 45), policy) + previous: string + difference: float + outputDifference: int64 + for step in 0 ..< 30: + var data: array[45, Fixed] + for i in 0 ..< data.len: + data[i] = Fixed(int32((i * 7919 + step * 1123) mod 131072 - 65536)) + vm.setArray("data", int32(i), toValue(data[i])) + vm.restart() + discard vm.run() + let + reference = case author + of "david": + inferDavid(loadDavid(original), previous, data) + of "andre": + inferAndre(loadAndre(original), previous, data) + else: + inferFly(loadFly(original), previous, data) + name = case author + of "david": + "tdState" + of "andre": + "taState" + else: + "tfState" + let actualOutputs = vm.values() + for i in 0 ..< actualOutputs.len: + outputDifference = max(outputDifference, + abs(int64(int32(actualOutputs[i])) - int64(int32(reference.outputs[i])))) + if author == "andre": + var actual: string + for layer in 0 ..< layers: + let handle = vm.getArrayValue("taStates", int32(layer)) + let bytes = vm.getBlob(handle) + actual.add bytes[16 .. ^1] + difference = max(difference, stateDifference(actual, reference.state)) + else: + difference = max(difference, + stateDifference(vm.stateBytes(name), reference.state)) + previous = reference.state + echo " layers=", layers, " maximum FP32 state difference=", difference + echo " maximum exported Q16.16 difference=", outputDifference, " raw units" + doAssert difference < 0.000002 + doAssert outputDifference <= 1 + vm.reset() + previous = "" + discard vm.run() + let reference = case author + of "david": + inferDavid(loadDavid(original), previous, newSeq[Fixed](45)) + of "andre": + inferAndre(loadAndre(original), previous, newSeq[Fixed](45)) + else: + inferFly(loadFly(original), previous, newSeq[Fixed](45)) + doAssert vm.values() == reference.outputs + +echo "Testing failed kernels leave destinations unchanged" +block: + var vm = makeVm(""" +dim shape(0) +shape(0) = 3 +if initialized = 0 then + destination = tensorCreate("fixed", shape) + input = tensorCreate("fixed", shape) + one = tensorScalar("fixed", "1") + zero = tensorScalar("fixed", "0") + tensorFill(destination, one) + initialized = 1 +end if +if bad then + tensorDivide(input, zero, destination) +end if +""") + discard vm.run() + let + handle = vm.getGlobalValue("destination") + before = vm.getBlob(handle) + vm.setGlobal("bad", 1) + vm.restart() + doAssert rejected(proc() = discard vm.run()) + doAssert vm.getBlob(handle) == before + var other = makeVm("x = 0") + doAssert rejected(proc() = discard other.tensorView(handle)) + vm.reset() + doAssert rejected(proc() = discard vm.tensorView(handle)) + +for manifest in ["not json", "{}", "{\"tensors\":[{\"name\":\"w\",\"dtype\":\"float32\",\"shape\":[3],\"resource\":\"x\",\"offset\":0}]}"]: + let policy = Policy(files: @[PolicyFile(name: "tensors.json", + bytes: manifest)]) + var vm = makeVm("w = tensorLoad(\"w\")", policy) + doAssert rejected(proc() = discard vm.run()) + +for value in ["nan", "inf", "bad", "1e100"]: + var vm = makeVm("x = tensorScalar(\"float32\", \"" & value & "\")") + doAssert rejected(proc() = discard vm.run()) + +echo "Testing native buffer limits independently of DIM arrays" +block: + var limits = defaultLimits() + limits.maxArrays = 1 + limits.maxNativeBuffers = 4 + let source = """ +dim shape(0) +shape(0) = 1 +a = tensorCreate("fixed", shape) +b = tensorCreate("fixed", shape) +c = tensorCreate("fixed", shape) +d = tensorCreate("fixed", shape) +""" + var vm = makeVm(source, limits = limits) + discard vm.run() + limits.maxNativeBuffers = 3 + var blocked = makeVm(source, limits = limits) + doAssert rejected(proc() = discard blocked.run()) + + +echo "Testing masks, ordered duplicate scatter, gather, reshape and CSR" +block: + var vm = makeVm(""" +dim shape(0) +dim matrix(1) +dim data(2) +shape(0) = 3 +x = tensorCreate("float32", shape) +y = tensorCreate("float32", shape) +mask = tensorCreate("int32", shape) +ids = tensorCreate("int32", shape) +data(0) = 1 +data(1) = 2 +data(2) = 3 +tensorImport(x, data) +data(0) = 0 +data(1) = 0 +data(2) = 1 +tensorImport(ids, data) +tensorScatterAdd(x, ids, y) +scatter = tensorExport(y) +tensorGather(y, ids, x) +gather = tensorExport(x) +one = tensorScalar("float32", "1") +zero = tensorScalar("float32", "0") +tensorCompare(x, one, "gt", mask) +tensorSelect(mask, x, zero, y) +answer = tensorExport(y) +best = tensorArgmaxMasked(y, mask) +izero = tensorScalar("int32", "0") +tensorFill(mask, izero) +none = tensorArgmaxMasked(y, mask) +matrix(0) = 1 +matrix(1) = 3 +reshaped = tensorCreate("float32", matrix) +tensorReshape(x, matrix, reshaped) +shape(0) = 4 +offsets = tensorCreate("int32", shape) +dim edges(3) +edges(0) = 0 +edges(1) = 2 +edges(2) = 3 +edges(3) = 3 +tensorImport(offsets, edges) +tensorCsr(x, offsets, ids, one, y) +""") + # The deliberately invalid CSR weight count fails after successful indexing. + doAssert rejected(proc() = discard vm.run()) + doAssert vm.values("scatter") == @[3'fx, 3'fx, 0'fx] + doAssert vm.values("gather") == @[3'fx, 3'fx, 3'fx] + doAssert vm.values() == @[3'fx, 3'fx, 3'fx] + doAssert vm.getGlobal("best") == 0 and vm.getGlobal("none") == -1 + let before = vm.getBlob(vm.getGlobalValue("y")) + doAssert before.len == 28 + +echo "Testing empty sparse tensors and invalid shapes" +block: + var vm = makeVm(""" +dim shape(0) +shape(0) = 0 +empty = tensorCreate("float32", shape) +noBest = tensorArgmax(empty) +""") + discard vm.run() + doAssert vm.getGlobal("noBest") == -1 +for source in [ + "dim s(0)\ns(0) = -1\nx = tensorCreate(\"fixed\", s)", + "dim s(0)\ns(0) = 4194305\nx = tensorCreate(\"fixed\", s)", + "x = tensorScalar(\"int32\", \"2147483648\")", + "x = tensorScalar(\"unknown\", \"1\")" +]: + var vm = makeVm(source) + doAssert rejected(proc() = discard vm.run()) + +echo "Testing immutable loaded weights and checked destination indices" +block: + let policy = tensorPolicy("richard", richardFixture()) + var vm = makeVm(""" +w = tensorLoad("weights.encoder") +one = tensorScalar("int32", "1") +tensorFill(w, one) +""", policy) + doAssert rejected(proc() = discard vm.run()) + doAssert vm.tensorView(vm.getGlobalValue("w")).immutable + +for call in [ + "tensorGather(x, indices, y)", + "tensorScatterAdd(x, indices, y)", + "tensorCopySlice(x, -1, y, 0, 1)", + "tensorSlice(x, 0, 2, y)" +]: + var vm = makeVm(""" +dim shape(0) +dim data(0) +shape(0) = 1 +x = tensorCreate("fixed", shape) +y = tensorCreate("fixed", shape) +indices = tensorCreate("int32", shape) +data(0) = 999 +tensorImport(indices, data) +""" & call) + doAssert rejected(proc() = discard vm.run()) + doAssert vm.stateBytes("y") == "\0\0\0\0" + +echo "Testing size-based work and bounded temporary memory" +block: + var limits = defaultLimits() + limits.maxWorkUnits = 20 + var vm = makeVm(""" +dim shape(0) +shape(0) = 10000 +x = tensorCreate("float32", shape) +""", limits = limits) + doAssert rejected(proc() = discard vm.run()) + limits = defaultLimits() + limits.maxNativeMemoryBytes = 200 + var constrained = makeVm(""" +dim shape(0) +shape(0) = 1000 +x = tensorCreate("float32", shape) +""", limits = limits) + doAssert rejected(proc() = discard constrained.run()) + +echo "Testing deterministic fixed nonlinear accuracy" +block: + var vm = makeVm(""" +dim shape(0) +dim data(0) +shape(0) = 1 +if initialized = 0 then + x = tensorCreate("fixed", shape) + y = tensorCreate("fixed", shape) + initialized = 1 +end if +tensorImport(x, data) +tensorSigmoid(x, y) +sigmoid = tensorExport(y) +tensorTanh(x, y) +hyperbolic = tensorExport(y) +tensorExp(x, y) +exponential = tensorExport(y) +""") + var maximumExpError, maximumSigmoidError, maximumTanhError: float + for i in -40 .. 40: + let input = float(i) / 8 + vm.setArray("data", 0, toValue(Fixed(int32(i * 8192)))) + vm.restart() + discard vm.run() + maximumExpError = max(maximumExpError, + abs(float(vm.values("exponential")[0].toFloat32) - exp(input))) + maximumSigmoidError = max(maximumSigmoidError, + abs(float(vm.values("sigmoid")[0].toFloat32) - 1 / (1 + exp(-input)))) + maximumTanhError = max(maximumTanhError, + abs(float(vm.values("hyperbolic")[0].toFloat32) - tanh(input))) + echo " fixed exp/sigmoid/tanh maximum errors=", maximumExpError, + "/", maximumSigmoidError, "/", maximumTanhError + doAssert maximumExpError < 0.0001 + doAssert maximumSigmoidError < 0.00004 + doAssert maximumTanhError < 0.00008 + + +echo "Testing empty CSR rows and nonfinite package weights" +block: + var vm = makeVm(""" +dim shape(0) +shape(0) = 2 +input = tensorCreate("float32", shape) +output = tensorCreate("float32", shape) +shape(0) = 3 +offsets = tensorCreate("int32", shape) +shape(0) = 0 +sources = tensorCreate("int32", shape) +weights = tensorCreate("float32", shape) +tensorCsr(input, offsets, sources, weights, output) +answer = tensorExport(output) +""") + discard vm.run() + doAssert vm.values() == @[0'fx, 0'fx] +block: + let package = tensorPolicy("david", davidFixture(45)) + for file in package.files.mitems: + if file.name == "tensors.bin": + file.bytes[0 .. 3] = "\0\0\x80\x7f" + var vm = makeVm("weight = tensorLoad(\"encoder\")", package) + doAssert rejected(proc() = discard vm.run()) + +echo "Tensor tests passed" From d45a533bccd349def757ca63832862a715c15021 Mon Sep 17 00:00:00 2001 From: treeform Date: Mon, 5 Oct 2026 13:08:15 -0700 Subject: [PATCH 2/2] pin merged Bassy tensor dependency --- coworld/dependencies.lock | 2 +- nimby.lock | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/coworld/dependencies.lock b/coworld/dependencies.lock index 3978472f..adcbb533 100644 --- a/coworld/dependencies.lock +++ b/coworld/dependencies.lock @@ -1,4 +1,4 @@ -bassy 0.1.0 https://github.com/treeform/bassy b1975841adf49320fb3ae28df062500ac5a2ff47 +bassy 0.1.0 https://github.com/treeform/bassy a02e6dd5889310b13d21725c891bde8d01539f08 curly 1.1.1 https://github.com/guzba/curly 29558e2d2b76244a2c65fea08e9f4efff16c1997 libcurl 1.0.0 https://github.com/Araq/libcurl eecbfc31e17ae2f09f0b6d2aaefcfb970c7d2be4 fixxy 0.1.0 https://github.com/treeform/fixxy 05e5446dffb70093056cebb0c57721a60deaf52a diff --git a/nimby.lock b/nimby.lock index 3978472f..adcbb533 100644 --- a/nimby.lock +++ b/nimby.lock @@ -1,4 +1,4 @@ -bassy 0.1.0 https://github.com/treeform/bassy b1975841adf49320fb3ae28df062500ac5a2ff47 +bassy 0.1.0 https://github.com/treeform/bassy a02e6dd5889310b13d21725c891bde8d01539f08 curly 1.1.1 https://github.com/guzba/curly 29558e2d2b76244a2c65fea08e9f4efff16c1997 libcurl 1.0.0 https://github.com/Araq/libcurl eecbfc31e17ae2f09f0b6d2aaefcfb970c7d2be4 fixxy 0.1.0 https://github.com/treeform/fixxy 05e5446dffb70093056cebb0c57721a60deaf52a