From b7957cf94ba9b42ed933b11f24507c28bfbe59c1 Mon Sep 17 00:00:00 2001 From: Martin Muskov <65186527+martin-k-m@users.noreply.github.com> Date: Fri, 21 Aug 2026 15:42:34 +0300 Subject: [PATCH] model: read_unverified was still reading .err off a Res selvedge moved arc.read to Res[Archive, Str]. shuttle kept reading an err field off the result, so every archive load failed at runtime with "cannot access field err of a non-record". twill check was clean and all five suites were green the whole time, because nothing under tests/ ever opened an archive. tests/model_test.tw is that test: it writes an archive with selvedge, loads it back through shuttle, and covers the missing-file case. It fails against the old code. Six suites now, not five, and the README says six. examples/serve.tw runs end to end once loom and selvedge have produced the archive, and the chain is written down. Two more bugs twill check does not catch, both in the example: an array literal with a non-literal element is a list rather than a Tensor, and reshape wants an Arr[I64]. twill has a clock. mono_ns and clock_now_ms both work, and "there is no clock" appeared in this README, docs/needs.md, four source files and the changelog. For the batcher the real blocker was never the clock, it is that there is no concurrency to hold a batch against. Said so. round over tensors exists and is half-away-from-zero, which closes needs entry 14. Entry 16, an import form that expresses a package dependency, is the one entry twill 1.7 did not move: this suite needs selvedge checked out beside it, and the README now says so rather than pretending otherwise. --- .github/workflows/ci.yml | 19 +++-- .gitignore | 5 ++ CHANGELOG.md | 14 ++-- README.md | 171 +++++++++++++++++++++++++++++++-------- docs/needs.md | 136 ++++++++++++++++++++----------- examples/serve.tw | 108 +++++++++++++++++++++++-- spool.toml | 2 +- src/batcher.tw | 10 ++- src/model.tw | 18 +++-- src/score.tw | 10 +-- src/warmup.tw | 5 +- tests/model_test.tw | 57 +++++++++++++ 12 files changed, 444 insertions(+), 111 deletions(-) create mode 100644 tests/model_test.tw diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 7a08bc9..1d59a80 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -158,10 +158,10 @@ jobs: # would have. Pinned rather than floating: a green run should mean this # code passed against a known compiler, not against whatever was newest # that morning. - - name: Install twill v1.7.0 + - name: Install twill v1.7.1 run: | curl -fsSL -o twill \ - https://github.com/twill-lang/twill/releases/download/v1.7.0/twill-v1.7.0-linux-amd64 + https://github.com/twill-lang/twill/releases/download/v1.7.1/twill-v1.7.1-linux-amd64 chmod +x twill ./twill --version @@ -193,6 +193,15 @@ jobs: - name: Run the tests run: ./twill test tests - # No example-run step. examples/serve.tw needs a model file that is not - # in the repository; it refuses cleanly when the file is absent, which is - # correct behaviour and not a smoke test of anything. + # Checking an example is not the same as running it. `twill check` + # follows imports for enum declarations only, and it does not read a + # builtin's argument kinds closely enough either: examples/serve.tw + # passed `[1, 3]` to `reshape`, which is a Tensor and not the Arr[I64] + # that builtin wants, and check was silent about it. + # + # It runs here without a published model. examples/serve.tw generates the + # request files it reads, and falls back to `md.from_params` over an + # untrained tree when models/blobs-1.2.0.slv is absent, saying so on the + # way past. + - name: Run the examples + run: ./twill run examples/serve.tw > /dev/null diff --git a/.gitignore b/.gitignore index ec13fba..ee8d1c2 100644 --- a/.gitignore +++ b/.gitignore @@ -7,3 +7,8 @@ twill_modules/ models/ *.slv scores/ + +# The request fixtures examples/serve.tw generates on first use. Generated +# rather than checked in, because the source gate refuses a file in this +# repository that is not twill or documentation. +examples/*.csv diff --git a/CHANGELOG.md b/CHANGELOG.md index abcb159..9d8c4e7 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,9 +4,12 @@ First cut of shuttle, inference and serving for twill, written in twill. -It does not run. twill's `mode systems` is still being built. See -`docs/needs.md` for what is missing and `README.md` for the status table. -Nothing below has ever executed. +It runs. `twill test tests` passes six suites and +`twill run examples/serve.tw` loads a published model and answers with it, both +on twill 1.7.1. This paragraph said the opposite until `mode systems` landed in +twill 1.6. See `docs/needs.md` for what the language still owes this library and +`README.md` for the status table, which names the test or the example behind +every row. Added: @@ -25,7 +28,7 @@ Added: configurations where a knob does nothing and a `report` that gives the realised trade next to the configured one. - Warmup, at named batch shapes, with a written-out list of what it covers and - what it does not, and a refusal to claim a saving there is no clock to measure. + what it does not, and a refusal to claim a saving nothing here measures. - int8, float16 and bfloat16 with min-max and percentile calibration, a size and accuracy table where every figure is labelled DERIVED, GATED or UNMEASURED, and a `compare` that measures the accuracy cost on the caller's own held-out data. @@ -41,7 +44,8 @@ Deliberately not included in v0.1, and in most cases not possible: - A network server, a port, or anything that listens. twill has no sockets. - Any overlap between waiting and computing. twill has no concurrency. -- A progress time estimate or a warmup saving. twill has no clock. +- A progress time estimate or a warmup saving. twill grew `mono_ns` in 1.7 and + no file here calls it. - A quantisation size win yet. The dtypes round for real, but the bytes drop only once twill NEEDS-111's packed buffer and a narrow archive encoding land. - Activation calibration. It would mean owning the forward pass, which shuttle diff --git a/README.md b/README.md index 84e0b1a..fd51ef5 100644 --- a/README.md +++ b/README.md @@ -24,36 +24,101 @@ `shuttle` is written in twill, in `.tw` files, using `mode systems`. That subset did not exist when this library was written, so for a long time none of the code here executed and this section said so. twill 1.6 is the release that closed it: -the 5 test suites under `tests/` pass, and CI runs them against a released -twill on every push rather than gating on the prose in this file. +the 6 test suites under `tests/` pass, the example loads a published model and +answers with it, and CI runs both against a released twill on every push rather +than gating on the prose in this file. + +You need twill 1.7.0 or newer. Get one: + +```bash +curl -fsSL -o twill https://github.com/twill-lang/twill/releases/download/v1.7.1/twill-v1.7.1-linux-amd64 +chmod +x twill +``` + +The asset name is `twill-v1.7.1--`: `linux-amd64`, `linux-arm64`, +`darwin-amd64`, `darwin-arm64`, `windows-amd64.exe`. + +The suite needs a checkout of [selvedge](https://github.com/twill-lang/selvedge) +beside this one, because `src/model.tw` imports its archive reader by path. +Then, from the repository root: + +``` +$ twill test tests +ok tests/batcher_test.tw +ok tests/model_test.tw +ok tests/predict_test.tw +ok tests/quant_test.tw +ok tests/score_test.tw +ok tests/signature_test.tw + +6 file(s): 6 passed, 0 failed +``` + +And the example. This is the run with the model selvedge publishes already in +place; see the pipeline below for how it gets there: + +``` +$ twill run examples/serve.tw +blobs@1.2.0 (1af561e0982a) f64, input [_, 4] -> [_, 3] logits +warmed 2 pass(es) at batch sizes 1 32 +class 0 with 0.999148 +blobs: shape error: the input axis 0 is 2 but the model expects 4 ([2] against [4]) +batches 3 rows 12 mean 4 configured max_batch 32 max_hold 4 refused 0 +scoring 192/192 (100%) +scored 192 rows +bf16: over 72 rows: mean |diff| 0.010978, max |diff| 0.047055, argmax disagreements 0 +f16: over 72 rows: mean |diff| 0.001167, max |diff| 0.004079, argmax disagreements 0 +int8: over 72 rows: mean |diff| 0.028892, max |diff| 0.160998, argmax disagreements 0 +stored today: 1048 bytes +int8 realised: 195 bytes +``` + +The shape error on the fourth line is deliberate: the example sends a two-element +request to a four-feature model to show what the refusal looks like. + +Without `examples/models/blobs-1.2.0.slv` the example still runs. It falls back +to `md.from_params` over an untrained tree and a signature declared in the file, +which is the weaker of the two load paths and is exactly the one this README +argues against, so it says so on the way past. + +## The pipeline + +Three repositories, one file handed along each edge. Checked out side by side: ```bash -twill test tests +cd loom && twill run examples/classifier.tw +cp loom/examples/runs/blobs.params selvedge/examples/runs/blobs.params +cd selvedge && twill run examples/publish.tw +cp selvedge/examples/models/blobs-1.2.0.slv shuttle/examples/models/blobs-1.2.0.slv +cd shuttle && twill run examples/serve.tw ``` -You need twill 1.7.0 or newer. `docs/needs.md` is still worth reading -- it -is the list of what this library asked the language for, and it now records -which of those arrived and which are still open. +`docs/needs.md` is still worth reading -- it is the list of what this library +asked the language for, and it now records which of those arrived and which are +still open. ## What shuttle is The layer between a trained model and the thing that uses it. loom trains, [selvedge](https://github.com/twill-lang/selvedge) ships, shuttle answers. +Every row below names the test or the example that runs it. A row that names +nothing is a row that claims nothing. + | Piece | State | | --- | --- | -| Load a selvedge archive, integrity verified before the weights are trusted | written, unrun | -| Single, batched and streaming prediction | written, unrun | -| Input validation against the model's declared shapes, with twill's kind of message | written, unrun | -| Dynamic batching with the latency-throughput trade as two required numbers | written, unrun | -| Warmup, with a written-out list of what it does and does not cover | written, unrun | -| int8, float16 and bfloat16 with min-max and percentile calibration | written, unrun | -| Batch scoring over a dataset, chunked, with progress | written, unrun | -| Accuracy evaluation over a labelled dataset | written, unrun | +| Load a selvedge archive, integrity verified before the weights are trusted | runs. `tests/model_test.tw` writes an archive and loads it back; the example loads the published one | +| Single, batched and streaming prediction | runs. `tests/predict_test.tw`, and the example does all three | +| Input validation against the model's declared shapes, with twill's kind of message | runs. `tests/signature_test.tw` asserts the message text and not only that it failed | +| Dynamic batching with the latency-throughput trade as two required numbers | runs. `tests/batcher_test.tw`, and the example drives 12 arrivals through it | +| Warmup, with a written-out list of what it does and does not cover | the example calls it and it reports the shapes it warmed. **No test under `tests/`** | +| int8, float16 and bfloat16 with min-max and percentile calibration | runs. `tests/quant_test.tw`, and the example compares all three on held-out rows | +| Batch scoring over a dataset, chunked, with progress | runs. `tests/score_test.tw`, and the example scores 192 rows | +| Accuracy evaluation over a labelled dataset | runs. `sc.evaluate`, covered by `tests/score_test.tw`. No example uses it | | A quantisation size win | **numerics real; bytes gated on twill NEEDS-111.** See below | -| A progress time estimate | **not possible.** There is no clock | +| A progress time estimate | **not wired.** twill 1.7 has `mono_ns`; shuttle does not call it. See below | | A network server, a port, a socket, a request thread | **not possible, and not planned** | -| Anything running end to end | **no** | +| Anything running end to end | runs. `twill run examples/serve.tw`, output above | ## The forward pass stays yours @@ -157,8 +222,19 @@ not doing anything and nobody would otherwise notice. ### The hold counts arrivals, not milliseconds -Every real dynamic batcher holds a batch for a duration. shuttle cannot: twill -has no clock. `max_hold` counts arrivals instead. +Every real dynamic batcher holds a batch for a duration. shuttle's does not: +`max_hold` counts arrivals instead. + +twill has a clock. `mono_ns()` and `clock_now_ms()` landed in 1.7 and this +README used to say they did not exist. `src/batcher.tw` calls neither, so the +hold is still counted in arrivals and the paragraph below is still what the knob +does. What changed is whose work it is: a duration-bounded hold is now a shuttle +change rather than a language request. It is not a small one. A hold that +expires on time needs something to notice the expiry, and with no concurrency +nothing runs between `submit` calls, so the deadline can only be checked when +the next request arrives, which is the case the arrivals count already covers. +The honest version needs a caller-driven tick, and that is a design decision +this library has not made. `docs/needs.md` entry 3. This is a worse knob and it is the honest one. Counting arrivals bounds the wait in requests and not in time, so under a trickle of traffic a request can wait @@ -201,8 +277,9 @@ request. **The honest summary.** For a dense float model in twill today, warmup buys the lazy work in your own forward function and the page faults. Real, and not large. For an int8 model it buys the dequantisation, which is large. shuttle reports -the shapes it warmed and refuses to claim a saving, because there is no clock to -measure one with. +the shapes it warmed and refuses to claim a saving. It could measure one now: +twill 1.7 has `mono_ns`, and `src/warmup.tw` does not call it. That is a real +gap and it is shuttle's, not twill's. Synthetic input is drawn from the standard normal, not zeros. A relu on a zero input is uniformly on the flat side and every comparison against zero takes one @@ -227,8 +304,10 @@ changes, because the rounding was always the hard half. Until then, `stored_byte with `realised: false` reports the true current footprint and with `true` the footprint after the gate clears; the default is never the aspiration. -Every number below is labelled with where it comes from. Nothing here is a -benchmark result: shuttle runs as of twill 1.6, but no benchmark has been run. +Every number below is labelled with where it comes from. Nothing in the table is +a benchmark result: no timing has been run, and the ratios are arithmetic. The +accuracy figures further down are measured, on one small model, and are labelled +as that rather than as a general result. **Quantising does not make the file smaller today.** twill stores every float as an f64 whatever dtype it carries, so rounding a weight to bfloat16 changes the @@ -272,6 +351,21 @@ f16: over N rows: mean |diff| ..., max |diff| ..., argmax disagreements ... int8: over N rows: mean |diff| ..., max |diff| ..., argmax disagreements ... ``` +`examples/serve.tw` fills that shape in on the model loom trains and selvedge +publishes, over 72 held-out rows: + +``` +bf16: over 72 rows: mean |diff| 0.010978, max |diff| 0.047055, argmax disagreements 0 +f16: over 72 rows: mean |diff| 0.001167, max |diff| 0.004079, argmax disagreements 0 +int8: over 72 rows: mean |diff| 0.028892, max |diff| 0.160998, argmax disagreements 0 +``` + +Those are MEASURED, on one four-tensor MLP over three well separated blobs, and +they are not a result about quantisation. A model with a decision boundary +anywhere near its data would move rows, and this one has none near it: zero +disagreements over 72 rows says the blobs are separable, not that int8 is free. +Run it on your own held-out data, which is what `compare` is for. + The argmax disagreement count is the one that decides whether a quantisation is acceptable, and it is not derivable from the weight error: two models can differ by 1e-6 everywhere and disagree on 3% of rows near the boundary. @@ -317,8 +411,10 @@ else, and there is a test that says so. has no streaming reader, so a file larger than memory cannot be scored. The chunking bounds the activations and not the input. `docs/needs.md` entry 5. -**No time estimate.** There is no clock. The useful part of a progress report on -a four-hour job is the time remaining, and shuttle prints a percentage. +**No time estimate.** shuttle prints a percentage, and the useful part of a +progress report on a four-hour job is the time remaining. twill 1.7 has +`mono_ns` and `src/score.tw` does not call it, so this is shuttle's omission and +no longer a language limit. `docs/needs.md` entry 3. ## Stated limits @@ -327,30 +423,39 @@ Collected, so none of them has to be discovered. - **No network server, no socket, no port, no request thread.** twill has none of the primitives and this is not planned. - **No concurrency.** The batcher accumulates and cuts; it does not overlap. -- **No clock.** The batching hold counts arrivals, warmup cannot report a - saving, and progress has no estimate. +- **Nothing here reads a clock.** The batching hold counts arrivals, warmup + reports no saving, and progress has no estimate. twill 1.7 has `mono_ns` and + `clock_now_ms`; no file in `src/` calls either, so all three of those are + shuttle's work and not the language's. - **Quantisation rounds for real but does not shrink the file yet.** The numerics are exact; the bytes drop once twill NEEDS-111 and a narrow archive encoding land. Until then it measures what shrinking will cost. - **The input to `score_csv` is read whole.** Only the activations are chunked. - **Lit progress lines, but no stateful bar.** The progress line is coloured - from twill's palette, vendored now that the terminal layer is reachable from a - package, and drops to plain text when piped. The rate-and-ETA bar from - `src/cli/progress.tw` is still not adopted; `docs/needs.md` entry 11. + from twill's palette. Nothing is vendored: `std/term/caps`, `std/term/ansi` + and `std/term/theme` are ordinary `std/` modules and `src/score.tw` imports + them directly. It drops to plain text when piped. There is no `std/cli`, so + there is no rate-and-ETA bar to adopt; `docs/needs.md` entry 11. - **`flush` is the caller's job.** Forget it and the last partial batch is never run. ## Install -Once spool and `mode systems` both work: +`mode systems` works. spool does not vendor shuttle for you yet, so until it +does the way in is a clone with selvedge beside it, or: ``` spool add shuttle https://github.com/twill-lang/shuttle ``` -spool vendors into `twill_modules/`, and twill's import is a path, so the import -lines are the long ones in the example above and they resolve relative to the -project root. That is twill's rule rather than shuttle's; see spool's README. +spool vendors into `twill_modules/`, and twill's import is a path, which is why +the import lines above are the long ones. **A path in twill, whether it is an +import or an argument to `read_csv` or `save`, resolves against the directory of +the file that contains it, not against the working directory.** So +`src/model.tw` reaches selvedge as `../../selvedge/src/archive.tw`, and +`twill run examples/serve.tw` reads and writes under `examples/` whatever +directory you invoke it from. That is twill's rule rather than shuttle's; see +spool's README. ## Repository layout diff --git a/docs/needs.md b/docs/needs.md index 4d50cb7..f013cab 100644 --- a/docs/needs.md +++ b/docs/needs.md @@ -1,9 +1,13 @@ # What shuttle needs from twill -shuttle is written in twill and does not run yet. This file is the reason: the -language and runtime features the source uses that twill does not provide today, -with the file and the function that needs each one, and what shuttle does in the -meantime. +shuttle is written in twill and it runs: `twill test tests` passes six suites +and `twill run examples/serve.tw` loads a published model and answers with it. +This file is no longer the reason it does not run. It is the record of what this +library asked the language for, with the file and the function that needed each +one, what shuttle did in the meantime, and, for the ones that have since +arrived, whether shuttle has taken them up. An entry the language delivered and +shuttle has not wired up says exactly that, because "twill cannot" and "shuttle +has not" are different sentences and only one of them is a language work item. It is a work queue for the language, not a complaint. Every entry was reached by writing real code and hitting the wall, which is the only way a list like this @@ -19,14 +23,14 @@ They are restated here with shuttle's call sites rather than cross-referenced, because a work queue that makes you read four repositories to find out what is blocking is a work queue nobody reads. -## Blocking: shuttle cannot run at all without these +## Was blocking: shuttle could not run at all without these ### 1. `mode systems` itself **Used by:** every file -**Status:** designed in `docs/self-hosting.md`, not implemented. +**Status:** DELIVERED in twill 1.6. Closed. -Nothing else on this list matters until this does. +Nothing else on this list mattered until this did. ### 2. A narrow tensor storage: int8, float16 and bfloat16 @@ -66,14 +70,26 @@ true footprint today, true is the footprint once the two pieces above land. **Used by:** `src/batcher.tw` (`max_hold`, which counts arrivals instead), `src/score.tw` (`Progress`, which has no time estimate), `src/warmup.tw` (which cannot report what it saved) -**Status:** not in the language. loom entry 16 and bobbin's first entry are the -same requirement. - -This is the most damaging absence in this repository and it damages the batcher -most. +**Status:** DELIVERED in twill 1.7, and shuttle has not taken it up. `mono_ns()` +returns a monotonic nanosecond count and `clock_now_ms()` a wall-clock +millisecond one; both were checked against the 1.7.1 binary. No file under +`src/` calls either. + +So all three consequences below still hold and none of them is twill's fault any +more. Two of the three are now small changes: `src/warmup.tw` can time its +passes and `src/score.tw` can extrapolate a remaining time, and neither needs +anything from the language. + +The batcher is the one that is still hard, and the reason is entry 13 rather +than this one. A hold that expires after 5 ms needs something to notice the +expiry, and with no concurrency nothing in this library runs between one +`submit` and the next, so a deadline can only ever be checked when the next +request arrives. That is what counting arrivals already does. A duration-bounded +hold needs either a caller-driven tick, which is an API decision shuttle has not +made, or concurrency, which is entry 13. Every real dynamic batcher holds a batch for a duration: "up to 5ms". shuttle -counts arrivals, because there is no clock. The knob is therefore bounded in +counts arrivals. The knob is therefore bounded in requests rather than in time, which means under thin traffic a queued request waits indefinitely for the next arrival and tail latency is unbounded. `flush` exists for that and calling it is the caller's job. The consequence, stated in @@ -88,8 +104,8 @@ anyone wants to read. Third, a progress report on a four-hour scoring job is useful because of the time remaining, and `src/score.tw` prints a percentage. -One primitive fixes all three. It is the same primitive three other repositories -in this ecosystem want. +That primitive exists now. Two of the three are shuttle's to write and the +third needs a design decision first. ### 4. Function values as parameters @@ -99,11 +115,13 @@ systems-mode function take `forward`; `predict_stream` also takes `sink`), `src/batcher.tw` (`run`, `flush`), `src/warmup.tw` (`warm`, `warm_with`), `src/score.tw` (`score`, `score_to`, `evaluate`), `src/quant.tw` (`compare`) -**Status:** functions are values in numeric twill; whether a systems-mode -function may take one, and how the type is spelled, is not stated anywhere. +**Status:** DELIVERED, including the closure. A systems-mode function takes a +function value, the type is spelled `fn(A, B) -> C`, and the closure +`score_to` passes to `predict_stream` captures its environment and runs. +`tests/score_test.tw` and `tests/predict_test.tw` cover both. Closed. Every entry point in this library takes the forward function. shuttle writes -`forward: fn(Tree, Tensor) -> Tensor` and assumes that syntax. +`forward: fn(Tree, Tensor) -> Tensor` and that is the syntax. This is not a convenience, in exactly the way loom's version of this entry is not. The caller owning the forward pass is the design: the thing that makes a @@ -119,8 +137,13 @@ the stronger version, a function value that captures its environment. **Needs:** a reader that yields part of a file, or `read_csv` with a row range **Used by:** `src/score.tw` (`score_csv`) -**Status:** `read_file` returns the whole file; `read_csv` returns the whole -tensor. `std/io` says as much at the top of itself. +**Status:** partly DELIVERED, and not usable for this. twill 1.7.1 has +`read_file_at(path, offset, length)` and `file_size(path)`, so a byte range of a +file is readable now. `read_csv` still returns the whole tensor and there is no +row range, so `score_csv` would have to find its own line boundaries inside a +byte window and parse the rows itself, which is a CSV reader in this repository. +What is wanted is still `read_csv` with a row range, or a reader that yields +rows. `score_csv` chunks the forward pass, which bounds the activations. It does not bound the input, because the whole CSV is in memory before the first chunk runs. @@ -135,10 +158,9 @@ by the word "streaming", which would otherwise be read as more than it is. **Needs:** a spelling for "a tensor, or a list or record nesting tensors" **Used by:** `src/model.tw` (`Model.params`), and every function that takes a forward function, since `Tree` is its first parameter -**Status:** the concept exists at runtime and has no name in the type language. -loom entry 2 and selvedge entry 9 are the same wall. - -Systems mode makes annotations mandatory, so `Model` is currently undeclarable. +**Status:** DELIVERED. `Tree` is the name, `Model.params` is declared with it, +and every entry point that takes a forward function runs. Closed. loom entry 2 +and selvedge entry 9 closed the same way. ### 7. Multiple return values, or `Res[T, E]` @@ -146,14 +168,21 @@ Systems mode makes annotations mandatory, so `Model` is currently undeclarable. `Labels`), `src/batcher.tw` (`Batch`), `src/warmup.tw` (`Warmup`), `src/quant.tw` (`Quantized`, `Comparison`), `src/score.tw` (`Score`, `Evaluation`) -**Status:** AVAILABLE in 1.6, and not yet taken up. +**Status:** DELIVERED in 1.6, and still not taken up. This is the entry that +cost something. 1.6 shipped `Res[T, E]`, `Opt[T]` and postfix `?`, so the language side of this entry is closed. The nine structs are still here, because converting them -changes every public return type in the repository at once and that is a -release of its own rather than a tidy-up. The order to do it in is -`src/model.tw` first, since `Loaded` is the one a caller meets before anything -else works. +changes every public return type in the repository at once and that is a release +of its own rather than a tidy-up. The order to do it in is `src/model.tw` first, +since `Loaded` is the one a caller meets before anything else works. + +selvedge did the conversion and shuttle did not, and the seam between them broke +without anything noticing: `src/model.tw` `read_unverified` went on reading an +`err` field off `arc.read`, which had become a `Res`, so every archive load +failed at runtime while `twill check` and the whole suite stayed green. It was +found by running `examples/serve.tw`. The fix is a `match`, and +`tests/model_test.tw` now loads an archive so the seam is covered. Nine structs in this repository exist to return a value alongside an error string. Not one of them is a type anyone wanted; each is a tuple with a name and @@ -165,7 +194,7 @@ serving library that is worse than it is in a trainer, because the value beside the ignored error is a tensor of zeros and a caller who skips the check gets predictions rather than a crash. -## Blocking: features the source assumes exist +## Was blocking: features the source assumed exist ### 8. Activation calibration, which needs the forward pass to be observable @@ -192,7 +221,9 @@ The entry stays so that the gap is a decision rather than an omission. **Needs:** `count(t)` over a comparison result, or `sum` accepting one **Used by:** `src/quant.tw` (`f64_of_bool`, `compare`), `src/score.tw` (`evaluate`) -**Status:** comparisons give a boolean tensor; `sum` wants numbers. +**Status:** DELIVERED, and taken up. `equal(a, b)` yields a 0/1 tensor that +`sum` adds directly, which is what `q.compare` counts argmax disagreements with; +`tests/quant_test.tw` and the example both exercise it. Closed. shuttle writes `where(t, 1.0, 0.0)` and sums that. Three call sites, one helper, and it allocates a full-size float tensor to count a handful of disagreements. @@ -218,7 +249,15 @@ than start a fourth copy. selvedge's byte-identical copy can go the same way. **Needs:** `src/term/` reachable from a package **Used by:** `src/score.tw` (`Progress`), which now calls it -**Status:** RESOLVED for colour and capability detection. +**Status:** DELIVERED for colour and capability detection, and taken up, but not +in the way recorded below. The terminal layer is under `std/`: +`std/term/caps`, `std/term/ansi` and `std/term/theme`, which `src/score.tw` +imports directly. Nothing is vendored and the import rule did not change. There +is no `std/cli`, so the rate-and-ETA bar this entry wanted does not exist to +adopt; building one here needs the clock from entry 3, which shuttle now has and +does not call. + +The older account, kept because it was wrong and the correction is the point: Resolved the same way as loom's entry 8: twill's terminal modules import each other by a path relative to the importer, so `src/score.tw` vendors the palette @@ -233,8 +272,11 @@ line, now lit from the shared palette so it never drifts in colour. ### 12. A test runner **Would improve:** `tests/` -**Status:** none. `tests/harness.tw` is a hand-rolled counter and `report` calls -`exit(1)`. +**Status:** DELIVERED. `twill test tests` collects `*_test.tw`, runs each in a +fresh interpreter and reports once. CI calls it and so does the README. +`tests/harness.tw` stays, because the runner names the file that failed and the +harness names the assertion inside it; deleting the three copies across three +repositories wants a `std/test`. `tests/harness.tw` is now the fourth identical copy of the same file across four repositories. A `twill test` that collected `*_test.tw`, ran each in a fresh @@ -261,18 +303,18 @@ throughput, and it says so at the top. ### 14. `abs` and `round` over tensors, confirmed **Would improve:** `src/quant.tw` (`simulate_f16`, `simulate_int8`) -**Status:** `abs` is listed as an elementwise builtin; `round` is not in the -table in the README. +**Status:** CONFIRMED on twill 1.7.1. Both exist over tensors, and `round` is +half away from zero: `round([0.5, 1.5, -0.5, 2.4])` gives `[1, 2, -1, 2]`. +Closed. -`src/quant.tw` calls `round` on a tensor in both quantisation paths. If it does -not exist, or exists only for scalars, both are unwritable as they stand and the -fallback is `floor(x + 0.5)`, which is wrong at the halfway point in a way that -biases every weight upward by a fraction of a step. That bias is exactly what -the file's "round, do not truncate" comment argues against, so it would be a -silent regression rather than a compile error. +That is the behaviour `src/quant.tw` wanted. The `floor(x + 0.5)` fallback this +entry feared, which biases every weight upward by a fraction of a step at the +halfway point, is not needed and should not be written. -Confirming it is a documentation fix if the builtin is there and a small -addition if it is not. +What is still not written down is the tie rule itself, in twill's own +documentation. It was established here by running it, and a behaviour a caller +has to discover by experiment is a behaviour that can change without anyone +calling it a break. ### 15. Temporary files, and cleaning up after a test @@ -290,8 +332,10 @@ up after. **Needs:** a way to import a dependency by name **Used by:** `src/model.tw`, which imports selvedge as `../../selvedge/src/...` -**Status:** twill resolves a non-`std/` import as a path relative to the -importing file, and there is no other form. +**Status:** unchanged in 1.7.1. twill resolves a non-`std/` import as a path +relative to the importing file, and there is no other form. This is the one +entry in this file that twill 1.7 did not move at all, and it is the reason the +CI workflow clones selvedge into `../selvedge` before it can run the tests. shuttle depends on selvedge, and the only way to say so in source is a relative path that walks out of shuttle's own tree and into a sibling. That works because diff --git a/examples/serve.tw b/examples/serve.tw index 4f02026..f9c355e 100644 --- a/examples/serve.tw +++ b/examples/serve.tw @@ -7,8 +7,28 @@ mode systems # be built on, driven by a program instead of by a socket, and the README says # so at the top rather than leaving it to be discovered. # -# This does not run. shuttle is written in twill's systems subset, which is -# still being implemented. See docs/needs.md. +# twill run examples/serve.tw +# +# The working directory does not matter: twill resolves an import and a relative +# path argument against the directory of the file that contains it, so +# everything below is read from and written to examples/. +# +# It runs standalone. The three request files it reads are generated on first +# use, under a fixed seed, rather than checked in, because CI's source gate +# refuses any file in this repository that is not twill or documentation. +# +# The model is the one selvedge publishes. From checkouts of loom and selvedge +# beside this one: +# +# cd ../loom && twill run examples/classifier.tw +# cp ../loom/examples/runs/blobs.params ../selvedge/examples/runs/blobs.params +# cd ../selvedge && twill run examples/publish.tw +# cp ../selvedge/examples/models/blobs-1.2.0.slv examples/models/blobs-1.2.0.slv +# +# Without that file this falls back to `md.from_params` over an untrained tree +# and a signature declared right here, which is the second of the two load paths +# src/model.tw documents and the weaker one: the signature becomes a claim +# instead of a fact. That is the point of showing both. import "../src/model.tw" as md import "../src/signature.tw" as sg @@ -25,6 +45,51 @@ fn forward(p: Tree, x: Tensor) -> Tensor { h @ transpose(p.w2) + p.b2 } +# ---- the fixtures ---------------------------------------------------------- + +let FEATURES = 4 +let CLASSES = 3 + +fn blob_rows(per_class: I64, s: I64) -> Tensor { + seed(s) + let centres = [[2.0, 0.0, -1.0, 0.5], [-2.0, 1.0, 0.0, -0.5], [0.0, -2.0, 1.5, 0.0]] + let xs: Arr[Tensor] = [] + let c = 0 + while c < CLASSES { + push(xs, randn(per_class, FEATURES) * 0.8 + centres[c]) + c = c + 1 + } + concat(xs, 0) +} + +fn ensure_csv(path: Str, per_class: I64, s: I64) { + if path_exists(path) { + return unit + } + let x = blob_rows(per_class, s) + let rows = shape(x)[0] + let out = "" + let i = 0 + while i < rows { + let line = "" + let j = 0 + while j < FEATURES { + if j > 0 { line = line + "," } + line = line + str(item(x[i][j])) + j = j + 1 + } + out = out + line + " +" + i = i + 1 + } + write_file(path, out) +} + +mkdir_all("models") +ensure_csv("requests.csv", 4, 20260811) +ensure_csv("batch.csv", 64, 20260810) +ensure_csv("holdout.csv", 24, 20260809) + # ---- load ------------------------------------------------------------------ # # From a selvedge archive, so the signature is what the model was published @@ -32,7 +97,29 @@ fn forward(p: Tree, x: Tensor) -> Tensor { # weights are trusted: a serving process is the last place anyone will look at # the file, and a corrupt payload that reaches a forward pass produces # predictions rather than an error. -let loaded = md.from_archive("models/blobs-1.2.0.slv") +fn untrained_tree() -> Str { + seed(20260807) + save({ + w1: randn(16, FEATURES) * 0.1, + b1: zeros(16), + w2: randn(CLASSES, 16) * 0.1, + b2: zeros(CLASSES), + }, "models/untrained.params") + "models/untrained.params" +} + +fn load_model() -> md.Loaded { + if path_exists("models/blobs-1.2.0.slv") { + return md.from_archive("models/blobs-1.2.0.slv") + } + print("models/blobs-1.2.0.slv is absent, so this serves an untrained tree") + print("through from_params. See the header of this file for the three") + print("commands that produce the archive.") + md.from_params(untrained_tree(), + sg.signature([-1, FEATURES], [-1, CLASSES], "logits"), "blobs-untrained") +} + +let loaded = load_model() if len(loaded.err) > 0 { print("shuttle: " + loaded.err) exit(1) @@ -49,9 +136,10 @@ print(md.describe(model)) # max_batch 32 throughput. Up to 32 rows in one forward pass. # max_hold 4 latency. A queued request waits through at most 4 arrivals. # -# max_hold counts arrivals and not milliseconds, because twill has no clock. -# That is a worse knob than a duration and it is the honest one; see the top of -# src/batcher.tw for what it means under thin traffic. +# max_hold counts arrivals and not milliseconds. twill has a clock; shuttle has +# no concurrency, which is the actual reason. That is a worse knob than a +# duration and it is the honest one; see the top of src/batcher.tw for what it +# means under thin traffic. let cfg = bat.config(32, 4) let cfg_err = bat.validate(cfg) if len(cfg_err) > 0 { @@ -70,11 +158,15 @@ print(wu.report(w)) # ---- one request ----------------------------------------------------------- -let one = pr.predict(model, [1.2, 0.4, 0.0 - 0.8, 2.1], forward) +let one = pr.predict(model, [1.2, 0.4, -0.8, 2.1], forward) if len(one.err) > 0 { print("shuttle: " + one.err) } else { - let probs = pr.probabilities(model, reshape(one.out, [1, 3])) + # `[1, CLASSES]` and not `[1, 3]`: an array literal whose elements are all + # numeric literals is a Tensor, and `reshape` wants the dimensions as an + # Arr[I64]. One name in the list is enough to make it one. + let one_row: Arr[I64] = [1, CLASSES] + let probs = pr.probabilities(model, reshape(one.out, one_row)) print("class " + str(item(argmax(one.out, 0))) + " with " + str(item(max(probs.idx)))) } diff --git a/spool.toml b/spool.toml index 921be87..e8637c0 100644 --- a/spool.toml +++ b/spool.toml @@ -11,7 +11,7 @@ entry = "src/predict.tw" # The language itself, pinned to the release whose std modules and tensor # builtins this source was written against. spool has no notion of a toolchain # dependency, so it is declared as an ordinary git dependency. -twill = { version = "^1.5.0", git = "https://github.com/twill-lang/twill" } +twill = { version = "^1.7.0", git = "https://github.com/twill-lang/twill" } # Reading model archives. A real dependency and not a convenience: loading # through selvedge gets the model's declared input and output shapes out of the diff --git a/src/batcher.tw b/src/batcher.tw index 9319320..f9b3a90 100644 --- a/src/batcher.tw +++ b/src/batcher.tw @@ -30,7 +30,15 @@ mode systems # ---- what a hold is measured in, and why it is not milliseconds ----------- # # Every real dynamic batcher holds a batch for a duration: "up to 5ms". shuttle -# cannot, because twill has no clock. `max_hold` counts arrivals instead. +# does not: `max_hold` counts arrivals instead. +# +# Not because twill has no clock. It has `mono_ns` and `clock_now_ms` as of +# 1.7, and this comment said otherwise for a release. The reason is that there +# is no concurrency, so nothing in this file runs between one `submit` and the +# next, and a deadline can only be noticed when the next request arrives, which +# is what an arrivals count already measures. A duration-bounded hold needs a +# caller-driven tick, which is a decision shuttle has not made. +# docs/needs.md entry 3. # # This is a worse knob and it is the honest one. Counting arrivals bounds the # wait in requests and not in time, so under a trickle of traffic a request can diff --git a/src/model.tw b/src/model.tw index edf58c1..6c86586 100644 --- a/src/model.tw +++ b/src/model.tw @@ -98,15 +98,23 @@ fn from_archive(path: Str) -> Loaded { read_unverified(path) } +# `arc.read` returns `Res[Archive, Str]`. It used to return a record with an +# `err` field, and this function read that field for a release after selvedge +# moved, which made every archive load fail at runtime with "cannot access field +# err of a non-record". Nothing caught it because nothing under tests/ loaded an +# archive; tests/model_test.tw does now. fn read_unverified(path: Str) -> Loaded { - let got = arc.read(path) - if len(got.err) > 0 { - return failed(got.err) + match arc.read(path) { + Ok(a) => from_read(a), + Err(e) => failed(e), } - let man = got.arc.man +} + +fn from_read(a: arc.Archive) -> Loaded { + let man = a.man Loaded { model: Model { - params: got.arc.params, + params: a.params, sig: sg.signature(man.sig.input, man.sig.output, man.sig.output_kind), name: man.name, version: mf.ref_str(man), diff --git a/src/score.tw b/src/score.tw index 6ffaa96..0dbc7bc 100644 --- a/src/score.tw +++ b/src/score.tw @@ -16,11 +16,11 @@ mode systems # ---- progress -------------------------------------------------------------- # # Rows done, rows total, percent. No time estimate, and that is a gap rather -# than a decision: twill has no clock, so there is no elapsed time to -# extrapolate from. The useful part of a progress report on a four-hour job is -# the time remaining, so this is worse than what a progress bar should be, and -# it is worse in exactly the way loom's is, for the same missing primitive. -# docs/needs.md entry 3. +# than a decision. twill has `mono_ns` as of 1.7 and this file does not call +# it, so the elapsed time is there for the taking and nobody has taken it. The +# useful part of a progress report on a four-hour job is the time remaining, so +# this is worse than what a progress bar should be, and it is worse in exactly +# the way loom's is, for the same unwritten code. docs/needs.md entry 3. # # Progress goes to stdout, and it is lit now. twill's terminal layer became # reachable from a package once its modules began importing each other by a path diff --git a/src/warmup.tw b/src/warmup.tw index 1f655d9..ef04c8e 100644 --- a/src/warmup.tw +++ b/src/warmup.tw @@ -65,8 +65,9 @@ mode systems # own forward function and the page faults. That is a real number and it is not # a large one. For an int8 model it buys the dequantisation, which is large. # Neither is guessed at here: `Warmup.passes` and the shapes it covered are -# reported, and the timing is yours to measure because twill has no clock -# (docs/needs.md entry 3). +# reported, and the timing is yours to measure because this file does not time +# itself. twill has `mono_ns` as of 1.7 and nothing here calls it, so that is +# an omission rather than a limit (docs/needs.md entry 3). import "signature.tw" as sg import "model.tw" as md diff --git a/tests/model_test.tw b/tests/model_test.tw new file mode 100644 index 0000000..93e4de6 --- /dev/null +++ b/tests/model_test.tw @@ -0,0 +1,57 @@ +mode systems + +# Loading a model, through both doors. +# +# This suite exists because of a defect it would have caught. selvedge's +# `arc.read` moved to `Res[Archive, Str]` and `src/model.tw` went on reading an +# `err` field off it, so every archive load failed at runtime with "cannot +# access field err of a non-record" while `twill check` and the other five +# suites stayed green. Nothing under tests/ opened an archive. +# +# So this one writes a real archive with selvedge, in this directory, and reads +# it back through shuttle. It is the only test in the repository that touches +# the filesystem, and it cleans up after itself. + +import "harness.tw" as t +import "../src/model.tw" as md +import "../src/signature.tw" as sg +import "../../selvedge/src/archive.tw" as arc +import "../../selvedge/src/manifest.tw" as mf +import "../../selvedge/src/version.tw" as ver + +let PATH = "tmp_model.slv" + +fn a_tree() -> Tree { + seed(11) + { w1: randn(2, 3), b1: zeros(2) } +} + +fn write_archive() -> Str { + let sig = mf.signature([-1, 3], [-1, 2], "logits") + let m = mf.manifest("toy", ver.parse("1.2.0"), "mlp", sig) + m.architecture_detail = "dense 3->2" + arc.write(arc.promote(a_tree(), m), PATH) +} + +fn an_archive_loads_and_carries_its_own_signature() { + t.equal_str("the archive is written", write_archive(), "") + let loaded = md.from_archive(PATH) + t.equal_str("the load reports no error", loaded.err, "") + t.equal_str("the model is named by the archive", loaded.model.name, "toy") + t.equal_str("and versioned by it", loaded.model.version, "toy@1.2.0") + t.check("the digest came out of the file", len(loaded.model.digest) == 64) + t.check("the signature came out of the file too", + sg.equal(loaded.model.sig, sg.rows_of(3, 2, "logits"))) + t.check("and the parameters are the tree that was promoted", + loaded.model.params == a_tree()) + remove_file(PATH) +} + +fn a_missing_archive_is_refused_by_name_rather_than_crashing() { + let loaded = md.from_archive("tmp_absent.slv") + t.check("the load reports an error", len(loaded.err) > 0) +} + +an_archive_loads_and_carries_its_own_signature() +a_missing_archive_is_refused_by_name_rather_than_crashing() +t.report("model")