diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 0c79492..df42071 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -158,10 +158,10 @@ jobs: # would have. Pinned rather than floating: a green run should mean this # code passed against a known compiler, not against whatever was newest # that morning. - - name: Install twill v1.9.0 + - name: Install twill v1.12.0 run: | curl -fsSL -o twill \ - https://github.com/twill-lang/twill/releases/download/v1.9.0/twill-v1.9.0-linux-amd64 + https://github.com/twill-lang/twill/releases/download/v1.12.0/twill-v1.12.0-linux-amd64 chmod +x twill ./twill --version diff --git a/CHANGELOG.md b/CHANGELOG.md index 76358df..6e609f3 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,15 @@ ## Unreleased +### Changed + +- **The suites are written with `std/test`.** `tests/harness.tw` is deleted. + It was the fourth copy of the same harness across four repositories, and + `docs/needs.md` entry 12 said a `std/test` was what would delete it. twill + 1.11 shipped one, so every `*_test.tw` imports `"std/test"` and calls the + same four assertions by the same names. The seven suites pass on twill + 1.12.0, and the pin moved from 1.9.0 to 1.12.0 in `spool.toml` and CI. + ### Added - **Warmup times itself.** Every pass is measured with `mono_ns`, and the report diff --git a/README.md b/README.md index d669a82..36ac996 100644 --- a/README.md +++ b/README.md @@ -28,14 +28,15 @@ the 6 test suites under `tests/` pass, the example loads a published model and answers with it, and CI runs both against a released twill on every push rather than gating on the prose in this file. -You need twill 1.9.0 or newer. Get one: +You need twill 1.11.0 or newer, because the suites are written with `std/test`, +which arrived in 1.11. Get one: ```bash -curl -fsSL -o twill https://github.com/twill-lang/twill/releases/download/v1.9.0/twill-v1.9.0-linux-amd64 +curl -fsSL -o twill https://github.com/twill-lang/twill/releases/download/v1.12.0/twill-v1.12.0-linux-amd64 chmod +x twill ``` -The asset name is `twill-v1.9.0--`: `linux-amd64`, `linux-arm64`, +The asset name is `twill-v1.12.0--`: `linux-amd64`, `linux-arm64`, `darwin-amd64`, `darwin-arm64`, `windows-amd64.exe`. The suite needs a checkout of [selvedge](https://github.com/twill-lang/selvedge) diff --git a/docs/needs.md b/docs/needs.md index b7d2007..84ecfa6 100644 --- a/docs/needs.md +++ b/docs/needs.md @@ -283,15 +283,20 @@ line, now lit from the shared palette so it never drifts in colour. ### 12. A test runner **Would improve:** `tests/` -**Status:** DELIVERED. `twill test tests` collects `*_test.tw`, runs each in a -fresh interpreter and reports once. CI calls it and so does the README. -`tests/harness.tw` stays, because the runner names the file that failed and the -harness names the assertion inside it; deleting the three copies across three -repositories wants a `std/test`. - -`tests/harness.tw` is now the fourth identical copy of the same file across four -repositories. A `twill test` that collected `*_test.tw`, ran each in a fresh -interpreter and reported once would delete all four. +**Status:** DELIVERED, both halves. `twill test tests` collects `*_test.tw`, +runs each in a fresh interpreter and reports once. CI calls it and so does the +README. twill 1.11 then shipped `std/test`, which is what this entry said would +delete the harness copies, and `tests/harness.tw` is deleted: every suite +imports `"std/test"` and calls `check`, `equal_str`, `equal_i64` and `near` by +the same names with the same signatures, so no assertion changed. `near` still +takes the tolerance and never defaults it. The summary is now printed in the +shape the runner reads, ` passed

failed ` and then `OK` or +`FAILED`, which the copy never was. + +What it said while the copy was here: `tests/harness.tw` stays, because the +runner names the file that failed and the harness names the assertion inside +it; deleting the three copies across three repositories wants a `std/test`. It +was the fourth identical copy of the same file across four repositories. ### 13. Concurrency, and a process interface diff --git a/spool.toml b/spool.toml index f08a57c..bb2f3d4 100644 --- a/spool.toml +++ b/spool.toml @@ -11,7 +11,7 @@ entry = "src/predict.tw" # The language itself, pinned to the release whose std modules and tensor # builtins this source was written against. spool has no notion of a toolchain # dependency, so it is declared as an ordinary git dependency. -twill = { version = "^1.9.0", git = "https://github.com/twill-lang/twill" } +twill = { version = "^1.12.0", git = "https://github.com/twill-lang/twill" } # Reading model archives. A real dependency and not a convenience: loading # through selvedge gets the model's declared input and output shapes out of the diff --git a/tests/batcher_test.tw b/tests/batcher_test.tw index e902383..1221415 100644 --- a/tests/batcher_test.tw +++ b/tests/batcher_test.tw @@ -1,6 +1,6 @@ mode systems -import "harness.tw" as t +import "std/test" as t import "../src/signature.tw" as sg import "../src/model.tw" as md import "../src/batcher.tw" as bat diff --git a/tests/harness.tw b/tests/harness.tw deleted file mode 100644 index aa5e960..0000000 --- a/tests/harness.tw +++ /dev/null @@ -1,71 +0,0 @@ -mode systems - -# A test harness, in twill. -# -# There is no test runner in the toolchain yet, so a test file is a program: it -# calls these and `report` exits non-zero if anything failed. Primitive, and -# enough to be a CI gate the day `mode systems` runs. -# -# This mirrors loom's and spool's tests/harness.tw deliberately, line for line. -# Two conventions for the same job across one ecosystem is one too many. It is -# also the third copy of this file, and docs/needs.md asks for the test runner -# that would delete all three. -# -# Test names are written as sentences, because a failing test's name is the only -# documentation anyone reads at the moment it fails. - -struct Counter { - passed: I64, - failed: I64, -} - -let TESTS = Counter { passed: 0, failed: 0 } - -fn check(name: Str, ok: Bool) { - if ok { - TESTS.passed = TESTS.passed + 1 - } else { - TESTS.failed = TESTS.failed + 1 - print("FAIL " + name) - } -} - -fn equal_str(name: Str, got: Str, want: Str) { - if got == want { - TESTS.passed = TESTS.passed + 1 - } else { - TESTS.failed = TESTS.failed + 1 - print("FAIL " + name) - print(" got: " + got) - print(" want: " + want) - } -} - -fn equal_i64(name: Str, got: I64, want: I64) { - equal_str(name, str(got), str(want)) -} - -# Floats are compared with a tolerance, and the tolerance is an argument rather -# than a constant in here. A metric test wants 1e-9 and a cosine schedule test -# wants 1e-6, and a single shared epsilon would be wrong for one of them. -fn near(name: Str, got: F64, want: F64, tol: F64) { - let d = got - want - if d < 0.0 { - d = -d - } - if d <= tol { - TESTS.passed = TESTS.passed + 1 - } else { - TESTS.failed = TESTS.failed + 1 - print("FAIL " + name) - print(" got: " + str(got)) - print(" want: " + str(want) + " within " + str(tol)) - } -} - -fn report(suite: Str) { - print(suite + ": " + str(TESTS.passed) + " passed, " + str(TESTS.failed) + " failed") - if TESTS.failed > 0 { - exit(1) - } -} diff --git a/tests/model_test.tw b/tests/model_test.tw index 93e4de6..bbab0e8 100644 --- a/tests/model_test.tw +++ b/tests/model_test.tw @@ -12,7 +12,7 @@ mode systems # it back through shuttle. It is the only test in the repository that touches # the filesystem, and it cleans up after itself. -import "harness.tw" as t +import "std/test" as t import "../src/model.tw" as md import "../src/signature.tw" as sg import "../../selvedge/src/archive.tw" as arc diff --git a/tests/predict_test.tw b/tests/predict_test.tw index 0259f06..a150b68 100644 --- a/tests/predict_test.tw +++ b/tests/predict_test.tw @@ -1,6 +1,6 @@ mode systems -import "harness.tw" as t +import "std/test" as t import "../src/signature.tw" as sg import "../src/model.tw" as md import "../src/predict.tw" as pr diff --git a/tests/quant_test.tw b/tests/quant_test.tw index d1b7a66..463a483 100644 --- a/tests/quant_test.tw +++ b/tests/quant_test.tw @@ -9,7 +9,7 @@ mode systems # unbiased, that the sizes are what the projection says, and that a saturating # calibration says how much it saturated. -import "harness.tw" as t +import "std/test" as t import "../src/signature.tw" as sg import "../src/model.tw" as md import "../src/quant.tw" as q diff --git a/tests/score_test.tw b/tests/score_test.tw index df39aab..5c33ab2 100644 --- a/tests/score_test.tw +++ b/tests/score_test.tw @@ -1,6 +1,6 @@ mode systems -import "harness.tw" as t +import "std/test" as t import "../src/signature.tw" as sg import "../src/model.tw" as md import "../src/score.tw" as sc diff --git a/tests/signature_test.tw b/tests/signature_test.tw index 51f6995..7ba4e4a 100644 --- a/tests/signature_test.tw +++ b/tests/signature_test.tw @@ -7,7 +7,7 @@ mode systems # debug, which is the situation twill's checker exists to prevent and which # serving otherwise reintroduces. -import "harness.tw" as t +import "std/test" as t import "../src/signature.tw" as sg import "std/text" as tx diff --git a/tests/warmup_test.tw b/tests/warmup_test.tw index 10d492e..449fa80 100644 --- a/tests/warmup_test.tw +++ b/tests/warmup_test.tw @@ -9,7 +9,7 @@ mode systems # saving is only claimed when there is enough evidence for one, that a first # pass no slower than the rest claims nothing, and that every pass is timed. -import "harness.tw" as t +import "std/test" as t import "std/text" as tx import "../src/signature.tw" as sg import "../src/model.tw" as md