From aa1c2e787b2dc4ca7a0aeeed61d7682fcd7144c6 Mon Sep 17 00:00:00 2001 From: Derek Date: Thu, 3 Sep 2026 07:22:17 +1000 Subject: [PATCH] fix(developer-rust): bound the build pool against free space, not just the clock The pool ceiling only binds while the prune is running, and the prune ran weekly. A box that grows the pool faster than that spends the gap over its ceiling, which is how dragonfly reached 89% with 289G pooled against a 115G ceiling. The README described the overshoot as deliberate and asked whoever owned the box to watch free space themselves. Nobody did. So automate that check rather than only shortening the schedule: - hyperi-rust-cache-prune takes --if-free-below SIZE|PERCENT. Above the floor it exits after one statvfs, before walking the pool and before prompting, so it can be scheduled hourly and still cost nothing almost every time. - The full prune moves from weekly to daily. Setting rust_cache_prune_schedule_weekday puts it back. - An hourly guard runs the same prune to the same ceiling once free space drops below rust_cache_prune_free_floor, 20% by default. systemd timer on Linux, a launchd agent on StartInterval for macOS. - A guard run that ends inside the ceiling says so instead of cutting deeper. The space went somewhere the prune does not own, and naming that beats evicting artefacts that were not the cause. Also fixes the accounting the guard's verdict rests on. remove() warned and returned on a failed rmtree while both callers subtracted the size anyway, so a pool whose evictions all failed reported "under the ceiling at 0B" with every byte still on disk. remove() now reports whether the bytes went, and prune_pool returns what the pool still holds. The macOS load sequence was five tasks and a second agent would have duplicated all of it, so it moves to tasks/launchd_agent.yml, included once per agent to keep each one's register scope. Verified on dragonfly: both timers live, the guard reports 371.3G free of 691.5G against a 138.3G floor and returns in under a second, and the role's second pass in the same run is all ok. 65 tests pass. --- ansible/roles/developer-rust/README.md | 22 +-- .../roles/developer-rust/defaults/main.yml | 19 ++- .../files/hyperi-rust-cache-prune | 126 +++++++++++++++-- ansible/roles/developer-rust/tasks/cache.yml | 130 ++++++++++++------ .../developer-rust/tasks/launchd_agent.yml | 60 ++++++++ .../hyperi-rust-cache-prune-guard.service.j2 | 13 ++ .../hyperi-rust-cache-prune-guard.timer.j2 | 12 ++ .../hyperi-rust-cache-prune.timer.j2 | 4 +- .../io.hyperi.rust-cache-prune-guard.plist.j2 | 43 ++++++ .../io.hyperi.rust-cache-prune.plist.j2 | 6 +- tools/tests/test_rust_cache_prune.py | 110 +++++++++++++++ 11 files changed, 476 insertions(+), 69 deletions(-) create mode 100644 ansible/roles/developer-rust/tasks/launchd_agent.yml create mode 100644 ansible/roles/developer-rust/templates/hyperi-rust-cache-prune-guard.service.j2 create mode 100644 ansible/roles/developer-rust/templates/hyperi-rust-cache-prune-guard.timer.j2 create mode 100644 ansible/roles/developer-rust/templates/io.hyperi.rust-cache-prune-guard.plist.j2 diff --git a/ansible/roles/developer-rust/README.md b/ansible/roles/developer-rust/README.md index fdefde7..48673f7 100644 --- a/ansible/roles/developer-rust/README.md +++ b/ansible/roles/developer-rust/README.md @@ -70,6 +70,8 @@ silently stops being cached while `sccache` is still installed and configured. `hyperi-rust-cache-prune` reports it rather than showing an empty ceiling, and `sccache --stop-server` clears it. +## Bounding the build-artefact pool + **Build artefacts have no upstream cap at all.** Cargo does not track them (rust-lang/cargo#13136), so nothing reclaims a `target/` ever, and one per repo across a tree full of them is what actually fills a disk. @@ -82,7 +84,7 @@ binaries, so `cargo run`, IDEs and anything globbing for a built artefact are unaffected. `hyperi-rust-cache-prune` then bounds that pool on a schedule -- a systemd timer -on Linux, a launchd agent on macOS, weekly and at idle IO priority. It drops +on Linux, a launchd agent on macOS, daily and at idle IO priority. It drops workspaces not built for `rust_cache_max_age_days`, then evicts least-recently-built ones until the pool is under `rust_cache_build_dir_max`. It touches no project `target/`, and reports the self-capping caches without @@ -94,14 +96,18 @@ derived from it chases itself downward. That puts a 692G build box at 115G and a 256G laptop at the floor, so one default suits both. Set `rust_cache_build_dir_max` to an explicit size to override it. -**The ceiling is a weekly reset, not a live cap.** Nothing enforces it between -runs, so a busy build box spends most of the week above it -- one added 37G in -two days and another 37G in a single afternoon. That is the design working, not -a prune that failed. It only matters if the peak, not the floor, would fill the -disk; check free space before reaching for a shorter schedule, and change -`rust_cache_prune_schedule_weekday` if the peak is genuinely too high. +**The ceiling binds while the tool runs, not between runs.** A pool that grows +faster than the schedule spends the gap above it, so the prune runs daily and an +hourly guard backs it up -- one statvfs while the disk has room, a prune to the +same ceiling once free space falls below `rust_cache_prune_free_floor` (20%). + +A guard run that finds the pool already under its ceiling stops and says so. The +space went somewhere the prune does not own, and naming that is more use than +evicting artefacts that were not the cause. Set +`rust_cache_prune_guard_enabled: false` to drop the guard, or +`rust_cache_prune_schedule_weekday` to go back to weekly. -sccache and ccache keep fixed ceilings; both enforce their own. +## Which sccache builds actually use **sccache comes from upstream's release, not the distro.** Ubuntu ships 0.13.0 against an upstream on 0.17.x, and a wrapper four versions behind is what a diff --git a/ansible/roles/developer-rust/defaults/main.yml b/ansible/roles/developer-rust/defaults/main.yml index fe73f33..2772c0f 100644 --- a/ansible/roles/developer-rust/defaults/main.yml +++ b/ansible/roles/developer-rust/defaults/main.yml @@ -37,12 +37,29 @@ rust_cache_build_dir_max: "auto" rust_cache_max_age_days: 14 # Off-hours, so a prune is unlikely to land mid-build. -rust_cache_prune_schedule_weekday: "Sun" +# +# Daily rather than weekly: the ceiling binds only while the tool runs, so the +# gap between runs is time the pool spends unbounded. Set a weekday for weekly. +rust_cache_prune_schedule_weekday: "" rust_cache_prune_schedule_hour: 3 +# Backstop for growth that outruns the daily prune. One statvfs an hour, and a +# prune only below the floor. +# +# A percentage because one size cannot suit a laptop and a build box. +# +# The guard prunes to the ordinary ceiling and no further. A pool already inside +# the ceiling means the space went elsewhere, and the run reports that. +rust_cache_prune_guard_enabled: true +rust_cache_prune_free_floor: "20%" +rust_cache_prune_guard_schedule: "hourly" +rust_cache_prune_guard_interval_seconds: 3600 + # macOS launchd only; systemd takes its unit name from the file. rust_cache_prune_label: "io.hyperi.rust-cache-prune" rust_cache_prune_plist: "{{ user_home }}/Library/LaunchAgents/{{ rust_cache_prune_label }}.plist" +rust_cache_prune_guard_label: "io.hyperi.rust-cache-prune-guard" +rust_cache_prune_guard_plist: "{{ user_home }}/Library/LaunchAgents/{{ rust_cache_prune_guard_label }}.plist" # Start the sccache server from a systemd user unit rather than leaving it to # whichever process compiles first, which decides the server's cgroup and diff --git a/ansible/roles/developer-rust/files/hyperi-rust-cache-prune b/ansible/roles/developer-rust/files/hyperi-rust-cache-prune index 7c06c01..89dc1ae 100644 --- a/ansible/roles/developer-rust/files/hyperi-rust-cache-prune +++ b/ansible/roles/developer-rust/files/hyperi-rust-cache-prune @@ -6,6 +6,7 @@ Run it yourself: hyperi-rust-cache-prune # report, then ask before deleting hyperi-rust-cache-prune --check # report only, delete nothing hyperi-rust-cache-prune --yes # no prompt (how the timer invokes it) + hyperi-rust-cache-prune --if-free-below 20% # only when the disk is tight Why this exists --------------- @@ -33,6 +34,25 @@ a workspace you compile against daily and one you only link against look the same. Age is measured from the newest thing cargo wrote, which is the closest honest proxy. +When it runs at all +------------------- +The ceiling is enforced when the tool runs and at no other moment, so a pool +that grows faster than the schedule spends the gap unbounded. Shortening the +schedule alone trades that for a walk of the whole pool every time. + +`--if-free-below` is the answer to that. It costs one statvfs, so the tool can +be scheduled often enough to matter and still do nothing almost every time: if +the filesystem holding the pool has more free space than the floor, it exits +before walking anything. + +The floor takes a percentage as well as a byte count, because the same number +cannot be right on a 256G laptop and a 692G build box. + +A guarded run that finds the pool already inside its ceiling does not go +looking for more to delete. It says the pool is not where the space went and +stops. Reclaiming artefacts that were not the problem would be the wrong +answer to a disk filled by something else. + What it does NOT touch ---------------------- Project `target/` directories, anywhere. `build.build-dir` moves intermediates @@ -73,6 +93,7 @@ AUTO_FLOOR = "40G" POOL_DEPTH = 2 _SIZE = re.compile(r"^\s*(\d+(?:\.\d+)?)\s*([KMGT]?)i?B?\s*$", re.IGNORECASE) +_PERCENT = re.compile(r"^\s*(\d+(?:\.\d+)?)\s*%\s*$") _MULTIPLIER = {"": 1, "K": 1024, "M": 1024**2, "G": 1024**3, "T": 1024**4} @@ -122,6 +143,49 @@ def parse_size_or_auto(text: str): return parse_size(text) +def parse_size_or_percent(text: str): + """A byte count, or a share of the filesystem written as a percentage. + + Returned as a (kind, value) pair rather than a number because a percentage + cannot be resolved until the pool path names a filesystem to take it of. + """ + match = _PERCENT.match(text) + if match: + share = float(match.group(1)) + if not 0 < share < 100: + raise argparse.ArgumentTypeError( + f"not a usable percentage: {text!r} -- give something above 0 and below 100" + ) + return ("percent", share) + return ("bytes", parse_size(text)) + + +def above_free_floor(pool: Path, floor, rep: Reporter) -> bool: + """Has the filesystem holding the pool got more free space than the floor? + + One statvfs, called before anything walks the pool, so a guarded run that + has nothing to do costs nothing. `free`, not `total - used`: reserved + blocks are not ours to spend and counting them would let the guard sit + quiet while writes are already failing. + """ + kind, value = floor + try: + usage = shutil.disk_usage(pool) + except OSError as exc: + # Unmeasurable is not the same as fine. Skipping the prune here would + # turn a broken statvfs into a silently unbounded cache. + rep.warn(f"could not measure free space at {pool} ({exc}) -- pruning anyway") + return False + + threshold = int(usage.total * value / 100) if kind == "percent" else value + where = f"{human(usage.free)} free of {human(usage.total)}" + if usage.free > threshold: + rep.info(f"{where}, above the {human(threshold)} floor -- nothing to do") + return True + rep.info(f"{where}, below the {human(threshold)} floor") + return False + + def resolve_max_size(value, pool: Path, rep: Reporter) -> int: """Turn `auto` into bytes for the filesystem that actually holds the pool.""" if value != "auto": @@ -277,27 +341,38 @@ def find_workspaces(pool: Path) -> list[Workspace]: return [Workspace(leaf) for leaf in leaves] -def remove(workspace: Workspace, reason: str, dry_run: bool, rep: Reporter) -> None: +def remove(workspace: Workspace, reason: str, dry_run: bool, rep: Reporter) -> bool: + """Drop one workspace. False means the bytes are still on disk. + + The caller has to know, because counting a failed removal against the pool + total reports a cache back under its ceiling while it is still over. + """ if dry_run: rep.info(f"[check] would drop {workspace.path.name} ({human(workspace.size)}, {reason})") rep.freed += workspace.size - return + return True try: shutil.rmtree(workspace.path) except OSError as exc: rep.warn(f"could not remove {workspace.path}: {exc}") - return + return False rep.change(f"dropped {workspace.path.name} ({human(workspace.size)}, {reason})") rep.freed += workspace.size + return True def prune_pool( pool: Path, dry_run: bool, rep: Reporter, *, max_size: int, max_age_days: int -) -> None: +) -> int: + """Bound the pool, and return what it still occupies. + + Callers want that rather than the freed count, which reads the same whether + there was nothing to free or every eviction failed. + """ workspaces = find_workspaces(pool) if not workspaces: rep.info(f"{pool}: empty, nothing to prune") - return + return 0 total = sum(item.size for item in workspaces) rep.info( @@ -306,21 +381,21 @@ def prune_pool( stale = [item for item in workspaces if item.age_days > max_age_days] for item in stale: - remove(item, f"idle {item.age_days:.0f}d", dry_run, rep) + if remove(item, f"idle {item.age_days:.0f}d", dry_run, rep): + total -= item.size remaining = [item for item in workspaces if item not in stale] - total -= sum(item.size for item in stale) if total <= max_size: rep.info(f"under the ceiling at {human(total)}") - return + return total # Oldest first: the least-recently-built workspace is the cheapest to lose. for item in sorted(remaining, key=lambda entry: entry.mtime): if total <= max_size: break - remove(item, f"over ceiling, {human(total)} used", dry_run, rep) - total -= item.size + if remove(item, f"over ceiling, {human(total)} used", dry_run, rep): + total -= item.size if total > max_size: rep.warn(f"still {human(total)} after pruning everything eligible") @@ -330,6 +405,8 @@ def prune_pool( if not dry_run: drop_empty_shards(pool) + return total + def drop_empty_shards(pool: Path) -> None: """Remove shard directories emptied by eviction, so the pool does not silt up.""" @@ -433,6 +510,16 @@ def main() -> int: metavar="DAYS", help=f"drop workspaces not built in this long (default: {DEFAULT_MAX_AGE_DAYS})", ) + parser.add_argument( + "--if-free-below", + type=parse_size_or_percent, + default=None, + metavar="SIZE|PERCENT", + help=( + "do nothing unless the filesystem holding the pool has less than this " + "free (e.g. 20%% or 80G) -- how the hourly guard invokes it" + ), + ) parser.add_argument( "--pool", type=Path, @@ -470,6 +557,12 @@ def main() -> int: print(f"hyperi-rust-cache-prune: {pool}") + # Before the walk and before the prompt: a guarded run with room to spare + # must cost one statvfs and must never ask the user anything. + guarded = args.if_free_below is not None + if guarded and above_free_floor(pool, args.if_free_below, rep): + return 0 + max_size = resolve_max_size(args.max_size, pool, rep) if not args.check and not args.yes: @@ -485,7 +578,7 @@ def main() -> int: return 0 rep.step("Pooled build artefacts") - prune_pool( + remaining = prune_pool( pool, args.check, rep, @@ -493,6 +586,17 @@ def main() -> int: max_age_days=args.max_age_days, ) + # A guarded run ending inside the ceiling means the pool is not what filled + # the disk. Tested on what the pool still holds, not on what was freed -- a + # failed eviction also frees nothing. + if guarded and remaining <= max_size: + rep.warn( + "below the free floor with the pool already inside its ceiling -- " + "nothing here to reclaim. Whatever filled the disk is outside what " + "this tool bounds, and the caches reported below are the only other " + "ones it can see." + ) + rep.step("Self-capping caches (reported, not pruned)") try: report_self_capping_caches(rep) diff --git a/ansible/roles/developer-rust/tasks/cache.yml b/ansible/roles/developer-rust/tasks/cache.yml index 59dafcd..863aa56 100644 --- a/ansible/roles/developer-rust/tasks/cache.yml +++ b/ansible/roles/developer-rust/tasks/cache.yml @@ -6,6 +6,10 @@ # hyperi-rust-cache-prune bounds that pool on a schedule, because cargo does # not track build artefacts and nothing upstream ever reclaims them. # +# Two schedules, because a ceiling only binds while the tool runs. The daily +# prune is the routine path. The guard runs hourly, costs one statvfs when the +# disk has room, and prunes when free space drops below the floor. +# # The whole file is optional: a machine with no cap still builds, so nothing # here may fail the run. @@ -43,6 +47,26 @@ mode: '0644' become: true + - name: Install the rust cache prune guard service + ansible.builtin.template: + src: hyperi-rust-cache-prune-guard.service.j2 + dest: /etc/systemd/system/hyperi-rust-cache-prune-guard.service + owner: root + group: root + mode: '0644' + become: true + when: rust_cache_prune_guard_enabled | bool + + - name: Install the rust cache prune guard timer + ansible.builtin.template: + src: hyperi-rust-cache-prune-guard.timer.j2 + dest: /etc/systemd/system/hyperi-rust-cache-prune-guard.timer + owner: root + group: root + mode: '0644' + become: true + when: rust_cache_prune_guard_enabled | bool + # Check mode does not write the unit files above, so systemd has nothing to # enable and the task would fail the run rather than preview it. - name: Enable and start the rust cache prune timer @@ -54,6 +78,41 @@ become: true when: not ansible_check_mode + - name: Enable and start the rust cache prune guard timer + ansible.builtin.systemd: + name: hyperi-rust-cache-prune-guard.timer + enabled: true + state: started + daemon_reload: true + become: true + when: + - not ansible_check_mode + - rust_cache_prune_guard_enabled | bool + + # Turning the guard off has to remove it, not just stop writing it. A timer + # left enabled from a previous run would keep firing against a flag that + # says it is disabled. + - name: Stop the rust cache prune guard timer when disabled + ansible.builtin.systemd: + name: hyperi-rust-cache-prune-guard.timer + enabled: false + state: stopped + become: true + failed_when: false + when: + - not ansible_check_mode + - not rust_cache_prune_guard_enabled | bool + + - name: Remove the rust cache prune guard units when disabled + ansible.builtin.file: + path: "/etc/systemd/system/{{ item }}" + state: absent + become: true + loop: + - hyperi-rust-cache-prune-guard.timer + - hyperi-rust-cache-prune-guard.service + when: not rust_cache_prune_guard_enabled | bool + - name: Schedule the prune (macOS) when: ansible_facts['distribution'] == 'MacOSX' block: @@ -66,57 +125,38 @@ mode: '0755' become: false - - name: Install the rust cache prune launch agent - ansible.builtin.template: - src: io.hyperi.rust-cache-prune.plist.j2 - dest: "{{ rust_cache_prune_plist }}" - mode: '0644' - become: false - register: developer_rust_prune_plist + # One include per agent rather than a loop over the tasks: each inclusion + # gets its own register scope, so the bootout/bootstrap decision reads the + # plist result for the agent in hand. + - name: Install and load the rust cache prune launch agents + ansible.builtin.include_tasks: launchd_agent.yml + loop: "{{ developer_rust_launchd_agents }}" + loop_control: + loop_var: launchd_agent + label: "{{ launchd_agent.label }}" + vars: + developer_rust_launchd_agents: >- + {{ [{'label': rust_cache_prune_label, + 'plist': rust_cache_prune_plist, + 'template': 'io.hyperi.rust-cache-prune.plist.j2'}] + + ([{'label': rust_cache_prune_guard_label, + 'plist': rust_cache_prune_guard_plist, + 'template': 'io.hyperi.rust-cache-prune-guard.plist.j2'}] + if rust_cache_prune_guard_enabled | bool else []) }} - # `launchctl print` is the only reliable "is it loaded" probe; `list` greps - # a label that may match a different agent's prefix. - - name: Check whether the launch agent is already loaded + - name: Unload the rust cache prune guard launch agent when disabled ansible.builtin.command: - cmd: "launchctl print gui/{{ ansible_facts['user_uid'] }}/{{ rust_cache_prune_label }}" - become: false - register: developer_rust_prune_loaded - changed_when: false - failed_when: false - - # bootout first: bootstrap on an already-loaded label is an error, and the - # plist may have changed underneath it. - - name: Unload the previous launch agent - ansible.builtin.command: - cmd: "launchctl bootout gui/{{ ansible_facts['user_uid'] }}/{{ rust_cache_prune_label }}" + cmd: "launchctl bootout gui/{{ ansible_facts['user_uid'] }}/{{ rust_cache_prune_guard_label }}" become: false changed_when: false failed_when: false when: - not ansible_check_mode - - developer_rust_prune_loaded.rc == 0 - - developer_rust_prune_plist.changed + - not rust_cache_prune_guard_enabled | bool - - name: Load the rust cache prune launch agent - ansible.builtin.command: - cmd: >- - launchctl bootstrap gui/{{ ansible_facts['user_uid'] }} - {{ rust_cache_prune_plist }} + - name: Remove the rust cache prune guard launch agent when disabled + ansible.builtin.file: + path: "{{ rust_cache_prune_guard_plist }}" + state: absent become: false - register: developer_rust_prune_bootstrap - changed_when: developer_rust_prune_bootstrap.rc == 0 - failed_when: false - when: - - not ansible_check_mode - - developer_rust_prune_loaded.rc != 0 or developer_rust_prune_plist.changed - - - name: Warn when the launch agent did not load - ansible.builtin.set_fact: - deploy_warnings: >- - {{ deploy_warnings | default([]) - + ['Load the rust cache prune launch agent: ' - ~ (developer_rust_prune_bootstrap.stderr | default('unknown error'))] }} - when: - - not ansible_check_mode - - developer_rust_prune_bootstrap is defined - - developer_rust_prune_bootstrap.rc | default(0) != 0 + when: not rust_cache_prune_guard_enabled | bool diff --git a/ansible/roles/developer-rust/tasks/launchd_agent.yml b/ansible/roles/developer-rust/tasks/launchd_agent.yml new file mode 100644 index 0000000..582ec68 --- /dev/null +++ b/ansible/roles/developer-rust/tasks/launchd_agent.yml @@ -0,0 +1,60 @@ +--- +# Load one launchd agent idempotently. Included once per agent from cache.yml, +# which passes `launchd_agent` as {label, plist, template}. +# +# launchd has no reload verb, so a changed plist has to be booted out first. + +- name: "Install launch agent {{ launchd_agent.label }}" + ansible.builtin.template: + src: "{{ launchd_agent.template }}" + dest: "{{ launchd_agent.plist }}" + mode: '0644' + become: false + register: developer_rust_agent_plist + +# `launchctl print` is the only reliable "is it loaded" probe; `list` greps +# a label that may match a different agent's prefix. +- name: "Check the loaded state of launch agent {{ launchd_agent.label }}" + ansible.builtin.command: + cmd: "launchctl print gui/{{ ansible_facts['user_uid'] }}/{{ launchd_agent.label }}" + become: false + register: developer_rust_agent_loaded + changed_when: false + failed_when: false + +# bootout first: bootstrap on an already-loaded label is an error, and the +# plist may have changed underneath it. +- name: "Unload the previous launch agent {{ launchd_agent.label }}" + ansible.builtin.command: + cmd: "launchctl bootout gui/{{ ansible_facts['user_uid'] }}/{{ launchd_agent.label }}" + become: false + changed_when: false + failed_when: false + when: + - not ansible_check_mode + - developer_rust_agent_loaded.rc == 0 + - developer_rust_agent_plist.changed + +- name: "Load launch agent {{ launchd_agent.label }}" + ansible.builtin.command: + cmd: >- + launchctl bootstrap gui/{{ ansible_facts['user_uid'] }} + {{ launchd_agent.plist }} + become: false + register: developer_rust_agent_bootstrap + changed_when: developer_rust_agent_bootstrap.rc == 0 + failed_when: false + when: + - not ansible_check_mode + - developer_rust_agent_loaded.rc != 0 or developer_rust_agent_plist.changed + +- name: "Warn about a launch agent that did not load: {{ launchd_agent.label }}" + ansible.builtin.set_fact: + deploy_warnings: >- + {{ deploy_warnings | default([]) + + ['Load the ' ~ launchd_agent.label ~ ' launch agent: ' + ~ (developer_rust_agent_bootstrap.stderr | default('unknown error'))] }} + when: + - not ansible_check_mode + - developer_rust_agent_bootstrap is defined + - developer_rust_agent_bootstrap.rc | default(0) != 0 diff --git a/ansible/roles/developer-rust/templates/hyperi-rust-cache-prune-guard.service.j2 b/ansible/roles/developer-rust/templates/hyperi-rust-cache-prune-guard.service.j2 new file mode 100644 index 0000000..67a662e --- /dev/null +++ b/ansible/roles/developer-rust/templates/hyperi-rust-cache-prune-guard.service.j2 @@ -0,0 +1,13 @@ +[Unit] +Description=Prune the pooled Rust build cache when free space runs short + +[Service] +Type=oneshot +User={{ actual_user }} +# Same prune with the same ceiling, gated on free space. Above the floor it +# exits after one statvfs without walking the pool. +ExecStart=/usr/local/bin/hyperi-rust-cache-prune --yes --max-size {{ rust_cache_build_dir_max }} --max-age-days {{ rust_cache_max_age_days }} --if-free-below {{ rust_cache_prune_free_floor }} +# Walking a large pool is IO-bound; it must never compete with an active build. +Nice=19 +IOSchedulingClass=idle +TimeoutStartSec=1800 diff --git a/ansible/roles/developer-rust/templates/hyperi-rust-cache-prune-guard.timer.j2 b/ansible/roles/developer-rust/templates/hyperi-rust-cache-prune-guard.timer.j2 new file mode 100644 index 0000000..9baa0af --- /dev/null +++ b/ansible/roles/developer-rust/templates/hyperi-rust-cache-prune-guard.timer.j2 @@ -0,0 +1,12 @@ +[Unit] +Description=Watch free space for the pooled Rust build cache + +[Timer] +OnCalendar={{ rust_cache_prune_guard_schedule }} +# Persistent so a box that was off catches up once, not once per missed hour. +Persistent=true +# Short jitter only. A guard that fires late is a guard that fired too late. +RandomizedDelaySec=5m + +[Install] +WantedBy=timers.target diff --git a/ansible/roles/developer-rust/templates/hyperi-rust-cache-prune.timer.j2 b/ansible/roles/developer-rust/templates/hyperi-rust-cache-prune.timer.j2 index 43c2283..83f2ba3 100644 --- a/ansible/roles/developer-rust/templates/hyperi-rust-cache-prune.timer.j2 +++ b/ansible/roles/developer-rust/templates/hyperi-rust-cache-prune.timer.j2 @@ -1,8 +1,8 @@ [Unit] -Description=Bound the pooled Rust build cache weekly +Description=Bound the pooled Rust build cache [Timer] -OnCalendar={{ rust_cache_prune_schedule_weekday }} *-*-* {{ '%02d' | format(rust_cache_prune_schedule_hour | int) }}:00:00 +OnCalendar={{ (rust_cache_prune_schedule_weekday ~ ' ') if rust_cache_prune_schedule_weekday else '' }}*-*-* {{ '%02d' | format(rust_cache_prune_schedule_hour | int) }}:00:00 # Persistent so a box that was off over the scheduled window still prunes. Persistent=true RandomizedDelaySec=30m diff --git a/ansible/roles/developer-rust/templates/io.hyperi.rust-cache-prune-guard.plist.j2 b/ansible/roles/developer-rust/templates/io.hyperi.rust-cache-prune-guard.plist.j2 new file mode 100644 index 0000000..e985b41 --- /dev/null +++ b/ansible/roles/developer-rust/templates/io.hyperi.rust-cache-prune-guard.plist.j2 @@ -0,0 +1,43 @@ + + + + + Label + {{ rust_cache_prune_guard_label }} + + ProgramArguments + + /usr/local/bin/hyperi-rust-cache-prune + --yes + --max-size + {{ rust_cache_build_dir_max }} + --max-age-days + {{ rust_cache_max_age_days }} + --if-free-below + {{ rust_cache_prune_free_floor }} + + + + StartInterval + {{ rust_cache_prune_guard_interval_seconds | int }} + + + RunAtLoad + + + + ProcessType + Background + LowPriorityIO + + Nice + 19 + + StandardOutPath + {{ user_home }}/Library/Logs/hyperi-rust-cache-prune.log + StandardErrorPath + {{ user_home }}/Library/Logs/hyperi-rust-cache-prune.log + + diff --git a/ansible/roles/developer-rust/templates/io.hyperi.rust-cache-prune.plist.j2 b/ansible/roles/developer-rust/templates/io.hyperi.rust-cache-prune.plist.j2 index e476ae2..70f8f9b 100644 --- a/ansible/roles/developer-rust/templates/io.hyperi.rust-cache-prune.plist.j2 +++ b/ansible/roles/developer-rust/templates/io.hyperi.rust-cache-prune.plist.j2 @@ -1,4 +1,4 @@ -{% set weekday_num = {'Sun': 0, 'Mon': 1, 'Tue': 2, 'Wed': 3, 'Thu': 4, 'Fri': 5, 'Sat': 6}[rust_cache_prune_schedule_weekday] %} +{% set weekday_num = {'Sun': 0, 'Mon': 1, 'Tue': 2, 'Wed': 3, 'Thu': 4, 'Fri': 5, 'Sat': 6}.get(rust_cache_prune_schedule_weekday) %} @@ -17,11 +17,13 @@ + what a laptop closed overnight needs. Omitting Weekday means daily. --> StartCalendarInterval +{% if weekday_num is not none %} Weekday {{ weekday_num }} +{% endif %} Hour {{ rust_cache_prune_schedule_hour | int }} Minute diff --git a/tools/tests/test_rust_cache_prune.py b/tools/tests/test_rust_cache_prune.py index dd99c3c..5d1f642 100644 --- a/tools/tests/test_rust_cache_prune.py +++ b/tools/tests/test_rust_cache_prune.py @@ -6,6 +6,7 @@ import importlib.util from pathlib import Path +from types import SimpleNamespace import pytest @@ -105,3 +106,112 @@ def test_parse_size_accepts_auto_and_sizes(prune): assert prune.parse_size_or_auto("auto") == "auto" assert prune.parse_size_or_auto(" AUTO ") == "auto" assert prune.parse_size_or_auto("40G") == prune.parse_size("40G") + + +# --------------------------------------------------------------------------- +# The free-space guard +# --------------------------------------------------------------------------- + + +def fake_usage(total, free): + """Stand in for shutil.disk_usage, which only .total and .free are read from.""" + return SimpleNamespace(total=total, used=total - free, free=free) + + +def test_floor_accepts_a_percentage(prune): + assert prune.parse_size_or_percent("20%") == ("percent", 20.0) + assert prune.parse_size_or_percent(" 7.5 % ") == ("percent", 7.5) + + +def test_floor_accepts_a_byte_count(prune): + assert prune.parse_size_or_percent("80G") == ("bytes", prune.parse_size("80G")) + + +@pytest.mark.parametrize("text", ["0%", "100%", "-5%", "abc", ""]) +def test_floor_rejects_what_cannot_be_a_floor(prune, text): + """0 and 100 are rejected as well as junk: neither can gate anything.""" + with pytest.raises(Exception): + prune.parse_size_or_percent(text) + + +def test_guard_does_nothing_while_there_is_room(prune, tmp_path, monkeypatch): + """The whole point of the guard is being cheap when it has no work. + + Above the floor it must answer from one statvfs and never reach the pool. + """ + monkeypatch.setattr(prune.shutil, "disk_usage", lambda _: fake_usage(1000, 500)) + assert prune.above_free_floor(tmp_path, ("percent", 20.0), prune.Reporter()) is True + + +def test_guard_lets_the_prune_through_when_space_is_short(prune, tmp_path, monkeypatch): + monkeypatch.setattr(prune.shutil, "disk_usage", lambda _: fake_usage(1000, 100)) + assert prune.above_free_floor(tmp_path, ("percent", 20.0), prune.Reporter()) is False + + +def test_guard_takes_a_byte_floor_as_well(prune, tmp_path, monkeypatch): + monkeypatch.setattr(prune.shutil, "disk_usage", lambda _: fake_usage(1000, 100)) + assert prune.above_free_floor(tmp_path, ("bytes", 50), prune.Reporter()) is True + assert prune.above_free_floor(tmp_path, ("bytes", 150), prune.Reporter()) is False + + +def make_pool(tmp_path, sizes): + """Build a pool at cargo's shard depth, one leaf per given byte size.""" + pool = tmp_path / "pool" + for index, size in enumerate(sizes): + leaf = pool / f"{index:02x}" / f"hash{index}" + leaf.mkdir(parents=True) + (leaf / "artefact.bin").write_bytes(b"\0" * size) + return pool + + +def test_the_reported_total_counts_only_evictions_that_worked(prune, tmp_path, monkeypatch): + """A failed rmtree must leave its bytes in the total. + + Crediting them anyway reports the pool back under its ceiling while it is + still over, which is the one thing the guard's verdict rests on. + """ + pool = make_pool(tmp_path, [200_000, 200_000, 200_000]) + + def refuse(_path): + raise OSError("permission denied") + + monkeypatch.setattr(prune.shutil, "rmtree", refuse) + rep = prune.Reporter() + remaining = prune_pool_at(prune, pool, rep, max_size=1) + + assert rep.freed == 0 + assert remaining > 1, "a pool whose evictions all failed is still over its ceiling" + assert rep.warnings + + +def test_the_reported_total_drops_when_evictions_succeed(prune, tmp_path): + pool = make_pool(tmp_path, [200_000, 200_000, 200_000]) + rep = prune.Reporter() + before = prune_pool_at(prune, pool, rep, max_size=10**9) + + rep_two = prune.Reporter() + after = prune_pool_at(prune, pool, rep_two, max_size=1) + + assert after < before + assert rep_two.freed > 0 + + +def prune_pool_at(prune, pool, rep, *, max_size): + """prune_pool with the age pass held off, so only the size pass is in play.""" + return prune.prune_pool(pool, False, rep, max_size=max_size, max_age_days=10**6) + + +def test_an_unmeasurable_filesystem_prunes_rather_than_skipping(prune, tmp_path, monkeypatch): + """Unknown free space must not be read as plenty. + + Treating a failed statvfs as room to spare would turn a broken probe into a + cache nothing ever bounds again. + """ + + def explode(_): + raise OSError("no") + + monkeypatch.setattr(prune.shutil, "disk_usage", explode) + rep = prune.Reporter() + assert prune.above_free_floor(tmp_path, ("percent", 20.0), rep) is False + assert rep.warnings