diff --git a/ansible/roles/developer-rust/README.md b/ansible/roles/developer-rust/README.md index fdefde7..48673f7 100644 --- a/ansible/roles/developer-rust/README.md +++ b/ansible/roles/developer-rust/README.md @@ -70,6 +70,8 @@ silently stops being cached while `sccache` is still installed and configured. `hyperi-rust-cache-prune` reports it rather than showing an empty ceiling, and `sccache --stop-server` clears it. +## Bounding the build-artefact pool + **Build artefacts have no upstream cap at all.** Cargo does not track them (rust-lang/cargo#13136), so nothing reclaims a `target/` ever, and one per repo across a tree full of them is what actually fills a disk. @@ -82,7 +84,7 @@ binaries, so `cargo run`, IDEs and anything globbing for a built artefact are unaffected. `hyperi-rust-cache-prune` then bounds that pool on a schedule -- a systemd timer -on Linux, a launchd agent on macOS, weekly and at idle IO priority. It drops +on Linux, a launchd agent on macOS, daily and at idle IO priority. It drops workspaces not built for `rust_cache_max_age_days`, then evicts least-recently-built ones until the pool is under `rust_cache_build_dir_max`. It touches no project `target/`, and reports the self-capping caches without @@ -94,14 +96,18 @@ derived from it chases itself downward. That puts a 692G build box at 115G and a 256G laptop at the floor, so one default suits both. Set `rust_cache_build_dir_max` to an explicit size to override it. -**The ceiling is a weekly reset, not a live cap.** Nothing enforces it between -runs, so a busy build box spends most of the week above it -- one added 37G in -two days and another 37G in a single afternoon. That is the design working, not -a prune that failed. It only matters if the peak, not the floor, would fill the -disk; check free space before reaching for a shorter schedule, and change -`rust_cache_prune_schedule_weekday` if the peak is genuinely too high. +**The ceiling binds while the tool runs, not between runs.** A pool that grows +faster than the schedule spends the gap above it, so the prune runs daily and an +hourly guard backs it up -- one statvfs while the disk has room, a prune to the +same ceiling once free space falls below `rust_cache_prune_free_floor` (20%). + +A guard run that finds the pool already under its ceiling stops and says so. The +space went somewhere the prune does not own, and naming that is more use than +evicting artefacts that were not the cause. Set +`rust_cache_prune_guard_enabled: false` to drop the guard, or +`rust_cache_prune_schedule_weekday` to go back to weekly. -sccache and ccache keep fixed ceilings; both enforce their own. +## Which sccache builds actually use **sccache comes from upstream's release, not the distro.** Ubuntu ships 0.13.0 against an upstream on 0.17.x, and a wrapper four versions behind is what a diff --git a/ansible/roles/developer-rust/defaults/main.yml b/ansible/roles/developer-rust/defaults/main.yml index fe73f33..2772c0f 100644 --- a/ansible/roles/developer-rust/defaults/main.yml +++ b/ansible/roles/developer-rust/defaults/main.yml @@ -37,12 +37,29 @@ rust_cache_build_dir_max: "auto" rust_cache_max_age_days: 14 # Off-hours, so a prune is unlikely to land mid-build. -rust_cache_prune_schedule_weekday: "Sun" +# +# Daily rather than weekly: the ceiling binds only while the tool runs, so the +# gap between runs is time the pool spends unbounded. Set a weekday for weekly. +rust_cache_prune_schedule_weekday: "" rust_cache_prune_schedule_hour: 3 +# Backstop for growth that outruns the daily prune. One statvfs an hour, and a +# prune only below the floor. +# +# A percentage because one size cannot suit a laptop and a build box. +# +# The guard prunes to the ordinary ceiling and no further. A pool already inside +# the ceiling means the space went elsewhere, and the run reports that. +rust_cache_prune_guard_enabled: true +rust_cache_prune_free_floor: "20%" +rust_cache_prune_guard_schedule: "hourly" +rust_cache_prune_guard_interval_seconds: 3600 + # macOS launchd only; systemd takes its unit name from the file. rust_cache_prune_label: "io.hyperi.rust-cache-prune" rust_cache_prune_plist: "{{ user_home }}/Library/LaunchAgents/{{ rust_cache_prune_label }}.plist" +rust_cache_prune_guard_label: "io.hyperi.rust-cache-prune-guard" +rust_cache_prune_guard_plist: "{{ user_home }}/Library/LaunchAgents/{{ rust_cache_prune_guard_label }}.plist" # Start the sccache server from a systemd user unit rather than leaving it to # whichever process compiles first, which decides the server's cgroup and diff --git a/ansible/roles/developer-rust/files/hyperi-rust-cache-prune b/ansible/roles/developer-rust/files/hyperi-rust-cache-prune index 7c06c01..89dc1ae 100644 --- a/ansible/roles/developer-rust/files/hyperi-rust-cache-prune +++ b/ansible/roles/developer-rust/files/hyperi-rust-cache-prune @@ -6,6 +6,7 @@ Run it yourself: hyperi-rust-cache-prune # report, then ask before deleting hyperi-rust-cache-prune --check # report only, delete nothing hyperi-rust-cache-prune --yes # no prompt (how the timer invokes it) + hyperi-rust-cache-prune --if-free-below 20% # only when the disk is tight Why this exists --------------- @@ -33,6 +34,25 @@ a workspace you compile against daily and one you only link against look the same. Age is measured from the newest thing cargo wrote, which is the closest honest proxy. +When it runs at all +------------------- +The ceiling is enforced when the tool runs and at no other moment, so a pool +that grows faster than the schedule spends the gap unbounded. Shortening the +schedule alone trades that for a walk of the whole pool every time. + +`--if-free-below` is the answer to that. It costs one statvfs, so the tool can +be scheduled often enough to matter and still do nothing almost every time: if +the filesystem holding the pool has more free space than the floor, it exits +before walking anything. + +The floor takes a percentage as well as a byte count, because the same number +cannot be right on a 256G laptop and a 692G build box. + +A guarded run that finds the pool already inside its ceiling does not go +looking for more to delete. It says the pool is not where the space went and +stops. Reclaiming artefacts that were not the problem would be the wrong +answer to a disk filled by something else. + What it does NOT touch ---------------------- Project `target/` directories, anywhere. `build.build-dir` moves intermediates @@ -73,6 +93,7 @@ AUTO_FLOOR = "40G" POOL_DEPTH = 2 _SIZE = re.compile(r"^\s*(\d+(?:\.\d+)?)\s*([KMGT]?)i?B?\s*$", re.IGNORECASE) +_PERCENT = re.compile(r"^\s*(\d+(?:\.\d+)?)\s*%\s*$") _MULTIPLIER = {"": 1, "K": 1024, "M": 1024**2, "G": 1024**3, "T": 1024**4} @@ -122,6 +143,49 @@ def parse_size_or_auto(text: str): return parse_size(text) +def parse_size_or_percent(text: str): + """A byte count, or a share of the filesystem written as a percentage. + + Returned as a (kind, value) pair rather than a number because a percentage + cannot be resolved until the pool path names a filesystem to take it of. + """ + match = _PERCENT.match(text) + if match: + share = float(match.group(1)) + if not 0 < share < 100: + raise argparse.ArgumentTypeError( + f"not a usable percentage: {text!r} -- give something above 0 and below 100" + ) + return ("percent", share) + return ("bytes", parse_size(text)) + + +def above_free_floor(pool: Path, floor, rep: Reporter) -> bool: + """Has the filesystem holding the pool got more free space than the floor? + + One statvfs, called before anything walks the pool, so a guarded run that + has nothing to do costs nothing. `free`, not `total - used`: reserved + blocks are not ours to spend and counting them would let the guard sit + quiet while writes are already failing. + """ + kind, value = floor + try: + usage = shutil.disk_usage(pool) + except OSError as exc: + # Unmeasurable is not the same as fine. Skipping the prune here would + # turn a broken statvfs into a silently unbounded cache. + rep.warn(f"could not measure free space at {pool} ({exc}) -- pruning anyway") + return False + + threshold = int(usage.total * value / 100) if kind == "percent" else value + where = f"{human(usage.free)} free of {human(usage.total)}" + if usage.free > threshold: + rep.info(f"{where}, above the {human(threshold)} floor -- nothing to do") + return True + rep.info(f"{where}, below the {human(threshold)} floor") + return False + + def resolve_max_size(value, pool: Path, rep: Reporter) -> int: """Turn `auto` into bytes for the filesystem that actually holds the pool.""" if value != "auto": @@ -277,27 +341,38 @@ def find_workspaces(pool: Path) -> list[Workspace]: return [Workspace(leaf) for leaf in leaves] -def remove(workspace: Workspace, reason: str, dry_run: bool, rep: Reporter) -> None: +def remove(workspace: Workspace, reason: str, dry_run: bool, rep: Reporter) -> bool: + """Drop one workspace. False means the bytes are still on disk. + + The caller has to know, because counting a failed removal against the pool + total reports a cache back under its ceiling while it is still over. + """ if dry_run: rep.info(f"[check] would drop {workspace.path.name} ({human(workspace.size)}, {reason})") rep.freed += workspace.size - return + return True try: shutil.rmtree(workspace.path) except OSError as exc: rep.warn(f"could not remove {workspace.path}: {exc}") - return + return False rep.change(f"dropped {workspace.path.name} ({human(workspace.size)}, {reason})") rep.freed += workspace.size + return True def prune_pool( pool: Path, dry_run: bool, rep: Reporter, *, max_size: int, max_age_days: int -) -> None: +) -> int: + """Bound the pool, and return what it still occupies. + + Callers want that rather than the freed count, which reads the same whether + there was nothing to free or every eviction failed. + """ workspaces = find_workspaces(pool) if not workspaces: rep.info(f"{pool}: empty, nothing to prune") - return + return 0 total = sum(item.size for item in workspaces) rep.info( @@ -306,21 +381,21 @@ def prune_pool( stale = [item for item in workspaces if item.age_days > max_age_days] for item in stale: - remove(item, f"idle {item.age_days:.0f}d", dry_run, rep) + if remove(item, f"idle {item.age_days:.0f}d", dry_run, rep): + total -= item.size remaining = [item for item in workspaces if item not in stale] - total -= sum(item.size for item in stale) if total <= max_size: rep.info(f"under the ceiling at {human(total)}") - return + return total # Oldest first: the least-recently-built workspace is the cheapest to lose. for item in sorted(remaining, key=lambda entry: entry.mtime): if total <= max_size: break - remove(item, f"over ceiling, {human(total)} used", dry_run, rep) - total -= item.size + if remove(item, f"over ceiling, {human(total)} used", dry_run, rep): + total -= item.size if total > max_size: rep.warn(f"still {human(total)} after pruning everything eligible") @@ -330,6 +405,8 @@ def prune_pool( if not dry_run: drop_empty_shards(pool) + return total + def drop_empty_shards(pool: Path) -> None: """Remove shard directories emptied by eviction, so the pool does not silt up.""" @@ -433,6 +510,16 @@ def main() -> int: metavar="DAYS", help=f"drop workspaces not built in this long (default: {DEFAULT_MAX_AGE_DAYS})", ) + parser.add_argument( + "--if-free-below", + type=parse_size_or_percent, + default=None, + metavar="SIZE|PERCENT", + help=( + "do nothing unless the filesystem holding the pool has less than this " + "free (e.g. 20%% or 80G) -- how the hourly guard invokes it" + ), + ) parser.add_argument( "--pool", type=Path, @@ -470,6 +557,12 @@ def main() -> int: print(f"hyperi-rust-cache-prune: {pool}") + # Before the walk and before the prompt: a guarded run with room to spare + # must cost one statvfs and must never ask the user anything. + guarded = args.if_free_below is not None + if guarded and above_free_floor(pool, args.if_free_below, rep): + return 0 + max_size = resolve_max_size(args.max_size, pool, rep) if not args.check and not args.yes: @@ -485,7 +578,7 @@ def main() -> int: return 0 rep.step("Pooled build artefacts") - prune_pool( + remaining = prune_pool( pool, args.check, rep, @@ -493,6 +586,17 @@ def main() -> int: max_age_days=args.max_age_days, ) + # A guarded run ending inside the ceiling means the pool is not what filled + # the disk. Tested on what the pool still holds, not on what was freed -- a + # failed eviction also frees nothing. + if guarded and remaining <= max_size: + rep.warn( + "below the free floor with the pool already inside its ceiling -- " + "nothing here to reclaim. Whatever filled the disk is outside what " + "this tool bounds, and the caches reported below are the only other " + "ones it can see." + ) + rep.step("Self-capping caches (reported, not pruned)") try: report_self_capping_caches(rep) diff --git a/ansible/roles/developer-rust/tasks/cache.yml b/ansible/roles/developer-rust/tasks/cache.yml index 59dafcd..863aa56 100644 --- a/ansible/roles/developer-rust/tasks/cache.yml +++ b/ansible/roles/developer-rust/tasks/cache.yml @@ -6,6 +6,10 @@ # hyperi-rust-cache-prune bounds that pool on a schedule, because cargo does # not track build artefacts and nothing upstream ever reclaims them. # +# Two schedules, because a ceiling only binds while the tool runs. The daily +# prune is the routine path. The guard runs hourly, costs one statvfs when the +# disk has room, and prunes when free space drops below the floor. +# # The whole file is optional: a machine with no cap still builds, so nothing # here may fail the run. @@ -43,6 +47,26 @@ mode: '0644' become: true + - name: Install the rust cache prune guard service + ansible.builtin.template: + src: hyperi-rust-cache-prune-guard.service.j2 + dest: /etc/systemd/system/hyperi-rust-cache-prune-guard.service + owner: root + group: root + mode: '0644' + become: true + when: rust_cache_prune_guard_enabled | bool + + - name: Install the rust cache prune guard timer + ansible.builtin.template: + src: hyperi-rust-cache-prune-guard.timer.j2 + dest: /etc/systemd/system/hyperi-rust-cache-prune-guard.timer + owner: root + group: root + mode: '0644' + become: true + when: rust_cache_prune_guard_enabled | bool + # Check mode does not write the unit files above, so systemd has nothing to # enable and the task would fail the run rather than preview it. - name: Enable and start the rust cache prune timer @@ -54,6 +78,41 @@ become: true when: not ansible_check_mode + - name: Enable and start the rust cache prune guard timer + ansible.builtin.systemd: + name: hyperi-rust-cache-prune-guard.timer + enabled: true + state: started + daemon_reload: true + become: true + when: + - not ansible_check_mode + - rust_cache_prune_guard_enabled | bool + + # Turning the guard off has to remove it, not just stop writing it. A timer + # left enabled from a previous run would keep firing against a flag that + # says it is disabled. + - name: Stop the rust cache prune guard timer when disabled + ansible.builtin.systemd: + name: hyperi-rust-cache-prune-guard.timer + enabled: false + state: stopped + become: true + failed_when: false + when: + - not ansible_check_mode + - not rust_cache_prune_guard_enabled | bool + + - name: Remove the rust cache prune guard units when disabled + ansible.builtin.file: + path: "/etc/systemd/system/{{ item }}" + state: absent + become: true + loop: + - hyperi-rust-cache-prune-guard.timer + - hyperi-rust-cache-prune-guard.service + when: not rust_cache_prune_guard_enabled | bool + - name: Schedule the prune (macOS) when: ansible_facts['distribution'] == 'MacOSX' block: @@ -66,57 +125,38 @@ mode: '0755' become: false - - name: Install the rust cache prune launch agent - ansible.builtin.template: - src: io.hyperi.rust-cache-prune.plist.j2 - dest: "{{ rust_cache_prune_plist }}" - mode: '0644' - become: false - register: developer_rust_prune_plist + # One include per agent rather than a loop over the tasks: each inclusion + # gets its own register scope, so the bootout/bootstrap decision reads the + # plist result for the agent in hand. + - name: Install and load the rust cache prune launch agents + ansible.builtin.include_tasks: launchd_agent.yml + loop: "{{ developer_rust_launchd_agents }}" + loop_control: + loop_var: launchd_agent + label: "{{ launchd_agent.label }}" + vars: + developer_rust_launchd_agents: >- + {{ [{'label': rust_cache_prune_label, + 'plist': rust_cache_prune_plist, + 'template': 'io.hyperi.rust-cache-prune.plist.j2'}] + + ([{'label': rust_cache_prune_guard_label, + 'plist': rust_cache_prune_guard_plist, + 'template': 'io.hyperi.rust-cache-prune-guard.plist.j2'}] + if rust_cache_prune_guard_enabled | bool else []) }} - # `launchctl print` is the only reliable "is it loaded" probe; `list` greps - # a label that may match a different agent's prefix. - - name: Check whether the launch agent is already loaded + - name: Unload the rust cache prune guard launch agent when disabled ansible.builtin.command: - cmd: "launchctl print gui/{{ ansible_facts['user_uid'] }}/{{ rust_cache_prune_label }}" - become: false - register: developer_rust_prune_loaded - changed_when: false - failed_when: false - - # bootout first: bootstrap on an already-loaded label is an error, and the - # plist may have changed underneath it. - - name: Unload the previous launch agent - ansible.builtin.command: - cmd: "launchctl bootout gui/{{ ansible_facts['user_uid'] }}/{{ rust_cache_prune_label }}" + cmd: "launchctl bootout gui/{{ ansible_facts['user_uid'] }}/{{ rust_cache_prune_guard_label }}" become: false changed_when: false failed_when: false when: - not ansible_check_mode - - developer_rust_prune_loaded.rc == 0 - - developer_rust_prune_plist.changed + - not rust_cache_prune_guard_enabled | bool - - name: Load the rust cache prune launch agent - ansible.builtin.command: - cmd: >- - launchctl bootstrap gui/{{ ansible_facts['user_uid'] }} - {{ rust_cache_prune_plist }} + - name: Remove the rust cache prune guard launch agent when disabled + ansible.builtin.file: + path: "{{ rust_cache_prune_guard_plist }}" + state: absent become: false - register: developer_rust_prune_bootstrap - changed_when: developer_rust_prune_bootstrap.rc == 0 - failed_when: false - when: - - not ansible_check_mode - - developer_rust_prune_loaded.rc != 0 or developer_rust_prune_plist.changed - - - name: Warn when the launch agent did not load - ansible.builtin.set_fact: - deploy_warnings: >- - {{ deploy_warnings | default([]) - + ['Load the rust cache prune launch agent: ' - ~ (developer_rust_prune_bootstrap.stderr | default('unknown error'))] }} - when: - - not ansible_check_mode - - developer_rust_prune_bootstrap is defined - - developer_rust_prune_bootstrap.rc | default(0) != 0 + when: not rust_cache_prune_guard_enabled | bool diff --git a/ansible/roles/developer-rust/tasks/launchd_agent.yml b/ansible/roles/developer-rust/tasks/launchd_agent.yml new file mode 100644 index 0000000..582ec68 --- /dev/null +++ b/ansible/roles/developer-rust/tasks/launchd_agent.yml @@ -0,0 +1,60 @@ +--- +# Load one launchd agent idempotently. Included once per agent from cache.yml, +# which passes `launchd_agent` as {label, plist, template}. +# +# launchd has no reload verb, so a changed plist has to be booted out first. + +- name: "Install launch agent {{ launchd_agent.label }}" + ansible.builtin.template: + src: "{{ launchd_agent.template }}" + dest: "{{ launchd_agent.plist }}" + mode: '0644' + become: false + register: developer_rust_agent_plist + +# `launchctl print` is the only reliable "is it loaded" probe; `list` greps +# a label that may match a different agent's prefix. +- name: "Check the loaded state of launch agent {{ launchd_agent.label }}" + ansible.builtin.command: + cmd: "launchctl print gui/{{ ansible_facts['user_uid'] }}/{{ launchd_agent.label }}" + become: false + register: developer_rust_agent_loaded + changed_when: false + failed_when: false + +# bootout first: bootstrap on an already-loaded label is an error, and the +# plist may have changed underneath it. +- name: "Unload the previous launch agent {{ launchd_agent.label }}" + ansible.builtin.command: + cmd: "launchctl bootout gui/{{ ansible_facts['user_uid'] }}/{{ launchd_agent.label }}" + become: false + changed_when: false + failed_when: false + when: + - not ansible_check_mode + - developer_rust_agent_loaded.rc == 0 + - developer_rust_agent_plist.changed + +- name: "Load launch agent {{ launchd_agent.label }}" + ansible.builtin.command: + cmd: >- + launchctl bootstrap gui/{{ ansible_facts['user_uid'] }} + {{ launchd_agent.plist }} + become: false + register: developer_rust_agent_bootstrap + changed_when: developer_rust_agent_bootstrap.rc == 0 + failed_when: false + when: + - not ansible_check_mode + - developer_rust_agent_loaded.rc != 0 or developer_rust_agent_plist.changed + +- name: "Warn about a launch agent that did not load: {{ launchd_agent.label }}" + ansible.builtin.set_fact: + deploy_warnings: >- + {{ deploy_warnings | default([]) + + ['Load the ' ~ launchd_agent.label ~ ' launch agent: ' + ~ (developer_rust_agent_bootstrap.stderr | default('unknown error'))] }} + when: + - not ansible_check_mode + - developer_rust_agent_bootstrap is defined + - developer_rust_agent_bootstrap.rc | default(0) != 0 diff --git a/ansible/roles/developer-rust/templates/hyperi-rust-cache-prune-guard.service.j2 b/ansible/roles/developer-rust/templates/hyperi-rust-cache-prune-guard.service.j2 new file mode 100644 index 0000000..67a662e --- /dev/null +++ b/ansible/roles/developer-rust/templates/hyperi-rust-cache-prune-guard.service.j2 @@ -0,0 +1,13 @@ +[Unit] +Description=Prune the pooled Rust build cache when free space runs short + +[Service] +Type=oneshot +User={{ actual_user }} +# Same prune with the same ceiling, gated on free space. Above the floor it +# exits after one statvfs without walking the pool. +ExecStart=/usr/local/bin/hyperi-rust-cache-prune --yes --max-size {{ rust_cache_build_dir_max }} --max-age-days {{ rust_cache_max_age_days }} --if-free-below {{ rust_cache_prune_free_floor }} +# Walking a large pool is IO-bound; it must never compete with an active build. +Nice=19 +IOSchedulingClass=idle +TimeoutStartSec=1800 diff --git a/ansible/roles/developer-rust/templates/hyperi-rust-cache-prune-guard.timer.j2 b/ansible/roles/developer-rust/templates/hyperi-rust-cache-prune-guard.timer.j2 new file mode 100644 index 0000000..9baa0af --- /dev/null +++ b/ansible/roles/developer-rust/templates/hyperi-rust-cache-prune-guard.timer.j2 @@ -0,0 +1,12 @@ +[Unit] +Description=Watch free space for the pooled Rust build cache + +[Timer] +OnCalendar={{ rust_cache_prune_guard_schedule }} +# Persistent so a box that was off catches up once, not once per missed hour. +Persistent=true +# Short jitter only. A guard that fires late is a guard that fired too late. +RandomizedDelaySec=5m + +[Install] +WantedBy=timers.target diff --git a/ansible/roles/developer-rust/templates/hyperi-rust-cache-prune.timer.j2 b/ansible/roles/developer-rust/templates/hyperi-rust-cache-prune.timer.j2 index 43c2283..83f2ba3 100644 --- a/ansible/roles/developer-rust/templates/hyperi-rust-cache-prune.timer.j2 +++ b/ansible/roles/developer-rust/templates/hyperi-rust-cache-prune.timer.j2 @@ -1,8 +1,8 @@ [Unit] -Description=Bound the pooled Rust build cache weekly +Description=Bound the pooled Rust build cache [Timer] -OnCalendar={{ rust_cache_prune_schedule_weekday }} *-*-* {{ '%02d' | format(rust_cache_prune_schedule_hour | int) }}:00:00 +OnCalendar={{ (rust_cache_prune_schedule_weekday ~ ' ') if rust_cache_prune_schedule_weekday else '' }}*-*-* {{ '%02d' | format(rust_cache_prune_schedule_hour | int) }}:00:00 # Persistent so a box that was off over the scheduled window still prunes. Persistent=true RandomizedDelaySec=30m diff --git a/ansible/roles/developer-rust/templates/io.hyperi.rust-cache-prune-guard.plist.j2 b/ansible/roles/developer-rust/templates/io.hyperi.rust-cache-prune-guard.plist.j2 new file mode 100644 index 0000000..e985b41 --- /dev/null +++ b/ansible/roles/developer-rust/templates/io.hyperi.rust-cache-prune-guard.plist.j2 @@ -0,0 +1,43 @@ + + + + + Label + {{ rust_cache_prune_guard_label }} + + ProgramArguments + + /usr/local/bin/hyperi-rust-cache-prune + --yes + --max-size + {{ rust_cache_build_dir_max }} + --max-age-days + {{ rust_cache_max_age_days }} + --if-free-below + {{ rust_cache_prune_free_floor }} + + + + StartInterval + {{ rust_cache_prune_guard_interval_seconds | int }} + + + RunAtLoad + + + + ProcessType + Background + LowPriorityIO + + Nice + 19 + + StandardOutPath + {{ user_home }}/Library/Logs/hyperi-rust-cache-prune.log + StandardErrorPath + {{ user_home }}/Library/Logs/hyperi-rust-cache-prune.log + + diff --git a/ansible/roles/developer-rust/templates/io.hyperi.rust-cache-prune.plist.j2 b/ansible/roles/developer-rust/templates/io.hyperi.rust-cache-prune.plist.j2 index e476ae2..70f8f9b 100644 --- a/ansible/roles/developer-rust/templates/io.hyperi.rust-cache-prune.plist.j2 +++ b/ansible/roles/developer-rust/templates/io.hyperi.rust-cache-prune.plist.j2 @@ -1,4 +1,4 @@ -{% set weekday_num = {'Sun': 0, 'Mon': 1, 'Tue': 2, 'Wed': 3, 'Thu': 4, 'Fri': 5, 'Sat': 6}[rust_cache_prune_schedule_weekday] %} +{% set weekday_num = {'Sun': 0, 'Mon': 1, 'Tue': 2, 'Wed': 3, 'Thu': 4, 'Fri': 5, 'Sat': 6}.get(rust_cache_prune_schedule_weekday) %} @@ -17,11 +17,13 @@ + what a laptop closed overnight needs. Omitting Weekday means daily. --> StartCalendarInterval +{% if weekday_num is not none %} Weekday {{ weekday_num }} +{% endif %} Hour {{ rust_cache_prune_schedule_hour | int }} Minute diff --git a/tools/tests/test_rust_cache_prune.py b/tools/tests/test_rust_cache_prune.py index dd99c3c..5d1f642 100644 --- a/tools/tests/test_rust_cache_prune.py +++ b/tools/tests/test_rust_cache_prune.py @@ -6,6 +6,7 @@ import importlib.util from pathlib import Path +from types import SimpleNamespace import pytest @@ -105,3 +106,112 @@ def test_parse_size_accepts_auto_and_sizes(prune): assert prune.parse_size_or_auto("auto") == "auto" assert prune.parse_size_or_auto(" AUTO ") == "auto" assert prune.parse_size_or_auto("40G") == prune.parse_size("40G") + + +# --------------------------------------------------------------------------- +# The free-space guard +# --------------------------------------------------------------------------- + + +def fake_usage(total, free): + """Stand in for shutil.disk_usage, which only .total and .free are read from.""" + return SimpleNamespace(total=total, used=total - free, free=free) + + +def test_floor_accepts_a_percentage(prune): + assert prune.parse_size_or_percent("20%") == ("percent", 20.0) + assert prune.parse_size_or_percent(" 7.5 % ") == ("percent", 7.5) + + +def test_floor_accepts_a_byte_count(prune): + assert prune.parse_size_or_percent("80G") == ("bytes", prune.parse_size("80G")) + + +@pytest.mark.parametrize("text", ["0%", "100%", "-5%", "abc", ""]) +def test_floor_rejects_what_cannot_be_a_floor(prune, text): + """0 and 100 are rejected as well as junk: neither can gate anything.""" + with pytest.raises(Exception): + prune.parse_size_or_percent(text) + + +def test_guard_does_nothing_while_there_is_room(prune, tmp_path, monkeypatch): + """The whole point of the guard is being cheap when it has no work. + + Above the floor it must answer from one statvfs and never reach the pool. + """ + monkeypatch.setattr(prune.shutil, "disk_usage", lambda _: fake_usage(1000, 500)) + assert prune.above_free_floor(tmp_path, ("percent", 20.0), prune.Reporter()) is True + + +def test_guard_lets_the_prune_through_when_space_is_short(prune, tmp_path, monkeypatch): + monkeypatch.setattr(prune.shutil, "disk_usage", lambda _: fake_usage(1000, 100)) + assert prune.above_free_floor(tmp_path, ("percent", 20.0), prune.Reporter()) is False + + +def test_guard_takes_a_byte_floor_as_well(prune, tmp_path, monkeypatch): + monkeypatch.setattr(prune.shutil, "disk_usage", lambda _: fake_usage(1000, 100)) + assert prune.above_free_floor(tmp_path, ("bytes", 50), prune.Reporter()) is True + assert prune.above_free_floor(tmp_path, ("bytes", 150), prune.Reporter()) is False + + +def make_pool(tmp_path, sizes): + """Build a pool at cargo's shard depth, one leaf per given byte size.""" + pool = tmp_path / "pool" + for index, size in enumerate(sizes): + leaf = pool / f"{index:02x}" / f"hash{index}" + leaf.mkdir(parents=True) + (leaf / "artefact.bin").write_bytes(b"\0" * size) + return pool + + +def test_the_reported_total_counts_only_evictions_that_worked(prune, tmp_path, monkeypatch): + """A failed rmtree must leave its bytes in the total. + + Crediting them anyway reports the pool back under its ceiling while it is + still over, which is the one thing the guard's verdict rests on. + """ + pool = make_pool(tmp_path, [200_000, 200_000, 200_000]) + + def refuse(_path): + raise OSError("permission denied") + + monkeypatch.setattr(prune.shutil, "rmtree", refuse) + rep = prune.Reporter() + remaining = prune_pool_at(prune, pool, rep, max_size=1) + + assert rep.freed == 0 + assert remaining > 1, "a pool whose evictions all failed is still over its ceiling" + assert rep.warnings + + +def test_the_reported_total_drops_when_evictions_succeed(prune, tmp_path): + pool = make_pool(tmp_path, [200_000, 200_000, 200_000]) + rep = prune.Reporter() + before = prune_pool_at(prune, pool, rep, max_size=10**9) + + rep_two = prune.Reporter() + after = prune_pool_at(prune, pool, rep_two, max_size=1) + + assert after < before + assert rep_two.freed > 0 + + +def prune_pool_at(prune, pool, rep, *, max_size): + """prune_pool with the age pass held off, so only the size pass is in play.""" + return prune.prune_pool(pool, False, rep, max_size=max_size, max_age_days=10**6) + + +def test_an_unmeasurable_filesystem_prunes_rather_than_skipping(prune, tmp_path, monkeypatch): + """Unknown free space must not be read as plenty. + + Treating a failed statvfs as room to spare would turn a broken probe into a + cache nothing ever bounds again. + """ + + def explode(_): + raise OSError("no") + + monkeypatch.setattr(prune.shutil, "disk_usage", explode) + rep = prune.Reporter() + assert prune.above_free_floor(tmp_path, ("percent", 20.0), rep) is False + assert rep.warnings