diff --git a/ansible/roles/developer-rust/README.md b/ansible/roles/developer-rust/README.md
index fdefde7..48673f7 100644
--- a/ansible/roles/developer-rust/README.md
+++ b/ansible/roles/developer-rust/README.md
@@ -70,6 +70,8 @@ silently stops being cached while `sccache` is still installed and configured.
`hyperi-rust-cache-prune` reports it rather than showing an empty ceiling, and
`sccache --stop-server` clears it.
+## Bounding the build-artefact pool
+
**Build artefacts have no upstream cap at all.** Cargo does not track them
(rust-lang/cargo#13136), so nothing reclaims a `target/` ever, and one per repo
across a tree full of them is what actually fills a disk.
@@ -82,7 +84,7 @@ binaries, so `cargo run`, IDEs and anything globbing for a built artefact are
unaffected.
`hyperi-rust-cache-prune` then bounds that pool on a schedule -- a systemd timer
-on Linux, a launchd agent on macOS, weekly and at idle IO priority. It drops
+on Linux, a launchd agent on macOS, daily and at idle IO priority. It drops
workspaces not built for `rust_cache_max_age_days`, then evicts
least-recently-built ones until the pool is under `rust_cache_build_dir_max`.
It touches no project `target/`, and reports the self-capping caches without
@@ -94,14 +96,18 @@ derived from it chases itself downward. That puts a 692G build box at 115G and a
256G laptop at the floor, so one default suits both. Set
`rust_cache_build_dir_max` to an explicit size to override it.
-**The ceiling is a weekly reset, not a live cap.** Nothing enforces it between
-runs, so a busy build box spends most of the week above it -- one added 37G in
-two days and another 37G in a single afternoon. That is the design working, not
-a prune that failed. It only matters if the peak, not the floor, would fill the
-disk; check free space before reaching for a shorter schedule, and change
-`rust_cache_prune_schedule_weekday` if the peak is genuinely too high.
+**The ceiling binds while the tool runs, not between runs.** A pool that grows
+faster than the schedule spends the gap above it, so the prune runs daily and an
+hourly guard backs it up -- one statvfs while the disk has room, a prune to the
+same ceiling once free space falls below `rust_cache_prune_free_floor` (20%).
+
+A guard run that finds the pool already under its ceiling stops and says so. The
+space went somewhere the prune does not own, and naming that is more use than
+evicting artefacts that were not the cause. Set
+`rust_cache_prune_guard_enabled: false` to drop the guard, or
+`rust_cache_prune_schedule_weekday` to go back to weekly.
-sccache and ccache keep fixed ceilings; both enforce their own.
+## Which sccache builds actually use
**sccache comes from upstream's release, not the distro.** Ubuntu ships 0.13.0
against an upstream on 0.17.x, and a wrapper four versions behind is what a
diff --git a/ansible/roles/developer-rust/defaults/main.yml b/ansible/roles/developer-rust/defaults/main.yml
index fe73f33..2772c0f 100644
--- a/ansible/roles/developer-rust/defaults/main.yml
+++ b/ansible/roles/developer-rust/defaults/main.yml
@@ -37,12 +37,29 @@ rust_cache_build_dir_max: "auto"
rust_cache_max_age_days: 14
# Off-hours, so a prune is unlikely to land mid-build.
-rust_cache_prune_schedule_weekday: "Sun"
+#
+# Daily rather than weekly: the ceiling binds only while the tool runs, so the
+# gap between runs is time the pool spends unbounded. Set a weekday for weekly.
+rust_cache_prune_schedule_weekday: ""
rust_cache_prune_schedule_hour: 3
+# Backstop for growth that outruns the daily prune. One statvfs an hour, and a
+# prune only below the floor.
+#
+# A percentage because one size cannot suit a laptop and a build box.
+#
+# The guard prunes to the ordinary ceiling and no further. A pool already inside
+# the ceiling means the space went elsewhere, and the run reports that.
+rust_cache_prune_guard_enabled: true
+rust_cache_prune_free_floor: "20%"
+rust_cache_prune_guard_schedule: "hourly"
+rust_cache_prune_guard_interval_seconds: 3600
+
# macOS launchd only; systemd takes its unit name from the file.
rust_cache_prune_label: "io.hyperi.rust-cache-prune"
rust_cache_prune_plist: "{{ user_home }}/Library/LaunchAgents/{{ rust_cache_prune_label }}.plist"
+rust_cache_prune_guard_label: "io.hyperi.rust-cache-prune-guard"
+rust_cache_prune_guard_plist: "{{ user_home }}/Library/LaunchAgents/{{ rust_cache_prune_guard_label }}.plist"
# Start the sccache server from a systemd user unit rather than leaving it to
# whichever process compiles first, which decides the server's cgroup and
diff --git a/ansible/roles/developer-rust/files/hyperi-rust-cache-prune b/ansible/roles/developer-rust/files/hyperi-rust-cache-prune
index 7c06c01..89dc1ae 100644
--- a/ansible/roles/developer-rust/files/hyperi-rust-cache-prune
+++ b/ansible/roles/developer-rust/files/hyperi-rust-cache-prune
@@ -6,6 +6,7 @@ Run it yourself:
hyperi-rust-cache-prune # report, then ask before deleting
hyperi-rust-cache-prune --check # report only, delete nothing
hyperi-rust-cache-prune --yes # no prompt (how the timer invokes it)
+ hyperi-rust-cache-prune --if-free-below 20% # only when the disk is tight
Why this exists
---------------
@@ -33,6 +34,25 @@ a workspace you compile against daily and one you only link against look the
same. Age is measured from the newest thing cargo wrote, which is the closest
honest proxy.
+When it runs at all
+-------------------
+The ceiling is enforced when the tool runs and at no other moment, so a pool
+that grows faster than the schedule spends the gap unbounded. Shortening the
+schedule alone trades that for a walk of the whole pool every time.
+
+`--if-free-below` is the answer to that. It costs one statvfs, so the tool can
+be scheduled often enough to matter and still do nothing almost every time: if
+the filesystem holding the pool has more free space than the floor, it exits
+before walking anything.
+
+The floor takes a percentage as well as a byte count, because the same number
+cannot be right on a 256G laptop and a 692G build box.
+
+A guarded run that finds the pool already inside its ceiling does not go
+looking for more to delete. It says the pool is not where the space went and
+stops. Reclaiming artefacts that were not the problem would be the wrong
+answer to a disk filled by something else.
+
What it does NOT touch
----------------------
Project `target/` directories, anywhere. `build.build-dir` moves intermediates
@@ -73,6 +93,7 @@ AUTO_FLOOR = "40G"
POOL_DEPTH = 2
_SIZE = re.compile(r"^\s*(\d+(?:\.\d+)?)\s*([KMGT]?)i?B?\s*$", re.IGNORECASE)
+_PERCENT = re.compile(r"^\s*(\d+(?:\.\d+)?)\s*%\s*$")
_MULTIPLIER = {"": 1, "K": 1024, "M": 1024**2, "G": 1024**3, "T": 1024**4}
@@ -122,6 +143,49 @@ def parse_size_or_auto(text: str):
return parse_size(text)
+def parse_size_or_percent(text: str):
+ """A byte count, or a share of the filesystem written as a percentage.
+
+ Returned as a (kind, value) pair rather than a number because a percentage
+ cannot be resolved until the pool path names a filesystem to take it of.
+ """
+ match = _PERCENT.match(text)
+ if match:
+ share = float(match.group(1))
+ if not 0 < share < 100:
+ raise argparse.ArgumentTypeError(
+ f"not a usable percentage: {text!r} -- give something above 0 and below 100"
+ )
+ return ("percent", share)
+ return ("bytes", parse_size(text))
+
+
+def above_free_floor(pool: Path, floor, rep: Reporter) -> bool:
+ """Has the filesystem holding the pool got more free space than the floor?
+
+ One statvfs, called before anything walks the pool, so a guarded run that
+ has nothing to do costs nothing. `free`, not `total - used`: reserved
+ blocks are not ours to spend and counting them would let the guard sit
+ quiet while writes are already failing.
+ """
+ kind, value = floor
+ try:
+ usage = shutil.disk_usage(pool)
+ except OSError as exc:
+ # Unmeasurable is not the same as fine. Skipping the prune here would
+ # turn a broken statvfs into a silently unbounded cache.
+ rep.warn(f"could not measure free space at {pool} ({exc}) -- pruning anyway")
+ return False
+
+ threshold = int(usage.total * value / 100) if kind == "percent" else value
+ where = f"{human(usage.free)} free of {human(usage.total)}"
+ if usage.free > threshold:
+ rep.info(f"{where}, above the {human(threshold)} floor -- nothing to do")
+ return True
+ rep.info(f"{where}, below the {human(threshold)} floor")
+ return False
+
+
def resolve_max_size(value, pool: Path, rep: Reporter) -> int:
"""Turn `auto` into bytes for the filesystem that actually holds the pool."""
if value != "auto":
@@ -277,27 +341,38 @@ def find_workspaces(pool: Path) -> list[Workspace]:
return [Workspace(leaf) for leaf in leaves]
-def remove(workspace: Workspace, reason: str, dry_run: bool, rep: Reporter) -> None:
+def remove(workspace: Workspace, reason: str, dry_run: bool, rep: Reporter) -> bool:
+ """Drop one workspace. False means the bytes are still on disk.
+
+ The caller has to know, because counting a failed removal against the pool
+ total reports a cache back under its ceiling while it is still over.
+ """
if dry_run:
rep.info(f"[check] would drop {workspace.path.name} ({human(workspace.size)}, {reason})")
rep.freed += workspace.size
- return
+ return True
try:
shutil.rmtree(workspace.path)
except OSError as exc:
rep.warn(f"could not remove {workspace.path}: {exc}")
- return
+ return False
rep.change(f"dropped {workspace.path.name} ({human(workspace.size)}, {reason})")
rep.freed += workspace.size
+ return True
def prune_pool(
pool: Path, dry_run: bool, rep: Reporter, *, max_size: int, max_age_days: int
-) -> None:
+) -> int:
+ """Bound the pool, and return what it still occupies.
+
+ Callers want that rather than the freed count, which reads the same whether
+ there was nothing to free or every eviction failed.
+ """
workspaces = find_workspaces(pool)
if not workspaces:
rep.info(f"{pool}: empty, nothing to prune")
- return
+ return 0
total = sum(item.size for item in workspaces)
rep.info(
@@ -306,21 +381,21 @@ def prune_pool(
stale = [item for item in workspaces if item.age_days > max_age_days]
for item in stale:
- remove(item, f"idle {item.age_days:.0f}d", dry_run, rep)
+ if remove(item, f"idle {item.age_days:.0f}d", dry_run, rep):
+ total -= item.size
remaining = [item for item in workspaces if item not in stale]
- total -= sum(item.size for item in stale)
if total <= max_size:
rep.info(f"under the ceiling at {human(total)}")
- return
+ return total
# Oldest first: the least-recently-built workspace is the cheapest to lose.
for item in sorted(remaining, key=lambda entry: entry.mtime):
if total <= max_size:
break
- remove(item, f"over ceiling, {human(total)} used", dry_run, rep)
- total -= item.size
+ if remove(item, f"over ceiling, {human(total)} used", dry_run, rep):
+ total -= item.size
if total > max_size:
rep.warn(f"still {human(total)} after pruning everything eligible")
@@ -330,6 +405,8 @@ def prune_pool(
if not dry_run:
drop_empty_shards(pool)
+ return total
+
def drop_empty_shards(pool: Path) -> None:
"""Remove shard directories emptied by eviction, so the pool does not silt up."""
@@ -433,6 +510,16 @@ def main() -> int:
metavar="DAYS",
help=f"drop workspaces not built in this long (default: {DEFAULT_MAX_AGE_DAYS})",
)
+ parser.add_argument(
+ "--if-free-below",
+ type=parse_size_or_percent,
+ default=None,
+ metavar="SIZE|PERCENT",
+ help=(
+ "do nothing unless the filesystem holding the pool has less than this "
+ "free (e.g. 20%% or 80G) -- how the hourly guard invokes it"
+ ),
+ )
parser.add_argument(
"--pool",
type=Path,
@@ -470,6 +557,12 @@ def main() -> int:
print(f"hyperi-rust-cache-prune: {pool}")
+ # Before the walk and before the prompt: a guarded run with room to spare
+ # must cost one statvfs and must never ask the user anything.
+ guarded = args.if_free_below is not None
+ if guarded and above_free_floor(pool, args.if_free_below, rep):
+ return 0
+
max_size = resolve_max_size(args.max_size, pool, rep)
if not args.check and not args.yes:
@@ -485,7 +578,7 @@ def main() -> int:
return 0
rep.step("Pooled build artefacts")
- prune_pool(
+ remaining = prune_pool(
pool,
args.check,
rep,
@@ -493,6 +586,17 @@ def main() -> int:
max_age_days=args.max_age_days,
)
+ # A guarded run ending inside the ceiling means the pool is not what filled
+ # the disk. Tested on what the pool still holds, not on what was freed -- a
+ # failed eviction also frees nothing.
+ if guarded and remaining <= max_size:
+ rep.warn(
+ "below the free floor with the pool already inside its ceiling -- "
+ "nothing here to reclaim. Whatever filled the disk is outside what "
+ "this tool bounds, and the caches reported below are the only other "
+ "ones it can see."
+ )
+
rep.step("Self-capping caches (reported, not pruned)")
try:
report_self_capping_caches(rep)
diff --git a/ansible/roles/developer-rust/tasks/cache.yml b/ansible/roles/developer-rust/tasks/cache.yml
index 59dafcd..863aa56 100644
--- a/ansible/roles/developer-rust/tasks/cache.yml
+++ b/ansible/roles/developer-rust/tasks/cache.yml
@@ -6,6 +6,10 @@
# hyperi-rust-cache-prune bounds that pool on a schedule, because cargo does
# not track build artefacts and nothing upstream ever reclaims them.
#
+# Two schedules, because a ceiling only binds while the tool runs. The daily
+# prune is the routine path. The guard runs hourly, costs one statvfs when the
+# disk has room, and prunes when free space drops below the floor.
+#
# The whole file is optional: a machine with no cap still builds, so nothing
# here may fail the run.
@@ -43,6 +47,26 @@
mode: '0644'
become: true
+ - name: Install the rust cache prune guard service
+ ansible.builtin.template:
+ src: hyperi-rust-cache-prune-guard.service.j2
+ dest: /etc/systemd/system/hyperi-rust-cache-prune-guard.service
+ owner: root
+ group: root
+ mode: '0644'
+ become: true
+ when: rust_cache_prune_guard_enabled | bool
+
+ - name: Install the rust cache prune guard timer
+ ansible.builtin.template:
+ src: hyperi-rust-cache-prune-guard.timer.j2
+ dest: /etc/systemd/system/hyperi-rust-cache-prune-guard.timer
+ owner: root
+ group: root
+ mode: '0644'
+ become: true
+ when: rust_cache_prune_guard_enabled | bool
+
# Check mode does not write the unit files above, so systemd has nothing to
# enable and the task would fail the run rather than preview it.
- name: Enable and start the rust cache prune timer
@@ -54,6 +78,41 @@
become: true
when: not ansible_check_mode
+ - name: Enable and start the rust cache prune guard timer
+ ansible.builtin.systemd:
+ name: hyperi-rust-cache-prune-guard.timer
+ enabled: true
+ state: started
+ daemon_reload: true
+ become: true
+ when:
+ - not ansible_check_mode
+ - rust_cache_prune_guard_enabled | bool
+
+ # Turning the guard off has to remove it, not just stop writing it. A timer
+ # left enabled from a previous run would keep firing against a flag that
+ # says it is disabled.
+ - name: Stop the rust cache prune guard timer when disabled
+ ansible.builtin.systemd:
+ name: hyperi-rust-cache-prune-guard.timer
+ enabled: false
+ state: stopped
+ become: true
+ failed_when: false
+ when:
+ - not ansible_check_mode
+ - not rust_cache_prune_guard_enabled | bool
+
+ - name: Remove the rust cache prune guard units when disabled
+ ansible.builtin.file:
+ path: "/etc/systemd/system/{{ item }}"
+ state: absent
+ become: true
+ loop:
+ - hyperi-rust-cache-prune-guard.timer
+ - hyperi-rust-cache-prune-guard.service
+ when: not rust_cache_prune_guard_enabled | bool
+
- name: Schedule the prune (macOS)
when: ansible_facts['distribution'] == 'MacOSX'
block:
@@ -66,57 +125,38 @@
mode: '0755'
become: false
- - name: Install the rust cache prune launch agent
- ansible.builtin.template:
- src: io.hyperi.rust-cache-prune.plist.j2
- dest: "{{ rust_cache_prune_plist }}"
- mode: '0644'
- become: false
- register: developer_rust_prune_plist
+ # One include per agent rather than a loop over the tasks: each inclusion
+ # gets its own register scope, so the bootout/bootstrap decision reads the
+ # plist result for the agent in hand.
+ - name: Install and load the rust cache prune launch agents
+ ansible.builtin.include_tasks: launchd_agent.yml
+ loop: "{{ developer_rust_launchd_agents }}"
+ loop_control:
+ loop_var: launchd_agent
+ label: "{{ launchd_agent.label }}"
+ vars:
+ developer_rust_launchd_agents: >-
+ {{ [{'label': rust_cache_prune_label,
+ 'plist': rust_cache_prune_plist,
+ 'template': 'io.hyperi.rust-cache-prune.plist.j2'}]
+ + ([{'label': rust_cache_prune_guard_label,
+ 'plist': rust_cache_prune_guard_plist,
+ 'template': 'io.hyperi.rust-cache-prune-guard.plist.j2'}]
+ if rust_cache_prune_guard_enabled | bool else []) }}
- # `launchctl print` is the only reliable "is it loaded" probe; `list` greps
- # a label that may match a different agent's prefix.
- - name: Check whether the launch agent is already loaded
+ - name: Unload the rust cache prune guard launch agent when disabled
ansible.builtin.command:
- cmd: "launchctl print gui/{{ ansible_facts['user_uid'] }}/{{ rust_cache_prune_label }}"
- become: false
- register: developer_rust_prune_loaded
- changed_when: false
- failed_when: false
-
- # bootout first: bootstrap on an already-loaded label is an error, and the
- # plist may have changed underneath it.
- - name: Unload the previous launch agent
- ansible.builtin.command:
- cmd: "launchctl bootout gui/{{ ansible_facts['user_uid'] }}/{{ rust_cache_prune_label }}"
+ cmd: "launchctl bootout gui/{{ ansible_facts['user_uid'] }}/{{ rust_cache_prune_guard_label }}"
become: false
changed_when: false
failed_when: false
when:
- not ansible_check_mode
- - developer_rust_prune_loaded.rc == 0
- - developer_rust_prune_plist.changed
+ - not rust_cache_prune_guard_enabled | bool
- - name: Load the rust cache prune launch agent
- ansible.builtin.command:
- cmd: >-
- launchctl bootstrap gui/{{ ansible_facts['user_uid'] }}
- {{ rust_cache_prune_plist }}
+ - name: Remove the rust cache prune guard launch agent when disabled
+ ansible.builtin.file:
+ path: "{{ rust_cache_prune_guard_plist }}"
+ state: absent
become: false
- register: developer_rust_prune_bootstrap
- changed_when: developer_rust_prune_bootstrap.rc == 0
- failed_when: false
- when:
- - not ansible_check_mode
- - developer_rust_prune_loaded.rc != 0 or developer_rust_prune_plist.changed
-
- - name: Warn when the launch agent did not load
- ansible.builtin.set_fact:
- deploy_warnings: >-
- {{ deploy_warnings | default([])
- + ['Load the rust cache prune launch agent: '
- ~ (developer_rust_prune_bootstrap.stderr | default('unknown error'))] }}
- when:
- - not ansible_check_mode
- - developer_rust_prune_bootstrap is defined
- - developer_rust_prune_bootstrap.rc | default(0) != 0
+ when: not rust_cache_prune_guard_enabled | bool
diff --git a/ansible/roles/developer-rust/tasks/launchd_agent.yml b/ansible/roles/developer-rust/tasks/launchd_agent.yml
new file mode 100644
index 0000000..582ec68
--- /dev/null
+++ b/ansible/roles/developer-rust/tasks/launchd_agent.yml
@@ -0,0 +1,60 @@
+---
+# Load one launchd agent idempotently. Included once per agent from cache.yml,
+# which passes `launchd_agent` as {label, plist, template}.
+#
+# launchd has no reload verb, so a changed plist has to be booted out first.
+
+- name: "Install launch agent {{ launchd_agent.label }}"
+ ansible.builtin.template:
+ src: "{{ launchd_agent.template }}"
+ dest: "{{ launchd_agent.plist }}"
+ mode: '0644'
+ become: false
+ register: developer_rust_agent_plist
+
+# `launchctl print` is the only reliable "is it loaded" probe; `list` greps
+# a label that may match a different agent's prefix.
+- name: "Check the loaded state of launch agent {{ launchd_agent.label }}"
+ ansible.builtin.command:
+ cmd: "launchctl print gui/{{ ansible_facts['user_uid'] }}/{{ launchd_agent.label }}"
+ become: false
+ register: developer_rust_agent_loaded
+ changed_when: false
+ failed_when: false
+
+# bootout first: bootstrap on an already-loaded label is an error, and the
+# plist may have changed underneath it.
+- name: "Unload the previous launch agent {{ launchd_agent.label }}"
+ ansible.builtin.command:
+ cmd: "launchctl bootout gui/{{ ansible_facts['user_uid'] }}/{{ launchd_agent.label }}"
+ become: false
+ changed_when: false
+ failed_when: false
+ when:
+ - not ansible_check_mode
+ - developer_rust_agent_loaded.rc == 0
+ - developer_rust_agent_plist.changed
+
+- name: "Load launch agent {{ launchd_agent.label }}"
+ ansible.builtin.command:
+ cmd: >-
+ launchctl bootstrap gui/{{ ansible_facts['user_uid'] }}
+ {{ launchd_agent.plist }}
+ become: false
+ register: developer_rust_agent_bootstrap
+ changed_when: developer_rust_agent_bootstrap.rc == 0
+ failed_when: false
+ when:
+ - not ansible_check_mode
+ - developer_rust_agent_loaded.rc != 0 or developer_rust_agent_plist.changed
+
+- name: "Warn about a launch agent that did not load: {{ launchd_agent.label }}"
+ ansible.builtin.set_fact:
+ deploy_warnings: >-
+ {{ deploy_warnings | default([])
+ + ['Load the ' ~ launchd_agent.label ~ ' launch agent: '
+ ~ (developer_rust_agent_bootstrap.stderr | default('unknown error'))] }}
+ when:
+ - not ansible_check_mode
+ - developer_rust_agent_bootstrap is defined
+ - developer_rust_agent_bootstrap.rc | default(0) != 0
diff --git a/ansible/roles/developer-rust/templates/hyperi-rust-cache-prune-guard.service.j2 b/ansible/roles/developer-rust/templates/hyperi-rust-cache-prune-guard.service.j2
new file mode 100644
index 0000000..67a662e
--- /dev/null
+++ b/ansible/roles/developer-rust/templates/hyperi-rust-cache-prune-guard.service.j2
@@ -0,0 +1,13 @@
+[Unit]
+Description=Prune the pooled Rust build cache when free space runs short
+
+[Service]
+Type=oneshot
+User={{ actual_user }}
+# Same prune with the same ceiling, gated on free space. Above the floor it
+# exits after one statvfs without walking the pool.
+ExecStart=/usr/local/bin/hyperi-rust-cache-prune --yes --max-size {{ rust_cache_build_dir_max }} --max-age-days {{ rust_cache_max_age_days }} --if-free-below {{ rust_cache_prune_free_floor }}
+# Walking a large pool is IO-bound; it must never compete with an active build.
+Nice=19
+IOSchedulingClass=idle
+TimeoutStartSec=1800
diff --git a/ansible/roles/developer-rust/templates/hyperi-rust-cache-prune-guard.timer.j2 b/ansible/roles/developer-rust/templates/hyperi-rust-cache-prune-guard.timer.j2
new file mode 100644
index 0000000..9baa0af
--- /dev/null
+++ b/ansible/roles/developer-rust/templates/hyperi-rust-cache-prune-guard.timer.j2
@@ -0,0 +1,12 @@
+[Unit]
+Description=Watch free space for the pooled Rust build cache
+
+[Timer]
+OnCalendar={{ rust_cache_prune_guard_schedule }}
+# Persistent so a box that was off catches up once, not once per missed hour.
+Persistent=true
+# Short jitter only. A guard that fires late is a guard that fired too late.
+RandomizedDelaySec=5m
+
+[Install]
+WantedBy=timers.target
diff --git a/ansible/roles/developer-rust/templates/hyperi-rust-cache-prune.timer.j2 b/ansible/roles/developer-rust/templates/hyperi-rust-cache-prune.timer.j2
index 43c2283..83f2ba3 100644
--- a/ansible/roles/developer-rust/templates/hyperi-rust-cache-prune.timer.j2
+++ b/ansible/roles/developer-rust/templates/hyperi-rust-cache-prune.timer.j2
@@ -1,8 +1,8 @@
[Unit]
-Description=Bound the pooled Rust build cache weekly
+Description=Bound the pooled Rust build cache
[Timer]
-OnCalendar={{ rust_cache_prune_schedule_weekday }} *-*-* {{ '%02d' | format(rust_cache_prune_schedule_hour | int) }}:00:00
+OnCalendar={{ (rust_cache_prune_schedule_weekday ~ ' ') if rust_cache_prune_schedule_weekday else '' }}*-*-* {{ '%02d' | format(rust_cache_prune_schedule_hour | int) }}:00:00
# Persistent so a box that was off over the scheduled window still prunes.
Persistent=true
RandomizedDelaySec=30m
diff --git a/ansible/roles/developer-rust/templates/io.hyperi.rust-cache-prune-guard.plist.j2 b/ansible/roles/developer-rust/templates/io.hyperi.rust-cache-prune-guard.plist.j2
new file mode 100644
index 0000000..e985b41
--- /dev/null
+++ b/ansible/roles/developer-rust/templates/io.hyperi.rust-cache-prune-guard.plist.j2
@@ -0,0 +1,43 @@
+
+
+
+
+ Label
+ {{ rust_cache_prune_guard_label }}
+
+ ProgramArguments
+
+ /usr/local/bin/hyperi-rust-cache-prune
+ --yes
+ --max-size
+ {{ rust_cache_build_dir_max }}
+ --max-age-days
+ {{ rust_cache_max_age_days }}
+ --if-free-below
+ {{ rust_cache_prune_free_floor }}
+
+
+
+ StartInterval
+ {{ rust_cache_prune_guard_interval_seconds | int }}
+
+
+ RunAtLoad
+
+
+
+ ProcessType
+ Background
+ LowPriorityIO
+
+ Nice
+ 19
+
+ StandardOutPath
+ {{ user_home }}/Library/Logs/hyperi-rust-cache-prune.log
+ StandardErrorPath
+ {{ user_home }}/Library/Logs/hyperi-rust-cache-prune.log
+
+
diff --git a/ansible/roles/developer-rust/templates/io.hyperi.rust-cache-prune.plist.j2 b/ansible/roles/developer-rust/templates/io.hyperi.rust-cache-prune.plist.j2
index e476ae2..70f8f9b 100644
--- a/ansible/roles/developer-rust/templates/io.hyperi.rust-cache-prune.plist.j2
+++ b/ansible/roles/developer-rust/templates/io.hyperi.rust-cache-prune.plist.j2
@@ -1,4 +1,4 @@
-{% set weekday_num = {'Sun': 0, 'Mon': 1, 'Tue': 2, 'Wed': 3, 'Thu': 4, 'Fri': 5, 'Sat': 6}[rust_cache_prune_schedule_weekday] %}
+{% set weekday_num = {'Sun': 0, 'Mon': 1, 'Tue': 2, 'Wed': 3, 'Thu': 4, 'Fri': 5, 'Sat': 6}.get(rust_cache_prune_schedule_weekday) %}
@@ -17,11 +17,13 @@
+ what a laptop closed overnight needs. Omitting Weekday means daily. -->
StartCalendarInterval
+{% if weekday_num is not none %}
Weekday
{{ weekday_num }}
+{% endif %}
Hour
{{ rust_cache_prune_schedule_hour | int }}
Minute
diff --git a/tools/tests/test_rust_cache_prune.py b/tools/tests/test_rust_cache_prune.py
index dd99c3c..5d1f642 100644
--- a/tools/tests/test_rust_cache_prune.py
+++ b/tools/tests/test_rust_cache_prune.py
@@ -6,6 +6,7 @@
import importlib.util
from pathlib import Path
+from types import SimpleNamespace
import pytest
@@ -105,3 +106,112 @@ def test_parse_size_accepts_auto_and_sizes(prune):
assert prune.parse_size_or_auto("auto") == "auto"
assert prune.parse_size_or_auto(" AUTO ") == "auto"
assert prune.parse_size_or_auto("40G") == prune.parse_size("40G")
+
+
+# ---------------------------------------------------------------------------
+# The free-space guard
+# ---------------------------------------------------------------------------
+
+
+def fake_usage(total, free):
+ """Stand in for shutil.disk_usage, which only .total and .free are read from."""
+ return SimpleNamespace(total=total, used=total - free, free=free)
+
+
+def test_floor_accepts_a_percentage(prune):
+ assert prune.parse_size_or_percent("20%") == ("percent", 20.0)
+ assert prune.parse_size_or_percent(" 7.5 % ") == ("percent", 7.5)
+
+
+def test_floor_accepts_a_byte_count(prune):
+ assert prune.parse_size_or_percent("80G") == ("bytes", prune.parse_size("80G"))
+
+
+@pytest.mark.parametrize("text", ["0%", "100%", "-5%", "abc", ""])
+def test_floor_rejects_what_cannot_be_a_floor(prune, text):
+ """0 and 100 are rejected as well as junk: neither can gate anything."""
+ with pytest.raises(Exception):
+ prune.parse_size_or_percent(text)
+
+
+def test_guard_does_nothing_while_there_is_room(prune, tmp_path, monkeypatch):
+ """The whole point of the guard is being cheap when it has no work.
+
+ Above the floor it must answer from one statvfs and never reach the pool.
+ """
+ monkeypatch.setattr(prune.shutil, "disk_usage", lambda _: fake_usage(1000, 500))
+ assert prune.above_free_floor(tmp_path, ("percent", 20.0), prune.Reporter()) is True
+
+
+def test_guard_lets_the_prune_through_when_space_is_short(prune, tmp_path, monkeypatch):
+ monkeypatch.setattr(prune.shutil, "disk_usage", lambda _: fake_usage(1000, 100))
+ assert prune.above_free_floor(tmp_path, ("percent", 20.0), prune.Reporter()) is False
+
+
+def test_guard_takes_a_byte_floor_as_well(prune, tmp_path, monkeypatch):
+ monkeypatch.setattr(prune.shutil, "disk_usage", lambda _: fake_usage(1000, 100))
+ assert prune.above_free_floor(tmp_path, ("bytes", 50), prune.Reporter()) is True
+ assert prune.above_free_floor(tmp_path, ("bytes", 150), prune.Reporter()) is False
+
+
+def make_pool(tmp_path, sizes):
+ """Build a pool at cargo's shard depth, one leaf per given byte size."""
+ pool = tmp_path / "pool"
+ for index, size in enumerate(sizes):
+ leaf = pool / f"{index:02x}" / f"hash{index}"
+ leaf.mkdir(parents=True)
+ (leaf / "artefact.bin").write_bytes(b"\0" * size)
+ return pool
+
+
+def test_the_reported_total_counts_only_evictions_that_worked(prune, tmp_path, monkeypatch):
+ """A failed rmtree must leave its bytes in the total.
+
+ Crediting them anyway reports the pool back under its ceiling while it is
+ still over, which is the one thing the guard's verdict rests on.
+ """
+ pool = make_pool(tmp_path, [200_000, 200_000, 200_000])
+
+ def refuse(_path):
+ raise OSError("permission denied")
+
+ monkeypatch.setattr(prune.shutil, "rmtree", refuse)
+ rep = prune.Reporter()
+ remaining = prune_pool_at(prune, pool, rep, max_size=1)
+
+ assert rep.freed == 0
+ assert remaining > 1, "a pool whose evictions all failed is still over its ceiling"
+ assert rep.warnings
+
+
+def test_the_reported_total_drops_when_evictions_succeed(prune, tmp_path):
+ pool = make_pool(tmp_path, [200_000, 200_000, 200_000])
+ rep = prune.Reporter()
+ before = prune_pool_at(prune, pool, rep, max_size=10**9)
+
+ rep_two = prune.Reporter()
+ after = prune_pool_at(prune, pool, rep_two, max_size=1)
+
+ assert after < before
+ assert rep_two.freed > 0
+
+
+def prune_pool_at(prune, pool, rep, *, max_size):
+ """prune_pool with the age pass held off, so only the size pass is in play."""
+ return prune.prune_pool(pool, False, rep, max_size=max_size, max_age_days=10**6)
+
+
+def test_an_unmeasurable_filesystem_prunes_rather_than_skipping(prune, tmp_path, monkeypatch):
+ """Unknown free space must not be read as plenty.
+
+ Treating a failed statvfs as room to spare would turn a broken probe into a
+ cache nothing ever bounds again.
+ """
+
+ def explode(_):
+ raise OSError("no")
+
+ monkeypatch.setattr(prune.shutil, "disk_usage", explode)
+ rep = prune.Reporter()
+ assert prune.above_free_floor(tmp_path, ("percent", 20.0), rep) is False
+ assert rep.warnings