From 560326513aa03404025a82f727fd28e6a9a909d8 Mon Sep 17 00:00:00 2001 From: Naman Rusia Date: Fri, 25 Sep 2026 14:42:32 -0400 Subject: [PATCH] Make complexity estimates hold up under gVisor In production two of the four reference solutions got the wrong class: runs there carry some 150 ms of fixed cost and CPU time comes in 10 ms steps, so the small runs the fit leaned on were mostly noise. A size now counts once the solution's own time is at least the run's fixed cost, short runs are timed twice and the faster kept, and only the largest sizes are fitted. Two Sum gets one more input size so it clears the floor at three sizes. A result faster than expected no longer reads as a match. --- apps/backend/app/runner/complexity.py | 34 ++++++++++++++++--- .../V14__longer_two_sum_analysis.sql | 10 ++++++ apps/web/src/components/ComplexityPanel.tsx | 5 +-- docs/judge-and-sandbox.md | 4 +-- 4 files changed, 44 insertions(+), 9 deletions(-) create mode 100644 apps/backend/migrations/V14__longer_two_sum_analysis.sql diff --git a/apps/backend/app/runner/complexity.py b/apps/backend/app/runner/complexity.py index 83e23d0..3135167 100644 --- a/apps/backend/app/runner/complexity.py +++ b/apps/backend/app/runner/complexity.py @@ -25,10 +25,21 @@ BUDGET_SECONDS = 6.0 PER_RUN_SECONDS = 2.5 ENOUGH_SECONDS = 1.0 - -# Below this a run is mostly noise, not the solution. +# Only runs shorter than this are repeated. Noise is a few tens of +# milliseconds, which does not matter to a longer run, and the time saved +# buys a slow solution one more size. +REPEAT_BELOW_SECONDS = 0.5 + +# A run counts once the solution's own time is at least the fixed cost taken +# off it, so an error in that estimate stays a fraction of what is measured. +# The fixed cost is some 15 ms on a laptop and 150 ms under gVisor, which also +# reports CPU time in 10 ms steps. This is the floor where it is smaller. MIN_MEASURABLE_MS = 20.0 +# Only the largest sizes are fitted: the smaller a run, the larger the share +# of its time that is noise and error in the fixed cost. +FITTED_POINTS = 3 + # Fitted exponent of n -> growth class, split halfway between the powers. # n log n fits at about 1.1 over the sizes used, too close to n to tell # apart, so they share a class. @@ -111,19 +122,32 @@ def analyze(code: str, generator: str, mem_limit_mb: int) -> dict: # and nothing more; left in, it flattens the growth of every run. tiny = generate(max(4, sizes[0] // 32)) startup = min(_run(program, workdir, tiny, PER_RUN_SECONDS, mem_limit_mb)[0] or 0.0 for _ in range(3)) + floor_ms = max(MIN_MEASURABLE_MS, startup * 1000) for n in sizes: remaining = BUDGET_SECONDS - spent if remaining <= 0.1: note = f"Stopped before n = {n}: out of time for this analysis." break - seconds, failure = _run(program, workdir, generate(n), min(PER_RUN_SECONDS, remaining), mem_limit_mb) + stdin = generate(n) + seconds, failure = _run(program, workdir, stdin, min(PER_RUN_SECONDS, remaining), mem_limit_mb) if seconds is None: note = f"Stopped at n = {n}: it {failure}." break spent += seconds + # Scheduling only ever makes a run slower, so a short measurable + # size is run twice and the faster run kept. + if ( + (seconds - startup) * 1000 >= floor_ms + and seconds < REPEAT_BELOW_SECONDS + and BUDGET_SECONDS - spent > seconds + ): + again, _ = _run(program, workdir, stdin, min(PER_RUN_SECONDS, BUDGET_SECONDS - spent), mem_limit_mb) + if again is not None: + spent += again + seconds = min(seconds, again) ms = (seconds - startup) * 1000 - if ms >= MIN_MEASURABLE_MS: + if ms >= floor_ms: points.append((n, ms)) if seconds >= ENOUGH_SECONDS: break @@ -135,7 +159,7 @@ def analyze(code: str, generator: str, mem_limit_mb: int) -> dict: "note": note, } if len(points) >= 2: - result["complexity"], slope = classify(points) + result["complexity"], slope = classify(points[-FITTED_POINTS:]) result["slope"] = round(slope, 2) if len(points) == 2: result["note"] = (note + " " if note else "") + "Only two sizes were measurable, so this is rough." diff --git a/apps/backend/migrations/V14__longer_two_sum_analysis.sql b/apps/backend/migrations/V14__longer_two_sum_analysis.sql new file mode 100644 index 0000000..096d185 --- /dev/null +++ b/apps/backend/migrations/V14__longer_two_sum_analysis.sql @@ -0,0 +1,10 @@ +-- V14: One more input size for Two Sum's complexity analysis +-- +-- A run only counts once the solution's own time is at least the run's fixed +-- cost (app/runner/complexity.py), and at half a million numbers the +-- reference solution clears that at only one or two sizes. A million numbers +-- fit in the problem's memory limit. + +UPDATE problems +SET complexity_generator = replace(complexity_generator, 'range(10)', 'range(11)') +WHERE slug = 'two-sum'; diff --git a/apps/web/src/components/ComplexityPanel.tsx b/apps/web/src/components/ComplexityPanel.tsx index d5a9729..9c15662 100644 --- a/apps/web/src/components/ComplexityPanel.tsx +++ b/apps/web/src/components/ComplexityPanel.tsx @@ -75,15 +75,16 @@ const ComplexityPanel = ({ submission, expected }: { submission: Submission; exp const measured = CLASSES.indexOf(analysis.complexity); const target = expected ? CLASSES.indexOf(expected) : -1; const slower = target >= 0 && measured > target; + const verdict = target < 0 ? null : measured > target ? "Slower than" : measured < target ? "Faster than" : "Matches"; return (
- {expected && ( + {expected && verdict && ( - {slower ? "Slower than" : "Matches"} the expected {LABEL[expected]} + {verdict} the expected {LABEL[expected]} )}
diff --git a/docs/judge-and-sandbox.md b/docs/judge-and-sandbox.md index 8e3f83a..3c8fad0 100644 --- a/docs/judge-and-sandbox.md +++ b/docs/judge-and-sandbox.md @@ -33,8 +33,8 @@ Without `SANDBOX_IMAGE`, the runner refuses to start unless `ALLOW_UNSANDBOXED=1 On request, an accepted run can be rerun on inputs of growing size to estimate how its running time grows. The problem's generator (`complexity_generator`, see [CONTRIBUTING.md](../CONTRIBUTING.md)) makes the inputs, and `app/runner/complexity.py` runs the program on each size in one sandbox container. - It measures CPU time from outside the program. Counting executed lines would miss work done inside built-ins, such as `x in some_list`, and would call a quadratic brute force linear. The program also can't report a time of its own. -- The fixed cost of every run (interpreter start, imports, reading input) is measured on a tiny input and subtracted. -- It fits the times to a power of n and names the class. n log n measures too close to n to tell apart, so the two share a class. +- The fixed cost of every run (interpreter start, imports, reading input) is measured on a tiny input and subtracted. Under gVisor it is some 150 ms, ten times a laptop's, and gVisor reports CPU time in 10 ms steps, so a size only counts once the solution's own time is at least that fixed cost. Short runs are timed twice and the faster kept, since scheduling only ever adds time. +- It fits the largest measurable sizes (`FITTED_POINTS`) to a power of n and names the class. n log n measures too close to n to tell apart, so the two share a class. - Each analysis has a time budget, and each size a limit (`BUDGET_SECONDS` and `PER_RUN_SECONDS` in `complexity.py`). A solution that outgrows them is stopped, and the note says where. - The runner analyzes only when no submission is waiting, so an analysis never holds up a verdict by more than the one in progress.