diff --git a/apps/backend/app/runner/complexity.py b/apps/backend/app/runner/complexity.py index 83e23d0..3135167 100644 --- a/apps/backend/app/runner/complexity.py +++ b/apps/backend/app/runner/complexity.py @@ -25,10 +25,21 @@ BUDGET_SECONDS = 6.0 PER_RUN_SECONDS = 2.5 ENOUGH_SECONDS = 1.0 - -# Below this a run is mostly noise, not the solution. +# Only runs shorter than this are repeated. Noise is a few tens of +# milliseconds, which does not matter to a longer run, and the time saved +# buys a slow solution one more size. +REPEAT_BELOW_SECONDS = 0.5 + +# A run counts once the solution's own time is at least the fixed cost taken +# off it, so an error in that estimate stays a fraction of what is measured. +# The fixed cost is some 15 ms on a laptop and 150 ms under gVisor, which also +# reports CPU time in 10 ms steps. This is the floor where it is smaller. MIN_MEASURABLE_MS = 20.0 +# Only the largest sizes are fitted: the smaller a run, the larger the share +# of its time that is noise and error in the fixed cost. +FITTED_POINTS = 3 + # Fitted exponent of n -> growth class, split halfway between the powers. # n log n fits at about 1.1 over the sizes used, too close to n to tell # apart, so they share a class. @@ -111,19 +122,32 @@ def analyze(code: str, generator: str, mem_limit_mb: int) -> dict: # and nothing more; left in, it flattens the growth of every run. tiny = generate(max(4, sizes[0] // 32)) startup = min(_run(program, workdir, tiny, PER_RUN_SECONDS, mem_limit_mb)[0] or 0.0 for _ in range(3)) + floor_ms = max(MIN_MEASURABLE_MS, startup * 1000) for n in sizes: remaining = BUDGET_SECONDS - spent if remaining <= 0.1: note = f"Stopped before n = {n}: out of time for this analysis." break - seconds, failure = _run(program, workdir, generate(n), min(PER_RUN_SECONDS, remaining), mem_limit_mb) + stdin = generate(n) + seconds, failure = _run(program, workdir, stdin, min(PER_RUN_SECONDS, remaining), mem_limit_mb) if seconds is None: note = f"Stopped at n = {n}: it {failure}." break spent += seconds + # Scheduling only ever makes a run slower, so a short measurable + # size is run twice and the faster run kept. + if ( + (seconds - startup) * 1000 >= floor_ms + and seconds < REPEAT_BELOW_SECONDS + and BUDGET_SECONDS - spent > seconds + ): + again, _ = _run(program, workdir, stdin, min(PER_RUN_SECONDS, BUDGET_SECONDS - spent), mem_limit_mb) + if again is not None: + spent += again + seconds = min(seconds, again) ms = (seconds - startup) * 1000 - if ms >= MIN_MEASURABLE_MS: + if ms >= floor_ms: points.append((n, ms)) if seconds >= ENOUGH_SECONDS: break @@ -135,7 +159,7 @@ def analyze(code: str, generator: str, mem_limit_mb: int) -> dict: "note": note, } if len(points) >= 2: - result["complexity"], slope = classify(points) + result["complexity"], slope = classify(points[-FITTED_POINTS:]) result["slope"] = round(slope, 2) if len(points) == 2: result["note"] = (note + " " if note else "") + "Only two sizes were measurable, so this is rough." diff --git a/apps/backend/migrations/V14__longer_two_sum_analysis.sql b/apps/backend/migrations/V14__longer_two_sum_analysis.sql new file mode 100644 index 0000000..096d185 --- /dev/null +++ b/apps/backend/migrations/V14__longer_two_sum_analysis.sql @@ -0,0 +1,10 @@ +-- V14: One more input size for Two Sum's complexity analysis +-- +-- A run only counts once the solution's own time is at least the run's fixed +-- cost (app/runner/complexity.py), and at half a million numbers the +-- reference solution clears that at only one or two sizes. A million numbers +-- fit in the problem's memory limit. + +UPDATE problems +SET complexity_generator = replace(complexity_generator, 'range(10)', 'range(11)') +WHERE slug = 'two-sum'; diff --git a/apps/web/src/components/ComplexityPanel.tsx b/apps/web/src/components/ComplexityPanel.tsx index d5a9729..9c15662 100644 --- a/apps/web/src/components/ComplexityPanel.tsx +++ b/apps/web/src/components/ComplexityPanel.tsx @@ -75,15 +75,16 @@ const ComplexityPanel = ({ submission, expected }: { submission: Submission; exp const measured = CLASSES.indexOf(analysis.complexity); const target = expected ? CLASSES.indexOf(expected) : -1; const slower = target >= 0 && measured > target; + const verdict = target < 0 ? null : measured > target ? "Slower than" : measured < target ? "Faster than" : "Matches"; return (
- {expected && ( + {expected && verdict && ( - {slower ? "Slower than" : "Matches"} the expected {LABEL[expected]} + {verdict} the expected {LABEL[expected]} )}
diff --git a/docs/judge-and-sandbox.md b/docs/judge-and-sandbox.md index 8e3f83a..3c8fad0 100644 --- a/docs/judge-and-sandbox.md +++ b/docs/judge-and-sandbox.md @@ -33,8 +33,8 @@ Without `SANDBOX_IMAGE`, the runner refuses to start unless `ALLOW_UNSANDBOXED=1 On request, an accepted run can be rerun on inputs of growing size to estimate how its running time grows. The problem's generator (`complexity_generator`, see [CONTRIBUTING.md](../CONTRIBUTING.md)) makes the inputs, and `app/runner/complexity.py` runs the program on each size in one sandbox container. - It measures CPU time from outside the program. Counting executed lines would miss work done inside built-ins, such as `x in some_list`, and would call a quadratic brute force linear. The program also can't report a time of its own. -- The fixed cost of every run (interpreter start, imports, reading input) is measured on a tiny input and subtracted. -- It fits the times to a power of n and names the class. n log n measures too close to n to tell apart, so the two share a class. +- The fixed cost of every run (interpreter start, imports, reading input) is measured on a tiny input and subtracted. Under gVisor it is some 150 ms, ten times a laptop's, and gVisor reports CPU time in 10 ms steps, so a size only counts once the solution's own time is at least that fixed cost. Short runs are timed twice and the faster kept, since scheduling only ever adds time. +- It fits the largest measurable sizes (`FITTED_POINTS`) to a power of n and names the class. n log n measures too close to n to tell apart, so the two share a class. - Each analysis has a time budget, and each size a limit (`BUDGET_SECONDS` and `PER_RUN_SECONDS` in `complexity.py`). A solution that outgrows them is stopped, and the note says where. - The runner analyzes only when no submission is waiting, so an analysis never holds up a verdict by more than the one in progress.