Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
34 changes: 29 additions & 5 deletions apps/backend/app/runner/complexity.py
Original file line number Diff line number Diff line change
Expand Up @@ -25,10 +25,21 @@
BUDGET_SECONDS = 6.0
PER_RUN_SECONDS = 2.5
ENOUGH_SECONDS = 1.0

# Below this a run is mostly noise, not the solution.
# Only runs shorter than this are repeated. Noise is a few tens of
# milliseconds, which does not matter to a longer run, and the time saved
# buys a slow solution one more size.
REPEAT_BELOW_SECONDS = 0.5

# A run counts once the solution's own time is at least the fixed cost taken
# off it, so an error in that estimate stays a fraction of what is measured.
# The fixed cost is some 15 ms on a laptop and 150 ms under gVisor, which also
# reports CPU time in 10 ms steps. This is the floor where it is smaller.
MIN_MEASURABLE_MS = 20.0

# Only the largest sizes are fitted: the smaller a run, the larger the share
# of its time that is noise and error in the fixed cost.
FITTED_POINTS = 3

# Fitted exponent of n -> growth class, split halfway between the powers.
# n log n fits at about 1.1 over the sizes used, too close to n to tell
# apart, so they share a class.
Expand Down Expand Up @@ -111,19 +122,32 @@ def analyze(code: str, generator: str, mem_limit_mb: int) -> dict:
# and nothing more; left in, it flattens the growth of every run.
tiny = generate(max(4, sizes[0] // 32))
startup = min(_run(program, workdir, tiny, PER_RUN_SECONDS, mem_limit_mb)[0] or 0.0 for _ in range(3))
floor_ms = max(MIN_MEASURABLE_MS, startup * 1000)

for n in sizes:
remaining = BUDGET_SECONDS - spent
if remaining <= 0.1:
note = f"Stopped before n = {n}: out of time for this analysis."
break
seconds, failure = _run(program, workdir, generate(n), min(PER_RUN_SECONDS, remaining), mem_limit_mb)
stdin = generate(n)
seconds, failure = _run(program, workdir, stdin, min(PER_RUN_SECONDS, remaining), mem_limit_mb)
if seconds is None:
note = f"Stopped at n = {n}: it {failure}."
break
spent += seconds
# Scheduling only ever makes a run slower, so a short measurable
# size is run twice and the faster run kept.
if (
(seconds - startup) * 1000 >= floor_ms
and seconds < REPEAT_BELOW_SECONDS
and BUDGET_SECONDS - spent > seconds
):
again, _ = _run(program, workdir, stdin, min(PER_RUN_SECONDS, BUDGET_SECONDS - spent), mem_limit_mb)
if again is not None:
spent += again
seconds = min(seconds, again)
ms = (seconds - startup) * 1000
if ms >= MIN_MEASURABLE_MS:
if ms >= floor_ms:
points.append((n, ms))
if seconds >= ENOUGH_SECONDS:
break
Expand All @@ -135,7 +159,7 @@ def analyze(code: str, generator: str, mem_limit_mb: int) -> dict:
"note": note,
}
if len(points) >= 2:
result["complexity"], slope = classify(points)
result["complexity"], slope = classify(points[-FITTED_POINTS:])
result["slope"] = round(slope, 2)
if len(points) == 2:
result["note"] = (note + " " if note else "") + "Only two sizes were measurable, so this is rough."
Expand Down
10 changes: 10 additions & 0 deletions apps/backend/migrations/V14__longer_two_sum_analysis.sql
Original file line number Diff line number Diff line change
@@ -0,0 +1,10 @@
-- V14: One more input size for Two Sum's complexity analysis
--
-- A run only counts once the solution's own time is at least the run's fixed
-- cost (app/runner/complexity.py), and at half a million numbers the
-- reference solution clears that at only one or two sizes. A million numbers
-- fit in the problem's memory limit.

UPDATE problems
SET complexity_generator = replace(complexity_generator, 'range(10)', 'range(11)')
WHERE slug = 'two-sum';
5 changes: 3 additions & 2 deletions apps/web/src/components/ComplexityPanel.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -75,15 +75,16 @@ const ComplexityPanel = ({ submission, expected }: { submission: Submission; exp
const measured = CLASSES.indexOf(analysis.complexity);
const target = expected ? CLASSES.indexOf(expected) : -1;
const slower = target >= 0 && measured > target;
const verdict = target < 0 ? null : measured > target ? "Slower than" : measured < target ? "Faster than" : "Matches";
return (
<div className="space-y-3 border-t-2 border-zinc-800 px-4 py-3 text-sm">
<div className="flex flex-wrap items-baseline gap-x-4 gap-y-1">
<span className={`font-pixel text-3xl leading-none ${slower ? "text-amber-300" : "text-emerald-400"}`}>
<Label name={analysis.complexity} />
</span>
{expected && (
{expected && verdict && (
<span className={muted}>
{slower ? "Slower than" : "Matches"} the expected {LABEL[expected]}
{verdict} the expected {LABEL[expected]}
</span>
)}
</div>
Expand Down
4 changes: 2 additions & 2 deletions docs/judge-and-sandbox.md
Original file line number Diff line number Diff line change
Expand Up @@ -33,8 +33,8 @@ Without `SANDBOX_IMAGE`, the runner refuses to start unless `ALLOW_UNSANDBOXED=1
On request, an accepted run can be rerun on inputs of growing size to estimate how its running time grows. The problem's generator (`complexity_generator`, see [CONTRIBUTING.md](../CONTRIBUTING.md)) makes the inputs, and `app/runner/complexity.py` runs the program on each size in one sandbox container.

- It measures CPU time from outside the program. Counting executed lines would miss work done inside built-ins, such as `x in some_list`, and would call a quadratic brute force linear. The program also can't report a time of its own.
- The fixed cost of every run (interpreter start, imports, reading input) is measured on a tiny input and subtracted.
- It fits the times to a power of n and names the class. n log n measures too close to n to tell apart, so the two share a class.
- The fixed cost of every run (interpreter start, imports, reading input) is measured on a tiny input and subtracted. Under gVisor it is some 150 ms, ten times a laptop's, and gVisor reports CPU time in 10 ms steps, so a size only counts once the solution's own time is at least that fixed cost. Short runs are timed twice and the faster kept, since scheduling only ever adds time.
- It fits the largest measurable sizes (`FITTED_POINTS`) to a power of n and names the class. n log n measures too close to n to tell apart, so the two share a class.
- Each analysis has a time budget, and each size a limit (`BUDGET_SECONDS` and `PER_RUN_SECONDS` in `complexity.py`). A solution that outgrows them is stopped, and the note says where.
- The runner analyzes only when no submission is waiting, so an analysis never holds up a verdict by more than the one in progress.

Expand Down
Loading