From fc04ead8786fa8e160892aa3ec24664a8f5ff591 Mon Sep 17 00:00:00 2001
From: Naman Rusia
Date: Fri, 25 Sep 2026 14:03:34 -0400
Subject: [PATCH 1/2] Estimate the time complexity of an accepted run on
request
An accepted run can be rerun on inputs of growing size from a per-problem
generator. The runner measures each size's CPU time in one sandbox container,
fits it to a power of n and names the class, and the room sees the result
next to the problem's expected class. Analyses only run when no submission is
waiting, within a time budget, one per person at a time and 20 an hour.
Two Sum, Valid Parentheses, Best Time to Buy and Sell Stock and 3Sum support
it to start.
---
CONTRIBUTING.md | 16 ++
apps/backend/app/dao/problems.py | 15 +-
apps/backend/app/dao/submissions.py | 57 ++++++-
apps/backend/app/routes/submissions.py | 15 ++
apps/backend/app/runner/__main__.py | 42 ++++-
apps/backend/app/runner/complexity.py | 161 ++++++++++++++++++
apps/backend/app/runner/sandbox.py | 42 +++--
apps/backend/app/schemas/problems.py | 2 +
apps/backend/app/schemas/submissions.py | 14 ++
apps/backend/app/services/submissions.py | 48 ++++++
.../migrations/V13__complexity_analysis.sql | 94 ++++++++++
apps/backend/tests/conftest.py | 3 +
apps/backend/tests/test_analysis.py | 114 +++++++++++++
apps/backend/tests/test_complexity.py | 43 +++++
apps/backend/tests/test_sandbox.py | 9 +-
apps/web/src/components/ComplexityPanel.tsx | 112 ++++++++++++
apps/web/src/hooks/UseWebSocket.ts | 10 ++
apps/web/src/lib/api.ts | 6 +
apps/web/src/routes/rooms/RoomView.tsx | 6 +
docs/architecture.md | 2 +
docs/judge-and-sandbox.md | 12 ++
21 files changed, 806 insertions(+), 17 deletions(-)
create mode 100644 apps/backend/app/runner/complexity.py
create mode 100644 apps/backend/migrations/V13__complexity_analysis.sql
create mode 100644 apps/backend/tests/test_analysis.py
create mode 100644 apps/backend/tests/test_complexity.py
create mode 100644 apps/web/src/components/ComplexityPanel.tsx
diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md
index f825145..5656b5a 100644
--- a/CONTRIBUTING.md
+++ b/CONTRIBUTING.md
@@ -125,6 +125,22 @@ in your migration, hidden ones included. It fails if:
- the starter code does not compile, has no `Solution` class, or passes the
tests without being filled in
+### Optional: support complexity analysis
+
+An accepted run can be rerun on bigger inputs to estimate how its time grows.
+A problem supports this when its migration also sets two columns:
+
+- `complexity_generator`: Python source defining `SIZES`, the input sizes to
+ try in order, and `generate(n)`, which returns the stdin for an input of
+ size `n`. Make the inputs force a full solve, for example by putting the
+ answer at the end.
+- `expected_complexity`: the class your reference solution achieves, one of
+ `constant`, `linear`, `quadratic` or `cubic`.
+
+The test suite measures your reference solution with your generator and fails
+if it does not land in the expected class. `V13__complexity_analysis.sql` has
+examples.
+
### 5. Check it
Run `make test`. To see the problem in the app,
diff --git a/apps/backend/app/dao/problems.py b/apps/backend/app/dao/problems.py
index c8bab92..87e6218 100644
--- a/apps/backend/app/dao/problems.py
+++ b/apps/backend/app/dao/problems.py
@@ -8,7 +8,10 @@
# The full test list never leaves the DAO through here: only the public
# samples do, so a client cannot read the hidden tests off the API.
-COLUMNS = "id, slug, title, difficulty, time_limit_ms, mem_limit_mb, statement, starter_code, tests"
+COLUMNS = (
+ "id, slug, title, difficulty, time_limit_ms, mem_limit_mb, statement, starter_code, tests, "
+ "complexity_generator IS NOT NULL AS analyzable, expected_complexity"
+)
def _map_problem(row: asyncpg.Record) -> dict:
@@ -25,6 +28,8 @@ def _map_problem(row: asyncpg.Record) -> dict:
"samples": [
{"input": t["input"], "expected": t["expected"]} for t in tests if not t.get("hidden")
],
+ "analyzable": row["analyzable"],
+ "expectedComplexity": row["expected_complexity"],
}
@@ -84,6 +89,14 @@ async def get_judge_spec(self, problem_id: UUID) -> dict:
"memLimitMb": row["mem_limit_mb"],
}
+ async def get_analysis_spec(self, problem_id: UUID) -> dict | None:
+ query = "SELECT complexity_generator, mem_limit_mb FROM problems WHERE id = $1"
+ async with self.pool.acquire() as conn:
+ row = await conn.fetchrow(query, problem_id)
+ if row is None or row["complexity_generator"] is None:
+ return None
+ return {"generator": row["complexity_generator"], "memLimitMb": row["mem_limit_mb"]}
+
async def exists(self, problem_id: UUID) -> bool:
query = "SELECT EXISTS(SELECT 1 FROM problems WHERE id = $1)"
async with self.pool.acquire() as conn:
diff --git a/apps/backend/app/dao/submissions.py b/apps/backend/app/dao/submissions.py
index 7c70840..c47cbf7 100644
--- a/apps/backend/app/dao/submissions.py
+++ b/apps/backend/app/dao/submissions.py
@@ -9,7 +9,7 @@
SELECT_SUBMISSION = """
SELECT s.id, s.room_id, s.user_id, s.problem_id, s.language, s.code, s.status,
s.time_ms, s.created_at, s.s3_key_stdout, s.s3_key_stderr,
- s.s3_key_result_json, s.result, u.name AS user_name
+ s.s3_key_result_json, s.result, s.analysis_status, s.analysis, u.name AS user_name
FROM submissions s
JOIN users u ON u.id = s.user_id
"""
@@ -40,6 +40,8 @@ def _map_submission(row: asyncpg.Record) -> dict:
"s3KeyStderr": row["s3_key_stderr"],
"s3KeyResultJson": row["s3_key_result_json"],
"result": json.loads(row["result"]) if row["result"] else None,
+ "analysisStatus": row["analysis_status"],
+ "analysis": json.loads(row["analysis"]) if row["analysis"] else None,
}
@@ -119,3 +121,56 @@ async def requeue_running(self) -> int:
async with self.pool.acquire() as conn:
tag = await conn.execute("UPDATE submissions SET status = 'pending' WHERE status = 'running'")
return int(tag.rsplit(" ", 1)[1])
+
+ async def count_analyses_since(self, user_id: str, seconds: int) -> int:
+ query = """
+ SELECT COUNT(*) FROM submissions
+ WHERE analysis_requested_by = $1 AND analysis_requested_at > NOW() - make_interval(secs => $2)
+ """
+ async with self.pool.acquire() as conn:
+ return await conn.fetchval(query, user_id, seconds)
+
+ async def request_analysis(self, submission_id: UUID, user_id: str) -> dict | None:
+ """Queue an analysis of an accepted run, unless one is queued or done already."""
+ query = """
+ UPDATE submissions
+ SET analysis_status = 'pending', analysis = NULL,
+ analysis_requested_by = $2, analysis_requested_at = NOW()
+ WHERE id = $1 AND status = 'accepted'
+ AND (analysis_status IS NULL OR analysis_status = 'failed')
+ RETURNING id
+ """
+ async with self.pool.acquire() as conn:
+ updated = await conn.fetchval(query, submission_id, user_id)
+ return await self.get_by_id(updated) if updated else None
+
+ async def claim_pending_analysis(self) -> dict | None:
+ query = """
+ UPDATE submissions SET analysis_status = 'running'
+ WHERE id = (
+ SELECT id FROM submissions
+ WHERE analysis_status = 'pending'
+ ORDER BY analysis_requested_at
+ LIMIT 1
+ FOR UPDATE SKIP LOCKED
+ )
+ RETURNING id
+ """
+ async with self.pool.acquire() as conn:
+ submission_id = await conn.fetchval(query)
+ return await self.get_by_id(submission_id) if submission_id else None
+
+ async def complete_analysis(self, submission_id: UUID, status: str, analysis: dict) -> None:
+ query = "UPDATE submissions SET analysis_status = $2, analysis = $3::jsonb WHERE id = $1"
+ async with self.pool.acquire() as conn:
+ async with conn.transaction():
+ await conn.execute(query, submission_id, status, json.dumps(analysis))
+ # The room hears about it the way it hears about a verdict.
+ await conn.execute("SELECT pg_notify($1, $2)", JUDGED_CHANNEL, str(submission_id))
+
+ async def requeue_running_analyses(self) -> int:
+ async with self.pool.acquire() as conn:
+ tag = await conn.execute(
+ "UPDATE submissions SET analysis_status = 'pending' WHERE analysis_status = 'running'"
+ )
+ return int(tag.rsplit(" ", 1)[1])
diff --git a/apps/backend/app/routes/submissions.py b/apps/backend/app/routes/submissions.py
index f1db71f..31a39d2 100644
--- a/apps/backend/app/routes/submissions.py
+++ b/apps/backend/app/routes/submissions.py
@@ -43,6 +43,21 @@ async def list_submissions(
return [SubmissionResponse.model_validate(submission) for submission in submissions]
+@router.post("/{submission_id}/analysis", response_model=SubmissionResponse)
+async def request_analysis(
+ submission_id: UUID,
+ request: Request,
+ caller_id: str = Depends(current_user_id),
+ service: SubmissionService = Depends(get_submission_service),
+) -> SubmissionResponse:
+ submission = await service.request_analysis(submission_id, caller_id)
+ # Everyone in the room sees it start; the result follows from the listener.
+ await request.app.state.room_chat_manager.broadcast(
+ submission["roomId"], {"type": "submission", "submission": submission}
+ )
+ return SubmissionResponse.model_validate(submission)
+
+
@router.get("/{submission_id}", response_model=SubmissionResponse)
async def get_submission(
submission_id: UUID,
diff --git a/apps/backend/app/runner/__main__.py b/apps/backend/app/runner/__main__.py
index 24cd9c8..6805b71 100644
--- a/apps/backend/app/runner/__main__.py
+++ b/apps/backend/app/runner/__main__.py
@@ -1,5 +1,9 @@
"""Claim pending submissions and judge them, one at a time.
+Complexity analyses share the runner but only run when no submission is
+waiting, so asking for one never delays anyone's verdict by more than the
+analysis already in progress.
+
The submissions table is the queue: a row is claimed by moving it to
``running`` under ``FOR UPDATE SKIP LOCKED``, so several runners can share a
database without judging the same submission twice.
@@ -16,6 +20,7 @@
from app.dao.submissions import SubmissionDAO
from app.database import create_pool
from app.runner import sandbox
+from app.runner.complexity import analyze
from app.runner.judge import RUNTIME_ERROR, Verdict, judge
logger = logging.getLogger(__name__)
@@ -36,6 +41,40 @@ def run_judge(code: str, tests: list[dict], time_limit_ms: int, mem_limit_mb: in
return judge(code, tests, time_limit_ms, mem_limit_mb)
+def run_analysis(code: str, generator: str, mem_limit_mb: int) -> dict:
+ if SANDBOX_IMAGE:
+ return sandbox.analyze_in_container(SANDBOX_IMAGE, code, generator, mem_limit_mb)
+ return analyze(code, generator, mem_limit_mb)
+
+
+async def analyze_next(submissions: SubmissionDAO, problems: ProblemDAO) -> bool:
+ """Analyze one accepted run if one is waiting. Returns whether it did."""
+ submission = await submissions.claim_pending_analysis()
+ if submission is None:
+ return False
+
+ try:
+ spec = await problems.get_analysis_spec(submission["problemId"])
+ if spec is None:
+ raise RuntimeError("the problem has no generator")
+ analysis = await asyncio.to_thread(run_analysis, submission["code"] or "", spec["generator"], spec["memLimitMb"])
+ except Exception:
+ logger.exception("Analysis failed for submission %s", submission["id"])
+ analysis = {"points": [], "complexity": None, "slope": None, "note": "The analysis could not run. Try again later."}
+
+ status = "done" if analysis["complexity"] else "failed"
+ try:
+ await submissions.complete_analysis(submission["id"], status, analysis)
+ except Exception:
+ logger.exception("Could not store the analysis for submission %s", submission["id"])
+ status = "failed"
+ await submissions.complete_analysis(
+ submission["id"], status, {"points": [], "complexity": None, "slope": None, "note": "The analysis could not be saved."}
+ )
+ logger.info("Submission %s analysis: %s", submission["id"], analysis["complexity"] or status)
+ return True
+
+
async def judge_next(submissions: SubmissionDAO, problems: ProblemDAO) -> bool:
"""Judge one submission if there is one waiting. Returns whether it did."""
submission = await submissions.claim_pending()
@@ -80,9 +119,10 @@ async def main() -> None:
# claimed again. One runner at a time is the local setup, so on boot
# anything still running is ours from before and goes back in the queue.
await submissions.requeue_running()
+ await submissions.requeue_running_analyses()
try:
while True:
- if not await judge_next(submissions, problems):
+ if not await judge_next(submissions, problems) and not await analyze_next(submissions, problems):
await asyncio.sleep(POLL_SECONDS)
finally:
await pool.close()
diff --git a/apps/backend/app/runner/complexity.py b/apps/backend/app/runner/complexity.py
new file mode 100644
index 0000000..83e23d0
--- /dev/null
+++ b/apps/backend/app/runner/complexity.py
@@ -0,0 +1,161 @@
+"""Estimate how a program's running time grows with the size of its input.
+
+The program runs on inputs of doubling size from the problem's generator, and
+the CPU time of each run is fitted to a power of n. CPU time, not a count of
+executed lines, because a line count sees `x in some_list` or `sorted(xs)` as
+one step and would call a quadratic brute force linear. It is measured from
+outside the program, so the program cannot report a time of its own.
+
+Like the judge, this file is shipped into a sandbox container as the program
+text, so it imports nothing from the app.
+"""
+
+from __future__ import annotations
+
+import math
+import resource
+import subprocess
+import sys
+import tempfile
+import time
+from pathlib import Path
+
+# Runner time one analysis may use, and the most one size may take. A run
+# that reaches a second has shown its growth; bigger inputs only cost time.
+BUDGET_SECONDS = 6.0
+PER_RUN_SECONDS = 2.5
+ENOUGH_SECONDS = 1.0
+
+# Below this a run is mostly noise, not the solution.
+MIN_MEASURABLE_MS = 20.0
+
+# Fitted exponent of n -> growth class, split halfway between the powers.
+# n log n fits at about 1.1 over the sizes used, too close to n to tell
+# apart, so they share a class.
+CLASSES = [(0.5, "constant"), (1.5, "linear"), (2.5, "quadratic"), (3.5, "cubic")]
+WORSE = "worse"
+
+PROGRAM_ENV = {"PATH": "/usr/local/bin:/usr/bin:/bin", "LANG": "C.UTF-8"}
+
+
+def _limits(mem_limit_mb: int):
+ # The same limits the judge applies; this file cannot import them.
+ def apply() -> None:
+ mem = mem_limit_mb * 1024 * 1024
+ resource.setrlimit(resource.RLIMIT_AS, (mem, mem))
+ resource.setrlimit(resource.RLIMIT_NPROC, (0, 0))
+ resource.setrlimit(resource.RLIMIT_FSIZE, (0, 0))
+
+ return apply
+
+
+def _children_cpu() -> float:
+ usage = resource.getrusage(resource.RUSAGE_CHILDREN)
+ return usage.ru_utime + usage.ru_stime
+
+
+def classify(points: list[tuple[int, float]]) -> tuple[str, float]:
+ """Least-squares slope of log(ms) against log(n), and its class."""
+ xs = [math.log(n) for n, _ in points]
+ ys = [math.log(ms) for _, ms in points]
+ mean_x, mean_y = sum(xs) / len(xs), sum(ys) / len(ys)
+ spread = sum((x - mean_x) ** 2 for x in xs)
+ slope = sum((x - mean_x) * (y - mean_y) for x, y in zip(xs, ys)) / spread
+ for bound, name in CLASSES:
+ if slope < bound:
+ return name, slope
+ return WORSE, slope
+
+
+def _run(program: Path, workdir: str, stdin: str, timeout: float, mem_limit_mb: int):
+ """CPU and wall seconds for one run, or None and a reason if it did not finish."""
+ cpu_before, started = _children_cpu(), time.perf_counter()
+ try:
+ completed = subprocess.run(
+ [sys.executable, "-I", str(program)],
+ input=stdin.encode(),
+ # Nothing is checked here, and a large input can mean a large answer.
+ stdout=subprocess.DEVNULL,
+ stderr=subprocess.PIPE,
+ timeout=timeout,
+ cwd=workdir,
+ env=PROGRAM_ENV,
+ preexec_fn=_limits(mem_limit_mb),
+ )
+ except subprocess.TimeoutExpired:
+ return None, f"took over {timeout:.1f} s"
+ wall = time.perf_counter() - started
+ if completed.returncode != 0:
+ # Postgres rejects NUL in jsonb, and a result that cannot be stored
+ # would leave the runner retrying the same analysis forever.
+ last = completed.stderr.decode("utf-8", errors="replace").replace("\x00", "").strip().splitlines()
+ return None, f"crashed: {last[-1][:200]}" if last else "crashed"
+ # Some sandboxes do not account children's CPU; wall time is the fallback.
+ cpu = _children_cpu() - cpu_before
+ return (cpu if cpu > 0 else wall), None
+
+
+def analyze(code: str, generator: str, mem_limit_mb: int) -> dict:
+ namespace: dict = {}
+ exec(generator, namespace)
+ generate, sizes = namespace["generate"], namespace["SIZES"]
+
+ points: list[tuple[int, float]] = []
+ note = None
+ spent = 0.0
+ with tempfile.TemporaryDirectory() as workdir:
+ program = Path(workdir) / "main.py"
+ program.write_text(code)
+ # Every run pays for starting the interpreter, the program's imports
+ # and reading its input. The program on a tiny input costs about that
+ # and nothing more; left in, it flattens the growth of every run.
+ tiny = generate(max(4, sizes[0] // 32))
+ startup = min(_run(program, workdir, tiny, PER_RUN_SECONDS, mem_limit_mb)[0] or 0.0 for _ in range(3))
+
+ for n in sizes:
+ remaining = BUDGET_SECONDS - spent
+ if remaining <= 0.1:
+ note = f"Stopped before n = {n}: out of time for this analysis."
+ break
+ seconds, failure = _run(program, workdir, generate(n), min(PER_RUN_SECONDS, remaining), mem_limit_mb)
+ if seconds is None:
+ note = f"Stopped at n = {n}: it {failure}."
+ break
+ spent += seconds
+ ms = (seconds - startup) * 1000
+ if ms >= MIN_MEASURABLE_MS:
+ points.append((n, ms))
+ if seconds >= ENOUGH_SECONDS:
+ break
+
+ result = {
+ "points": [{"n": n, "ms": round(ms, 1)} for n, ms in points],
+ "complexity": None,
+ "slope": None,
+ "note": note,
+ }
+ if len(points) >= 2:
+ result["complexity"], slope = classify(points)
+ result["slope"] = round(slope, 2)
+ if len(points) == 2:
+ result["note"] = (note + " " if note else "") + "Only two sizes were measurable, so this is rough."
+ elif not points and note is None:
+ # Even the largest input ran too fast to measure: it barely grows.
+ result["complexity"], result["slope"] = "constant", 0.0
+ elif note is None:
+ result["note"] = "Only one size was measurable, which is not enough to see growth."
+ return result
+
+
+if __name__ == "__main__":
+ # Entry point inside a sandbox container: the spec arrives on stdin and
+ # the result leaves on stdout, both as JSON.
+ import ctypes
+ import json
+
+ # As in the judge: the program must not be able to print our result for us.
+ PR_SET_DUMPABLE = 4
+ ctypes.CDLL(None).prctl(PR_SET_DUMPABLE, 0)
+
+ spec = json.load(sys.stdin)
+ print(json.dumps(analyze(spec["code"], spec["generator"], spec["memLimitMb"])))
diff --git a/apps/backend/app/runner/sandbox.py b/apps/backend/app/runner/sandbox.py
index b0b5caf..548a499 100644
--- a/apps/backend/app/runner/sandbox.py
+++ b/apps/backend/app/runner/sandbox.py
@@ -17,15 +17,17 @@
import uuid
from pathlib import Path
+from app.runner import complexity
from app.runner.judge import RUNTIME_ERROR, TestOutcome, Verdict
logger = logging.getLogger(__name__)
JUDGE_SOURCE = (Path(__file__).parent / "judge.py").read_text()
+COMPLEXITY_SOURCE = (Path(__file__).parent / "complexity.py").read_text()
# Room for the interpreter and the judge on top of the program's own limit.
CONTAINER_MEMORY_OVERHEAD_MB = 64
-# Container start and interpreter start, on top of the tests' own limits.
+# Container start and interpreter start, on top of the program's own limits.
STARTUP_ALLOWANCE_SECONDS = 20
# `runsc` in production: gVisor answers the program's system calls itself, so
@@ -40,9 +42,8 @@ def pull(image: str) -> None:
subprocess.run(["docker", "pull", "--quiet", image], check=True, capture_output=True, timeout=600)
-def judge_in_container(
- image: str, code: str, tests: list[dict], time_limit_ms: int, mem_limit_mb: int
-) -> Verdict:
+def _run_in_container(image: str, source: str, spec: dict, mem_limit_mb: int, timeout: float) -> dict | None:
+ """Run a program file from this package in a fresh container; its JSON result, or None if it hung."""
name = f"judge-{uuid.uuid4().hex[:12]}"
command = [
"docker", "run", "--rm", "--interactive", "--name", name,
@@ -61,27 +62,32 @@ def judge_in_container(
"--env", "PYTHONDONTWRITEBYTECODE=1",
*(["--runtime", SANDBOX_RUNTIME] if SANDBOX_RUNTIME else []),
image,
- "python", "-I", "-c", JUDGE_SOURCE,
+ "python", "-I", "-c", source,
]
- spec = json.dumps(
- {"code": code, "tests": tests, "timeLimitMs": time_limit_ms, "memLimitMb": mem_limit_mb}
- )
- timeout = len(tests) * time_limit_ms / 1000 + STARTUP_ALLOWANCE_SECONDS
try:
completed = subprocess.run(
- command, input=spec.encode(), capture_output=True, timeout=timeout
+ command, input=json.dumps(spec).encode(), capture_output=True, timeout=timeout
)
except subprocess.TimeoutExpired:
- # The judge inside enforces per-test limits; reaching this means the
+ # The program inside enforces its own limits; reaching this means the
# container itself hung. Make sure it is gone.
subprocess.run(["docker", "rm", "--force", name], capture_output=True, timeout=60)
logger.error("Sandbox %s exceeded %.0fs and was removed", name, timeout)
- return Verdict(status=RUNTIME_ERROR, timeMs=0, passed=0, total=len(tests))
+ return None
if completed.returncode != 0:
raise RuntimeError(f"Sandbox exited {completed.returncode}: {completed.stderr.decode(errors='replace')[:500]}")
+ return json.loads(completed.stdout)
+
- data = json.loads(completed.stdout)
+def judge_in_container(
+ image: str, code: str, tests: list[dict], time_limit_ms: int, mem_limit_mb: int
+) -> Verdict:
+ spec = {"code": code, "tests": tests, "timeLimitMs": time_limit_ms, "memLimitMb": mem_limit_mb}
+ timeout = len(tests) * time_limit_ms / 1000 + STARTUP_ALLOWANCE_SECONDS
+ data = _run_in_container(image, JUDGE_SOURCE, spec, mem_limit_mb, timeout)
+ if data is None:
+ return Verdict(status=RUNTIME_ERROR, timeMs=0, passed=0, total=len(tests))
return Verdict(
status=data["status"],
timeMs=data["timeMs"],
@@ -89,3 +95,13 @@ def judge_in_container(
total=data["total"],
tests=[TestOutcome(**outcome) for outcome in data["tests"]],
)
+
+
+def analyze_in_container(image: str, code: str, generator: str, mem_limit_mb: int) -> dict:
+ spec = {"code": code, "generator": generator, "memLimitMb": mem_limit_mb}
+ # Generating the inputs takes a moment on top of the measured runs.
+ timeout = complexity.BUDGET_SECONDS + complexity.PER_RUN_SECONDS + STARTUP_ALLOWANCE_SECONDS
+ data = _run_in_container(image, COMPLEXITY_SOURCE, spec, mem_limit_mb, timeout)
+ if data is None:
+ return {"points": [], "complexity": None, "slope": None, "note": "The analysis did not finish in time."}
+ return data
diff --git a/apps/backend/app/schemas/problems.py b/apps/backend/app/schemas/problems.py
index 5659b6a..773f735 100644
--- a/apps/backend/app/schemas/problems.py
+++ b/apps/backend/app/schemas/problems.py
@@ -20,3 +20,5 @@ class ProblemResponse(BaseModel):
statement: str | None
starterCode: str | None
samples: list[SampleTest]
+ analyzable: bool = False
+ expectedComplexity: str | None = None
diff --git a/apps/backend/app/schemas/submissions.py b/apps/backend/app/schemas/submissions.py
index 4a12f8b..79ea8c3 100644
--- a/apps/backend/app/schemas/submissions.py
+++ b/apps/backend/app/schemas/submissions.py
@@ -34,6 +34,18 @@ class SubmissionResult(BaseModel):
tests: list[TestOutcome]
+class ComplexityPoint(BaseModel):
+ n: int
+ ms: float
+
+
+class ComplexityAnalysis(BaseModel):
+ points: list[ComplexityPoint]
+ complexity: str | None
+ slope: float | None
+ note: str | None
+
+
class SubmissionResponse(BaseModel):
id: UUID
roomId: str
@@ -49,3 +61,5 @@ class SubmissionResponse(BaseModel):
s3KeyStderr: str | None = None
s3KeyResultJson: str | None = None
result: SubmissionResult | None = None
+ analysisStatus: str | None = None
+ analysis: ComplexityAnalysis | None = None
diff --git a/apps/backend/app/services/submissions.py b/apps/backend/app/services/submissions.py
index 1d99f54..5ad5b61 100644
--- a/apps/backend/app/services/submissions.py
+++ b/apps/backend/app/services/submissions.py
@@ -17,6 +17,8 @@
# The runner is one process judging one run at a time. A pair practising
# hard stays well under this; a script holding the runner does not.
RUNS_PER_HOUR = 120
+# An analysis costs the runner several seconds, a run well under one.
+ANALYSES_PER_HOUR = 20
class SubmissionService:
@@ -97,6 +99,52 @@ async def get_submission(self, submission_id, caller_id: str) -> dict:
await ensure_room_member(self.room_member_dao, room, caller_id)
return submission
+ async def request_analysis(self, submission_id, caller_id: str) -> dict:
+ submission = await self.submission_dao.get_by_id(submission_id)
+ if not submission:
+ raise HTTPException(
+ status_code=status.HTTP_404_NOT_FOUND,
+ detail=f"Submission not found: {submission_id}",
+ )
+ room_id = submission["roomId"]
+ await get_active_room(self.room_dao, room_id)
+ if not await self.room_member_dao.is_present(room_id, caller_id):
+ raise HTTPException(
+ status_code=status.HTTP_403_FORBIDDEN,
+ detail="Join the room to analyze runs in it",
+ )
+ # A wrong answer's timings describe a program that does not solve
+ # the problem, so they say nothing useful.
+ if submission["status"] != "accepted":
+ raise HTTPException(
+ status_code=status.HTTP_409_CONFLICT,
+ detail="Only accepted runs can be analyzed",
+ )
+ problem = await self.problem_dao.get_by_id(submission["problemId"])
+ if not problem or not problem["analyzable"]:
+ raise HTTPException(
+ status_code=status.HTTP_409_CONFLICT,
+ detail="This problem does not support complexity analysis yet",
+ )
+ # Both people in a room can press the button; the second press sees
+ # the first one's analysis instead of queueing another.
+ if submission["analysisStatus"] in ("pending", "running", "done"):
+ return submission
+ if await self.submission_dao.count_analyses_since(caller_id, 3600) >= ANALYSES_PER_HOUR:
+ raise HTTPException(
+ status_code=status.HTTP_429_TOO_MANY_REQUESTS,
+ detail="Too many analyses this hour. Try again later.",
+ )
+ try:
+ queued = await self.submission_dao.request_analysis(submission_id, caller_id)
+ except asyncpg.UniqueViolationError as exc:
+ raise HTTPException(
+ status_code=status.HTTP_429_TOO_MANY_REQUESTS,
+ detail="Wait for your current analysis to finish",
+ ) from exc
+ # None means someone else queued it between the read and the update.
+ return queued or await self.submission_dao.get_by_id(submission_id)
+
async def list_submissions(self, room_id: str, caller_id: str, user_id: str | None = None) -> list[dict]:
room = await get_active_room(self.room_dao, room_id)
await ensure_room_member(self.room_member_dao, room, caller_id)
diff --git a/apps/backend/migrations/V13__complexity_analysis.sql b/apps/backend/migrations/V13__complexity_analysis.sql
new file mode 100644
index 0000000..79396c0
--- /dev/null
+++ b/apps/backend/migrations/V13__complexity_analysis.sql
@@ -0,0 +1,94 @@
+-- V13: Complexity analysis of accepted runs
+--
+-- On request, an accepted run is rerun on inputs of growing size to estimate
+-- how its running time grows (app/runner/complexity.py). A problem opts in
+-- with a generator: Python source defining SIZES, the input sizes to try in
+-- order, and generate(n), the stdin for an input of size n. Its expected
+-- class is what the reference solution achieves.
+
+ALTER TABLE problems
+ ADD COLUMN complexity_generator TEXT,
+ ADD COLUMN expected_complexity TEXT,
+ ADD CONSTRAINT problems_expected_complexity_known
+ CHECK (expected_complexity IN ('constant', 'linear', 'quadratic', 'cubic')),
+ ADD CONSTRAINT problems_complexity_needs_both
+ CHECK ((complexity_generator IS NULL) = (expected_complexity IS NULL));
+
+ALTER TABLE submissions
+ ADD COLUMN analysis_status TEXT
+ CHECK (analysis_status IN ('pending', 'running', 'done', 'failed')),
+ ADD COLUMN analysis JSONB,
+ ADD COLUMN analysis_requested_by TEXT REFERENCES users(id),
+ ADD COLUMN analysis_requested_at TIMESTAMPTZ;
+
+-- One analysis in flight per person, for the same reason as one run.
+CREATE UNIQUE INDEX submissions_one_analysis_in_flight_per_user
+ ON submissions (analysis_requested_by)
+ WHERE analysis_status IN ('pending', 'running');
+
+-- The only matching pair is the last two numbers, so every solution has to
+-- read to the end. Every other number is even and the target is odd, which
+-- makes that pair the only answer.
+UPDATE problems SET expected_complexity = 'linear', complexity_generator = $gen$
+import random
+
+SIZES = [1000 * 2 ** k for k in range(10)]
+
+
+def generate(n):
+ rng = random.Random(n)
+ evens = [2 * v for v in rng.sample(range(1, 10 * n), n - 1)]
+ even = evens.pop()
+ odd = 2 * rng.randrange(10 * n) + 1
+ nums = evens + [even, odd]
+ return " ".join(map(str, nums)) + "\n" + str(even + odd) + "\n"
+$gen$
+WHERE slug = 'two-sum';
+
+-- Balanced, so a correct solution reads every bracket.
+UPDATE problems SET expected_complexity = 'linear', complexity_generator = $gen$
+import random
+
+SIZES = [1000 * 2 ** k for k in range(13)]
+PAIRS = {"(": ")", "[": "]", "{": "}"}
+
+
+def generate(n):
+ rng = random.Random(n)
+ out, stack = [], []
+ for i in range(n):
+ if stack and (len(stack) == n - i or rng.random() < 0.5):
+ out.append(PAIRS[stack.pop()])
+ else:
+ stack.append(rng.choice("([{"))
+ out.append(stack[-1])
+ return "".join(out) + "\n"
+$gen$
+WHERE slug = 'valid-parentheses';
+
+UPDATE problems SET expected_complexity = 'linear', complexity_generator = $gen$
+import random
+
+SIZES = [1000 * 2 ** k for k in range(11)]
+
+
+def generate(n):
+ rng = random.Random(n)
+ return " ".join(str(rng.randint(1, 10000)) for _ in range(n)) + "\n"
+$gen$
+WHERE slug = 'best-time-to-buy-sell-stock';
+
+-- A wide range of values keeps the number of zero-sum triplets small, so the
+-- time is the search, not the answer. Sizes grow by a factor of about 1.4, as
+-- a cubic brute force outgrows its budget within a few doublings.
+UPDATE problems SET expected_complexity = 'quadratic', complexity_generator = $gen$
+import random
+
+SIZES = [int(100 * 2 ** (k / 2)) for k in range(14)]
+
+
+def generate(n):
+ rng = random.Random(n)
+ return " ".join(str(rng.randint(-10 ** 6, 10 ** 6)) for _ in range(n)) + "\n"
+$gen$
+WHERE slug = 'three-sum';
diff --git a/apps/backend/tests/conftest.py b/apps/backend/tests/conftest.py
index 249f25e..427cb5d 100644
--- a/apps/backend/tests/conftest.py
+++ b/apps/backend/tests/conftest.py
@@ -56,6 +56,9 @@ async def _settle_runs(pool) -> None:
await conn.execute(
"UPDATE submissions SET status = 'runtime_error' WHERE status IN ('pending', 'running')"
)
+ await conn.execute(
+ "UPDATE submissions SET analysis_status = 'failed' WHERE analysis_status IN ('pending', 'running')"
+ )
@pytest.fixture(autouse=True)
diff --git a/apps/backend/tests/test_analysis.py b/apps/backend/tests/test_analysis.py
new file mode 100644
index 0000000..1ab728b
--- /dev/null
+++ b/apps/backend/tests/test_analysis.py
@@ -0,0 +1,114 @@
+"""Asking for a complexity analysis of an accepted run."""
+
+from __future__ import annotations
+
+from pathlib import Path
+
+
+from app.dao.problems import ProblemDAO
+from app.dao.submissions import SubmissionDAO
+from app.runner.__main__ import analyze_next
+from app.runner.complexity import analyze
+from tests.rig import auth, room_socket, roster
+from tests.test_submissions import TWO_SUM, drain_queue, next_submission_event, submit
+
+SOLUTIONS = Path(__file__).parent / "solutions"
+
+
+def analyze_queue(client) -> int:
+ pool = client.app.state.db_pool
+ done = 0
+ while client.portal.call(analyze_next, SubmissionDAO(pool), ProblemDAO(pool)):
+ done += 1
+ return done
+
+
+def accepted_run(client, room) -> dict:
+ submission = submit(client, room, TWO_SUM).json()
+ drain_queue(client)
+ return submission
+
+
+def ask(client, room, submission_id: str, user_id: str = "user_alice"):
+ with room_socket(client, room["id"], user_id):
+ return client.post(f"/api/submissions/{submission_id}/analysis", headers=auth(user_id))
+
+
+def test_an_accepted_run_is_analyzed_and_the_room_hears_it(client, room):
+ submission = accepted_run(client, room)
+ with room_socket(client, room["id"], "user_bob") as bob:
+ roster(bob)
+ response = ask(client, room, submission["id"])
+ assert response.status_code == 200, response.text
+ assert response.json()["analysisStatus"] == "pending"
+ assert next_submission_event(bob)["analysisStatus"] == "pending"
+
+ assert analyze_queue(client) == 1
+ analyzed = next_submission_event(bob)
+ assert analyzed["analysisStatus"] == "done"
+ assert analyzed["analysis"]["complexity"] == "linear"
+ assert len(analyzed["analysis"]["points"]) >= 2
+
+
+def test_asking_twice_does_not_queue_a_second_analysis(client, room):
+ submission = accepted_run(client, room)
+ assert ask(client, room, submission["id"]).status_code == 200
+ again = ask(client, room, submission["id"], "user_bob")
+ assert again.status_code == 200, again.text
+ assert again.json()["analysisStatus"] == "pending"
+ assert analyze_queue(client) == 1
+
+
+def test_only_accepted_runs_can_be_analyzed(client, room):
+ submission = submit(client, room, "print('0 0')\n").json()
+ drain_queue(client)
+ response = ask(client, room, submission["id"])
+ assert response.status_code == 409
+ assert "accepted" in response.json()["detail"]
+
+
+def test_only_people_in_the_room_can_ask(client, room):
+ submission = accepted_run(client, room)
+ response = client.post(f"/api/submissions/{submission['id']}/analysis", headers=auth("user_bob"))
+ assert response.status_code == 403
+
+
+def test_a_problem_without_a_generator_cannot_be_analyzed(client, room):
+ problem = client.get("/api/problems/slug/climbing-stairs", headers=auth("user_alice")).json()
+ assert problem["analyzable"] is False
+ with room_socket(client, room["id"], "user_alice"):
+ submission = client.post(
+ "/api/submissions",
+ json={
+ "roomId": room["id"],
+ "problemId": problem["id"],
+ "language": "python",
+ "code": (SOLUTIONS / "climbing-stairs.py").read_text(),
+ },
+ headers=auth("user_alice"),
+ ).json()
+ drain_queue(client)
+ response = ask(client, room, submission["id"])
+ assert response.status_code == 409
+ assert "does not support" in response.json()["detail"]
+
+
+def test_an_analysis_left_running_by_a_dead_runner_is_requeued(client, room):
+ submission = accepted_run(client, room)
+ ask(client, room, submission["id"])
+ dao = SubmissionDAO(client.app.state.db_pool)
+ assert str(client.portal.call(dao.claim_pending_analysis)["id"]) == submission["id"]
+ assert analyze_queue(client) == 0
+ assert client.portal.call(dao.requeue_running_analyses) == 1
+ assert analyze_queue(client) == 1
+
+
+def test_every_reference_solution_has_the_expected_growth(client):
+ """A problem's generator is only right if its reference solution measures as expected."""
+ dao = ProblemDAO(client.app.state.db_pool)
+ analyzable = [problem for problem in client.portal.call(dao.list_all) if problem["analyzable"]]
+ assert analyzable
+ for problem in analyzable:
+ spec = client.portal.call(dao.get_analysis_spec, problem["id"])
+ result = analyze((SOLUTIONS / f"{problem['slug']}.py").read_text(), spec["generator"], spec["memLimitMb"])
+ assert result["complexity"] == problem["expectedComplexity"], (problem["slug"], result)
diff --git a/apps/backend/tests/test_complexity.py b/apps/backend/tests/test_complexity.py
new file mode 100644
index 0000000..aea2b5d
--- /dev/null
+++ b/apps/backend/tests/test_complexity.py
@@ -0,0 +1,43 @@
+"""Estimating growth from timed runs, with no database in sight."""
+
+from __future__ import annotations
+
+from app.runner.complexity import analyze, classify
+
+COUNT_TO_N = "n = int(input())\ntotal = 0\nfor i in range(n):\n total += i\nprint(total)\n"
+PAIRS_UP_TO_N = (
+ "n = int(input())\ntotal = 0\nfor i in range(n):\n for j in range(n):\n total += 1\nprint(total)\n"
+)
+
+
+def sizes(start: int, factor: float, count: int) -> str:
+ return f"SIZES = [int({start} * {factor} ** k) for k in range({count})]\ndef generate(n):\n return f'{{n}}\\n'\n"
+
+
+def test_classify_reads_the_power_of_n():
+ for power, name in [(0, "constant"), (1, "linear"), (2, "quadratic"), (3, "cubic"), (4, "worse")]:
+ points = [(n, float(n**power)) for n in (1000, 2000, 4000, 8000)]
+ assert classify(points)[0] == name, power
+
+
+def test_a_single_loop_is_linear():
+ result = analyze(COUNT_TO_N, sizes(20_000, 2, 10), 256)
+ assert result["complexity"] == "linear", result
+
+
+def test_a_nested_loop_is_quadratic():
+ result = analyze(PAIRS_UP_TO_N, sizes(300, 2**0.5, 14), 256)
+ assert result["complexity"] == "quadratic", result
+
+
+def test_a_program_that_crashes_on_big_inputs_says_where():
+ code = "n = int(input())\nif n > 5000:\n raise ValueError('too big')\nprint(n)\n"
+ result = analyze(code, sizes(1000, 2, 5), 256)
+ assert result["complexity"] is None
+ assert "n = 8000" in result["note"] and "too big" in result["note"]
+
+
+def test_a_program_too_fast_to_measure_barely_grows():
+ result = analyze("input()\nprint(1)\n", sizes(1000, 2, 5), 256)
+ assert result["complexity"] == "constant"
+ assert result["points"] == []
diff --git a/apps/backend/tests/test_sandbox.py b/apps/backend/tests/test_sandbox.py
index bb12555..a45c753 100644
--- a/apps/backend/tests/test_sandbox.py
+++ b/apps/backend/tests/test_sandbox.py
@@ -8,7 +8,7 @@
import pytest
from app.runner.judge import ACCEPTED, RUNTIME_ERROR
-from app.runner.sandbox import judge_in_container, pull
+from app.runner.sandbox import analyze_in_container, judge_in_container, pull
IMAGE = "python:3.11-slim"
@@ -57,3 +57,10 @@ def test_program_cannot_write_to_the_judges_stdout():
verdict = judge_in_container(IMAGE, code, TESTS, 2000, 128)
assert verdict.status == RUNTIME_ERROR
assert "PermissionError" in verdict.tests[0].stderr
+
+
+def test_growth_is_measured_in_the_container():
+ code = "n = int(input())\ntotal = 0\nfor i in range(n):\n total += i\nprint(total)\n"
+ generator = "SIZES = [20000 * 2 ** k for k in range(10)]\ndef generate(n):\n return f'{n}\\n'\n"
+ result = analyze_in_container(IMAGE, code, generator, 128)
+ assert result["complexity"] == "linear", result
diff --git a/apps/web/src/components/ComplexityPanel.tsx b/apps/web/src/components/ComplexityPanel.tsx
new file mode 100644
index 0000000..f40a17e
--- /dev/null
+++ b/apps/web/src/components/ComplexityPanel.tsx
@@ -0,0 +1,112 @@
+import { useState } from "react";
+import type { Submission } from "../hooks/UseWebSocket";
+import { submissionApi } from "../lib/api";
+import { button, muted } from "../lib/ui";
+
+// Growth classes from the runner, slowest-growing first. n log n measures
+// too close to n to tell apart, so the two share a class.
+const CLASSES = ["constant", "linear", "quadratic", "cubic", "worse"];
+const LABEL: Record = {
+ constant: "O(1) or O(log n)",
+ linear: "O(n) or O(n log n)",
+ quadratic: "O(n²)",
+ cubic: "O(n³)",
+ worse: "Worse than O(n³)",
+};
+
+// The pixel face has no superscript digits, so exponents are drawn raised.
+const Label = ({ name }: { name: string }) => (
+ <>
+ {LABEL[name].split(/([²³])/).map((part, i) =>
+ part === "²" || part === "³" ? {part === "²" ? 2 : 3} : part,
+ )}
+ >
+);
+
+const compact = (n: number) => (n >= 1_000_000 ? `${n / 1_000_000}M` : n >= 1000 ? `${Math.round(n / 1000)}k` : `${n}`);
+
+const Chart = ({ points }: { points: { n: number; ms: number }[] }) => {
+ const top = Math.max(...points.map((point) => point.ms));
+ return (
+
+ {points.map((point) => (
+
+ {Math.round(point.ms)} ms
+
+ n={compact(point.n)}
+
+ Estimated from CPU time on inputs of growing size. A guide, not a proof.
+
+
+ );
+ }
+
+ return (
+
+
+
+ {state === "failed" && analysis?.note ? analysis.note : "Reruns this solution on bigger inputs to see how its time grows."}
+
+ {error && {error}}
+
)}
{shown && }
+ {shown?.status === "accepted" && problem?.analyzable && shown.problemId === problem.id && (
+
+ )}
diff --git a/docs/architecture.md b/docs/architecture.md
index ed0d276..3c59c93 100644
--- a/docs/architecture.md
+++ b/docs/architecture.md
@@ -37,6 +37,8 @@ Rooms are public or unlisted, and the owner can ask for a partner, which highlig
2. The runner (`app/runner/__main__.py`) claims the oldest pending submission, judges it (see [judge and sandbox](judge-and-sandbox.md)), and stores the verdict.
3. Storing a verdict sends a Postgres `NOTIFY`. The API listens for it (`app/websocket/verdicts.py`) and broadcasts the verdict to the room.
+Complexity analysis of an accepted run (`POST /api/submissions/{id}/analysis`) goes through the same queue and the same notification, and the runner takes it only when no submission is waiting. See [judge and sandbox](judge-and-sandbox.md).
+
The runner polls for work and handles one submission at a time. A submission left running by a runner that died is put back in the queue when a runner starts.
## Replay
diff --git a/docs/judge-and-sandbox.md b/docs/judge-and-sandbox.md
index 1f37ba0..8e3f83a 100644
--- a/docs/judge-and-sandbox.md
+++ b/docs/judge-and-sandbox.md
@@ -28,6 +28,18 @@ The runner starts containers through the host's Docker socket, which makes the r
Without `SANDBOX_IMAGE`, the runner refuses to start unless `ALLOW_UNSANDBOXED=1` is set. Then it judges in its own process with resource limits only, which the test suite uses and nothing else should.
+## Complexity analysis
+
+On request, an accepted run can be rerun on inputs of growing size to estimate how its running time grows. The problem's generator (`complexity_generator`, see [CONTRIBUTING.md](../CONTRIBUTING.md)) makes the inputs, and `app/runner/complexity.py` runs the program on each size in one sandbox container.
+
+- It measures CPU time from outside the program. Counting executed lines would miss work done inside built-ins, such as `x in some_list`, and would call a quadratic brute force linear. The program also can't report a time of its own.
+- The fixed cost of every run (interpreter start, imports, reading input) is measured on a tiny input and subtracted.
+- It fits the times to a power of n and names the class. n log n measures too close to n to tell apart, so the two share a class.
+- Each analysis has a time budget, and each size a limit (`BUDGET_SECONDS` and `PER_RUN_SECONDS` in `complexity.py`). A solution that outgrows them is stopped, and the note says where.
+- The runner analyzes only when no submission is waiting, so an analysis never holds up a verdict by more than the one in progress.
+
+The result is an estimate, and the app says so. Nothing about it affects the verdict.
+
## Reporting a problem with it
A way out of the sandbox, or a way to see or change other people's runs, is a security issue. Report it privately as described in [SECURITY.md](../SECURITY.md).
From 6475893dab674e0c8bc3a0e3b732cda55ad26f4d Mon Sep 17 00:00:00 2001
From: Naman Rusia
Date: Fri, 25 Sep 2026 14:13:41 -0400
Subject: [PATCH 2/2] Make a failed analysis final, so retries cannot slip past
the hourly limit
---
apps/backend/app/dao/submissions.py | 8 +++-----
apps/backend/app/runner/__main__.py | 2 +-
apps/backend/app/services/submissions.py | 9 ++++++---
apps/backend/tests/test_analysis.py | 15 +++++++++++++++
apps/web/src/components/ComplexityPanel.tsx | 14 ++++++++++----
5 files changed, 35 insertions(+), 13 deletions(-)
diff --git a/apps/backend/app/dao/submissions.py b/apps/backend/app/dao/submissions.py
index c47cbf7..ad71726 100644
--- a/apps/backend/app/dao/submissions.py
+++ b/apps/backend/app/dao/submissions.py
@@ -131,13 +131,11 @@ async def count_analyses_since(self, user_id: str, seconds: int) -> int:
return await conn.fetchval(query, user_id, seconds)
async def request_analysis(self, submission_id: UUID, user_id: str) -> dict | None:
- """Queue an analysis of an accepted run, unless one is queued or done already."""
+ """Queue an analysis of an accepted run that has never had one."""
query = """
UPDATE submissions
- SET analysis_status = 'pending', analysis = NULL,
- analysis_requested_by = $2, analysis_requested_at = NOW()
- WHERE id = $1 AND status = 'accepted'
- AND (analysis_status IS NULL OR analysis_status = 'failed')
+ SET analysis_status = 'pending', analysis_requested_by = $2, analysis_requested_at = NOW()
+ WHERE id = $1 AND status = 'accepted' AND analysis_status IS NULL
RETURNING id
"""
async with self.pool.acquire() as conn:
diff --git a/apps/backend/app/runner/__main__.py b/apps/backend/app/runner/__main__.py
index 6805b71..4f7b6c6 100644
--- a/apps/backend/app/runner/__main__.py
+++ b/apps/backend/app/runner/__main__.py
@@ -60,7 +60,7 @@ async def analyze_next(submissions: SubmissionDAO, problems: ProblemDAO) -> bool
analysis = await asyncio.to_thread(run_analysis, submission["code"] or "", spec["generator"], spec["memLimitMb"])
except Exception:
logger.exception("Analysis failed for submission %s", submission["id"])
- analysis = {"points": [], "complexity": None, "slope": None, "note": "The analysis could not run. Try again later."}
+ analysis = {"points": [], "complexity": None, "slope": None, "note": "The analysis could not run."}
status = "done" if analysis["complexity"] else "failed"
try:
diff --git a/apps/backend/app/services/submissions.py b/apps/backend/app/services/submissions.py
index 5ad5b61..6ad2dfa 100644
--- a/apps/backend/app/services/submissions.py
+++ b/apps/backend/app/services/submissions.py
@@ -126,9 +126,12 @@ async def request_analysis(self, submission_id, caller_id: str) -> dict:
status_code=status.HTTP_409_CONFLICT,
detail="This problem does not support complexity analysis yet",
)
- # Both people in a room can press the button; the second press sees
- # the first one's analysis instead of queueing another.
- if submission["analysisStatus"] in ("pending", "running", "done"):
+ # One analysis per run, whatever it found. Both people in a room can
+ # press the button, and the second press sees the first one's. A
+ # failed one is final too: the hourly limit counts runs, so retries of
+ # one run would never reach it, and the same program on the same
+ # inputs would fail the same way.
+ if submission["analysisStatus"] is not None:
return submission
if await self.submission_dao.count_analyses_since(caller_id, 3600) >= ANALYSES_PER_HOUR:
raise HTTPException(
diff --git a/apps/backend/tests/test_analysis.py b/apps/backend/tests/test_analysis.py
index 1ab728b..00803f9 100644
--- a/apps/backend/tests/test_analysis.py
+++ b/apps/backend/tests/test_analysis.py
@@ -59,6 +59,21 @@ def test_asking_twice_does_not_queue_a_second_analysis(client, room):
assert analyze_queue(client) == 1
+def test_a_failed_analysis_is_final(client, room):
+ """Retries would reuse the run's row, which the hourly limit counts once."""
+ submission = accepted_run(client, room)
+ ask(client, room, submission["id"])
+ dao = SubmissionDAO(client.app.state.db_pool)
+ claimed = client.portal.call(dao.claim_pending_analysis)
+ failed = {"points": [], "complexity": None, "slope": None, "note": "Stopped at n = 1000."}
+ client.portal.call(dao.complete_analysis, claimed["id"], "failed", failed)
+
+ again = ask(client, room, submission["id"])
+ assert again.status_code == 200, again.text
+ assert again.json()["analysisStatus"] == "failed"
+ assert analyze_queue(client) == 0
+
+
def test_only_accepted_runs_can_be_analyzed(client, room):
submission = submit(client, room, "print('0 0')\n").json()
drain_queue(client)
diff --git a/apps/web/src/components/ComplexityPanel.tsx b/apps/web/src/components/ComplexityPanel.tsx
index f40a17e..d5a9729 100644
--- a/apps/web/src/components/ComplexityPanel.tsx
+++ b/apps/web/src/components/ComplexityPanel.tsx
@@ -96,14 +96,20 @@ const ComplexityPanel = ({ submission, expected }: { submission: Submission; exp
);
}
+ if (state === "failed") {
+ return (
+
+ No estimate for this run. {analysis?.note ?? ""}
+
+ );
+ }
+
return (
-
- {state === "failed" && analysis?.note ? analysis.note : "Reruns this solution on bigger inputs to see how its time grows."}
-
+ Reruns this solution on bigger inputs to see how its time grows.
{error && {error}}