From fc04ead8786fa8e160892aa3ec24664a8f5ff591 Mon Sep 17 00:00:00 2001 From: Naman Rusia Date: Fri, 25 Sep 2026 14:03:34 -0400 Subject: [PATCH 1/2] Estimate the time complexity of an accepted run on request An accepted run can be rerun on inputs of growing size from a per-problem generator. The runner measures each size's CPU time in one sandbox container, fits it to a power of n and names the class, and the room sees the result next to the problem's expected class. Analyses only run when no submission is waiting, within a time budget, one per person at a time and 20 an hour. Two Sum, Valid Parentheses, Best Time to Buy and Sell Stock and 3Sum support it to start. --- CONTRIBUTING.md | 16 ++ apps/backend/app/dao/problems.py | 15 +- apps/backend/app/dao/submissions.py | 57 ++++++- apps/backend/app/routes/submissions.py | 15 ++ apps/backend/app/runner/__main__.py | 42 ++++- apps/backend/app/runner/complexity.py | 161 ++++++++++++++++++ apps/backend/app/runner/sandbox.py | 42 +++-- apps/backend/app/schemas/problems.py | 2 + apps/backend/app/schemas/submissions.py | 14 ++ apps/backend/app/services/submissions.py | 48 ++++++ .../migrations/V13__complexity_analysis.sql | 94 ++++++++++ apps/backend/tests/conftest.py | 3 + apps/backend/tests/test_analysis.py | 114 +++++++++++++ apps/backend/tests/test_complexity.py | 43 +++++ apps/backend/tests/test_sandbox.py | 9 +- apps/web/src/components/ComplexityPanel.tsx | 112 ++++++++++++ apps/web/src/hooks/UseWebSocket.ts | 10 ++ apps/web/src/lib/api.ts | 6 + apps/web/src/routes/rooms/RoomView.tsx | 6 + docs/architecture.md | 2 + docs/judge-and-sandbox.md | 12 ++ 21 files changed, 806 insertions(+), 17 deletions(-) create mode 100644 apps/backend/app/runner/complexity.py create mode 100644 apps/backend/migrations/V13__complexity_analysis.sql create mode 100644 apps/backend/tests/test_analysis.py create mode 100644 apps/backend/tests/test_complexity.py create mode 100644 apps/web/src/components/ComplexityPanel.tsx diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index f825145..5656b5a 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -125,6 +125,22 @@ in your migration, hidden ones included. It fails if: - the starter code does not compile, has no `Solution` class, or passes the tests without being filled in +### Optional: support complexity analysis + +An accepted run can be rerun on bigger inputs to estimate how its time grows. +A problem supports this when its migration also sets two columns: + +- `complexity_generator`: Python source defining `SIZES`, the input sizes to + try in order, and `generate(n)`, which returns the stdin for an input of + size `n`. Make the inputs force a full solve, for example by putting the + answer at the end. +- `expected_complexity`: the class your reference solution achieves, one of + `constant`, `linear`, `quadratic` or `cubic`. + +The test suite measures your reference solution with your generator and fails +if it does not land in the expected class. `V13__complexity_analysis.sql` has +examples. + ### 5. Check it Run `make test`. To see the problem in the app, diff --git a/apps/backend/app/dao/problems.py b/apps/backend/app/dao/problems.py index c8bab92..87e6218 100644 --- a/apps/backend/app/dao/problems.py +++ b/apps/backend/app/dao/problems.py @@ -8,7 +8,10 @@ # The full test list never leaves the DAO through here: only the public # samples do, so a client cannot read the hidden tests off the API. -COLUMNS = "id, slug, title, difficulty, time_limit_ms, mem_limit_mb, statement, starter_code, tests" +COLUMNS = ( + "id, slug, title, difficulty, time_limit_ms, mem_limit_mb, statement, starter_code, tests, " + "complexity_generator IS NOT NULL AS analyzable, expected_complexity" +) def _map_problem(row: asyncpg.Record) -> dict: @@ -25,6 +28,8 @@ def _map_problem(row: asyncpg.Record) -> dict: "samples": [ {"input": t["input"], "expected": t["expected"]} for t in tests if not t.get("hidden") ], + "analyzable": row["analyzable"], + "expectedComplexity": row["expected_complexity"], } @@ -84,6 +89,14 @@ async def get_judge_spec(self, problem_id: UUID) -> dict: "memLimitMb": row["mem_limit_mb"], } + async def get_analysis_spec(self, problem_id: UUID) -> dict | None: + query = "SELECT complexity_generator, mem_limit_mb FROM problems WHERE id = $1" + async with self.pool.acquire() as conn: + row = await conn.fetchrow(query, problem_id) + if row is None or row["complexity_generator"] is None: + return None + return {"generator": row["complexity_generator"], "memLimitMb": row["mem_limit_mb"]} + async def exists(self, problem_id: UUID) -> bool: query = "SELECT EXISTS(SELECT 1 FROM problems WHERE id = $1)" async with self.pool.acquire() as conn: diff --git a/apps/backend/app/dao/submissions.py b/apps/backend/app/dao/submissions.py index 7c70840..c47cbf7 100644 --- a/apps/backend/app/dao/submissions.py +++ b/apps/backend/app/dao/submissions.py @@ -9,7 +9,7 @@ SELECT_SUBMISSION = """ SELECT s.id, s.room_id, s.user_id, s.problem_id, s.language, s.code, s.status, s.time_ms, s.created_at, s.s3_key_stdout, s.s3_key_stderr, - s.s3_key_result_json, s.result, u.name AS user_name + s.s3_key_result_json, s.result, s.analysis_status, s.analysis, u.name AS user_name FROM submissions s JOIN users u ON u.id = s.user_id """ @@ -40,6 +40,8 @@ def _map_submission(row: asyncpg.Record) -> dict: "s3KeyStderr": row["s3_key_stderr"], "s3KeyResultJson": row["s3_key_result_json"], "result": json.loads(row["result"]) if row["result"] else None, + "analysisStatus": row["analysis_status"], + "analysis": json.loads(row["analysis"]) if row["analysis"] else None, } @@ -119,3 +121,56 @@ async def requeue_running(self) -> int: async with self.pool.acquire() as conn: tag = await conn.execute("UPDATE submissions SET status = 'pending' WHERE status = 'running'") return int(tag.rsplit(" ", 1)[1]) + + async def count_analyses_since(self, user_id: str, seconds: int) -> int: + query = """ + SELECT COUNT(*) FROM submissions + WHERE analysis_requested_by = $1 AND analysis_requested_at > NOW() - make_interval(secs => $2) + """ + async with self.pool.acquire() as conn: + return await conn.fetchval(query, user_id, seconds) + + async def request_analysis(self, submission_id: UUID, user_id: str) -> dict | None: + """Queue an analysis of an accepted run, unless one is queued or done already.""" + query = """ + UPDATE submissions + SET analysis_status = 'pending', analysis = NULL, + analysis_requested_by = $2, analysis_requested_at = NOW() + WHERE id = $1 AND status = 'accepted' + AND (analysis_status IS NULL OR analysis_status = 'failed') + RETURNING id + """ + async with self.pool.acquire() as conn: + updated = await conn.fetchval(query, submission_id, user_id) + return await self.get_by_id(updated) if updated else None + + async def claim_pending_analysis(self) -> dict | None: + query = """ + UPDATE submissions SET analysis_status = 'running' + WHERE id = ( + SELECT id FROM submissions + WHERE analysis_status = 'pending' + ORDER BY analysis_requested_at + LIMIT 1 + FOR UPDATE SKIP LOCKED + ) + RETURNING id + """ + async with self.pool.acquire() as conn: + submission_id = await conn.fetchval(query) + return await self.get_by_id(submission_id) if submission_id else None + + async def complete_analysis(self, submission_id: UUID, status: str, analysis: dict) -> None: + query = "UPDATE submissions SET analysis_status = $2, analysis = $3::jsonb WHERE id = $1" + async with self.pool.acquire() as conn: + async with conn.transaction(): + await conn.execute(query, submission_id, status, json.dumps(analysis)) + # The room hears about it the way it hears about a verdict. + await conn.execute("SELECT pg_notify($1, $2)", JUDGED_CHANNEL, str(submission_id)) + + async def requeue_running_analyses(self) -> int: + async with self.pool.acquire() as conn: + tag = await conn.execute( + "UPDATE submissions SET analysis_status = 'pending' WHERE analysis_status = 'running'" + ) + return int(tag.rsplit(" ", 1)[1]) diff --git a/apps/backend/app/routes/submissions.py b/apps/backend/app/routes/submissions.py index f1db71f..31a39d2 100644 --- a/apps/backend/app/routes/submissions.py +++ b/apps/backend/app/routes/submissions.py @@ -43,6 +43,21 @@ async def list_submissions( return [SubmissionResponse.model_validate(submission) for submission in submissions] +@router.post("/{submission_id}/analysis", response_model=SubmissionResponse) +async def request_analysis( + submission_id: UUID, + request: Request, + caller_id: str = Depends(current_user_id), + service: SubmissionService = Depends(get_submission_service), +) -> SubmissionResponse: + submission = await service.request_analysis(submission_id, caller_id) + # Everyone in the room sees it start; the result follows from the listener. + await request.app.state.room_chat_manager.broadcast( + submission["roomId"], {"type": "submission", "submission": submission} + ) + return SubmissionResponse.model_validate(submission) + + @router.get("/{submission_id}", response_model=SubmissionResponse) async def get_submission( submission_id: UUID, diff --git a/apps/backend/app/runner/__main__.py b/apps/backend/app/runner/__main__.py index 24cd9c8..6805b71 100644 --- a/apps/backend/app/runner/__main__.py +++ b/apps/backend/app/runner/__main__.py @@ -1,5 +1,9 @@ """Claim pending submissions and judge them, one at a time. +Complexity analyses share the runner but only run when no submission is +waiting, so asking for one never delays anyone's verdict by more than the +analysis already in progress. + The submissions table is the queue: a row is claimed by moving it to ``running`` under ``FOR UPDATE SKIP LOCKED``, so several runners can share a database without judging the same submission twice. @@ -16,6 +20,7 @@ from app.dao.submissions import SubmissionDAO from app.database import create_pool from app.runner import sandbox +from app.runner.complexity import analyze from app.runner.judge import RUNTIME_ERROR, Verdict, judge logger = logging.getLogger(__name__) @@ -36,6 +41,40 @@ def run_judge(code: str, tests: list[dict], time_limit_ms: int, mem_limit_mb: in return judge(code, tests, time_limit_ms, mem_limit_mb) +def run_analysis(code: str, generator: str, mem_limit_mb: int) -> dict: + if SANDBOX_IMAGE: + return sandbox.analyze_in_container(SANDBOX_IMAGE, code, generator, mem_limit_mb) + return analyze(code, generator, mem_limit_mb) + + +async def analyze_next(submissions: SubmissionDAO, problems: ProblemDAO) -> bool: + """Analyze one accepted run if one is waiting. Returns whether it did.""" + submission = await submissions.claim_pending_analysis() + if submission is None: + return False + + try: + spec = await problems.get_analysis_spec(submission["problemId"]) + if spec is None: + raise RuntimeError("the problem has no generator") + analysis = await asyncio.to_thread(run_analysis, submission["code"] or "", spec["generator"], spec["memLimitMb"]) + except Exception: + logger.exception("Analysis failed for submission %s", submission["id"]) + analysis = {"points": [], "complexity": None, "slope": None, "note": "The analysis could not run. Try again later."} + + status = "done" if analysis["complexity"] else "failed" + try: + await submissions.complete_analysis(submission["id"], status, analysis) + except Exception: + logger.exception("Could not store the analysis for submission %s", submission["id"]) + status = "failed" + await submissions.complete_analysis( + submission["id"], status, {"points": [], "complexity": None, "slope": None, "note": "The analysis could not be saved."} + ) + logger.info("Submission %s analysis: %s", submission["id"], analysis["complexity"] or status) + return True + + async def judge_next(submissions: SubmissionDAO, problems: ProblemDAO) -> bool: """Judge one submission if there is one waiting. Returns whether it did.""" submission = await submissions.claim_pending() @@ -80,9 +119,10 @@ async def main() -> None: # claimed again. One runner at a time is the local setup, so on boot # anything still running is ours from before and goes back in the queue. await submissions.requeue_running() + await submissions.requeue_running_analyses() try: while True: - if not await judge_next(submissions, problems): + if not await judge_next(submissions, problems) and not await analyze_next(submissions, problems): await asyncio.sleep(POLL_SECONDS) finally: await pool.close() diff --git a/apps/backend/app/runner/complexity.py b/apps/backend/app/runner/complexity.py new file mode 100644 index 0000000..83e23d0 --- /dev/null +++ b/apps/backend/app/runner/complexity.py @@ -0,0 +1,161 @@ +"""Estimate how a program's running time grows with the size of its input. + +The program runs on inputs of doubling size from the problem's generator, and +the CPU time of each run is fitted to a power of n. CPU time, not a count of +executed lines, because a line count sees `x in some_list` or `sorted(xs)` as +one step and would call a quadratic brute force linear. It is measured from +outside the program, so the program cannot report a time of its own. + +Like the judge, this file is shipped into a sandbox container as the program +text, so it imports nothing from the app. +""" + +from __future__ import annotations + +import math +import resource +import subprocess +import sys +import tempfile +import time +from pathlib import Path + +# Runner time one analysis may use, and the most one size may take. A run +# that reaches a second has shown its growth; bigger inputs only cost time. +BUDGET_SECONDS = 6.0 +PER_RUN_SECONDS = 2.5 +ENOUGH_SECONDS = 1.0 + +# Below this a run is mostly noise, not the solution. +MIN_MEASURABLE_MS = 20.0 + +# Fitted exponent of n -> growth class, split halfway between the powers. +# n log n fits at about 1.1 over the sizes used, too close to n to tell +# apart, so they share a class. +CLASSES = [(0.5, "constant"), (1.5, "linear"), (2.5, "quadratic"), (3.5, "cubic")] +WORSE = "worse" + +PROGRAM_ENV = {"PATH": "/usr/local/bin:/usr/bin:/bin", "LANG": "C.UTF-8"} + + +def _limits(mem_limit_mb: int): + # The same limits the judge applies; this file cannot import them. + def apply() -> None: + mem = mem_limit_mb * 1024 * 1024 + resource.setrlimit(resource.RLIMIT_AS, (mem, mem)) + resource.setrlimit(resource.RLIMIT_NPROC, (0, 0)) + resource.setrlimit(resource.RLIMIT_FSIZE, (0, 0)) + + return apply + + +def _children_cpu() -> float: + usage = resource.getrusage(resource.RUSAGE_CHILDREN) + return usage.ru_utime + usage.ru_stime + + +def classify(points: list[tuple[int, float]]) -> tuple[str, float]: + """Least-squares slope of log(ms) against log(n), and its class.""" + xs = [math.log(n) for n, _ in points] + ys = [math.log(ms) for _, ms in points] + mean_x, mean_y = sum(xs) / len(xs), sum(ys) / len(ys) + spread = sum((x - mean_x) ** 2 for x in xs) + slope = sum((x - mean_x) * (y - mean_y) for x, y in zip(xs, ys)) / spread + for bound, name in CLASSES: + if slope < bound: + return name, slope + return WORSE, slope + + +def _run(program: Path, workdir: str, stdin: str, timeout: float, mem_limit_mb: int): + """CPU and wall seconds for one run, or None and a reason if it did not finish.""" + cpu_before, started = _children_cpu(), time.perf_counter() + try: + completed = subprocess.run( + [sys.executable, "-I", str(program)], + input=stdin.encode(), + # Nothing is checked here, and a large input can mean a large answer. + stdout=subprocess.DEVNULL, + stderr=subprocess.PIPE, + timeout=timeout, + cwd=workdir, + env=PROGRAM_ENV, + preexec_fn=_limits(mem_limit_mb), + ) + except subprocess.TimeoutExpired: + return None, f"took over {timeout:.1f} s" + wall = time.perf_counter() - started + if completed.returncode != 0: + # Postgres rejects NUL in jsonb, and a result that cannot be stored + # would leave the runner retrying the same analysis forever. + last = completed.stderr.decode("utf-8", errors="replace").replace("\x00", "").strip().splitlines() + return None, f"crashed: {last[-1][:200]}" if last else "crashed" + # Some sandboxes do not account children's CPU; wall time is the fallback. + cpu = _children_cpu() - cpu_before + return (cpu if cpu > 0 else wall), None + + +def analyze(code: str, generator: str, mem_limit_mb: int) -> dict: + namespace: dict = {} + exec(generator, namespace) + generate, sizes = namespace["generate"], namespace["SIZES"] + + points: list[tuple[int, float]] = [] + note = None + spent = 0.0 + with tempfile.TemporaryDirectory() as workdir: + program = Path(workdir) / "main.py" + program.write_text(code) + # Every run pays for starting the interpreter, the program's imports + # and reading its input. The program on a tiny input costs about that + # and nothing more; left in, it flattens the growth of every run. + tiny = generate(max(4, sizes[0] // 32)) + startup = min(_run(program, workdir, tiny, PER_RUN_SECONDS, mem_limit_mb)[0] or 0.0 for _ in range(3)) + + for n in sizes: + remaining = BUDGET_SECONDS - spent + if remaining <= 0.1: + note = f"Stopped before n = {n}: out of time for this analysis." + break + seconds, failure = _run(program, workdir, generate(n), min(PER_RUN_SECONDS, remaining), mem_limit_mb) + if seconds is None: + note = f"Stopped at n = {n}: it {failure}." + break + spent += seconds + ms = (seconds - startup) * 1000 + if ms >= MIN_MEASURABLE_MS: + points.append((n, ms)) + if seconds >= ENOUGH_SECONDS: + break + + result = { + "points": [{"n": n, "ms": round(ms, 1)} for n, ms in points], + "complexity": None, + "slope": None, + "note": note, + } + if len(points) >= 2: + result["complexity"], slope = classify(points) + result["slope"] = round(slope, 2) + if len(points) == 2: + result["note"] = (note + " " if note else "") + "Only two sizes were measurable, so this is rough." + elif not points and note is None: + # Even the largest input ran too fast to measure: it barely grows. + result["complexity"], result["slope"] = "constant", 0.0 + elif note is None: + result["note"] = "Only one size was measurable, which is not enough to see growth." + return result + + +if __name__ == "__main__": + # Entry point inside a sandbox container: the spec arrives on stdin and + # the result leaves on stdout, both as JSON. + import ctypes + import json + + # As in the judge: the program must not be able to print our result for us. + PR_SET_DUMPABLE = 4 + ctypes.CDLL(None).prctl(PR_SET_DUMPABLE, 0) + + spec = json.load(sys.stdin) + print(json.dumps(analyze(spec["code"], spec["generator"], spec["memLimitMb"]))) diff --git a/apps/backend/app/runner/sandbox.py b/apps/backend/app/runner/sandbox.py index b0b5caf..548a499 100644 --- a/apps/backend/app/runner/sandbox.py +++ b/apps/backend/app/runner/sandbox.py @@ -17,15 +17,17 @@ import uuid from pathlib import Path +from app.runner import complexity from app.runner.judge import RUNTIME_ERROR, TestOutcome, Verdict logger = logging.getLogger(__name__) JUDGE_SOURCE = (Path(__file__).parent / "judge.py").read_text() +COMPLEXITY_SOURCE = (Path(__file__).parent / "complexity.py").read_text() # Room for the interpreter and the judge on top of the program's own limit. CONTAINER_MEMORY_OVERHEAD_MB = 64 -# Container start and interpreter start, on top of the tests' own limits. +# Container start and interpreter start, on top of the program's own limits. STARTUP_ALLOWANCE_SECONDS = 20 # `runsc` in production: gVisor answers the program's system calls itself, so @@ -40,9 +42,8 @@ def pull(image: str) -> None: subprocess.run(["docker", "pull", "--quiet", image], check=True, capture_output=True, timeout=600) -def judge_in_container( - image: str, code: str, tests: list[dict], time_limit_ms: int, mem_limit_mb: int -) -> Verdict: +def _run_in_container(image: str, source: str, spec: dict, mem_limit_mb: int, timeout: float) -> dict | None: + """Run a program file from this package in a fresh container; its JSON result, or None if it hung.""" name = f"judge-{uuid.uuid4().hex[:12]}" command = [ "docker", "run", "--rm", "--interactive", "--name", name, @@ -61,27 +62,32 @@ def judge_in_container( "--env", "PYTHONDONTWRITEBYTECODE=1", *(["--runtime", SANDBOX_RUNTIME] if SANDBOX_RUNTIME else []), image, - "python", "-I", "-c", JUDGE_SOURCE, + "python", "-I", "-c", source, ] - spec = json.dumps( - {"code": code, "tests": tests, "timeLimitMs": time_limit_ms, "memLimitMb": mem_limit_mb} - ) - timeout = len(tests) * time_limit_ms / 1000 + STARTUP_ALLOWANCE_SECONDS try: completed = subprocess.run( - command, input=spec.encode(), capture_output=True, timeout=timeout + command, input=json.dumps(spec).encode(), capture_output=True, timeout=timeout ) except subprocess.TimeoutExpired: - # The judge inside enforces per-test limits; reaching this means the + # The program inside enforces its own limits; reaching this means the # container itself hung. Make sure it is gone. subprocess.run(["docker", "rm", "--force", name], capture_output=True, timeout=60) logger.error("Sandbox %s exceeded %.0fs and was removed", name, timeout) - return Verdict(status=RUNTIME_ERROR, timeMs=0, passed=0, total=len(tests)) + return None if completed.returncode != 0: raise RuntimeError(f"Sandbox exited {completed.returncode}: {completed.stderr.decode(errors='replace')[:500]}") + return json.loads(completed.stdout) + - data = json.loads(completed.stdout) +def judge_in_container( + image: str, code: str, tests: list[dict], time_limit_ms: int, mem_limit_mb: int +) -> Verdict: + spec = {"code": code, "tests": tests, "timeLimitMs": time_limit_ms, "memLimitMb": mem_limit_mb} + timeout = len(tests) * time_limit_ms / 1000 + STARTUP_ALLOWANCE_SECONDS + data = _run_in_container(image, JUDGE_SOURCE, spec, mem_limit_mb, timeout) + if data is None: + return Verdict(status=RUNTIME_ERROR, timeMs=0, passed=0, total=len(tests)) return Verdict( status=data["status"], timeMs=data["timeMs"], @@ -89,3 +95,13 @@ def judge_in_container( total=data["total"], tests=[TestOutcome(**outcome) for outcome in data["tests"]], ) + + +def analyze_in_container(image: str, code: str, generator: str, mem_limit_mb: int) -> dict: + spec = {"code": code, "generator": generator, "memLimitMb": mem_limit_mb} + # Generating the inputs takes a moment on top of the measured runs. + timeout = complexity.BUDGET_SECONDS + complexity.PER_RUN_SECONDS + STARTUP_ALLOWANCE_SECONDS + data = _run_in_container(image, COMPLEXITY_SOURCE, spec, mem_limit_mb, timeout) + if data is None: + return {"points": [], "complexity": None, "slope": None, "note": "The analysis did not finish in time."} + return data diff --git a/apps/backend/app/schemas/problems.py b/apps/backend/app/schemas/problems.py index 5659b6a..773f735 100644 --- a/apps/backend/app/schemas/problems.py +++ b/apps/backend/app/schemas/problems.py @@ -20,3 +20,5 @@ class ProblemResponse(BaseModel): statement: str | None starterCode: str | None samples: list[SampleTest] + analyzable: bool = False + expectedComplexity: str | None = None diff --git a/apps/backend/app/schemas/submissions.py b/apps/backend/app/schemas/submissions.py index 4a12f8b..79ea8c3 100644 --- a/apps/backend/app/schemas/submissions.py +++ b/apps/backend/app/schemas/submissions.py @@ -34,6 +34,18 @@ class SubmissionResult(BaseModel): tests: list[TestOutcome] +class ComplexityPoint(BaseModel): + n: int + ms: float + + +class ComplexityAnalysis(BaseModel): + points: list[ComplexityPoint] + complexity: str | None + slope: float | None + note: str | None + + class SubmissionResponse(BaseModel): id: UUID roomId: str @@ -49,3 +61,5 @@ class SubmissionResponse(BaseModel): s3KeyStderr: str | None = None s3KeyResultJson: str | None = None result: SubmissionResult | None = None + analysisStatus: str | None = None + analysis: ComplexityAnalysis | None = None diff --git a/apps/backend/app/services/submissions.py b/apps/backend/app/services/submissions.py index 1d99f54..5ad5b61 100644 --- a/apps/backend/app/services/submissions.py +++ b/apps/backend/app/services/submissions.py @@ -17,6 +17,8 @@ # The runner is one process judging one run at a time. A pair practising # hard stays well under this; a script holding the runner does not. RUNS_PER_HOUR = 120 +# An analysis costs the runner several seconds, a run well under one. +ANALYSES_PER_HOUR = 20 class SubmissionService: @@ -97,6 +99,52 @@ async def get_submission(self, submission_id, caller_id: str) -> dict: await ensure_room_member(self.room_member_dao, room, caller_id) return submission + async def request_analysis(self, submission_id, caller_id: str) -> dict: + submission = await self.submission_dao.get_by_id(submission_id) + if not submission: + raise HTTPException( + status_code=status.HTTP_404_NOT_FOUND, + detail=f"Submission not found: {submission_id}", + ) + room_id = submission["roomId"] + await get_active_room(self.room_dao, room_id) + if not await self.room_member_dao.is_present(room_id, caller_id): + raise HTTPException( + status_code=status.HTTP_403_FORBIDDEN, + detail="Join the room to analyze runs in it", + ) + # A wrong answer's timings describe a program that does not solve + # the problem, so they say nothing useful. + if submission["status"] != "accepted": + raise HTTPException( + status_code=status.HTTP_409_CONFLICT, + detail="Only accepted runs can be analyzed", + ) + problem = await self.problem_dao.get_by_id(submission["problemId"]) + if not problem or not problem["analyzable"]: + raise HTTPException( + status_code=status.HTTP_409_CONFLICT, + detail="This problem does not support complexity analysis yet", + ) + # Both people in a room can press the button; the second press sees + # the first one's analysis instead of queueing another. + if submission["analysisStatus"] in ("pending", "running", "done"): + return submission + if await self.submission_dao.count_analyses_since(caller_id, 3600) >= ANALYSES_PER_HOUR: + raise HTTPException( + status_code=status.HTTP_429_TOO_MANY_REQUESTS, + detail="Too many analyses this hour. Try again later.", + ) + try: + queued = await self.submission_dao.request_analysis(submission_id, caller_id) + except asyncpg.UniqueViolationError as exc: + raise HTTPException( + status_code=status.HTTP_429_TOO_MANY_REQUESTS, + detail="Wait for your current analysis to finish", + ) from exc + # None means someone else queued it between the read and the update. + return queued or await self.submission_dao.get_by_id(submission_id) + async def list_submissions(self, room_id: str, caller_id: str, user_id: str | None = None) -> list[dict]: room = await get_active_room(self.room_dao, room_id) await ensure_room_member(self.room_member_dao, room, caller_id) diff --git a/apps/backend/migrations/V13__complexity_analysis.sql b/apps/backend/migrations/V13__complexity_analysis.sql new file mode 100644 index 0000000..79396c0 --- /dev/null +++ b/apps/backend/migrations/V13__complexity_analysis.sql @@ -0,0 +1,94 @@ +-- V13: Complexity analysis of accepted runs +-- +-- On request, an accepted run is rerun on inputs of growing size to estimate +-- how its running time grows (app/runner/complexity.py). A problem opts in +-- with a generator: Python source defining SIZES, the input sizes to try in +-- order, and generate(n), the stdin for an input of size n. Its expected +-- class is what the reference solution achieves. + +ALTER TABLE problems + ADD COLUMN complexity_generator TEXT, + ADD COLUMN expected_complexity TEXT, + ADD CONSTRAINT problems_expected_complexity_known + CHECK (expected_complexity IN ('constant', 'linear', 'quadratic', 'cubic')), + ADD CONSTRAINT problems_complexity_needs_both + CHECK ((complexity_generator IS NULL) = (expected_complexity IS NULL)); + +ALTER TABLE submissions + ADD COLUMN analysis_status TEXT + CHECK (analysis_status IN ('pending', 'running', 'done', 'failed')), + ADD COLUMN analysis JSONB, + ADD COLUMN analysis_requested_by TEXT REFERENCES users(id), + ADD COLUMN analysis_requested_at TIMESTAMPTZ; + +-- One analysis in flight per person, for the same reason as one run. +CREATE UNIQUE INDEX submissions_one_analysis_in_flight_per_user + ON submissions (analysis_requested_by) + WHERE analysis_status IN ('pending', 'running'); + +-- The only matching pair is the last two numbers, so every solution has to +-- read to the end. Every other number is even and the target is odd, which +-- makes that pair the only answer. +UPDATE problems SET expected_complexity = 'linear', complexity_generator = $gen$ +import random + +SIZES = [1000 * 2 ** k for k in range(10)] + + +def generate(n): + rng = random.Random(n) + evens = [2 * v for v in rng.sample(range(1, 10 * n), n - 1)] + even = evens.pop() + odd = 2 * rng.randrange(10 * n) + 1 + nums = evens + [even, odd] + return " ".join(map(str, nums)) + "\n" + str(even + odd) + "\n" +$gen$ +WHERE slug = 'two-sum'; + +-- Balanced, so a correct solution reads every bracket. +UPDATE problems SET expected_complexity = 'linear', complexity_generator = $gen$ +import random + +SIZES = [1000 * 2 ** k for k in range(13)] +PAIRS = {"(": ")", "[": "]", "{": "}"} + + +def generate(n): + rng = random.Random(n) + out, stack = [], [] + for i in range(n): + if stack and (len(stack) == n - i or rng.random() < 0.5): + out.append(PAIRS[stack.pop()]) + else: + stack.append(rng.choice("([{")) + out.append(stack[-1]) + return "".join(out) + "\n" +$gen$ +WHERE slug = 'valid-parentheses'; + +UPDATE problems SET expected_complexity = 'linear', complexity_generator = $gen$ +import random + +SIZES = [1000 * 2 ** k for k in range(11)] + + +def generate(n): + rng = random.Random(n) + return " ".join(str(rng.randint(1, 10000)) for _ in range(n)) + "\n" +$gen$ +WHERE slug = 'best-time-to-buy-sell-stock'; + +-- A wide range of values keeps the number of zero-sum triplets small, so the +-- time is the search, not the answer. Sizes grow by a factor of about 1.4, as +-- a cubic brute force outgrows its budget within a few doublings. +UPDATE problems SET expected_complexity = 'quadratic', complexity_generator = $gen$ +import random + +SIZES = [int(100 * 2 ** (k / 2)) for k in range(14)] + + +def generate(n): + rng = random.Random(n) + return " ".join(str(rng.randint(-10 ** 6, 10 ** 6)) for _ in range(n)) + "\n" +$gen$ +WHERE slug = 'three-sum'; diff --git a/apps/backend/tests/conftest.py b/apps/backend/tests/conftest.py index 249f25e..427cb5d 100644 --- a/apps/backend/tests/conftest.py +++ b/apps/backend/tests/conftest.py @@ -56,6 +56,9 @@ async def _settle_runs(pool) -> None: await conn.execute( "UPDATE submissions SET status = 'runtime_error' WHERE status IN ('pending', 'running')" ) + await conn.execute( + "UPDATE submissions SET analysis_status = 'failed' WHERE analysis_status IN ('pending', 'running')" + ) @pytest.fixture(autouse=True) diff --git a/apps/backend/tests/test_analysis.py b/apps/backend/tests/test_analysis.py new file mode 100644 index 0000000..1ab728b --- /dev/null +++ b/apps/backend/tests/test_analysis.py @@ -0,0 +1,114 @@ +"""Asking for a complexity analysis of an accepted run.""" + +from __future__ import annotations + +from pathlib import Path + + +from app.dao.problems import ProblemDAO +from app.dao.submissions import SubmissionDAO +from app.runner.__main__ import analyze_next +from app.runner.complexity import analyze +from tests.rig import auth, room_socket, roster +from tests.test_submissions import TWO_SUM, drain_queue, next_submission_event, submit + +SOLUTIONS = Path(__file__).parent / "solutions" + + +def analyze_queue(client) -> int: + pool = client.app.state.db_pool + done = 0 + while client.portal.call(analyze_next, SubmissionDAO(pool), ProblemDAO(pool)): + done += 1 + return done + + +def accepted_run(client, room) -> dict: + submission = submit(client, room, TWO_SUM).json() + drain_queue(client) + return submission + + +def ask(client, room, submission_id: str, user_id: str = "user_alice"): + with room_socket(client, room["id"], user_id): + return client.post(f"/api/submissions/{submission_id}/analysis", headers=auth(user_id)) + + +def test_an_accepted_run_is_analyzed_and_the_room_hears_it(client, room): + submission = accepted_run(client, room) + with room_socket(client, room["id"], "user_bob") as bob: + roster(bob) + response = ask(client, room, submission["id"]) + assert response.status_code == 200, response.text + assert response.json()["analysisStatus"] == "pending" + assert next_submission_event(bob)["analysisStatus"] == "pending" + + assert analyze_queue(client) == 1 + analyzed = next_submission_event(bob) + assert analyzed["analysisStatus"] == "done" + assert analyzed["analysis"]["complexity"] == "linear" + assert len(analyzed["analysis"]["points"]) >= 2 + + +def test_asking_twice_does_not_queue_a_second_analysis(client, room): + submission = accepted_run(client, room) + assert ask(client, room, submission["id"]).status_code == 200 + again = ask(client, room, submission["id"], "user_bob") + assert again.status_code == 200, again.text + assert again.json()["analysisStatus"] == "pending" + assert analyze_queue(client) == 1 + + +def test_only_accepted_runs_can_be_analyzed(client, room): + submission = submit(client, room, "print('0 0')\n").json() + drain_queue(client) + response = ask(client, room, submission["id"]) + assert response.status_code == 409 + assert "accepted" in response.json()["detail"] + + +def test_only_people_in_the_room_can_ask(client, room): + submission = accepted_run(client, room) + response = client.post(f"/api/submissions/{submission['id']}/analysis", headers=auth("user_bob")) + assert response.status_code == 403 + + +def test_a_problem_without_a_generator_cannot_be_analyzed(client, room): + problem = client.get("/api/problems/slug/climbing-stairs", headers=auth("user_alice")).json() + assert problem["analyzable"] is False + with room_socket(client, room["id"], "user_alice"): + submission = client.post( + "/api/submissions", + json={ + "roomId": room["id"], + "problemId": problem["id"], + "language": "python", + "code": (SOLUTIONS / "climbing-stairs.py").read_text(), + }, + headers=auth("user_alice"), + ).json() + drain_queue(client) + response = ask(client, room, submission["id"]) + assert response.status_code == 409 + assert "does not support" in response.json()["detail"] + + +def test_an_analysis_left_running_by_a_dead_runner_is_requeued(client, room): + submission = accepted_run(client, room) + ask(client, room, submission["id"]) + dao = SubmissionDAO(client.app.state.db_pool) + assert str(client.portal.call(dao.claim_pending_analysis)["id"]) == submission["id"] + assert analyze_queue(client) == 0 + assert client.portal.call(dao.requeue_running_analyses) == 1 + assert analyze_queue(client) == 1 + + +def test_every_reference_solution_has_the_expected_growth(client): + """A problem's generator is only right if its reference solution measures as expected.""" + dao = ProblemDAO(client.app.state.db_pool) + analyzable = [problem for problem in client.portal.call(dao.list_all) if problem["analyzable"]] + assert analyzable + for problem in analyzable: + spec = client.portal.call(dao.get_analysis_spec, problem["id"]) + result = analyze((SOLUTIONS / f"{problem['slug']}.py").read_text(), spec["generator"], spec["memLimitMb"]) + assert result["complexity"] == problem["expectedComplexity"], (problem["slug"], result) diff --git a/apps/backend/tests/test_complexity.py b/apps/backend/tests/test_complexity.py new file mode 100644 index 0000000..aea2b5d --- /dev/null +++ b/apps/backend/tests/test_complexity.py @@ -0,0 +1,43 @@ +"""Estimating growth from timed runs, with no database in sight.""" + +from __future__ import annotations + +from app.runner.complexity import analyze, classify + +COUNT_TO_N = "n = int(input())\ntotal = 0\nfor i in range(n):\n total += i\nprint(total)\n" +PAIRS_UP_TO_N = ( + "n = int(input())\ntotal = 0\nfor i in range(n):\n for j in range(n):\n total += 1\nprint(total)\n" +) + + +def sizes(start: int, factor: float, count: int) -> str: + return f"SIZES = [int({start} * {factor} ** k) for k in range({count})]\ndef generate(n):\n return f'{{n}}\\n'\n" + + +def test_classify_reads_the_power_of_n(): + for power, name in [(0, "constant"), (1, "linear"), (2, "quadratic"), (3, "cubic"), (4, "worse")]: + points = [(n, float(n**power)) for n in (1000, 2000, 4000, 8000)] + assert classify(points)[0] == name, power + + +def test_a_single_loop_is_linear(): + result = analyze(COUNT_TO_N, sizes(20_000, 2, 10), 256) + assert result["complexity"] == "linear", result + + +def test_a_nested_loop_is_quadratic(): + result = analyze(PAIRS_UP_TO_N, sizes(300, 2**0.5, 14), 256) + assert result["complexity"] == "quadratic", result + + +def test_a_program_that_crashes_on_big_inputs_says_where(): + code = "n = int(input())\nif n > 5000:\n raise ValueError('too big')\nprint(n)\n" + result = analyze(code, sizes(1000, 2, 5), 256) + assert result["complexity"] is None + assert "n = 8000" in result["note"] and "too big" in result["note"] + + +def test_a_program_too_fast_to_measure_barely_grows(): + result = analyze("input()\nprint(1)\n", sizes(1000, 2, 5), 256) + assert result["complexity"] == "constant" + assert result["points"] == [] diff --git a/apps/backend/tests/test_sandbox.py b/apps/backend/tests/test_sandbox.py index bb12555..a45c753 100644 --- a/apps/backend/tests/test_sandbox.py +++ b/apps/backend/tests/test_sandbox.py @@ -8,7 +8,7 @@ import pytest from app.runner.judge import ACCEPTED, RUNTIME_ERROR -from app.runner.sandbox import judge_in_container, pull +from app.runner.sandbox import analyze_in_container, judge_in_container, pull IMAGE = "python:3.11-slim" @@ -57,3 +57,10 @@ def test_program_cannot_write_to_the_judges_stdout(): verdict = judge_in_container(IMAGE, code, TESTS, 2000, 128) assert verdict.status == RUNTIME_ERROR assert "PermissionError" in verdict.tests[0].stderr + + +def test_growth_is_measured_in_the_container(): + code = "n = int(input())\ntotal = 0\nfor i in range(n):\n total += i\nprint(total)\n" + generator = "SIZES = [20000 * 2 ** k for k in range(10)]\ndef generate(n):\n return f'{n}\\n'\n" + result = analyze_in_container(IMAGE, code, generator, 128) + assert result["complexity"] == "linear", result diff --git a/apps/web/src/components/ComplexityPanel.tsx b/apps/web/src/components/ComplexityPanel.tsx new file mode 100644 index 0000000..f40a17e --- /dev/null +++ b/apps/web/src/components/ComplexityPanel.tsx @@ -0,0 +1,112 @@ +import { useState } from "react"; +import type { Submission } from "../hooks/UseWebSocket"; +import { submissionApi } from "../lib/api"; +import { button, muted } from "../lib/ui"; + +// Growth classes from the runner, slowest-growing first. n log n measures +// too close to n to tell apart, so the two share a class. +const CLASSES = ["constant", "linear", "quadratic", "cubic", "worse"]; +const LABEL: Record = { + constant: "O(1) or O(log n)", + linear: "O(n) or O(n log n)", + quadratic: "O(n²)", + cubic: "O(n³)", + worse: "Worse than O(n³)", +}; + +// The pixel face has no superscript digits, so exponents are drawn raised. +const Label = ({ name }: { name: string }) => ( + <> + {LABEL[name].split(/([²³])/).map((part, i) => + part === "²" || part === "³" ? {part === "²" ? 2 : 3} : part, + )} + +); + +const compact = (n: number) => (n >= 1_000_000 ? `${n / 1_000_000}M` : n >= 1000 ? `${Math.round(n / 1000)}k` : `${n}`); + +const Chart = ({ points }: { points: { n: number; ms: number }[] }) => { + const top = Math.max(...points.map((point) => point.ms)); + return ( +
+ {points.map((point) => ( +
+ {Math.round(point.ms)} ms +
+ n={compact(point.n)} +
+ ))} +
+ ); +}; + +const ComplexityPanel = ({ submission, expected }: { submission: Submission; expected: string | null }) => { + const [asking, setAsking] = useState(false); + const [error, setError] = useState(null); + const { analysisStatus: state, analysis } = submission; + + const ask = async () => { + setAsking(true); + setError(null); + try { + await submissionApi.requestAnalysis(submission.id); + } catch (err) { + const detail = (err as { response?: { data?: { detail?: unknown } } }).response?.data?.detail; + setError(typeof detail === "string" ? detail : "The analysis could not be started."); + } finally { + setAsking(false); + } + }; + + if (state === "pending" || state === "running") { + return ( +
+ Measuring on bigger inputs + + {Array.from({ length: 5 }, (_, i) => ( + + ))} + +
+ ); + } + + if (state === "done" && analysis?.complexity) { + const measured = CLASSES.indexOf(analysis.complexity); + const target = expected ? CLASSES.indexOf(expected) : -1; + const slower = target >= 0 && measured > target; + return ( +
+
+ + + {expected && ( + + {slower ? "Slower than" : "Matches"} the expected {LABEL[expected]} + + )} +
+ {analysis.points.length > 0 && } + {analysis.note &&

{analysis.note}

} +

+ Estimated from CPU time on inputs of growing size. A guide, not a proof. +

+
+ ); + } + + return ( +
+ + + {state === "failed" && analysis?.note ? analysis.note : "Reruns this solution on bigger inputs to see how its time grows."} + + {error && {error}} +
+ ); +}; + +export default ComplexityPanel; diff --git a/apps/web/src/hooks/UseWebSocket.ts b/apps/web/src/hooks/UseWebSocket.ts index 5c315ea..06aa6de 100644 --- a/apps/web/src/hooks/UseWebSocket.ts +++ b/apps/web/src/hooks/UseWebSocket.ts @@ -33,10 +33,20 @@ export interface TestOutcome { export interface Submission { id: string; userId: string; + problemId: string; userName: string | null; status: string; createdAt: string; result: { passed: number; total: number; timeMs: number; tests: TestOutcome[] } | null; + analysisStatus?: "pending" | "running" | "done" | "failed" | null; + analysis?: ComplexityAnalysis | null; +} + +export interface ComplexityAnalysis { + points: { n: number; ms: number }[]; + complexity: string | null; + slope: number | null; + note: string | null; } // Newest first; a submission arrives once as pending and again judged. diff --git a/apps/web/src/lib/api.ts b/apps/web/src/lib/api.ts index 8d20fe7..ee075b2 100644 --- a/apps/web/src/lib/api.ts +++ b/apps/web/src/lib/api.ts @@ -124,6 +124,12 @@ export const submissionApi = { return response.data; }, + // The result arrives over the room socket, like a verdict. + requestAnalysis: async (id: string) => { + const response = await api.post(`/submissions/${id}/analysis`); + return response.data; + }, + getSubmissionsForRoom: async (roomId: string, userId?: string) => { const params = userId ? { userId } : {}; const response = await api.get(`/submissions/room/${roomId}`, { params }); diff --git a/apps/web/src/routes/rooms/RoomView.tsx b/apps/web/src/routes/rooms/RoomView.tsx index 096b235..e810a93 100644 --- a/apps/web/src/routes/rooms/RoomView.tsx +++ b/apps/web/src/routes/rooms/RoomView.tsx @@ -6,6 +6,7 @@ import RequireSignIn from "../../components/RequireSignIn"; import RoomChatComponent from "../../components/RoomChatComponent"; import RoomMembersPanel from "../../components/RoomMembersPanel"; import CollaborativeEditor from "../../components/CollaborativeEditor"; +import ComplexityPanel from "../../components/ComplexityPanel"; import InviteButton from "../../components/InviteButton"; import { useUser } from "../../hooks/useUser"; import useWebSocket from "../../hooks/UseWebSocket"; @@ -25,6 +26,8 @@ type Problem = { statement: string | null; starterCode: string | null; samples: { input: string; expected: string }[]; + analyzable: boolean; + expectedComplexity: string | null; }; type Room = { @@ -455,6 +458,9 @@ const Room = ({ roomId }: { roomId: string }) => {

)} {shown && } + {shown?.status === "accepted" && problem?.analyzable && shown.problemId === problem.id && ( + + )}
diff --git a/docs/architecture.md b/docs/architecture.md index ed0d276..3c59c93 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -37,6 +37,8 @@ Rooms are public or unlisted, and the owner can ask for a partner, which highlig 2. The runner (`app/runner/__main__.py`) claims the oldest pending submission, judges it (see [judge and sandbox](judge-and-sandbox.md)), and stores the verdict. 3. Storing a verdict sends a Postgres `NOTIFY`. The API listens for it (`app/websocket/verdicts.py`) and broadcasts the verdict to the room. +Complexity analysis of an accepted run (`POST /api/submissions/{id}/analysis`) goes through the same queue and the same notification, and the runner takes it only when no submission is waiting. See [judge and sandbox](judge-and-sandbox.md). + The runner polls for work and handles one submission at a time. A submission left running by a runner that died is put back in the queue when a runner starts. ## Replay diff --git a/docs/judge-and-sandbox.md b/docs/judge-and-sandbox.md index 1f37ba0..8e3f83a 100644 --- a/docs/judge-and-sandbox.md +++ b/docs/judge-and-sandbox.md @@ -28,6 +28,18 @@ The runner starts containers through the host's Docker socket, which makes the r Without `SANDBOX_IMAGE`, the runner refuses to start unless `ALLOW_UNSANDBOXED=1` is set. Then it judges in its own process with resource limits only, which the test suite uses and nothing else should. +## Complexity analysis + +On request, an accepted run can be rerun on inputs of growing size to estimate how its running time grows. The problem's generator (`complexity_generator`, see [CONTRIBUTING.md](../CONTRIBUTING.md)) makes the inputs, and `app/runner/complexity.py` runs the program on each size in one sandbox container. + +- It measures CPU time from outside the program. Counting executed lines would miss work done inside built-ins, such as `x in some_list`, and would call a quadratic brute force linear. The program also can't report a time of its own. +- The fixed cost of every run (interpreter start, imports, reading input) is measured on a tiny input and subtracted. +- It fits the times to a power of n and names the class. n log n measures too close to n to tell apart, so the two share a class. +- Each analysis has a time budget, and each size a limit (`BUDGET_SECONDS` and `PER_RUN_SECONDS` in `complexity.py`). A solution that outgrows them is stopped, and the note says where. +- The runner analyzes only when no submission is waiting, so an analysis never holds up a verdict by more than the one in progress. + +The result is an estimate, and the app says so. Nothing about it affects the verdict. + ## Reporting a problem with it A way out of the sandbox, or a way to see or change other people's runs, is a security issue. Report it privately as described in [SECURITY.md](../SECURITY.md). From 6475893dab674e0c8bc3a0e3b732cda55ad26f4d Mon Sep 17 00:00:00 2001 From: Naman Rusia Date: Fri, 25 Sep 2026 14:13:41 -0400 Subject: [PATCH 2/2] Make a failed analysis final, so retries cannot slip past the hourly limit --- apps/backend/app/dao/submissions.py | 8 +++----- apps/backend/app/runner/__main__.py | 2 +- apps/backend/app/services/submissions.py | 9 ++++++--- apps/backend/tests/test_analysis.py | 15 +++++++++++++++ apps/web/src/components/ComplexityPanel.tsx | 14 ++++++++++---- 5 files changed, 35 insertions(+), 13 deletions(-) diff --git a/apps/backend/app/dao/submissions.py b/apps/backend/app/dao/submissions.py index c47cbf7..ad71726 100644 --- a/apps/backend/app/dao/submissions.py +++ b/apps/backend/app/dao/submissions.py @@ -131,13 +131,11 @@ async def count_analyses_since(self, user_id: str, seconds: int) -> int: return await conn.fetchval(query, user_id, seconds) async def request_analysis(self, submission_id: UUID, user_id: str) -> dict | None: - """Queue an analysis of an accepted run, unless one is queued or done already.""" + """Queue an analysis of an accepted run that has never had one.""" query = """ UPDATE submissions - SET analysis_status = 'pending', analysis = NULL, - analysis_requested_by = $2, analysis_requested_at = NOW() - WHERE id = $1 AND status = 'accepted' - AND (analysis_status IS NULL OR analysis_status = 'failed') + SET analysis_status = 'pending', analysis_requested_by = $2, analysis_requested_at = NOW() + WHERE id = $1 AND status = 'accepted' AND analysis_status IS NULL RETURNING id """ async with self.pool.acquire() as conn: diff --git a/apps/backend/app/runner/__main__.py b/apps/backend/app/runner/__main__.py index 6805b71..4f7b6c6 100644 --- a/apps/backend/app/runner/__main__.py +++ b/apps/backend/app/runner/__main__.py @@ -60,7 +60,7 @@ async def analyze_next(submissions: SubmissionDAO, problems: ProblemDAO) -> bool analysis = await asyncio.to_thread(run_analysis, submission["code"] or "", spec["generator"], spec["memLimitMb"]) except Exception: logger.exception("Analysis failed for submission %s", submission["id"]) - analysis = {"points": [], "complexity": None, "slope": None, "note": "The analysis could not run. Try again later."} + analysis = {"points": [], "complexity": None, "slope": None, "note": "The analysis could not run."} status = "done" if analysis["complexity"] else "failed" try: diff --git a/apps/backend/app/services/submissions.py b/apps/backend/app/services/submissions.py index 5ad5b61..6ad2dfa 100644 --- a/apps/backend/app/services/submissions.py +++ b/apps/backend/app/services/submissions.py @@ -126,9 +126,12 @@ async def request_analysis(self, submission_id, caller_id: str) -> dict: status_code=status.HTTP_409_CONFLICT, detail="This problem does not support complexity analysis yet", ) - # Both people in a room can press the button; the second press sees - # the first one's analysis instead of queueing another. - if submission["analysisStatus"] in ("pending", "running", "done"): + # One analysis per run, whatever it found. Both people in a room can + # press the button, and the second press sees the first one's. A + # failed one is final too: the hourly limit counts runs, so retries of + # one run would never reach it, and the same program on the same + # inputs would fail the same way. + if submission["analysisStatus"] is not None: return submission if await self.submission_dao.count_analyses_since(caller_id, 3600) >= ANALYSES_PER_HOUR: raise HTTPException( diff --git a/apps/backend/tests/test_analysis.py b/apps/backend/tests/test_analysis.py index 1ab728b..00803f9 100644 --- a/apps/backend/tests/test_analysis.py +++ b/apps/backend/tests/test_analysis.py @@ -59,6 +59,21 @@ def test_asking_twice_does_not_queue_a_second_analysis(client, room): assert analyze_queue(client) == 1 +def test_a_failed_analysis_is_final(client, room): + """Retries would reuse the run's row, which the hourly limit counts once.""" + submission = accepted_run(client, room) + ask(client, room, submission["id"]) + dao = SubmissionDAO(client.app.state.db_pool) + claimed = client.portal.call(dao.claim_pending_analysis) + failed = {"points": [], "complexity": None, "slope": None, "note": "Stopped at n = 1000."} + client.portal.call(dao.complete_analysis, claimed["id"], "failed", failed) + + again = ask(client, room, submission["id"]) + assert again.status_code == 200, again.text + assert again.json()["analysisStatus"] == "failed" + assert analyze_queue(client) == 0 + + def test_only_accepted_runs_can_be_analyzed(client, room): submission = submit(client, room, "print('0 0')\n").json() drain_queue(client) diff --git a/apps/web/src/components/ComplexityPanel.tsx b/apps/web/src/components/ComplexityPanel.tsx index f40a17e..d5a9729 100644 --- a/apps/web/src/components/ComplexityPanel.tsx +++ b/apps/web/src/components/ComplexityPanel.tsx @@ -96,14 +96,20 @@ const ComplexityPanel = ({ submission, expected }: { submission: Submission; exp ); } + if (state === "failed") { + return ( +

+ No estimate for this run. {analysis?.note ?? ""} +

+ ); + } + return (
- - {state === "failed" && analysis?.note ? analysis.note : "Reruns this solution on bigger inputs to see how its time grows."} - + Reruns this solution on bigger inputs to see how its time grows. {error && {error}}
);