From 81f745586274eed6f6db3e66dc65994b8fdc906b Mon Sep 17 00:00:00 2001 From: Karthik Suresh <7954591+k21993@users.noreply.github.com> Date: Wed, 23 Sep 2026 23:11:44 -0700 Subject: [PATCH 1/2] log_parsers: read unittest and Django test-runner output Python unittest and Django commands previously fell back to the pytest parser and produced no oracle. Add TextTestRunner parsing in both parser copies, runner detection, and the verbosity flags needed to expose test identities. Group subtests under stable parent identities, accept nested parameter values, preserve the worst status, and keep unittest count footers out of the pytest fallback. Closes #167 --- docs/pipelines/pr_runtime.md | 4 +- src/repo2rlenv/log_parsers/__init__.py | 11 +- src/repo2rlenv/log_parsers/pytest_parser.py | 5 + src/repo2rlenv/log_parsers/unittest_parser.py | 140 ++++++ .../pipelines/_pr_runtime_verifier.py | 87 +++- src/repo2rlenv/pipelines/pr_runtime.py | 16 + tests/test_pipeline_pr_runtime.py | 21 + tests/test_unittest_output.py | 434 ++++++++++++++++++ 8 files changed, 714 insertions(+), 4 deletions(-) create mode 100644 src/repo2rlenv/log_parsers/unittest_parser.py create mode 100644 tests/test_unittest_output.py diff --git a/docs/pipelines/pr_runtime.md b/docs/pipelines/pr_runtime.md index 2bb6b667..0383bb94 100644 --- a/docs/pipelines/pr_runtime.md +++ b/docs/pipelines/pr_runtime.md @@ -82,8 +82,8 @@ Per-language log parsers map raw test output to `{PASSED, FAILED, SKIPPED, ERROR | Language | Parser | Notes | |---|---|---| -| Python (pytest) | `log_parsers/python.py` | Handles `PASSED tests/foo.py::test_x` lines | -| Python (django) | same module, django variant | Django runs are slightly different | +| Python (pytest) | `log_parsers/pytest_parser.py` | Handles `PASSED tests/foo.py::test_x` lines, plus pytest-xdist's `[gw1]` labels | +| Python (unittest, django) | `log_parsers/unittest_parser.py` | `test_x (pkg.Case.test_x) ... ok`; needs `-v` (`-v 2` for Django) | | JS / TS (mocha, jest) | `log_parsers/javascript.py` | From SWE-bench-Live multi-language | | Go (`go test`) | new (we write it) | `--- PASS:` / `--- FAIL:` | | Rust (`cargo test`) | new | `test foo ... ok` / `FAILED` | diff --git a/src/repo2rlenv/log_parsers/__init__.py b/src/repo2rlenv/log_parsers/__init__.py index ef4ab7f4..aae3eacc 100644 --- a/src/repo2rlenv/log_parsers/__init__.py +++ b/src/repo2rlenv/log_parsers/__init__.py @@ -14,6 +14,7 @@ * `parse_go_test` — `go test -v` * `parse_cargo_test`— `cargo test` * `parse_jest` — Jest / Mocha / Vitest + * `parse_unittest` — `python -m unittest` / Django's runner * `parse_logs(language, test_cmds, log)` — dispatch by runner. Inspects the test_cmds string for runner keywords first; falls back to the @@ -32,6 +33,7 @@ from repo2rlenv.log_parsers.go_parser import parse_go_test from repo2rlenv.log_parsers.jest_parser import parse_jest from repo2rlenv.log_parsers.pytest_parser import TestStatus, parse_pytest +from repo2rlenv.log_parsers.unittest_parser import parse_unittest __all__ = [ "TestStatus", @@ -40,13 +42,14 @@ "parse_jest", "parse_logs", "parse_pytest", + "parse_unittest", ] def _detect_runner(test_cmds: list[str]) -> str: """Inspect the bootstrap-recorded test commands and name the runner. - Returns one of: "pytest", "go", "cargo", "jest", "unknown". + Returns one of: "pytest", "unittest", "go", "cargo", "jest", "unknown". Why inspect test_cmds rather than the LanguageHint alone? A single language often has multiple runners — Python has pytest + unittest + @@ -57,6 +60,10 @@ def _detect_runner(test_cmds: list[str]) -> str: joined = " ".join(test_cmds).lower() if "pytest" in joined: return "pytest" + # unittest's own runner, Django's `manage.py test`, and Django's in-repo + # `runtests.py` all print TextTestRunner output. + if re.search(r"\bunittest\b|manage\.py\s+test\b|runtests\.py", joined): + return "unittest" if re.search(r"\bgo\s+test\b", joined): return "go" if re.search(r"\bcargo\s+test\b", joined): @@ -105,6 +112,8 @@ def parse_logs( if runner == "pytest": return parse_pytest(log) + if runner == "unittest": + return parse_unittest(log) if runner == "go": return parse_go_test(log) if runner == "cargo": diff --git a/src/repo2rlenv/log_parsers/pytest_parser.py b/src/repo2rlenv/log_parsers/pytest_parser.py index 6882d6cb..14775fcf 100644 --- a/src/repo2rlenv/log_parsers/pytest_parser.py +++ b/src/repo2rlenv/log_parsers/pytest_parser.py @@ -85,6 +85,11 @@ def parse_pytest(log: str) -> dict[str, TestStatus]: break if leading_status is not None: work = line[len(leading_status) :].strip() + # unittest's footer is `FAILED (failures=1, errors=1)`, which is a + # count, not a node ID. Without this it lands as a test named + # `(errors=1)` whenever a unittest log reaches this parser. + if work.startswith("(") and work.endswith(")"): + continue if leading_status == "SKIPPED" and re.match(r"^\[\d+\](?:\s|$)", work): # Folded skips report a file location, not a parametrized ID. tokens = work.split(maxsplit=2) diff --git a/src/repo2rlenv/log_parsers/unittest_parser.py b/src/repo2rlenv/log_parsers/unittest_parser.py new file mode 100644 index 00000000..b5255a25 --- /dev/null +++ b/src/repo2rlenv/log_parsers/unittest_parser.py @@ -0,0 +1,140 @@ +"""`python -m unittest` / Django test-runner output parser. + +unittest's verbose runner prints one line per test, with the status after an +ellipsis: + + test_add (tests.test_math.MathTests.test_add) ... ok + test_broken (tests.test_math.MathTests.test_broken) ... FAIL + test_error (tests.test_math.MathTests.test_error) ... ERROR + test_skipped (tests.test_math.MathTests.test_skipped) ... skipped 'later' + +Two shapes need care: + + * A test with a docstring prints the name, then the docstring and the + status on the NEXT line, because `getDescription` joins them with a + newline. The status therefore belongs to the name printed just above. + * A subtest failure is reported on its own indented line carrying the + parameters (`... (i=2) ... FAIL`), while the parent test's line ends + after the ellipsis with no status at all. Subtests are recorded under + the parent's identity, and the worst status wins, because the repaired + run prints only `parent ... ok`: keeping the parameters would leave + FAIL_TO_PASS with a name that never appears again. Parameters can nest + parentheses (`(value=(1, 2))`), so they are matched loosely. + +Test names are canonicalized to the dotted id you could re-run, so the key +is stable across Python versions: 3.11+ prints the full path inside the +parentheses, older versions print only the class path with the method name +outside. + +`expected failure` counts as PASSED and `unexpected success` as FAILED, +matching unittest's own verdict (`wasSuccessful()`), so the F2P/P2P sets +agree with the suite's exit code. + +Django's runner (`manage.py test -v 2`) delegates to unittest's +TextTestRunner, so this parses its output too. + +Released under Apache-2.0. +""" + +from __future__ import annotations + +import re + +from repo2rlenv.log_parsers.pytest_parser import TestStatus + +# `test_add (pkg.mod.Class.test_add)` plus any subtest parameters, e.g. +# ` (i=2)` or ` (value=(1, 2))`, which are matched but not kept. +_NAME_RE = re.compile( + r"^(?P[^\s()]+) \((?P[\w.]+)\)(?P \(.*\))?$", +) + +_STATUS_WORDS: dict[str, TestStatus] = { + "ok": "PASSED", + "FAIL": "FAILED", + "ERROR": "ERROR", + "expected failure": "PASSED", + "unexpected success": "FAILED", +} + +# Failure blocks below the run repeat each name: `FAIL: test_x (pkg.Class.test_x)`. +_BLOCK_RE = re.compile(r"^(?PFAIL|ERROR):\s+(?P.+?)\s*$") + + +# Worst status wins when a test reports more than once, i.e. a parent whose +# subtests each report separately. +_RANK: dict[TestStatus, int] = {"SKIPPED": 0, "PASSED": 1, "FAILED": 2, "ERROR": 3} + + +def _canonical(method: str, dotted: str) -> str: + """Return the dotted id, e.g. `tests.test_math.MathTests.test_add`. + + 3.11+ prints the full path inside the parentheses; older versions print + the class path there and the method name outside. + """ + return dotted if dotted.endswith(f".{method}") else f"{dotted}.{method}" + + +def _record(out: dict[str, TestStatus], name: str, status: TestStatus) -> None: + if name in out and _RANK[out[name]] >= _RANK[status]: + return + out[name] = status + + +def _status_for(tail: str) -> TestStatus | None: + """Map the text after the ellipsis to a status, or None when unknown.""" + text = tail.strip() + if not text: + # A parent test whose subtests report separately. + return None + if text.startswith("skipped"): + return "SKIPPED" + return _STATUS_WORDS.get(text) + + +def parse_unittest(log: str) -> dict[str, TestStatus]: + """Return {test_name -> status} parsed from unittest/Django output. + + Lines that are neither a result line nor a failure-block header are + ignored, so tracebacks and the `Ran N tests` footer contribute nothing. + """ + out: dict[str, TestStatus] = {} + if not log: + return out + + # A name printed without a status is only claimed by the very next line. + pending: str | None = None + for raw in log.split("\n"): + line = raw.rstrip() + claimed, pending = pending, None + if not line.strip(): + continue + + head, sep, tail = line.strip().partition(" ... ") + if sep: + m = _NAME_RE.match(head) + if m: + status = _status_for(tail) + if status is None: + # Parent of subtests: its own results follow, indented. + continue + _record(out, _canonical(m["method"], m["dotted"]), status) + elif claimed is not None: + # ` ... ok` for the name on the previous line. + status = _status_for(tail) + if status is not None: + _record(out, claimed, status) + continue + + m = _NAME_RE.match(line.strip()) + if m: + pending = _canonical(m["method"], m["dotted"]) + continue + + m = _BLOCK_RE.match(line) + if m: + name = _NAME_RE.match(m["name"]) + if name: + key = _canonical(name["method"], name["dotted"]) + _record(out, key, _STATUS_WORDS[m["status"]]) + + return out diff --git a/src/repo2rlenv/pipelines/_pr_runtime_verifier.py b/src/repo2rlenv/pipelines/_pr_runtime_verifier.py index 1b811160..43bb3e0d 100644 --- a/src/repo2rlenv/pipelines/_pr_runtime_verifier.py +++ b/src/repo2rlenv/pipelines/_pr_runtime_verifier.py @@ -104,6 +104,9 @@ def parse_pytest(log: str) -> dict[str, str]: break if leading is not None: work = line[len(leading) :].strip() + # unittest's `FAILED (failures=1)` footer is a count, not a node ID. + if work.startswith("(") and work.endswith(")"): + continue if leading == SKIPPED and re.match(r"^\[\d+\](?:\s|$)", work): # Folded skips report a file location, not a parametrized ID. tokens = work.split(maxsplit=2) @@ -126,6 +129,83 @@ def parse_pytest(log: str) -> dict[str, str]: return out +# Keep these patterns in sync with log_parsers/unittest_parser.py. +_UT_NAME_RE = re.compile(r"^(?P[^\s()]+) \((?P[\w.]+)\)(?P \(.*\))?$") +_UT_BLOCK_RE = re.compile(r"^(?PFAIL|ERROR):\s+(?P.+?)\s*$") +_UT_STATUS = { + "ok": PASSED, + "FAIL": FAILED, + "ERROR": ERROR, + "expected failure": PASSED, + "unexpected success": FAILED, +} + + +# Subtests report under the parent's identity, worst status first, because a +# repaired run prints only `parent ... ok`. +_UT_RANK = {SKIPPED: 0, PASSED: 1, FAILED: 2, ERROR: 3} + + +def _ut_canonical(method: str, dotted: str) -> str: + return dotted if dotted.endswith(f".{method}") else f"{dotted}.{method}" + + +def _ut_record(out: dict[str, str], name: str, status: str) -> None: + if name in out and _UT_RANK[out[name]] >= _UT_RANK[status]: + return + out[name] = status + + +def _ut_status(tail: str) -> str | None: + text = tail.strip() + if not text: + return None + if text.startswith("skipped"): + return SKIPPED + return _UT_STATUS.get(text) + + +def parse_unittest(log: str) -> dict[str, str]: + """{test_name -> status} from `python -m unittest -v` / Django output. + + A docstring pushes the status onto the line after the name, and subtest + failures report on their own indented line while the parent prints none. + """ + out: dict[str, str] = {} + if not log: + return out + pending: str | None = None + for raw in log.split("\n"): + line = raw.rstrip() + claimed, pending = pending, None + if not line.strip(): + continue + head, sep, tail = line.strip().partition(" ... ") + if sep: + m = _UT_NAME_RE.match(head) + if m: + status = _ut_status(tail) + if status is not None: + _ut_record(out, _ut_canonical(m["method"], m["dotted"]), status) + elif claimed is not None: + status = _ut_status(tail) + if status is not None: + _ut_record(out, claimed, status) + continue + m = _UT_NAME_RE.match(line.strip()) + if m: + pending = _ut_canonical(m["method"], m["dotted"]) + continue + m = _UT_BLOCK_RE.match(line) + if m: + name = _UT_NAME_RE.match(m["name"]) + if name: + _ut_record( + out, _ut_canonical(name["method"], name["dotted"]), _UT_STATUS[m["status"]] + ) + return out + + _GO_TEST_RE = re.compile(r"^\s*---\s+(?PPASS|FAIL|SKIP):\s+(?P\S+)") _GO_STATUS = {"PASS": PASSED, "FAIL": FAILED, "SKIP": SKIPPED} @@ -310,6 +390,8 @@ def _detect_runner(test_cmds: str) -> str: joined = test_cmds.lower() if "pytest" in joined: return "pytest" + if re.search(r"\bunittest\b|manage\.py\s+test\b|runtests\.py", joined): + return "unittest" if re.search(r"\bgo\s+test\b", joined): return "go" if re.search(r"\bcargo\s+test\b", joined): @@ -323,6 +405,8 @@ def parse_logs(runner: str, log: str) -> dict[str, str]: """Dispatch to the right per-runner parser. Empty dict if unknown.""" if runner == "pytest": return parse_pytest(log) + if runner == "unittest": + return parse_unittest(log) if runner == "go": return parse_go_test(log) if runner == "cargo": @@ -448,7 +532,7 @@ def main(argv: list[str] | None = None) -> int: p.add_argument("--log", required=True, help="captured test-run log file") p.add_argument("--f2p", required=True, help="JSON file: FAIL_TO_PASS test names") p.add_argument("--p2p", required=True, help="JSON file: PASS_TO_PASS test names") - p.add_argument("--runner", default="", help="pytest|go|cargo|jest (else auto-detect)") + p.add_argument("--runner", default="", help="pytest|unittest|go|cargo|jest (else auto-detect)") p.add_argument("--test-cmds", default="", help="test command string (runner auto-detect)") p.add_argument("--exit-code", type=int, default=1, help="test suite exit code (fallback)") p.add_argument("--out-dir", default="/logs/verifier", help="where to write reward.{txt,json}") @@ -521,4 +605,5 @@ def main(argv: list[str] | None = None) -> int: "parse_jest", "parse_logs", "parse_pytest", + "parse_unittest", ] diff --git a/src/repo2rlenv/pipelines/pr_runtime.py b/src/repo2rlenv/pipelines/pr_runtime.py index 97f5b979..f7d86cfc 100644 --- a/src/repo2rlenv/pipelines/pr_runtime.py +++ b/src/repo2rlenv/pipelines/pr_runtime.py @@ -629,6 +629,11 @@ def normalize_test_cmds_for_runtime(test_cmds: list[str]) -> list[str]: - Drop `--collect-only` / `--co` so pytest actually runs tests - Drop `-q` / `--quiet`: suppresses per-test names; cancels `-v` in pytest 9 - Add `-v` if no verbosity flag is present + python unittest / Django: + - Add a verbosity flag when none is present, since these runners + print only dots at the default level: `-v` for unittest, `-v 2` + for `manage.py test`, and `--verbosity 2` for Django's in-repo + `tests/runtests.py`, which takes no `-v` count. go test: - Add `-v` if missing (default `go test` doesn't print --- PASS lines) cargo test: @@ -665,6 +670,17 @@ def normalize_test_cmds_for_runtime(test_cmds: list[str]) -> list[str]: if not re.search(r"\s-v\b|\s--verbose\b|-vv\b", cleaned): cleaned = cleaned.rstrip() + " -v" + # --- python unittest / Django (manage.py test, tests/runtests.py) --- + elif re.search(r"\bunittest\b|manage\.py\s+test\b|runtests\.py", cleaned): + if not re.search(r"\s-v\b|\s--verbose\b|\s--verbosity\b|-vv\b", cleaned): + if re.search(r"runtests\.py", cleaned): + flag = " --verbosity 2" + elif re.search(r"manage\.py\s+test\b", cleaned): + flag = " -v 2" # Django's runner counts verbosity + else: + flag = " -v" # unittest's own flag + cleaned = cleaned.rstrip() + flag + # --- go test --- elif re.search(r"\bgo\s+test\b", cleaned): if not re.search(r"\s-v\b", cleaned): diff --git a/tests/test_pipeline_pr_runtime.py b/tests/test_pipeline_pr_runtime.py index 7fceb822..e877ee0d 100644 --- a/tests/test_pipeline_pr_runtime.py +++ b/tests/test_pipeline_pr_runtime.py @@ -507,6 +507,27 @@ def test_normalize_preserves_existing_verbose(): assert normalize_test_cmds_for_runtime(["pytest -vv"]) == ["pytest -vv"] +def test_normalize_unittest_gets_verbosity(): + # Without -v, unittest prints only dots and no test name is parseable. + assert normalize_test_cmds_for_runtime(["python -m unittest"]) == ["python -m unittest -v"] + assert normalize_test_cmds_for_runtime(["python -m unittest discover -s tests"]) == [ + "python -m unittest discover -s tests -v" + ] + # Django's runner counts verbosity instead of toggling it. + assert normalize_test_cmds_for_runtime(["./manage.py test"]) == ["./manage.py test -v 2"] + # Already verbose ⇒ keep + assert normalize_test_cmds_for_runtime(["python manage.py test demo -v 2"]) == [ + "python manage.py test demo -v 2" + ] + # Django's in-repo runtests.py takes --verbosity, not a -v count. + assert normalize_test_cmds_for_runtime(["python tests/runtests.py"]) == [ + "python tests/runtests.py --verbosity 2" + ] + assert normalize_test_cmds_for_runtime(["python tests/runtests.py --verbosity 3"]) == [ + "python tests/runtests.py --verbosity 3" + ] + + def test_normalize_go_test_gets_v_flag(): """`go test` without -v doesn't print --- PASS lines — parser needs them.""" assert normalize_test_cmds_for_runtime(["go test ./..."]) == ["go test -v ./..."] diff --git a/tests/test_unittest_output.py b/tests/test_unittest_output.py new file mode 100644 index 00000000..086ff7d5 --- /dev/null +++ b/tests/test_unittest_output.py @@ -0,0 +1,434 @@ +"""unittest and Django runs must be parsed by both parsers, not read as pytest.""" + +from __future__ import annotations + +import json +import subprocess +import sys +from pathlib import Path + +import pytest + +from repo2rlenv.log_parsers import parse_logs +from repo2rlenv.log_parsers.unittest_parser import parse_unittest +from repo2rlenv.pipelines import _pr_runtime_verifier as verifier + + +@pytest.fixture(params=[parse_unittest, verifier.parse_unittest], ids=["canonical", "standalone"]) +def parser(request): + return request.param + + +# Real `python -m unittest -v` output (CPython 3.14): every status unittest can +# report, a test whose docstring pushes the status onto the next line, and a +# subtest failure reported under its parent. +_VERBOSE = ( + "test_add (tests.test_math.MathTests.test_add) ... ok\n" + "test_broken (tests.test_math.MathTests.test_broken) ... FAIL\n" + "test_error (tests.test_math.MathTests.test_error) ... ERROR\n" + "test_expected_failure (tests.test_math.MathTests.test_expected_failure) ... expected failure\n" + "test_skipped (tests.test_math.MathTests.test_skipped) ... skipped 'later'\n" + "test_subtests (tests.test_math.MathTests.test_subtests) ... \n" + " test_subtests (tests.test_math.MathTests.test_subtests) (i=2) ... FAIL\n" + "test_with_docstring (tests.test_math.MathTests.test_with_docstring)\n" + "Adds two negatives. ... ok\n" + "test_subtests_pass (tests.test_math.MoreTests.test_subtests_pass) ... ok\n" + "test_unexpected_success (tests.test_math.MoreTests.test_unexpected_success) ... unexpected success\n" + "\n" + "======================================================================\n" + "ERROR: test_error (tests.test_math.MathTests.test_error)\n" + "----------------------------------------------------------------------\n" + "Traceback (most recent call last):\n" + ' File "/repo/tests/test_math.py", line 16, in test_error\n' + ' raise RuntimeError("boom")\n' + "RuntimeError: boom\n" + "\n" + "======================================================================\n" + "FAIL: test_broken (tests.test_math.MathTests.test_broken)\n" + "----------------------------------------------------------------------\n" + "Traceback (most recent call last):\n" + ' File "/repo/tests/test_math.py", line 13, in test_broken\n' + " self.assertEqual(1 + 1, 3)\n" + " ~~~~~~~~~~~~~~~~^^^^^^^^^^\n" + "AssertionError: 2 != 3\n" + "\n" + "======================================================================\n" + "FAIL: test_subtests (tests.test_math.MathTests.test_subtests) (i=2)\n" + "----------------------------------------------------------------------\n" + "Traceback (most recent call last):\n" + ' File "/repo/tests/test_math.py", line 29, in test_subtests\n' + " self.assertEqual(i % 2, 1)\n" + " ~~~~~~~~~~~~~~~~^^^^^^^^^^\n" + "AssertionError: 0 != 1\n" + "\n" + "======================================================================\n" + "UNEXPECTED SUCCESS: test_unexpected_success (tests.test_math.MoreTests.test_unexpected_success)\n" + "----------------------------------------------------------------------\n" + "Ran 9 tests in 0.001s\n" + "\n" + "FAILED (failures=2, errors=1, skipped=1, expected failures=1, unexpected successes=1)\n" +) + +# The same suite without -v: dots, then the failure blocks. +_PLAIN = ( + ".FExsF..u\n" + "======================================================================\n" + "ERROR: test_error (tests.test_math.MathTests.test_error)\n" + "----------------------------------------------------------------------\n" + "Traceback (most recent call last):\n" + ' File "/repo/tests/test_math.py", line 16, in test_error\n' + ' raise RuntimeError("boom")\n' + "RuntimeError: boom\n" + "\n" + "======================================================================\n" + "FAIL: test_broken (tests.test_math.MathTests.test_broken)\n" + "----------------------------------------------------------------------\n" + "Traceback (most recent call last):\n" + ' File "/repo/tests/test_math.py", line 13, in test_broken\n' + " self.assertEqual(1 + 1, 3)\n" + " ~~~~~~~~~~~~~~~~^^^^^^^^^^\n" + "AssertionError: 2 != 3\n" + "\n" + "======================================================================\n" + "FAIL: test_subtests (tests.test_math.MathTests.test_subtests) (i=2)\n" + "----------------------------------------------------------------------\n" + "Traceback (most recent call last):\n" + ' File "/repo/tests/test_math.py", line 29, in test_subtests\n' + " self.assertEqual(i % 2, 1)\n" + " ~~~~~~~~~~~~~~~~^^^^^^^^^^\n" + "AssertionError: 0 != 1\n" + "\n" + "======================================================================\n" + "UNEXPECTED SUCCESS: test_unexpected_success (tests.test_math.MoreTests.test_unexpected_success)\n" + "----------------------------------------------------------------------\n" + "Ran 9 tests in 0.001s\n" + "\n" + "FAILED (failures=2, errors=1, skipped=1, expected failures=1, unexpected successes=1)\n" +) + +# Real `manage.py test -v 2` output (Django 6.1.1), which uses TextTestRunner. +_DJANGO = ( + "test_add (demo.tests.MathTests.test_add) ... ok\n" + "test_broken (demo.tests.MathTests.test_broken) ... FAIL\n" + "test_with_docstring (demo.tests.MathTests.test_with_docstring)\n" + "Adds two negatives. ... ok\n" + "\n" + "======================================================================\n" + "FAIL: test_broken (demo.tests.MathTests.test_broken)\n" + "----------------------------------------------------------------------\n" + "Traceback (most recent call last):\n" + ' File "/repo/dj/demo/tests.py", line 13, in test_broken\n' + " self.assertEqual(1 + 1, 3)\n" + "AssertionError: 2 != 3\n" + "\n" + "----------------------------------------------------------------------\n" + "Ran 3 tests in 0.000s\n" + "\n" + "FAILED (failures=1)\n" + "Found 3 test(s).\n" + "Skipping setup of unused database(s): default.\n" + "System check identified no issues (0 silenced).\n" +) + + +def test_every_status_is_recorded(parser): + assert parser(_VERBOSE) == { + "tests.test_math.MathTests.test_add": "PASSED", + "tests.test_math.MathTests.test_broken": "FAILED", + "tests.test_math.MathTests.test_error": "ERROR", + "tests.test_math.MathTests.test_expected_failure": "PASSED", + "tests.test_math.MathTests.test_skipped": "SKIPPED", + "tests.test_math.MathTests.test_subtests": "FAILED", + "tests.test_math.MathTests.test_with_docstring": "PASSED", + "tests.test_math.MoreTests.test_subtests_pass": "PASSED", + "tests.test_math.MoreTests.test_unexpected_success": "FAILED", + } + + +def test_plain_run_records_only_the_failure_blocks(parser): + assert parser(_PLAIN) == { + "tests.test_math.MathTests.test_broken": "FAILED", + "tests.test_math.MathTests.test_error": "ERROR", + "tests.test_math.MathTests.test_subtests": "FAILED", + } + + +def test_django_runner(parser): + assert parser(_DJANGO) == { + "demo.tests.MathTests.test_add": "PASSED", + "demo.tests.MathTests.test_broken": "FAILED", + "demo.tests.MathTests.test_with_docstring": "PASSED", + } + + +def test_pre_311_name_layout(parser): + # Before 3.11 the parentheses held only the class path. + log = "test_add (tests.test_math.MathTests) ... ok\n" + assert parser(log) == {"tests.test_math.MathTests.test_add": "PASSED"} + + +@pytest.mark.parametrize( + ("tail", "status"), + [ + ("ok", "PASSED"), + ("FAIL", "FAILED"), + ("ERROR", "ERROR"), + ("expected failure", "PASSED"), + ("unexpected success", "FAILED"), + ("skipped 'needs network'", "SKIPPED"), + ("skipped", "SKIPPED"), + ], +) +def test_status_words(parser, tail, status): + log = f"test_x (pkg.mod.Case.test_x) ... {tail}\n" + assert parser(log) == {"pkg.mod.Case.test_x": status} + + +def test_docstring_status_only_binds_to_the_line_above(parser): + # A name with no status, then an unrelated line, must record nothing. + log = "test_x (pkg.mod.Case.test_x)\nsome other output ... ok\n" + assert parser(log) == {"pkg.mod.Case.test_x": "PASSED"} + log = "test_x (pkg.mod.Case.test_x)\n\nnoise\nmore noise ... ok\n" + assert parser(log) == {} + + +def test_tracebacks_and_footers_are_ignored(parser): + log = ( + "======================================================================\n" + "FAIL: test_broken (pkg.mod.Case.test_broken)\n" + "----------------------------------------------------------------------\n" + "Traceback (most recent call last):\n" + ' File "/repo/tests/test_math.py", line 13, in test_broken\n' + " self.assertEqual(1 + 1, 3)\n" + "AssertionError: 2 != 3\n" + "\n" + "Ran 9 tests in 0.001s\n" + "\n" + "FAILED (failures=1, errors=1, skipped=1)\n" + ) + assert parser(log) == {"pkg.mod.Case.test_broken": "FAILED"} + + +def test_dispatcher_routes_unittest_commands(): + for cmd in [ + "python -m unittest -v", + "python -m unittest discover -s tests -v", + "python manage.py test -v 2", + "python tests/runtests.py --verbosity 2", + ]: + assert parse_logs([cmd], _VERBOSE), cmd + # The standalone verifier detects the same commands from --test-cmds. + assert verifier.parse_logs(verifier._detect_runner("python -m unittest -v"), _VERBOSE) + + +def test_unittest_footer_is_not_a_pytest_node(): + # A wrapper command lands in the pytest parser; its footer must not + # become a test called `(errors=1)`. + from repo2rlenv.log_parsers.pytest_parser import parse_pytest + + for parse in (parse_pytest, verifier.parse_pytest): + assert parse(_VERBOSE) == {} + assert parse("FAILED tests/test_a.py::test_x - AssertionError") == { + "tests/test_a.py::test_x": "FAILED" + } + + +def test_long_lines_do_not_stall_parsing(): + # Isolate the timeout so a backtracking regression cannot hang the suite. + result = subprocess.run( + [ + sys.executable, + "-c", + "from repo2rlenv.log_parsers.unittest_parser import parse_unittest\n" + "from repo2rlenv.pipelines._pr_runtime_verifier import parse_unittest as standalone\n" + "noise = ['t (' + 'a' * 200000 + ')', 'x' * 200000 + ' ... ok',\n" + " 't (pkg.C.t)' + ' (i=1)' * 50000 + ' ... FAIL',\n" + " 'FAIL: t (' + 'b.' * 100000 + 'C.t)']\n" + "log = '\\n'.join(noise) + '\\ntest_x (pkg.mod.Case.test_x) ... ok\\n'\n" + "for parser in (parse_unittest, standalone):\n" + " assert parser(log)['pkg.mod.Case.test_x'] == 'PASSED'\n", + ], + capture_output=True, + text=True, + timeout=10, + ) + assert result.returncode == 0, result.stdout + result.stderr + + +# A real fix under unittest: `div` gains a zero check, and the test that +# covers it has a docstring, so its status is printed on the following line. +_PRE = ( + "test_divides (tests.test_calc.DivTests.test_divides) ... ok\n" + "test_rejects_zero (tests.test_calc.DivTests.test_rejects_zero)\n" + "Rejects a zero divisor. ... ERROR\n" + "\n" + "======================================================================\n" + "ERROR: test_rejects_zero (tests.test_calc.DivTests.test_rejects_zero)\n" + "Rejects a zero divisor.\n" + "----------------------------------------------------------------------\n" + "Traceback (most recent call last):\n" + ' File "/repo/tests/test_calc.py", line 13, in test_rejects_zero\n' + " div(1, 0)\n" + " ~~~^^^^^^\n" + ' File "/repo/calc.py", line 2, in div\n' + " return a / b\n" + " ~~^~~\n" + "ZeroDivisionError: division by zero\n" + "\n" + "----------------------------------------------------------------------\n" + "Ran 2 tests in 0.001s\n" + "\n" + "FAILED (errors=1)\n" +) + +_POST = ( + "test_divides (tests.test_calc.DivTests.test_divides) ... ok\n" + "test_rejects_zero (tests.test_calc.DivTests.test_rejects_zero)\n" + "Rejects a zero divisor. ... ok\n" + "\n" + "----------------------------------------------------------------------\n" + "Ran 2 tests in 0.000s\n" + "\n" + "OK\n" +) + + +def test_unittest_fix_yields_an_oracle_and_grades_it(tmp_path: Path): + cmd = ["python -m unittest -v"] + pre, post = parse_logs(cmd, _PRE), parse_logs(cmd, _POST) + f2p = sorted( + name for name, st in pre.items() if st in ("FAILED", "ERROR") and post.get(name) == "PASSED" + ) + p2p = sorted(name for name, st in pre.items() if st == "PASSED" and post.get(name) == "PASSED") + assert f2p == ["tests.test_calc.DivTests.test_rejects_zero"] + assert p2p == ["tests.test_calc.DivTests.test_divides"] + + # Run the verifier in isolation, with no installed repo2rlenv imports. + standalone = tmp_path / "verifier.py" + standalone.write_text(Path(verifier.__file__).read_text(encoding="utf-8"), encoding="utf-8") + (tmp_path / "out.log").write_text(_POST, encoding="utf-8") + (tmp_path / "f2p.json").write_text(json.dumps(f2p), encoding="utf-8") + (tmp_path / "p2p.json").write_text(json.dumps(p2p), encoding="utf-8") + result = subprocess.run( + [ + sys.executable, + "-I", + str(standalone), + "--log", + "out.log", + "--f2p", + "f2p.json", + "--p2p", + "p2p.json", + "--test-cmds", + "python -m unittest -v", + "--exit-code", + "0", + "--out-dir", + "rewards", + ], + cwd=tmp_path, + capture_output=True, + text=True, + timeout=30, + ) + assert result.returncode == 0, result.stdout + result.stderr + details = json.loads((tmp_path / "rewards/reward-details.json").read_text()) + assert details["runner"] == "unittest" + assert details["reward"] == 1.0 + assert details["resolved"] is True + + +# A real fix whose failing tests are subtests: `is_odd` is corrected, and one +# subtest value is a tuple, so its parameters contain nested parentheses. +_SUB_PRE = ( + "test_all_odd (tests.test_sub.SubTests.test_all_odd) ... \n" + " test_all_odd (tests.test_sub.SubTests.test_all_odd) (value=3) ... FAIL\n" + "test_pairs (tests.test_sub.SubTests.test_pairs) ... \n" + " test_pairs (tests.test_sub.SubTests.test_pairs) (value=(3, 4)) ... FAIL\n" + "\n" + "======================================================================\n" + "FAIL: test_all_odd (tests.test_sub.SubTests.test_all_odd) (value=3)\n" + "----------------------------------------------------------------------\n" + "Traceback (most recent call last):\n" + ' File "/repo/tests/test_sub.py", line 10, in test_all_odd\n' + " self.assertTrue(is_odd(value))\n" + " ~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^\n" + "AssertionError: False is not true\n" + "\n" + "======================================================================\n" + "FAIL: test_pairs (tests.test_sub.SubTests.test_pairs) (value=(3, 4))\n" + "----------------------------------------------------------------------\n" + "Traceback (most recent call last):\n" + ' File "/repo/tests/test_sub.py", line 15, in test_pairs\n' + " self.assertTrue(is_odd(pair[0]))\n" + " ~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^\n" + "AssertionError: False is not true\n" + "\n" + "----------------------------------------------------------------------\n" + "Ran 2 tests in 0.001s\n" + "\n" + "FAILED (failures=2)\n" +) + +_SUB_POST = ( + "test_all_odd (tests.test_sub.SubTests.test_all_odd) ... ok\n" + "test_pairs (tests.test_sub.SubTests.test_pairs) ... ok\n" + "\n" + "----------------------------------------------------------------------\n" + "Ran 2 tests in 0.000s\n" + "\n" + "OK\n" +) + + +def test_subtest_failures_report_under_the_parent(parser): + assert parser(_SUB_PRE) == { + "tests.test_sub.SubTests.test_all_odd": "FAILED", + "tests.test_sub.SubTests.test_pairs": "FAILED", + } + assert parser(_SUB_POST) == { + "tests.test_sub.SubTests.test_all_odd": "PASSED", + "tests.test_sub.SubTests.test_pairs": "PASSED", + } + + +def test_subtests_can_transition_from_failing_to_passing(): + # Keeping the parameters would leave a FAIL_TO_PASS name that the repaired + # run never prints, since it reports only `parent ... ok`. + cmd = ["python -m unittest -v"] + pre, post = parse_logs(cmd, _SUB_PRE), parse_logs(cmd, _SUB_POST) + f2p = sorted( + name for name, st in pre.items() if st in ("FAILED", "ERROR") and post.get(name) == "PASSED" + ) + assert f2p == [ + "tests.test_sub.SubTests.test_all_odd", + "tests.test_sub.SubTests.test_pairs", + ] + + +@pytest.mark.parametrize( + "params", + ["(i=2)", "(value=(1, 2))", "(value={'a': 1})", "(msg='a (b) c')", "(a=1) (b=2)"], +) +def test_subtest_parameters_are_matched_but_not_kept(parser, params): + log = f" test_x (pkg.mod.Case.test_x) {params} ... FAIL\n" + assert parser(log) == {"pkg.mod.Case.test_x": "FAILED"} + + +@pytest.mark.parametrize( + ("first", "second", "expected"), + [ + ("ok", "FAIL", "FAILED"), + ("FAIL", "ok", "FAILED"), + ("FAIL", "ERROR", "ERROR"), + ("skipped 'x'", "ok", "PASSED"), + ("ok", "skipped 'x'", "PASSED"), + ], +) +def test_worst_status_wins_when_a_test_reports_twice(parser, first, second, expected): + log = ( + f" test_x (pkg.mod.Case.test_x) (i=1) ... {first}\n" + f" test_x (pkg.mod.Case.test_x) (i=2) ... {second}\n" + ) + assert parser(log) == {"pkg.mod.Case.test_x": expected} From 34e42c996e1fa1b82434613d7ee1f57ff0ffba9d Mon Sep 17 00:00:00 2001 From: Karthik Suresh <7954591+k21993@users.noreply.github.com> Date: Tue, 29 Sep 2026 02:04:21 -0700 Subject: [PATCH 2/2] log_parsers: parse named unittest subtests TextTestRunner includes the subtest message before its parameters. Match and discard that full suffix so both parser copies retain the stable parent identity, including when the message contains an ellipsis. --- src/repo2rlenv/log_parsers/unittest_parser.py | 15 ++++++----- .../pipelines/_pr_runtime_verifier.py | 4 +-- tests/test_unittest_output.py | 25 +++++++++++++++++++ 3 files changed, 36 insertions(+), 8 deletions(-) diff --git a/src/repo2rlenv/log_parsers/unittest_parser.py b/src/repo2rlenv/log_parsers/unittest_parser.py index b5255a25..a705bd75 100644 --- a/src/repo2rlenv/log_parsers/unittest_parser.py +++ b/src/repo2rlenv/log_parsers/unittest_parser.py @@ -18,8 +18,9 @@ after the ellipsis with no status at all. Subtests are recorded under the parent's identity, and the worst status wins, because the repaired run prints only `parent ... ok`: keeping the parameters would leave - FAIL_TO_PASS with a name that never appears again. Parameters can nest - parentheses (`(value=(1, 2))`), so they are matched loosely. + FAIL_TO_PASS with a name that never appears again. Named subtests add a + message before their parameters (`[edge case] (value=(1, 2))`), so the + whole suffix is matched loosely and discarded. Test names are canonicalized to the dotted id you could re-run, so the key is stable across Python versions: 3.11+ prints the full path inside the @@ -42,10 +43,10 @@ from repo2rlenv.log_parsers.pytest_parser import TestStatus -# `test_add (pkg.mod.Class.test_add)` plus any subtest parameters, e.g. -# ` (i=2)` or ` (value=(1, 2))`, which are matched but not kept. +# `test_add (pkg.mod.Class.test_add)` plus any subtest description, e.g. +# ` [edge case] (i=2)` or ` (value=(1, 2))`, which is matched but not kept. _NAME_RE = re.compile( - r"^(?P[^\s()]+) \((?P[\w.]+)\)(?P \(.*\))?$", + r"^(?P[^\s()]+) \((?P[\w.]+)\)(?: .+)?$", ) _STATUS_WORDS: dict[str, TestStatus] = { @@ -109,7 +110,9 @@ def parse_unittest(log: str) -> dict[str, TestStatus]: if not line.strip(): continue - head, sep, tail = line.strip().partition(" ... ") + # A named subtest message may itself contain ` ... `, so split at the + # final separator before the result word. + head, sep, tail = line.strip().rpartition(" ... ") if sep: m = _NAME_RE.match(head) if m: diff --git a/src/repo2rlenv/pipelines/_pr_runtime_verifier.py b/src/repo2rlenv/pipelines/_pr_runtime_verifier.py index 43bb3e0d..01cd8527 100644 --- a/src/repo2rlenv/pipelines/_pr_runtime_verifier.py +++ b/src/repo2rlenv/pipelines/_pr_runtime_verifier.py @@ -130,7 +130,7 @@ def parse_pytest(log: str) -> dict[str, str]: # Keep these patterns in sync with log_parsers/unittest_parser.py. -_UT_NAME_RE = re.compile(r"^(?P[^\s()]+) \((?P[\w.]+)\)(?P \(.*\))?$") +_UT_NAME_RE = re.compile(r"^(?P[^\s()]+) \((?P[\w.]+)\)(?: .+)?$") _UT_BLOCK_RE = re.compile(r"^(?PFAIL|ERROR):\s+(?P.+?)\s*$") _UT_STATUS = { "ok": PASSED, @@ -180,7 +180,7 @@ def parse_unittest(log: str) -> dict[str, str]: claimed, pending = pending, None if not line.strip(): continue - head, sep, tail = line.strip().partition(" ... ") + head, sep, tail = line.strip().rpartition(" ... ") if sep: m = _UT_NAME_RE.match(head) if m: diff --git a/tests/test_unittest_output.py b/tests/test_unittest_output.py index 086ff7d5..0059cdc7 100644 --- a/tests/test_unittest_output.py +++ b/tests/test_unittest_output.py @@ -2,9 +2,11 @@ from __future__ import annotations +import io import json import subprocess import sys +import unittest from pathlib import Path import pytest @@ -416,6 +418,29 @@ def test_subtest_parameters_are_matched_but_not_kept(parser, params): assert parser(log) == {"pkg.mod.Case.test_x": "FAILED"} +def test_named_subtests_from_text_test_runner(parser): + def run_named_subtests(case): + with case.subTest("edge ... case", i=1): + case.assertEqual(1, 2) + with case.subTest("message only"): + case.assertEqual(1, 2) + + case_type = type( + "NamedSubTests", + (unittest.TestCase,), + {"__module__": "pkg", "test_named": run_named_subtests}, + ) + stream = io.StringIO() + unittest.TextTestRunner(stream=stream, verbosity=2).run( + unittest.defaultTestLoader.loadTestsFromTestCase(case_type) + ) + log = stream.getvalue() + + assert "[edge ... case] (i=1) ... FAIL" in log + assert "[message only] ... FAIL" in log + assert parser(log) == {"pkg.NamedSubTests.test_named": "FAILED"} + + @pytest.mark.parametrize( ("first", "second", "expected"), [