diff --git a/docs/pipelines/pr_runtime.md b/docs/pipelines/pr_runtime.md index 2bb6b667..0383bb94 100644 --- a/docs/pipelines/pr_runtime.md +++ b/docs/pipelines/pr_runtime.md @@ -82,8 +82,8 @@ Per-language log parsers map raw test output to `{PASSED, FAILED, SKIPPED, ERROR | Language | Parser | Notes | |---|---|---| -| Python (pytest) | `log_parsers/python.py` | Handles `PASSED tests/foo.py::test_x` lines | -| Python (django) | same module, django variant | Django runs are slightly different | +| Python (pytest) | `log_parsers/pytest_parser.py` | Handles `PASSED tests/foo.py::test_x` lines, plus pytest-xdist's `[gw1]` labels | +| Python (unittest, django) | `log_parsers/unittest_parser.py` | `test_x (pkg.Case.test_x) ... ok`; needs `-v` (`-v 2` for Django) | | JS / TS (mocha, jest) | `log_parsers/javascript.py` | From SWE-bench-Live multi-language | | Go (`go test`) | new (we write it) | `--- PASS:` / `--- FAIL:` | | Rust (`cargo test`) | new | `test foo ... ok` / `FAILED` | diff --git a/src/repo2rlenv/log_parsers/__init__.py b/src/repo2rlenv/log_parsers/__init__.py index ef4ab7f4..aae3eacc 100644 --- a/src/repo2rlenv/log_parsers/__init__.py +++ b/src/repo2rlenv/log_parsers/__init__.py @@ -14,6 +14,7 @@ * `parse_go_test` — `go test -v` * `parse_cargo_test`— `cargo test` * `parse_jest` — Jest / Mocha / Vitest + * `parse_unittest` — `python -m unittest` / Django's runner * `parse_logs(language, test_cmds, log)` — dispatch by runner. Inspects the test_cmds string for runner keywords first; falls back to the @@ -32,6 +33,7 @@ from repo2rlenv.log_parsers.go_parser import parse_go_test from repo2rlenv.log_parsers.jest_parser import parse_jest from repo2rlenv.log_parsers.pytest_parser import TestStatus, parse_pytest +from repo2rlenv.log_parsers.unittest_parser import parse_unittest __all__ = [ "TestStatus", @@ -40,13 +42,14 @@ "parse_jest", "parse_logs", "parse_pytest", + "parse_unittest", ] def _detect_runner(test_cmds: list[str]) -> str: """Inspect the bootstrap-recorded test commands and name the runner. - Returns one of: "pytest", "go", "cargo", "jest", "unknown". + Returns one of: "pytest", "unittest", "go", "cargo", "jest", "unknown". Why inspect test_cmds rather than the LanguageHint alone? A single language often has multiple runners — Python has pytest + unittest + @@ -57,6 +60,10 @@ def _detect_runner(test_cmds: list[str]) -> str: joined = " ".join(test_cmds).lower() if "pytest" in joined: return "pytest" + # unittest's own runner, Django's `manage.py test`, and Django's in-repo + # `runtests.py` all print TextTestRunner output. + if re.search(r"\bunittest\b|manage\.py\s+test\b|runtests\.py", joined): + return "unittest" if re.search(r"\bgo\s+test\b", joined): return "go" if re.search(r"\bcargo\s+test\b", joined): @@ -105,6 +112,8 @@ def parse_logs( if runner == "pytest": return parse_pytest(log) + if runner == "unittest": + return parse_unittest(log) if runner == "go": return parse_go_test(log) if runner == "cargo": diff --git a/src/repo2rlenv/log_parsers/pytest_parser.py b/src/repo2rlenv/log_parsers/pytest_parser.py index 6882d6cb..14775fcf 100644 --- a/src/repo2rlenv/log_parsers/pytest_parser.py +++ b/src/repo2rlenv/log_parsers/pytest_parser.py @@ -85,6 +85,11 @@ def parse_pytest(log: str) -> dict[str, TestStatus]: break if leading_status is not None: work = line[len(leading_status) :].strip() + # unittest's footer is `FAILED (failures=1, errors=1)`, which is a + # count, not a node ID. Without this it lands as a test named + # `(errors=1)` whenever a unittest log reaches this parser. + if work.startswith("(") and work.endswith(")"): + continue if leading_status == "SKIPPED" and re.match(r"^\[\d+\](?:\s|$)", work): # Folded skips report a file location, not a parametrized ID. tokens = work.split(maxsplit=2) diff --git a/src/repo2rlenv/log_parsers/unittest_parser.py b/src/repo2rlenv/log_parsers/unittest_parser.py new file mode 100644 index 00000000..a705bd75 --- /dev/null +++ b/src/repo2rlenv/log_parsers/unittest_parser.py @@ -0,0 +1,143 @@ +"""`python -m unittest` / Django test-runner output parser. + +unittest's verbose runner prints one line per test, with the status after an +ellipsis: + + test_add (tests.test_math.MathTests.test_add) ... ok + test_broken (tests.test_math.MathTests.test_broken) ... FAIL + test_error (tests.test_math.MathTests.test_error) ... ERROR + test_skipped (tests.test_math.MathTests.test_skipped) ... skipped 'later' + +Two shapes need care: + + * A test with a docstring prints the name, then the docstring and the + status on the NEXT line, because `getDescription` joins them with a + newline. The status therefore belongs to the name printed just above. + * A subtest failure is reported on its own indented line carrying the + parameters (`... (i=2) ... FAIL`), while the parent test's line ends + after the ellipsis with no status at all. Subtests are recorded under + the parent's identity, and the worst status wins, because the repaired + run prints only `parent ... ok`: keeping the parameters would leave + FAIL_TO_PASS with a name that never appears again. Named subtests add a + message before their parameters (`[edge case] (value=(1, 2))`), so the + whole suffix is matched loosely and discarded. + +Test names are canonicalized to the dotted id you could re-run, so the key +is stable across Python versions: 3.11+ prints the full path inside the +parentheses, older versions print only the class path with the method name +outside. + +`expected failure` counts as PASSED and `unexpected success` as FAILED, +matching unittest's own verdict (`wasSuccessful()`), so the F2P/P2P sets +agree with the suite's exit code. + +Django's runner (`manage.py test -v 2`) delegates to unittest's +TextTestRunner, so this parses its output too. + +Released under Apache-2.0. +""" + +from __future__ import annotations + +import re + +from repo2rlenv.log_parsers.pytest_parser import TestStatus + +# `test_add (pkg.mod.Class.test_add)` plus any subtest description, e.g. +# ` [edge case] (i=2)` or ` (value=(1, 2))`, which is matched but not kept. +_NAME_RE = re.compile( + r"^(?P[^\s()]+) \((?P[\w.]+)\)(?: .+)?$", +) + +_STATUS_WORDS: dict[str, TestStatus] = { + "ok": "PASSED", + "FAIL": "FAILED", + "ERROR": "ERROR", + "expected failure": "PASSED", + "unexpected success": "FAILED", +} + +# Failure blocks below the run repeat each name: `FAIL: test_x (pkg.Class.test_x)`. +_BLOCK_RE = re.compile(r"^(?PFAIL|ERROR):\s+(?P.+?)\s*$") + + +# Worst status wins when a test reports more than once, i.e. a parent whose +# subtests each report separately. +_RANK: dict[TestStatus, int] = {"SKIPPED": 0, "PASSED": 1, "FAILED": 2, "ERROR": 3} + + +def _canonical(method: str, dotted: str) -> str: + """Return the dotted id, e.g. `tests.test_math.MathTests.test_add`. + + 3.11+ prints the full path inside the parentheses; older versions print + the class path there and the method name outside. + """ + return dotted if dotted.endswith(f".{method}") else f"{dotted}.{method}" + + +def _record(out: dict[str, TestStatus], name: str, status: TestStatus) -> None: + if name in out and _RANK[out[name]] >= _RANK[status]: + return + out[name] = status + + +def _status_for(tail: str) -> TestStatus | None: + """Map the text after the ellipsis to a status, or None when unknown.""" + text = tail.strip() + if not text: + # A parent test whose subtests report separately. + return None + if text.startswith("skipped"): + return "SKIPPED" + return _STATUS_WORDS.get(text) + + +def parse_unittest(log: str) -> dict[str, TestStatus]: + """Return {test_name -> status} parsed from unittest/Django output. + + Lines that are neither a result line nor a failure-block header are + ignored, so tracebacks and the `Ran N tests` footer contribute nothing. + """ + out: dict[str, TestStatus] = {} + if not log: + return out + + # A name printed without a status is only claimed by the very next line. + pending: str | None = None + for raw in log.split("\n"): + line = raw.rstrip() + claimed, pending = pending, None + if not line.strip(): + continue + + # A named subtest message may itself contain ` ... `, so split at the + # final separator before the result word. + head, sep, tail = line.strip().rpartition(" ... ") + if sep: + m = _NAME_RE.match(head) + if m: + status = _status_for(tail) + if status is None: + # Parent of subtests: its own results follow, indented. + continue + _record(out, _canonical(m["method"], m["dotted"]), status) + elif claimed is not None: + # ` ... ok` for the name on the previous line. + status = _status_for(tail) + if status is not None: + _record(out, claimed, status) + continue + + m = _NAME_RE.match(line.strip()) + if m: + pending = _canonical(m["method"], m["dotted"]) + continue + + m = _BLOCK_RE.match(line) + if m: + name = _NAME_RE.match(m["name"]) + if name: + key = _canonical(name["method"], name["dotted"]) + _record(out, key, _STATUS_WORDS[m["status"]]) + + return out diff --git a/src/repo2rlenv/pipelines/_pr_runtime_verifier.py b/src/repo2rlenv/pipelines/_pr_runtime_verifier.py index 1b811160..01cd8527 100644 --- a/src/repo2rlenv/pipelines/_pr_runtime_verifier.py +++ b/src/repo2rlenv/pipelines/_pr_runtime_verifier.py @@ -104,6 +104,9 @@ def parse_pytest(log: str) -> dict[str, str]: break if leading is not None: work = line[len(leading) :].strip() + # unittest's `FAILED (failures=1)` footer is a count, not a node ID. + if work.startswith("(") and work.endswith(")"): + continue if leading == SKIPPED and re.match(r"^\[\d+\](?:\s|$)", work): # Folded skips report a file location, not a parametrized ID. tokens = work.split(maxsplit=2) @@ -126,6 +129,83 @@ def parse_pytest(log: str) -> dict[str, str]: return out +# Keep these patterns in sync with log_parsers/unittest_parser.py. +_UT_NAME_RE = re.compile(r"^(?P[^\s()]+) \((?P[\w.]+)\)(?: .+)?$") +_UT_BLOCK_RE = re.compile(r"^(?PFAIL|ERROR):\s+(?P.+?)\s*$") +_UT_STATUS = { + "ok": PASSED, + "FAIL": FAILED, + "ERROR": ERROR, + "expected failure": PASSED, + "unexpected success": FAILED, +} + + +# Subtests report under the parent's identity, worst status first, because a +# repaired run prints only `parent ... ok`. +_UT_RANK = {SKIPPED: 0, PASSED: 1, FAILED: 2, ERROR: 3} + + +def _ut_canonical(method: str, dotted: str) -> str: + return dotted if dotted.endswith(f".{method}") else f"{dotted}.{method}" + + +def _ut_record(out: dict[str, str], name: str, status: str) -> None: + if name in out and _UT_RANK[out[name]] >= _UT_RANK[status]: + return + out[name] = status + + +def _ut_status(tail: str) -> str | None: + text = tail.strip() + if not text: + return None + if text.startswith("skipped"): + return SKIPPED + return _UT_STATUS.get(text) + + +def parse_unittest(log: str) -> dict[str, str]: + """{test_name -> status} from `python -m unittest -v` / Django output. + + A docstring pushes the status onto the line after the name, and subtest + failures report on their own indented line while the parent prints none. + """ + out: dict[str, str] = {} + if not log: + return out + pending: str | None = None + for raw in log.split("\n"): + line = raw.rstrip() + claimed, pending = pending, None + if not line.strip(): + continue + head, sep, tail = line.strip().rpartition(" ... ") + if sep: + m = _UT_NAME_RE.match(head) + if m: + status = _ut_status(tail) + if status is not None: + _ut_record(out, _ut_canonical(m["method"], m["dotted"]), status) + elif claimed is not None: + status = _ut_status(tail) + if status is not None: + _ut_record(out, claimed, status) + continue + m = _UT_NAME_RE.match(line.strip()) + if m: + pending = _ut_canonical(m["method"], m["dotted"]) + continue + m = _UT_BLOCK_RE.match(line) + if m: + name = _UT_NAME_RE.match(m["name"]) + if name: + _ut_record( + out, _ut_canonical(name["method"], name["dotted"]), _UT_STATUS[m["status"]] + ) + return out + + _GO_TEST_RE = re.compile(r"^\s*---\s+(?PPASS|FAIL|SKIP):\s+(?P\S+)") _GO_STATUS = {"PASS": PASSED, "FAIL": FAILED, "SKIP": SKIPPED} @@ -310,6 +390,8 @@ def _detect_runner(test_cmds: str) -> str: joined = test_cmds.lower() if "pytest" in joined: return "pytest" + if re.search(r"\bunittest\b|manage\.py\s+test\b|runtests\.py", joined): + return "unittest" if re.search(r"\bgo\s+test\b", joined): return "go" if re.search(r"\bcargo\s+test\b", joined): @@ -323,6 +405,8 @@ def parse_logs(runner: str, log: str) -> dict[str, str]: """Dispatch to the right per-runner parser. Empty dict if unknown.""" if runner == "pytest": return parse_pytest(log) + if runner == "unittest": + return parse_unittest(log) if runner == "go": return parse_go_test(log) if runner == "cargo": @@ -448,7 +532,7 @@ def main(argv: list[str] | None = None) -> int: p.add_argument("--log", required=True, help="captured test-run log file") p.add_argument("--f2p", required=True, help="JSON file: FAIL_TO_PASS test names") p.add_argument("--p2p", required=True, help="JSON file: PASS_TO_PASS test names") - p.add_argument("--runner", default="", help="pytest|go|cargo|jest (else auto-detect)") + p.add_argument("--runner", default="", help="pytest|unittest|go|cargo|jest (else auto-detect)") p.add_argument("--test-cmds", default="", help="test command string (runner auto-detect)") p.add_argument("--exit-code", type=int, default=1, help="test suite exit code (fallback)") p.add_argument("--out-dir", default="/logs/verifier", help="where to write reward.{txt,json}") @@ -521,4 +605,5 @@ def main(argv: list[str] | None = None) -> int: "parse_jest", "parse_logs", "parse_pytest", + "parse_unittest", ] diff --git a/src/repo2rlenv/pipelines/pr_runtime.py b/src/repo2rlenv/pipelines/pr_runtime.py index 97f5b979..f7d86cfc 100644 --- a/src/repo2rlenv/pipelines/pr_runtime.py +++ b/src/repo2rlenv/pipelines/pr_runtime.py @@ -629,6 +629,11 @@ def normalize_test_cmds_for_runtime(test_cmds: list[str]) -> list[str]: - Drop `--collect-only` / `--co` so pytest actually runs tests - Drop `-q` / `--quiet`: suppresses per-test names; cancels `-v` in pytest 9 - Add `-v` if no verbosity flag is present + python unittest / Django: + - Add a verbosity flag when none is present, since these runners + print only dots at the default level: `-v` for unittest, `-v 2` + for `manage.py test`, and `--verbosity 2` for Django's in-repo + `tests/runtests.py`, which takes no `-v` count. go test: - Add `-v` if missing (default `go test` doesn't print --- PASS lines) cargo test: @@ -665,6 +670,17 @@ def normalize_test_cmds_for_runtime(test_cmds: list[str]) -> list[str]: if not re.search(r"\s-v\b|\s--verbose\b|-vv\b", cleaned): cleaned = cleaned.rstrip() + " -v" + # --- python unittest / Django (manage.py test, tests/runtests.py) --- + elif re.search(r"\bunittest\b|manage\.py\s+test\b|runtests\.py", cleaned): + if not re.search(r"\s-v\b|\s--verbose\b|\s--verbosity\b|-vv\b", cleaned): + if re.search(r"runtests\.py", cleaned): + flag = " --verbosity 2" + elif re.search(r"manage\.py\s+test\b", cleaned): + flag = " -v 2" # Django's runner counts verbosity + else: + flag = " -v" # unittest's own flag + cleaned = cleaned.rstrip() + flag + # --- go test --- elif re.search(r"\bgo\s+test\b", cleaned): if not re.search(r"\s-v\b", cleaned): diff --git a/tests/test_pipeline_pr_runtime.py b/tests/test_pipeline_pr_runtime.py index 7fceb822..e877ee0d 100644 --- a/tests/test_pipeline_pr_runtime.py +++ b/tests/test_pipeline_pr_runtime.py @@ -507,6 +507,27 @@ def test_normalize_preserves_existing_verbose(): assert normalize_test_cmds_for_runtime(["pytest -vv"]) == ["pytest -vv"] +def test_normalize_unittest_gets_verbosity(): + # Without -v, unittest prints only dots and no test name is parseable. + assert normalize_test_cmds_for_runtime(["python -m unittest"]) == ["python -m unittest -v"] + assert normalize_test_cmds_for_runtime(["python -m unittest discover -s tests"]) == [ + "python -m unittest discover -s tests -v" + ] + # Django's runner counts verbosity instead of toggling it. + assert normalize_test_cmds_for_runtime(["./manage.py test"]) == ["./manage.py test -v 2"] + # Already verbose ⇒ keep + assert normalize_test_cmds_for_runtime(["python manage.py test demo -v 2"]) == [ + "python manage.py test demo -v 2" + ] + # Django's in-repo runtests.py takes --verbosity, not a -v count. + assert normalize_test_cmds_for_runtime(["python tests/runtests.py"]) == [ + "python tests/runtests.py --verbosity 2" + ] + assert normalize_test_cmds_for_runtime(["python tests/runtests.py --verbosity 3"]) == [ + "python tests/runtests.py --verbosity 3" + ] + + def test_normalize_go_test_gets_v_flag(): """`go test` without -v doesn't print --- PASS lines — parser needs them.""" assert normalize_test_cmds_for_runtime(["go test ./..."]) == ["go test -v ./..."] diff --git a/tests/test_unittest_output.py b/tests/test_unittest_output.py new file mode 100644 index 00000000..0059cdc7 --- /dev/null +++ b/tests/test_unittest_output.py @@ -0,0 +1,459 @@ +"""unittest and Django runs must be parsed by both parsers, not read as pytest.""" + +from __future__ import annotations + +import io +import json +import subprocess +import sys +import unittest +from pathlib import Path + +import pytest + +from repo2rlenv.log_parsers import parse_logs +from repo2rlenv.log_parsers.unittest_parser import parse_unittest +from repo2rlenv.pipelines import _pr_runtime_verifier as verifier + + +@pytest.fixture(params=[parse_unittest, verifier.parse_unittest], ids=["canonical", "standalone"]) +def parser(request): + return request.param + + +# Real `python -m unittest -v` output (CPython 3.14): every status unittest can +# report, a test whose docstring pushes the status onto the next line, and a +# subtest failure reported under its parent. +_VERBOSE = ( + "test_add (tests.test_math.MathTests.test_add) ... ok\n" + "test_broken (tests.test_math.MathTests.test_broken) ... FAIL\n" + "test_error (tests.test_math.MathTests.test_error) ... ERROR\n" + "test_expected_failure (tests.test_math.MathTests.test_expected_failure) ... expected failure\n" + "test_skipped (tests.test_math.MathTests.test_skipped) ... skipped 'later'\n" + "test_subtests (tests.test_math.MathTests.test_subtests) ... \n" + " test_subtests (tests.test_math.MathTests.test_subtests) (i=2) ... FAIL\n" + "test_with_docstring (tests.test_math.MathTests.test_with_docstring)\n" + "Adds two negatives. ... ok\n" + "test_subtests_pass (tests.test_math.MoreTests.test_subtests_pass) ... ok\n" + "test_unexpected_success (tests.test_math.MoreTests.test_unexpected_success) ... unexpected success\n" + "\n" + "======================================================================\n" + "ERROR: test_error (tests.test_math.MathTests.test_error)\n" + "----------------------------------------------------------------------\n" + "Traceback (most recent call last):\n" + ' File "/repo/tests/test_math.py", line 16, in test_error\n' + ' raise RuntimeError("boom")\n' + "RuntimeError: boom\n" + "\n" + "======================================================================\n" + "FAIL: test_broken (tests.test_math.MathTests.test_broken)\n" + "----------------------------------------------------------------------\n" + "Traceback (most recent call last):\n" + ' File "/repo/tests/test_math.py", line 13, in test_broken\n' + " self.assertEqual(1 + 1, 3)\n" + " ~~~~~~~~~~~~~~~~^^^^^^^^^^\n" + "AssertionError: 2 != 3\n" + "\n" + "======================================================================\n" + "FAIL: test_subtests (tests.test_math.MathTests.test_subtests) (i=2)\n" + "----------------------------------------------------------------------\n" + "Traceback (most recent call last):\n" + ' File "/repo/tests/test_math.py", line 29, in test_subtests\n' + " self.assertEqual(i % 2, 1)\n" + " ~~~~~~~~~~~~~~~~^^^^^^^^^^\n" + "AssertionError: 0 != 1\n" + "\n" + "======================================================================\n" + "UNEXPECTED SUCCESS: test_unexpected_success (tests.test_math.MoreTests.test_unexpected_success)\n" + "----------------------------------------------------------------------\n" + "Ran 9 tests in 0.001s\n" + "\n" + "FAILED (failures=2, errors=1, skipped=1, expected failures=1, unexpected successes=1)\n" +) + +# The same suite without -v: dots, then the failure blocks. +_PLAIN = ( + ".FExsF..u\n" + "======================================================================\n" + "ERROR: test_error (tests.test_math.MathTests.test_error)\n" + "----------------------------------------------------------------------\n" + "Traceback (most recent call last):\n" + ' File "/repo/tests/test_math.py", line 16, in test_error\n' + ' raise RuntimeError("boom")\n' + "RuntimeError: boom\n" + "\n" + "======================================================================\n" + "FAIL: test_broken (tests.test_math.MathTests.test_broken)\n" + "----------------------------------------------------------------------\n" + "Traceback (most recent call last):\n" + ' File "/repo/tests/test_math.py", line 13, in test_broken\n' + " self.assertEqual(1 + 1, 3)\n" + " ~~~~~~~~~~~~~~~~^^^^^^^^^^\n" + "AssertionError: 2 != 3\n" + "\n" + "======================================================================\n" + "FAIL: test_subtests (tests.test_math.MathTests.test_subtests) (i=2)\n" + "----------------------------------------------------------------------\n" + "Traceback (most recent call last):\n" + ' File "/repo/tests/test_math.py", line 29, in test_subtests\n' + " self.assertEqual(i % 2, 1)\n" + " ~~~~~~~~~~~~~~~~^^^^^^^^^^\n" + "AssertionError: 0 != 1\n" + "\n" + "======================================================================\n" + "UNEXPECTED SUCCESS: test_unexpected_success (tests.test_math.MoreTests.test_unexpected_success)\n" + "----------------------------------------------------------------------\n" + "Ran 9 tests in 0.001s\n" + "\n" + "FAILED (failures=2, errors=1, skipped=1, expected failures=1, unexpected successes=1)\n" +) + +# Real `manage.py test -v 2` output (Django 6.1.1), which uses TextTestRunner. +_DJANGO = ( + "test_add (demo.tests.MathTests.test_add) ... ok\n" + "test_broken (demo.tests.MathTests.test_broken) ... FAIL\n" + "test_with_docstring (demo.tests.MathTests.test_with_docstring)\n" + "Adds two negatives. ... ok\n" + "\n" + "======================================================================\n" + "FAIL: test_broken (demo.tests.MathTests.test_broken)\n" + "----------------------------------------------------------------------\n" + "Traceback (most recent call last):\n" + ' File "/repo/dj/demo/tests.py", line 13, in test_broken\n' + " self.assertEqual(1 + 1, 3)\n" + "AssertionError: 2 != 3\n" + "\n" + "----------------------------------------------------------------------\n" + "Ran 3 tests in 0.000s\n" + "\n" + "FAILED (failures=1)\n" + "Found 3 test(s).\n" + "Skipping setup of unused database(s): default.\n" + "System check identified no issues (0 silenced).\n" +) + + +def test_every_status_is_recorded(parser): + assert parser(_VERBOSE) == { + "tests.test_math.MathTests.test_add": "PASSED", + "tests.test_math.MathTests.test_broken": "FAILED", + "tests.test_math.MathTests.test_error": "ERROR", + "tests.test_math.MathTests.test_expected_failure": "PASSED", + "tests.test_math.MathTests.test_skipped": "SKIPPED", + "tests.test_math.MathTests.test_subtests": "FAILED", + "tests.test_math.MathTests.test_with_docstring": "PASSED", + "tests.test_math.MoreTests.test_subtests_pass": "PASSED", + "tests.test_math.MoreTests.test_unexpected_success": "FAILED", + } + + +def test_plain_run_records_only_the_failure_blocks(parser): + assert parser(_PLAIN) == { + "tests.test_math.MathTests.test_broken": "FAILED", + "tests.test_math.MathTests.test_error": "ERROR", + "tests.test_math.MathTests.test_subtests": "FAILED", + } + + +def test_django_runner(parser): + assert parser(_DJANGO) == { + "demo.tests.MathTests.test_add": "PASSED", + "demo.tests.MathTests.test_broken": "FAILED", + "demo.tests.MathTests.test_with_docstring": "PASSED", + } + + +def test_pre_311_name_layout(parser): + # Before 3.11 the parentheses held only the class path. + log = "test_add (tests.test_math.MathTests) ... ok\n" + assert parser(log) == {"tests.test_math.MathTests.test_add": "PASSED"} + + +@pytest.mark.parametrize( + ("tail", "status"), + [ + ("ok", "PASSED"), + ("FAIL", "FAILED"), + ("ERROR", "ERROR"), + ("expected failure", "PASSED"), + ("unexpected success", "FAILED"), + ("skipped 'needs network'", "SKIPPED"), + ("skipped", "SKIPPED"), + ], +) +def test_status_words(parser, tail, status): + log = f"test_x (pkg.mod.Case.test_x) ... {tail}\n" + assert parser(log) == {"pkg.mod.Case.test_x": status} + + +def test_docstring_status_only_binds_to_the_line_above(parser): + # A name with no status, then an unrelated line, must record nothing. + log = "test_x (pkg.mod.Case.test_x)\nsome other output ... ok\n" + assert parser(log) == {"pkg.mod.Case.test_x": "PASSED"} + log = "test_x (pkg.mod.Case.test_x)\n\nnoise\nmore noise ... ok\n" + assert parser(log) == {} + + +def test_tracebacks_and_footers_are_ignored(parser): + log = ( + "======================================================================\n" + "FAIL: test_broken (pkg.mod.Case.test_broken)\n" + "----------------------------------------------------------------------\n" + "Traceback (most recent call last):\n" + ' File "/repo/tests/test_math.py", line 13, in test_broken\n' + " self.assertEqual(1 + 1, 3)\n" + "AssertionError: 2 != 3\n" + "\n" + "Ran 9 tests in 0.001s\n" + "\n" + "FAILED (failures=1, errors=1, skipped=1)\n" + ) + assert parser(log) == {"pkg.mod.Case.test_broken": "FAILED"} + + +def test_dispatcher_routes_unittest_commands(): + for cmd in [ + "python -m unittest -v", + "python -m unittest discover -s tests -v", + "python manage.py test -v 2", + "python tests/runtests.py --verbosity 2", + ]: + assert parse_logs([cmd], _VERBOSE), cmd + # The standalone verifier detects the same commands from --test-cmds. + assert verifier.parse_logs(verifier._detect_runner("python -m unittest -v"), _VERBOSE) + + +def test_unittest_footer_is_not_a_pytest_node(): + # A wrapper command lands in the pytest parser; its footer must not + # become a test called `(errors=1)`. + from repo2rlenv.log_parsers.pytest_parser import parse_pytest + + for parse in (parse_pytest, verifier.parse_pytest): + assert parse(_VERBOSE) == {} + assert parse("FAILED tests/test_a.py::test_x - AssertionError") == { + "tests/test_a.py::test_x": "FAILED" + } + + +def test_long_lines_do_not_stall_parsing(): + # Isolate the timeout so a backtracking regression cannot hang the suite. + result = subprocess.run( + [ + sys.executable, + "-c", + "from repo2rlenv.log_parsers.unittest_parser import parse_unittest\n" + "from repo2rlenv.pipelines._pr_runtime_verifier import parse_unittest as standalone\n" + "noise = ['t (' + 'a' * 200000 + ')', 'x' * 200000 + ' ... ok',\n" + " 't (pkg.C.t)' + ' (i=1)' * 50000 + ' ... FAIL',\n" + " 'FAIL: t (' + 'b.' * 100000 + 'C.t)']\n" + "log = '\\n'.join(noise) + '\\ntest_x (pkg.mod.Case.test_x) ... ok\\n'\n" + "for parser in (parse_unittest, standalone):\n" + " assert parser(log)['pkg.mod.Case.test_x'] == 'PASSED'\n", + ], + capture_output=True, + text=True, + timeout=10, + ) + assert result.returncode == 0, result.stdout + result.stderr + + +# A real fix under unittest: `div` gains a zero check, and the test that +# covers it has a docstring, so its status is printed on the following line. +_PRE = ( + "test_divides (tests.test_calc.DivTests.test_divides) ... ok\n" + "test_rejects_zero (tests.test_calc.DivTests.test_rejects_zero)\n" + "Rejects a zero divisor. ... ERROR\n" + "\n" + "======================================================================\n" + "ERROR: test_rejects_zero (tests.test_calc.DivTests.test_rejects_zero)\n" + "Rejects a zero divisor.\n" + "----------------------------------------------------------------------\n" + "Traceback (most recent call last):\n" + ' File "/repo/tests/test_calc.py", line 13, in test_rejects_zero\n' + " div(1, 0)\n" + " ~~~^^^^^^\n" + ' File "/repo/calc.py", line 2, in div\n' + " return a / b\n" + " ~~^~~\n" + "ZeroDivisionError: division by zero\n" + "\n" + "----------------------------------------------------------------------\n" + "Ran 2 tests in 0.001s\n" + "\n" + "FAILED (errors=1)\n" +) + +_POST = ( + "test_divides (tests.test_calc.DivTests.test_divides) ... ok\n" + "test_rejects_zero (tests.test_calc.DivTests.test_rejects_zero)\n" + "Rejects a zero divisor. ... ok\n" + "\n" + "----------------------------------------------------------------------\n" + "Ran 2 tests in 0.000s\n" + "\n" + "OK\n" +) + + +def test_unittest_fix_yields_an_oracle_and_grades_it(tmp_path: Path): + cmd = ["python -m unittest -v"] + pre, post = parse_logs(cmd, _PRE), parse_logs(cmd, _POST) + f2p = sorted( + name for name, st in pre.items() if st in ("FAILED", "ERROR") and post.get(name) == "PASSED" + ) + p2p = sorted(name for name, st in pre.items() if st == "PASSED" and post.get(name) == "PASSED") + assert f2p == ["tests.test_calc.DivTests.test_rejects_zero"] + assert p2p == ["tests.test_calc.DivTests.test_divides"] + + # Run the verifier in isolation, with no installed repo2rlenv imports. + standalone = tmp_path / "verifier.py" + standalone.write_text(Path(verifier.__file__).read_text(encoding="utf-8"), encoding="utf-8") + (tmp_path / "out.log").write_text(_POST, encoding="utf-8") + (tmp_path / "f2p.json").write_text(json.dumps(f2p), encoding="utf-8") + (tmp_path / "p2p.json").write_text(json.dumps(p2p), encoding="utf-8") + result = subprocess.run( + [ + sys.executable, + "-I", + str(standalone), + "--log", + "out.log", + "--f2p", + "f2p.json", + "--p2p", + "p2p.json", + "--test-cmds", + "python -m unittest -v", + "--exit-code", + "0", + "--out-dir", + "rewards", + ], + cwd=tmp_path, + capture_output=True, + text=True, + timeout=30, + ) + assert result.returncode == 0, result.stdout + result.stderr + details = json.loads((tmp_path / "rewards/reward-details.json").read_text()) + assert details["runner"] == "unittest" + assert details["reward"] == 1.0 + assert details["resolved"] is True + + +# A real fix whose failing tests are subtests: `is_odd` is corrected, and one +# subtest value is a tuple, so its parameters contain nested parentheses. +_SUB_PRE = ( + "test_all_odd (tests.test_sub.SubTests.test_all_odd) ... \n" + " test_all_odd (tests.test_sub.SubTests.test_all_odd) (value=3) ... FAIL\n" + "test_pairs (tests.test_sub.SubTests.test_pairs) ... \n" + " test_pairs (tests.test_sub.SubTests.test_pairs) (value=(3, 4)) ... FAIL\n" + "\n" + "======================================================================\n" + "FAIL: test_all_odd (tests.test_sub.SubTests.test_all_odd) (value=3)\n" + "----------------------------------------------------------------------\n" + "Traceback (most recent call last):\n" + ' File "/repo/tests/test_sub.py", line 10, in test_all_odd\n' + " self.assertTrue(is_odd(value))\n" + " ~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^\n" + "AssertionError: False is not true\n" + "\n" + "======================================================================\n" + "FAIL: test_pairs (tests.test_sub.SubTests.test_pairs) (value=(3, 4))\n" + "----------------------------------------------------------------------\n" + "Traceback (most recent call last):\n" + ' File "/repo/tests/test_sub.py", line 15, in test_pairs\n' + " self.assertTrue(is_odd(pair[0]))\n" + " ~~~~~~~~~~~~~~~^^^^^^^^^^^^^^^^^\n" + "AssertionError: False is not true\n" + "\n" + "----------------------------------------------------------------------\n" + "Ran 2 tests in 0.001s\n" + "\n" + "FAILED (failures=2)\n" +) + +_SUB_POST = ( + "test_all_odd (tests.test_sub.SubTests.test_all_odd) ... ok\n" + "test_pairs (tests.test_sub.SubTests.test_pairs) ... ok\n" + "\n" + "----------------------------------------------------------------------\n" + "Ran 2 tests in 0.000s\n" + "\n" + "OK\n" +) + + +def test_subtest_failures_report_under_the_parent(parser): + assert parser(_SUB_PRE) == { + "tests.test_sub.SubTests.test_all_odd": "FAILED", + "tests.test_sub.SubTests.test_pairs": "FAILED", + } + assert parser(_SUB_POST) == { + "tests.test_sub.SubTests.test_all_odd": "PASSED", + "tests.test_sub.SubTests.test_pairs": "PASSED", + } + + +def test_subtests_can_transition_from_failing_to_passing(): + # Keeping the parameters would leave a FAIL_TO_PASS name that the repaired + # run never prints, since it reports only `parent ... ok`. + cmd = ["python -m unittest -v"] + pre, post = parse_logs(cmd, _SUB_PRE), parse_logs(cmd, _SUB_POST) + f2p = sorted( + name for name, st in pre.items() if st in ("FAILED", "ERROR") and post.get(name) == "PASSED" + ) + assert f2p == [ + "tests.test_sub.SubTests.test_all_odd", + "tests.test_sub.SubTests.test_pairs", + ] + + +@pytest.mark.parametrize( + "params", + ["(i=2)", "(value=(1, 2))", "(value={'a': 1})", "(msg='a (b) c')", "(a=1) (b=2)"], +) +def test_subtest_parameters_are_matched_but_not_kept(parser, params): + log = f" test_x (pkg.mod.Case.test_x) {params} ... FAIL\n" + assert parser(log) == {"pkg.mod.Case.test_x": "FAILED"} + + +def test_named_subtests_from_text_test_runner(parser): + def run_named_subtests(case): + with case.subTest("edge ... case", i=1): + case.assertEqual(1, 2) + with case.subTest("message only"): + case.assertEqual(1, 2) + + case_type = type( + "NamedSubTests", + (unittest.TestCase,), + {"__module__": "pkg", "test_named": run_named_subtests}, + ) + stream = io.StringIO() + unittest.TextTestRunner(stream=stream, verbosity=2).run( + unittest.defaultTestLoader.loadTestsFromTestCase(case_type) + ) + log = stream.getvalue() + + assert "[edge ... case] (i=1) ... FAIL" in log + assert "[message only] ... FAIL" in log + assert parser(log) == {"pkg.NamedSubTests.test_named": "FAILED"} + + +@pytest.mark.parametrize( + ("first", "second", "expected"), + [ + ("ok", "FAIL", "FAILED"), + ("FAIL", "ok", "FAILED"), + ("FAIL", "ERROR", "ERROR"), + ("skipped 'x'", "ok", "PASSED"), + ("ok", "skipped 'x'", "PASSED"), + ], +) +def test_worst_status_wins_when_a_test_reports_twice(parser, first, second, expected): + log = ( + f" test_x (pkg.mod.Case.test_x) (i=1) ... {first}\n" + f" test_x (pkg.mod.Case.test_x) (i=2) ... {second}\n" + ) + assert parser(log) == {"pkg.mod.Case.test_x": expected}