Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions docs/pipelines/pr_runtime.md
Original file line number Diff line number Diff line change
Expand Up @@ -82,8 +82,8 @@ Per-language log parsers map raw test output to `{PASSED, FAILED, SKIPPED, ERROR

| Language | Parser | Notes |
|---|---|---|
| Python (pytest) | `log_parsers/python.py` | Handles `PASSED tests/foo.py::test_x` lines |
| Python (django) | same module, django variant | Django runs are slightly different |
| Python (pytest) | `log_parsers/pytest_parser.py` | Handles `PASSED tests/foo.py::test_x` lines, plus pytest-xdist's `[gw1]` labels |
| Python (unittest, django) | `log_parsers/unittest_parser.py` | `test_x (pkg.Case.test_x) ... ok`; needs `-v` (`-v 2` for Django) |
| JS / TS (mocha, jest) | `log_parsers/javascript.py` | From SWE-bench-Live multi-language |
| Go (`go test`) | new (we write it) | `--- PASS:` / `--- FAIL:` |
| Rust (`cargo test`) | new | `test foo ... ok` / `FAILED` |
Expand Down
11 changes: 10 additions & 1 deletion src/repo2rlenv/log_parsers/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -14,6 +14,7 @@
* `parse_go_test` — `go test -v`
* `parse_cargo_test`— `cargo test`
* `parse_jest` — Jest / Mocha / Vitest
* `parse_unittest` — `python -m unittest` / Django's runner

* `parse_logs(language, test_cmds, log)` — dispatch by runner. Inspects
the test_cmds string for runner keywords first; falls back to the
Expand All @@ -32,6 +33,7 @@
from repo2rlenv.log_parsers.go_parser import parse_go_test
from repo2rlenv.log_parsers.jest_parser import parse_jest
from repo2rlenv.log_parsers.pytest_parser import TestStatus, parse_pytest
from repo2rlenv.log_parsers.unittest_parser import parse_unittest

__all__ = [
"TestStatus",
Expand All @@ -40,13 +42,14 @@
"parse_jest",
"parse_logs",
"parse_pytest",
"parse_unittest",
]


def _detect_runner(test_cmds: list[str]) -> str:
"""Inspect the bootstrap-recorded test commands and name the runner.

Returns one of: "pytest", "go", "cargo", "jest", "unknown".
Returns one of: "pytest", "unittest", "go", "cargo", "jest", "unknown".

Why inspect test_cmds rather than the LanguageHint alone? A single
language often has multiple runners — Python has pytest + unittest +
Expand All @@ -57,6 +60,10 @@ def _detect_runner(test_cmds: list[str]) -> str:
joined = " ".join(test_cmds).lower()
if "pytest" in joined:
return "pytest"
# unittest's own runner, Django's `manage.py test`, and Django's in-repo
# `runtests.py` all print TextTestRunner output.
if re.search(r"\bunittest\b|manage\.py\s+test\b|runtests\.py", joined):
return "unittest"
if re.search(r"\bgo\s+test\b", joined):
return "go"
if re.search(r"\bcargo\s+test\b", joined):
Expand Down Expand Up @@ -105,6 +112,8 @@ def parse_logs(

if runner == "pytest":
return parse_pytest(log)
if runner == "unittest":
return parse_unittest(log)
if runner == "go":
return parse_go_test(log)
if runner == "cargo":
Expand Down
5 changes: 5 additions & 0 deletions src/repo2rlenv/log_parsers/pytest_parser.py
Original file line number Diff line number Diff line change
Expand Up @@ -85,6 +85,11 @@ def parse_pytest(log: str) -> dict[str, TestStatus]:
break
if leading_status is not None:
work = line[len(leading_status) :].strip()
# unittest's footer is `FAILED (failures=1, errors=1)`, which is a
# count, not a node ID. Without this it lands as a test named
# `(errors=1)` whenever a unittest log reaches this parser.
if work.startswith("(") and work.endswith(")"):
continue
if leading_status == "SKIPPED" and re.match(r"^\[\d+\](?:\s|$)", work):
# Folded skips report a file location, not a parametrized ID.
tokens = work.split(maxsplit=2)
Expand Down
143 changes: 143 additions & 0 deletions src/repo2rlenv/log_parsers/unittest_parser.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,143 @@
"""`python -m unittest` / Django test-runner output parser.

unittest's verbose runner prints one line per test, with the status after an
ellipsis:

test_add (tests.test_math.MathTests.test_add) ... ok
test_broken (tests.test_math.MathTests.test_broken) ... FAIL
test_error (tests.test_math.MathTests.test_error) ... ERROR
test_skipped (tests.test_math.MathTests.test_skipped) ... skipped 'later'

Two shapes need care:

* A test with a docstring prints the name, then the docstring and the
status on the NEXT line, because `getDescription` joins them with a
newline. The status therefore belongs to the name printed just above.
* A subtest failure is reported on its own indented line carrying the
parameters (`... (i=2) ... FAIL`), while the parent test's line ends
after the ellipsis with no status at all. Subtests are recorded under
the parent's identity, and the worst status wins, because the repaired
run prints only `parent ... ok`: keeping the parameters would leave
FAIL_TO_PASS with a name that never appears again. Named subtests add a
message before their parameters (`[edge case] (value=(1, 2))`), so the
whole suffix is matched loosely and discarded.

Test names are canonicalized to the dotted id you could re-run, so the key
is stable across Python versions: 3.11+ prints the full path inside the
parentheses, older versions print only the class path with the method name
outside.

`expected failure` counts as PASSED and `unexpected success` as FAILED,
matching unittest's own verdict (`wasSuccessful()`), so the F2P/P2P sets
agree with the suite's exit code.

Django's runner (`manage.py test -v 2`) delegates to unittest's
TextTestRunner, so this parses its output too.

Released under Apache-2.0.
"""

from __future__ import annotations

import re

from repo2rlenv.log_parsers.pytest_parser import TestStatus

# `test_add (pkg.mod.Class.test_add)` plus any subtest description, e.g.
# ` [edge case] (i=2)` or ` (value=(1, 2))`, which is matched but not kept.
_NAME_RE = re.compile(
r"^(?P<method>[^\s()]+) \((?P<dotted>[\w.]+)\)(?: .+)?$",
)

_STATUS_WORDS: dict[str, TestStatus] = {
"ok": "PASSED",
"FAIL": "FAILED",
"ERROR": "ERROR",
"expected failure": "PASSED",
"unexpected success": "FAILED",
}

# Failure blocks below the run repeat each name: `FAIL: test_x (pkg.Class.test_x)`.
_BLOCK_RE = re.compile(r"^(?P<status>FAIL|ERROR):\s+(?P<name>.+?)\s*$")


# Worst status wins when a test reports more than once, i.e. a parent whose
# subtests each report separately.
_RANK: dict[TestStatus, int] = {"SKIPPED": 0, "PASSED": 1, "FAILED": 2, "ERROR": 3}


def _canonical(method: str, dotted: str) -> str:
"""Return the dotted id, e.g. `tests.test_math.MathTests.test_add`.

3.11+ prints the full path inside the parentheses; older versions print
the class path there and the method name outside.
"""
return dotted if dotted.endswith(f".{method}") else f"{dotted}.{method}"


def _record(out: dict[str, TestStatus], name: str, status: TestStatus) -> None:
if name in out and _RANK[out[name]] >= _RANK[status]:
return
out[name] = status


def _status_for(tail: str) -> TestStatus | None:
"""Map the text after the ellipsis to a status, or None when unknown."""
text = tail.strip()
if not text:
# A parent test whose subtests report separately.
return None
if text.startswith("skipped"):
return "SKIPPED"
return _STATUS_WORDS.get(text)


def parse_unittest(log: str) -> dict[str, TestStatus]:
"""Return {test_name -> status} parsed from unittest/Django output.

Lines that are neither a result line nor a failure-block header are
ignored, so tracebacks and the `Ran N tests` footer contribute nothing.
"""
out: dict[str, TestStatus] = {}
if not log:
return out

# A name printed without a status is only claimed by the very next line.
pending: str | None = None
for raw in log.split("\n"):
line = raw.rstrip()
claimed, pending = pending, None
if not line.strip():
continue

# A named subtest message may itself contain ` ... `, so split at the
# final separator before the result word.
head, sep, tail = line.strip().rpartition(" ... ")
if sep:
m = _NAME_RE.match(head)
if m:
status = _status_for(tail)
if status is None:
# Parent of subtests: its own results follow, indented.
continue
_record(out, _canonical(m["method"], m["dotted"]), status)
elif claimed is not None:
# `<docstring> ... ok` for the name on the previous line.
status = _status_for(tail)
if status is not None:
_record(out, claimed, status)
continue

m = _NAME_RE.match(line.strip())
if m:
pending = _canonical(m["method"], m["dotted"])
continue

m = _BLOCK_RE.match(line)
if m:
name = _NAME_RE.match(m["name"])
if name:
key = _canonical(name["method"], name["dotted"])
_record(out, key, _STATUS_WORDS[m["status"]])

return out
87 changes: 86 additions & 1 deletion src/repo2rlenv/pipelines/_pr_runtime_verifier.py
Original file line number Diff line number Diff line change
Expand Up @@ -104,6 +104,9 @@ def parse_pytest(log: str) -> dict[str, str]:
break
if leading is not None:
work = line[len(leading) :].strip()
# unittest's `FAILED (failures=1)` footer is a count, not a node ID.
if work.startswith("(") and work.endswith(")"):
continue
if leading == SKIPPED and re.match(r"^\[\d+\](?:\s|$)", work):
# Folded skips report a file location, not a parametrized ID.
tokens = work.split(maxsplit=2)
Expand All @@ -126,6 +129,83 @@ def parse_pytest(log: str) -> dict[str, str]:
return out


# Keep these patterns in sync with log_parsers/unittest_parser.py.
_UT_NAME_RE = re.compile(r"^(?P<method>[^\s()]+) \((?P<dotted>[\w.]+)\)(?: .+)?$")
_UT_BLOCK_RE = re.compile(r"^(?P<status>FAIL|ERROR):\s+(?P<name>.+?)\s*$")
_UT_STATUS = {
"ok": PASSED,
"FAIL": FAILED,
"ERROR": ERROR,
"expected failure": PASSED,
"unexpected success": FAILED,
}


# Subtests report under the parent's identity, worst status first, because a
# repaired run prints only `parent ... ok`.
_UT_RANK = {SKIPPED: 0, PASSED: 1, FAILED: 2, ERROR: 3}


def _ut_canonical(method: str, dotted: str) -> str:
return dotted if dotted.endswith(f".{method}") else f"{dotted}.{method}"


def _ut_record(out: dict[str, str], name: str, status: str) -> None:
if name in out and _UT_RANK[out[name]] >= _UT_RANK[status]:
return
out[name] = status


def _ut_status(tail: str) -> str | None:
text = tail.strip()
if not text:
return None
if text.startswith("skipped"):
return SKIPPED
return _UT_STATUS.get(text)


def parse_unittest(log: str) -> dict[str, str]:
"""{test_name -> status} from `python -m unittest -v` / Django output.

A docstring pushes the status onto the line after the name, and subtest
failures report on their own indented line while the parent prints none.
"""
out: dict[str, str] = {}
if not log:
return out
pending: str | None = None
for raw in log.split("\n"):
line = raw.rstrip()
claimed, pending = pending, None
if not line.strip():
continue
head, sep, tail = line.strip().rpartition(" ... ")
if sep:
m = _UT_NAME_RE.match(head)
if m:
status = _ut_status(tail)
if status is not None:
_ut_record(out, _ut_canonical(m["method"], m["dotted"]), status)
elif claimed is not None:
status = _ut_status(tail)
if status is not None:
_ut_record(out, claimed, status)
continue
m = _UT_NAME_RE.match(line.strip())
if m:
pending = _ut_canonical(m["method"], m["dotted"])
continue
m = _UT_BLOCK_RE.match(line)
if m:
name = _UT_NAME_RE.match(m["name"])
if name:
_ut_record(
out, _ut_canonical(name["method"], name["dotted"]), _UT_STATUS[m["status"]]
)
return out


_GO_TEST_RE = re.compile(r"^\s*---\s+(?P<status>PASS|FAIL|SKIP):\s+(?P<name>\S+)")
_GO_STATUS = {"PASS": PASSED, "FAIL": FAILED, "SKIP": SKIPPED}

Expand Down Expand Up @@ -310,6 +390,8 @@ def _detect_runner(test_cmds: str) -> str:
joined = test_cmds.lower()
if "pytest" in joined:
return "pytest"
if re.search(r"\bunittest\b|manage\.py\s+test\b|runtests\.py", joined):
return "unittest"
if re.search(r"\bgo\s+test\b", joined):
return "go"
if re.search(r"\bcargo\s+test\b", joined):
Expand All @@ -323,6 +405,8 @@ def parse_logs(runner: str, log: str) -> dict[str, str]:
"""Dispatch to the right per-runner parser. Empty dict if unknown."""
if runner == "pytest":
return parse_pytest(log)
if runner == "unittest":
return parse_unittest(log)
if runner == "go":
return parse_go_test(log)
if runner == "cargo":
Expand Down Expand Up @@ -448,7 +532,7 @@ def main(argv: list[str] | None = None) -> int:
p.add_argument("--log", required=True, help="captured test-run log file")
p.add_argument("--f2p", required=True, help="JSON file: FAIL_TO_PASS test names")
p.add_argument("--p2p", required=True, help="JSON file: PASS_TO_PASS test names")
p.add_argument("--runner", default="", help="pytest|go|cargo|jest (else auto-detect)")
p.add_argument("--runner", default="", help="pytest|unittest|go|cargo|jest (else auto-detect)")
p.add_argument("--test-cmds", default="", help="test command string (runner auto-detect)")
p.add_argument("--exit-code", type=int, default=1, help="test suite exit code (fallback)")
p.add_argument("--out-dir", default="/logs/verifier", help="where to write reward.{txt,json}")
Expand Down Expand Up @@ -521,4 +605,5 @@ def main(argv: list[str] | None = None) -> int:
"parse_jest",
"parse_logs",
"parse_pytest",
"parse_unittest",
]
16 changes: 16 additions & 0 deletions src/repo2rlenv/pipelines/pr_runtime.py
Original file line number Diff line number Diff line change
Expand Up @@ -629,6 +629,11 @@ def normalize_test_cmds_for_runtime(test_cmds: list[str]) -> list[str]:
- Drop `--collect-only` / `--co` so pytest actually runs tests
- Drop `-q` / `--quiet`: suppresses per-test names; cancels `-v` in pytest 9
- Add `-v` if no verbosity flag is present
python unittest / Django:
- Add a verbosity flag when none is present, since these runners
print only dots at the default level: `-v` for unittest, `-v 2`
for `manage.py test`, and `--verbosity 2` for Django's in-repo
`tests/runtests.py`, which takes no `-v` count.
go test:
- Add `-v` if missing (default `go test` doesn't print --- PASS lines)
cargo test:
Expand Down Expand Up @@ -665,6 +670,17 @@ def normalize_test_cmds_for_runtime(test_cmds: list[str]) -> list[str]:
if not re.search(r"\s-v\b|\s--verbose\b|-vv\b", cleaned):
cleaned = cleaned.rstrip() + " -v"

# --- python unittest / Django (manage.py test, tests/runtests.py) ---
elif re.search(r"\bunittest\b|manage\.py\s+test\b|runtests\.py", cleaned):
if not re.search(r"\s-v\b|\s--verbose\b|\s--verbosity\b|-vv\b", cleaned):
if re.search(r"runtests\.py", cleaned):
flag = " --verbosity 2"
elif re.search(r"manage\.py\s+test\b", cleaned):
flag = " -v 2" # Django's runner counts verbosity
else:
flag = " -v" # unittest's own flag
cleaned = cleaned.rstrip() + flag

# --- go test ---
elif re.search(r"\bgo\s+test\b", cleaned):
if not re.search(r"\s-v\b", cleaned):
Expand Down
Loading