From ad45ffc04e01ce31799b3d87d60612912dc17419 Mon Sep 17 00:00:00 2001 From: Arjun Sharma Date: Wed, 7 Oct 2026 06:34:42 +0000 Subject: [PATCH 1/5] task template: pytest regression check runs the base tests and a true base tree Before the agent-side run, pytest_runner.py puts test files the agent modified or deleted back to FC_BASE and removes test files it added. It writes the list to /logs/artifacts/test_edits.json (also under "test_edits" in test_results.json). Test selection still uses the changed source files, plus the restored test files. The base-side run no longer uses git stash. It copies every file that differs from FC_BASE (committed, uncommitted and untracked), checks out the FC_BASE versions, removes added files, runs, and copies the agent files and the git index back. It then compares file hashes with the state before the run. A mismatch or a failed patched rebuild sets regression.restore_ok to false and the runner exits 3. --- .../template/tests/pytest_runner.py | 158 +++++++++++++++--- tests/docker/test_pytest_runner_regression.py | 85 +++++++++- tests/docker/test_template_lint.py | 7 + 3 files changed, 228 insertions(+), 22 deletions(-) diff --git a/src/datasmith/harbor_adapter/template/tests/pytest_runner.py b/src/datasmith/harbor_adapter/template/tests/pytest_runner.py index 0fe6f6fb..02fc8df9 100644 --- a/src/datasmith/harbor_adapter/template/tests/pytest_runner.py +++ b/src/datasmith/harbor_adapter/template/tests/pytest_runner.py @@ -3,11 +3,13 @@ from __future__ import print_function import argparse +import hashlib import json import os import re import shlex import subprocess +import sys import time from glob import glob from pathlib import Path @@ -196,6 +198,8 @@ def discover_test_files_from_changed(changed_files, repo_root): continue if is_test_file(rel): + if not os.path.isfile(abs_path): + continue mapping.setdefault(rel, []) mapping[rel].append(rel) selected.add(rel) @@ -302,23 +306,102 @@ class _BaseRebuildFailed(Exception): pass -def run_base_and_diff(selected_tests, extra_args, repo_root, agent_results): - """Stash the agent's edits, rerun the same tests at base, restore; flag base-pass -> agent-fail.""" +def _changed_vs(base_ref, repo_root): + """{path: git status letter} for the working tree against ``base_ref``; untracked files are 'A'.""" + code, out, err = _run(["git", "diff", "--name-status", "--no-renames", "-z", base_ref], cwd=repo_root) + if code != 0: + raise RuntimeError("git diff %s failed: %s" % (base_ref, err.strip()[-200:])) + fields = out.split("\0") + changed = {fields[i + 1]: fields[i][:1] for i in range(0, len(fields) - 1, 2) if fields[i]} + code, out, _ = _run(["git", "ls-files", "-z", "--others", "--exclude-standard"], cwd=repo_root) + changed.update((p, "A") for p in out.split("\0") if p and code == 0) + return changed + + +def _checkout(base_ref, paths, repo_root): + if paths: + code, _, err = _run(["git", "--literal-pathspecs", "checkout", base_ref, "--"] + list(paths), cwd=repo_root) + if code != 0: + raise RuntimeError("git checkout %s failed: %s" % (base_ref, err.strip()[-200:])) + + +def _remove(path): + if os.path.lexists(path): + os.remove(path) + try: + os.removedirs(os.path.dirname(path)) + except OSError: + pass + + +def _fingerprint(paths, repo_root): + """{path: sha256 of the file bytes or symlink target, None if missing}.""" + fp = {} + for path in paths: + full = os.path.join(repo_root, path) + if os.path.islink(full): + fp[path] = hashlib.sha256(os.readlink(full).encode()).hexdigest() + elif os.path.isfile(full): + with open(full, "rb") as fh: + fp[path] = hashlib.sha256(fh.read()).hexdigest() + else: + fp[path] = None + return fp + + +def is_test_path(path): + """Test code: test_*.py, *_test.py, conftest.py, pytest.ini, or any file under a tests/test/_tests directory.""" + parts = path.split("/") + return is_test_file(path) or parts[-1] in ("conftest.py", "pytest.ini") or any(p in _TEST_DIRS for p in parts[:-1]) + + +def restore_test_edits(base_ref, repo_root): + """Put the test files back to ``base_ref`` so the agent cannot weaken the tests that judge it.""" + edits = {"modified": [], "added": [], "deleted": [], "config_changed": []} + for path, status in sorted(_changed_vs(base_ref, repo_root).items()): + if os.path.basename(path) in ("pyproject.toml", "setup.cfg", "tox.ini"): + edits["config_changed"].append(path) + elif is_test_path(path): + edits[{"A": "added", "D": "deleted"}.get(status, "modified")].append(path) + _checkout(base_ref, edits["modified"] + edits["deleted"], repo_root) + edits["added_sha256"] = _fingerprint(edits["added"], repo_root) + for path in edits["added"]: + _remove(os.path.join(repo_root, path)) + return edits + + +def run_base_and_diff(selected_tests, extra_args, repo_root, agent_results, base_ref="HEAD"): + """Put every changed file back to ``base_ref``, rerun the same tests, restore the agent tree; flag base-pass -> agent-fail.""" + import shutil import sys as _sys import tempfile as _tempfile agent_oc = _final_outcomes(agent_results) - # Tracked changes only: untracked generated files (e.g. setuptools-scm _version.py) must stay importable. - code, out, _ = _run( - ["git", "stash", "push", "-m", "fc-regression"], cwd=repo_root - ) - stashed = code == 0 and "No local changes" not in (out or "") + changed = sorted(_changed_vs(base_ref, repo_root)) + # Committed and untracked agent changes are reverted too, unlike git stash. + backup = _tempfile.mkdtemp(prefix="fcagent_") + index = os.path.join(repo_root, _run(["git", "rev-parse", "--git-path", "index"], cwd=repo_root)[1].strip()) + if os.path.isfile(index): + shutil.copy2(index, os.path.join(backup, ".index")) + for path in changed: + src = os.path.join(repo_root, path) + if os.path.lexists(src): + dst = os.path.join(backup, "tree", path) + os.makedirs(os.path.dirname(dst), exist_ok=True) + shutil.copy2(src, dst, follow_symlinks=False) + before = _fingerprint(changed, repo_root) + reverted = bool(changed) + restore_ok = True base_oc = {} base_error = None # test.sh sets this when the patch touches compiled sources; base must not import the patched .so. rebuild = os.environ.get("FC_REBUILD_CMD") - if stashed: + if reverted: try: + at_base = [p for p in changed if _run(["git", "cat-file", "-e", "%s:%s" % (base_ref, p)], cwd=repo_root)[0] == 0] + for path in set(changed) - set(at_base): + _remove(os.path.join(repo_root, path)) + _checkout(base_ref, at_base, repo_root) if rebuild: rc, _, err = _run(["bash", "-c", rebuild], cwd=repo_root) if rc != 0: @@ -358,17 +441,39 @@ def run_base_and_diff(selected_tests, extra_args, repo_root, agent_results): pass except _BaseRebuildFailed as e: base_error = "base rebuild failed: %s" % e + except Exception as e: # noqa: BLE001 + base_error = "base run failed: %s: %s" % (type(e).__name__, e) finally: - _run(["git", "stash", "pop"], cwd=repo_root) + try: + for path in changed: + saved = os.path.join(backup, "tree", path) + dst = os.path.join(repo_root, path) + _remove(dst) + if os.path.lexists(saved): + os.makedirs(os.path.dirname(dst), exist_ok=True) + shutil.copy2(saved, dst, follow_symlinks=False) + if os.path.isfile(os.path.join(backup, ".index")): + shutil.copy2(os.path.join(backup, ".index"), index) + # Timing runs after this on the same tree, so check it is the agent's tree again. + if _fingerprint(changed, repo_root) != before: + restore_ok = False + base_error = (base_error or "") + "; agent tree differs after restore" + except Exception as e: # noqa: BLE001 + restore_ok = False + base_error = (base_error or "") + "; restore failed: %s: %s" % (type(e).__name__, e) if rebuild and _run(["bash", "-c", rebuild], cwd=repo_root)[0] != 0: - base_error = (base_error or "") + "; patched rebuild after stash pop failed" + restore_ok = False + base_error = (base_error or "") + "; patched rebuild after restore failed" + shutil.rmtree(backup, ignore_errors=True) regressed = sorted( n for n, o in agent_oc.items() if o in ("failed", "error") and base_oc.get(n) == "passed" ) out = { - "ran": bool(stashed and selected_tests and base_oc), - "stashed": stashed, + "ran": bool(reverted and selected_tests and base_oc), + "reverted": reverted, + "restore_ok": restore_ok, + "base_ref": base_ref, "n_selected_tests": len(selected_tests), "n_agent_tests": len(agent_oc), "n_base_pass": sum(1 for o in base_oc.values() if o == "passed"), @@ -812,6 +917,9 @@ def main(args) -> dict: repo_root=repo_root, ) + # Test files restored from base still belong to the agent's change, so their tests run. + changed = sorted(set(changed) | set(getattr(args, "restored_tests", []))) + if args.only != "all": filtered = [] for p in changed: @@ -910,6 +1018,20 @@ def main(args) -> dict: args = parse_args() # The retries below mutate extra_args; the regression gate uses the originals. _orig_extra = args.extra_args + # Must match parser.py's LOG_DIR lookup; /logs/test_results.json is a best-effort legacy copy. + logs_root = Path( + os.environ.get("T_BENCH_TASK_LOGS_PATH") + or os.environ.get("T_BENCH_CONTAINER_LOGS_PATH") + or "/logs/artifacts" + ) + logs_root.mkdir(parents=True, exist_ok=True) + # test.sh has already saved patch.diff, so the reward still sees the agent's test edits. + try: + test_edits = restore_test_edits(args.base_ref, args.repo_root or detect_repo_root()) + except Exception as e: # noqa: BLE001 + test_edits = {"error": "%s: %s" % (type(e).__name__, e)} + (logs_root / "test_edits.json").write_text(json.dumps(test_edits, sort_keys=True)) + args.restored_tests = test_edits.get("modified", []) + test_edits.get("deleted", []) output = main(args) # Selected tests but nothing collected: retry without extra args, then with importlib mode @@ -941,7 +1063,7 @@ def main(args) -> dict: output["strategy"] = strategy if selected: output["regression"] = run_base_and_diff( - selected, extra, repo_root, output["results"] + selected, extra, repo_root, output["results"], base_ref=args.base_ref ) else: output["regression"] = {"ran": False, "reason": "no tests selected (direct or fallback)"} @@ -949,15 +1071,9 @@ def main(args) -> dict: output["regression"] = {"ran": False, "error": "%s: %s" % (type(e).__name__, e)} output["fc_runner"] = "regression-v1+template" + output["test_edits"] = test_edits - # Must match parser.py's LOG_DIR lookup; /logs/test_results.json is a best-effort legacy copy. - logs_root = Path( - os.environ.get("T_BENCH_TASK_LOGS_PATH") - or os.environ.get("T_BENCH_CONTAINER_LOGS_PATH") - or "/logs/artifacts" - ) _payload = json.dumps(output, sort_keys=True) - logs_root.mkdir(parents=True, exist_ok=True) (logs_root / "test_results.json").write_text(_payload) _legacy = Path("/logs/test_results.json") if _legacy != logs_root / "test_results.json": @@ -966,3 +1082,5 @@ def main(args) -> dict: _legacy.write_text(_payload) except Exception: # noqa: BLE001 -- best-effort; the parser path above is the critical one pass + if output["regression"].get("restore_ok") is False: + sys.exit(3) diff --git a/tests/docker/test_pytest_runner_regression.py b/tests/docker/test_pytest_runner_regression.py index e8c98db3..f30c0912 100644 --- a/tests/docker/test_pytest_runner_regression.py +++ b/tests/docker/test_pytest_runner_regression.py @@ -1,6 +1,8 @@ -"""run_base_and_diff: `ran` only when the base side produced results.""" +"""run_base_and_diff and restore_test_edits: the base side runs at the base commit and the agent cannot weaken tests.""" import importlib.util +import json +import os import subprocess import sys from pathlib import Path @@ -46,7 +48,7 @@ def test_no_base_results_is_not_ran(runner, repo: Path, monkeypatch) -> None: monkeypatch.setattr(runner, "_run", lambda cmd, cwd=None: (1, "", "boom") if cmd[0] == sys.executable else real(cmd, cwd)) out = runner.run_base_and_diff(["test_m.py"], "", str(repo), AGENT) assert out["ran"] is False - assert out["stashed"] is True + assert out["reverted"] is True assert "no results" in out["base_error"] assert "assert False" in (repo / "test_m.py").read_text() @@ -71,3 +73,82 @@ def test_rebuild_runs_at_base_and_again_after_restore(runner, repo: Path, monkey monkeypatch.setenv("FC_REBUILD_CMD", "false") out = runner.run_base_and_diff(["test_m.py"], "", str(repo), AGENT) assert out["ran"] is False and "base rebuild failed" in out["base_error"] + + +def _git(cwd: Path, *args) -> str: + return subprocess.run(["git", "-c", "user.name=t", "-c", "user.email=t@t", *args], cwd=cwd, check=True, capture_output=True, text=True).stdout + + +@pytest.fixture +def src_repo(tmp_path: Path) -> tuple[Path, str]: + root = tmp_path / "repo" + (root / "pkg").mkdir(parents=True) + _git(root, "init", "-q") + (root / "pkg" / "m.py").write_text("def f():\n return 1\n") + (root / "pkg" / "test_m.py").write_text("from m import f\n\ndef test_f():\n assert f() == 1\n") + (root / "old.py").write_text("x = 1\n") + _git(root, "add", "-A") + _git(root, "commit", "-qm", "fc-baseline") + return root, _git(root, "rev-parse", "HEAD").strip() + + +def _run_runner(root: Path, base: str, logs: Path) -> tuple[dict, dict]: + env = dict(os.environ, T_BENCH_TASK_LOGS_PATH=str(logs)) + subprocess.run([sys.executable, str(_RUNNER), "--base", base, "--root", str(root)], cwd=root, env=env, check=True, capture_output=True) + return json.loads((logs / "test_results.json").read_text()), json.loads((logs / "test_edits.json").read_text()) + + +def test_weakened_test_is_restored_and_regression_reported(src_repo, tmp_path: Path) -> None: + root, base = src_repo + (root / "pkg" / "m.py").write_text("def f():\n return 2\n") + (root / "pkg" / "test_m.py").write_text("from m import f\n\ndef test_f():\n assert f() == 2\n") + (root / "pkg" / "test_new.py").write_text("def test_n():\n assert True\n") + results, edits = _run_runner(root, base, tmp_path / "logs") + assert edits["modified"] == ["pkg/test_m.py"] and edits["added"] == ["pkg/test_new.py"] + assert not (root / "pkg" / "test_new.py").exists() + assert results["regression"]["regressed"] == ["pkg/test_m.py::test_f"] + + +def test_committed_agent_change_is_reverted_for_base_run(src_repo, tmp_path: Path) -> None: + root, base = src_repo + (root / "pkg" / "m.py").write_text("def f():\n return 2\n") + _git(root, "commit", "-qam", "agent") + results, _ = _run_runner(root, base, tmp_path / "logs") + assert results["regression"]["ran"] is True + assert results["regression"]["regressed"] == ["pkg/test_m.py::test_f"] + + +def test_agent_tree_is_restored_byte_for_byte(runner, src_repo, monkeypatch) -> None: + root, base = src_repo + (root / "pkg" / "m.py").write_text("def f():\n return 2\n") + _git(root, "commit", "-qam", "agent") + (root / "pkg" / "test_m.py").write_text("from m import f\n\ndef test_f():\n assert f() == 3\n") + (root / "new").mkdir() + (root / "new" / "new.py").write_bytes(b"y = 2\r\n") + (root / "old.py").unlink() + _git(root, "add", "pkg/test_m.py") + + def snapshot(): + files = {p.relative_to(root): p.read_bytes() for p in root.rglob("*") if p.is_file() and not str(p.relative_to(root)).startswith(".") and "__pycache__" not in p.parts} + return files, _git(root, "status", "--porcelain"), _git(root, "rev-parse", "HEAD") + + before = snapshot() + real = runner._run + monkeypatch.setattr(runner, "_run", lambda cmd, cwd=None: (_ for _ in ()).throw(KeyboardInterrupt) if cmd[0] == sys.executable else real(cmd, cwd)) + with pytest.raises(KeyboardInterrupt): + runner.run_base_and_diff(["pkg/test_m.py"], "", str(root), AGENT, base_ref=base) + assert snapshot() == before + monkeypatch.setattr(runner, "_run", real) + assert runner.run_base_and_diff(["pkg/test_m.py"], "", str(root), AGENT, base_ref=base)["ran"] is True + assert snapshot() == before + + +def test_failed_restore_sets_restore_ok_false_and_exits_nonzero(src_repo, tmp_path: Path) -> None: + root, base = src_repo + (root / "pkg" / "m.py").write_text("def f():\n return 2\n") + logs = tmp_path / "logs" + env = dict(os.environ, T_BENCH_TASK_LOGS_PATH=str(logs), FC_REBUILD_CMD="grep -q 'return 1' pkg/m.py") + proc = subprocess.run([sys.executable, str(_RUNNER), "--base", base, "--root", str(root)], cwd=root, env=env, capture_output=True) + regression = json.loads((logs / "test_results.json").read_text())["regression"] + assert proc.returncode == 3 + assert regression["restore_ok"] is False and "patched rebuild after restore failed" in regression["base_error"] diff --git a/tests/docker/test_template_lint.py b/tests/docker/test_template_lint.py index aca8286e..4cf88396 100644 --- a/tests/docker/test_template_lint.py +++ b/tests/docker/test_template_lint.py @@ -62,3 +62,10 @@ def test_no_undefined_names_in_templates(rel: str) -> None: f"These files are excluded from ruff, so nothing else will catch this.\n" f"{proc.stdout}\n{proc.stderr}" ) + + +@pytest.mark.parametrize("path", sorted((_ROOT / "src/datasmith/harbor_adapter/template").rglob("*.py")), ids=lambda p: p.name) +def test_harbor_template_parses_as_python_3_8(path: Path) -> None: + import ast + + ast.parse(path.read_text(), feature_version=(3, 8)) From 39051abd4ed625f631e1536e6184dfd8590adcec Mon Sep 17 00:00:00 2001 From: Arjun Sharma Date: Thu, 8 Oct 2026 01:16:29 +0000 Subject: [PATCH 2/5] template: harden the separate verifier against diffs that hide or run code, and count skipped tests as regressions The verifier's patch.diff uses --text --no-ext-diff --no-textconv with GIT_NO_REPLACE_OBJECTS, so a .gitattributes file in the agent's diff cannot hide edited benchmark lines from the tamper check; the collect command drops external diff drivers too. prepare.sh refuses a tree.diff that creates a symlink or adds *.pth, sitecustomize.py or usercustomize.py. A test that passes at base and is skipped, xfailed or not collected with the agent now counts as a regression. --- .../harbor_adapter/template/shared/collect.sh | 2 +- .../harbor_adapter/template/tests/prepare.sh | 9 +++- .../template/tests/pytest_runner.py | 6 +-- .../harbor_adapter/template/tests/test.sh | 3 +- tests/docker/test_pytest_runner_regression.py | 14 +++++ tests/docker/test_separate_verifier_tree.py | 51 +++++++++++++++++++ 6 files changed, 78 insertions(+), 7 deletions(-) diff --git a/src/datasmith/harbor_adapter/template/shared/collect.sh b/src/datasmith/harbor_adapter/template/shared/collect.sh index 66d43d5e..d594a6c8 100644 --- a/src/datasmith/harbor_adapter/template/shared/collect.sh +++ b/src/datasmith/harbor_adapter/template/shared/collect.sh @@ -4,7 +4,7 @@ mkdir -p /fc_submission export GIT_INDEX_FILE="$(mktemp -d)/index" git read-tree {{ base_commit }} {% include "shared/stage_tree.sh" %} -git diff --cached --binary {{ base_commit }} > /fc_submission/tree.diff +GIT_NO_REPLACE_OBJECTS=1 git diff --cached --binary --no-ext-diff --no-textconv {{ base_commit }} > /fc_submission/tree.diff # Processes the agent left running would compete for CPU with the timing; kill -1 spares PID 1 and this shell. kill -9 -1 2>/dev/null true diff --git a/src/datasmith/harbor_adapter/template/tests/prepare.sh b/src/datasmith/harbor_adapter/template/tests/prepare.sh index 7291c71c..569a3365 100644 --- a/src/datasmith/harbor_adapter/template/tests/prepare.sh +++ b/src/datasmith/harbor_adapter/template/tests/prepare.sh @@ -144,7 +144,14 @@ echo "[$(ts)] [prepare] LSV init complete." SETUP_PHASE="agent_tree" # The agent's tree is the base commit plus tree.diff; ignored files (the build) stay as the image has them. # A missing or broken tree.diff leaves the starting tree, so the trial has no patch (no_patch). -if [ -s /fc_submission/tree.diff ]; then +# Symlinks, *.pth and site/usercustomize.py run code outside the patched sources, so such a diff is refused. +_refused="$( { git apply --summary /fc_submission/tree.diff 2>/dev/null | grep -E ' 120000( |$)' || true; \ + git apply --numstat /fc_submission/tree.diff 2>/dev/null | cut -f3 | grep -E '(^|/)(site|user)customize\.py$|\.pth$' || true; } )" +if [ -n "${_refused}" ]; then + echo "[$(ts)] [prepare] tree.diff refused: ${_refused}" >&2 + echo tree_diff_refused > /logs/artifacts/agent_tree.txt + printf '%s\n' "${_refused}" > /logs/artifacts/tree_diff_refused.txt +elif [ -s /fc_submission/tree.diff ]; then git read-tree -u --reset {{ base_commit }} if ! git apply --binary --whitespace=nowarn /fc_submission/tree.diff; then echo "[$(ts)] [prepare] tree.diff does not apply; using the starting tree." >&2 diff --git a/src/datasmith/harbor_adapter/template/tests/pytest_runner.py b/src/datasmith/harbor_adapter/template/tests/pytest_runner.py index 163efd5b..40ed47da 100644 --- a/src/datasmith/harbor_adapter/template/tests/pytest_runner.py +++ b/src/datasmith/harbor_adapter/template/tests/pytest_runner.py @@ -465,10 +465,8 @@ def run_base_and_diff(selected_tests, extra_args, repo_root, agent_results, base restore_ok = False base_error = (base_error or "") + "; patched rebuild after restore failed" shutil.rmtree(backup, ignore_errors=True) - regressed = sorted( - n for n, o in agent_oc.items() - if o in ("failed", "error") and base_oc.get(n) == "passed" - ) + # Passing at base and not passing with the agent: failed, error, skipped, xfailed, or not collected. + regressed = sorted(n for n, o in base_oc.items() if o == "passed" and agent_oc.get(n) not in ("passed", "xpassed")) out = { "ran": bool(reverted and selected_tests and base_oc), "reverted": reverted, diff --git a/src/datasmith/harbor_adapter/template/tests/test.sh b/src/datasmith/harbor_adapter/template/tests/test.sh index d5e66e62..a677d2ff 100644 --- a/src/datasmith/harbor_adapter/template/tests/test.sh +++ b/src/datasmith/harbor_adapter/template/tests/test.sh @@ -65,7 +65,8 @@ FC_BASE="$(cat /opt/fc_baseline_sha 2>/dev/null || echo {{ base_commit }})" FC_INDEX="$(mktemp)" cp .git/index "${FC_INDEX}" 2>/dev/null || rm -f "${FC_INDEX}" GIT_INDEX_FILE="${FC_INDEX}" git add -A -N 2>/dev/null || true -fc_diff() { GIT_INDEX_FILE="${FC_INDEX}" git diff "${FC_BASE}" "$@" 2>/dev/null; } +# --text and no drivers: a .gitattributes file in the agent's diff must not hide edited lines from the tamper check. +fc_diff() { GIT_INDEX_FILE="${FC_INDEX}" GIT_NO_REPLACE_OBJECTS=1 git diff --text --no-ext-diff --no-textconv "${FC_BASE}" "$@" 2>/dev/null; } fc_diff > "${LOG_DIR}/patch.diff" || true patch_files=$(fc_diff --name-only | wc -l | tr -d ' ') patch_numstat=$(fc_diff --numstat || true) diff --git a/tests/docker/test_pytest_runner_regression.py b/tests/docker/test_pytest_runner_regression.py index 2d6f222e..2ef26761 100644 --- a/tests/docker/test_pytest_runner_regression.py +++ b/tests/docker/test_pytest_runner_regression.py @@ -199,3 +199,17 @@ def test_failed_restore_sets_restore_ok_false_and_exits_nonzero(src_repo, tmp_pa regression = json.loads((logs / "test_results.json").read_text())["regression"] assert proc.returncode == 3 assert regression["restore_ok"] is False and "patched rebuild after restore failed" in regression["base_error"] + + +def test_base_pass_test_skipped_by_agent_code_is_a_regression(src_repo, tmp_path: Path) -> None: + root, base = src_repo + (root / "pkg" / "m.py").write_text("import pytest\n\ndef f():\n pytest.skip('no')\n") + results, _ = _run_runner(root, base, tmp_path / "logs") + assert results["regression"]["regressed"] == ["pkg/test_m.py::test_f"] + + +def test_base_pass_test_not_collected_is_a_regression(src_repo, tmp_path: Path) -> None: + root, base = src_repo + (root / "pkg" / "m.py").write_text("raise ImportError('broken')\n") + results, _ = _run_runner(root, base, tmp_path / "logs") + assert results["regression"]["regressed"] == ["pkg/test_m.py::test_f"] diff --git a/tests/docker/test_separate_verifier_tree.py b/tests/docker/test_separate_verifier_tree.py index a1652914..1a7ee5ad 100644 --- a/tests/docker/test_separate_verifier_tree.py +++ b/tests/docker/test_separate_verifier_tree.py @@ -107,3 +107,54 @@ def test_broken_tree_diff_leaves_starting_tree(tmp_path): bash(prepare_block("agent_tree", "complete", sha, base).replace("/fc_submission", str(sub)), image) assert {k: v for k, v in files(image).items() if k not in ("sha", "agent_tree.txt")} == before assert (image / "agent_tree.txt").read_text().strip() == "tree_diff_failed" + + +def _collect_and_recreate(tmp_path: Path, edit) -> tuple[Path, Path]: + image, base = image_repo(tmp_path) + agent, verifier = tmp_path / "agent", tmp_path / "verifier" + subprocess.run(["cp", "-a", str(image), str(agent)], check=True) + subprocess.run(["cp", "-a", str(image), str(verifier)], check=True) + edit(agent) + sub = tmp_path / "submission" + bash(collect_script(base, sub), agent) + sha = verifier / "sha" + bash(prepare_block("baseline_commit", "base_copy", sha, base), verifier) + bash(prepare_block("agent_tree", "complete", sha, base).replace("/fc_submission", str(sub)), verifier) + return verifier, sha + + +def test_gitattributes_cannot_hide_a_benchmark_edit_from_the_verifier_diff(tmp_path): + def edit(agent: Path) -> None: + (agent / ".gitattributes").write_text("benchmarks/** -diff\n") + (agent / "benchmarks/bench.py").write_text("def time_x(): return 0\n") + + verifier, sha = _collect_and_recreate(tmp_path, edit) + test_sh = render_template( + "tests/test.sh", + base_commit="x", + run_pytest=False, + rounds=None, + task_id="t", + owner="o", + repo="r", + issue_number=1, + ) + fc_diff = next(line for line in test_sh.splitlines() if line.startswith("fc_diff()")) + setup = "\n".join( + line for line in test_sh.splitlines() if line.startswith(("FC_INDEX=", "cp .git/index", "GIT_INDEX_FILE=")) + ) + script = f"FC_BASE={sha.read_text().strip()}\n{setup}\n{fc_diff}\nfc_diff" + out = subprocess.run(["bash", "-c", script], cwd=verifier, check=True, capture_output=True, text=True).stdout + assert "+def time_x(): return 0" in out + + +def test_diff_with_pth_file_or_symlink_is_refused(tmp_path): + def edit(agent: Path) -> None: + (agent / "pkg.py").write_text("x = 2\n") + (agent / "evil.pth").write_text("import os\n") + (agent / "link").symlink_to("/etc/passwd") + + verifier, _ = _collect_and_recreate(tmp_path, edit) + assert (verifier / "agent_tree.txt").read_text().strip() == "tree_diff_refused" + assert not (verifier / "evil.pth").exists() and not (verifier / "link").is_symlink() + assert (verifier / "pkg.py").read_text() == "x = 1 # image edit\n" From 6a74b9f15f6bc919295241e144d93b9736b927d3 Mon Sep 17 00:00:00 2001 From: Arjun Sharma Date: Thu, 8 Oct 2026 01:21:17 +0000 Subject: [PATCH 3/5] template: protect testing helpers and pytest config sections, record refused diffs, parse paths with -z Files under testing/ and _testing/ (numpy.testing, pandas._testing, xarray.testing) count as test code and are restored from base before pytest. The pytest sections of pyproject.toml, setup.cfg and tox.ini are put back to base while other settings stay, so addopts cannot load an agent plugin. parser.py records prepare.sh's tree_diff_refused or tree_diff_failed as patch.submission_rejected. The refusal check reads git apply --numstat -z, so a quoted non-ASCII path is matched. --- .../harbor_adapter/template/tests/parser.py | 17 ++++--- .../harbor_adapter/template/tests/prepare.sh | 3 +- .../template/tests/pytest_runner.py | 49 +++++++++++++++++-- tests/docker/test_pytest_runner_regression.py | 38 ++++++++++++++ tests/docker/test_separate_verifier_tree.py | 11 +++++ 5 files changed, 107 insertions(+), 11 deletions(-) diff --git a/src/datasmith/harbor_adapter/template/tests/parser.py b/src/datasmith/harbor_adapter/template/tests/parser.py index e2170e62..d5f44ca1 100644 --- a/src/datasmith/harbor_adapter/template/tests/parser.py +++ b/src/datasmith/harbor_adapter/template/tests/parser.py @@ -287,12 +287,17 @@ def summarize_snapshots(log_dir: Path) -> dict: def load_patch_info(log_dir: Path) -> dict: path = log_dir / "patch_info.json" - if not path.exists(): - return {"applied": False, "files": 0, "added_lines": 0, "removed_lines": 0} - try: - return json.loads(path.read_text()) - except (json.JSONDecodeError, OSError): - return {"applied": False, "files": 0, "added_lines": 0, "removed_lines": 0} + info = {"applied": False, "files": 0, "added_lines": 0, "removed_lines": 0} + if path.exists(): + try: + info = json.loads(path.read_text()) + except (json.JSONDecodeError, OSError): + pass + # prepare.sh writes tree_diff_refused or tree_diff_failed; the host fails such a trial instead of an empty patch. + tree = log_dir / "agent_tree.txt" + if tree.exists(): + info["submission_rejected"] = tree.read_text().strip() + return info def load_setup_status(log_dir: Path) -> dict: diff --git a/src/datasmith/harbor_adapter/template/tests/prepare.sh b/src/datasmith/harbor_adapter/template/tests/prepare.sh index 569a3365..1d40089d 100644 --- a/src/datasmith/harbor_adapter/template/tests/prepare.sh +++ b/src/datasmith/harbor_adapter/template/tests/prepare.sh @@ -146,7 +146,8 @@ SETUP_PHASE="agent_tree" # A missing or broken tree.diff leaves the starting tree, so the trial has no patch (no_patch). # Symlinks, *.pth and site/usercustomize.py run code outside the patched sources, so such a diff is refused. _refused="$( { git apply --summary /fc_submission/tree.diff 2>/dev/null | grep -E ' 120000( |$)' || true; \ - git apply --numstat /fc_submission/tree.diff 2>/dev/null | cut -f3 | grep -E '(^|/)(site|user)customize\.py$|\.pth$' || true; } )" + git apply --numstat -z /fc_submission/tree.diff 2>/dev/null | tr '\0' '\n' | sed -E 's/^[0-9-]+\t[0-9-]+\t//' \ + | grep -E '(^|/)(site|user)customize\.py$|\.pth$' || true; } )" if [ -n "${_refused}" ]; then echo "[$(ts)] [prepare] tree.diff refused: ${_refused}" >&2 echo tree_diff_refused > /logs/artifacts/agent_tree.txt diff --git a/src/datasmith/harbor_adapter/template/tests/pytest_runner.py b/src/datasmith/harbor_adapter/template/tests/pytest_runner.py index 40ed47da..0a9a9f06 100644 --- a/src/datasmith/harbor_adapter/template/tests/pytest_runner.py +++ b/src/datasmith/harbor_adapter/template/tests/pytest_runner.py @@ -349,18 +349,59 @@ def _fingerprint(paths, repo_root): return fp +# Assertion helpers in package code (numpy/testing, pandas/_testing, xarray/testing) judge the tests too. +_TEST_HELPER_DIRS = ("testing", "_testing") + + def is_test_path(path): - """Test code: test_*.py, *_test.py, conftest.py, pytest.ini, or any file under a tests/test/_tests directory.""" + """Test code: test_*.py, *_test.py, conftest.py, pytest.ini, or a file under a tests or testing directory.""" parts = path.split("/") - return is_test_file(path) or parts[-1] in ("conftest.py", "pytest.ini") or any(p in _TEST_DIRS for p in parts[:-1]) + dirs = _TEST_DIRS + _TEST_HELPER_DIRS + return is_test_file(path) or parts[-1] in ("conftest.py", "pytest.ini") or any(p in dirs for p in parts[:-1]) + + +_PYTEST_SECTIONS = {"pyproject.toml": r"tool\.pytest(\..+)?", "setup.cfg": r"tool:pytest", "tox.ini": r"pytest"} +_SECTION_HEADER = re.compile(r"^\s*\[\[?\s*([^\]]+?)\s*\]\]?\s*(#.*)?$") + + +def _sections(text): + """[(section name or None, text)] split at each [section] header line.""" + blocks = [[None, []]] + for line in text.splitlines(True): + m = _SECTION_HEADER.match(line) + if m: + blocks.append([m.group(1), []]) + blocks[-1][1].append(line) + return [(name, "".join(lines)) for name, lines in blocks] + + +def restore_pytest_sections(base_text, agent_text, pattern): + """``agent_text`` with its pytest sections replaced by those of ``base_text``; None when they already match.""" + rx = re.compile(pattern + "$") + pick = lambda text, keep: [b for n, b in _sections(text) if bool(n and rx.match(n)) == keep] # noqa: E731 + base_pytest, agent_pytest = "".join(pick(base_text, True)), "".join(pick(agent_text, True)) + if base_pytest.strip() == agent_pytest.strip(): + return None + rest = "".join(pick(agent_text, False)) + return rest.rstrip("\n") + ("\n\n" + base_pytest if base_pytest else "\n") def restore_test_edits(base_ref, repo_root): """Put the test files back to ``base_ref`` so the agent cannot weaken the tests that judge it.""" - edits = {"modified": [], "added": [], "deleted": [], "config_changed": []} + edits = {"modified": [], "added": [], "deleted": [], "config_changed": [], "config_restored": []} for path, status in sorted(_changed_vs(base_ref, repo_root).items()): - if os.path.basename(path) in ("pyproject.toml", "setup.cfg", "tox.ini"): + name = os.path.basename(path) + if name in _PYTEST_SECTIONS: edits["config_changed"].append(path) + full = os.path.join(repo_root, path) + code, base_text, _ = _run(["git", "show", "%s:%s" % (base_ref, path)], cwd=repo_root) + if status != "D" and os.path.isfile(full): + with open(full) as fh: + fixed = restore_pytest_sections(base_text if code == 0 else "", fh.read(), _PYTEST_SECTIONS[name]) + if fixed is not None: + with open(full, "w") as fh: + fh.write(fixed) + edits["config_restored"].append(path) elif is_test_path(path): edits[{"A": "added", "D": "deleted"}.get(status, "modified")].append(path) _checkout(base_ref, edits["modified"] + edits["deleted"], repo_root) diff --git a/tests/docker/test_pytest_runner_regression.py b/tests/docker/test_pytest_runner_regression.py index 2ef26761..50cc4000 100644 --- a/tests/docker/test_pytest_runner_regression.py +++ b/tests/docker/test_pytest_runner_regression.py @@ -213,3 +213,41 @@ def test_base_pass_test_not_collected_is_a_regression(src_repo, tmp_path: Path) (root / "pkg" / "m.py").write_text("raise ImportError('broken')\n") results, _ = _run_runner(root, base, tmp_path / "logs") assert results["regression"]["regressed"] == ["pkg/test_m.py::test_f"] + + +def test_weakened_testing_helper_is_restored(src_repo, tmp_path: Path) -> None: + root, _ = src_repo + (root / "pkg" / "testing").mkdir() + (root / "pkg" / "testing" / "__init__.py").write_text("") + (root / "pkg" / "testing" / "helpers.py").write_text("def check(x):\n assert x == 1\n") + (root / "pkg" / "test_m.py").write_text( + "from m import f\nfrom testing.helpers import check\n\ndef test_f():\n check(f())\n" + ) + _git(root, "add", "-A") + _git(root, "commit", "-qm", "helpers") + base = _git(root, "rev-parse", "HEAD").strip() + (root / "pkg" / "m.py").write_text("def f():\n return 2\n") + (root / "pkg" / "testing" / "helpers.py").write_text("def check(x):\n pass\n") + results, edits = _run_runner(root, base, tmp_path / "logs") + assert edits["modified"] == ["pkg/testing/helpers.py"] + assert results["regression"]["regressed"] == ["pkg/test_m.py::test_f"] + + +def test_pytest_sections_are_restored_and_build_settings_kept(runner) -> None: + base = '[project]\nname = "p"\n\n[tool.pytest.ini_options]\naddopts = "-ra"\n' + agent = '[project]\nname = "p"\nversion = "2"\n\n[tool.pytest.ini_options]\naddopts = "-p pkg._plugin"\n' + fixed = runner.restore_pytest_sections(base, agent, runner._PYTEST_SECTIONS["pyproject.toml"]) + assert 'version = "2"' in fixed and 'addopts = "-ra"' in fixed and "pkg._plugin" not in fixed + assert runner.restore_pytest_sections(base, base, runner._PYTEST_SECTIONS["pyproject.toml"]) is None + added = runner.restore_pytest_sections( + "[metadata]\nname = p\n", "[metadata]\nname = p\n\n[tool:pytest]\naddopts = -p x\n", "tool:pytest" + ) + assert "tool:pytest" not in added and "name = p" in added + + +def test_pyproject_addopts_plugin_is_restored_before_the_run(src_repo, tmp_path: Path) -> None: + root, base = src_repo + (root / "pyproject.toml").write_text('[tool.pytest.ini_options]\naddopts = "-p no_such_agent_plugin"\n') + _, edits = _run_runner(root, base, tmp_path / "logs") + assert edits["config_restored"] == ["pyproject.toml"] + assert "no_such_agent_plugin" not in (root / "pyproject.toml").read_text() diff --git a/tests/docker/test_separate_verifier_tree.py b/tests/docker/test_separate_verifier_tree.py index 1a7ee5ad..94393061 100644 --- a/tests/docker/test_separate_verifier_tree.py +++ b/tests/docker/test_separate_verifier_tree.py @@ -158,3 +158,14 @@ def edit(agent: Path) -> None: assert (verifier / "agent_tree.txt").read_text().strip() == "tree_diff_refused" assert not (verifier / "evil.pth").exists() and not (verifier / "link").is_symlink() assert (verifier / "pkg.py").read_text() == "x = 1 # image edit\n" + + +def test_quoted_path_sitecustomize_is_refused(tmp_path): + def edit(agent: Path) -> None: + (agent / "pkg.py").write_text("x = 2\n") + (agent / "é").mkdir() + (agent / "é" / "sitecustomize.py").write_text("import os\n") + + verifier, _ = _collect_and_recreate(tmp_path, edit) + assert (verifier / "agent_tree.txt").read_text().strip() == "tree_diff_refused" + assert not (verifier / "é").exists() From 185e0ded4a1a964f916f9fca890d800033d87c82 Mon Sep 17 00:00:00 2001 From: Arjun Sharma Date: Thu, 8 Oct 2026 01:27:02 +0000 Subject: [PATCH 4/5] template: pytest reads the base commit's config through -c, and a plugin only the agent run loads is a regression Both pytest runs get -c with a copy of the base commit's pytest config file (pytest.ini, .pytest.ini, pyproject.toml, tox.ini or setup.cfg, in pytest's order) and --rootdir at the repo, so the agent's config files, dotted TOML keys and an added .pytest.ini are never read. The section editing of the previous commit is removed. Each run records its loaded plugins; one that only the agent run loads (a pytest11 entry point from the agent's package) adds plugin: to regressed. prepare.sh also refuses dist-info and egg-info files in tree.diff. --- .../harbor_adapter/template/tests/prepare.sh | 4 +- .../template/tests/pytest_runner.py | 68 ++++++++++++------- tests/docker/test_pytest_runner_regression.py | 51 ++++++++++---- tests/docker/test_separate_verifier_tree.py | 9 +++ 4 files changed, 92 insertions(+), 40 deletions(-) diff --git a/src/datasmith/harbor_adapter/template/tests/prepare.sh b/src/datasmith/harbor_adapter/template/tests/prepare.sh index 1d40089d..ec963d20 100644 --- a/src/datasmith/harbor_adapter/template/tests/prepare.sh +++ b/src/datasmith/harbor_adapter/template/tests/prepare.sh @@ -144,10 +144,10 @@ echo "[$(ts)] [prepare] LSV init complete." SETUP_PHASE="agent_tree" # The agent's tree is the base commit plus tree.diff; ignored files (the build) stay as the image has them. # A missing or broken tree.diff leaves the starting tree, so the trial has no patch (no_patch). -# Symlinks, *.pth and site/usercustomize.py run code outside the patched sources, so such a diff is refused. +# Symlinks, *.pth, site/usercustomize.py and package metadata (entry points) run code outside the patched sources. _refused="$( { git apply --summary /fc_submission/tree.diff 2>/dev/null | grep -E ' 120000( |$)' || true; \ git apply --numstat -z /fc_submission/tree.diff 2>/dev/null | tr '\0' '\n' | sed -E 's/^[0-9-]+\t[0-9-]+\t//' \ - | grep -E '(^|/)(site|user)customize\.py$|\.pth$' || true; } )" + | grep -E '(^|/)(site|user)customize\.py$|\.pth$|\.(dist|egg)-info/' || true; } )" if [ -n "${_refused}" ]; then echo "[$(ts)] [prepare] tree.diff refused: ${_refused}" >&2 echo tree_diff_refused > /logs/artifacts/agent_tree.txt diff --git a/src/datasmith/harbor_adapter/template/tests/pytest_runner.py b/src/datasmith/harbor_adapter/template/tests/pytest_runner.py index 0a9a9f06..0ae484d3 100644 --- a/src/datasmith/harbor_adapter/template/tests/pytest_runner.py +++ b/src/datasmith/harbor_adapter/template/tests/pytest_runner.py @@ -10,6 +10,7 @@ import shlex import subprocess import sys +import tempfile import time from glob import glob from pathlib import Path @@ -375,33 +376,35 @@ def _sections(text): return [(name, "".join(lines)) for name, lines in blocks] -def restore_pytest_sections(base_text, agent_text, pattern): - """``agent_text`` with its pytest sections replaced by those of ``base_text``; None when they already match.""" - rx = re.compile(pattern + "$") - pick = lambda text, keep: [b for n, b in _sections(text) if bool(n and rx.match(n)) == keep] # noqa: E731 - base_pytest, agent_pytest = "".join(pick(base_text, True)), "".join(pick(agent_text, True)) - if base_pytest.strip() == agent_pytest.strip(): - return None - rest = "".join(pick(agent_text, False)) - return rest.rstrip("\n") + ("\n\n" + base_pytest if base_pytest else "\n") +_PYTEST_CONFIG_FILES = ("pytest.ini", ".pytest.ini", "pyproject.toml", "tox.ini", "setup.cfg") + + +def write_base_pytest_config(base_ref, repo_root): + """A copy of the base commit's pytest config file, picked in pytest's order, for ``-c``; an empty one if none.""" + out = tempfile.mkdtemp(prefix="fc_pytest_cfg_") + for name in _PYTEST_CONFIG_FILES: + code, text, _ = _run(["git", "show", "%s:%s" % (base_ref, name)], cwd=repo_root) + if code != 0: + continue + if name in _PYTEST_SECTIONS and not any(n and re.match(_PYTEST_SECTIONS[name] + "$", n) for n, _ in _sections(text)): + continue + path = os.path.join(out, name) + with open(path, "w") as fh: + fh.write(text) + return path + path = os.path.join(out, "pytest.ini") + with open(path, "w") as fh: + fh.write("[pytest]\n") + return path def restore_test_edits(base_ref, repo_root): """Put the test files back to ``base_ref`` so the agent cannot weaken the tests that judge it.""" - edits = {"modified": [], "added": [], "deleted": [], "config_changed": [], "config_restored": []} + edits = {"modified": [], "added": [], "deleted": [], "config_changed": []} for path, status in sorted(_changed_vs(base_ref, repo_root).items()): - name = os.path.basename(path) - if name in _PYTEST_SECTIONS: + # pytest reads the base config through -c (write_base_pytest_config), so config edits are only recorded. + if os.path.basename(path) in _PYTEST_CONFIG_FILES: edits["config_changed"].append(path) - full = os.path.join(repo_root, path) - code, base_text, _ = _run(["git", "show", "%s:%s" % (base_ref, path)], cwd=repo_root) - if status != "D" and os.path.isfile(full): - with open(full) as fh: - fixed = restore_pytest_sections(base_text if code == 0 else "", fh.read(), _PYTEST_SECTIONS[name]) - if fixed is not None: - with open(full, "w") as fh: - fh.write(fixed) - edits["config_restored"].append(path) elif is_test_path(path): edits[{"A": "added", "D": "deleted"}.get(status, "modified")].append(path) _checkout(base_ref, edits["modified"] + edits["deleted"], repo_root) @@ -434,6 +437,7 @@ def run_base_and_diff(selected_tests, extra_args, repo_root, agent_results, base reverted = bool(changed) restore_ok = True base_oc = {} + base_plugins = None base_error = None # test.sh sets this when the patch touches compiled sources; base must not import the patched .so. rebuild = os.environ.get("FC_REBUILD_CMD") @@ -456,7 +460,7 @@ def run_base_and_diff(selected_tests, extra_args, repo_root, agent_results, base "spec = importlib.util.spec_from_file_location('fc_pytest_runner', %r); " "P = importlib.util.module_from_spec(spec); spec.loader.exec_module(P); " "r = P.run_pytest_and_collect(%r, extra_args=%r, cwd=%r); " - "open(%r, 'w').write(json.dumps(r.get('tests', [])))" + "open(%r, 'w').write(json.dumps({'tests': r.get('tests', []), 'plugins': r.get('plugins', [])}))" ) % (os.path.dirname(self_path), self_path, list(selected_tests), extra_args, repo_root, tf) # Own pycache: a base file with the agent file's size and mtime would load the agent bytecode. _old_pyc = os.environ.get("PYTHONPYCACHEPREFIX") @@ -470,7 +474,9 @@ def run_base_and_diff(selected_tests, extra_args, repo_root, agent_results, base os.environ["PYTHONPYCACHEPREFIX"] = _old_pyc try: with open(tf) as _fh: - base_oc = _final_outcomes({"tests": json.load(_fh)}) + base_res = json.load(_fh) + base_oc = _final_outcomes(base_res) + base_plugins = base_res.get("plugins", []) except Exception: # noqa: BLE001 — treat an unreadable base run as "no base data" base_oc = {} base_error = "base run produced no results (rc=%s): %s" % ( @@ -508,11 +514,16 @@ def run_base_and_diff(selected_tests, extra_args, repo_root, agent_results, base shutil.rmtree(backup, ignore_errors=True) # Passing at base and not passing with the agent: failed, error, skipped, xfailed, or not collected. regressed = sorted(n for n, o in base_oc.items() if o == "passed" and agent_oc.get(n) not in ("passed", "xpassed")) + # A pytest plugin that only the agent run loads (an entry point from the agent's package) can rewrite reports. + agent_only_plugins = sorted(set((agent_results or {}).get("plugins", [])) - set(base_plugins or [])) + if base_plugins is not None: + regressed += ["plugin:%s" % n for n in agent_only_plugins] out = { "ran": bool(reverted and selected_tests and base_oc), "reverted": reverted, "restore_ok": restore_ok, "base_ref": base_ref, + "agent_only_plugins": agent_only_plugins, "n_selected_tests": len(selected_tests), "n_agent_tests": len(agent_oc), "n_base_pass": sum(1 for o in base_oc.values() if o == "passed"), @@ -622,6 +633,10 @@ def pytest_runtest_logreport(self, report): def pytest_sessionfinish(self, session, exitstatus): self.results["exit_code"] = int(exitstatus) self.results["duration"] = time.time() - self.results["start_time"] + config = session.config + self.results["config_file"] = str(getattr(config, "inipath", None) or getattr(config, "inifile", None) or "") + # Anonymous plugins are named by id(); the rest (entry points, -p modules, conftests) must match the base run. + self.results["plugins"] = sorted(n for n, _ in config.pluginmanager.list_name_plugin() if n and not n.isdigit()) def pytest_collectreport(self, report): if report.failed: @@ -712,6 +727,9 @@ def run_pytest_and_collect(test_paths, extra_args=None, cwd=None): sys.path.insert(0, repo_dir) args = unique_paths + list(extra_args) + # Both runs use the base commit's pytest configuration; the agent's config files are never read. + if os.environ.get("FC_PYTEST_CONFIG"): + args = ["-c", os.environ["FC_PYTEST_CONFIG"], "--rootdir", repo_dir] + args exit_code = pytest.main(args=args, plugins=[plugin]) plugin.results["exit_code"] = int(exit_code) plugin.results["selected_paths"] = unique_paths @@ -1076,6 +1094,10 @@ def main(args) -> dict: test_edits = {"error": "%s: %s" % (type(e).__name__, e)} (logs_root / "test_edits.json").write_text(json.dumps(test_edits, sort_keys=True)) args.restored_tests = test_edits.get("modified", []) + test_edits.get("deleted", []) + try: + os.environ["FC_PYTEST_CONFIG"] = write_base_pytest_config(args.base_ref, args.repo_root or detect_repo_root()) + except Exception as e: # noqa: BLE001 + test_edits["config_error"] = "%s: %s" % (type(e).__name__, e) output = main(args) # Selected tests but nothing collected: retry without extra args, then with importlib mode diff --git a/tests/docker/test_pytest_runner_regression.py b/tests/docker/test_pytest_runner_regression.py index 50cc4000..5673814f 100644 --- a/tests/docker/test_pytest_runner_regression.py +++ b/tests/docker/test_pytest_runner_regression.py @@ -233,21 +233,42 @@ def test_weakened_testing_helper_is_restored(src_repo, tmp_path: Path) -> None: assert results["regression"]["regressed"] == ["pkg/test_m.py::test_f"] -def test_pytest_sections_are_restored_and_build_settings_kept(runner) -> None: - base = '[project]\nname = "p"\n\n[tool.pytest.ini_options]\naddopts = "-ra"\n' - agent = '[project]\nname = "p"\nversion = "2"\n\n[tool.pytest.ini_options]\naddopts = "-p pkg._plugin"\n' - fixed = runner.restore_pytest_sections(base, agent, runner._PYTEST_SECTIONS["pyproject.toml"]) - assert 'version = "2"' in fixed and 'addopts = "-ra"' in fixed and "pkg._plugin" not in fixed - assert runner.restore_pytest_sections(base, base, runner._PYTEST_SECTIONS["pyproject.toml"]) is None - added = runner.restore_pytest_sections( - "[metadata]\nname = p\n", "[metadata]\nname = p\n\n[tool:pytest]\naddopts = -p x\n", "tool:pytest" - ) - assert "tool:pytest" not in added and "name = p" in added +def _base_config_in_setup_cfg(root: Path) -> str: + (root / "setup.cfg").write_text("[tool:pytest]\naddopts = -ra\n") + _git(root, "add", "-A") + _git(root, "commit", "-qm", "config") + return _git(root, "rev-parse", "HEAD").strip() + + +def test_agent_pyproject_dotted_key_config_is_not_read(src_repo, tmp_path: Path) -> None: + root, _ = src_repo + base = _base_config_in_setup_cfg(root) + (root / "pyproject.toml").write_text('[tool]\npytest.ini_options.addopts = "--deselect pkg/test_m.py::test_f"\n') + (root / "pkg" / "m.py").write_text("def f():\n return 2\n") + results, _ = _run_runner(root, base, tmp_path / "logs") + cfg = results["results"]["config_file"] + assert cfg.endswith("setup.cfg") and "fc_pytest_cfg_" in cfg + assert results["regression"]["regressed"] == ["pkg/test_m.py::test_f"] -def test_pyproject_addopts_plugin_is_restored_before_the_run(src_repo, tmp_path: Path) -> None: +def test_added_dot_pytest_ini_is_not_read(src_repo, tmp_path: Path) -> None: + root, _ = src_repo + base = _base_config_in_setup_cfg(root) + (root / ".pytest.ini").write_text("[pytest]\naddopts = --deselect pkg/test_m.py::test_f\n") + (root / "pkg" / "m.py").write_text("def f():\n return 2\n") + results, _ = _run_runner(root, base, tmp_path / "logs") + assert "fc_pytest_cfg_" in results["results"]["config_file"] + assert results["regression"]["regressed"] == ["pkg/test_m.py::test_f"] + + +def test_plugin_loaded_only_in_the_agent_run_is_a_regression(src_repo, tmp_path: Path) -> None: root, base = src_repo - (root / "pyproject.toml").write_text('[tool.pytest.ini_options]\naddopts = "-p no_such_agent_plugin"\n') - _, edits = _run_runner(root, base, tmp_path / "logs") - assert edits["config_restored"] == ["pyproject.toml"] - assert "no_such_agent_plugin" not in (root / "pyproject.toml").read_text() + dist = root / "fcevil-0.dist-info" + dist.mkdir() + (dist / "METADATA").write_text("Metadata-Version: 2.1\nName: fcevil\nVersion: 0\n") + (dist / "entry_points.txt").write_text("[pytest11]\nfcevil = fcevil_mod\n") + (root / "fcevil_mod.py").write_text("") + (root / "pkg" / "m.py").write_text("def f():\n return 1 # agent\n") + results, _ = _run_runner(root, base, tmp_path / "logs") + assert results["regression"]["agent_only_plugins"] == ["fcevil"] + assert "plugin:fcevil" in results["regression"]["regressed"] diff --git a/tests/docker/test_separate_verifier_tree.py b/tests/docker/test_separate_verifier_tree.py index 94393061..54f31bb9 100644 --- a/tests/docker/test_separate_verifier_tree.py +++ b/tests/docker/test_separate_verifier_tree.py @@ -169,3 +169,12 @@ def edit(agent: Path) -> None: verifier, _ = _collect_and_recreate(tmp_path, edit) assert (verifier / "agent_tree.txt").read_text().strip() == "tree_diff_refused" assert not (verifier / "é").exists() + + +def test_diff_with_package_metadata_is_refused(tmp_path): + def edit(agent: Path) -> None: + (agent / "evil-0.dist-info").mkdir() + (agent / "evil-0.dist-info" / "entry_points.txt").write_text("[pytest11]\nevil = evil\n") + + verifier, _ = _collect_and_recreate(tmp_path, edit) + assert (verifier / "agent_tree.txt").read_text().strip() == "tree_diff_refused" From 5f318560a4efce9eb9fbc6210b7a06b87b3fd461 Mon Sep 17 00:00:00 2001 From: Arjun Sharma Date: Thu, 8 Oct 2026 01:29:29 +0000 Subject: [PATCH 5/5] template: the pytest check fails when the base config cannot be written Without the base config pytest may read the agent's config files, so a config_error adds config:unavailable to regressed. --- .../harbor_adapter/template/tests/pytest_runner.py | 9 +++++++++ tests/docker/test_pytest_runner_regression.py | 9 +++++++++ 2 files changed, 18 insertions(+) diff --git a/src/datasmith/harbor_adapter/template/tests/pytest_runner.py b/src/datasmith/harbor_adapter/template/tests/pytest_runner.py index 0ae484d3..75df82c8 100644 --- a/src/datasmith/harbor_adapter/template/tests/pytest_runner.py +++ b/src/datasmith/harbor_adapter/template/tests/pytest_runner.py @@ -398,6 +398,14 @@ def write_base_pytest_config(base_ref, repo_root): return path +def fail_without_base_config(output, test_edits): + """Without the base config pytest may have read the agent's config files, so the check fails.""" + if "config_error" in test_edits: + reg = dict(output.get("regression") or {}, ran=True) + reg["regressed"] = list(reg.get("regressed") or []) + ["config:unavailable"] + output["regression"] = reg + + def restore_test_edits(base_ref, repo_root): """Put the test files back to ``base_ref`` so the agent cannot weaken the tests that judge it.""" edits = {"modified": [], "added": [], "deleted": [], "config_changed": []} @@ -1139,6 +1147,7 @@ def main(args) -> dict: output["fc_runner"] = "regression-v1+template" output["test_edits"] = test_edits + fail_without_base_config(output, test_edits) _payload = json.dumps(output, sort_keys=True) (logs_root / "test_results.json").write_text(_payload) _legacy = Path("/logs/test_results.json") diff --git a/tests/docker/test_pytest_runner_regression.py b/tests/docker/test_pytest_runner_regression.py index 5673814f..1966ed38 100644 --- a/tests/docker/test_pytest_runner_regression.py +++ b/tests/docker/test_pytest_runner_regression.py @@ -272,3 +272,12 @@ def test_plugin_loaded_only_in_the_agent_run_is_a_regression(src_repo, tmp_path: results, _ = _run_runner(root, base, tmp_path / "logs") assert results["regression"]["agent_only_plugins"] == ["fcevil"] assert "plugin:fcevil" in results["regression"]["regressed"] + + +def test_missing_base_config_fails_the_check(runner) -> None: + out = {"regression": {"ran": False, "reason": "no tests selected"}} + runner.fail_without_base_config(out, {"config_error": "OSError: x"}) + assert out["regression"]["ran"] is True and out["regression"]["regressed"] == ["config:unavailable"] + ok = {"regression": {"ran": True, "regressed": []}} + runner.fail_without_base_config(ok, {}) + assert ok["regression"]["regressed"] == []