diff --git a/src/datasmith/harbor_adapter/template/shared/collect.sh b/src/datasmith/harbor_adapter/template/shared/collect.sh index 66d43d5e..d594a6c8 100644 --- a/src/datasmith/harbor_adapter/template/shared/collect.sh +++ b/src/datasmith/harbor_adapter/template/shared/collect.sh @@ -4,7 +4,7 @@ mkdir -p /fc_submission export GIT_INDEX_FILE="$(mktemp -d)/index" git read-tree {{ base_commit }} {% include "shared/stage_tree.sh" %} -git diff --cached --binary {{ base_commit }} > /fc_submission/tree.diff +GIT_NO_REPLACE_OBJECTS=1 git diff --cached --binary --no-ext-diff --no-textconv {{ base_commit }} > /fc_submission/tree.diff # Processes the agent left running would compete for CPU with the timing; kill -1 spares PID 1 and this shell. kill -9 -1 2>/dev/null true diff --git a/src/datasmith/harbor_adapter/template/tests/parser.py b/src/datasmith/harbor_adapter/template/tests/parser.py index e2170e62..d5f44ca1 100644 --- a/src/datasmith/harbor_adapter/template/tests/parser.py +++ b/src/datasmith/harbor_adapter/template/tests/parser.py @@ -287,12 +287,17 @@ def summarize_snapshots(log_dir: Path) -> dict: def load_patch_info(log_dir: Path) -> dict: path = log_dir / "patch_info.json" - if not path.exists(): - return {"applied": False, "files": 0, "added_lines": 0, "removed_lines": 0} - try: - return json.loads(path.read_text()) - except (json.JSONDecodeError, OSError): - return {"applied": False, "files": 0, "added_lines": 0, "removed_lines": 0} + info = {"applied": False, "files": 0, "added_lines": 0, "removed_lines": 0} + if path.exists(): + try: + info = json.loads(path.read_text()) + except (json.JSONDecodeError, OSError): + pass + # prepare.sh writes tree_diff_refused or tree_diff_failed; the host fails such a trial instead of an empty patch. + tree = log_dir / "agent_tree.txt" + if tree.exists(): + info["submission_rejected"] = tree.read_text().strip() + return info def load_setup_status(log_dir: Path) -> dict: diff --git a/src/datasmith/harbor_adapter/template/tests/prepare.sh b/src/datasmith/harbor_adapter/template/tests/prepare.sh index 7291c71c..ec963d20 100644 --- a/src/datasmith/harbor_adapter/template/tests/prepare.sh +++ b/src/datasmith/harbor_adapter/template/tests/prepare.sh @@ -144,7 +144,15 @@ echo "[$(ts)] [prepare] LSV init complete." SETUP_PHASE="agent_tree" # The agent's tree is the base commit plus tree.diff; ignored files (the build) stay as the image has them. # A missing or broken tree.diff leaves the starting tree, so the trial has no patch (no_patch). -if [ -s /fc_submission/tree.diff ]; then +# Symlinks, *.pth, site/usercustomize.py and package metadata (entry points) run code outside the patched sources. +_refused="$( { git apply --summary /fc_submission/tree.diff 2>/dev/null | grep -E ' 120000( |$)' || true; \ + git apply --numstat -z /fc_submission/tree.diff 2>/dev/null | tr '\0' '\n' | sed -E 's/^[0-9-]+\t[0-9-]+\t//' \ + | grep -E '(^|/)(site|user)customize\.py$|\.pth$|\.(dist|egg)-info/' || true; } )" +if [ -n "${_refused}" ]; then + echo "[$(ts)] [prepare] tree.diff refused: ${_refused}" >&2 + echo tree_diff_refused > /logs/artifacts/agent_tree.txt + printf '%s\n' "${_refused}" > /logs/artifacts/tree_diff_refused.txt +elif [ -s /fc_submission/tree.diff ]; then git read-tree -u --reset {{ base_commit }} if ! git apply --binary --whitespace=nowarn /fc_submission/tree.diff; then echo "[$(ts)] [prepare] tree.diff does not apply; using the starting tree." >&2 diff --git a/src/datasmith/harbor_adapter/template/tests/pytest_runner.py b/src/datasmith/harbor_adapter/template/tests/pytest_runner.py index 2af9d8be..75df82c8 100644 --- a/src/datasmith/harbor_adapter/template/tests/pytest_runner.py +++ b/src/datasmith/harbor_adapter/template/tests/pytest_runner.py @@ -3,12 +3,14 @@ from __future__ import print_function import argparse +import hashlib import json import os import re import shlex import subprocess import sys +import tempfile import time from glob import glob from pathlib import Path @@ -197,6 +199,8 @@ def discover_test_files_from_changed(changed_files, repo_root): continue if is_test_file(rel): + if not os.path.isfile(abs_path): + continue mapping.setdefault(rel, []) mapping[rel].append(rel) selected.add(rel) @@ -303,23 +307,154 @@ class _BaseRebuildFailed(Exception): pass -def run_base_and_diff(selected_tests, extra_args, repo_root, agent_results): - """Stash the agent's edits, rerun the same tests at base, restore; flag base-pass -> agent-fail.""" +def _changed_vs(base_ref, repo_root): + """{path: git status letter} for the working tree against ``base_ref``; untracked files are 'A'.""" + code, out, err = _run(["git", "diff", "--name-status", "--no-renames", "-z", base_ref], cwd=repo_root) + if code != 0: + raise RuntimeError("git diff %s failed: %s" % (base_ref, err.strip()[-200:])) + fields = out.split("\0") + changed = {fields[i + 1]: fields[i][:1] for i in range(0, len(fields) - 1, 2) if fields[i]} + code, out, _ = _run(["git", "ls-files", "-z", "--others", "--exclude-standard"], cwd=repo_root) + changed.update((p, "A") for p in out.split("\0") if p and code == 0) + return changed + + +def _checkout(base_ref, paths, repo_root): + if paths: + code, _, err = _run(["git", "--literal-pathspecs", "checkout", base_ref, "--"] + list(paths), cwd=repo_root) + if code != 0: + raise RuntimeError("git checkout %s failed: %s" % (base_ref, err.strip()[-200:])) + + +def _remove(path): + if os.path.lexists(path): + os.remove(path) + try: + os.removedirs(os.path.dirname(path)) + except OSError: + pass + + +def _fingerprint(paths, repo_root): + """{path: sha256 of the file bytes or symlink target, None if missing}.""" + fp = {} + for path in paths: + full = os.path.join(repo_root, path) + if os.path.islink(full): + fp[path] = hashlib.sha256(os.readlink(full).encode()).hexdigest() + elif os.path.isfile(full): + with open(full, "rb") as fh: + fp[path] = hashlib.sha256(fh.read()).hexdigest() + else: + fp[path] = None + return fp + + +# Assertion helpers in package code (numpy/testing, pandas/_testing, xarray/testing) judge the tests too. +_TEST_HELPER_DIRS = ("testing", "_testing") + + +def is_test_path(path): + """Test code: test_*.py, *_test.py, conftest.py, pytest.ini, or a file under a tests or testing directory.""" + parts = path.split("/") + dirs = _TEST_DIRS + _TEST_HELPER_DIRS + return is_test_file(path) or parts[-1] in ("conftest.py", "pytest.ini") or any(p in dirs for p in parts[:-1]) + + +_PYTEST_SECTIONS = {"pyproject.toml": r"tool\.pytest(\..+)?", "setup.cfg": r"tool:pytest", "tox.ini": r"pytest"} +_SECTION_HEADER = re.compile(r"^\s*\[\[?\s*([^\]]+?)\s*\]\]?\s*(#.*)?$") + + +def _sections(text): + """[(section name or None, text)] split at each [section] header line.""" + blocks = [[None, []]] + for line in text.splitlines(True): + m = _SECTION_HEADER.match(line) + if m: + blocks.append([m.group(1), []]) + blocks[-1][1].append(line) + return [(name, "".join(lines)) for name, lines in blocks] + + +_PYTEST_CONFIG_FILES = ("pytest.ini", ".pytest.ini", "pyproject.toml", "tox.ini", "setup.cfg") + + +def write_base_pytest_config(base_ref, repo_root): + """A copy of the base commit's pytest config file, picked in pytest's order, for ``-c``; an empty one if none.""" + out = tempfile.mkdtemp(prefix="fc_pytest_cfg_") + for name in _PYTEST_CONFIG_FILES: + code, text, _ = _run(["git", "show", "%s:%s" % (base_ref, name)], cwd=repo_root) + if code != 0: + continue + if name in _PYTEST_SECTIONS and not any(n and re.match(_PYTEST_SECTIONS[name] + "$", n) for n, _ in _sections(text)): + continue + path = os.path.join(out, name) + with open(path, "w") as fh: + fh.write(text) + return path + path = os.path.join(out, "pytest.ini") + with open(path, "w") as fh: + fh.write("[pytest]\n") + return path + + +def fail_without_base_config(output, test_edits): + """Without the base config pytest may have read the agent's config files, so the check fails.""" + if "config_error" in test_edits: + reg = dict(output.get("regression") or {}, ran=True) + reg["regressed"] = list(reg.get("regressed") or []) + ["config:unavailable"] + output["regression"] = reg + + +def restore_test_edits(base_ref, repo_root): + """Put the test files back to ``base_ref`` so the agent cannot weaken the tests that judge it.""" + edits = {"modified": [], "added": [], "deleted": [], "config_changed": []} + for path, status in sorted(_changed_vs(base_ref, repo_root).items()): + # pytest reads the base config through -c (write_base_pytest_config), so config edits are only recorded. + if os.path.basename(path) in _PYTEST_CONFIG_FILES: + edits["config_changed"].append(path) + elif is_test_path(path): + edits[{"A": "added", "D": "deleted"}.get(status, "modified")].append(path) + _checkout(base_ref, edits["modified"] + edits["deleted"], repo_root) + edits["added_sha256"] = _fingerprint(edits["added"], repo_root) + for path in edits["added"]: + _remove(os.path.join(repo_root, path)) + return edits + + +def run_base_and_diff(selected_tests, extra_args, repo_root, agent_results, base_ref="HEAD"): + """Put every changed file back to ``base_ref``, rerun the same tests, restore the agent tree; flag base-pass -> agent-fail.""" + import shutil import sys as _sys import tempfile as _tempfile agent_oc = _final_outcomes(agent_results) - # Tracked changes only: untracked generated files (e.g. setuptools-scm _version.py) must stay importable. - code, out, _ = _run( - ["git", "stash", "push", "-m", "fc-regression"], cwd=repo_root - ) - stashed = code == 0 and "No local changes" not in (out or "") + changed = sorted(_changed_vs(base_ref, repo_root)) + # Committed and untracked agent changes are reverted too, unlike git stash. + backup = _tempfile.mkdtemp(prefix="fcagent_") + index = os.path.join(repo_root, _run(["git", "rev-parse", "--git-path", "index"], cwd=repo_root)[1].strip()) + if os.path.isfile(index): + shutil.copy2(index, os.path.join(backup, ".index")) + for path in changed: + src = os.path.join(repo_root, path) + if os.path.lexists(src): + dst = os.path.join(backup, "tree", path) + os.makedirs(os.path.dirname(dst), exist_ok=True) + shutil.copy2(src, dst, follow_symlinks=False) + before = _fingerprint(changed, repo_root) + reverted = bool(changed) + restore_ok = True base_oc = {} + base_plugins = None base_error = None # test.sh sets this when the patch touches compiled sources; base must not import the patched .so. rebuild = os.environ.get("FC_REBUILD_CMD") - if stashed: + if reverted: try: + at_base = [p for p in changed if _run(["git", "cat-file", "-e", "%s:%s" % (base_ref, p)], cwd=repo_root)[0] == 0] + for path in set(changed) - set(at_base): + _remove(os.path.join(repo_root, path)) + _checkout(base_ref, at_base, repo_root) if rebuild: rc, _, err = _run(["bash", "-c", rebuild], cwd=repo_root) if rc != 0: @@ -333,7 +468,7 @@ def run_base_and_diff(selected_tests, extra_args, repo_root, agent_results): "spec = importlib.util.spec_from_file_location('fc_pytest_runner', %r); " "P = importlib.util.module_from_spec(spec); spec.loader.exec_module(P); " "r = P.run_pytest_and_collect(%r, extra_args=%r, cwd=%r); " - "open(%r, 'w').write(json.dumps(r.get('tests', [])))" + "open(%r, 'w').write(json.dumps({'tests': r.get('tests', []), 'plugins': r.get('plugins', [])}))" ) % (os.path.dirname(self_path), self_path, list(selected_tests), extra_args, repo_root, tf) # Own pycache: a base file with the agent file's size and mtime would load the agent bytecode. _old_pyc = os.environ.get("PYTHONPYCACHEPREFIX") @@ -347,7 +482,9 @@ def run_base_and_diff(selected_tests, extra_args, repo_root, agent_results): os.environ["PYTHONPYCACHEPREFIX"] = _old_pyc try: with open(tf) as _fh: - base_oc = _final_outcomes({"tests": json.load(_fh)}) + base_res = json.load(_fh) + base_oc = _final_outcomes(base_res) + base_plugins = base_res.get("plugins", []) except Exception: # noqa: BLE001 — treat an unreadable base run as "no base data" base_oc = {} base_error = "base run produced no results (rc=%s): %s" % ( @@ -359,17 +496,42 @@ def run_base_and_diff(selected_tests, extra_args, repo_root, agent_results): pass except _BaseRebuildFailed as e: base_error = "base rebuild failed: %s" % e + except Exception as e: # noqa: BLE001 + base_error = "base run failed: %s: %s" % (type(e).__name__, e) finally: - _run(["git", "stash", "pop"], cwd=repo_root) + try: + for path in changed: + saved = os.path.join(backup, "tree", path) + dst = os.path.join(repo_root, path) + _remove(dst) + if os.path.lexists(saved): + os.makedirs(os.path.dirname(dst), exist_ok=True) + shutil.copy2(saved, dst, follow_symlinks=False) + if os.path.isfile(os.path.join(backup, ".index")): + shutil.copy2(os.path.join(backup, ".index"), index) + # Timing runs after this on the same tree, so check it is the agent's tree again. + if _fingerprint(changed, repo_root) != before: + restore_ok = False + base_error = (base_error or "") + "; agent tree differs after restore" + except Exception as e: # noqa: BLE001 + restore_ok = False + base_error = (base_error or "") + "; restore failed: %s: %s" % (type(e).__name__, e) if rebuild and _run(["bash", "-c", rebuild], cwd=repo_root)[0] != 0: - base_error = (base_error or "") + "; patched rebuild after stash pop failed" - regressed = sorted( - n for n, o in agent_oc.items() - if o in ("failed", "error") and base_oc.get(n) == "passed" - ) + restore_ok = False + base_error = (base_error or "") + "; patched rebuild after restore failed" + shutil.rmtree(backup, ignore_errors=True) + # Passing at base and not passing with the agent: failed, error, skipped, xfailed, or not collected. + regressed = sorted(n for n, o in base_oc.items() if o == "passed" and agent_oc.get(n) not in ("passed", "xpassed")) + # A pytest plugin that only the agent run loads (an entry point from the agent's package) can rewrite reports. + agent_only_plugins = sorted(set((agent_results or {}).get("plugins", [])) - set(base_plugins or [])) + if base_plugins is not None: + regressed += ["plugin:%s" % n for n in agent_only_plugins] out = { - "ran": bool(stashed and selected_tests and base_oc), - "stashed": stashed, + "ran": bool(reverted and selected_tests and base_oc), + "reverted": reverted, + "restore_ok": restore_ok, + "base_ref": base_ref, + "agent_only_plugins": agent_only_plugins, "n_selected_tests": len(selected_tests), "n_agent_tests": len(agent_oc), "n_base_pass": sum(1 for o in base_oc.values() if o == "passed"), @@ -479,6 +641,10 @@ def pytest_runtest_logreport(self, report): def pytest_sessionfinish(self, session, exitstatus): self.results["exit_code"] = int(exitstatus) self.results["duration"] = time.time() - self.results["start_time"] + config = session.config + self.results["config_file"] = str(getattr(config, "inipath", None) or getattr(config, "inifile", None) or "") + # Anonymous plugins are named by id(); the rest (entry points, -p modules, conftests) must match the base run. + self.results["plugins"] = sorted(n for n, _ in config.pluginmanager.list_name_plugin() if n and not n.isdigit()) def pytest_collectreport(self, report): if report.failed: @@ -569,6 +735,9 @@ def run_pytest_and_collect(test_paths, extra_args=None, cwd=None): sys.path.insert(0, repo_dir) args = unique_paths + list(extra_args) + # Both runs use the base commit's pytest configuration; the agent's config files are never read. + if os.environ.get("FC_PYTEST_CONFIG"): + args = ["-c", os.environ["FC_PYTEST_CONFIG"], "--rootdir", repo_dir] + args exit_code = pytest.main(args=args, plugins=[plugin]) plugin.results["exit_code"] = int(exit_code) plugin.results["selected_paths"] = unique_paths @@ -818,6 +987,9 @@ def main(args) -> dict: repo_root=repo_root, ) + # Test files restored from base still belong to the agent's change, so their tests run. + changed = sorted(set(changed) | set(getattr(args, "restored_tests", []))) + if args.only != "all": filtered = [] for p in changed: @@ -916,6 +1088,24 @@ def main(args) -> dict: args = parse_args() # The retries below mutate extra_args; the regression gate uses the originals. _orig_extra = args.extra_args + # Must match parser.py's LOG_DIR lookup; /logs/test_results.json is a best-effort legacy copy. + logs_root = Path( + os.environ.get("T_BENCH_TASK_LOGS_PATH") + or os.environ.get("T_BENCH_CONTAINER_LOGS_PATH") + or "/logs/artifacts" + ) + logs_root.mkdir(parents=True, exist_ok=True) + # test.sh has already saved patch.diff, so the reward still sees the agent's test edits. + try: + test_edits = restore_test_edits(args.base_ref, args.repo_root or detect_repo_root()) + except Exception as e: # noqa: BLE001 + test_edits = {"error": "%s: %s" % (type(e).__name__, e)} + (logs_root / "test_edits.json").write_text(json.dumps(test_edits, sort_keys=True)) + args.restored_tests = test_edits.get("modified", []) + test_edits.get("deleted", []) + try: + os.environ["FC_PYTEST_CONFIG"] = write_base_pytest_config(args.base_ref, args.repo_root or detect_repo_root()) + except Exception as e: # noqa: BLE001 + test_edits["config_error"] = "%s: %s" % (type(e).__name__, e) output = main(args) # Selected tests but nothing collected: retry without extra args, then with importlib mode @@ -947,7 +1137,7 @@ def main(args) -> dict: output["strategy"] = strategy if selected: output["regression"] = run_base_and_diff( - selected, extra, repo_root, output["results"] + selected, extra, repo_root, output["results"], base_ref=args.base_ref ) else: output["regression"] = {"ran": False, "reason": "no tests selected (direct or fallback)"} @@ -955,15 +1145,10 @@ def main(args) -> dict: output["regression"] = {"ran": False, "error": "%s: %s" % (type(e).__name__, e)} output["fc_runner"] = "regression-v1+template" + output["test_edits"] = test_edits - # Must match parser.py's LOG_DIR lookup; /logs/test_results.json is a best-effort legacy copy. - logs_root = Path( - os.environ.get("T_BENCH_TASK_LOGS_PATH") - or os.environ.get("T_BENCH_CONTAINER_LOGS_PATH") - or "/logs/artifacts" - ) + fail_without_base_config(output, test_edits) _payload = json.dumps(output, sort_keys=True) - logs_root.mkdir(parents=True, exist_ok=True) (logs_root / "test_results.json").write_text(_payload) _legacy = Path("/logs/test_results.json") if _legacy != logs_root / "test_results.json": @@ -972,3 +1157,5 @@ def main(args) -> dict: _legacy.write_text(_payload) except Exception: # noqa: BLE001 -- best-effort; the parser path above is the critical one pass + if output["regression"].get("restore_ok") is False: + sys.exit(3) diff --git a/src/datasmith/harbor_adapter/template/tests/test.sh b/src/datasmith/harbor_adapter/template/tests/test.sh index d5e66e62..a677d2ff 100644 --- a/src/datasmith/harbor_adapter/template/tests/test.sh +++ b/src/datasmith/harbor_adapter/template/tests/test.sh @@ -65,7 +65,8 @@ FC_BASE="$(cat /opt/fc_baseline_sha 2>/dev/null || echo {{ base_commit }})" FC_INDEX="$(mktemp)" cp .git/index "${FC_INDEX}" 2>/dev/null || rm -f "${FC_INDEX}" GIT_INDEX_FILE="${FC_INDEX}" git add -A -N 2>/dev/null || true -fc_diff() { GIT_INDEX_FILE="${FC_INDEX}" git diff "${FC_BASE}" "$@" 2>/dev/null; } +# --text and no drivers: a .gitattributes file in the agent's diff must not hide edited lines from the tamper check. +fc_diff() { GIT_INDEX_FILE="${FC_INDEX}" GIT_NO_REPLACE_OBJECTS=1 git diff --text --no-ext-diff --no-textconv "${FC_BASE}" "$@" 2>/dev/null; } fc_diff > "${LOG_DIR}/patch.diff" || true patch_files=$(fc_diff --name-only | wc -l | tr -d ' ') patch_numstat=$(fc_diff --numstat || true) diff --git a/tests/docker/test_pytest_runner_regression.py b/tests/docker/test_pytest_runner_regression.py index 859a6bf2..1966ed38 100644 --- a/tests/docker/test_pytest_runner_regression.py +++ b/tests/docker/test_pytest_runner_regression.py @@ -1,4 +1,4 @@ -"""run_base_and_diff: `ran` only when the base side produced results.""" +"""run_base_and_diff and restore_test_edits: the base side runs at the base commit and the agent cannot weaken tests.""" import importlib.util import json @@ -45,10 +45,12 @@ def test_base_results_mean_ran_and_regressions_count(runner, repo: Path) -> None def test_no_base_results_is_not_ran(runner, repo: Path, monkeypatch) -> None: real = runner._run - monkeypatch.setattr(runner, "_run", lambda cmd, cwd=None: (1, "", "boom") if cmd[0] == sys.executable else real(cmd, cwd)) + monkeypatch.setattr( + runner, "_run", lambda cmd, cwd=None: (1, "", "boom") if cmd[0] == sys.executable else real(cmd, cwd) + ) out = runner.run_base_and_diff(["test_m.py"], "", str(repo), AGENT) assert out["ran"] is False - assert out["stashed"] is True + assert out["reverted"] is True assert "no results" in out["base_error"] assert "assert False" in (repo / "test_m.py").read_text() @@ -100,3 +102,182 @@ def test_repo_package_wins_over_a_copy_in_site_packages(tmp_path_factory) -> Non proc = subprocess.run([sys.executable, "-c", script], cwd=repo, env=env, capture_output=True, text=True) summary = json.loads(proc.stdout.strip().splitlines()[-1]) assert summary["passed"] == 1, proc.stdout + proc.stderr + + +def _git(cwd: Path, *args) -> str: + return subprocess.run( + ["git", "-c", "user.name=t", "-c", "user.email=t@t", *args], cwd=cwd, check=True, capture_output=True, text=True + ).stdout + + +@pytest.fixture +def src_repo(tmp_path: Path) -> tuple[Path, str]: + root = tmp_path / "repo" + (root / "pkg").mkdir(parents=True) + _git(root, "init", "-q") + (root / "pkg" / "m.py").write_text("def f():\n return 1\n") + (root / "pkg" / "test_m.py").write_text("from m import f\n\ndef test_f():\n assert f() == 1\n") + (root / "old.py").write_text("x = 1\n") + _git(root, "add", "-A") + _git(root, "commit", "-qm", "fc-baseline") + return root, _git(root, "rev-parse", "HEAD").strip() + + +def _run_runner(root: Path, base: str, logs: Path) -> tuple[dict, dict]: + env = dict(os.environ, T_BENCH_TASK_LOGS_PATH=str(logs)) + subprocess.run( + [sys.executable, str(_RUNNER), "--base", base, "--root", str(root)], + cwd=root, + env=env, + check=True, + capture_output=True, + ) + return json.loads((logs / "test_results.json").read_text()), json.loads((logs / "test_edits.json").read_text()) + + +def test_weakened_test_is_restored_and_regression_reported(src_repo, tmp_path: Path) -> None: + root, base = src_repo + (root / "pkg" / "m.py").write_text("def f():\n return 2\n") + (root / "pkg" / "test_m.py").write_text("from m import f\n\ndef test_f():\n assert f() == 2\n") + (root / "pkg" / "test_new.py").write_text("def test_n():\n assert True\n") + results, edits = _run_runner(root, base, tmp_path / "logs") + assert edits["modified"] == ["pkg/test_m.py"] and edits["added"] == ["pkg/test_new.py"] + assert not (root / "pkg" / "test_new.py").exists() + assert results["regression"]["regressed"] == ["pkg/test_m.py::test_f"] + + +def test_committed_agent_change_is_reverted_for_base_run(src_repo, tmp_path: Path) -> None: + root, base = src_repo + (root / "pkg" / "m.py").write_text("def f():\n return 2\n") + _git(root, "commit", "-qam", "agent") + results, _ = _run_runner(root, base, tmp_path / "logs") + assert results["regression"]["ran"] is True + assert results["regression"]["regressed"] == ["pkg/test_m.py::test_f"] + + +def test_agent_tree_is_restored_byte_for_byte(runner, src_repo, monkeypatch) -> None: + root, base = src_repo + (root / "pkg" / "m.py").write_text("def f():\n return 2\n") + _git(root, "commit", "-qam", "agent") + (root / "pkg" / "test_m.py").write_text("from m import f\n\ndef test_f():\n assert f() == 3\n") + (root / "new").mkdir() + (root / "new" / "new.py").write_bytes(b"y = 2\r\n") + (root / "old.py").unlink() + _git(root, "add", "pkg/test_m.py") + + def snapshot(): + files = { + p.relative_to(root): p.read_bytes() + for p in root.rglob("*") + if p.is_file() and not str(p.relative_to(root)).startswith(".") and "__pycache__" not in p.parts + } + return files, _git(root, "status", "--porcelain"), _git(root, "rev-parse", "HEAD") + + before = snapshot() + real = runner._run + monkeypatch.setattr( + runner, + "_run", + lambda cmd, cwd=None: (_ for _ in ()).throw(KeyboardInterrupt) if cmd[0] == sys.executable else real(cmd, cwd), + ) + with pytest.raises(KeyboardInterrupt): + runner.run_base_and_diff(["pkg/test_m.py"], "", str(root), AGENT, base_ref=base) + assert snapshot() == before + monkeypatch.setattr(runner, "_run", real) + assert runner.run_base_and_diff(["pkg/test_m.py"], "", str(root), AGENT, base_ref=base)["ran"] is True + assert snapshot() == before + + +def test_failed_restore_sets_restore_ok_false_and_exits_nonzero(src_repo, tmp_path: Path) -> None: + root, base = src_repo + (root / "pkg" / "m.py").write_text("def f():\n return 2\n") + logs = tmp_path / "logs" + env = dict(os.environ, T_BENCH_TASK_LOGS_PATH=str(logs), FC_REBUILD_CMD="grep -q 'return 1' pkg/m.py") + proc = subprocess.run( + [sys.executable, str(_RUNNER), "--base", base, "--root", str(root)], cwd=root, env=env, capture_output=True + ) + regression = json.loads((logs / "test_results.json").read_text())["regression"] + assert proc.returncode == 3 + assert regression["restore_ok"] is False and "patched rebuild after restore failed" in regression["base_error"] + + +def test_base_pass_test_skipped_by_agent_code_is_a_regression(src_repo, tmp_path: Path) -> None: + root, base = src_repo + (root / "pkg" / "m.py").write_text("import pytest\n\ndef f():\n pytest.skip('no')\n") + results, _ = _run_runner(root, base, tmp_path / "logs") + assert results["regression"]["regressed"] == ["pkg/test_m.py::test_f"] + + +def test_base_pass_test_not_collected_is_a_regression(src_repo, tmp_path: Path) -> None: + root, base = src_repo + (root / "pkg" / "m.py").write_text("raise ImportError('broken')\n") + results, _ = _run_runner(root, base, tmp_path / "logs") + assert results["regression"]["regressed"] == ["pkg/test_m.py::test_f"] + + +def test_weakened_testing_helper_is_restored(src_repo, tmp_path: Path) -> None: + root, _ = src_repo + (root / "pkg" / "testing").mkdir() + (root / "pkg" / "testing" / "__init__.py").write_text("") + (root / "pkg" / "testing" / "helpers.py").write_text("def check(x):\n assert x == 1\n") + (root / "pkg" / "test_m.py").write_text( + "from m import f\nfrom testing.helpers import check\n\ndef test_f():\n check(f())\n" + ) + _git(root, "add", "-A") + _git(root, "commit", "-qm", "helpers") + base = _git(root, "rev-parse", "HEAD").strip() + (root / "pkg" / "m.py").write_text("def f():\n return 2\n") + (root / "pkg" / "testing" / "helpers.py").write_text("def check(x):\n pass\n") + results, edits = _run_runner(root, base, tmp_path / "logs") + assert edits["modified"] == ["pkg/testing/helpers.py"] + assert results["regression"]["regressed"] == ["pkg/test_m.py::test_f"] + + +def _base_config_in_setup_cfg(root: Path) -> str: + (root / "setup.cfg").write_text("[tool:pytest]\naddopts = -ra\n") + _git(root, "add", "-A") + _git(root, "commit", "-qm", "config") + return _git(root, "rev-parse", "HEAD").strip() + + +def test_agent_pyproject_dotted_key_config_is_not_read(src_repo, tmp_path: Path) -> None: + root, _ = src_repo + base = _base_config_in_setup_cfg(root) + (root / "pyproject.toml").write_text('[tool]\npytest.ini_options.addopts = "--deselect pkg/test_m.py::test_f"\n') + (root / "pkg" / "m.py").write_text("def f():\n return 2\n") + results, _ = _run_runner(root, base, tmp_path / "logs") + cfg = results["results"]["config_file"] + assert cfg.endswith("setup.cfg") and "fc_pytest_cfg_" in cfg + assert results["regression"]["regressed"] == ["pkg/test_m.py::test_f"] + + +def test_added_dot_pytest_ini_is_not_read(src_repo, tmp_path: Path) -> None: + root, _ = src_repo + base = _base_config_in_setup_cfg(root) + (root / ".pytest.ini").write_text("[pytest]\naddopts = --deselect pkg/test_m.py::test_f\n") + (root / "pkg" / "m.py").write_text("def f():\n return 2\n") + results, _ = _run_runner(root, base, tmp_path / "logs") + assert "fc_pytest_cfg_" in results["results"]["config_file"] + assert results["regression"]["regressed"] == ["pkg/test_m.py::test_f"] + + +def test_plugin_loaded_only_in_the_agent_run_is_a_regression(src_repo, tmp_path: Path) -> None: + root, base = src_repo + dist = root / "fcevil-0.dist-info" + dist.mkdir() + (dist / "METADATA").write_text("Metadata-Version: 2.1\nName: fcevil\nVersion: 0\n") + (dist / "entry_points.txt").write_text("[pytest11]\nfcevil = fcevil_mod\n") + (root / "fcevil_mod.py").write_text("") + (root / "pkg" / "m.py").write_text("def f():\n return 1 # agent\n") + results, _ = _run_runner(root, base, tmp_path / "logs") + assert results["regression"]["agent_only_plugins"] == ["fcevil"] + assert "plugin:fcevil" in results["regression"]["regressed"] + + +def test_missing_base_config_fails_the_check(runner) -> None: + out = {"regression": {"ran": False, "reason": "no tests selected"}} + runner.fail_without_base_config(out, {"config_error": "OSError: x"}) + assert out["regression"]["ran"] is True and out["regression"]["regressed"] == ["config:unavailable"] + ok = {"regression": {"ran": True, "regressed": []}} + runner.fail_without_base_config(ok, {}) + assert ok["regression"]["regressed"] == [] diff --git a/tests/docker/test_separate_verifier_tree.py b/tests/docker/test_separate_verifier_tree.py index a1652914..54f31bb9 100644 --- a/tests/docker/test_separate_verifier_tree.py +++ b/tests/docker/test_separate_verifier_tree.py @@ -107,3 +107,74 @@ def test_broken_tree_diff_leaves_starting_tree(tmp_path): bash(prepare_block("agent_tree", "complete", sha, base).replace("/fc_submission", str(sub)), image) assert {k: v for k, v in files(image).items() if k not in ("sha", "agent_tree.txt")} == before assert (image / "agent_tree.txt").read_text().strip() == "tree_diff_failed" + + +def _collect_and_recreate(tmp_path: Path, edit) -> tuple[Path, Path]: + image, base = image_repo(tmp_path) + agent, verifier = tmp_path / "agent", tmp_path / "verifier" + subprocess.run(["cp", "-a", str(image), str(agent)], check=True) + subprocess.run(["cp", "-a", str(image), str(verifier)], check=True) + edit(agent) + sub = tmp_path / "submission" + bash(collect_script(base, sub), agent) + sha = verifier / "sha" + bash(prepare_block("baseline_commit", "base_copy", sha, base), verifier) + bash(prepare_block("agent_tree", "complete", sha, base).replace("/fc_submission", str(sub)), verifier) + return verifier, sha + + +def test_gitattributes_cannot_hide_a_benchmark_edit_from_the_verifier_diff(tmp_path): + def edit(agent: Path) -> None: + (agent / ".gitattributes").write_text("benchmarks/** -diff\n") + (agent / "benchmarks/bench.py").write_text("def time_x(): return 0\n") + + verifier, sha = _collect_and_recreate(tmp_path, edit) + test_sh = render_template( + "tests/test.sh", + base_commit="x", + run_pytest=False, + rounds=None, + task_id="t", + owner="o", + repo="r", + issue_number=1, + ) + fc_diff = next(line for line in test_sh.splitlines() if line.startswith("fc_diff()")) + setup = "\n".join( + line for line in test_sh.splitlines() if line.startswith(("FC_INDEX=", "cp .git/index", "GIT_INDEX_FILE=")) + ) + script = f"FC_BASE={sha.read_text().strip()}\n{setup}\n{fc_diff}\nfc_diff" + out = subprocess.run(["bash", "-c", script], cwd=verifier, check=True, capture_output=True, text=True).stdout + assert "+def time_x(): return 0" in out + + +def test_diff_with_pth_file_or_symlink_is_refused(tmp_path): + def edit(agent: Path) -> None: + (agent / "pkg.py").write_text("x = 2\n") + (agent / "evil.pth").write_text("import os\n") + (agent / "link").symlink_to("/etc/passwd") + + verifier, _ = _collect_and_recreate(tmp_path, edit) + assert (verifier / "agent_tree.txt").read_text().strip() == "tree_diff_refused" + assert not (verifier / "evil.pth").exists() and not (verifier / "link").is_symlink() + assert (verifier / "pkg.py").read_text() == "x = 1 # image edit\n" + + +def test_quoted_path_sitecustomize_is_refused(tmp_path): + def edit(agent: Path) -> None: + (agent / "pkg.py").write_text("x = 2\n") + (agent / "é").mkdir() + (agent / "é" / "sitecustomize.py").write_text("import os\n") + + verifier, _ = _collect_and_recreate(tmp_path, edit) + assert (verifier / "agent_tree.txt").read_text().strip() == "tree_diff_refused" + assert not (verifier / "é").exists() + + +def test_diff_with_package_metadata_is_refused(tmp_path): + def edit(agent: Path) -> None: + (agent / "evil-0.dist-info").mkdir() + (agent / "evil-0.dist-info" / "entry_points.txt").write_text("[pytest11]\nevil = evil\n") + + verifier, _ = _collect_and_recreate(tmp_path, edit) + assert (verifier / "agent_tree.txt").read_text().strip() == "tree_diff_refused" diff --git a/tests/docker/test_template_lint.py b/tests/docker/test_template_lint.py index aca8286e..4cf88396 100644 --- a/tests/docker/test_template_lint.py +++ b/tests/docker/test_template_lint.py @@ -62,3 +62,10 @@ def test_no_undefined_names_in_templates(rel: str) -> None: f"These files are excluded from ruff, so nothing else will catch this.\n" f"{proc.stdout}\n{proc.stderr}" ) + + +@pytest.mark.parametrize("path", sorted((_ROOT / "src/datasmith/harbor_adapter/template").rglob("*.py")), ids=lambda p: p.name) +def test_harbor_template_parses_as_python_3_8(path: Path) -> None: + import ast + + ast.parse(path.read_text(), feature_version=(3, 8))