diff --git a/src/datasmith/harbor_adapter/template/tests/lsv_init.py b/src/datasmith/harbor_adapter/template/tests/lsv_init.py index 173b3a2d..4f366c4f 100644 --- a/src/datasmith/harbor_adapter/template/tests/lsv_init.py +++ b/src/datasmith/harbor_adapter/template/tests/lsv_init.py @@ -30,6 +30,7 @@ def _ts() -> str: REPO_ROOT = Path("/workspace/repo") +BASELINE_SHA_FILE = "/opt/fc_baseline_sha" # iris benchmarks generate their data with DATA_GEN_PYTHON and write it to BENCHMARK_DATA (else inside the repo). os.environ.setdefault("DATA_GEN_PYTHON", sys.executable) os.environ.setdefault("BENCHMARK_DATA", os.path.join(tempfile.gettempdir(), "fc_benchmark_data")) @@ -577,6 +578,16 @@ def image_reuse_reason(image_meta: dict | None, head: str | None, current_fp: di return "fingerprint mismatch (" + "; ".join(diffs) + ")" if diffs else None +def image_commit(head: str | None) -> str | None: + """The commit the image build ran at: the parent of setup.sh's baseline commit when HEAD is that commit.""" + try: + if head and Path(BASELINE_SHA_FILE).read_text().strip() == head: + return subprocess.check_output(["git", "rev-parse", f"{head}^"], cwd=str(REPO_ROOT), text=True).strip() + except (OSError, subprocess.CalledProcessError): + pass + return head + + def _bound_build_cpus() -> bool: """Pin the image build to the first FORMULACODE_IMAGE_CPUS allowed CPUs (BuildKit has no per-RUN CPU quota).""" try: @@ -700,7 +711,7 @@ def main() -> None: _image = json.loads(_image_meta.read_text()) except (OSError, ValueError): _image = None - _reuse_reason = image_reuse_reason(_image, _head, _fingerprint, _paired) + _reuse_reason = image_reuse_reason(_image, image_commit(_head), _fingerprint, _paired) if _reuse_reason is None: from shutil import copy2 @@ -723,6 +734,8 @@ def main() -> None: # baseline are exactly the unmeasurable ones -> no speedup contribution -> # reward unchanged. The image lsv_init_results.json carries baseline_sha for # invariant #15. + # The image was measured on this same tree before setup.sh committed it. + _image["image_baseline_sha"], _image["baseline_sha"] = _image.get("baseline_sha"), _head _image["baseline_reuse"] = {"reused": True, "reason": None, "paired": _paired, "trial_fingerprint": _fingerprint} (OUTPUT_DIR / "lsv_init_results.json").write_text(json.dumps(_image, indent=2)) print( diff --git a/src/datasmith/harbor_adapter/template/tests/lsv_measure.py b/src/datasmith/harbor_adapter/template/tests/lsv_measure.py index d5f3b8cf..94f9f1bd 100644 --- a/src/datasmith/harbor_adapter/template/tests/lsv_measure.py +++ b/src/datasmith/harbor_adapter/template/tests/lsv_measure.py @@ -161,11 +161,7 @@ def _bench_dir_for(cfg_path: Path) -> Path | None: def get_changed_files(base_commit: str) -> list[str]: - """Get files changed between base_commit and current working tree. - - Uses `git diff --name-only` which catches both committed - and uncommitted changes relative to the base. - """ + """Files that differ from base_commit in the working tree (committed or not), plus new untracked files.""" result = subprocess.run( ["git", "diff", base_commit, "--name-only"], capture_output=True, @@ -175,9 +171,11 @@ def get_changed_files(base_commit: str) -> list[str]: if result.returncode != 0: print(f"WARNING: git diff failed: {result.stderr}") return [] + untracked = subprocess.run(["git", "ls-files", "--others", "--exclude-standard"], capture_output=True, text=True, + cwd=str(REPO_ROOT)).stdout files = [] - for line in result.stdout.strip().splitlines(): + for line in sorted(set((result.stdout + "\n" + untracked).splitlines())): line = line.strip() if line: path = REPO_ROOT / line diff --git a/src/datasmith/harbor_adapter/template/tests/setup.sh b/src/datasmith/harbor_adapter/template/tests/setup.sh index 3aa5faca..d536561e 100644 --- a/src/datasmith/harbor_adapter/template/tests/setup.sh +++ b/src/datasmith/harbor_adapter/template/tests/setup.sh @@ -110,6 +110,22 @@ PYEOF SETUP_PHASE="extra_setup_commands" {{ extra_setup_commands }} +SETUP_PHASE="baseline_commit" +# The grader diffs against a commit of this starting tree, so the image's own edits and untracked benchmark suites +# are not agent work. Run output stays out of it; a nested repository (astropy-benchmarks/) goes in as plain files. +if [ "$(cat /opt/fc_baseline_sha 2>/dev/null)" != "$(git rev-parse HEAD)" ]; then + printf '%s\n' asv_benchmarks.txt solution_patch.diff '*.orig' '*.rej' __pycache__/ '*.py[co]' .asv/ \ + '/tmp.*' '/*.npz' /test_array.csv '*.nc' '*.nc4' .hypothesis/ .pytest_cache/ >> .git/info/exclude + python -c 'import json, os; c = json.load(open("/workspace/asv.conf.json")); print("\n".join("/" + os.path.relpath(c[k], "/workspace/repo") + "/" for k in ("results_dir", "env_dir", "html_dir") if (c.get(k) or "").startswith("/workspace/repo/")))' >> .git/info/exclude || true + git ls-files --others --exclude-standard | { grep '/$' || true; } | while read -r d; do + git -C "$d" ls-files -co --exclude-standard | sed "s|^|$d|" + done | git update-index --add --stdin + git add -A + git -c user.name=fc -c user.email=fc@local commit -q --no-verify --allow-empty -m fc-baseline + git rev-parse HEAD > /opt/fc_baseline_sha + chmod 444 /opt/fc_baseline_sha +fi + SETUP_PHASE="base_copy" # lsv_measure.py times this unpatched copy (with its build) against the patched repo. Overlayfs cannot rename # image directories, so the repo is moved out and copied back to make both trees renamable. diff --git a/src/datasmith/harbor_adapter/template/tests/test.sh b/src/datasmith/harbor_adapter/template/tests/test.sh index 339c4e4f..c462348f 100644 --- a/src/datasmith/harbor_adapter/template/tests/test.sh +++ b/src/datasmith/harbor_adapter/template/tests/test.sh @@ -50,9 +50,16 @@ test_start=$(date +%s) # ── Capture patch + detect whether the agent actually changed anything ─── echo "[$(ts)] [test] Capturing patch..." -git diff {{ base_commit }} > "${LOG_DIR}/patch.diff" 2>/dev/null || true -patch_files=$(git diff {{ base_commit }} --name-only 2>/dev/null | wc -l | tr -d ' ') -patch_numstat=$(git diff {{ base_commit }} --numstat 2>/dev/null || true) +# setup.sh commits the starting tree and writes its sha; without that file (an older setup.sh) the upstream base is used. +FC_BASE="$(cat /opt/fc_baseline_sha 2>/dev/null || echo {{ base_commit }})" +# Intent-to-add entries show new files in the diff. They go in a copy of the index, because git stash rejects them. +FC_INDEX="$(mktemp)" +cp .git/index "${FC_INDEX}" 2>/dev/null || rm -f "${FC_INDEX}" +GIT_INDEX_FILE="${FC_INDEX}" git add -A -N 2>/dev/null || true +fc_diff() { GIT_INDEX_FILE="${FC_INDEX}" git diff "${FC_BASE}" "$@" 2>/dev/null; } +fc_diff > "${LOG_DIR}/patch.diff" || true +patch_files=$(fc_diff --name-only | wc -l | tr -d ' ') +patch_numstat=$(fc_diff --numstat || true) patch_added=$(printf '%s\n' "${patch_numstat}" | awk '{a+=$1} END {print a+0}') patch_removed=$(printf '%s\n' "${patch_numstat}" | awk '{d+=$2} END {print d+0}') python - "${LOG_DIR}" "${patch_files}" "${patch_added}" "${patch_removed}" <<'PYEOF' @@ -73,7 +80,7 @@ echo "[$(ts)] [test] patch: files=${patch_files} +${patch_added}/-${patch_remove # pytest_runner.py reruns FC_REBUILD_CMD around its base-side run. FC_REBUILD_CMD="" # No pipe into grep -q: under pipefail the early exit of grep can fail the test (SIGPIPE on the writer). -_changed="$({ git diff {{ base_commit }} --name-only; git ls-files --others --exclude-standard; } 2>/dev/null || true)" +_changed="$(fc_diff --name-only || true)" if grep -qE '\.(pyx|pxd|pxi|c|cc|cpp|cxx|h|hh|hpp)$|(^|/)(setup\.py|setup\.cfg|pyproject\.toml|meson\.build|CMakeLists\.txt)$' <<< "${_changed}"; then FC_REBUILD_CMD="bash /tests/rebuild.sh" echo "[$(ts)] [test] Patch touches compiled sources; rebuilding (log: ${LOG_DIR}/rebuild.log)..." @@ -90,7 +97,7 @@ echo "[$(ts)] [test] Running LSV measure..." # The exit code is re-raised after release so a failure never leaks the lease. lsv_measure_start=$(date +%s) mg_acquire measure -if python /tests/lsv_measure.py --base-commit {{ base_commit }}{% if rounds is not none %} --rounds {{ rounds }}{% endif %}; then _mg_rc=0; else _mg_rc=$?; fi +if python /tests/lsv_measure.py --base-commit "${FC_BASE}"{% if rounds is not none %} --rounds {{ rounds }}{% endif %}; then _mg_rc=0; else _mg_rc=$?; fi mg_release if [ "${_mg_rc:-0}" -ne 0 ]; then exit "${_mg_rc}"; fi lsv_measure_end=$(date +%s) @@ -153,7 +160,7 @@ pytest_start=$(date +%s) {%- if run_pytest %} echo "[$(ts)] [test] Running pytest..." # A hung test (e.g. a stuck plot-export browser) would hold its gate slot until the verifier limit; timeout ends its process group. -if timeout --kill-after=60 "${FC_PYTEST_TIMEOUT:-1800}" python /tests/pytest_runner.py --base {{ base_commit }} --extra-args "-p jinja_patch_plugin_pandas"; then _py_rc=0; else _py_rc=$?; fi +if timeout --kill-after=60 "${FC_PYTEST_TIMEOUT:-1800}" python /tests/pytest_runner.py --base "${FC_BASE}" --extra-args "-p jinja_patch_plugin_pandas"; then _py_rc=0; else _py_rc=$?; fi [ "${_py_rc}" = 124 ] && echo "[$(ts)] [test] pytest timed out after ${FC_PYTEST_TIMEOUT:-1800}s" >&2 {% else %} mkdir -p "$LOG_DIR" @@ -184,7 +191,7 @@ PYEOF # ── Parser: compute reward ─────────────────────────────────────────────── echo "[$(ts)] [test] Computing reward -> /logs/verifier/reward.{json,txt} ..." -python /tests/parser.py --owner "${OWNER}" --repo "${REPO}" --issue-number "${ISSUE_NUMBER}" --agent-key "${AGENT_KEY}" --base-commit "{{ base_commit }}" +python /tests/parser.py --owner "${OWNER}" --repo "${REPO}" --issue-number "${ISSUE_NUMBER}" --agent-key "${AGENT_KEY}" --base-commit "${FC_BASE}" # ── Upload to Supabase (if configured) ─────────────────────────────────── if [ -n "${SUPABASE_URL:-}" ] && [ -n "${SUPABASE_ANON_KEY:-}" ] && [ -z "${FORMULACODE_NO_UPLOAD:-}" ]; then diff --git a/tests/docker/test_lsv_baseline_fingerprint.py b/tests/docker/test_lsv_baseline_fingerprint.py index 69c81ccf..c692af30 100644 --- a/tests/docker/test_lsv_baseline_fingerprint.py +++ b/tests/docker/test_lsv_baseline_fingerprint.py @@ -92,3 +92,24 @@ def test_paired_reuse_ignores_cpu_and_rounds(m): ) def test_paired_reuse_still_needs_sha_and_lsv(m, image, head, trial): assert m.image_reuse_reason(image, head, trial, paired=True) + + +def test_baseline_commit_reuses_the_image_measured_at_its_parent(m, tmp_path, monkeypatch): + import subprocess + + def git(*a): + return subprocess.run(["git", "-c", "user.name=t", "-c", "user.email=t@t", *a], cwd=tmp_path, check=True, + capture_output=True, text=True).stdout.strip() + + git("init", "-q") + git("commit", "-q", "--allow-empty", "-m", "upstream") + upstream = git("rev-parse", "HEAD") + git("commit", "-q", "--allow-empty", "-m", "fc-baseline") + head = git("rev-parse", "HEAD") + sha_file = tmp_path / "fc_baseline_sha" + monkeypatch.setattr(m, "REPO_ROOT", tmp_path) + monkeypatch.setattr(m, "BASELINE_SHA_FILE", str(sha_file)) + assert m.image_commit(head) == head + sha_file.write_text(head + "\n") + assert m.image_commit(head) == upstream + assert m.image_reuse_reason(_image(baseline_sha=upstream), m.image_commit(head), dict(FP)) is None diff --git a/tests/docker/test_template_pytest_step.py b/tests/docker/test_template_pytest_step.py index 546cf964..d0ec8c13 100644 --- a/tests/docker/test_template_pytest_step.py +++ b/tests/docker/test_template_pytest_step.py @@ -25,6 +25,7 @@ def _run(tmp_path: Path, runner: str) -> subprocess.CompletedProcess: (tmp_path / "test_results.json").write_text("{}") script = f"""set -euo pipefail LOG_DIR={tmp_path} +FC_BASE=BASE ts() {{ date +%s; }} mg_release() {{ :; }} python() {{ {runner}; }}