Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
88 changes: 77 additions & 11 deletions src/datasmith/harbor_adapter/template/tests/lsv_measure.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,7 +9,7 @@
timed in the same container instead, alternating round by round (see measure_paired).

Usage:
python /tests/lsv_measure.py --base-commit <sha> [--rounds N]
python /tests/lsv_measure.py --base-commit <sha> [--rounds N] [--only FILE] [--out NAME]
"""

from __future__ import annotations
Expand Down Expand Up @@ -44,6 +44,7 @@ def _ts() -> str:
# setup.sh copies the unpatched repo (with its build) here; when it exists, base and patched are timed in pairs.
BASE_COPY = Path("/workspace/.fc_base")
PARKED = Path("/workspace/.fc_patched")
DEFAULT_OUT = "lsv_measure_results.json"


def _strip_jsonc(text: str) -> str:
Expand Down Expand Up @@ -277,8 +278,47 @@ def paired_stats(rounds: list[dict], min_pairs: int) -> dict:
return out


def measure_paired(session, changed: list[str], args) -> dict:
"""Time LSV's diff-selected benchmarks on the base copy and the patched repo, alternating in one container."""
def read_only(path: str) -> tuple[set, dict]:
"""--only file: a JSON list of benchmark ids, or {"ids": [...], ...} whose other keys go to the output's top level."""
data = json.loads(Path(path).read_text())
if isinstance(data, list):
return set(map(str, data)), {}
return set(map(str, data["ids"])), {k: v for k, v in data.items() if k != "ids"}


def only_names(only: set, names: set) -> set:
"""Benchmark names that hold the listed ids; a parameterized id is "<name>-<param index>"."""
out = set()
for bid in only:
head, _, idx = bid.rpartition("-")
out.add(bid if bid in names or not idx.isdigit() else head)
return out & names


def confirm_set(results: dict, conf: dict) -> list:
"""Benchmarks to time again: first-pass |median ln speedup| above their theta, plus the oracle's improved ones."""
bench = results.get("benchmarks") or {}
if not bench:
return []
theta, default = conf.get("theta") or {}, float(conf["theta_default"])
unsure = {bid for bid, b in bench.items()
if (b.get("paired") or {}).get("median_log_ratio") is not None
and abs(b["paired"]["median_log_ratio"]) > float(theta.get(bid, default))}
return sorted(unsure | set(map(str, conf.get("oracle_up") or [])))


def write_confirm_set(conf_path: str, out_dir: str) -> int:
"""test.sh: write <out_dir>/confirm_set.json for --only; returns the rounds to run (0: nothing to confirm)."""
conf = json.loads(Path(conf_path).read_text())
ids = confirm_set(json.loads((Path(out_dir) / DEFAULT_OUT).read_text()), conf)
meta = {"ids": ids, "confirm_count": len(ids), "theta_source": conf.get("theta_source")}
(Path(out_dir) / "confirm_set.json").write_text(json.dumps(meta, indent=2))
return int(conf.get("rounds", 6)) if ids else 0


def measure_paired(session, changed: list[str], args, only: set | None = None) -> dict:
"""Time LSV's diff-selected benchmarks on the base copy and the patched repo, alternating in one container.
only: keep just these benchmark ids (LSV still has to select them)."""
from asv.contrib.lightspeed.deps_db import LightspeedDB
from asv.contrib.lightspeed.fingerprint import changed_files_with_fingerprints
from asv.contrib.lightspeed.session import _all_bids, _extract_deltas, _fmt, _timing_params
Expand All @@ -292,6 +332,8 @@ def measure_paired(session, changed: list[str], args) -> dict:
db = LightspeedDB(str(session.deps_db_path))
changes = changed_files_with_fingerprints(changed, db.get_stored_fshas())
names = {bid.name for bid in db.get_affected_benchmark_ids(changes)}
if only is not None:
names = only_names(only, names)
selected = benchmarks.filter_out(set(benchmarks.keys()) - names)
env = session._get_env()
launch = getattr(session._conf, "launch_method", None) or "auto"
Expand All @@ -305,6 +347,9 @@ def run() -> dict:

rounds = run_paired(run, args.rounds) if names else []
stats = paired_stats(rounds, (args.rounds + 1) // 2)
if only is not None:
# asv times every parameter combination of a benchmark; keep the listed ones.
stats = {bid: s for bid, s in stats.items() if bid in only}
bench = {}
for bid, s in stats.items():
base = statistics.median(t for t in s["base_times"] if t)
Expand All @@ -313,7 +358,7 @@ def run() -> dict:
"baseline_str": _fmt(base), "current_str": _fmt(cur), "params": params.get(bid), "paired": s}
print(f" {bid}: {_fmt(base)} -> {_fmt(cur)} (paired speedup {s['speedup']:.3f}, mad ln {s['mad_log_ratio']:.3f})")
n_selected = len(selected) if names else 0
dropped = [b for b in map(str, _all_bids(selected)) if b not in stats] if names else []
dropped = [b for b in map(str, _all_bids(selected)) if b not in stats and (only is None or b in only)] if names else []
return {
"benchmarks": bench,
"selected_count": n_selected,
Expand Down Expand Up @@ -345,7 +390,12 @@ def main() -> None:
default=None,
help="Warmup seconds before timing (default: asv auto)",
)
parser.add_argument("--only", default=None, help="JSON file of benchmark ids to time (see read_only); paired mode only")
parser.add_argument("--out", default=DEFAULT_OUT, help=f"Output file name in OUTPUT_DIR (default: {DEFAULT_OUT})")
args = parser.parse_args()
only, args.extra = read_only(args.only) if args.only else (None, {})
if only is not None:
args.extra["only_count"] = len(only)

OUTPUT_DIR.mkdir(parents=True, exist_ok=True)
os.chdir(REPO_ROOT)
Expand Down Expand Up @@ -376,7 +426,7 @@ def main() -> None:
"skipped_count": 0,
"timing": {"total_s": 0.0},
}
_write_combined_results(empty_measure)
_emit(args, empty_measure)
return

from asv.contrib.lightspeed import LightspeedSession
Expand Down Expand Up @@ -405,8 +455,13 @@ def main() -> None:
# Timing would measure the other copy, not the repo, so it is skipped.
err = "project shadowed: " + ", ".join(f"{k} -> {v}" for k, v in shadow.items())
print(f"[{_ts()}] [lsv_measure] ERROR: {err}; timing skipped")
_write_combined_results({"benchmarks": {}, "selected_count": 0, "total_count": 0, "skipped_count": 0,
"timing": {"total_s": 0.0}, "error": err}, shadow)
_emit(args, {"benchmarks": {}, "selected_count": 0, "total_count": 0, "skipped_count": 0,
"timing": {"total_s": 0.0}, "error": err}, shadow)
return

if only is not None and not BASE_COPY.is_dir():
_emit(args, {"benchmarks": {}, "selected_count": 0, "total_count": 0, "skipped_count": 0,
"timing": {"total_s": 0.0}, "error": f"--only needs the paired base copy at {BASE_COPY}"}, shadow)
return

if BASE_COPY.is_dir():
Expand All @@ -415,13 +470,15 @@ def main() -> None:
if os.environ.get("FC_REBUILD_CMD"):
print(f"[{_ts()}] [lsv_measure] rebuilding the base copy and the patched tree the same way (log: rebuild_both.log)")
rebuild_both_sides(os.environ["FC_REBUILD_CMD"], OUTPUT_DIR / "rebuild_both.log")
measure_data = measure_paired(session, changed, args)
measure_data = measure_paired(session, changed, args, only)
except Exception as e: # noqa: BLE001 -- a crashed measure still writes results with the error
print(f"[{_ts()}] [lsv_measure] ERROR: paired measure raised: {e}")
measure_data = {"benchmarks": {}, "selected_count": 0, "total_count": 0, "skipped_count": 0,
"timing": {"total_s": 0.0}, "error": f"LSV paired measure raised: {e}"}
(OUTPUT_DIR / "lsv_measure_results.json").write_text(json.dumps(_finite(measure_data), indent=2))
_write_combined_results(_finite(measure_data), shadow)
measure_data = _finite({**measure_data, **args.extra})
(OUTPUT_DIR / args.out).write_text(json.dumps(measure_data, indent=2))
if args.out == DEFAULT_OUT:
_write_combined_results(measure_data, shadow)
return

# Run measure_impacted
Expand Down Expand Up @@ -461,7 +518,7 @@ def main() -> None:
"timing": {"total_s": 0.0},
"error": err,
}
_write_combined_results(empty_measure, shadow)
_emit(args, empty_measure, shadow)
return

print(f" selected: {measure_result.selected_count}/{measure_result.total_count}")
Expand Down Expand Up @@ -550,6 +607,15 @@ def main() -> None:
print(f"[{_ts()}] [lsv_measure] Complete. Results at {OUTPUT_DIR}")


def _emit(args, measure_data: dict, shadow: dict | None = None) -> None:
"""Results of a run that timed nothing. The default output goes only into lsv_results.json, as before."""
if args.out == DEFAULT_OUT:
_write_combined_results(measure_data, shadow)
return
data = {**measure_data, **args.extra, **({"project_shadowed": shadow} if shadow is not None else {})}
(OUTPUT_DIR / args.out).write_text(json.dumps(_finite(data), indent=2))


def _write_combined_results(measure_data: dict, shadow: dict | None = None) -> None:
"""Merge init and measure results into a single lsv_results.json. shadow: packages imported from outside the repo ({} = checked, none)."""
if shadow is not None:
Expand Down
22 changes: 20 additions & 2 deletions src/datasmith/harbor_adapter/template/tests/test.sh
Original file line number Diff line number Diff line change
Expand Up @@ -102,6 +102,23 @@ mg_release
if [ "${_mg_rc:-0}" -ne 0 ]; then exit "${_mg_rc}"; fi
lsv_measure_end=$(date +%s)

# ── LSV confirmation pass (only with /tests/confirm.json) ───────────────
# More paired rounds in this container on the benchmarks the first pass left unsure; written to lsv_measure_confirm.json.
lsv_confirm_start=$(date +%s)
if [ -f /tests/confirm.json ]; then
_lsv_out="${LSV_OUTPUT_DIR:-${LOG_DIR}/lsv}"
_confirm_rounds="$(python -c 'import sys; sys.path.insert(0, "/tests"); from lsv_measure import write_confirm_set; print(write_confirm_set(*sys.argv[1:]))' /tests/confirm.json "${_lsv_out}" || echo 0)"
if [ "${_confirm_rounds}" -gt 0 ] 2>/dev/null; then
echo "[$(ts)] [test] Running LSV confirmation pass (${_confirm_rounds} rounds)..."
mg_acquire confirm
# The first pass already rebuilt both trees. A failed confirmation keeps the first pass's results.
_confirm_args=(--only "${_lsv_out}/confirm_set.json" --rounds "${_confirm_rounds}" --out lsv_measure_confirm.json)
FC_REBUILD_CMD="" python /tests/lsv_measure.py --base-commit "${FC_BASE}" "${_confirm_args[@]}" || echo "[$(ts)] [test] WARNING: LSV confirmation pass failed" >&2
mg_release
fi
fi
lsv_confirm_end=$(date +%s)

# ── Snapshot vars ───────────────────────────────────────────────────────
SNAPSHOT_DIR="${LOG_DIR}/.snapshots"
SNAPSHOT_FILTER="${FORMULACODE_SNAPSHOT_FILTER:-^(lexer|verifier)\\.}"
Expand Down Expand Up @@ -174,15 +191,16 @@ if [ "${_py_rc:-0}" -ne 0 ]; then echo "[$(ts)] [test] pytest runner exited ${_p
# Per-step timings; parser.py merges them with setup_timings.json into reward.json.
test_end=$(date +%s)
python - "${LOG_DIR}" "${test_start}" "${lsv_measure_start}" "${lsv_measure_end}" \
"${snapshot_start}" "${snapshot_end}" "${pytest_start}" \
"${lsv_confirm_start}" "${lsv_confirm_end}" "${snapshot_start}" "${snapshot_end}" "${pytest_start}" \
"${pytest_end}" "${test_end}" <<'PYEOF'
import json, sys, pathlib
args = [int(a) for a in sys.argv[2:]]
(test_start, lsv_ms, lsv_me, snap_s, snap_e, py_s, py_e, test_end) = args
(test_start, lsv_ms, lsv_me, conf_s, conf_e, snap_s, snap_e, py_s, py_e, test_end) = args
log_dir = sys.argv[1]
timings = {
"test_total_s": test_end - test_start,
"lsv_measure_s": lsv_me - lsv_ms,
"lsv_confirm_s": conf_e - conf_s,
"snapshot_s": snap_e - snap_s,
"pytest_s": py_e - py_s,
}
Expand Down
Loading
Loading