diff --git a/tests/test_cli_inprocess.py b/tests/test_cli_inprocess.py index 5115705..08fd7a7 100644 --- a/tests/test_cli_inprocess.py +++ b/tests/test_cli_inprocess.py @@ -12,6 +12,7 @@ from __future__ import annotations import json +import re import sys import pytest @@ -882,6 +883,51 @@ def test_it_reports_what_moved_and_what_broke(self, two_versions, capsys): assert "agreement" in out assert "fixed" in out and "broke" in out + def test_it_still_prints_the_lines_the_readme_shows(self, two_versions, capsys): + """The README opens with a transcript of this command. Keep it a transcript. + + Shapes, not figures: the sample's data is invented, but its labels and + its columns are the program's. A label renamed or a column rewidened + leaves the front page showing output the program stopped producing, and + nothing else here would notice -- the tests above assert that + "agreement" and "fixed" appear somewhere, which survives both. + + The rest of the sample -- that its numbers follow from each other, and + that both READMEs show the same transcript -- is checked in + test_readme_session.py, which needs no fixtures at all. + """ + from test_readme_session import ACCURACY_ROW, figures, session + + good, weak = two_versions + assert run("diff", "demo.Sentiment", str(good), str(weak), "-n", "120") == 0 + printed = capsys.readouterr().out + + sample = session("README.md") + for label, value in figures(sample).items(): + start = sample.index(f" {label}") + column = sample.index(value, start) - sample.rindex(chr(10), 0, start) - 1 + match = re.search(rf"^ {label}( +)\S", printed, re.M) + assert match, f"the program no longer prints a {label!r} line" + printed_column = len(label) + 2 + len(match.group(1)) + assert printed_column == column, ( + f"{label!r} puts its value at column {printed_column}, " + f"the README shows column {column}" + ) + + # The accuracy rows carry a version number rather than a label, and the + # versions under test are whatever this module's metastore is up to. So + # the spacing is compared and the version is not. + shown = ACCURACY_ROW.search(sample) + printed_row = ACCURACY_ROW.search(printed) + assert printed_row, "the program no longer prints a per-version accuracy row" + assert printed_row.groups() == shown.groups(), ( + f"the accuracy row is spaced {printed_row.groups()}, the README shows {shown.groups()}" + ) + + assert "Against the labels" in printed, "the labelled-data block is gone" + for phrase in ("row(s) wrong in", "row(s) right in"): + assert phrase in printed, f"the program no longer says {phrase!r}" + def test_json_carries_the_whole_report(self, two_versions, capsys): good, weak = two_versions assert run("diff", "demo.Sentiment", str(good), str(weak), "-n", "80", "--json") == 0 diff --git a/tests/test_readme_session.py b/tests/test_readme_session.py new file mode 100644 index 0000000..d383531 --- /dev/null +++ b/tests/test_readme_session.py @@ -0,0 +1,111 @@ +"""The terminal session in the README has to be one the program could print. + +The front page opens with a `manage.py diff` run: agreement, the transitions, +the accuracy pair, fixed against broken, and a verdict with a p-value. It is the +first thing anyone reads, it is the argument for the whole tool, and nothing +checked any part of it. + +Three things can go wrong, and they go wrong silently: + + the numbers stop adding up somebody edits one figure in the sample and the + others no longer follow from it -- 500 rows at + 92.0% agreement is 40 changed rows, and 40 + changed rows is 29 fixed plus 11 broken + + the two READMEs drift the session is a terminal transcript, identical + in both languages; one gets updated and the other + does not + + the program stops saying it a label gets renamed, a column gets rewidened, + "row(s) wrong in" becomes "case(s) failing in" -- + and the README goes on showing output the program + no longer produces + +The first two are here, and need nothing but the files. The third lives beside +the diff tests in test_cli_inprocess.py, because that is where the fixtures that +train a model are, and it compares shapes rather than figures: the sample's data +is invented, but the lines it is laid out in are not. +""" + +from __future__ import annotations + +import re +from math import comb +from pathlib import Path + +ROOT = Path(__file__).resolve().parent.parent +READMES = ("README.md", "README.ru.md") + +SESSION = re.compile(r"```bash\n(\$ python manage\.py diff.*?)\n```", re.S) + +#: Every line of the sample that is "two spaces, a label, a value". +LABELLED = re.compile(r"^ (\w+) {2,}(.*)$", re.M) + +#: The two accuracy rows are not labelled lines: `v1` and `v2` are the version +#: numbers of that particular run, and a real run has whatever numbers its +#: metastore is up to. Treating them as labels made the live check assert that +#: the program prints a line starting " v1", which is true only when the two +#: versions under test happen to be 1 and 2 -- as they were on the laptop this +#: was written on, and were not in CI. The fixture that trains them says as much +#: in its own docstring; the shape of these rows is checked separately, by +#: ACCURACY_ROW, which does not care which versions they are. +VERSION = re.compile(r"^v\d+$") + +#: " v1 accuracy 0.7700", with the version left open. +ACCURACY_ROW = re.compile(r"^ v\d+( +)accuracy( +)[\d.]+", re.M) + + +def session(name: str) -> str: + found = SESSION.search((ROOT / name).read_text(encoding="utf-8")) + assert found, f"{name}: the `manage.py diff` sample session is gone" + return found.group(1) + + +def figures(text: str) -> dict[str, str]: + """The labelled lines of the sample, without the per-run version rows.""" + return {label: value for label, value in LABELLED.findall(text) if not VERSION.match(label)} + + +def test_the_two_readmes_show_the_same_session() -> None: + """It is a transcript of a command, so translating it would be inventing it.""" + assert session("README.md") == session("README.ru.md") + + +def test_the_sample_session_adds_up() -> None: + """Every figure in the sample follows from the others, so check that it does. + + A sample that contradicts itself is worse than no sample: it is an + advertisement for a tool whose whole pitch is that aggregate numbers hide + what moved. + """ + text = session("README.md") + values = figures(text) + + rows = int(re.search(r"on (\d+) rows", text).group(1)) + agreement = float(values["agreement"].rstrip("%")) / 100 + changed = int(values["changed"].split()[0]) + assert round(rows * (1 - agreement)) == changed, "agreement and changed disagree" + + transitions = [int(n) for n in re.findall(r"^ \S+ → \S+ +(\d+)$", text, re.M)] + assert transitions, "the transition breakdown is gone" + assert sum(transitions) == changed, "the transitions do not sum to the changed rows" + + accuracies = [float(a) for a in re.findall(r"accuracy +([\d.]+)", text)] + assert len(accuracies) == 2, "the sample no longer shows an accuracy for each version" + delta = float(re.search(r"accuracy +[\d.]+ +\+([\d.]+)", text).group(1)) + assert round(accuracies[1] - accuracies[0], 4) == delta, "the accuracy delta is wrong" + + fixed = int(values["fixed"].split()[0]) + broke = int(values["broke"].split()[0]) + assert fixed + broke == changed, "fixed and broke do not account for every changed row" + assert round(rows * accuracies[1]) - round(rows * accuracies[0]) == fixed - broke, ( + "the accuracy gain does not match fixed minus broken" + ) + + # McNemar, exact two-sided binomial over the discordant pairs -- which is + # what the verdict's p-value is, and the one figure here that cannot be + # checked by squinting. + quoted = float(re.search(r"\(p=([\d.]+)\)", values["verdict"]).group(1)) + n, k = fixed + broke, min(fixed, broke) + exact = 2 * sum(comb(n, i) for i in range(k + 1)) / 2**n + assert round(exact, 3) == quoted, f"verdict says p={quoted}, McNemar gives p={exact:.6f}"