From 09e76e642f6eafe3f552d905df02d4ce82495e89 Mon Sep 17 00:00:00 2001 From: Denis_Drobyshev Date: Sun, 20 Sep 2026 22:22:03 +0300 Subject: [PATCH 1/2] Keep the README's terminal session a transcript The front page opens with a `manage.py diff` run -- agreement, the transitions, the accuracy pair, fixed against broken, a verdict with a p-value. It is the first thing anyone reads and the argument for the whole tool, and nothing checked any part of it. Three ways it rots, all silent: A figure gets edited and the others stop following from it. Five hundred rows at 92.0% agreement is forty changed rows; forty changed rows is twenty-two plus eighteen transitions, and twenty-nine fixed plus eleven broken; 0.7700 to 0.8060 over five hundred rows is eighteen rows, which is twenty-nine minus eleven. All of that is now asserted, including the verdict's p-value, which is the exact two-sided McNemar over the discordant pairs and comes to 0.006427 against the 0.006 printed. A sample that contradicts itself would be a poor advertisement for a tool whose whole pitch is that aggregate numbers hide what moved. The two READMEs drift apart. It is a transcript of a command, identical in both languages, so translating it would be inventing it -- they are compared character for character. The program stops saying it. A label renamed or a column rewidened leaves the front page showing output that no longer exists. The existing diff tests assert that "agreement" and "fixed" appear somewhere, which survives both changes, so the new test runs the real command over two real trained versions and compares the shape: every label in the sample must be printed, with its value starting at the same column, and the labelled-data block must still say "row(s) wrong in" and "row(s) right in" rather than the "case(s) failing in" the eval path uses. Nothing was wrong. Every figure already added up and every line already matched, which is worth saying: this pins a page that was right, rather than fixing one that was not. Split where the fixtures are. The two static tests need nothing but the files and run in half a second; the live one sits beside TestDiff, which already has the machinery to train a model. Checked by breaking a copy five ways -- a figure edited, the Russian README left behind, the p-value adjusted, a label renamed in diff.py, a column widened in diff.py. Each is caught, and each says which: "agreement and changed disagree", "verdict says p=0.001, McNemar gives p=0.006427", "the program no longer prints a 'agreement' line", "'changed' puts its value at column 21, the README shows column 17". --- tests/test_cli_inprocess.py | 36 +++++++++++++ tests/test_readme_session.py | 97 ++++++++++++++++++++++++++++++++++++ 2 files changed, 133 insertions(+) create mode 100644 tests/test_readme_session.py diff --git a/tests/test_cli_inprocess.py b/tests/test_cli_inprocess.py index 5115705..c0167fc 100644 --- a/tests/test_cli_inprocess.py +++ b/tests/test_cli_inprocess.py @@ -12,6 +12,7 @@ from __future__ import annotations import json +import re import sys import pytest @@ -882,6 +883,41 @@ def test_it_reports_what_moved_and_what_broke(self, two_versions, capsys): assert "agreement" in out assert "fixed" in out and "broke" in out + def test_it_still_prints_the_lines_the_readme_shows(self, two_versions, capsys): + """The README opens with a transcript of this command. Keep it a transcript. + + Shapes, not figures: the sample's data is invented, but its labels and + its columns are the program's. A label renamed or a column rewidened + leaves the front page showing output the program stopped producing, and + nothing else here would notice -- the tests above assert that + "agreement" and "fixed" appear somewhere, which survives both. + + The rest of the sample -- that its numbers follow from each other, and + that both READMEs show the same transcript -- is checked in + test_readme_session.py, which needs no fixtures at all. + """ + from test_readme_session import figures, session + + good, weak = two_versions + assert run("diff", "demo.Sentiment", str(good), str(weak), "-n", "120") == 0 + printed = capsys.readouterr().out + + sample = session("README.md") + for label, value in figures(sample).items(): + start = sample.index(f" {label}") + column = sample.index(value, start) - sample.rindex(chr(10), 0, start) - 1 + match = re.search(rf"^ {label}( +)\S", printed, re.M) + assert match, f"the program no longer prints a {label!r} line" + printed_column = len(label) + 2 + len(match.group(1)) + assert printed_column == column, ( + f"{label!r} puts its value at column {printed_column}, " + f"the README shows column {column}" + ) + + assert "Against the labels" in printed, "the labelled-data block is gone" + for phrase in ("row(s) wrong in", "row(s) right in"): + assert phrase in printed, f"the program no longer says {phrase!r}" + def test_json_carries_the_whole_report(self, two_versions, capsys): good, weak = two_versions assert run("diff", "demo.Sentiment", str(good), str(weak), "-n", "80", "--json") == 0 diff --git a/tests/test_readme_session.py b/tests/test_readme_session.py new file mode 100644 index 0000000..3040d2a --- /dev/null +++ b/tests/test_readme_session.py @@ -0,0 +1,97 @@ +"""The terminal session in the README has to be one the program could print. + +The front page opens with a `manage.py diff` run: agreement, the transitions, +the accuracy pair, fixed against broken, and a verdict with a p-value. It is the +first thing anyone reads, it is the argument for the whole tool, and nothing +checked any part of it. + +Three things can go wrong, and they go wrong silently: + + the numbers stop adding up somebody edits one figure in the sample and the + others no longer follow from it -- 500 rows at + 92.0% agreement is 40 changed rows, and 40 + changed rows is 29 fixed plus 11 broken + + the two READMEs drift the session is a terminal transcript, identical + in both languages; one gets updated and the other + does not + + the program stops saying it a label gets renamed, a column gets rewidened, + "row(s) wrong in" becomes "case(s) failing in" -- + and the README goes on showing output the program + no longer produces + +The first two are here, and need nothing but the files. The third lives beside +the diff tests in test_cli_inprocess.py, because that is where the fixtures that +train a model are, and it compares shapes rather than figures: the sample's data +is invented, but the lines it is laid out in are not. +""" + +from __future__ import annotations + +import re +from math import comb +from pathlib import Path + +ROOT = Path(__file__).resolve().parent.parent +READMES = ("README.md", "README.ru.md") + +SESSION = re.compile(r"```bash\n(\$ python manage\.py diff.*?)\n```", re.S) + +#: Every line of the sample that is "two spaces, a label, a value". +LABELLED = re.compile(r"^ (\w+) {2,}(.*)$", re.M) + + +def session(name: str) -> str: + found = SESSION.search((ROOT / name).read_text(encoding="utf-8")) + assert found, f"{name}: the `manage.py diff` sample session is gone" + return found.group(1) + + +def figures(text: str) -> dict[str, str]: + return dict(LABELLED.findall(text)) + + +def test_the_two_readmes_show_the_same_session() -> None: + """It is a transcript of a command, so translating it would be inventing it.""" + assert session("README.md") == session("README.ru.md") + + +def test_the_sample_session_adds_up() -> None: + """Every figure in the sample follows from the others, so check that it does. + + A sample that contradicts itself is worse than no sample: it is an + advertisement for a tool whose whole pitch is that aggregate numbers hide + what moved. + """ + text = session("README.md") + values = figures(text) + + rows = int(re.search(r"on (\d+) rows", text).group(1)) + agreement = float(values["agreement"].rstrip("%")) / 100 + changed = int(values["changed"].split()[0]) + assert round(rows * (1 - agreement)) == changed, "agreement and changed disagree" + + transitions = [int(n) for n in re.findall(r"^ \S+ → \S+ +(\d+)$", text, re.M)] + assert transitions, "the transition breakdown is gone" + assert sum(transitions) == changed, "the transitions do not sum to the changed rows" + + accuracies = [float(a) for a in re.findall(r"accuracy +([\d.]+)", text)] + assert len(accuracies) == 2, "the sample no longer shows an accuracy for each version" + delta = float(re.search(r"accuracy +[\d.]+ +\+([\d.]+)", text).group(1)) + assert round(accuracies[1] - accuracies[0], 4) == delta, "the accuracy delta is wrong" + + fixed = int(values["fixed"].split()[0]) + broke = int(values["broke"].split()[0]) + assert fixed + broke == changed, "fixed and broke do not account for every changed row" + assert round(rows * accuracies[1]) - round(rows * accuracies[0]) == fixed - broke, ( + "the accuracy gain does not match fixed minus broken" + ) + + # McNemar, exact two-sided binomial over the discordant pairs -- which is + # what the verdict's p-value is, and the one figure here that cannot be + # checked by squinting. + quoted = float(re.search(r"\(p=([\d.]+)\)", values["verdict"]).group(1)) + n, k = fixed + broke, min(fixed, broke) + exact = 2 * sum(comb(n, i) for i in range(k + 1)) / 2**n + assert round(exact, 3) == quoted, f"verdict says p={quoted}, McNemar gives p={exact:.6f}" From adbff30efc622a8e26554074008e3623774a55f2 Mon Sep 17 00:00:00 2001 From: Denis_Drobyshev Date: Sun, 20 Sep 2026 22:30:28 +0300 Subject: [PATCH 2/2] The version numbers in the sample are not labels CI went red on every test job with AssertionError: the program no longer prints a 'v1' line The sample's accuracy rows are v1 accuracy 0.7700 v2 accuracy 0.8060 +0.0360 and the pattern that picks out labelled lines picked up `v1` and `v2` as labels. So the live check asserted that a real run prints a line beginning " v1", which is true only when the two versions under test happen to be numbered 1 and 2. On the laptop this was written on they were, because running that one test alone trains exactly two. In CI the whole module runs, TestTrain has already trained several by then, and they were not. The fixture that produces them says this in its own docstring -- "the version numbers are read back rather than assumed: this module shares one metastore, so how many versions exist by now depends on which other tests have run" -- and I read it, quoted its behaviour in the test I was writing, and then hard-coded the numbers anyway. `figures()` now drops the version rows, and their shape is checked by ACCURACY_ROW, which compares the two runs of spacing and does not care which versions it is looking at. Run both ways since: the module whole, where the versions are well past two, and that test alone, where they are one and two. --- tests/test_cli_inprocess.py | 12 +++++++++++- tests/test_readme_session.py | 16 +++++++++++++++- 2 files changed, 26 insertions(+), 2 deletions(-) diff --git a/tests/test_cli_inprocess.py b/tests/test_cli_inprocess.py index c0167fc..08fd7a7 100644 --- a/tests/test_cli_inprocess.py +++ b/tests/test_cli_inprocess.py @@ -896,7 +896,7 @@ def test_it_still_prints_the_lines_the_readme_shows(self, two_versions, capsys): that both READMEs show the same transcript -- is checked in test_readme_session.py, which needs no fixtures at all. """ - from test_readme_session import figures, session + from test_readme_session import ACCURACY_ROW, figures, session good, weak = two_versions assert run("diff", "demo.Sentiment", str(good), str(weak), "-n", "120") == 0 @@ -914,6 +914,16 @@ def test_it_still_prints_the_lines_the_readme_shows(self, two_versions, capsys): f"the README shows column {column}" ) + # The accuracy rows carry a version number rather than a label, and the + # versions under test are whatever this module's metastore is up to. So + # the spacing is compared and the version is not. + shown = ACCURACY_ROW.search(sample) + printed_row = ACCURACY_ROW.search(printed) + assert printed_row, "the program no longer prints a per-version accuracy row" + assert printed_row.groups() == shown.groups(), ( + f"the accuracy row is spaced {printed_row.groups()}, the README shows {shown.groups()}" + ) + assert "Against the labels" in printed, "the labelled-data block is gone" for phrase in ("row(s) wrong in", "row(s) right in"): assert phrase in printed, f"the program no longer says {phrase!r}" diff --git a/tests/test_readme_session.py b/tests/test_readme_session.py index 3040d2a..d383531 100644 --- a/tests/test_readme_session.py +++ b/tests/test_readme_session.py @@ -41,6 +41,19 @@ #: Every line of the sample that is "two spaces, a label, a value". LABELLED = re.compile(r"^ (\w+) {2,}(.*)$", re.M) +#: The two accuracy rows are not labelled lines: `v1` and `v2` are the version +#: numbers of that particular run, and a real run has whatever numbers its +#: metastore is up to. Treating them as labels made the live check assert that +#: the program prints a line starting " v1", which is true only when the two +#: versions under test happen to be 1 and 2 -- as they were on the laptop this +#: was written on, and were not in CI. The fixture that trains them says as much +#: in its own docstring; the shape of these rows is checked separately, by +#: ACCURACY_ROW, which does not care which versions they are. +VERSION = re.compile(r"^v\d+$") + +#: " v1 accuracy 0.7700", with the version left open. +ACCURACY_ROW = re.compile(r"^ v\d+( +)accuracy( +)[\d.]+", re.M) + def session(name: str) -> str: found = SESSION.search((ROOT / name).read_text(encoding="utf-8")) @@ -49,7 +62,8 @@ def session(name: str) -> str: def figures(text: str) -> dict[str, str]: - return dict(LABELLED.findall(text)) + """The labelled lines of the sample, without the per-run version rows.""" + return {label: value for label, value in LABELLED.findall(text) if not VERSION.match(label)} def test_the_two_readmes_show_the_same_session() -> None: