Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
46 changes: 46 additions & 0 deletions tests/test_cli_inprocess.py
Original file line number Diff line number Diff line change
Expand Up @@ -12,6 +12,7 @@
from __future__ import annotations

import json
import re
import sys

import pytest
Expand Down Expand Up @@ -882,6 +883,51 @@
assert "agreement" in out
assert "fixed" in out and "broke" in out

def test_it_still_prints_the_lines_the_readme_shows(self, two_versions, capsys):
"""The README opens with a transcript of this command. Keep it a transcript.

Shapes, not figures: the sample's data is invented, but its labels and
its columns are the program's. A label renamed or a column rewidened
leaves the front page showing output the program stopped producing, and
nothing else here would notice -- the tests above assert that
"agreement" and "fixed" appear somewhere, which survives both.

The rest of the sample -- that its numbers follow from each other, and
that both READMEs show the same transcript -- is checked in
test_readme_session.py, which needs no fixtures at all.
"""
from test_readme_session import ACCURACY_ROW, figures, session

good, weak = two_versions
assert run("diff", "demo.Sentiment", str(good), str(weak), "-n", "120") == 0
printed = capsys.readouterr().out

sample = session("README.md")
for label, value in figures(sample).items():
start = sample.index(f" {label}")
column = sample.index(value, start) - sample.rindex(chr(10), 0, start) - 1
match = re.search(rf"^ {label}( +)\S", printed, re.M)
assert match, f"the program no longer prints a {label!r} line"
printed_column = len(label) + 2 + len(match.group(1))
assert printed_column == column, (
f"{label!r} puts its value at column {printed_column}, "
f"the README shows column {column}"
)

# The accuracy rows carry a version number rather than a label, and the
# versions under test are whatever this module's metastore is up to. So
# the spacing is compared and the version is not.
shown = ACCURACY_ROW.search(sample)
printed_row = ACCURACY_ROW.search(printed)
assert printed_row, "the program no longer prints a per-version accuracy row"
assert printed_row.groups() == shown.groups(), (
f"the accuracy row is spaced {printed_row.groups()}, the README shows {shown.groups()}"
)

assert "Against the labels" in printed, "the labelled-data block is gone"
for phrase in ("row(s) wrong in", "row(s) right in"):
assert phrase in printed, f"the program no longer says {phrase!r}"

def test_json_carries_the_whole_report(self, two_versions, capsys):
good, weak = two_versions
assert run("diff", "demo.Sentiment", str(good), str(weak), "-n", "80", "--json") == 0
Expand Down Expand Up @@ -1303,7 +1349,7 @@
A scaffold that creates a file the test command then ignores teaches
people the file does not matter.
"""
import re

Check notice

Code scanning / CodeQL

Module is imported more than once Note test

This import of module re is redundant, as it was previously imported
on line 15
.

def passed() -> int:
"""How many tests the last run reported.
Expand Down
111 changes: 111 additions & 0 deletions tests/test_readme_session.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,111 @@
"""The terminal session in the README has to be one the program could print.

The front page opens with a `manage.py diff` run: agreement, the transitions,
the accuracy pair, fixed against broken, and a verdict with a p-value. It is the
first thing anyone reads, it is the argument for the whole tool, and nothing
checked any part of it.

Three things can go wrong, and they go wrong silently:

the numbers stop adding up somebody edits one figure in the sample and the
others no longer follow from it -- 500 rows at
92.0% agreement is 40 changed rows, and 40
changed rows is 29 fixed plus 11 broken

the two READMEs drift the session is a terminal transcript, identical
in both languages; one gets updated and the other
does not

the program stops saying it a label gets renamed, a column gets rewidened,
"row(s) wrong in" becomes "case(s) failing in" --
and the README goes on showing output the program
no longer produces

The first two are here, and need nothing but the files. The third lives beside
the diff tests in test_cli_inprocess.py, because that is where the fixtures that
train a model are, and it compares shapes rather than figures: the sample's data
is invented, but the lines it is laid out in are not.
"""

from __future__ import annotations

import re
from math import comb
from pathlib import Path

ROOT = Path(__file__).resolve().parent.parent
READMES = ("README.md", "README.ru.md")

SESSION = re.compile(r"```bash\n(\$ python manage\.py diff.*?)\n```", re.S)

#: Every line of the sample that is "two spaces, a label, a value".
LABELLED = re.compile(r"^ (\w+) {2,}(.*)$", re.M)

#: The two accuracy rows are not labelled lines: `v1` and `v2` are the version
#: numbers of that particular run, and a real run has whatever numbers its
#: metastore is up to. Treating them as labels made the live check assert that
#: the program prints a line starting " v1", which is true only when the two
#: versions under test happen to be 1 and 2 -- as they were on the laptop this
#: was written on, and were not in CI. The fixture that trains them says as much
#: in its own docstring; the shape of these rows is checked separately, by
#: ACCURACY_ROW, which does not care which versions they are.
VERSION = re.compile(r"^v\d+$")

#: " v1 accuracy 0.7700", with the version left open.
ACCURACY_ROW = re.compile(r"^ v\d+( +)accuracy( +)[\d.]+", re.M)


def session(name: str) -> str:
found = SESSION.search((ROOT / name).read_text(encoding="utf-8"))
assert found, f"{name}: the `manage.py diff` sample session is gone"
return found.group(1)


def figures(text: str) -> dict[str, str]:
"""The labelled lines of the sample, without the per-run version rows."""
return {label: value for label, value in LABELLED.findall(text) if not VERSION.match(label)}


def test_the_two_readmes_show_the_same_session() -> None:
"""It is a transcript of a command, so translating it would be inventing it."""
assert session("README.md") == session("README.ru.md")


def test_the_sample_session_adds_up() -> None:
"""Every figure in the sample follows from the others, so check that it does.

A sample that contradicts itself is worse than no sample: it is an
advertisement for a tool whose whole pitch is that aggregate numbers hide
what moved.
"""
text = session("README.md")
values = figures(text)

rows = int(re.search(r"on (\d+) rows", text).group(1))
agreement = float(values["agreement"].rstrip("%")) / 100
changed = int(values["changed"].split()[0])
assert round(rows * (1 - agreement)) == changed, "agreement and changed disagree"

transitions = [int(n) for n in re.findall(r"^ \S+ → \S+ +(\d+)$", text, re.M)]
assert transitions, "the transition breakdown is gone"
assert sum(transitions) == changed, "the transitions do not sum to the changed rows"

accuracies = [float(a) for a in re.findall(r"accuracy +([\d.]+)", text)]
assert len(accuracies) == 2, "the sample no longer shows an accuracy for each version"
delta = float(re.search(r"accuracy +[\d.]+ +\+([\d.]+)", text).group(1))
assert round(accuracies[1] - accuracies[0], 4) == delta, "the accuracy delta is wrong"

fixed = int(values["fixed"].split()[0])
broke = int(values["broke"].split()[0])
assert fixed + broke == changed, "fixed and broke do not account for every changed row"
assert round(rows * accuracies[1]) - round(rows * accuracies[0]) == fixed - broke, (
"the accuracy gain does not match fixed minus broken"
)

# McNemar, exact two-sided binomial over the discordant pairs -- which is
# what the verdict's p-value is, and the one figure here that cannot be
# checked by squinting.
quoted = float(re.search(r"\(p=([\d.]+)\)", values["verdict"]).group(1))
n, k = fixed + broke, min(fixed, broke)
exact = 2 * sum(comb(n, i) for i in range(k + 1)) / 2**n
assert round(exact, 3) == quoted, f"verdict says p={quoted}, McNemar gives p={exact:.6f}"
Loading