diff --git a/.github/workflows/pages.yml b/.github/workflows/pages.yml
index b0f6cf8..1109ad2 100644
--- a/.github/workflows/pages.yml
+++ b/.github/workflows/pages.yml
@@ -38,9 +38,31 @@ jobs:
run: |
mkdir -p _site
cp index.html 404.html robots.txt sitemap.xml .nojekyll _site/
- cp -r assets legal ru _site/
+ cp -r assets checks legal ru _site/
find _site -type f | sort
+ # Deliberate naming above means a forgotten directory is silently absent
+ # from the published site rather than from the repository: /checks/ was
+ # committed, passed every structural check, and 404ed for twenty minutes
+ # while /ru/checks/ worked, because `ru` is copied whole and `checks` was
+ # not on the list. check_site.py knows which pages exist; this asks it.
+ - name: Every page the checker knows about is actually published
+ run: |
+ python - <<'PY'
+ import pathlib, re, sys
+ source = pathlib.Path("scripts/check_site.py").read_text(encoding="utf-8")
+ match = re.search(r"^PAGES = \[(.*?)\]", source, re.MULTILINE | re.DOTALL)
+ if not match:
+ sys.exit("check_site.py no longer declares PAGES in the expected form")
+ pages = re.findall(r'"([^"]+)"', match.group(1))
+ missing = [p for p in pages if not (pathlib.Path("_site") / p).is_file()]
+ for page in missing:
+ print(f" not published: {page}")
+ if missing:
+ sys.exit(f"{len(missing)} page(s) in the repository never reach the site")
+ print(f" all {len(pages)} pages published")
+ PY
+
- uses: actions/configure-pages@v6
- uses: actions/upload-pages-artifact@v5
with:
diff --git a/checks/index.html b/checks/index.html
index 94e7695..c52219a 100644
--- a/checks/index.html
+++ b/checks/index.html
@@ -150,6 +150,11 @@
Numbers pinned to code
praxis README, both languages |
tests/test_eval.py |
+
+ | praxis's offline retrieval figures are what a run actually produces |
+ the quality table in both praxis READMEs |
+ tests/test_eval.py, under PRAXIS_OFFLINE=1 |
+
| praxis depends on this organisation's own packages, not on strangers' names |
pyproject.toml extras |
@@ -242,12 +247,13 @@ What nothing checks yet
-
- praxis retrieval metrics
+ - praxis on real models
-
- recall@5 0.92 and MRR 0.94 on the full corpus are stated in prose. They depend on
- which models are installed, so holding them needs a scheduled re-run like
- decisionrl's, not a unit test. The golden set's size and a recall floor of 0.7 are
- pinned; the headline figures are not.
+ recall@5 0.92 and MRR 0.94 on the full corpus are stated in prose and measured on a
+ GPU that CI does not have. The offline column of the same table is now held by a test
+ — the offline path has no models, no network and no seed, so a run of it is
+ reproducible — but the GPU column needs a scheduled re-run on real hardware, the way
+ decisionrl re-verifies its applied claims nightly.
diff --git a/ru/checks/index.html b/ru/checks/index.html
index b989ae3..284a0d3 100644
--- a/ru/checks/index.html
+++ b/ru/checks/index.html
@@ -151,6 +151,11 @@
Числа, прибитые к коду
README praxis, обе версии |
tests/test_eval.py |
+
+ | Офлайн-метрики поиска praxis — это то, что печатает прогон |
+ таблица качества в обоих README praxis |
+ tests/test_eval.py, под PRAXIS_OFFLINE=1 |
+
| praxis зависит от пакетов этой организации, а не от чужих имён |
extras в pyproject.toml |
@@ -242,12 +247,13 @@ Что пока не проверяется
-
- Метрики поиска praxis
+ - praxis на реальных моделях
-
- recall@5 0.92 и MRR 0.94 по полному корпусу заявлены прозой. Они зависят от того,
- какие модели установлены, поэтому держать их нужно перезапуском по расписанию, как в
- decisionrl, а не юнит-тестом. Размер золотого набора и нижняя граница recall 0.7
- прибиты; сами заголовочные числа — нет.
+ recall@5 0.92 и MRR 0.94 по полному корпусу заявлены прозой и замерены на GPU,
+ которого в CI нет. Офлайн-колонка той же таблицы теперь держится тестом — у офлайн-пути
+ нет ни моделей, ни сети, ни случайности, поэтому его прогон воспроизводим, — а колонка
+ с реальными моделями требует перезапуска по расписанию на настоящем железе, как
+ decisionrl перепроверяет свои прикладные заявления еженощно.