From 0792c325fc866b6dc808ee56e575cb1d66cc3aa7 Mon Sep 17 00:00:00 2001 From: xuefei-wang Date: Tue, 21 Jul 2026 16:57:07 -0700 Subject: [PATCH 1/9] docs: restyle README header with centered logo and for-the-badge badges Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01KhxqgzVNZB1P82smfaM2Lo --- README.md | 22 +++++++++++++++++----- 1 file changed, 17 insertions(+), 5 deletions(-) diff --git a/README.md b/README.md index 1b3d376..aa7db4c 100644 --- a/README.md +++ b/README.md @@ -1,10 +1,22 @@ +
+ +KSI + # KSI — Knowledge-centric Self-Improvement -[![CI](https://github.com/recursive-knowledge/KSI/actions/workflows/ci.yml/badge.svg)](https://github.com/recursive-knowledge/KSI/actions/workflows/ci.yml) -[![Docs](https://img.shields.io/badge/docs-site-3f51b5?logo=materialformkdocs&logoColor=white)](https://recursive-knowledge.github.io/KSI/) -[![Blog](https://img.shields.io/badge/blog-paper_page-c45e3b?logo=githubpages&logoColor=white)](https://recursive-knowledge.github.io/knowledge-centric-self-improvement/) -[![License](https://img.shields.io/badge/license-Apache--2.0-blue)](./LICENSE) -[![Python](https://img.shields.io/badge/python-3.12%2B-blue?logo=python&logoColor=white)](./pyproject.toml) +### *Disposable agents attempt your tasks, share what worked, and distill knowledge that seeds the next generation.* + +

+ CI + Docs + Blog + License + Python +

+ +
+ +--- KSI runs a population of disposable agents on **your own tasks**, each working independently in a sandboxed container. They compare notes on what worked in a From a414b7c268567f3bfc33430c4609f9bea4c49d4b Mon Sep 17 00:00:00 2001 From: xuefei-wang Date: Tue, 21 Jul 2026 17:41:57 -0700 Subject: [PATCH 2/9] docs: remove CI badge from README header Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01KhxqgzVNZB1P82smfaM2Lo --- README.md | 1 - 1 file changed, 1 deletion(-) diff --git a/README.md b/README.md index aa7db4c..1dafe9e 100644 --- a/README.md +++ b/README.md @@ -7,7 +7,6 @@ ### *Disposable agents attempt your tasks, share what worked, and distill knowledge that seeds the next generation.*

- CI Docs Blog License From 6f048949e598ded968e3e3fb190b0ff2f0af8af4 Mon Sep 17 00:00:00 2001 From: "Xuefei (Julie) Wang" <33240530+xuefei-wang@users.noreply.github.com> Date: Tue, 21 Jul 2026 18:08:35 -0700 Subject: [PATCH 3/9] Update README description for clarity --- README.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/README.md b/README.md index 1dafe9e..5a94517 100644 --- a/README.md +++ b/README.md @@ -4,7 +4,7 @@ # KSI — Knowledge-centric Self-Improvement -### *Disposable agents attempt your tasks, share what worked, and distill knowledge that seeds the next generation.* +### *Disposable agents attempt your tasks, share what worked, and distill knowledge that seeds the next attempt.*

Docs From 6a3308ebc65e8335b47594652099d4eb2ba45646 Mon Sep 17 00:00:00 2001 From: xuefei-wang Date: Tue, 21 Jul 2026 20:48:25 -0700 Subject: [PATCH 4/9] fix: surface provider profile errors and validate reasoning_effort The doctor reported a generic "no usable key" even when a profile had a key but was missing MODEL_PROVIDER/MODEL; it now surfaces the specific ProviderConfigError per profile. load_provider_profile now rejects an invalid REASONING_EFFORT (e.g. 'minimal') for openai up front, instead of letting every agent/forum/distill call 400 with the error buried in the runtime sidecar DB. Co-Authored-By: Claude Opus 4.8 (1M context) Claude-Session: https://claude.ai/code/session_01XvzJBrjK196FnT8cG3hVFf --- src/ksi/doctor.py | 15 ++++++++++----- src/ksi/providers.py | 12 ++++++++++++ tests/test_doctor.py | 15 +++++++++++++++ tests/test_providers_unit.py | 23 +++++++++++++++++++++++ 4 files changed, 60 insertions(+), 5 deletions(-) diff --git a/src/ksi/doctor.py b/src/ksi/doctor.py index 33c5815..d743801 100644 --- a/src/ksi/doctor.py +++ b/src/ksi/doctor.py @@ -197,20 +197,25 @@ def _check_providers(r: Report) -> None: return usable: list[str] = [] + reasons: list[str] = [] for prof in profiles: try: load_provider_profile(str(prof)) usable.append(prof.name) - except ProviderConfigError: - pass + except ProviderConfigError as exc: + reasons.append(f"{prof.name}: {exc}") if usable: r.ok("provider profile ready", ", ".join(usable)) else: + detail = f"checked {len(profiles)} profile(s) in {PROVIDERS_DIR}" + if reasons: + detail += " — " + "; ".join(reasons) r.fail( - "no provider profile has a usable key", - f"checked {len(profiles)} profile(s) in {PROVIDERS_DIR}", - "add a real key, e.g. set ANTHROPIC_API_KEY in configs/ksi/.env.haiku", + "no usable provider profile", + detail, + "fix the issue above — a profile needs MODEL_PROVIDER, MODEL, and a real key " + "(e.g. set OPENAI_API_KEY in configs/ksi/.env.openai)", ) diff --git a/src/ksi/providers.py b/src/ksi/providers.py index 806ed5d..7e7b103 100644 --- a/src/ksi/providers.py +++ b/src/ksi/providers.py @@ -12,6 +12,12 @@ class ProviderConfigError(KsiError, RuntimeError): pass +# Valid OpenAI GPT-5.x reasoning_effort values. The API rejects anything else +# (e.g. 'minimal') with a 400 on every call, which otherwise surfaces only deep +# in the runtime traces — validate at profile load so it fails loudly up front. +_OPENAI_REASONING_EFFORTS = frozenset({"none", "low", "medium", "high", "xhigh"}) + + _PROFILE_MOVES = (("configs/providers", "configs/ksi"),) @@ -102,6 +108,12 @@ def load_provider_profile(profile_path: str) -> dict[str, str]: if not key: raise ProviderConfigError("openai/api mode requires OPENAI_API_KEY.") out["OPENAI_API_KEY"] = key + effort = cfg.get("REASONING_EFFORT", "").strip() or os.environ.get("REASONING_EFFORT", "").strip() + if effort and effort.lower() not in _OPENAI_REASONING_EFFORTS: + raise ProviderConfigError( + f"Invalid REASONING_EFFORT={effort!r} for openai. " + f"Supported values: {', '.join(sorted(_OPENAI_REASONING_EFFORTS))}." + ) # Optional pass-throughs. for key in ( diff --git a/tests/test_doctor.py b/tests/test_doctor.py index 0d75a5b..eabecdd 100644 --- a/tests/test_doctor.py +++ b/tests/test_doctor.py @@ -103,3 +103,18 @@ def test_check_vector_status_not_a_knowledge_db(tmp_path, capsys) -> None: doctor._check_vector_status(r, str(db_path)) assert r.hard_failures == 0 assert "vector_status table missing" in capsys.readouterr().out + + +def test_check_providers_surfaces_incomplete_profile_reason(tmp_path, monkeypatch, capsys) -> None: + # A profile with a key but missing MODEL_PROVIDER/MODEL must report WHY it is + # unusable, not the generic "no usable key". + (tmp_path / ".env.openai").write_text("OPENAI_API_KEY=sk-test-123\n") + monkeypatch.setattr(doctor, "PROVIDERS_DIR", tmp_path) + + r = doctor.Report() + doctor._check_providers(r) + out = capsys.readouterr().out + + assert r.hard_failures == 1 + assert ".env.openai" in out + assert "MODEL_PROVIDER" in out diff --git a/tests/test_providers_unit.py b/tests/test_providers_unit.py index 2d23686..a791445 100644 --- a/tests/test_providers_unit.py +++ b/tests/test_providers_unit.py @@ -110,3 +110,26 @@ def test_optional_passthrough_keys_from_host_env(self, tmp_path, monkeypatch): cfg = load_provider_profile(str(p)) assert cfg["KSI_DISABLE_VECTOR"] == "1" assert cfg["MEMORY_ENABLE_SEMANTIC_SEARCH"] == "0" + + +def test_openai_invalid_reasoning_effort_raises(tmp_path, monkeypatch): + monkeypatch.delenv("REASONING_EFFORT", raising=False) + p = tmp_path / ".env.test" + p.write_text( + "MODEL_PROVIDER=openai\nMODEL=gpt-5.4-mini\nMODEL_AUTH_MODE=api\n" + "OPENAI_API_KEY=sk-test-key-123\nREASONING_EFFORT=minimal\n" + ) + with pytest.raises(ProviderConfigError, match="Invalid REASONING_EFFORT"): + load_provider_profile(str(p)) + + +def test_openai_valid_reasoning_effort_accepted(tmp_path, monkeypatch): + monkeypatch.delenv("REASONING_EFFORT", raising=False) + for effort in ("none", "low", "medium", "high", "xhigh"): + p = tmp_path / ".env.test" + p.write_text( + "MODEL_PROVIDER=openai\nMODEL=gpt-5.4-mini\nMODEL_AUTH_MODE=api\n" + f"OPENAI_API_KEY=sk-test-key-123\nREASONING_EFFORT={effort}\n" + ) + cfg = load_provider_profile(str(p)) + assert cfg["REASONING_EFFORT"] == effort From 62de3bbcb9a6fdef693ab7249f1e5227612f34d6 Mon Sep 17 00:00:00 2001 From: xuefei-wang Date: Tue, 21 Jul 2026 20:48:25 -0700 Subject: [PATCH 5/9] fix(quickstart): honor TASKS_PATH override TASKS_PATH was a hardcoded assignment while PROFILE/EXPERIMENT_NAME were overridable, so 'TASKS_PATH=... bash scripts/quickstart.sh' silently ran the bundled demo. Make it ${TASKS_PATH:-...} and document the knob. Co-Authored-By: Claude Opus 4.8 (1M context) Claude-Session: https://claude.ai/code/session_01XvzJBrjK196FnT8cG3hVFf --- scripts/quickstart.sh | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/scripts/quickstart.sh b/scripts/quickstart.sh index 375dc1f..ae78d7e 100755 --- a/scripts/quickstart.sh +++ b/scripts/quickstart.sh @@ -18,6 +18,9 @@ # PROFILE=configs/ksi/.env.openai bash scripts/quickstart.sh # # Env knobs: +# TASKS_PATH= run your own tasks .jsonl/.json instead of the bundled demo +# EXPERIMENT_NAME=x name the run (default: quickstart_demo) +# PROFILE= provider profile to use (default: configs/ksi/.env.haiku) # SKIP_BOOTSTRAP=1 don't build the image / install deps / synthesize a profile # SKIP_DOCTOR=1 skip the readiness check # DRY_RUN=true print the run command (skips the image build / npm install, @@ -30,7 +33,7 @@ KSI_ROOT="$(cd "$SCRIPT_DIR/.." && pwd)" cd "$KSI_ROOT" PROFILE="${PROFILE:-configs/ksi/.env.haiku}" -TASKS_PATH="examples/custom_tasks/tasks.jsonl" +TASKS_PATH="${TASKS_PATH:-examples/custom_tasks/tasks.jsonl}" EXPERIMENT_NAME="${EXPERIMENT_NAME:-quickstart_demo}" AGENT_IMAGE="ksi-agent:bench" From 35ff7813af11bcd031d22edda979bf606bea53b0 Mon Sep 17 00:00:00 2001 From: xuefei-wang Date: Tue, 21 Jul 2026 20:48:25 -0700 Subject: [PATCH 6/9] fix(distiller): downgrade benign cross-task 0-rows warning to info MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 0 cross-task posts is expected when the cross-task forum is off, at gen 1, or when none were produced — it fired as a scary WARNING on clean runs. Downgrade to INFO and clarify the message. Co-Authored-By: Claude Opus 4.8 (1M context) Claude-Session: https://claude.ai/code/session_01XvzJBrjK196FnT8cG3hVFf --- src/ksi/distillation/distiller.py | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/src/ksi/distillation/distiller.py b/src/ksi/distillation/distiller.py index 9da94cc..80168bb 100644 --- a/src/ksi/distillation/distiller.py +++ b/src/ksi/distillation/distiller.py @@ -667,10 +667,11 @@ def _load_cross_task_posts_all_gens(inp: DistillInput) -> list[dict[str, Any]]: } ) if not out: - log.warning( + log.info( "cross_task_forum cross-gen query returned 0 rows up to gen=%d " - "(window=%d) — no cross-task posts in window. Check that the " - "cross-task forum phase actually ran.", + "(window=%d) — no cross-task posts yet. Expected when the cross-task " + "forum is disabled (--cross-task-forum-rounds 0), at gen 1, or when no " + "posts were produced in the window.", inp.generation, window, ) From 14472937a2cd55c2250c7ad3cc08fe61681e2deb Mon Sep 17 00:00:00 2001 From: xuefei-wang Date: Tue, 21 Jul 2026 20:48:25 -0700 Subject: [PATCH 7/9] docs(getting-started): explain early-stop, expand forums, collapse output details Add a note on why the demo doesn't show steps 3-5 (single generation, forums off, all-solved -> early stop). Expand step 3 into per-task and cross-task forums. Move the verbose sample-output/artifacts/DB-check block into a collapsible details block. Co-Authored-By: Claude Opus 4.8 (1M context) Claude-Session: https://claude.ai/code/session_01XvzJBrjK196FnT8cG3hVFf --- docs/getting-started.md | 88 +++++++++++++++++++++-------------------- 1 file changed, 46 insertions(+), 42 deletions(-) diff --git a/docs/getting-started.md b/docs/getting-started.md index d30a7c1..7e167f2 100644 --- a/docs/getting-started.md +++ b/docs/getting-started.md @@ -60,46 +60,41 @@ The run logs each attempt and its score as it progresses. When it finishes, resu | Score summary (optional — only when `--output-json` is set) | `results/.json` | | Execution traces | `analysis/traces//` | -For the quickstart, `` defaults to `quickstart_demo`. The -quickstart itself doesn't pass `--output-json`, so it won't produce a -`results/quickstart_demo.json` — pass that flag yourself (or use one of the -`benchmarks/` run presets, which set it for you) if you want a score -summary written to disk. - -The direct CLI defaults traces to `analysis/traces//`; set -`KSI_TRACE_DIR` before launching if your environment needs a different trace -root. - -Here's a real excerpt from a quickstart run against `claude-haiku-4-5-20251001` -(the default `configs/ksi/.env.haiku` profile), with timestamps trimmed. The task -names, `solved=3/3`, and the `[tokens]` fields stay the same across runs; the -elapsed times and token counts won't, since they depend on the model and how the -agent solves each task: - -```text -INFO ksi.orchestrator.execution_phase: [gen 1] task=reverse-words agent=agent-1 done elapsed=27.4s score=1.0000 -INFO ksi.orchestrator.execution_phase: [gen 1] task=fizzbuzz agent=agent-0 done elapsed=28.1s score=1.0000 -INFO ksi.orchestrator.execution_phase: [gen 1] task=anagram-groups agent=agent-2 done elapsed=33.0s score=1.0000 -INFO ksi.orchestrator.engine: completed traces=3 tasks=3 solved=3/3 (100.0%) -INFO ksi.orchestrator.persistence: [tokens] total=418,329 cached_input=346,149 uncached_input=9,129 output=5,090 cache_create=57,961 -``` - -**Knowledge DB check** — every solved attempt writes an `entry_type='attempt'` -row plus an `insight` row. The quickstart turns off both forums for speed -(`--per-task-forum-rounds 0 --cross-task-forum-rounds 0`), so there are no -discussion posts — and with nothing unsolved in this single-generation run, -distillation has nothing to write either: - -```console -$ sqlite3 runtime_state/knowledge/quickstart_demo/quickstart_demo_knowledge.sqlite \ - "select entry_type, source_phase, count(*) from knowledge group by entry_type, source_phase order by entry_type, source_phase;" -attempt|execution|3 -insight|execution|3 -``` - -If your own run's numbers differ (different elapsed times, different token -counts), that's expected — `solved=3/3 (100.0%)` is the signal that your -environment is set up correctly. +For the quickstart, `` defaults to `quickstart_demo`. The run prints +each task's score as it goes and ends with a `completed … solved=3/3 (100.0%)` +line — that's the signal your environment is set up correctly. Elapsed times and +token counts vary by model and run; the task names and `solved=3/3` don't. + +??? note "A closer look — sample output, optional artifacts, and the knowledge DB" + + The quickstart doesn't pass `--output-json`, so it writes no + `results/quickstart_demo.json`; pass that flag yourself (or use a `benchmarks/` + run preset, which sets it for you) for a score summary on disk. Traces default + to `analysis/traces//` — set `KSI_TRACE_DIR` to change the root. + + A real excerpt from a run against `claude-haiku-4-5-20251001` (the default + `configs/ksi/.env.haiku` profile), timestamps trimmed: + + ```text + INFO ksi.orchestrator.execution_phase: [gen 1] task=reverse-words agent=agent-1 done elapsed=27.4s score=1.0000 + INFO ksi.orchestrator.execution_phase: [gen 1] task=fizzbuzz agent=agent-0 done elapsed=28.1s score=1.0000 + INFO ksi.orchestrator.execution_phase: [gen 1] task=anagram-groups agent=agent-2 done elapsed=33.0s score=1.0000 + INFO ksi.orchestrator.engine: completed traces=3 tasks=3 solved=3/3 (100.0%) + INFO ksi.orchestrator.persistence: [tokens] total=418,329 cached_input=346,149 uncached_input=9,129 output=5,090 cache_create=57,961 + ``` + + **Knowledge DB check** — every solved attempt writes an `entry_type='attempt'` + row plus an `insight` row. The quickstart turns both forums off for speed + (`--per-task-forum-rounds 0 --cross-task-forum-rounds 0`), so there are no + discussion posts, and with nothing unsolved in this single-generation run + distillation has nothing to write either: + + ```console + $ sqlite3 runtime_state/knowledge/quickstart_demo/quickstart_demo_knowledge.sqlite \ + "select entry_type, source_phase, count(*) from knowledge group by entry_type, source_phase order by entry_type, source_phase;" + attempt|execution|3 + insight|execution|3 + ``` ## What just happened? @@ -107,10 +102,19 @@ KSI runs a knowledge-refinement loop across generations: 1. A population of agents each attempt the tasks in isolated containers. 2. They record every attempt in the knowledge database. -3. They [*discuss*](glossary.md#forum) what worked (the *[forum](glossary.md#forum)*). -4. The system [*distills*](glossary.md#distillation) the discussion into reusable guidance. +3. They [*discuss*](glossary.md#forum) what worked in two [forums](glossary.md#forum): a **per-task forum**, where the agents that attempted the same task compare their approaches, and a **cross-task forum**, where lessons that generalize beyond a single task are shared across the whole population. +4. The system [*distills*](glossary.md#distillation) those discussions into reusable guidance. 5. The next generation is [*seeded*](glossary.md#seeding) with that guidance. +!!! note "Why the demo doesn't show steps 3–5" + The quickstart runs a single generation with both forums off, so steps 3–5 + don't fire here. And because every task solves on the first attempt, there + would be nothing to learn anyway: a solved task is dropped from later + generations (`--drop-solved`, on by default), so a multi-generation run + **stops early** once everything is solved. To watch the full loop, turn the + forums on, request several generations, and use tasks hard enough that some + fail — see [experiments.md](experiments.md). + ## Next steps - **Bring your own tasks** — the record schema, the `command` evaluator's From c957110d728149d7ad46c562e6b5315c87b538ba Mon Sep 17 00:00:00 2001 From: xuefei-wang Date: Tue, 21 Jul 2026 20:48:25 -0700 Subject: [PATCH 8/9] docs: nav restructure and site polish Enable pymdownx.details (collapsible blocks); promote Getting started to a single nav link and move Contributing to its own top-level entry. Render the landing tagline emphasis with (raw HTML block didn't process **). Drop 'benchmark' from the KSI definition and remove the --improvement-strategy FAQ. Remove the TB2 Verifier Trust Boundary (architecture) and Removed compatibility flags (experiments) sections. Pin the header title to the site name so it doesn't swap to the section on scroll. Co-Authored-By: Claude Opus 4.8 (1M context) Claude-Session: https://claude.ai/code/session_01XvzJBrjK196FnT8cG3hVFf --- docs/architecture.md | 68 +------------------------------------- docs/experiments.md | 21 ------------ docs/faq.md | 17 +--------- docs/index.md | 2 +- docs/stylesheets/extra.css | 12 +++++++ mkdocs.yml | 6 ++-- 6 files changed, 18 insertions(+), 108 deletions(-) diff --git a/docs/architecture.md b/docs/architecture.md index f174b17..0269b57 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -493,73 +493,7 @@ proxy container name still resolves. two concurrent campaigns sharing the same value intentionally share one egress proxy/network lease and must have the same provider allowlist needs. -## 11. TB2 Verifier Trust Boundary - -Terminal-Bench 2 (TB2) runs the agent and the task verifier **inside the same -container**. After the agent phase finishes, KSI injects `/tests` and invokes -the verifier through a trusted `/bin/bash` extracted from the pristine task -image, with that trusted directory prepended to `PATH`. If the trusted -toolchain cannot be established, KSI refuses the legacy `bash /tests/test.sh` -fallback and records an unscored -`trial_status=verifier_fail_closed_untrusted_toolchain` by default. Set -`KSI_TB2_REQUIRE_TRUSTED_VERIFIER=0` only to reproduce legacy comparisons. The -resulting reward is read back from an agent-visible bind mount -(`-v {logs_root}:/logs`; `verifier/reward.txt` → `resolved = bool(reward is not -None and reward >= 1.0)`). - -**Threat model.** The TB2 agent runs as **root in the same container** as the -verifier, the verifier trusts the in-container toolchain and an -**unsanitized `PATH`**, and the reward flows back over a mount the agent can -write. A determined root agent could therefore trojan the toolchain (shadow -`bash`/coreutils on `PATH`, tamper with `/tests`, or write `/logs/.../reward.txt` -directly) to force `resolved = True` without actually solving the task. This is -a **reward-hacking** surface, not an egress/exfil surface — §10 isolation does -not address it. - -**Why this is deferred (document + defer).** The exposure is **symmetric -across A/B arms**: every arm runs the agent and verifier under identical trust -assumptions, so both have equal opportunity to hack the reward. Past A/B deltas -are therefore **not contaminated** by this surface — a systematic exploit would -lift both arms equally. What it *does* mean is that **absolute TB2 numbers are -not adversarial-reward-hack-robust**: they should be read as "solve rate under a -cooperative agent," not as a hardened, exploit-proof score. - -**The self-improvement loop weakens the "cooperative agent" assumption.** Unlike -a one-shot solver, KSI distills and seeds strategies that scored `reward >= 1.0` -into later generations. If an agent ever *accidentally* discovers a reward-path -write (shadow `bash`, tamper `/tests`, write `/logs/verifier/reward.txt`), the -loop selects for and amplifies it across generations rather than treating it as a -one-off. That amplification risk — not a single gamed task — is the trigger for -implementing the pristine verifier: watch the detection audit below for -any reward-path *write* or `/solution` *read*, and prioritize the rebuild if one -appears in a recorded campaign. - -**Hardening state.** KSI now hardens the verifier entrypoint by injecting a -trusted shell from the pristine image and failing closed when that injection is -unavailable. The remaining robust fix is a **pristine verifier**: run `test.sh` -from a clean `docker commit` / fresh image of the pre-agent container with a -fully sanitized toolchain, so every interpreter and subprocess the verifier -uses is independent of anything the agent touched. That architectural rebuild -remains intentionally deferred because it would -change runtime behavior and re-baseline TB2. Separately, -the agent-runs-as-root fact is confirmed, but for the maintained 89-task native -corpus the grading assets are protected *architecturally* — the task `/tests` -directory is injected only **after** the agent phase (root-proof) and the single -in-image reference (`adaptive-rejection-sampler`) ships an AES-encrypted -`protected.tar.gz.enc` — so `/protected` file permissions were never the -protection surface. The `chmod 700 /protected` plaintext-answer leak was specific -to the now-abandoned TB1-converted tasks, not the native corpus. - -**Detection in the meantime.** A post-hoc audit scans recorded TB2 tool-traces -for agent reads/writes of verifier-controlled paths (`/protected`, `/tests`, -`/solution`, `/logs/verifier`) and grades each finding **high** (answer leak / -reward tamper — fails the gate) or **info** (recon of an empty/encrypted surface). -It surfaces suspicious access after the fact over a runtime DB -(`attempts.tool_trace_json`) or transcript dir, covering both structured -`*.json` / `*.jsonl` tool-traces and free-text `agent.transcript.md`. -This mirrors the ARC mount-answer audit precedent. - -## 12. Operational References +## 11. Operational References - CLI flags: `uv run python -m ksi.cli --help` - Run presets: [benchmarks/README.md](https://github.com/recursive-knowledge/KSI/blob/main/benchmarks/README.md) diff --git a/docs/experiments.md b/docs/experiments.md index d774581..8f05ad0 100644 --- a/docs/experiments.md +++ b/docs/experiments.md @@ -83,27 +83,6 @@ as the relevance signal and delivered only to the agent attempting it. Pass `--cross-task-distill-target-conditioning false` to use a single broadcast bundle shared across all agents instead. -## Removed compatibility flags - -The following CLI flags were deprecated compatibility aliases and have been -removed. Passing one of the forum flags fails with an error naming the -replacement; `--runtime openai` fails argparse choices validation. - -| Removed flag | Use instead | -|--------------|-------------| -| `--agents N` | `--max-concurrent-tasks N` (agent count now derives from the filtered task pool) | -| `--runtime openai` | `--runtime container` with `MODEL_PROVIDER=openai` in the provider profile | -| `--forum-rounds N` | `--per-task-forum-rounds N` **and** `--cross-task-forum-rounds N` | -| `--forum-mode off` | `--per-task-forum-rounds 0 --cross-task-forum-rounds 0` | -| `--forum-mode self` / `--forum-mode multi` | nothing — this was the default behavior (forums on) | -| `--forum-ablate-r3` | `--distill-enabled=false` | - -Note: `--knowledge-db-path` is **not** deprecated — it is the canonical -knowledge-substrate flag. The removed flag is `--memory-db-path`; use -`--knowledge-db-path` (authoritative substrate) or `--runtime-db-path` -(optional audit sidecar) instead. See -[Database Ownership](architecture.md#5-database-ownership). - ## See also - [benchmarks/docs/BENCHMARK_PREPARE.md](https://github.com/recursive-knowledge/KSI/blob/main/benchmarks/docs/BENCHMARK_PREPARE.md) — diff --git a/docs/faq.md b/docs/faq.md index 3c8787d..51cda64 100644 --- a/docs/faq.md +++ b/docs/faq.md @@ -4,7 +4,7 @@ Common questions about KSI — from first-time setup through research use. ## What is KSI in one sentence? -KSI (Knowledge-centric Self-Improvement) is a benchmark framework that treats +KSI (Knowledge-centric Self-Improvement) is a framework that treats agents as disposable workers and keeps improvement in a shared knowledge store rather than in any single agent's memory. @@ -161,21 +161,6 @@ source of truth). A complete runnable example is at for the full API reference and migration notes if you are coming from an older programmatic interface. -## What does `--improvement-strategy knowledge` vs `raw_attempts` do? - -`knowledge` (the default) runs the full self-improvement loop each generation: -per-task forum, cross-task forum, distillation, and seeding. It is -behavior-preserving relative to the engine's historical defaults. - -`raw_attempts` is the true knowledge-off ablation baseline: forums, -distillation, knowledge-guided seeding, and same-task enrichment (prior-attempt -history, best-score, memory-snapshot injection) are all skipped, regardless of -`--no-memory`; agents receive only raw-attempts seeding. Use it to measure how -much of the performance gain comes from the knowledge loop versus simply -accumulating more attempts. See -[improvement_strategies.md](improvement_strategies.md) for details and the -`register_strategy` seam to add custom strategies. - ## My run failed or produced an empty knowledge DB — where do I start? Run `uv run ksi-doctor` first; it prints a checklist of everything a run diff --git a/docs/index.md b/docs/index.md index 8987970..5d1f0d1 100644 --- a/docs/index.md +++ b/docs/index.md @@ -9,7 +9,7 @@ hide: # Knowledge-centric Self-Improvement

-A population of disposable agents attempts **your own tasks**, compares notes +A population of disposable agents attempts your own tasks, compares notes on what worked, distills the lessons into reusable knowledge, and seeds the next generation with it.

diff --git a/docs/stylesheets/extra.css b/docs/stylesheets/extra.css index c7cbef0..505cb14 100644 --- a/docs/stylesheets/extra.css +++ b/docs/stylesheets/extra.css @@ -376,3 +376,15 @@ body.mermaid-zoom-lock { max-height: 82vh; user-select: none; } + +/* Header title stays "KSI" — don't swap to the current page/section title as the + page H1 scrolls out of view (Material's default header-topic behavior). */ +.md-header__topic[data-md-component="header-topic"] { + display: none; +} +.md-header__title--active .md-header__topic:first-child { + opacity: 1; + transform: none; + pointer-events: auto; + z-index: 0; +} diff --git a/mkdocs.yml b/mkdocs.yml index 83c285d..5bbd7a4 100644 --- a/mkdocs.yml +++ b/mkdocs.yml @@ -66,6 +66,7 @@ plugins: markdown_extensions: - admonition + - pymdownx.details - attr_list - md_in_html - toc: @@ -83,9 +84,7 @@ markdown_extensions: nav: - Home: index.md - - Getting started: - - getting-started.md - - Contributing: CONTRIBUTING.md + - Getting started: getting-started.md - Your own tasks: your_own_tasks.md - FAQ: faq.md - Programmatic API: @@ -103,4 +102,5 @@ nav: - Artifacts & cleanup: artifacts.md - Runtime startup performance: runtime-startup-performance.md - Benchmarks: benchmarks.md + - Contributing: CONTRIBUTING.md - Blog ↗: https://recursive-knowledge.github.io/knowledge-centric-self-improvement/ From b006fa6f2559c9b66269f095126711b3748e8e7e Mon Sep 17 00:00:00 2001 From: xuefei-wang Date: Tue, 21 Jul 2026 21:11:29 -0700 Subject: [PATCH 9/9] docs: fix cd typo, brand casing, and spelling in contributor/extension guides - CONTRIBUTING: cd ksi -> cd KSI (clone creates KSI/, errors on case-sensitive FS) - CONTRIBUTING/extending: brand product name as KSI in prose/headings (code/import/path references to `ksi` left unchanged) - improvement_strategies: behaviour -> behavior for spelling consistency Co-Authored-By: Claude Opus 4.8 (1M context) --- docs/CONTRIBUTING.md | 8 ++++---- docs/extending.md | 10 +++++----- docs/improvement_strategies.md | 4 ++-- 3 files changed, 11 insertions(+), 11 deletions(-) diff --git a/docs/CONTRIBUTING.md b/docs/CONTRIBUTING.md index b8cbff5..71e6561 100644 --- a/docs/CONTRIBUTING.md +++ b/docs/CONTRIBUTING.md @@ -1,4 +1,4 @@ -# Contributing to ksi +# Contributing to KSI Thanks for your interest in extending the knowledge-centric self-improvement agent. This guide gets you from clone to a tested change. @@ -10,7 +10,7 @@ runs through `uv run`. ```bash git clone https://github.com/recursive-knowledge/KSI -cd ksi +cd KSI bash scripts/setup_all.sh # installs deps + builds the agent container image # or, deps only: uv sync --extra memory @@ -52,9 +52,9 @@ dispatch site or a moved symbol can break a test in a file you never opened; running only the files you touched hides that. (The pre-commit hook runs ruff and the TypeScript typechecks, but not pytest.) -## Extending ksi (the seams) +## Extending KSI (the seams) -ksi is built around small, typed extension points so you can add a benchmark, +KSI is built around small, typed extension points so you can add a benchmark, evaluator, runtime, or improvement strategy **without editing core engine code**. Each is a `Protocol` + a registry. Start here: diff --git a/docs/extending.md b/docs/extending.md index 18a5ffd..39bf5e4 100644 --- a/docs/extending.md +++ b/docs/extending.md @@ -1,6 +1,6 @@ -# Extending ksi +# Extending KSI -ksi is built around small, typed **seams** so researchers and developers can +KSI is built around small, typed **seams** so researchers and developers can implement variants without editing core engine code. Every seam is the same shape: a `Protocol` that defines the contract + a **registry** you register into. No `if name == ...` dispatch edits anywhere. @@ -57,8 +57,8 @@ dict directly. Direct mutation bypasses the duplicate-name detection, so the registry module's `__all__`). Read the registered set with `resolve_` / `supported_s`. -Every exception ksi raises subclasses `ksi.KsiError`, so a programmatic -caller can `except ksi.KsiError` to catch any ksi-originated failure (the +Every exception KSI raises subclasses `ksi.KsiError`, so a programmatic +caller can `except ksi.KsiError` to catch any KSI-originated failure (the concrete types keep their historical `RuntimeError` / `ValueError` base too). ### Constructing a registered component programmatically @@ -91,7 +91,7 @@ the base runtime; the Terminal-Bench-2 delegation wrapper is a CLI-only concern. > no arguments, so `ksi.register_strategy`'d strategies are already buildable > via `get_strategy_spec(name).factory()`. -## Driving ksi programmatically +## Driving KSI programmatically Once your seam is registered (or constructed directly), drive a run from Python with `ksi.run(...)` — see [programmatic_api.md](./programmatic_api.md). diff --git a/docs/improvement_strategies.md b/docs/improvement_strategies.md index 75fd788..93f5db8 100644 --- a/docs/improvement_strategies.md +++ b/docs/improvement_strategies.md @@ -70,7 +70,7 @@ orch.set_improvement_strategy(RawAttemptsStrategy()) ## Bundled strategies -- **`DefaultKnowledgeStrategy`** — current behaviour, extracted verbatim via +- **`DefaultKnowledgeStrategy`** — current behavior, extracted verbatim via phase-service delegation. Adds no new branching; every gating flag still works as before. - **`RawAttemptsStrategy`** — the true knowledge-off ablation: no forums, no @@ -118,7 +118,7 @@ registries). Two are built in: `knowledge` (default) and `raw_attempts` ``` (Or wire it ad-hoc via `orch.set_improvement_strategy(MyStrategy())`.) -5. Add an equivalence/behaviour test alongside +5. Add an equivalence/behavior test alongside `tests/test_improvement_strategy.py`. ## What remains engine-owned