From fb4664f79de881f02463e6d5594d414a18ab528d Mon Sep 17 00:00:00 2001 From: fangliquanflq Date: Sat, 8 Aug 2026 07:08:43 +0800 Subject: [PATCH] fix(learn): process large sources incrementally --- agent/learn_prompt.py | 16 ++++++++++++++-- tests/agent/test_learn_prompt.py | 6 ++++++ 2 files changed, 20 insertions(+), 2 deletions(-) diff --git a/agent/learn_prompt.py b/agent/learn_prompt.py index 7268b6b51c..5b8e8eb32d 100644 --- a/agent/learn_prompt.py +++ b/agent/learn_prompt.py @@ -128,6 +128,11 @@ expansive skill: decision rules, anti-patterns, key numbers and tables, with chapter/section refs back to the source. Bullet-dense, roughly 100-150 lines per file. +- Process large sources incrementally: inventory the chapters/topics first, + then read, distill, and persist ONE chapter or topic at a time before moving + to the next. Never load an entire large corpus into conversation context at + once. After all units are written, reconcile the SKILL.md index against the + actual reference files so none are missing or stale. - Add cross-cutting files when the source earns them: a `references/` glossary (terms with chapter refs), patterns/techniques, and a cheatsheet of decision tables. Skip any that would be padding. @@ -192,10 +197,13 @@ def build_learn_prompt(user_request: str) -> str: "deprecated\" as authoring requirements. Never fetch the first source " "and ignore the rest.\n\n" "Do this:\n" - "1. Gather every source the user named, using the tools you already " + "1. Inventory every source the user named, using the tools you already " "have — `read_file`/`search_files` for local files or directories, " "`web_extract` for URLs, the current conversation history if they " "referred to something you just did, and the text they pasted as-is. " + "Gather a small source now. For a large source, inspect enough to map " + "its chapters or major topics, but do not load the whole corpus into " + "conversation context; process it incrementally in step 2b. " "If the request is ambiguous about scope, make a reasonable choice " "and note it; do not stall.\n" "1b. Apply every requirement, focus, and constraint in the request to " @@ -215,7 +223,11 @@ def build_learn_prompt(user_request: str) -> str: "docs corpus gets the knowledge-base layout below — a lean SKILL.md " "index plus per-chapter `references/` files added with `skill_manage` " "write_file. If a single SKILL.md would force you to summarize away " - "most of the material, that is the signal to go expansive.\n\n" + "most of the material, that is the signal to go expansive. For this " + "layout, create or load the skill after inventorying the source, then " + "read, distill, and persist one chapter/topic at a time before reading " + "the next; finish by reconciling the SKILL.md index with every " + "reference file you wrote.\n\n" f"{_SOURCE_HYGIENE}\n\n" f"{_AUTHORING_STANDARDS}\n\n" f"{_KNOWLEDGE_SKILL_STANDARDS}\n\n" diff --git a/tests/agent/test_learn_prompt.py b/tests/agent/test_learn_prompt.py index 45ed02f83b..4308e0bead 100644 --- a/tests/agent/test_learn_prompt.py +++ b/tests/agent/test_learn_prompt.py @@ -75,6 +75,11 @@ class TestBuildLearnPrompt: assert "never reproduce" in kb # Extend an existing skill rather than minting a near-duplicate. assert "fold-in" in kb + # Large inputs must be persisted incrementally instead of overflowing + # the live conversation context before any reference file is written. + assert "one chapter or topic at a time" in kb + assert "never load an entire large corpus" in kb + assert "reconcile the skill.md index" in kb def test_prompt_embeds_all_three_standards_blocks(self): prompt = build_learn_prompt("~/books/ddia.pdf") @@ -84,6 +89,7 @@ class TestBuildLearnPrompt: # The shape decision is explicit: small source -> one file, large # prose source -> knowledge-base layout. assert "Pick the shape by the source" in prompt + assert "process it incrementally in step 2b" in prompt def test_source_hygiene_covers_invisible_unicode(self): # Extracted document text is an injection vector (Trojan Source /