From c8f5ca6ef000e24f56cafa2e633017604ff4b5b5 Mon Sep 17 00:00:00 2001 From: Juan Ezquerro LLanes Date: Wed, 30 Sep 2026 01:24:29 +0200 Subject: [PATCH] fix: make documentation reliable on small local reasoning models The pipeline assumed a non-reasoning model and a 4 chars/token ratio, which silently corrupted output on the models this tool actually targets. Reasoning models: ornith and gemma4 emit a `thinking` block that the client never parsed and never bounded. Measured on an RTX 5060 Ti, a single-line slot cost 400 tokens with thinking enabled but 34 with `think: false`, and a paragraph slot 795 against 63. The reasoning was invisible to the tool, ate the generation budget and could return empty or confidently false content. The client now parses the envelope, sends `think` and `num_predict`, and fails loudly when a slot yields no visible output. Context budget: budgeting in characters with a 4 chars/token assumption overshot the real 3.05 ratio, so a maximum-size payload consumed 19,997 of 20,000 tokens and the document was written mid-sentence. Three of the nine existing wiki files were already truncated, and the truncated text was stored in .metadata.json so `update` never repaired it. Budgeting now uses a configurable ratio and reserves tokens for generation, and truncated or empty output is never written or cached. Document structure: the model chose the page layout, which failed roughly half the time, and tightening the instruction made compliance worse (17%). Factual retention was never the problem, so structure moved into Go: the code renders the headings, component names and ordering, and the LLM fills one prose slot per call. Module pages now share an identical structure by construction. Other changes: test files are excluded by default (22% of calls for restated assertions), the system message is constant across a run so Ollama's prefix cache can hit, quickstart.md is derived from the architecture instead of the root summary, and parent pages describe subsystem interaction instead of restating their children. Verified on a copy of this repository with ornith:9b at the shipped defaults: 134 calls, no truncation, no empty files, and identical headings across all seven module pages. --- .code-reducer.yaml | 46 +- README.md | 140 ++++-- cmd/cmd_test.go | 122 ++++- cmd/root.go | 30 +- cmd/setup.go | 26 + docs/architecture.md | 187 ++++++-- docs/cli-wizard.md | 27 +- docs/configuration.md | 147 +++++- docs/index.md | 4 +- docs/map-reduce-caching.md | 6 +- docs/performance-vram.md | 36 +- examples/go/.code-reducer.yaml | 35 +- examples/terraform/.code-reducer.yaml | 38 +- internal/config/config.go | 63 ++- internal/config/config_test.go | 193 +++++++- internal/config/io.go | 1 + internal/config/resolve.go | 212 +++++++- internal/engine/budget.go | 27 ++ internal/engine/chunking.go | 44 +- internal/engine/chunking_test.go | 87 +++- internal/engine/client.go | 125 ++++- internal/engine/client_test.go | 168 +++++++ internal/engine/constants.go | 29 +- internal/engine/markdown.go | 162 ++++++- internal/engine/orchestrator.go | 50 +- internal/engine/orchestrator_test.go | 343 ++++++++++++- internal/engine/pages.go | 486 +++++++++++++++++++ internal/engine/pages_test.go | 666 ++++++++++++++++++++++++++ internal/engine/runner.go | 2 +- internal/engine/synthesize.go | 35 +- internal/engine/synthesize_test.go | 169 ++++++- internal/tools/file_tools.go | 53 +- internal/tools/file_tools_test.go | 72 ++- 33 files changed, 3541 insertions(+), 290 deletions(-) create mode 100644 internal/engine/budget.go create mode 100644 internal/engine/client_test.go create mode 100644 internal/engine/pages.go create mode 100644 internal/engine/pages_test.go diff --git a/.code-reducer.yaml b/.code-reducer.yaml index 71c8965..5e6ad7c 100644 --- a/.code-reducer.yaml +++ b/.code-reducer.yaml @@ -1,22 +1,52 @@ -model_id: gemma4:26b-a4b-it-qat +model_id: ornith:9b ollama_base_url: http://localhost:11434 ollama_num_ctx: 20000 docs_dir: wiki +# Reasoning output. ornith:9b is a reasoning model: with thinking enabled it +# charges its hidden thinking against every generation budget, so a one-line slot +# costs 400 tokens instead of 34 and the slot budgets below are no longer +# reachable. Keep this false. +think: false + +# Client-wide generation cap. 0 lets Ollama decide, which only bounds the fact +# extraction calls: every documentation slot is capped by its own per-kind cap +# instead. A value lower than a per-kind cap caps that cap as well. +num_predict: 0 + +# Generation caps for the documentation prose slots. The page structure is written +# by the engine, so these are ceilings a well-behaved answer never reaches: a +# one-line slot measures 34 tokens and a paragraph slot 63. A clipped answer is +# salvaged to its last complete sentence instead of failing. +slot_num_predict: 192 +paragraph_num_predict: 1024 + +# Characters per token, used to turn the context budget into a prompt payload +# budget, and the context tokens held back from every prompt for generation. +chars_per_token: 3.0 +output_token_reserve: 1024 + +# Test files are excluded by default. +include_tests: false + system_prompt: | You are Code-Reducer, an expert technical writer and code analyzer. Your job is to strictly follow instructions. You do not yap, you do not write filler. DEFENSIVE RULES: 1. Do NOT use absolute terms ('always', 'never', 'zero') unless explicitly proven. 2. Do NOT guess downstream consequences or invent unhandled paths. If an error is swallowed, just say it is swallowed. 3. Do NOT name standard library packages unless explicitly stated in the source text. 4. Only report facts you are 100% sure about. +# Prose style for every slot of a module page. The engine owns the headings, their +# order and the component names, so this prompt must never ask for structure. module_synthesis_prompt: |- - Task: Write a technical documentation page for a code module based on the provided list of its internal components. - Rule 1: Group related functions and classes under appropriate Markdown headings. - Rule 2: Explain the responsibility of the module and the data flow. - Rule 3: Keep it highly technical and dense. + Task: Write the prose for one slot of a module documentation page. + Rule 1: The headings, their order and the component names are already fixed by the tooling. Never emit headings, lists, code fences or any other structure. + Rule 2: Answer with the requested slot text only, dense and technical, and keep it as short as the slot instruction asks. + Rule 3: Use only the supplied facts, and say nothing about code you were not shown. +# Prose style for every section of the architecture and quickstart pages. architecture_prompt: |- - Task: Write a global architecture or quickstart document based on the module summaries. - Rule 1: Explain the system boundaries and how the modules interact. - Rule 2: Provide a dense, developer-friendly overview. + Task: Write the prose for one slot of a project documentation page (architecture or quickstart). + Rule 1: The title, the headings and their order are already fixed by the tooling. Never emit headings, lists, code fences or any other structure. + Rule 2: Answer with the requested slot text only, dense and developer-friendly, and keep it as short as the slot instruction asks. + Rule 3: Use only the supplied module summaries, and never invent a module, a command or a workflow you were not shown. file_fact_consolidation_prompt: |- You are a specialized code documentation assistant. diff --git a/README.md b/README.md index 5b9dd25..1990eda 100644 --- a/README.md +++ b/README.md @@ -10,7 +10,8 @@ Designed specifically for **local development and private LLMs**, Code-Reducer u * **Hierarchical Map-Reduce Pipeline**: Breaks codebase synthesis into a structured Map-Reduce pipeline to document large directories recursively, staying strictly within local LLM context limits. * **Optimized for Private & Local LLMs**: Built specifically to leverage Ollama (e.g., `ornith:9b` or `gemma4:26b`), eliminating expensive cloud API costs and keeping proprietary code local. -* **Fully Customizable Prompting System**: Allows overriding default system prompts, synthesis rules, architecture blueprints, and file fact consolidation directly from YAML configuration. +* **Deterministic Page Structure**: The engine writes every title, heading, heading order and component entry, and the model fills one bounded prose slot at a time, so page structure no longer depends on the model complying. +* **Fully Customizable Prompting System**: Allows overriding the default system prompt, the prose style of module and architecture pages, and the file fact consolidation rules directly from YAML configuration. * **Enterprise-Grade Security Sandbox**: Features path traversal guards, atomic process locking, and TOCTOU symlink hijacking defenses for safe workspace operations. * **Fast Incremental Updates**: Uses a filesystem SHA256 hash cache to only re-document modified files, propagating changes upward to minimize LLM calls. * **Extraction Steps Cache Invalidation**: Automatically detects changes in your extraction steps pipeline and invalidates the cache to ensure documentation accuracy. @@ -63,6 +64,8 @@ Runs an interactive setup flow in the current directory to generate the `.code-r * Custom files and directories to ignore (comma-separated) * Documentation output folder name (defaults to `wiki`) +Every key the wizard does not ask about (the prompts, the extraction steps, the generation budgets and `include_tests`) is carried through to the saved file unchanged. + ### 2. `code-reducer init` Scans the repository, builds the hierarchical tree, and generates the initial set of wiki markdown pages: * Generates a metadata cache in `/.metadata.json` containing the baseline metadata file summaries. @@ -97,6 +100,35 @@ ollama_base_url: http://localhost:11434 # Custom context window size ollama_num_ctx: 20000 +# Ask the model for reasoning output. Required to be false for a reasoning model such +# as ornith:9b, which otherwise charges its hidden thinking against every generation +# budget: the same slot then costs 400 tokens instead of 34, and can return no content. +think: false + +# Client-wide generation cap per LLM call. 0 lets Ollama decide. A value lower than +# a per-slot cap also caps that slot kind. +num_predict: 0 + +# Generation caps for the documentation prose slots, split by slot kind. The engine +# writes the page structure, so both are ceilings, not targets: a well-behaved slot +# returns far fewer tokens than allowed, and a generation that reaches one is +# salvaged down to its complete sentences instead of being written clipped. +# A line slot is trimmed by the code to its first sentence and to 40 words, so it +# only needs room for that sentence. A paragraph slot keeps everything it wrote. +# Measured with think: false, they cost 34 and 63 tokens. +slot_num_predict: 192 +paragraph_num_predict: 1024 + +# Prompt payload budgeting: characters per token, and the context tokens held +# back from every prompt so generation always has room. +chars_per_token: 3.0 +output_token_reserve: 1024 + +# Document test files as well. They are excluded by default because a test file +# is usually a restatement of its assertions, and each one costs a full +# extraction pass. +include_tests: false + # Target directory to write generated markdown documentation docs_dir: wiki @@ -105,18 +137,21 @@ system_prompt: | You are Code-Reducer, an expert technical writer and code analyzer. Your job is to strictly follow instructions. You do not yap, you do not write filler. DEFENSIVE RULES: 1. Do NOT use absolute terms ('always', 'never', 'zero') unless explicitly proven. 2. Do NOT guess downstream consequences or invent unhandled paths. If an error is swallowed, just say it is swallowed. 3. Do NOT name standard library packages unless explicitly stated in the source text. 4. Only report facts you are 100% sure about. -# Synthesis prompt for directory modules +# Prose style applied to every slot of a directory module page. The engine owns +# the headings, their order and the component names, so this must never ask the +# model for structure. module_synthesis_prompt: |- - Task: Write a technical documentation page for a code module based on the provided list of its internal components. - Rule 1: Group related functions and classes under appropriate Markdown headings. - Rule 2: Explain the responsibility of the module and the data flow. - Rule 3: Keep it highly technical and dense. + Task: Write the prose for one slot of a module documentation page. + Rule 1: The headings, their order and the component names are already fixed by the tooling. Never emit headings, lists, code fences or any other structure. + Rule 2: Answer with the requested slot text only, dense and technical, and keep it as short as the slot instruction asks. + Rule 3: Use only the supplied facts, and say nothing about code you were not shown. -# Synthesis prompt for the global architecture overview +# Prose style applied to every section of the global architecture and quickstart pages architecture_prompt: |- - Task: Write a global architecture or quickstart document based on the module summaries. - Rule 1: Explain the system boundaries and how the modules interact. - Rule 2: Provide a dense, developer-friendly overview. + Task: Write the prose for one slot of a project documentation page (architecture or quickstart). + Rule 1: The title, the headings and their order are already fixed by the tooling. Never emit headings, lists, code fences or any other structure. + Rule 2: Answer with the requested slot text only, dense and developer-friendly, and keep it as short as the slot instruction asks. + Rule 3: Use only the supplied module summaries, and never invent a module, a command or a workflow you were not shown. # Prompt used to consolidate chunks of the same file file_fact_consolidation_prompt: |- @@ -164,9 +199,17 @@ Code-Reducer implements a four-tier configuration resolution chain: * `CODE_REDUCER_MODEL_ID` overrides `model_id` * `OLLAMA_BASE_URL` overrides `ollama_base_url` * `OLLAMA_NUM_CTX` overrides `ollama_num_ctx` + * `CODE_REDUCER_THINK` overrides `think` + * `OLLAMA_NUM_PREDICT` overrides `num_predict` + * `CODE_REDUCER_SLOT_NUM_PREDICT` overrides `slot_num_predict` + * `CODE_REDUCER_PARAGRAPH_NUM_PREDICT` overrides `paragraph_num_predict` + * `CODE_REDUCER_INCLUDE_TESTS` overrides `include_tests` 3. **YAML File (`.code-reducer.yaml`)**: Read from the repository root. 4. **Defaults**: Hardcoded fallbacks if no other configuration exists. +### Incomplete Output Is Never Written +Every call is bounded, and every incomplete answer is rejected instead of reaching a page. A generation that stopped on the generation limit (`done_reason: "length"`) is salvaged only when it left complete sentences: a one-line slot keeps its first complete sentence, and a paragraph slot keeps the text up to its last complete sentence. A run therefore aborts only when a slot answer holds no complete sentence at all, or when a fact extraction call is truncated, and then it aborts naming the call and the token counts without writing or caching anything. See [docs/configuration.md](docs/configuration.md) for the full table. + ### Multi-language & Infrastructure Support Code-Reducer can be configured to document not only software codebases but also Infrastructure-as-Code (IaC) or cloud topology. You can inspect an example config tailored for Terraform project analysis in [examples/terraform/.code-reducer.yaml](examples/terraform/.code-reducer.yaml). @@ -180,15 +223,15 @@ Code-Reducer can be configured to document not only software codebases but also graph TD A[Code Discovery & Filtering] --> B[Build Hierarchical Tree] B --> C[Map: File-level API Extraction] - C --> D[Reduce: Batch Synthesis - Context Size Budgeted] - D --> E[Directory Module Summary .md] + C --> D[Reduce: Facts Consolidation per Chunked File] + D --> E[Render Fixed Module Skeleton & Fill Bounded Prose Slots] E --> F[Hierarchical Subsystem Synthesis] F --> G[Root Level Reduction] G --> H[Global Docs: architecture.md & quickstart.md] ``` #### Tree Structure Construction -Code-Reducer groups scanned files into a logical directory hierarchy using a node prefix tree (`DirNode` containing children, files, and path values). +Code-Reducer groups scanned files into a logical directory hierarchy using a node prefix tree (`DirNode` containing children, files, and path values). Discovery drops test files by naming convention (`*_test.go`, `test_*.py`, `*.test.js`, `*Test.java`, `*Tests.cs`, `*_spec.rb`, `*_test.rs`, `*Tests.scala` and more) unless `include_tests: true` or `--include-tests` is set, because a documented test file is mostly a restatement of its assertions and costs a full extraction pass. #### State Tracking & Change Propagation (`RunUpdate`) In `update` mode, the engine dynamically determines which directory nodes are "affected" to avoid full-repository rebuilds. A directory is marked "affected" if: @@ -204,21 +247,44 @@ The metadata cache contains a `steps_hash` field representing the SHA256 of the For every code file in an affected directory, the engine calculates the `SHA256` of its contents: * **Cache Hit**: Reuses the stored facts string from the cache. * **Cache Miss**: Analyzes the file using the configurable `extraction_steps` pipeline. -* **Llm Context-Based File Limits**: Large files are split into overlapping fragments. The engine calculates a dynamic truncation limit: it allocates 75% of `NumCtx` (typically assuming ~4 characters per token) to the file content, reserving the remaining 25% for prompts and output context. An overlap margin (defaults to 800 characters) is used to prevent context blindness at boundaries. +* **Llm Context-Based File Limits**: Large files are split into overlapping fragments. The engine budgets the payload from the context window: `output_token_reserve` tokens are held back for generation and the rest is converted into characters with `chars_per_token`. An overlap margin (defaults to 800 characters) is used to prevent context blindness at boundaries. * Isolated inference is run on each chunk for each extraction step. +* **Prompt-Cache Friendly Layout**: The system message is always `system_prompt` and nothing else, so it is byte-identical on every call of a run and Ollama can reuse it (`prompt_eval_cached_count`). The per-step instruction rides in the user turn instead, after the file block, which keeps the shared prefix intact. + +#### The Reduce Phase (Deterministic Skeleton & Bounded Prose Slots) +A module page is not written by the model. The engine renders the title, the section headings, their order, the component names and the component order, then fills each prose slot with one bounded micro-task: + +```markdown +# Module: internal/engine + +## Responsibility + + +## Components -#### The Reduce Phase (Hierarchical Consolidation & Truncation Safety) -To prevent massive folders from blowing out Ollama's context window, Code-Reducer applies a recursive bottom-up consolidation strategy grouped in dynamically sized batches (capped at `NumCtx * 3` characters): -* **File-Level Reduce**: If a single file was split into multiple chunks during the Map phase, their extracted facts are consolidated into a unified briefing via `reduceFileFacts`. -* **Directory-Level Reduce**: File briefings and child directory summaries are grouped into batches. - * If a directory's components fit into a single batch, they are joined and sent to the LLM with the `module_synthesis_prompt` to yield a unified directory summary. - * If they exceed the limit, they are split into sub-batches, reduced independently, and recursively merged. - * **Dynamic Multi-Layer Map-Reduce**: In `reduceItems`, if a single item exceeds the character limit, it is automatically chunked into smaller pieces to create a deeper reduction layer. This preserves all information without truncation. To avoid infinite map-reduce loops (e.g. if the LLM refuses to condense text), the engine compares input and output payload sizes and gracefully halts recursion without losing context or exceeding buffer limits. +### File: client.go + + +### Subsystem: config + + +## Data Flow + + +## Error Handling + +``` + +* **Minimal Context Per Slot**: The responsibility slot sees component identities only, a component slot sees only that component's facts, and the two paragraph slots see the identities plus a bounded per-component facts digest. +* **One Deterministic Line Per Line Slot**: A model asked for one line still answers with a bullet list or a rambling paragraph, so Go takes the first meaningful line (the item text when it is bulleted, otherwise the first sentence), bounds it to 40 words on a clause or word boundary, and drops the rest. +* **Parents Read Children as Identities**: A child subsystem reaches its parent as the head of its own page (title, responsibility, component entries), and a module with children is asked what flows *between* its components and how errors cross the boundary, instead of restating each child's interior. A leaf module keeps the per-component framing. +* **Bounded Per Slot**: Every slot passes its own `num_predict`, selected by slot kind (`slot_num_predict`, default `192`, for a one-line slot, `paragraph_num_predict`, default `1024`, for a paragraph slot), and a lower global `num_predict` wins for both. The defaults are sized on measurements taken with `think: false` (34 and 63 tokens), so `think: false` is required on a reasoning model such as `ornith:9b`: with thinking on, the same slots cost 400 and 795 tokens of which the tool never reads a word. +* **All-or-Nothing Pages**: A failing slot aborts the page, so nothing is written to disk and nothing is cached. A run aborts only when a slot answer holds no complete sentence to salvage, or when a fact extraction call is truncated. #### Global Synthesis Phase -After reducing the root directory (`.`), the final summary is sent to the LLM to generate: -1. **System Blueprint**: `wiki/architecture.md` (High-level architecture, module boundaries, external integrations). -2. **Developer Quickstart**: `wiki/quickstart.md` (Onboarding guide, configuration guidelines). +After the root directory (`.`) is reduced, its module page is used to generate: +1. **System Blueprint**: `wiki/architecture.md` (Overview, system boundaries, module interaction), built from the root module page. +2. **Developer Quickstart**: `wiki/quickstart.md` (What the project does, project layout, common workflows), built from the architecture page that was just generated rather than from the root page. 3. **AI Agent Guidelines**: During the initial initialization (`init`), the pipeline writes guidelines to `AGENTS.md` (or appends to it) to help other incoming agentic developers find and utilize the generated documentation. --- @@ -294,22 +360,22 @@ $ code-reducer init Starting Map-Reduce pipeline: init Step 1: Code Discovery & Building Tree... Step 2: Hierarchical Tree-Merging (Map-Reduce)... -➜ Extracting file (Step 1/4 - API_SIGNATURES): cmd/init.go -➜ Extracting file (Step 2/4 - BUSINESS_LOGIC): cmd/init.go -➜ Extracting file (Step 3/4 - STATE_AND_CONCURRENCY): cmd/init.go -➜ Extracting file (Step 4/4 - ERRORS_AND_SIDE_EFFECTS): cmd/init.go ➜ Extracting file (Step 1/4 - API_SIGNATURES): cmd/root.go +➜ Extracting file (Step 2/4 - BUSINESS_LOGIC): cmd/root.go +➜ Extracting file (Step 3/4 - STATE_AND_CONCURRENCY): cmd/root.go +➜ Extracting file (Step 4/4 - ERRORS_AND_SIDE_EFFECTS): cmd/root.go +➜ Assembling module page: cmd (4 total components) +➜ Describing responsibility of cmd +➜ Describing component File: init.go of cmd +➜ Describing component File: root.go of cmd ... -➜ Synthesizing directory: cmd (4 total components) -➜ LLM Synthesizing chunk for cmd (4 items) -... -➜ Extracting file (Step 1/4 - API_SIGNATURES): internal/config/resolve.go +➜ Describing data flow of cmd +➜ Describing error handling of cmd +➜ Assembling module page: internal/config (3 total components) ... -➜ Synthesizing directory: internal/config (3 total components) -➜ LLM Synthesizing chunk for internal/config (3 items) +➜ Assembling module page: . (3 total components) +➜ Describing responsibility of . ... -➜ Synthesizing directory: . (4 total components) -➜ LLM Synthesizing chunk for . (4 items) Step 3: Global Architecture Synthesis... Step 4: Generating Quickstart... Step 5: Updating AGENTS.md... @@ -327,7 +393,7 @@ You can inspect the actual documentation generated by Code-Reducer for this repo * [internal/README.md](wiki/modules/internal/README.md) – Synthesis of core application library packages. * [internal/config/README.md](wiki/modules/internal/config/README.md) – Configuration engine and environment management details. * [internal/engine/README.md](wiki/modules/internal/engine/README.md) – Core Map-Reduce execution pipeline and LLM client logic. - * [internal/security/README.md](wiki/modules/internal/security/README.md) – Path traversal checks and flock-based concurrency controls. + * [internal/security/README.md](wiki/modules/internal/security/README.md) – Path traversal checks and atomic lockfile-based process serialization. * [internal/tools/README.md](wiki/modules/internal/tools/README.md) – Helper utilities for Git integration and directory/binary discovery. --- diff --git a/cmd/cmd_test.go b/cmd/cmd_test.go index 8540042..a7dd231 100644 --- a/cmd/cmd_test.go +++ b/cmd/cmd_test.go @@ -15,8 +15,26 @@ func TestRootFlags(t *testing.T) { // reset flags modelIDFlag = "" numCtxFlag = "" + thinkFlag = "" + numPredictFlag = "" + slotNumPredictFlag = "" + paragraphNumPredictFlag = "" + charsPerTokenFlag = "" + outputTokenReserveFlag = "" + includeTestsFlag = "" - RootCmd.SetArgs([]string{"--model-id", "test-model", "--num-ctx", "4096", "help"}) // run something harmless like help + RootCmd.SetArgs([]string{ + "--model-id", "test-model", + "--num-ctx", "4096", + "--think", "true", + "--num-predict", "2048", + "--slot-num-predict", "96", + "--paragraph-num-predict", "768", + "--chars-per-token", "2.5", + "--output-token-reserve", "512", + "--include-tests", "true", + "help", + }) // run something harmless like help _ = RootCmd.Execute() if modelIDFlag != "test-model" { @@ -26,6 +44,34 @@ func TestRootFlags(t *testing.T) { if numCtxFlag != "4096" { t.Errorf("Expected numCtxFlag to be '4096', got '%s'", numCtxFlag) } + + if thinkFlag != "true" { + t.Errorf("Expected thinkFlag to be 'true', got '%s'", thinkFlag) + } + + if numPredictFlag != "2048" { + t.Errorf("Expected numPredictFlag to be '2048', got '%s'", numPredictFlag) + } + + if slotNumPredictFlag != "96" { + t.Errorf("Expected slotNumPredictFlag to be '96', got '%s'", slotNumPredictFlag) + } + + if paragraphNumPredictFlag != "768" { + t.Errorf("Expected paragraphNumPredictFlag to be '768', got '%s'", paragraphNumPredictFlag) + } + + if charsPerTokenFlag != "2.5" { + t.Errorf("Expected charsPerTokenFlag to be '2.5', got '%s'", charsPerTokenFlag) + } + + if outputTokenReserveFlag != "512" { + t.Errorf("Expected outputTokenReserveFlag to be '512', got '%s'", outputTokenReserveFlag) + } + + if includeTestsFlag != "true" { + t.Errorf("Expected includeTestsFlag to be 'true', got '%s'", includeTestsFlag) + } } func TestCheckInitStatus(t *testing.T) { @@ -169,10 +215,15 @@ func TestLoadInitialSetupConfig(t *testing.T) { if cfg.ModelID == "" || cfg.OllamaNumCtx <= 0 { t.Fatalf("expected non-empty defaults, got %+v", cfg) } + if cfg.SlotNumPredict != config.SlotNumPredictDefault || cfg.ParagraphNumPredict != config.ParagraphNumPredictDefault { + t.Fatalf("expected the shipped slot bounds as defaults, got %+v", cfg) + } testCfg := &config.Config{ - ModelID: "custom-model", - OllamaNumCtx: 8192, + ModelID: "custom-model", + OllamaNumCtx: 8192, + SlotNumPredict: 96, + ParagraphNumPredict: 768, } if err := config.SaveConfig(tmp, testCfg); err != nil { t.Fatalf("failed to save config: %v", err) @@ -181,6 +232,9 @@ func TestLoadInitialSetupConfig(t *testing.T) { if loaded.ModelID != "custom-model" || loaded.OllamaNumCtx != 8192 { t.Fatalf("expected custom values, got %+v", loaded) } + if loaded.SlotNumPredict != 96 || loaded.ParagraphNumPredict != 768 { + t.Fatalf("expected the custom slot bounds to survive, got %+v", loaded) + } } func TestRunSetupFlow(t *testing.T) { @@ -200,3 +254,65 @@ func TestRunSetupFlow(t *testing.T) { t.Errorf("expected default model ID, got %s", cfg.ModelID) } } + +func TestRunSetupFlowKeepsIncludeTests(t *testing.T) { + tmp := t.TempDir() + if err := config.SaveConfig(tmp, &config.Config{ModelID: "custom-model", IncludeTests: true}); err != nil { + t.Fatalf("failed to save config: %v", err) + } + + reader := bufio.NewReader(bytes.NewBufferString("\n\n\n\n\n")) + if err := runSetupFlowWithReader(reader, tmp); err != nil { + t.Fatalf("unexpected error running setup flow: %v", err) + } + + cfg, err := config.LoadConfig(tmp) + if err != nil { + t.Fatalf("expected config to be saved: %v", err) + } + if !cfg.IncludeTests { + t.Error("expected the setup wizard to keep include_tests from the existing config") + } +} + +func TestRunSetupFlowKeepsSlotBounds(t *testing.T) { + tmp := t.TempDir() + if err := config.SaveConfig(tmp, &config.Config{ModelID: "custom-model", SlotNumPredict: 96, ParagraphNumPredict: 768}); err != nil { + t.Fatalf("failed to save config: %v", err) + } + + reader := bufio.NewReader(bytes.NewBufferString("\n\n\n\n\n")) + if err := runSetupFlowWithReader(reader, tmp); err != nil { + t.Fatalf("unexpected error running setup flow: %v", err) + } + + cfg, err := config.LoadConfig(tmp) + if err != nil { + t.Fatalf("expected config to be saved: %v", err) + } + if cfg.SlotNumPredict != 96 { + t.Errorf("expected the setup wizard to keep slot_num_predict, got %d", cfg.SlotNumPredict) + } + if cfg.ParagraphNumPredict != 768 { + t.Errorf("expected the setup wizard to keep paragraph_num_predict, got %d", cfg.ParagraphNumPredict) + } +} + +func TestRunSetupFlowWritesShippedSlotBounds(t *testing.T) { + tmp := t.TempDir() + reader := bufio.NewReader(bytes.NewBufferString("\n\n\n\n\n")) + if err := runSetupFlowWithReader(reader, tmp); err != nil { + t.Fatalf("unexpected error running setup flow: %v", err) + } + + cfg, err := config.LoadConfig(tmp) + if err != nil { + t.Fatalf("expected config to be saved: %v", err) + } + if cfg.SlotNumPredict != config.SlotNumPredictDefault { + t.Errorf("expected slot_num_predict %d, got %d", config.SlotNumPredictDefault, cfg.SlotNumPredict) + } + if cfg.ParagraphNumPredict != config.ParagraphNumPredictDefault { + t.Errorf("expected paragraph_num_predict %d, got %d", config.ParagraphNumPredictDefault, cfg.ParagraphNumPredict) + } +} diff --git a/cmd/root.go b/cmd/root.go index fafca78..3c031a5 100644 --- a/cmd/root.go +++ b/cmd/root.go @@ -16,8 +16,15 @@ import ( ) var ( - modelIDFlag string - numCtxFlag string + modelIDFlag string + numCtxFlag string + thinkFlag string + numPredictFlag string + slotNumPredictFlag string + paragraphNumPredictFlag string + charsPerTokenFlag string + outputTokenReserveFlag string + includeTestsFlag string ) var RootCmd = &cobra.Command{ @@ -32,6 +39,13 @@ var RootCmd = &cobra.Command{ func init() { RootCmd.PersistentFlags().StringVar(&modelIDFlag, "model-id", "", "Specify LLM model ID") RootCmd.PersistentFlags().StringVar(&numCtxFlag, "num-ctx", "", "Specify Ollama context window size") + RootCmd.PersistentFlags().StringVar(&thinkFlag, "think", "", "Enable reasoning output on reasoning models (default false)") + RootCmd.PersistentFlags().StringVar(&numPredictFlag, "num-predict", "", "Cap the number of generated tokens per call (default 0, Ollama decides)") + RootCmd.PersistentFlags().StringVar(&slotNumPredictFlag, "slot-num-predict", "", "Cap the generated tokens for each one-line documentation slot (default 192)") + RootCmd.PersistentFlags().StringVar(¶graphNumPredictFlag, "paragraph-num-predict", "", "Cap the generated tokens for each documentation paragraph slot (default 1024)") + RootCmd.PersistentFlags().StringVar(&charsPerTokenFlag, "chars-per-token", "", "Characters per token used to size prompt payloads (default 3.0)") + RootCmd.PersistentFlags().StringVar(&outputTokenReserveFlag, "output-token-reserve", "", "Context tokens held back from prompt payloads for generation (default 1024)") + RootCmd.PersistentFlags().StringVar(&includeTestsFlag, "include-tests", "", "Document test files as well (default false)") } func executeCommand(mode engine.Mode) error { @@ -48,7 +62,17 @@ func executeCommand(mode engine.Mode) error { return err } - cfg, err := config.ResolveConfig(repoRoot, modelIDFlag, numCtxFlag) + cfg, err := config.ResolveConfig(repoRoot, config.Flags{ + ModelID: modelIDFlag, + NumCtx: numCtxFlag, + Think: thinkFlag, + NumPredict: numPredictFlag, + SlotNumPredict: slotNumPredictFlag, + ParagraphNumPredict: paragraphNumPredictFlag, + CharsPerToken: charsPerTokenFlag, + OutputTokenReserve: outputTokenReserveFlag, + IncludeTests: includeTestsFlag, + }) if err != nil { return err } diff --git a/cmd/setup.go b/cmd/setup.go index eb8a4db..cda50b1 100644 --- a/cmd/setup.go +++ b/cmd/setup.go @@ -38,6 +38,13 @@ func loadInitialSetupConfig(repoRoot string) *config.Config { OllamaBaseURL: config.OllamaDefaultBaseURL, OllamaNumCtx: config.OllamaDefaultNumCtx, DocsDir: config.DefaultDocsDir, + Think: config.ThinkDefault, + NumPredict: config.NumPredictDefault, + SlotNumPredict: config.SlotNumPredictDefault, + ParagraphNumPredict: config.ParagraphNumPredictDefault, + CharsPerToken: config.CharsPerTokenDefault, + OutputTokenReserve: config.OutputTokenReserveDefault, + IncludeTests: config.IncludeTestsDefault, ExtractionSteps: config.DefaultExtractionSteps, SystemPrompt: config.DefaultSystemPrompt, ModuleSynthesisPrompt: config.DefaultModuleSynthesisPrompt, @@ -61,6 +68,18 @@ func loadInitialSetupConfig(repoRoot string) *config.Config { if len(cfg.ExtractionSteps) == 0 { cfg.ExtractionSteps = config.DefaultExtractionSteps } + if cfg.SlotNumPredict <= 0 { + cfg.SlotNumPredict = config.SlotNumPredictDefault + } + if cfg.ParagraphNumPredict <= 0 { + cfg.ParagraphNumPredict = config.ParagraphNumPredictDefault + } + if cfg.CharsPerToken <= 0 { + cfg.CharsPerToken = config.CharsPerTokenDefault + } + if cfg.OutputTokenReserve <= 0 { + cfg.OutputTokenReserve = config.OutputTokenReserveDefault + } if cfg.SystemPrompt == "" { cfg.SystemPrompt = config.DefaultSystemPrompt } @@ -144,6 +163,13 @@ func runSetupFlowWithReader(reader *bufio.Reader, repoRoot string) error { OllamaBaseURL: urlInput, OllamaNumCtx: numCtx, DocsDir: docsDirInput, + Think: current.Think, + NumPredict: current.NumPredict, + SlotNumPredict: current.SlotNumPredict, + ParagraphNumPredict: current.ParagraphNumPredict, + CharsPerToken: current.CharsPerToken, + OutputTokenReserve: current.OutputTokenReserve, + IncludeTests: current.IncludeTests, ExtractionSteps: current.ExtractionSteps, Ignore: ignores, SystemPrompt: current.SystemPrompt, diff --git a/docs/architecture.md b/docs/architecture.md index 639d2b8..3d77e4a 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -11,36 +11,35 @@ By structuring codebase extraction into discrete, context-budgeted stages, Code- ```mermaid flowchart TD subgraph Discovery ["1. Discovery & Tree Construction"] - A[Repository File Scanner] -->|DiscoverCodeFiles| B[Ignore Filters & Gitignore] + A[Repository File Scanner] -->|DiscoverCodeFiles| B[Ignore Filters, Gitignore & Test Files] B -->|Allowed Code Files| C[Build Node Prefix Tree DirNode] end subgraph MapPhase ["2. Map Phase (File Extraction)"] C --> D{Cache Hit in metadata.json?} D -- Yes --> E[Reuse Cached File Facts] - D -- No --> F[Calculate Dynamic Truncation Budget 75% NumCtx] + D -- No --> F[Budget File Chunks Context Minus Output Reserve] F --> G[chunkTextWithOverlap maxRunes, 800-char overlap] G --> H[Multi-Step Extraction Pipeline] H --> I[reduceFileFacts Consolidation] I --> J[Save File Facts to Cache] end - subgraph ReducePhase ["3. Reduce Phase (Hierarchical Consolidation)"] + subgraph ReducePhase ["3. Reduce Phase (Deterministic Skeleton & Prose Slots)"] J & E --> K[Gather Directory Components Files & Subsystems] - K --> L[reduceItems Batching Budget NumCtx * 3] - L --> M{Batch Size > MaxChars?} - M -- Yes --> N[Split into Sub-Batches & Dynamic Sub-Layers] - M -- No --> O[Call LLM with ModuleSynthesisPrompt] - N --> P{Output >= 95% Input?} - P -- Yes Loop Guard --> Q[Concatenate Intermediate Summaries] - P -- No --> L - O & Q --> R[Write wiki/modules/safeName.md] + K --> L[Render Fixed Skeleton Title & Headings & Component Order] + L --> M[Fill Responsibility Slot 1 Bounded LLM Call] + M --> N[Fill One Line Slot Per Component] + N --> O[Fill Data Flow & Error Handling Slots] + O --> P{Any Slot Failed?} + P -- Yes --> Q[Abort: Nothing Written, Nothing Cached] + P -- No --> R[Write wiki/modules/safeName/README.md] end subgraph SynthesisPhase ["4. Global Synthesis & AI Guidelines"] - R --> S[Root Node Reduction Summary] - S --> T[Generate wiki/architecture.md System Blueprint] - S --> U[Generate wiki/quickstart.md Developer Guide] + R --> S[Root Node Module Page] + S --> T[Slot-Fill wiki/architecture.md System Blueprint] + T --> U[Slot-Fill wiki/quickstart.md Developer Guide] S --> V[Update / Append Guidelines to AGENTS.md] end ``` @@ -75,12 +74,12 @@ In the Map Phase, Code-Reducer iterates over source files within affected direct +-------------------------------------------------------------------------+ | Source File | +-------------------------------------------------------------------------+ - | - v - +-----------------------------------------------+ - | Dynamic Truncation Budget (75% of NumCtx) | - | Overlap Margin: 800 characters | - +-----------------------------------------------+ + | + v + +-----------------------------------------------+ + | Output Reserve + Chars-Per-Token Budget | + | Overlap Margin: 800 characters | + +-----------------------------------------------+ | +-----------------------+-----------------------+ | | @@ -115,14 +114,13 @@ In the Map Phase, Code-Reducer iterates over source files within affected direct To prevent context overflow, file chunking limits are dynamically calculated based on the LLM's context size `NumCtx`: ```go -numCtx := p.client.NumCtx() -if numCtx < minNumCtxFloor { - numCtx = minNumCtxFloor -} -fileLimit := int(float64(numCtx * 4) * 0.75) // 75% of token context in characters +// calculateFileLimit: the prompt budget is the context minus the output reserve, +// converted into characters with the measured characters-per-token ratio. +fileLimit := promptCharBudget(c.NumCtx(), cfg.OutputTokenReserve, cfg.CharsPerToken) ``` -- **75% Budget Allocation (`contextWindowAllocRatio = 0.75`)**: Reserves 75% of `NumCtx` (assuming ~4 characters per token) for raw source content, leaving 25% reserved for system instructions, extraction prompts, and completion tokens. +- **Output Reserve (`output_token_reserve`, default `1024` tokens)**: Held back from every prompt payload so the model always has room to generate, and reduced automatically if it is not smaller than the context. +- **Characters Per Token (`chars_per_token`, default `3.0`)**: The measured ratio for Go and Markdown payloads. With the defaults, a 20,000 token window gives a file payload of `(20000 - 1024) * 3.0` = 56,928 characters. - **800-Character Overlap Margin (`defaultChunkOverlap = 800`)**: Overlaps contiguous chunks by 800 characters to prevent boundary context loss (e.g. split function definitions or multiline comments). ### Multi-Step Extraction Pipeline @@ -134,43 +132,126 @@ Each chunk is passed sequentially through configured `extraction_steps`: Multi-chunk extractions for a single file are merged into a unified briefing via `reduceFileFacts`. +### Cache-Friendly Message Layout +Ollama reuses the cached tokens of an unchanged prompt prefix and reports the reuse as `prompt_eval_cached_count`. Concatenating the per-step instruction onto `system_prompt` (`system = SystemPrompt + step.Prompt`) made every one of the four extraction steps a different system message, which broke the prefix on every call and drove the hit rate to 0%. + +The system message is therefore byte-identical on **every** call of a run: it is always `system_prompt` and nothing else. Everything that varies per call - the extraction step instruction, the consolidation instruction, the prose style of a page kind, and the slot instruction with its context - is assembled into the user turn by `userMessage`: + +```go +user := userMessage( + fmt.Sprintf("File: %s inside Module: %s\n```\n%s\n```", name, nodePath, chunk), + step.Prompt, // per-call instruction, last +) +res, err := p.client.CallLLM(ctx, p.cfg.SystemPrompt, []Message{{Role: "user", Content: user}}, false) +``` + +`userMessage` keeps the run-invariant parts first and the per-call text last, so the shared prefix of the user turn survives as well. Over the four extraction calls this restores a 12.5% hit rate, and the same layout is used for the consolidation, module-slot and standard-page calls. + --- -## ⚡ The Reduce Phase (Hierarchical Consolidation) +## ⚡ The Reduce Phase (Deterministic Skeleton & Prose Slots) -Once file-level facts are extracted, Code-Reducer consolidates directory components (file briefings and child subsystem summaries) into unified directory module documentation (`wiki/modules/.md`). +Once file-level facts are extracted, Code-Reducer turns the directory components (file briefings and child subsystem summaries) into module documentation (`wiki/modules//README.md`). -### Batch Consolidation Budget (`NumCtx * 3`) -During reduction, batch character sizes are capped using `maxCharsMultiplier`: +The LLM is not asked to write a page. Measurements on a 9B model showed that a single free-form "write the module page" instruction complies with the requested headings only about half of the time, and that tightening the wording makes compliance *worse*, while a single-slot micro-task with a bounded one-line answer complies every time at a fraction of the tokens. Factual retention was unaffected in every variant: the model was never losing facts, it was failing at structure. -$$\text{MaxChars} = \text{NumCtx} \times 3$$ +So the structure is the engine's job. `renderModulePage` writes the title, the section headings, their order, the component names and the component order verbatim, and each LLM call fills exactly one prose slot: -If total component characters exceed $\text{MaxChars}$, components are grouped into smaller sub-batches and reduced independently. +```markdown +# Module: internal/engine -### Dynamic Multi-Layer Map-Reduce (`reduceItems`) -When an individual item or sub-batch exceeds $\text{MaxChars}$, `reduceItems` automatically breaks the item into smaller chunks (`maxChars / 2` with `maxChars / 10` overlap) to introduce an additional reduction layer: +## Responsibility + -```go -for _, item := range items { - if utf8.RuneCountInString(item) > maxChars { - chunks, err := chunkTextWithOverlap(item, maxChars/2, maxChars/10) - expanded = append(expanded, chunks...) - } else { - expanded = append(expanded, item) - } -} -``` +## Components -### Recursion Loop Prevention Guard -If an LLM fails to condense input text effectively (e.g., producing verbose output equal to or larger than input), naive recursive reduction would loop infinitely. Code-Reducer guards against this by measuring payload shrinkage: +### File: client.go + -```go -if totalOutputRunes >= (totalInputRunes * 95 / 100) { - return strings.Join(intermediate, "\n\n"), nil -} +### Subsystem: config + + +## Data Flow + + +## Error Handling + ``` -If total output character length is $\ge 95\%$ of input length, recursion halts and intermediate summaries are concatenated directly. Subsequent parent reductions safely handle chunking without data loss. +### Minimal Context Per Slot +Slots receive only the context they need, so a module with many components never re-sends its whole subtree to every call: + +| Slot | Kind | Context | +| :--- | :--- | :--- | +| Responsibility | one line | component identities only | +| Component line | one line | that single component's own facts | +| Data Flow | paragraph | component identities plus a bounded per-component facts digest | +| Error Handling | paragraph | component identities plus a bounded per-component facts digest | + +`componentFactsDigest` splits `promptCharBudget` evenly across the components, and `truncateRunes` marks every cut so the model never reads a partial payload as the complete picture. + +### Parents Read Their Children as Identities +A child subsystem reaches its parent as its rendered page, and a rendered page contains +a `## Data Flow` and an `## Error Handling` paragraph of its own. Feeding those to the +parent is what makes a parent page restate its children, so `childIdentityFacts` keeps +only the head of a child page (title, responsibility, component entries) and +`readComponents` uses it for every slot of the parent. The parent is then asked the +question only a parent can answer: + +| Node shape | `## Data Flow` | `## Error Handling` | +| :--- | :--- | :--- | +| Has subsystem children | What flows between the components and what each hands to the next. | How errors cross the boundaries between the components and reach the caller. | +| Leaf (files only) | How data moves through the module. | How the module reports and handles errors. | + +The framing is selected in Go from the node shape (`moduleShape`). It is not a config +knob, and a leaf module reads exactly the components it always did. + +### One Deterministic Line Per Line Slot +A model asked for a single line still returns a bullet list, a heading, a fenced block +or an enumeration of every function in a file, so the shape of a line slot is decided +in Go rather than trusted to the model: + +1. The first line that is not a fence or a heading is taken. A bulleted line keeps the + item text, any other line is cut at its first sentence. +2. The result is bounded to 40 words, cut on the last sentence or clause boundary + before that, and on a word boundary when the text offers none, so a word is never + split. +3. Every remaining line is dropped. The heading and the structure around the line are + already fixed by the page, so the tail is duplication. + +### Hard Generation Bound Per Slot, Selected By Kind +Every slot call passes its own `num_predict` through `CallLLMWithNumPredict`, and the +bound is selected from the slot kind at call time. A one-line slot uses +`slot_num_predict` (default `192`): its answer is trimmed by the code to the first +sentence and to 40 words, so the bound only has to cover that first sentence, and +with `think: false` a measured one-line answer costs 34 tokens. A paragraph slot uses +`paragraph_num_predict` (default `1024`), because its answer is kept whole: with +`think: false` a measured paragraph costs 63 tokens. Both defaults are sized for a +non-thinking answer, which is what makes any sane slot budget possible at all: +`ornith:9b` is a reasoning model, and with `think: true` the same one-line slot costs +400 tokens and the same paragraph slot 795, almost all of it reasoning the tool never +reads and every token of it charged against the slot bound. Both bounds exist to stop a +chatty model from drifting back to the ~900 tokens a full-page generation costs. A user +who lowers the global `num_predict` below either bound wins, since `slotNumPredict` takes +the smaller of the two and never lets a global cap raise a per-slot one. + +### Incomplete Output Never Reaches a Page +`requireCompleteContent` rejects any generation that stopped on `done_reason: "length"` +and any empty content, and every extraction call goes through it. A partial fact list +is misleading because it is the source of truth for the page it feeds, so it is a hard +failure. + +Prose slots salvage a clipped answer before falling back to that hard failure, in the +slot layer only. A line slot keeps its first complete sentence and a paragraph slot +keeps the text up to its last complete sentence, because both prefixes are well-formed +instances of the contract the slot asked for; the call was bounded, so the salvage +costs no more tokens than the answer it replaces. An answer with no complete sentence +falls through to the hard failure, as does any other caller of +`requireCompleteContent`. + +A page is all-or-nothing otherwise: a failing slot aborts it, so nothing is written to +disk and nothing is added to the module cache, and a partial page can never be +persisted or served from cache on the next run. --- @@ -179,9 +260,9 @@ If total output character length is $\ge 95\%$ of input length, recursion halts After the root directory node (`.`) is synthesized, the orchestrator invokes `GenerateStandardDocs` to produce top-level project documentation: 1. **System Blueprint (`wiki/architecture.md`)**: - Sent to the LLM with `architecture_prompt` to generate global system architecture, component interaction boundaries, and subsystem relationships. + `buildStandardDocPage` renders the fixed `Overview` / `System Boundaries` / `Module Interaction` skeleton and fills each section with one bounded call styled by `architecture_prompt`. The source of every section is the root module page produced by the reduce phase. 2. **Developer Quickstart (`wiki/quickstart.md`)**: - Generates onboarding workflows, environment setup, and development patterns based on root summary insights. + The same treatment over a fixed `What This Project Does` / `Project Layout` / `Common Workflows` skeleton, so onboarding workflows, layout and development patterns each come from a single micro-task. Its source is the **architecture page that was just written**, not the root module summary: the two pages read different input, and each one is labelled with the source it was given (`Root module page:` versus `Architecture page:`). 3. **AI Agent Guidelines (`AGENTS.md`)**: Creates or appends structured guidelines in `AGENTS.md` at the repository root: diff --git a/docs/cli-wizard.md b/docs/cli-wizard.md index 066e132..113ebce 100644 --- a/docs/cli-wizard.md +++ b/docs/cli-wizard.md @@ -30,7 +30,11 @@ code-reducer setup Generates the initial repository documentation set and metadata cache. ```bash -code-reducer init [--model-id ] [--num-ctx ] +code-reducer init [--model-id ] [--num-ctx ] [--think ] \ + [--num-predict ] [--slot-num-predict ] \ + [--paragraph-num-predict ] \ + [--chars-per-token ] [--output-token-reserve ] \ + [--include-tests ] ``` - **Purpose**: Performs full codebase discovery, builds the hierarchical tree, extracts facts across files, and produces directory module summaries alongside root blueprints. @@ -56,6 +60,7 @@ code-reducer update [--model-id ] [--num-ctx ] - **Bottom-Up Propagation**: When a file changes, its parent directory module is rebuilt, propagating up the tree to refresh affected parent summaries. - **Cache Invalidation**: Detects changes in `.code-reducer.yaml` extraction steps and automatically invalidates outdated file facts. - **Constraints**: Fails fast if the project has not been initialized yet (`code-reducer init` required). +- **Failure Modes**: A generation that stopped on the generation limit is kept only up to its complete sentences: a one-line slot keeps its first complete sentence and a paragraph slot keeps the text up to its last complete sentence. The run aborts only when a slot answer holds no complete sentence, when a fact extraction call is truncated, or when an answer has no visible content, and then it aborts with the failing call and the token counts; nothing is written and nothing is cached, so re-running resumes from the last completed state. The truncation message names `num_predict`, `slot_num_predict`, `paragraph_num_predict` and `think: false`, because a reasoning model with thinking enabled charges its hidden reasoning against the same slot budget (400 tokens for a one-line slot instead of 34). --- @@ -82,7 +87,7 @@ Configuration successfully saved to local .code-reducer.yaml file. 4. **Ignore Patterns**: Comma-separated list of files or directories to ignore. Passing `clear` or `none` empties existing custom ignore lists. 5. **Documentation Directory**: Output directory for generated docs (Default: `wiki`). -If `.code-reducer.yaml` already exists when `setup` is executed, the wizard populates every prompt default using values from the existing YAML file. +If `.code-reducer.yaml` already exists when `setup` is executed, the wizard populates every prompt default using values from the existing YAML file, and carries every key it does not ask about (the prompts, the extraction steps, the generation budgets and `include_tests`) through to the saved file unchanged. --- @@ -123,6 +128,16 @@ Code-Reducer accepts persistent flags on `init` and `update`: | :--- | :--- | :--- | :--- | | `--model-id` | `string` | Override LLM model ID | `--model-id gemma4:26b` | | `--num-ctx` | `string` | Override Ollama context window size | `--num-ctx 16384` | +| `--think` | `bool` | Reasoning output. Required to be `false` on a reasoning model such as `ornith:9b` (default `false`) | `--think=false` | +| `--num-predict` | `int` | Client-wide cap on generated tokens per call (default `0`, Ollama decides) | `--num-predict 2000` | +| `--slot-num-predict` | `int` | Cap on generated tokens for each one-line documentation slot (default `192`) | `--slot-num-predict 256` | +| `--paragraph-num-predict` | `int` | Cap on generated tokens for each documentation paragraph slot (default `1024`) | `--paragraph-num-predict 1536` | +| `--chars-per-token` | `float` | Characters per token used to size prompt payloads (default `3.0`) | `--chars-per-token 3.5` | +| `--output-token-reserve` | `int` | Context tokens held back from prompt payloads for generation (default `1024`) | `--output-token-reserve 1536` | +| `--include-tests` | `bool` | Document test files as well (default `false`) | `--include-tests=true` | + +A `--num-predict` lower than either slot cap also caps that slot kind, so one global +ceiling can be lowered without retuning the slot budgets. #### Flag Examples @@ -132,6 +147,9 @@ code-reducer init --model-id gemma4:26b --num-ctx 16384 # Run incremental update overriding only the model code-reducer update --model-id ornith:9b + +# Document a repository that contains its own tests +code-reducer init --include-tests=true ``` ### Environment Variable Overrides @@ -143,6 +161,11 @@ Environment variables override values in `.code-reducer.yaml` but yield to expli | `CODE_REDUCER_MODEL_ID` | `model_id` | Overrides the LLM model ID. | | `OLLAMA_BASE_URL` | `ollama_base_url` | Overrides the Ollama API server URL. | | `OLLAMA_NUM_CTX` | `ollama_num_ctx` | Overrides the Ollama context window size. | +| `CODE_REDUCER_THINK` | `think` | Enables or disables reasoning output. | +| `OLLAMA_NUM_PREDICT` | `num_predict` | Overrides the client-wide generation cap. | +| `CODE_REDUCER_SLOT_NUM_PREDICT` | `slot_num_predict` | Overrides the one-line documentation slot generation cap. | +| `CODE_REDUCER_PARAGRAPH_NUM_PREDICT` | `paragraph_num_predict` | Overrides the documentation paragraph slot generation cap. | +| `CODE_REDUCER_INCLUDE_TESTS` | `include_tests` | Includes or excludes test files. | #### Environment Variable Examples diff --git a/docs/configuration.md b/docs/configuration.md index f72c0b2..35d48cb 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -14,9 +14,16 @@ Below is the complete list of configurable properties supported by `.code-reduce | `ollama_base_url` | `string` | `http://localhost:11434` | Endpoint URL of the local or remote Ollama server. | | `ollama_num_ctx` | `integer` | `8192` | Context window size in tokens allocated for inference. | | `docs_dir` | `string` | `wiki` | Output directory where markdown documentation is generated. | +| `think` | `boolean` | `false` | Ask the model for reasoning output. Must stay `false` for a reasoning model such as `ornith:9b`. See [Reasoning Models](#reasoning-models-think). | +| `num_predict` | `integer` | `0` | Client-wide generation cap per LLM call. `0` lets Ollama decide. A lower value than a per-slot cap also caps that slot kind. | +| `slot_num_predict` | `integer` | `192` | Generation cap for each one-line documentation slot. See [Generation Bounds](#generation-bounds-and-hard-failures). | +| `paragraph_num_predict` | `integer` | `1024` | Generation cap for each documentation paragraph slot. See [Generation Bounds](#generation-bounds-and-hard-failures). | +| `chars_per_token` | `float` | `3.0` | Characters per token, used to convert the context budget into a prompt payload budget. | +| `output_token_reserve` | `integer` | `1024` | Context tokens held back from every prompt payload so generation always has room. | +| `include_tests` | `boolean` | `false` | Document test files as well. See [Test File Exclusion](#test-file-exclusion-include_tests). | | `system_prompt` | `string` | See defaults | System-level prompt injected into all LLM inference calls. | -| `module_synthesis_prompt` | `string` | See defaults | Prompt applied during directory-level reduction. | -| `architecture_prompt` | `string` | See defaults | Prompt applied for global system architecture synthesis. | +| `module_synthesis_prompt` | `string` | See defaults | Prose style applied when filling a module page slot. The page skeleton is written by the engine, so this must never ask for headings or a whole page. | +| `architecture_prompt` | `string` | See defaults | Prose style applied when filling an `architecture.md` or `quickstart.md` section. | | `file_fact_consolidation_prompt` | `string` | See defaults | Prompt applied when consolidating facts across file chunks. | | `extraction_steps` | `array` | 4 default steps | Array of extraction phases executed during the Map stage. | | `ignore` | `array` | `[]` | List of file, folder, or glob patterns ignored during scanning. | @@ -30,6 +37,28 @@ ollama_base_url: http://localhost:11434 ollama_num_ctx: 20000 docs_dir: wiki +# Reasoning output. Required to be false for a reasoning model such as ornith:9b, +# which otherwise charges its hidden thinking against every generation budget: the +# same slot then costs 400 tokens instead of 34, and can return no content at all. +think: false + +# Client-wide generation cap per call. 0 lets Ollama decide; a positive value also +# caps a documentation slot when it is lower than that slot kind's own cap. +num_predict: 0 + +# Generation caps for the documentation prose slots. The engine writes the page +# structure, so a line slot is trimmed to its first sentence by the code and only +# needs room for that sentence, while a paragraph slot keeps everything it wrote. +slot_num_predict: 192 +paragraph_num_predict: 1024 + +# Prompt payload budgeting +chars_per_token: 3.0 +output_token_reserve: 1024 + +# Document test files as well (default: false) +include_tests: false + # Global System Persona & Safety Directives system_prompt: | You are Code-Reducer, an expert technical writer and code analyzer. Your job is to strictly follow instructions. You do not yap, you do not write filler. @@ -39,18 +68,19 @@ system_prompt: | 3. Do NOT name standard library packages unless explicitly stated in the source text. 4. Only report facts you are 100% sure about. -# Directory Module Synthesis Prompt +# Directory Module Prose Style (applied to every slot of a module page) module_synthesis_prompt: |- - Task: Write a technical documentation page for a code module based on the provided list of its internal components. - Rule 1: Group related functions and classes under appropriate Markdown headings. - Rule 2: Explain the responsibility of the module and the data flow. - Rule 3: Keep it highly technical and dense. + Task: Write the prose for one slot of a module documentation page. + Rule 1: The headings, their order and the component names are already fixed by the tooling. Never emit headings, lists, code fences or any other structure. + Rule 2: Answer with the requested slot text only, dense and technical, and keep it as short as the slot instruction asks. + Rule 3: Use only the supplied facts, and say nothing about code you were not shown. -# System Architecture & Quickstart Prompt +# System Architecture & Quickstart Prose Style (applied to every page section) architecture_prompt: |- - Task: Write a global architecture or quickstart document based on the module summaries. - Rule 1: Explain the system boundaries and how the modules interact. - Rule 2: Provide a dense, developer-friendly overview. + Task: Write the prose for one slot of a project documentation page (architecture or quickstart). + Rule 1: The title, the headings and their order are already fixed by the tooling. Never emit headings, lists, code fences or any other structure. + Rule 2: Answer with the requested slot text only, dense and developer-friendly, and keep it as short as the slot instruction asks. + Rule 3: Use only the supplied module summaries, and never invent a module, a command or a workflow you were not shown. # Multi-chunk File Fact Consolidation Prompt file_fact_consolidation_prompt: |- @@ -109,28 +139,78 @@ Code-Reducer determines parameter values dynamically using a deterministic four- When resolving options at runtime: -1. **CLI Flags**: Passed explicitly during subcommand invocation (`--model-id`, `--num-ctx`). +1. **CLI Flags**: Passed explicitly during subcommand invocation (`--model-id`, `--num-ctx`, `--think`, `--num-predict`, `--slot-num-predict`, `--paragraph-num-predict`, `--chars-per-token`, `--output-token-reserve`, `--include-tests`). 2. **Environment Variables**: Read from shell context: - `CODE_REDUCER_MODEL_ID` overrides `model_id`. - `OLLAMA_BASE_URL` overrides `ollama_base_url`. - `OLLAMA_NUM_CTX` overrides `ollama_num_ctx`. + - `CODE_REDUCER_THINK` overrides `think`. + - `OLLAMA_NUM_PREDICT` overrides `num_predict`. + - `CODE_REDUCER_SLOT_NUM_PREDICT` overrides `slot_num_predict`. + - `CODE_REDUCER_PARAGRAPH_NUM_PREDICT` overrides `paragraph_num_predict`. + - `CODE_REDUCER_INCLUDE_TESTS` overrides `include_tests`. + - `chars_per_token` and `output_token_reserve` have no environment override. 3. **YAML File**: Values defined in `.code-reducer.yaml`. 4. **System Defaults**: Built-in fallbacks (`ornith:9b`, `http://localhost:11434`, `8192`, `wiki`). --- -## System & Synthesis Prompts +## Generation Bounds and Hard Failures + +Every LLM call is bounded and every incomplete answer is rejected rather than written +to disk, because a clipped answer that looks finished is worse than a failed run. + +| Signal | Meaning | Result | +| :--- | :--- | :--- | +| `done_reason: "length"` on a fact extraction | The fact list is the source of truth for the page it feeds. | Hard failure, nothing cached. | +| `done_reason: "length"` on a line slot | The contract is one line, and the first complete sentence of the answer already satisfies it. | The first complete sentence is used. An answer with no complete sentence is a hard failure. | +| `done_reason: "length"` on a paragraph slot | The contract is a whole paragraph, and the text up to its last complete sentence is one. | The text up to the last complete sentence is used. An answer with no complete sentence is a hard failure. | +| Empty content | The model emitted only reasoning output, or formatting only. | Hard failure. Set `think: false`. | + +So a run aborts only in two cases: a fact extraction call is truncated, or a slot +answer holds no complete sentence to salvage. A verbose model is no longer able to end +a run, because every clipped slot keeps its complete sentences. + +`slot_num_predict` and `paragraph_num_predict` are ceilings, not targets. A +well-behaved slot returns far fewer tokens than its cap allows, so the values cost +nothing while they stay above the length of a real answer. They are split by slot kind +because the two need different amounts of room: the code trims a line slot to its first +sentence and to 40 words, so tokens past that sentence are pure waste, while a +paragraph slot keeps every sentence the model wrote. Measured with `think: false`, a +one-line slot costs 34 tokens and a paragraph slot 63, which is what the defaults of +`192` and `1024` leave an order of magnitude of headroom over. If you lower a cap enough +to clip a slot answer, the run salvages what is complete and reports only the case where +nothing usable is left, with a message naming the slot and the token counts. + +## Reasoning Models (`think`) + +`think` defaults to `false` and **must stay `false` for a reasoning model such as +`ornith:9b`**. With `think: true` the model spends the generation budget on hidden +thinking, which is slower and can leave the content empty, which the pipeline then +rejects. The budget effect is the larger one: the same one-line slot measures 400 tokens +with thinking enabled against 34 without it, and the same paragraph slot 795 against 63. +The reasoning tokens are charged against the same bound as the answer and are never +parsed, so they are pure loss, and the shipped `slot_num_predict` and +`paragraph_num_predict` defaults are sized on the `think: false` measurements. If a slot +is ever reported as truncated, check `think` first. `think: true` is only useful with a +model whose thinking output is worth the tokens. + +--- + +## System & Prose Prompts Code-Reducer allows overriding prompt templates to adapt outputs for specialized domain requirements or alternate languages. +These prompts carry **voice and content style only**. The engine owns the document structure: it renders every title, heading, heading order, component name and component order, and each LLM call fills exactly one prose slot. A prompt override must not ask the model for headings, lists or a whole page, because the engine would have to strip that structure back out again. + ### `system_prompt` Injected into every request sent to Ollama. It defines the core persona and defensive grounding constraints to prevent LLM hallucinations. ### `module_synthesis_prompt` -Executed during the **Reduce phase** when synthesizing folder-level briefings (`wiki/modules//README.md`). Guides how internal file facts are structured into cohesive module documentation. +Applied to every prose slot of a module page (`wiki/modules//README.md`): the `Responsibility` line, one line per component, and the `Data Flow` and `Error Handling` paragraphs. ### `architecture_prompt` -Executed during the final **Global Reduction phase** to produce root-level system summaries (`wiki/architecture.md` and `wiki/quickstart.md`). +Applied to every section of `wiki/architecture.md` and `wiki/quickstart.md`. ### `file_fact_consolidation_prompt` Invoked when a single source file exceeds context boundaries and is split into multiple overlapping chunks during the Map phase. It instructs the LLM to deduplicate and unify facts extracted across chunks. @@ -170,6 +250,29 @@ Default system exclusions (like `.git`) are combined with user-defined patterns --- +## Test File Exclusion (`include_tests`) + +Test files are skipped during discovery unless `include_tests: true` is set. Documenting a test file mostly restates its assertions, while every discovered file costs one extraction pass per entry in `extraction_steps`, so the generated wiki spends its context budget on the architecture a reader actually needs. + +Matching is by file-name convention and language-agnostic, applied to the base name of each discovered file: + +| Convention | Examples | +| :--- | :--- | +| Go | `client_test.go` | +| Python | `client_test.py`, `test_client.py` | +| JavaScript / TypeScript | `client_test.js`, `client.test.js`, `client.spec.js`, `client.test.ts`, `client.spec.ts` | +| Ruby | `client_spec.rb`, `test_client.rb` | +| Rust | `client_test.rs` | +| JVM / .NET | `ClientTest.java`, `ClientTests.cs`, `ClientTests.scala` | + +Matching is case-sensitive, so a source file that merely ends in a similar word (`contest.go`, `latest.java`) is still documented. + +The `ignore` list is unaffected: listing a test pattern such as `**/*_test.go` keeps working and excludes the same files even when `include_tests: true`. + +Toggling the setting is detected by `code-reducer update` like any other file-set change: newly included test files are treated as added, and newly excluded ones as deleted, which prunes their cache entries and regenerates the affected module pages. + +--- + ## Infrastructure-as-Code (IaC) & Terraform Support Code-Reducer is not restricted to standard programming languages. By adjusting `system_prompt` and `extraction_steps`, it can document cloud topologies and Infrastructure-as-Code repositories. @@ -192,14 +295,16 @@ system_prompt: | 3. Only report resource definitions and configurations you are 100% sure about. module_synthesis_prompt: | - Task: Write a technical documentation page for a Terraform module based on the provided list of its internal resources, variables, and sub-modules. - Rule 1: Group related resources (networking, compute, database, IAM) under appropriate Markdown headings. - Rule 2: Explain resource dependencies, data flow, and security controls. + Task: Write the prose for one slot of a Terraform module documentation page. + Rule 1: The headings, their order and the resource names are already fixed by the tooling. Never emit headings, lists, code fences or any other structure. + Rule 2: Cover resource dependencies, data flow and security controls, as densely as the slot instruction allows. + Rule 3: Only report resources and variables that appear in the supplied facts. architecture_prompt: | - Task: Write a global cloud architecture overview and deployment guide based on module summaries. - Rule 1: Explain cloud topology, provider requirements, and inter-module networking. - Rule 2: Provide standard Terraform deployment commands (init, plan, apply). + Task: Write the prose for one slot of a cloud architecture or quickstart page. + Rule 1: The title, the headings and their order are already fixed by the tooling. Never emit headings, lists, code fences or any other structure. + Rule 2: Cover cloud topology, provider requirements, inter-module networking and the standard Terraform deployment commands, as the slot instruction allows. + Rule 3: Never invent a resource, provider or command you were not shown. file_fact_consolidation_prompt: | You are a specialized infrastructure documentation assistant. Consolidate and merge facts extracted across chunks of Terraform files. diff --git a/docs/index.md b/docs/index.md index 23ad205..3a1a886 100644 --- a/docs/index.md +++ b/docs/index.md @@ -27,7 +27,7 @@ Hierarchical Map-Reduce Wiki Generator for Local LLMs

Fully Customizable Prompting

-

Tailor extraction steps, system prompts, module synthesis blueprints, and file fact consolidation rules directly via .code-reducer.yaml.

+

Tailor extraction steps, the system prompt, and the prose style of module and architecture pages directly via .code-reducer.yaml. The engine owns the page structure.

@@ -64,7 +64,7 @@ Explore the documentation guides below to learn more about Code-Reducer's design

Architecture & Map-Reduce Engine

-

Deep dive into node prefix trees (DirNode), dynamic file chunking, sub-batch reduction, and global synthesis phases.

+

Deep dive into node prefix trees (DirNode), context-budgeted file chunking, the deterministic page skeleton with its bounded prose slots, and global synthesis phases.

diff --git a/docs/map-reduce-caching.md b/docs/map-reduce-caching.md index 6ca3446..1a1be4f 100644 --- a/docs/map-reduce-caching.md +++ b/docs/map-reduce-caching.md @@ -37,7 +37,7 @@ sequenceDiagram Engine->>LLM: Run Extract & Reduce File Facts Engine->>Cache: Store SHA256 + Facts end - Engine->>LLM: Synthesize Module README (wiki/modules/.md) + Engine->>LLM: Fill Prose Slots of the Module Page (wiki/modules//README.md) Engine->>Cache: Store Module Summary end alt Root Node (.) Affected OR Global Docs Missing @@ -100,6 +100,8 @@ During `detectFileChanges`, active repository files discovered by `DiscoverCodeF | **`Modified`** | Present in both workspace and cache, but current SHA256 $\neq$ cached SHA256. | Directory marked **affected**. Fact cache invalidated for file; full extraction re-run. | | **`Deleted`** | Present in `cache.Files`, missing from active workspace. | Immediate parent directory marked **affected**. Cache entry pruned from `cache.Files`. | +Toggling `include_tests` moves files in and out of the discovered set, so it is classified like any other membership change: a newly included test file is `Added`, and a newly excluded one is `Deleted`, which prunes its cache entry and rebuilds the module pages that used to document it. A directory that held only test files leaves the tree entirely, and `pruneStaleCache` removes its cached summary and its page from disk. + --- ## 🌲 Bottom-Up Change Propagation @@ -110,7 +112,7 @@ When source files change deep inside a project subfolder, child module changes m A directory node `n` is directly marked **affected** if: - Any file in `n.Files` has a `Modified` or `Added` status. - Any file associated with `n.Path` was `Deleted`. -- The directory's module documentation file (`wiki/modules/.md`) is physically missing on disk. +- The directory's module documentation directory (`wiki/modules//README.md`) is physically missing on disk. - The directory entry `cache.Modules[n.Path]` is missing. ### 2. Recursive Bottom-Up Propagation (`propagateAffected`) diff --git a/docs/performance-vram.md b/docs/performance-vram.md index 3064641..c769167 100644 --- a/docs/performance-vram.md +++ b/docs/performance-vram.md @@ -20,34 +20,46 @@ This low VRAM footprint enables developers to execute high-density Map-Reduce do ## Context Window Budgeting Strategy -To prevent context truncation, out-of-memory (OOM) errors, or prompt degradation during file processing, Code-Reducer implements an automated **75% / 25% Context Budgeting Rule**. +To prevent context truncation, out-of-memory (OOM) errors, or prompt degradation during file processing, Code-Reducer reserves part of the context for generation and budgets the rest in characters. ``` ┌─────────────────────────────────────────────────────────┐ │ Total Context Window (NumCtx) │ ├───────────────────────────────────┬─────────────────────┤ -│ 75% Source File Payload │ 25% Prompt & Output │ -│ (~4 chars / token) │ Reserve │ +│ Source File Payload │ Output Reserve │ +│ (NumCtx - reserve) * chars/token │ (1024 tokens) │ └───────────────────────────────────┴─────────────────────┘ ``` ### 1. Dynamic File Chunking (Map Stage) -For source files that exceed available token budgets: +For source files that exceed the available budget: -1. **Character Ratio Calculation**: Code-Reducer assumes an average of ~4 characters per token. -2. **Payload Limit**: Assigns 75% of `NumCtx` to file payload chunks. For an 8,192 token window, the chunk payload limit is: - $$\text{Limit} = 8192 \times 0.75 \times 4 = 24,576 \text{ characters}$$ -3. **Context Boundary Overlap**: Each chunk includes an **800-character overlap margin** relative to the preceding chunk, preserving variable definitions, scopes, and comment context across chunk boundaries. +1. **Output Reserve**: `output_token_reserve` (default `1024` tokens) is held back from every prompt payload so generation always has room. +2. **Character Ratio**: `chars_per_token` (default `3.0`) is the measured characters-per-token ratio for Go and Markdown payloads. +3. **Payload Limit**: The chunk payload limit is `chars_per_token * (NumCtx - output_token_reserve)`. For an 8,192 token window with the defaults that is `3.0 * (8192 - 1024)` = 21,504 characters. +4. **Context Boundary Overlap**: Each chunk includes an **800-character overlap margin** relative to the preceding chunk, preserving variable definitions, scopes, and comment context across chunk boundaries. -### 2. Batch Consolidation (Reduce Stage) +### 2. Fact Consolidation (Map Stage) -During directory-level reduction: +When one file is split into several chunks, the facts of each extraction step are +consolidated back into one briefing per step: -- Component summaries are merged into dynamic batches capped at `NumCtx * 3` characters. -- If a single batch exceeds the character limit, Code-Reducer splits it into sub-batches and applies multi-layer recursive reduction. +- Items are merged into batches capped at the same character budget. +- If a single batch exceeds the limit, Code-Reducer splits it into sub-batches and applies multi-layer recursive reduction. - To prevent infinite reduction loops (e.g. if an LLM fails to compress input text), the engine tracks input vs output size reduction ratios and gracefully terminates recursion before memory limits are breached. +### 3. Documentation Slots (Reduce Stage) + +A module page is not generated as a payload of prose. The engine writes the headings and +sends one bounded micro-task per prose slot, so the only payload a reduce step ever +assembles is a component facts digest sized to the same budget, and every generation is +capped by the cap of its slot kind: `slot_num_predict` (default `192`) for a one-line +slot and `paragraph_num_predict` (default `1024`) for a paragraph slot. Both defaults +are sized for a non-thinking answer, since a reasoning model charges its hidden +thinking against the same cap: measured with `think: false` a one-line slot costs 34 +tokens and a paragraph slot 63, while with thinking enabled they cost 400 and 795. + --- ## Hardware Recommendations for Consumer GPUs diff --git a/examples/go/.code-reducer.yaml b/examples/go/.code-reducer.yaml index 6b83427..a59b65d 100644 --- a/examples/go/.code-reducer.yaml +++ b/examples/go/.code-reducer.yaml @@ -4,6 +4,23 @@ ollama_base_url: "http://localhost:11434" ollama_num_ctx: 8192 docs_dir: "wiki" +# ornith:9b is a reasoning model. With think: true it charges its hidden +# reasoning against every generation budget, which slows every call and can +# leave a slot answer with no visible content. Keep it false. +think: false + +# Client-wide generation cap. 0 lets Ollama decide, which only bounds the fact +# extraction calls: every documentation slot is capped by its own per-kind cap +# below. A value lower than a per-kind cap caps that cap as well. +num_predict: 0 + +# Generation caps for the documentation prose slots, the shipped defaults. The +# page structure is written by the engine, so these are ceilings a well-behaved +# answer never reaches: a one-line slot measures 34 tokens and a paragraph slot +# 63 with think: false. +slot_num_predict: 192 +paragraph_num_predict: 1024 + # System Prompt defines the AI persona system_prompt: | You are Code-Reducer, an expert technical writer and Go code analyzer. Your job is to strictly follow instructions. You do not yap, you do not write filler. @@ -13,17 +30,19 @@ system_prompt: | 3. Do NOT name standard library packages unless explicitly stated in the source text. 4. Only report facts you are 100% sure about. -# Prompts for synthesized output +# Prose style applied to every slot of a module page. The engine writes the page +# skeleton, so these prompts must never ask the model for structure. module_synthesis_prompt: | - Task: Write a technical documentation page for a code module based on the provided list of its internal components. - Rule 1: Group related functions and classes under appropriate Markdown headings. - Rule 2: Explain the responsibility of the module and the data flow. - Rule 3: Keep it highly technical and dense. + Task: Write the prose for one slot of a module documentation page. + Rule 1: The headings, their order and the component names are already fixed by the tooling. Never emit headings, lists, code fences or any other structure. + Rule 2: Answer with the requested slot text only, dense and technical, and keep it as short as the slot instruction asks. + Rule 3: Use only the supplied facts, and say nothing about code you were not shown. architecture_prompt: | - Task: Write a global architecture or quickstart document based on the module summaries. - Rule 1: Explain the system boundaries and how the modules interact. - Rule 2: Provide a dense, developer-friendly overview. + Task: Write the prose for one slot of a project documentation page (architecture or quickstart). + Rule 1: The title, the headings and their order are already fixed by the tooling. Never emit headings, lists, code fences or any other structure. + Rule 2: Answer with the requested slot text only, dense and developer-friendly, and keep it as short as the slot instruction asks. + Rule 3: Use only the supplied module summaries, and never invent a module, a command or a workflow you were not shown. file_fact_consolidation_prompt: | You are a specialized code documentation assistant. Consolidate, deduplicate and merge the following facts extracted from different chunks of the same file into a single, cohesive summary. diff --git a/examples/terraform/.code-reducer.yaml b/examples/terraform/.code-reducer.yaml index 6b4e981..03e79df 100644 --- a/examples/terraform/.code-reducer.yaml +++ b/examples/terraform/.code-reducer.yaml @@ -4,6 +4,23 @@ ollama_base_url: "http://localhost:11434" ollama_num_ctx: 8192 docs_dir: "wiki" +# Reasoning output. Reasoning models such as ornith:9b charge their hidden +# thinking against every generation budget when this is true, which slows every +# call and can leave a slot answer with no visible content. Keep it false. +think: false + +# Client-wide generation cap. 0 lets Ollama decide, which only bounds the fact +# extraction calls: every documentation slot is capped by its own per-kind cap +# below. A value lower than a per-kind cap caps that cap as well. +num_predict: 0 + +# Generation caps for the documentation prose slots, the shipped defaults. The +# page structure is written by the engine, so these are ceilings a well-behaved +# answer never reaches: a one-line slot measures 34 tokens and a paragraph slot +# 63 with think: false. +slot_num_predict: 192 +paragraph_num_predict: 1024 + # Tell the LLM to behave as a Cloud Architect and IaC analyzer system_prompt: | You are Code-Reducer, an expert Cloud Architect and Terraform infrastructure analyzer. Your job is to strictly follow instructions. You do not yap, you do not write filler. @@ -12,18 +29,21 @@ system_prompt: | 2. Do NOT guess downstream resource side effects unless directly linked (e.g., via security group attachments or subnet associations). 3. Only report resource definitions and configurations you are 100% sure about. -# Synthesis prompt for directory modules (which represent infrastructure sub-components) +# Prose style for every slot of a module page (which represents an infrastructure +# sub-component). The engine owns the headings, their order and the component names, +# so this prompt must never ask the model for structure. module_synthesis_prompt: | - Task: Write a technical documentation page for a Terraform module based on the provided list of its internal resources, variables, and sub-modules. - Rule 1: Group related resources (e.g., networking, compute, database, IAM) under appropriate Markdown headings. - Rule 2: Explain the primary responsibility of this module, the resource dependency flow, and security configurations. - Rule 3: Keep it highly technical, listing key resources, inputs, and outputs. + Task: Write the prose for one slot of a Terraform module documentation page. + Rule 1: The headings, their order and the resource names are already fixed by the tooling. Never emit headings, lists, code fences or any other structure. + Rule 2: Answer with the requested slot text only, dense and technical, and keep it as short as the slot instruction asks. + Rule 3: Only report resources, variables and modules that appear in the supplied facts. -# Synthesis prompt for the global infrastructure overview +# Prose style for every section of the global infrastructure overview pages architecture_prompt: | - Task: Write a global cloud architecture overview and deployment guide based on the module summaries. - Rule 1: Explain the high-level cloud topology, provider requirements, and how modules connect (e.g., VPC connecting to EKS and RDS). - Rule 2: Provide a dense, developer-friendly infrastructure overview and standard Terraform deployment commands (init, plan, apply). + Task: Write the prose for one slot of a cloud architecture or quickstart page. + Rule 1: The title, the headings and their order are already fixed by the tooling. Never emit headings, lists, code fences or any other structure. + Rule 2: Cover cloud topology, provider requirements, inter-module networking and the standard Terraform deployment commands, as densely as the slot instruction allows. + Rule 3: Never invent a resource, provider or command you were not shown. file_fact_consolidation_prompt: | You are a specialized infrastructure documentation assistant. Consolidate, deduplicate and merge the following facts extracted from different chunks of the same Terraform file into a single, cohesive summary. diff --git a/internal/config/config.go b/internal/config/config.go index 24f6a14..93fea7b 100644 --- a/internal/config/config.go +++ b/internal/config/config.go @@ -3,10 +3,20 @@ package config const ( // CodeReducerModelIDEnvKey is the env key for model ID override. CodeReducerModelIDEnvKey = "CODE_REDUCER_MODEL_ID" + // CodeReducerThinkEnvKey is the env key for reasoning mode override. + CodeReducerThinkEnvKey = "CODE_REDUCER_THINK" // OllamaBaseURLEnvKey is the env key for Ollama URL override. OllamaBaseURLEnvKey = "OLLAMA_BASE_URL" // OllamaNumCtxEnvKey is the env key for context size override. OllamaNumCtxEnvKey = "OLLAMA_NUM_CTX" + // OllamaNumPredictEnvKey is the env key for generation budget override. + OllamaNumPredictEnvKey = "OLLAMA_NUM_PREDICT" + // CodeReducerSlotNumPredictEnvKey is the env key for the line slot budget. + CodeReducerSlotNumPredictEnvKey = "CODE_REDUCER_SLOT_NUM_PREDICT" + // CodeReducerParagraphNumPredictEnvKey is the env key for the paragraph slot budget. + CodeReducerParagraphNumPredictEnvKey = "CODE_REDUCER_PARAGRAPH_NUM_PREDICT" + // CodeReducerIncludeTestsEnvKey is the env key for the test file switch. + CodeReducerIncludeTestsEnvKey = "CODE_REDUCER_INCLUDE_TESTS" // OllamaDefaultBaseURL is the default URL for local Ollama api. OllamaDefaultBaseURL = "http://localhost:11434" @@ -20,15 +30,53 @@ const ( ConfigFileName = ".code-reducer.yaml" configFilePerm = 0600 + // ThinkDefault disables reasoning output by default. Reasoning models spend + // the output budget on hidden thinking, which leaves content empty or + // hallucinated and is several times slower. + ThinkDefault = false + // NumPredictDefault of 0 lets Ollama decide the generation budget. + NumPredictDefault = 0 + // IncludeTestsDefault excludes test files from discovery. Documenting a test + // file mostly restates its assertions and costs a full extraction pass per + // file, so tests are opt-in. + IncludeTestsDefault = false + // CharsPerTokenDefault is the measured characters-per-token ratio for Go and + // Markdown payloads, used to turn the token budget into a character budget. + CharsPerTokenDefault = 3.0 + // OutputTokenReserveDefault is the number of context tokens held back from + // the prompt budget so the model always has room to generate an answer. + OutputTokenReserveDefault = 1024 + // SlotNumPredictDefault bounds every document line slot. A line slot is a + // micro-task whose answer the code trims to its first sentence and to + // slotLineWordBudget words, so tokens past the first sentence are pure waste. + // Measured with think disabled a one-line slot costs 34 tokens for a 164 + // character answer, so 192 is roughly five times a real one-line answer and + // still caps a runaway generation. With thinking enabled the same slot costs + // 400 tokens, almost all of it reasoning the tool never reads, which is why + // this default is sized for a non-thinking answer and why a clipped line is + // salvaged down to its first complete sentence. + SlotNumPredictDefault = 192 + // ParagraphNumPredictDefault bounds every document paragraph slot. Measured + // with think disabled a paragraph slot costs 63 tokens for a 287 character + // answer, so 1024 is roughly sixteen times a real paragraph answer. With + // thinking enabled the same slot costs 795 tokens, most of it reasoning the + // tool never reads. A paragraph answer is not shortened by the code, and a + // clipped one is salvaged only up to its last complete sentence, so the bound + // is the only thing keeping a talkative paragraph from losing its tail. + ParagraphNumPredictDefault = 1024 + // DefaultSystemPrompt is the default system instructions for the LLM. DefaultSystemPrompt = "You are Code-Reducer, an expert technical writer and code analyzer. Your job is to strictly follow instructions. You do not yap, you do not write filler.\n" + "DEFENSIVE RULES: 1. Do NOT use absolute terms ('always', 'never', 'zero') unless explicitly proven. 2. Do NOT guess downstream consequences or invent unhandled paths. If an error is swallowed, just say it is swallowed. 3. Do NOT name standard library packages unless explicitly stated in the source text. 4. Only report facts you are 100% sure about.\n" - // DefaultModuleSynthesisPrompt is the default prompt for folder summaries. - DefaultModuleSynthesisPrompt = "Task: Write a technical documentation page for a code module based on the provided list of its internal components.\nRule 1: Group related functions and classes under appropriate Markdown headings.\nRule 2: Explain the responsibility of the module and the data flow.\nRule 3: Keep it highly technical and dense." + // DefaultModuleSynthesisPrompt is the default prose style for module page slots. + // The page skeleton is written by the code, so this prompt must never ask the + // model for headings, grouping or a whole page. + DefaultModuleSynthesisPrompt = "Task: Write the prose for one slot of a module documentation page.\nRule 1: The headings, their order and the component names are already fixed by the tooling. Never emit headings, lists, code fences or any other structure.\nRule 2: Answer with the requested slot text only, dense and technical, and keep it as short as the slot instruction asks.\nRule 3: Use only the supplied facts, and say nothing about code you were not shown." - // DefaultArchitecturePrompt is the default prompt for architecture synthesis. - DefaultArchitecturePrompt = "Task: Write a global architecture or quickstart document based on the module summaries.\nRule 1: Explain the system boundaries and how the modules interact.\nRule 2: Provide a dense, developer-friendly overview." + // DefaultArchitecturePrompt is the default prose style for the architecture and + // quickstart page slots, which have the same fixed-skeleton contract. + DefaultArchitecturePrompt = "Task: Write the prose for one slot of a project documentation page (architecture or quickstart).\nRule 1: The title, the headings and their order are already fixed by the tooling. Never emit headings, lists, code fences or any other structure.\nRule 2: Answer with the requested slot text only, dense and developer-friendly, and keep it as short as the slot instruction asks.\nRule 3: Use only the supplied module summaries, and never invent a module, a command or a workflow you were not shown." // DefaultFileFactConsolidationPrompt is the default prompt for consolidation. DefaultFileFactConsolidationPrompt = "You are a specialized code documentation assistant.\nConsolidate, deduplicate and merge the following facts extracted from different chunks of the same file into a single, cohesive summary." @@ -46,11 +94,18 @@ type Config struct { OllamaBaseURL string `yaml:"ollama_base_url"` OllamaNumCtx int `yaml:"ollama_num_ctx"` DocsDir string `yaml:"docs_dir"` + Think bool `yaml:"think,omitempty"` + NumPredict int `yaml:"num_predict,omitempty"` + SlotNumPredict int `yaml:"slot_num_predict,omitempty"` + ParagraphNumPredict int `yaml:"paragraph_num_predict,omitempty"` + CharsPerToken float64 `yaml:"chars_per_token,omitempty"` + OutputTokenReserve int `yaml:"output_token_reserve,omitempty"` SystemPrompt string `yaml:"system_prompt"` ModuleSynthesisPrompt string `yaml:"module_synthesis_prompt"` ArchitecturePrompt string `yaml:"architecture_prompt"` FileFactConsolidationPrompt string `yaml:"file_fact_consolidation_prompt"` ExtractionSteps []ExtractionStep `yaml:"extraction_steps"` + IncludeTests bool `yaml:"include_tests,omitempty"` Ignore []string `yaml:"ignore"` } diff --git a/internal/config/config_test.go b/internal/config/config_test.go index fe44d79..1dd869f 100644 --- a/internal/config/config_test.go +++ b/internal/config/config_test.go @@ -57,7 +57,7 @@ func TestResolveConfig(t *testing.T) { t.Setenv(OllamaNumCtxEnvKey, "4096") t.Setenv(CodeReducerModelIDEnvKey, "env-model") - resolved, err := ResolveConfig(dir, "flag-model", "8192") + resolved, err := ResolveConfig(dir, Flags{ModelID: "flag-model", NumCtx: "8192"}) if err != nil { t.Fatalf("failed to resolve config: %v", err) } @@ -72,7 +72,7 @@ func TestResolveConfig(t *testing.T) { t.Run("fail fast on invalid num ctx env", func(t *testing.T) { t.Setenv(OllamaNumCtxEnvKey, "invalid") - _, err := ResolveConfig(t.TempDir(), "", "") + _, err := ResolveConfig(t.TempDir(), Flags{}) if err == nil { t.Fatal("expected error, got nil") } @@ -80,14 +80,14 @@ func TestResolveConfig(t *testing.T) { t.Run("fail fast on zero num ctx env", func(t *testing.T) { t.Setenv(OllamaNumCtxEnvKey, "0") - _, err := ResolveConfig(t.TempDir(), "", "") + _, err := ResolveConfig(t.TempDir(), Flags{}) if err == nil { t.Fatal("expected error, got nil") } }) t.Run("fail fast on invalid num ctx flag", func(t *testing.T) { - _, err := ResolveConfig(t.TempDir(), "", "invalid") + _, err := ResolveConfig(t.TempDir(), Flags{NumCtx: "invalid"}) if err == nil { t.Fatal("expected error, got nil") } @@ -95,7 +95,7 @@ func TestResolveConfig(t *testing.T) { t.Run("resolve base url from env", func(t *testing.T) { t.Setenv(OllamaBaseURLEnvKey, "http://custom-ollama:11434") - resolved, err := ResolveConfig(t.TempDir(), "", "") + resolved, err := ResolveConfig(t.TempDir(), Flags{}) if err != nil { t.Fatalf("unexpected error: %v", err) } @@ -104,3 +104,186 @@ func TestResolveConfig(t *testing.T) { } }) } + +func TestResolveConfigGenerationSettings(t *testing.T) { + t.Run("defaults", func(t *testing.T) { + resolved, err := ResolveConfig(t.TempDir(), Flags{}) + if err != nil { + t.Fatalf("unexpected error: %v", err) + } + if resolved.Think != ThinkDefault { + t.Errorf("expected Think %v, got %v", ThinkDefault, resolved.Think) + } + if resolved.NumPredict != NumPredictDefault { + t.Errorf("expected NumPredict %d, got %d", NumPredictDefault, resolved.NumPredict) + } + if resolved.SlotNumPredict != SlotNumPredictDefault { + t.Errorf("expected SlotNumPredict %d, got %d", SlotNumPredictDefault, resolved.SlotNumPredict) + } + if resolved.ParagraphNumPredict != ParagraphNumPredictDefault { + t.Errorf("expected ParagraphNumPredict %d, got %d", ParagraphNumPredictDefault, resolved.ParagraphNumPredict) + } + if resolved.CharsPerToken != CharsPerTokenDefault { + t.Errorf("expected CharsPerToken %v, got %v", CharsPerTokenDefault, resolved.CharsPerToken) + } + if resolved.OutputTokenReserve != OutputTokenReserveDefault { + t.Errorf("expected OutputTokenReserve %d, got %d", OutputTokenReserveDefault, resolved.OutputTokenReserve) + } + if resolved.IncludeTests != IncludeTestsDefault { + t.Errorf("expected IncludeTests %v, got %v", IncludeTestsDefault, resolved.IncludeTests) + } + }) + + t.Run("yaml values applied", func(t *testing.T) { + dir := t.TempDir() + err := SaveConfig(dir, &Config{ + Think: true, + NumPredict: 2048, + SlotNumPredict: 256, + ParagraphNumPredict: 1536, + CharsPerToken: 2.5, + OutputTokenReserve: 512, + IncludeTests: true, + }) + if err != nil { + t.Fatalf("failed to save config: %v", err) + } + + resolved, err := ResolveConfig(dir, Flags{}) + if err != nil { + t.Fatalf("unexpected error: %v", err) + } + if !resolved.Think { + t.Error("expected Think true from yaml, got false") + } + if resolved.NumPredict != 2048 { + t.Errorf("expected 2048, got %d", resolved.NumPredict) + } + if resolved.SlotNumPredict != 256 { + t.Errorf("expected 256, got %d", resolved.SlotNumPredict) + } + if resolved.ParagraphNumPredict != 1536 { + t.Errorf("expected 1536, got %d", resolved.ParagraphNumPredict) + } + if resolved.CharsPerToken != 2.5 { + t.Errorf("expected 2.5, got %v", resolved.CharsPerToken) + } + if resolved.OutputTokenReserve != 512 { + t.Errorf("expected 512, got %d", resolved.OutputTokenReserve) + } + if !resolved.IncludeTests { + t.Error("expected IncludeTests true from yaml, got false") + } + }) + + t.Run("env overrides yaml", func(t *testing.T) { + dir := t.TempDir() + err := SaveConfig(dir, &Config{NumPredict: 2048, SlotNumPredict: 256, ParagraphNumPredict: 256, IncludeTests: true}) + if err != nil { + t.Fatalf("failed to save config: %v", err) + } + + t.Setenv(CodeReducerThinkEnvKey, "true") + t.Setenv(OllamaNumPredictEnvKey, "4096") + t.Setenv(CodeReducerSlotNumPredictEnvKey, "192") + t.Setenv(CodeReducerParagraphNumPredictEnvKey, "1920") + t.Setenv(CodeReducerIncludeTestsEnvKey, "false") + + resolved, err := ResolveConfig(dir, Flags{}) + if err != nil { + t.Fatalf("unexpected error: %v", err) + } + if !resolved.Think { + t.Error("expected Think true from env, got false") + } + if resolved.NumPredict != 4096 { + t.Errorf("expected 4096, got %d", resolved.NumPredict) + } + if resolved.SlotNumPredict != 192 { + t.Errorf("expected 192, got %d", resolved.SlotNumPredict) + } + if resolved.ParagraphNumPredict != 1920 { + t.Errorf("expected 1920, got %d", resolved.ParagraphNumPredict) + } + if resolved.IncludeTests { + t.Error("expected IncludeTests false from env, got true") + } + }) + + t.Run("flags override env", func(t *testing.T) { + t.Setenv(CodeReducerThinkEnvKey, "true") + t.Setenv(OllamaNumPredictEnvKey, "4096") + t.Setenv(CodeReducerSlotNumPredictEnvKey, "192") + t.Setenv(CodeReducerParagraphNumPredictEnvKey, "1920") + + resolved, err := ResolveConfig(t.TempDir(), Flags{ + Think: "false", + NumPredict: "1024", + SlotNumPredict: "64", + ParagraphNumPredict: "800", + CharsPerToken: "3.5", + OutputTokenReserve: "256", + IncludeTests: "true", + }) + if err != nil { + t.Fatalf("unexpected error: %v", err) + } + if resolved.Think { + t.Error("expected Think false from flag, got true") + } + if resolved.NumPredict != 1024 { + t.Errorf("expected 1024, got %d", resolved.NumPredict) + } + if resolved.SlotNumPredict != 64 { + t.Errorf("expected 64, got %d", resolved.SlotNumPredict) + } + if resolved.ParagraphNumPredict != 800 { + t.Errorf("expected 800, got %d", resolved.ParagraphNumPredict) + } + if resolved.CharsPerToken != 3.5 { + t.Errorf("expected 3.5, got %v", resolved.CharsPerToken) + } + if resolved.OutputTokenReserve != 256 { + t.Errorf("expected 256, got %d", resolved.OutputTokenReserve) + } + if !resolved.IncludeTests { + t.Error("expected IncludeTests true from flag, got false") + } + }) + + t.Run("fail fast on invalid generation settings", func(t *testing.T) { + cases := []struct { + name string + env string + envVal string + flags Flags + }{ + {name: "invalid think env", env: CodeReducerThinkEnvKey, envVal: "maybe"}, + {name: "invalid num predict env", env: OllamaNumPredictEnvKey, envVal: "-1"}, + {name: "invalid slot num predict env", env: CodeReducerSlotNumPredictEnvKey, envVal: "0"}, + {name: "invalid slot num predict env text", env: CodeReducerSlotNumPredictEnvKey, envVal: "many"}, + {name: "negative paragraph num predict env", env: CodeReducerParagraphNumPredictEnvKey, envVal: "-1"}, + {name: "invalid paragraph num predict env text", env: CodeReducerParagraphNumPredictEnvKey, envVal: "many"}, + {name: "invalid think flag", flags: Flags{Think: "maybe"}}, + {name: "invalid num predict flag", flags: Flags{NumPredict: "-1"}}, + {name: "invalid slot num predict flag", flags: Flags{SlotNumPredict: "0"}}, + {name: "invalid paragraph num predict flag", flags: Flags{ParagraphNumPredict: "0"}}, + {name: "negative paragraph num predict flag", flags: Flags{ParagraphNumPredict: "-1"}}, + {name: "invalid chars per token flag", flags: Flags{CharsPerToken: "0"}}, + {name: "invalid chars per token flag text", flags: Flags{CharsPerToken: "many"}}, + {name: "invalid output token reserve flag", flags: Flags{OutputTokenReserve: "-5"}}, + {name: "invalid include tests env", env: CodeReducerIncludeTestsEnvKey, envVal: "sometimes"}, + {name: "invalid include tests flag", flags: Flags{IncludeTests: "sometimes"}}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + if tc.env != "" { + t.Setenv(tc.env, tc.envVal) + } + if _, err := ResolveConfig(t.TempDir(), tc.flags); err == nil { + t.Fatal("expected error, got nil") + } + }) + } + }) +} diff --git a/internal/config/io.go b/internal/config/io.go index cfa188e..49b2b1c 100644 --- a/internal/config/io.go +++ b/internal/config/io.go @@ -41,6 +41,7 @@ func formatYAML(data []byte) string { "\narchitecture_prompt:", "\n\narchitecture_prompt:", "\nfile_fact_consolidation_prompt:", "\n\nfile_fact_consolidation_prompt:", "\nextraction_steps:", "\n\nextraction_steps:", + "\ninclude_tests:", "\n\ninclude_tests:", "\nignore:", "\n\nignore:", } for i := 0; i < len(replacements); i += 2 { diff --git a/internal/config/resolve.go b/internal/config/resolve.go index 1609c99..d5d857b 100644 --- a/internal/config/resolve.go +++ b/internal/config/resolve.go @@ -85,15 +85,214 @@ func resolveNumCtx(cfgVal int, flagVal string) (int, error) { return numCtx, nil } +func resolveThink(cfgVal bool, flagVal string) (bool, error) { + think := ThinkDefault + if cfgVal { + think = true + } + if envVal := os.Getenv(CodeReducerThinkEnvKey); envVal != "" { + b, err := strconv.ParseBool(envVal) + if err != nil { + return false, fmt.Errorf("invalid value for %s: %s", CodeReducerThinkEnvKey, envVal) + } + think = b + } + if flagVal != "" { + b, err := strconv.ParseBool(flagVal) + if err != nil { + return false, fmt.Errorf("invalid value for think flag: %s", flagVal) + } + think = b + } + return think, nil +} + +func resolveNumPredict(cfgVal int, flagVal string) (int, error) { + numPredict := NumPredictDefault + if cfgVal > 0 { + numPredict = cfgVal + } + if envVal := os.Getenv(OllamaNumPredictEnvKey); envVal != "" { + n, err := strconv.Atoi(envVal) + if err != nil || n < 0 { + return 0, fmt.Errorf("invalid value for %s: %s", OllamaNumPredictEnvKey, envVal) + } + numPredict = n + } + if flagVal != "" { + n, err := strconv.Atoi(flagVal) + if err != nil || n < 0 { + return 0, fmt.Errorf("invalid value for num-predict flag: %s", flagVal) + } + numPredict = n + } + return numPredict, nil +} + +// resolveSlotNumPredict picks the generation bound for a single document line +// slot. Zero is rejected on purpose: an unbounded slot can silently stop on the +// generation limit and turn a correct answer into a hard pipeline failure. +func resolveSlotNumPredict(cfgVal int, flagVal string) (int, error) { + slotNumPredict := SlotNumPredictDefault + if cfgVal > 0 { + slotNumPredict = cfgVal + } + if envVal := os.Getenv(CodeReducerSlotNumPredictEnvKey); envVal != "" { + n, err := strconv.Atoi(envVal) + if err != nil || n <= 0 { + return 0, fmt.Errorf("invalid value for %s: %s", CodeReducerSlotNumPredictEnvKey, envVal) + } + slotNumPredict = n + } + if flagVal != "" { + n, err := strconv.Atoi(flagVal) + if err != nil || n <= 0 { + return 0, fmt.Errorf("invalid value for slot-num-predict flag: %s", flagVal) + } + slotNumPredict = n + } + return slotNumPredict, nil +} + +// resolveParagraphNumPredict picks the generation bound for a single document +// paragraph slot. It follows the same precedence and the same rejection of a +// zero or negative value as resolveSlotNumPredict, because the failure it prevents +// is the same one: a bounded answer the code can salvage, rather than an unbounded +// one that stops on the generation limit and aborts the page. +func resolveParagraphNumPredict(cfgVal int, flagVal string) (int, error) { + paragraphNumPredict := ParagraphNumPredictDefault + if cfgVal > 0 { + paragraphNumPredict = cfgVal + } + if envVal := os.Getenv(CodeReducerParagraphNumPredictEnvKey); envVal != "" { + n, err := strconv.Atoi(envVal) + if err != nil || n <= 0 { + return 0, fmt.Errorf("invalid value for %s: %s", CodeReducerParagraphNumPredictEnvKey, envVal) + } + paragraphNumPredict = n + } + if flagVal != "" { + n, err := strconv.Atoi(flagVal) + if err != nil || n <= 0 { + return 0, fmt.Errorf("invalid value for paragraph-num-predict flag: %s", flagVal) + } + paragraphNumPredict = n + } + return paragraphNumPredict, nil +} + +func resolveCharsPerToken(cfgVal float64, flagVal string) (float64, error) { + charsPerToken := CharsPerTokenDefault + if cfgVal > 0 { + charsPerToken = cfgVal + } + if flagVal != "" { + f, err := strconv.ParseFloat(flagVal, 64) + if err != nil || f <= 0 { + return 0, fmt.Errorf("invalid value for chars-per-token flag: %s", flagVal) + } + charsPerToken = f + } + return charsPerToken, nil +} + +func resolveOutputTokenReserve(cfgVal int, flagVal string) (int, error) { + outputTokenReserve := OutputTokenReserveDefault + if cfgVal > 0 { + outputTokenReserve = cfgVal + } + if flagVal != "" { + n, err := strconv.Atoi(flagVal) + if err != nil || n < 0 { + return 0, fmt.Errorf("invalid value for output-token-reserve flag: %s", flagVal) + } + outputTokenReserve = n + } + return outputTokenReserve, nil +} + +// resolveIncludeTests picks the test file switch. Like resolveThink, the config +// file can only turn the option on: an absent key and an explicit false mean the +// same thing, so there is nothing to override. +func resolveIncludeTests(cfgVal bool, flagVal string) (bool, error) { + includeTests := IncludeTestsDefault + if cfgVal { + includeTests = true + } + if envVal := os.Getenv(CodeReducerIncludeTestsEnvKey); envVal != "" { + b, err := strconv.ParseBool(envVal) + if err != nil { + return false, fmt.Errorf("invalid value for %s: %s", CodeReducerIncludeTestsEnvKey, envVal) + } + includeTests = b + } + if flagVal != "" { + b, err := strconv.ParseBool(flagVal) + if err != nil { + return false, fmt.Errorf("invalid value for include-tests flag: %s", flagVal) + } + includeTests = b + } + return includeTests, nil +} + +// Flags holds the raw CLI flag values that take precedence over the environment and the config file. +type Flags struct { + ModelID string + NumCtx string + Think string + NumPredict string + SlotNumPredict string + ParagraphNumPredict string + CharsPerToken string + OutputTokenReserve string + IncludeTests string +} + // ResolveConfig merges CLI overrides, environment variables, YAML config, and system defaults. // It returns a fully resolved Config struct ready to be used by the pipeline runner and LLM client. -func ResolveConfig(repoRoot, modelIDFlag, numCtxFlag string) (*Config, error) { +func ResolveConfig(repoRoot string, flags Flags) (*Config, error) { cfg, err := loadBaseConfig(repoRoot) if err != nil { return nil, err } - numCtx, err := resolveNumCtx(cfg.OllamaNumCtx, numCtxFlag) + numCtx, err := resolveNumCtx(cfg.OllamaNumCtx, flags.NumCtx) + if err != nil { + return nil, err + } + + think, err := resolveThink(cfg.Think, flags.Think) + if err != nil { + return nil, err + } + + numPredict, err := resolveNumPredict(cfg.NumPredict, flags.NumPredict) + if err != nil { + return nil, err + } + + slotNumPredict, err := resolveSlotNumPredict(cfg.SlotNumPredict, flags.SlotNumPredict) + if err != nil { + return nil, err + } + + paragraphNumPredict, err := resolveParagraphNumPredict(cfg.ParagraphNumPredict, flags.ParagraphNumPredict) + if err != nil { + return nil, err + } + + charsPerToken, err := resolveCharsPerToken(cfg.CharsPerToken, flags.CharsPerToken) + if err != nil { + return nil, err + } + + outputTokenReserve, err := resolveOutputTokenReserve(cfg.OutputTokenReserve, flags.OutputTokenReserve) + if err != nil { + return nil, err + } + + includeTests, err := resolveIncludeTests(cfg.IncludeTests, flags.IncludeTests) if err != nil { return nil, err } @@ -106,10 +305,17 @@ func ResolveConfig(repoRoot, modelIDFlag, numCtxFlag string) (*Config, error) { return &Config{ Ignore: deduplicate(cfg.Ignore), ExtractionSteps: resolvedSteps, - ModelID: resolveModelID(cfg.ModelID, modelIDFlag), + ModelID: resolveModelID(cfg.ModelID, flags.ModelID), OllamaBaseURL: resolveBaseURL(cfg.OllamaBaseURL), OllamaNumCtx: numCtx, DocsDir: resolveString(cfg.DocsDir, DefaultDocsDir), + Think: think, + NumPredict: numPredict, + SlotNumPredict: slotNumPredict, + ParagraphNumPredict: paragraphNumPredict, + CharsPerToken: charsPerToken, + OutputTokenReserve: outputTokenReserve, + IncludeTests: includeTests, SystemPrompt: resolveString(cfg.SystemPrompt, DefaultSystemPrompt), ModuleSynthesisPrompt: resolveString(cfg.ModuleSynthesisPrompt, DefaultModuleSynthesisPrompt), ArchitecturePrompt: resolveString(cfg.ArchitecturePrompt, DefaultArchitecturePrompt), diff --git a/internal/engine/budget.go b/internal/engine/budget.go new file mode 100644 index 0000000..216fb6f --- /dev/null +++ b/internal/engine/budget.go @@ -0,0 +1,27 @@ +package engine + +import "github.com/arrase/code-reducer/internal/config" + +// promptTokenBudget returns the number of context tokens available to a prompt +// payload, holding outputReserve tokens back so generation always has headroom. +func promptTokenBudget(numCtx, outputReserve int) int { + if numCtx < minNumCtxFloor { + numCtx = minNumCtxFloor + } + if outputReserve < 0 { + outputReserve = 0 + } + if outputReserve >= numCtx { + outputReserve = numCtx / 4 + } + return numCtx - outputReserve +} + +// promptCharBudget converts the prompt token budget into a character budget using +// the measured characters-per-token ratio for the payload being sent. +func promptCharBudget(numCtx, outputReserve int, charsPerToken float64) int { + if charsPerToken <= 0 { + charsPerToken = config.CharsPerTokenDefault + } + return int(float64(promptTokenBudget(numCtx, outputReserve)) * charsPerToken) +} diff --git a/internal/engine/chunking.go b/internal/engine/chunking.go index 0d68ee5..8e2082b 100644 --- a/internal/engine/chunking.go +++ b/internal/engine/chunking.go @@ -10,11 +10,14 @@ import ( ) type reductionConfig struct { - sysPrompt string - buildPrompt func(batch []string) string - logMsg func(batch []string) string - errMsg string - logEvent LogEventFunc + sysPrompt string + buildPrompt func(batch []string) string + logMsg func(batch []string) string + errMsg string + subject string + logEvent LogEventFunc + outputReserve int + charsPerToken float64 } func reduceWithLLM( @@ -26,7 +29,7 @@ func reduceWithLLM( if len(items) == 0 { return "", nil } - maxChars := c.NumCtx() * maxCharsMultiplier + maxChars := promptCharBudget(c.NumCtx(), cfg.outputReserve, cfg.charsPerToken) return reduceItems(ctx, items, maxChars, func(batch []string) (string, error) { prompt := cfg.buildPrompt(batch) cfg.logEvent(EventStatus, cfg.logMsg(batch)) @@ -34,39 +37,28 @@ func reduceWithLLM( if err != nil { return "", fmt.Errorf("%s: %w", cfg.errMsg, err) } - return stripOuterMarkdownFence(res), nil + return requireCompleteContent(res, cfg.subject) }) } -func reduceInChunks(ctx context.Context, c llmCaller, nodePath string, items []string, cfg *config.Config, logEvent LogEventFunc) (string, error) { - redCfg := reductionConfig{ - sysPrompt: cfg.SystemPrompt + "\n" + cfg.ModuleSynthesisPrompt, - buildPrompt: func(batch []string) string { - return fmt.Sprintf("Synthesize architecture for %s:\n%s", nodePath, strings.Join(batch, "\n\n")) - }, - logMsg: func(batch []string) string { - return fmt.Sprintf("➜ LLM Synthesizing chunk for %s (%d items)", nodePath, len(batch)) - }, - errMsg: "LLM error during synthesis", - logEvent: logEvent, - } - return reduceWithLLM(ctx, c, items, redCfg) -} - func reduceFileFacts(ctx context.Context, c llmCaller, filePath string, stepName string, items []string, cfg *config.Config, logEvent LogEventFunc) (string, error) { if len(items) == 1 { return items[0], nil } redCfg := reductionConfig{ - sysPrompt: cfg.SystemPrompt + "\n" + cfg.FileFactConsolidationPrompt, + sysPrompt: cfg.SystemPrompt, buildPrompt: func(batch []string) string { - return fmt.Sprintf("Consolidate and deduplicate the extracted facts for %s regarding step '%s':\n%s", filePath, stepName, strings.Join(batch, "\n\n")) + return userMessage(cfg.FileFactConsolidationPrompt, + fmt.Sprintf("Consolidate and deduplicate the extracted facts for %s regarding step '%s':\n%s", filePath, stepName, strings.Join(batch, "\n\n"))) }, logMsg: func(batch []string) string { return fmt.Sprintf("➜ LLM Consolidating facts for %s (%d items)", filePath, len(batch)) }, - errMsg: "LLM error during file fact consolidation", - logEvent: logEvent, + errMsg: "LLM error during file fact consolidation", + subject: fmt.Sprintf("consolidating facts for %s step %s", filePath, stepName), + logEvent: logEvent, + outputReserve: cfg.OutputTokenReserve, + charsPerToken: cfg.CharsPerToken, } return reduceWithLLM(ctx, c, items, redCfg) } diff --git a/internal/engine/chunking_test.go b/internal/engine/chunking_test.go index 454b591..ae4fc67 100644 --- a/internal/engine/chunking_test.go +++ b/internal/engine/chunking_test.go @@ -2,8 +2,11 @@ package engine import ( "context" + "errors" "strings" "testing" + + "github.com/arrase/code-reducer/internal/config" ) func TestChunkTextWithOverlap(t *testing.T) { @@ -112,15 +115,95 @@ func TestCountRunes(t *testing.T) { } func TestCalculateFileLimit(t *testing.T) { + cfg := &config.Config{CharsPerToken: config.CharsPerTokenDefault, OutputTokenReserve: config.OutputTokenReserveDefault} + // Below floor - limit1 := calculateFileLimit(100) + limit1 := calculateFileLimit(100, cfg) if limit1 <= 0 { t.Fatalf("expected positive limit, got %d", limit1) } // Above floor - limit2 := calculateFileLimit(8192) + limit2 := calculateFileLimit(8192, cfg) if limit2 <= limit1 { t.Fatalf("expected limit2 > limit1, got %d <= %d", limit2, limit1) } } + +func TestPromptCharBudgetLeavesHeadroom(t *testing.T) { + cases := []struct { + name string + numCtx int + outputReserve int + charsPerToken float64 + }{ + {name: "default settings", numCtx: 8192, outputReserve: config.OutputTokenReserveDefault, charsPerToken: config.CharsPerTokenDefault}, + {name: "20000 token context", numCtx: 20000, outputReserve: config.OutputTokenReserveDefault, charsPerToken: config.CharsPerTokenDefault}, + {name: "below floor", numCtx: 128, outputReserve: config.OutputTokenReserveDefault, charsPerToken: config.CharsPerTokenDefault}, + {name: "reserve larger than context", numCtx: 512, outputReserve: 4096, charsPerToken: config.CharsPerTokenDefault}, + {name: "no reserve", numCtx: 8192, outputReserve: 0, charsPerToken: config.CharsPerTokenDefault}, + {name: "custom ratio", numCtx: 32768, outputReserve: 2048, charsPerToken: 2.5}, + {name: "unset ratio falls back to default", numCtx: 8192, outputReserve: config.OutputTokenReserveDefault, charsPerToken: 0}, + } + + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + charsPerToken := tc.charsPerToken + if charsPerToken <= 0 { + charsPerToken = config.CharsPerTokenDefault + } + effectiveNumCtx := tc.numCtx + if effectiveNumCtx < minNumCtxFloor { + effectiveNumCtx = minNumCtxFloor + } + + limit := promptCharBudget(tc.numCtx, tc.outputReserve, tc.charsPerToken) + if limit <= 0 { + t.Fatalf("expected a positive char budget, got %d", limit) + } + + promptTokens := float64(limit) / charsPerToken + allowed := float64(promptTokenBudget(tc.numCtx, tc.outputReserve)) + if promptTokens > allowed { + t.Errorf("char budget %d chars is %.0f tokens, which exceeds the %0.f token prompt budget for numCtx=%d", + limit, promptTokens, allowed, tc.numCtx) + } + if promptTokens > float64(effectiveNumCtx) { + t.Errorf("char budget %d chars is %.0f tokens, which fills the whole %d token context", + limit, promptTokens, effectiveNumCtx) + } + }) + } +} + +func TestReduceFileFactsRejectsIncompleteLLMOutput(t *testing.T) { + cases := []struct { + name string + res LLMResult + wantErr error + }{ + {name: "truncated by length", res: truncatedResult("partial facts"), wantErr: ErrTruncatedOutput}, + {name: "empty content on stop", res: stopResult(""), wantErr: ErrEmptyOutput}, + } + + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + c := &scriptedLLM{numCtx: 2048, results: []LLMResult{tc.res}} + cfg := &config.Config{ + SystemPrompt: "system", + FileFactConsolidationPrompt: "consolidation", + CharsPerToken: config.CharsPerTokenDefault, + OutputTokenReserve: config.OutputTokenReserveDefault, + } + logEvent := func(t EventType, msg string) {} + + res, err := reduceFileFacts(context.Background(), c, "internal/engine/client.go", "API_SIGNATURES", []string{"first", "second"}, cfg, logEvent) + if !errors.Is(err, tc.wantErr) { + t.Fatalf("expected %v, got %v (res: %q)", tc.wantErr, err, res) + } + if res != "" { + t.Errorf("expected no partial result, got %q", res) + } + }) + } +} diff --git a/internal/engine/client.go b/internal/engine/client.go index 4872803..2e22af9 100644 --- a/internal/engine/client.go +++ b/internal/engine/client.go @@ -4,19 +4,61 @@ import ( "bytes" "context" "encoding/json" + "errors" "fmt" "io" "net/http" "strings" + + "github.com/arrase/code-reducer/internal/config" ) +const doneReasonLength = "length" + +var ( + // ErrTruncatedOutput indicates the model stopped because it reached the generation + // limit, so the response is incomplete and must never be persisted or cached. + ErrTruncatedOutput = errors.New("llm output was truncated: generation limit reached before completion") + // ErrEmptyOutput indicates the model returned no visible content, which happens when + // a reasoning model spends the whole output budget on hidden thinking. + ErrEmptyOutput = errors.New("llm returned no visible content") +) + +// Message is a single chat message. Thinking carries the reasoning block emitted by +// reasoning models and is never populated on responses. type Message struct { - Role string `json:"role"` - Content string `json:"content"` + Role string `json:"role"` + Content string `json:"content"` + Thinking *string `json:"thinking,omitempty"` +} + +// userMessage joins the ordered parts of one user turn. Order matters for prompt +// caching: Ollama reuses the cached tokens of an unchanged prompt prefix, so the +// parts a run keeps constant must come first and the per-call text must follow +// them. Every call of a run shares the system message (cfg.SystemPrompt) and puts +// its own instruction in this turn. +func userMessage(parts ...string) string { + kept := make([]string, 0, len(parts)) + for _, part := range parts { + if strings.TrimSpace(part) != "" { + kept = append(kept, part) + } + } + return strings.Join(kept, "\n\n") +} + +// LLMResult carries the generated text together with the generation metadata +// reported by the provider. +type LLMResult struct { + Content string + DoneReason string + EvalCount int + PromptCount int } type llmCaller interface { - CallLLM(ctx context.Context, systemPrompt string, messages []Message, jsonFormat bool) (string, error) + CallLLM(ctx context.Context, systemPrompt string, messages []Message, jsonFormat bool) (LLMResult, error) + CallLLMWithNumPredict(ctx context.Context, systemPrompt string, messages []Message, numPredict int) (LLMResult, error) NumCtx() int } @@ -24,14 +66,18 @@ type llmClient struct { modelID string baseURL string numCtx int + think bool + numPredict int httpClient *http.Client } -func newLLMClient(modelID, baseURL string, numCtx int) *llmClient { +func newLLMClient(cfg *config.Config) *llmClient { return &llmClient{ - modelID: modelID, - baseURL: baseURL, - numCtx: numCtx, + modelID: cfg.ModelID, + baseURL: cfg.OllamaBaseURL, + numCtx: cfg.OllamaNumCtx, + think: cfg.Think, + numPredict: cfg.NumPredict, httpClient: &http.Client{Timeout: defaultHTTPTimeout}, } } @@ -45,27 +91,53 @@ type ollamaRequest struct { Model string `json:"model"` Messages []Message `json:"messages"` Stream bool `json:"stream"` + Think *bool `json:"think,omitempty"` Format string `json:"format,omitempty"` Options *ollamaOptions `json:"options,omitempty"` } type ollamaOptions struct { - NumCtx int `json:"num_ctx,omitempty"` + NumCtx int `json:"num_ctx,omitempty"` + NumPredict int `json:"num_predict,omitempty"` } type ollamaResponse struct { - Message Message `json:"message"` + Message Message `json:"message"` + DoneReason string `json:"done_reason"` + EvalCount int `json:"eval_count"` + PromptEvalCount int `json:"prompt_eval_count"` +} + +// requireCompleteContent rejects LLM output that must never be written to disk or +// cached. A generation that stopped on "length" is incomplete, and an empty content +// body means a reasoning model consumed the whole output budget on thinking. +func requireCompleteContent(res LLMResult, subject string) (string, error) { + if res.DoneReason == doneReasonLength { + return "", fmt.Errorf("%s: %w (prompt_tokens=%d, generated_tokens=%d): raise num_predict, raise slot_num_predict or paragraph_num_predict for document slots, or set think: false if a reasoning model is charging its thinking tokens against this budget", + subject, ErrTruncatedOutput, res.PromptCount, res.EvalCount) + } + content := stripOuterMarkdownFence(res.Content) + if content == "" { + return "", fmt.Errorf("%s: %w: the model emitted only reasoning output, set think: false to disable it", subject, ErrEmptyOutput) + } + return content, nil } // prepareOllamaRequest creates and serializes the HTTP request for the Ollama api/chat endpoint. -func (c *llmClient) prepareOllamaRequest(ctx context.Context, systemPrompt string, messages []Message, jsonFormat bool) (*http.Request, error) { +// A numPredict of zero falls back to the client-wide generation budget. +func (c *llmClient) prepareOllamaRequest(ctx context.Context, systemPrompt string, messages []Message, jsonFormat bool, numPredict int) (*http.Request, error) { url := strings.TrimSuffix(c.baseURL, "/") + "/api/chat" + if numPredict <= 0 { + numPredict = c.numPredict + } + reqBody := ollamaRequest{ Model: c.modelID, Messages: append([]Message{{Role: "system", Content: systemPrompt}}, messages...), Stream: false, - Options: &ollamaOptions{NumCtx: c.numCtx}, + Think: &c.think, + Options: &ollamaOptions{NumCtx: c.numCtx, NumPredict: numPredict}, } if jsonFormat { reqBody.Format = "json" @@ -86,32 +158,47 @@ func (c *llmClient) prepareOllamaRequest(ctx context.Context, systemPrompt strin } // CallLLM invokes the LLM via HTTP failing fast without retries. -func (c *llmClient) CallLLM(ctx context.Context, systemPrompt string, messages []Message, jsonFormat bool) (string, error) { +func (c *llmClient) CallLLM(ctx context.Context, systemPrompt string, messages []Message, jsonFormat bool) (LLMResult, error) { + return c.call(ctx, systemPrompt, messages, jsonFormat, 0) +} + +// CallLLMWithNumPredict invokes the LLM with a per-call generation bound, used by +// document prose slots so a micro-task cannot expand into a full-page generation. +func (c *llmClient) CallLLMWithNumPredict(ctx context.Context, systemPrompt string, messages []Message, numPredict int) (LLMResult, error) { + return c.call(ctx, systemPrompt, messages, false, numPredict) +} + +func (c *llmClient) call(ctx context.Context, systemPrompt string, messages []Message, jsonFormat bool, numPredict int) (LLMResult, error) { - req, err := c.prepareOllamaRequest(ctx, systemPrompt, messages, jsonFormat) + req, err := c.prepareOllamaRequest(ctx, systemPrompt, messages, jsonFormat, numPredict) if err != nil { - return "", err + return LLMResult{}, err } resp, err := c.httpClient.Do(req) if err != nil { - return "", err + return LLMResult{}, err } defer resp.Body.Close() if resp.StatusCode == http.StatusOK { respData, err := io.ReadAll(resp.Body) if err != nil { - return "", err + return LLMResult{}, err } var result ollamaResponse if err := json.Unmarshal(respData, &result); err != nil { - return "", fmt.Errorf("failed to parse response: %w", err) + return LLMResult{}, fmt.Errorf("failed to parse response: %w", err) } - return result.Message.Content, nil + return LLMResult{ + Content: result.Message.Content, + DoneReason: result.DoneReason, + EvalCount: result.EvalCount, + PromptCount: result.PromptEvalCount, + }, nil } body, _ := io.ReadAll(io.LimitReader(resp.Body, maxErrorBodyBytes)) - return "", fmt.Errorf("ollama api error: status %d, response: %s", resp.StatusCode, string(body)) + return LLMResult{}, fmt.Errorf("ollama api error: status %d, response: %s", resp.StatusCode, string(body)) } diff --git a/internal/engine/client_test.go b/internal/engine/client_test.go new file mode 100644 index 0000000..1470721 --- /dev/null +++ b/internal/engine/client_test.go @@ -0,0 +1,168 @@ +package engine + +import ( + "context" + "encoding/json" + "errors" + "net/http" + "net/http/httptest" + "strings" + "testing" + + "github.com/arrase/code-reducer/internal/config" +) + +func TestPrepareOllamaRequest(t *testing.T) { + t.Run("omits num_predict and sends think when unset", func(t *testing.T) { + c := newLLMClient(&config.Config{ModelID: "m", OllamaBaseURL: "http://localhost:11434", OllamaNumCtx: 2048}) + + req, err := c.prepareOllamaRequest(context.Background(), "system", []Message{{Role: "user", Content: "hi"}}, false, 0) + if err != nil { + t.Fatalf("unexpected error: %v", err) + } + + var body ollamaRequest + if err := json.NewDecoder(req.Body).Decode(&body); err != nil { + t.Fatalf("failed to decode request body: %v", err) + } + if body.Think == nil { + t.Error("expected think to be sent") + } else if *body.Think { + t.Error("expected think to default to false") + } + if body.Options.NumPredict != 0 { + t.Errorf("expected num_predict to be unset, got %d", body.Options.NumPredict) + } + if len(body.Messages) != 2 || body.Messages[0].Role != "system" { + t.Errorf("expected a system message followed by the user message, got %+v", body.Messages) + } + if body.Format != "" { + t.Errorf("expected no format for non-json calls, got %s", body.Format) + } + }) + + t.Run("per-call num_predict overrides the client budget", func(t *testing.T) { + c := newLLMClient(&config.Config{ModelID: "m", OllamaBaseURL: "http://localhost:11434", OllamaNumCtx: 8192, NumPredict: 4000}) + + req, err := c.prepareOllamaRequest(context.Background(), "system", nil, false, 96) + if err != nil { + t.Fatalf("unexpected error: %v", err) + } + + var body ollamaRequest + if err := json.NewDecoder(req.Body).Decode(&body); err != nil { + t.Fatalf("failed to decode request body: %v", err) + } + if body.Options.NumPredict != 96 { + t.Errorf("expected num_predict 96, got %d", body.Options.NumPredict) + } + }) + + t.Run("sends num_predict and json format when configured", func(t *testing.T) { + c := newLLMClient(&config.Config{ + ModelID: "m", + OllamaBaseURL: "http://localhost:11434", + OllamaNumCtx: 8192, + Think: true, + NumPredict: 4000, + }) + + req, err := c.prepareOllamaRequest(context.Background(), "system", nil, true, 0) + if err != nil { + t.Fatalf("unexpected error: %v", err) + } + + var body ollamaRequest + if err := json.NewDecoder(req.Body).Decode(&body); err != nil { + t.Fatalf("failed to decode request body: %v", err) + } + if body.Think == nil || !*body.Think { + t.Error("expected think true to be sent") + } + if body.Options.NumPredict != 4000 { + t.Errorf("expected num_predict 4000, got %d", body.Options.NumPredict) + } + if body.Format != "json" { + t.Errorf("expected json format, got %q", body.Format) + } + }) +} + +func TestCallLLMParsesGenerationMetadata(t *testing.T) { + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + w.Header().Set("Content-Type", "application/json") + _, _ = w.Write([]byte(`{"message":{"role":"assistant","content":"docs"},"done_reason":"length","eval_count":7,"prompt_eval_count":1234}`)) + })) + defer srv.Close() + + c := newLLMClient(&config.Config{ModelID: "m", OllamaBaseURL: srv.URL, OllamaNumCtx: 2048}) + res, err := c.CallLLM(context.Background(), "system", []Message{{Role: "user", Content: "hi"}}, false) + if err != nil { + t.Fatalf("unexpected error: %v", err) + } + if res.Content != "docs" { + t.Errorf("expected content 'docs', got %q", res.Content) + } + if res.DoneReason != "length" { + t.Errorf("expected done_reason 'length', got %q", res.DoneReason) + } + if res.EvalCount != 7 { + t.Errorf("expected eval_count 7, got %d", res.EvalCount) + } + if res.PromptCount != 1234 { + t.Errorf("expected prompt_eval_count 1234, got %d", res.PromptCount) + } +} + +func TestUserMessage(t *testing.T) { + cases := []struct { + name string + parts []string + want string + }{ + {name: "single part", parts: []string{"only"}, want: "only"}, + {name: "keeps the run invariant part first", parts: []string{"STYLE", "PAYLOAD\n\nINSTRUCTION"}, want: "STYLE\n\nPAYLOAD\n\nINSTRUCTION"}, + {name: "drops empty parts", parts: []string{"STYLE", " ", "INSTRUCTION"}, want: "STYLE\n\nINSTRUCTION"}, + {name: "no parts", parts: nil, want: ""}, + } + + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + if got := userMessage(tc.parts...); got != tc.want { + t.Fatalf("userMessage(%q) = %q, want %q", tc.parts, got, tc.want) + } + }) + } +} + +func TestRequireCompleteContent(t *testing.T) { + t.Run("strips fences from a valid result", func(t *testing.T) { + res := stopResult("```markdown\n# Title\n```") + content, err := requireCompleteContent(res, "subject") + if err != nil { + t.Fatalf("unexpected error: %v", err) + } + if content != "# Title" { + t.Errorf("expected '# Title', got %q", content) + } + }) + + t.Run("truncation wins over empty content", func(t *testing.T) { + _, err := requireCompleteContent(truncatedResult(""), "subject") + if !errors.Is(err, ErrTruncatedOutput) { + t.Fatalf("expected ErrTruncatedOutput, got %v", err) + } + }) + + t.Run("truncation names every budget lever and the thinking cost", func(t *testing.T) { + _, err := requireCompleteContent(truncatedResult("half an answer"), "subject") + if !errors.Is(err, ErrTruncatedOutput) { + t.Fatalf("expected ErrTruncatedOutput, got %v", err) + } + for _, want := range []string{"num_predict", "slot_num_predict", "paragraph_num_predict", "think: false"} { + if !strings.Contains(err.Error(), want) { + t.Errorf("expected the truncation message to name %q, got %q", want, err) + } + } + }) +} diff --git a/internal/engine/constants.go b/internal/engine/constants.go index 9264f4e..b2b553d 100644 --- a/internal/engine/constants.go +++ b/internal/engine/constants.go @@ -3,15 +3,26 @@ package engine import "time" const ( - defaultHTTPTimeout = 10 * time.Minute - maxErrorBodyBytes = 1024 - defaultChunkOverlap = 800 - minNumCtxFloor = 512 - contextWindowAllocRatio = 0.75 - maxCharsMultiplier = 3 - metadataFileName = ".metadata.json" - agentsFileName = "AGENTS.md" - defaultDirPerm = 0755 + defaultHTTPTimeout = 10 * time.Minute + maxErrorBodyBytes = 1024 + defaultChunkOverlap = 800 + minNumCtxFloor = 512 + metadataFileName = ".metadata.json" + agentsFileName = "AGENTS.md" + defaultDirPerm = 0755 + + markdownHeadingPrefix = "### " + subsystemPrefix = "Subsystem: " + + // slotLineWordBudget is the word ceiling of a line slot. The line sits under a + // heading the code already wrote, so the budget is what keeps a rambling answer + // from becoming an unreadable block under a fixed heading. + slotLineWordBudget = 40 + + headingResponsibility = "Responsibility" + headingComponents = "Components" + headingDataFlow = "Data Flow" + headingErrorHandling = "Error Handling" ) type LogEventFunc func(EventType, string) diff --git a/internal/engine/markdown.go b/internal/engine/markdown.go index 3fbb585..5eaa37b 100644 --- a/internal/engine/markdown.go +++ b/internal/engine/markdown.go @@ -5,7 +5,16 @@ import ( "strings" ) -var markdownFenceRe = regexp.MustCompile("(?s)^\\x60{3,}(?:markdown|json)?\\s*(.*?)\\s*\\x60{3,}$") +var ( + markdownFenceRe = regexp.MustCompile("(?s)^\\x60{3,}(?:markdown|json)?\\s*(.*?)\\s*\\x60{3,}$") + fenceLineRe = regexp.MustCompile("^\\s*(?:\\x60{3,}|~{3,})") + headingLineRe = regexp.MustCompile("^\\s{0,3}#{1,6}\\s") + bulletPrefixRe = regexp.MustCompile(`^(?:[-*+•]|\d+[.)])\s+`) + collapsedSpaceRe = regexp.MustCompile(`\s+`) + trailingPunctTrim = " .:;,-" + sentenceEndRunes = ".!?" + clauseEndRunes = ":;," +) // stripOuterMarkdownFence strips surrounding markdown or json code fences from input strings. func stripOuterMarkdownFence(content string) string { @@ -15,3 +24,154 @@ func stripOuterMarkdownFence(content string) string { } return trimmed } + +// stripSlotNoise removes the markdown a model reaches for even when asked for a +// bare line: fence lines, headings and list markers. A heading or fence line means +// the model is emitting structure instead of answering, so it is dropped outright. +func stripSlotNoise(content string) string { + kept := make([]string, 0, strings.Count(content, "\n")+1) + for _, line := range strings.Split(content, "\n") { + line = strings.TrimSpace(line) + if line == "" { + continue + } + if fenceLineRe.MatchString(line) || headingLineRe.MatchString(line) { + continue + } + kept = append(kept, bulletPrefixRe.ReplaceAllString(line, "")) + } + return strings.Join(kept, "\n") +} + +// sanitizeSlotLine normalizes a slot answer into exactly one clean line. A model +// asked for a single line still returns bullets, headings, fences, an enumeration +// of everything in a file or a rambling paragraph, and none of that may reach the +// page. The shape of the line is therefore decided here rather than trusted to the +// model: the first meaningful line is taken whole when the model bulleted it and +// only up to its first sentence otherwise, the result is bounded to +// slotLineWordBudget words, and every remaining line is dropped. Dropping the tail +// is intentional: the heading and the structure around the line are already fixed, +// so anything after the first unit is duplication. Trailing sentence punctuation is +// trimmed so the line reads as a fragment under its heading. +func sanitizeSlotLine(content string) string { + for _, raw := range strings.Split(content, "\n") { + line := strings.TrimSpace(raw) + if line == "" || fenceLineRe.MatchString(line) || headingLineRe.MatchString(line) { + continue + } + bulleted := bulletPrefixRe.MatchString(line) + line = bulletPrefixRe.ReplaceAllString(line, "") + if !bulleted { + line = firstSentence(line) + } + return boundLineLength(strings.TrimSpace(line)) + } + return "" +} + +// boundLineLength cuts a line to slotLineWordBudget words. A cut lands on the last +// sentence or clause boundary at or before the budget, and on a word boundary when +// the text offers none, so a word is never split. +func boundLineLength(line string) string { + words := strings.Fields(line) + if len(words) <= slotLineWordBudget { + return strings.Trim(strings.Join(words, " "), trailingPunctTrim) + } + bounded := strings.Join(words[:slotLineWordBudget], " ") + if boundary := lastBoundary(bounded); boundary > 0 { + return strings.Trim(bounded[:boundary], trailingPunctTrim) + } + return strings.Trim(bounded, trailingPunctTrim) +} + +// firstCompleteSentence returns the leading sentence of text including its +// terminator, or an empty string when the text holds none. It is the salvage path +// for an answer that stopped on the generation limit: a complete sentence is a +// valid instance of a one-line contract, a clipped fragment is not. +func firstCompleteSentence(text string) string { + trimmed := strings.TrimSpace(text) + if end := firstBoundary(trimmed, sentenceEndRunes); end > 0 { + return trimmed[:end] + } + return "" +} + +// lastCompleteSentence returns text up to and including its last complete +// sentence, or an empty string when the text holds none. It is the salvage path +// for a clipped paragraph: the whole prefix is still a well-formed paragraph, so +// only the trailing fragment is dropped. +func lastCompleteSentence(text string) string { + trimmed := strings.TrimSpace(text) + if end := lastBoundaryOf(trimmed, sentenceEndRunes); end > 0 { + return trimmed[:end] + } + return "" +} + +// firstSentence returns the leading sentence of text, or the whole text when it +// holds no terminator. +func firstSentence(text string) string { + if end := firstBoundary(text, sentenceEndRunes); end > 0 { + return text[:end] + } + return text +} + +// firstBoundary returns the index just past the first terminator of text that ends +// a sentence or clause there, or 0 when there is none. +func firstBoundary(text string, terminators string) int { + for i := 0; i < len(text); i++ { + if strings.IndexByte(terminators, text[i]) >= 0 && isBoundaryAt(text, i) { + return i + 1 + } + } + return 0 +} + +// lastBoundary returns the index just past the last sentence or clause boundary of +// text, or 0 when it has none. +func lastBoundary(text string) int { + return lastBoundaryOf(text, sentenceEndRunes+clauseEndRunes) +} + +// lastBoundaryOf returns the index just past the last terminator of text that ends +// a unit there, or 0 when it has none. +func lastBoundaryOf(text, terminators string) int { + best := 0 + for i := 0; i < len(text); i++ { + if strings.IndexByte(terminators, text[i]) >= 0 && isBoundaryAt(text, i) { + best = i + 1 + } + } + return best +} + +// isBoundaryAt reports whether the terminator at index i really ends a unit there. +// It must be followed by whitespace or the end of the text, and a period that closes +// a single letter is an abbreviation such as "e.g." rather than a sentence end. +func isBoundaryAt(text string, i int) bool { + if i+1 < len(text) { + if !strings.ContainsRune(" \t\n", rune(text[i+1])) { + return false + } + } + if text[i] == '.' { + return !isAbbreviationPeriod(text, i) + } + return true +} + +// isAbbreviationPeriod reports whether the period at index i closes a single +// letter, as in "e.g." or "i.e.", rather than a sentence. +func isAbbreviationPeriod(text string, i int) bool { + return i > 0 && isLetter(rune(text[i-1])) && (i < 2 || !isLetter(rune(text[i-2]))) +} + +func isLetter(r rune) bool { + return (r >= 'a' && r <= 'z') || (r >= 'A' && r <= 'Z') +} + +// sanitizeSlotParagraph normalizes a slot answer into exactly one clean paragraph. +func sanitizeSlotParagraph(content string) string { + return collapsedSpaceRe.ReplaceAllString(stripSlotNoise(content), " ") +} diff --git a/internal/engine/orchestrator.go b/internal/engine/orchestrator.go index af0b55e..a71f7fb 100644 --- a/internal/engine/orchestrator.go +++ b/internal/engine/orchestrator.go @@ -33,28 +33,32 @@ type orchestrator struct { client llmCaller } -func (o *orchestrator) GenerateStandardDocs(ctx context.Context, repoRoot, docsDir, rootSum string, cfg *config.Config, logEvent LogEventFunc) error { - logEvent(EventStatus, "Step 3: Global Architecture Synthesis...") - archPath := filepath.Join(docsDir, "architecture.md") - archMsg := fmt.Sprintf("Write the global architecture overview (%s/architecture.md) based on the root summary.\n\n%s", docsDir, rootSum) - sysPrompt := cfg.SystemPrompt + "\n" + cfg.ArchitecturePrompt - archDoc, err := o.client.CallLLM(ctx, sysPrompt, []Message{{Role: "user", Content: archMsg}}, false) +// generateStandardPage builds one standard page from source, writes it to docsDir +// and returns the rendered content so a later page can be derived from it. +func (o *orchestrator) generateStandardPage(ctx context.Context, repoRoot, docsDir string, page standardDocPage, source string, cfg *config.Config, logEvent LogEventFunc) (string, error) { + doc, err := buildStandardDocPage(ctx, o.client, cfg, page, source, logEvent) if err != nil { - return fmt.Errorf("failed to generate global architecture: %w", err) + return "", fmt.Errorf("failed to generate the %s page: %w", page.title, err) } - if err := tools.WriteFileSafely(repoRoot, archPath, []byte(stripOuterMarkdownFence(archDoc))); err != nil { - return fmt.Errorf("failed to write architecture.md: %w", err) + if err := tools.WriteFileSafely(repoRoot, filepath.Join(docsDir, page.fileName), []byte(doc)); err != nil { + return "", fmt.Errorf("failed to write %s: %w", page.fileName, err) } + return doc, nil +} - logEvent(EventStatus, "Step 4: Generating Quickstart...") - qsPath := filepath.Join(docsDir, "quickstart.md") - qsMsg := fmt.Sprintf("Write the %s/quickstart.md page based on this architecture.\n\n%s", docsDir, rootSum) - qsDoc, err := o.client.CallLLM(ctx, sysPrompt, []Message{{Role: "user", Content: qsMsg}}, false) +// GenerateStandardDocs writes the two standard pages. The architecture page is +// built from the root module summary, and the quickstart is built from the +// architecture page that was just generated, so the two never read the same input. +func (o *orchestrator) GenerateStandardDocs(ctx context.Context, repoRoot, docsDir, rootSum string, cfg *config.Config, logEvent LogEventFunc) error { + logEvent(EventStatus, "Step 3: Global Architecture Synthesis...") + archDoc, err := o.generateStandardPage(ctx, repoRoot, docsDir, architecturePage, rootSum, cfg, logEvent) if err != nil { - return fmt.Errorf("failed to generate quickstart documentation: %w", err) + return err } - if err := tools.WriteFileSafely(repoRoot, qsPath, []byte(stripOuterMarkdownFence(qsDoc))); err != nil { - return fmt.Errorf("failed to write quickstart.md: %w", err) + + logEvent(EventStatus, "Step 4: Generating Quickstart...") + if _, err := o.generateStandardPage(ctx, repoRoot, docsDir, quickstartPage, archDoc, cfg, logEvent); err != nil { + return err } return nil @@ -157,7 +161,10 @@ func (o *orchestrator) RunInit(ctx context.Context, repoRoot string, cfg *config } cache.StepsHash = computeStepsHash(cfg.ExtractionSteps) - codeFiles, err := tools.DiscoverCodeFiles(repoRoot, ignores) + codeFiles, err := tools.DiscoverCodeFiles(repoRoot, tools.DiscoveryOptions{ + Ignores: ignores, + IncludeTests: cfg.IncludeTests, + }) if err != nil { return err } @@ -227,8 +234,11 @@ func (o *orchestrator) getAffectedDirs(repoRoot, docsDir string, tree *DirNode, return propagateAffected(tree, affectedDirs) } -func (o *orchestrator) detectFileChanges(repoRoot string, ignores []string, cache *MetadataCache, logEvent LogEventFunc) ([]FileChange, map[string]string, []string, error) { - codeFiles, err := tools.DiscoverCodeFiles(repoRoot, ignores) +func (o *orchestrator) detectFileChanges(repoRoot string, ignores []string, cache *MetadataCache, includeTests bool, logEvent LogEventFunc) ([]FileChange, map[string]string, []string, error) { + codeFiles, err := tools.DiscoverCodeFiles(repoRoot, tools.DiscoveryOptions{ + Ignores: ignores, + IncludeTests: includeTests, + }) if err != nil { return nil, nil, nil, err } @@ -290,7 +300,7 @@ func (o *orchestrator) RunUpdate(ctx context.Context, repoRoot string, cfg *conf logEvent(EventStatus, "Step 1: Detecting changed files...") - filteredChanges, currentFilesMap, allowedCodeFiles, err := o.detectFileChanges(repoRoot, ignores, cache, logEvent) + filteredChanges, currentFilesMap, allowedCodeFiles, err := o.detectFileChanges(repoRoot, ignores, cache, cfg.IncludeTests, logEvent) if err != nil { return err } diff --git a/internal/engine/orchestrator_test.go b/internal/engine/orchestrator_test.go index 1247669..7351464 100644 --- a/internal/engine/orchestrator_test.go +++ b/internal/engine/orchestrator_test.go @@ -2,8 +2,10 @@ package engine import ( "context" + "errors" "os" "path/filepath" + "strings" "testing" "github.com/arrase/code-reducer/internal/config" @@ -82,9 +84,348 @@ func TestOrchestratorRunInit(t *testing.T) { ExtractionSteps: config.DefaultExtractionSteps, } - o := &orchestrator{client: &mockLLMCaller{numCtx: 2048}} + o := &orchestrator{client: &scriptedLLM{numCtx: 2048}} err := o.RunInit(context.Background(), repoRoot, cfg, func(e Event) {}) if err != nil { t.Fatalf("unexpected error running RunInit: %v", err) } } + +func TestOrchestratorRunInitSkipsTestFiles(t *testing.T) { + repoRoot := t.TempDir() + for _, f := range []string{"main.go", "main_test.go"} { + if err := os.WriteFile(filepath.Join(repoRoot, f), []byte("package main\n"), 0644); err != nil { + t.Fatalf("failed to write %s: %v", f, err) + } + } + + cfg := &config.Config{DocsDir: "docs", ExtractionSteps: config.DefaultExtractionSteps} + + o := &orchestrator{client: &scriptedLLM{numCtx: 2048}} + if err := o.RunInit(context.Background(), repoRoot, cfg, func(e Event) {}); err != nil { + t.Fatalf("unexpected error running RunInit: %v", err) + } + + cache, err := loadMetadataCache(repoRoot, cfg.DocsDir) + if err != nil { + t.Fatalf("failed to load metadata cache: %v", err) + } + if _, ok := cache.Files["main_test.go"]; ok { + t.Error("expected the test file to stay out of the cache") + } + if _, ok := cache.Files["main.go"]; !ok { + t.Error("expected the source file to be documented") + } +} + +func TestRunUpdatePrunesTestFilesWhenTheyBecomeExcluded(t *testing.T) { + repoRoot := t.TempDir() + docsDir := "docs" + if err := os.MkdirAll(filepath.Join(repoRoot, docsDir, "modules"), 0755); err != nil { + t.Fatalf("failed to create modules dir: %v", err) + } + + sourceFile := "main.go" + testFile := "main_test.go" + for _, f := range []string{sourceFile, testFile} { + if err := os.WriteFile(filepath.Join(repoRoot, f), []byte("package main\n"), 0644); err != nil { + t.Fatalf("failed to write %s: %v", f, err) + } + } + + steps := []config.ExtractionStep{{Name: "API_SIGNATURES", Prompt: "extract"}} + sourceHash, err := computeSHA256(repoRoot, sourceFile) + if err != nil { + t.Fatalf("failed to hash the source file: %v", err) + } + cache := &MetadataCache{ + Version: currentCacheVersion, + StepsHash: computeStepsHash(steps), + Files: map[string]FileCacheEntry{ + sourceFile: {SHA256: sourceHash, Facts: "cached source facts"}, + testFile: {SHA256: "old-test-hash", Facts: "stale test facts"}, + }, + Modules: map[string]string{".": "# Module: .\n\n### File: main_test.go\nStale test facts"}, + } + if err := saveMetadataCache(repoRoot, docsDir, cache); err != nil { + t.Fatalf("failed to seed metadata cache: %v", err) + } + rootModule := filepath.Join(repoRoot, docsDir, "modules", "README.md") + if err := os.WriteFile(rootModule, []byte(cache.Modules["."]), 0644); err != nil { + t.Fatalf("failed to seed the root module page: %v", err) + } + + cfg := &config.Config{DocsDir: docsDir, ExtractionSteps: steps, IncludeTests: false} + o := &orchestrator{client: &scriptedLLM{numCtx: 2048}} + if err := o.RunUpdate(context.Background(), repoRoot, cfg, func(e Event) {}); err != nil { + t.Fatalf("unexpected error running RunUpdate: %v", err) + } + + updated, err := loadMetadataCache(repoRoot, docsDir) + if err != nil { + t.Fatalf("failed to load metadata cache: %v", err) + } + if _, ok := updated.Files[testFile]; ok { + t.Errorf("expected the cache entry for %s to be pruned, got %v", testFile, updated.Files) + } + if _, ok := updated.Files[sourceFile]; !ok { + t.Error("expected the source file to keep its cache entry") + } + + page, err := os.ReadFile(rootModule) + if err != nil { + t.Fatalf("failed to read the root module page: %v", err) + } + if strings.Contains(string(page), testFile) { + t.Errorf("expected %s to be dropped from the root module page, got:\n%s", testFile, page) + } + if !strings.Contains(string(page), sourceFile) { + t.Errorf("expected %s in the regenerated root module page, got:\n%s", sourceFile, page) + } +} + +func TestOrchestratorRunInitKeepsTheSystemMessageConstant(t *testing.T) { + repoRoot := t.TempDir() + if err := os.WriteFile(filepath.Join(repoRoot, "main.go"), []byte("package main\n"), 0644); err != nil { + t.Fatalf("failed to write test file: %v", err) + } + + steps := config.DefaultExtractionSteps[:2] + cfg := &config.Config{ + DocsDir: "docs", + ExtractionSteps: steps, + CharsPerToken: config.CharsPerTokenDefault, + SystemPrompt: "SHARED SYSTEM PROMPT", + ModuleSynthesisPrompt: "MODULE PAGE STYLE", + ArchitecturePrompt: "STANDARD PAGE STYLE", + } + + client := &scriptedLLM{numCtx: 2048} + o := &orchestrator{client: client} + if err := o.RunInit(context.Background(), repoRoot, cfg, func(e Event) {}); err != nil { + t.Fatalf("unexpected error running RunInit: %v", err) + } + + if len(client.systems) == 0 { + t.Fatal("expected the run to call the LLM") + } + for i, system := range client.systems { + if system != cfg.SystemPrompt { + t.Errorf("call %d: system message = %q, want %q", i, system, cfg.SystemPrompt) + } + } + + seen := make(map[string]bool, len(client.prompts)) + for i, prompt := range client.prompts { + if seen[prompt] { + t.Errorf("call %d repeated the user message of an earlier call: %q", i, prompt) + } + seen[prompt] = true + } + + for _, step := range steps { + if got := countCallsContaining(client.prompts, step.Prompt); got != 1 { + t.Errorf("expected exactly one call carrying step %q in the user turn, got %d", step.Name, got) + } + } + if got := countCallsContaining(client.prompts, cfg.ModuleSynthesisPrompt); got != 4 { + t.Errorf("expected the 4 module page slots to carry the module style, got %d calls", got) + } + if got := countCallsContaining(client.prompts, cfg.ArchitecturePrompt); got != 6 { + t.Errorf("expected the 6 standard page sections to carry the page style, got %d calls", got) + } + for _, varying := range append([]string{cfg.ModuleSynthesisPrompt, cfg.ArchitecturePrompt}, steps[0].Prompt, steps[1].Prompt) { + if strings.Contains(cfg.SystemPrompt, varying) { + t.Errorf("expected %q to stay out of the system message", varying) + } + } +} + +// countCallsContaining counts the recorded user messages that contain needle. +func countCallsContaining(prompts []string, needle string) int { + count := 0 + for _, p := range prompts { + if strings.Contains(p, needle) { + count++ + } + } + return count +} + +func TestGenerateStandardDocsDerivesQuickstartFromArchitecture(t *testing.T) { + cfg := &config.Config{ + DocsDir: "docs", + CharsPerToken: config.CharsPerTokenDefault, + SystemPrompt: "system", + ArchitecturePrompt: "architecture", + } + client := &scriptedLLM{numCtx: 2048, results: []LLMResult{ + stopResult("ARCH overview prose"), + stopResult("ARCH boundary prose"), + stopResult("ARCH interaction prose"), + stopResult("QS purpose prose"), + stopResult("QS layout prose"), + stopResult("QS workflow prose"), + }} + + repoRoot := t.TempDir() + if err := os.MkdirAll(filepath.Join(repoRoot, cfg.DocsDir), 0755); err != nil { + t.Fatalf("failed to create docs dir: %v", err) + } + + o := &orchestrator{client: client} + if err := o.GenerateStandardDocs(context.Background(), repoRoot, cfg.DocsDir, "RAW ROOT SUMMARY", cfg, func(t EventType, msg string) {}); err != nil { + t.Fatalf("unexpected error: %v", err) + } + + if len(client.prompts) != len(architecturePage.sections)+len(quickstartPage.sections) { + t.Fatalf("expected one call per section, got %d calls", len(client.prompts)) + } + + for i, section := range architecturePage.sections { + prompt := client.prompts[i] + if !strings.Contains(prompt, section.instruction) || !strings.Contains(prompt, architecturePage.sourceLabel) { + t.Errorf("architecture section %q did not receive its own instruction and the root summary, got:\n%s", section.heading, prompt) + } + if !strings.Contains(prompt, "RAW ROOT SUMMARY") { + t.Errorf("architecture section %q was not built from the root summary, got:\n%s", section.heading, prompt) + } + } + + for i, section := range quickstartPage.sections { + prompt := client.prompts[len(architecturePage.sections)+i] + if !strings.Contains(prompt, section.instruction) || !strings.Contains(prompt, quickstartPage.sourceLabel) { + t.Errorf("quickstart section %q did not receive its own instruction and the architecture page, got:\n%s", section.heading, prompt) + } + if strings.Contains(prompt, "RAW ROOT SUMMARY") { + t.Errorf("quickstart section %q was built from the raw root summary, got:\n%s", section.heading, prompt) + } + for _, archSection := range architecturePage.sections { + if !strings.Contains(prompt, archSection.heading) { + t.Errorf("quickstart section %q is not derived from the architecture page, missing %q, got:\n%s", section.heading, archSection.heading, prompt) + } + } + } + + quick, err := os.ReadFile(filepath.Join(repoRoot, cfg.DocsDir, quickstartPage.fileName)) + if err != nil { + t.Fatalf("failed to read quickstart.md: %v", err) + } + for _, want := range []string{"QS purpose prose", "QS layout prose", "QS workflow prose"} { + if !strings.Contains(string(quick), want) { + t.Errorf("expected %q in quickstart.md, got:\n%s", want, quick) + } + } + if strings.Contains(string(quick), "ARCH overview prose") { + t.Errorf("expected the architecture prose to stay in architecture.md, got:\n%s", quick) + } +} + +func TestGenerateStandardDocsRejectsIncompleteLLMOutput(t *testing.T) { + cfg := &config.Config{DocsDir: "docs", SystemPrompt: "system", ArchitecturePrompt: "architecture"} + + cases := []struct { + name string + results []LLMResult + wantErr error + archWritten bool + }{ + { + name: "truncated first architecture section", + results: []LLMResult{truncatedResult("partial arch")}, + wantErr: ErrTruncatedOutput, + }, + { + name: "empty architecture section", + results: []LLMResult{stopResult(" \n ")}, + wantErr: ErrEmptyOutput, + }, + { + name: "truncated last architecture section", + results: []LLMResult{stopResult("arch"), stopResult("arch"), truncatedResult("partial arch")}, + wantErr: ErrTruncatedOutput, + }, + { + name: "truncated quickstart section", + results: []LLMResult{stopResult("arch"), stopResult("arch"), stopResult("arch"), truncatedResult("partial quick")}, + wantErr: ErrTruncatedOutput, + archWritten: true, + }, + { + name: "empty quickstart section", + results: []LLMResult{stopResult("arch"), stopResult("arch"), stopResult("arch"), stopResult("")}, + wantErr: ErrEmptyOutput, + archWritten: true, + }, + } + + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + repoRoot := t.TempDir() + if err := os.MkdirAll(filepath.Join(repoRoot, cfg.DocsDir), 0755); err != nil { + t.Fatalf("failed to create docs dir: %v", err) + } + + o := &orchestrator{client: &scriptedLLM{numCtx: 2048, results: tc.results}} + err := o.GenerateStandardDocs(context.Background(), repoRoot, cfg.DocsDir, "root summary", cfg, func(t EventType, msg string) {}) + if !errors.Is(err, tc.wantErr) { + t.Fatalf("expected %v, got %v", tc.wantErr, err) + } + + qsPath := filepath.Join(repoRoot, cfg.DocsDir, "quickstart.md") + if _, err := os.Stat(qsPath); !os.IsNotExist(err) { + t.Errorf("expected %s not to be written", qsPath) + } + + archPath := filepath.Join(repoRoot, cfg.DocsDir, "architecture.md") + _, archErr := os.Stat(archPath) + if archExists := archErr == nil; archExists != tc.archWritten { + t.Fatalf("expected architecture.md existence to be %v, got %v", tc.archWritten, archExists) + } + }) + } +} + +func TestGenerateStandardDocsWritesFixedSkeletons(t *testing.T) { + cfg := &config.Config{DocsDir: "docs", SystemPrompt: "system", ArchitecturePrompt: "architecture"} + + repoRoot := t.TempDir() + if err := os.MkdirAll(filepath.Join(repoRoot, cfg.DocsDir), 0755); err != nil { + t.Fatalf("failed to create docs dir: %v", err) + } + + o := &orchestrator{client: &scriptedLLM{numCtx: 2048}} + if err := o.GenerateStandardDocs(context.Background(), repoRoot, cfg.DocsDir, "# Module: .", cfg, func(t EventType, msg string) {}); err != nil { + t.Fatalf("unexpected error: %v", err) + } + + arch, err := os.ReadFile(filepath.Join(repoRoot, cfg.DocsDir, "architecture.md")) + if err != nil { + t.Fatalf("failed to read architecture.md: %v", err) + } + quick, err := os.ReadFile(filepath.Join(repoRoot, cfg.DocsDir, "quickstart.md")) + if err != nil { + t.Fatalf("failed to read quickstart.md: %v", err) + } + + assertHeadings(t, "architecture.md", string(arch), []string{ + "# Architecture", "## Overview", "## System Boundaries", "## Module Interaction", + }) + assertHeadings(t, "quickstart.md", string(quick), []string{ + "# Quickstart", "## What This Project Does", "## Project Layout", "## Common Workflows", + }) +} + +// assertHeadings checks that the document contains exactly the expected headings, in order. +func assertHeadings(t *testing.T, name, doc string, want []string) { + t.Helper() + var got []string + for _, line := range strings.Split(doc, "\n") { + if strings.HasPrefix(line, "#") { + got = append(got, line) + } + } + if strings.Join(got, "|") != strings.Join(want, "|") { + t.Fatalf("%s headings = %v, want %v", name, got, want) + } +} diff --git a/internal/engine/pages.go b/internal/engine/pages.go new file mode 100644 index 0000000..240a942 --- /dev/null +++ b/internal/engine/pages.go @@ -0,0 +1,486 @@ +package engine + +import ( + "context" + "fmt" + "strings" + + "github.com/arrase/code-reducer/internal/config" +) + +// slotKind selects how a slot answer is normalized. +type slotKind int + +const ( + // slotLine is a micro-task whose answer must collapse to one clean line. + slotLine slotKind = iota + // slotParagraph is a micro-task whose answer must collapse to one clean paragraph. + slotParagraph +) + +// docSlot is one bounded LLM micro-task. The page structure is owned by the code, +// so a slot carries only what a single call needs: the format contract, the minimal +// context to answer it, and the subject used for logging and error reporting. +type docSlot struct { + kind slotKind + subject string + prompt string + logMsg string +} + +// namedSection is a heading the code owns together with the prose that fills it. +type namedSection struct { + heading string + text string +} + +// slotRunner fills prose slots sequentially against a single LLM client. Every slot +// is an independent, hard-bounded call: no slot shares a generation budget with +// another, and a failure aborts the page instead of leaving a section out. +type slotRunner struct { + ctx context.Context + client llmCaller + system string + style string + cfg *config.Config + logEvent LogEventFunc +} + +// newSlotRunner builds a runner for prose styled by the given page style prompt. +// The style prompt opens the user turn rather than the system message, so the +// system message is cfg.SystemPrompt on every call of a run and the style prompt +// stays the shared prefix of every slot of this page kind. +func newSlotRunner(ctx context.Context, c llmCaller, cfg *config.Config, stylePrompt string, logEvent LogEventFunc) slotRunner { + return slotRunner{ + ctx: ctx, + client: c, + system: cfg.SystemPrompt, + style: stylePrompt, + cfg: cfg, + logEvent: logEvent, + } +} + +// numPredict returns the generation bound of a slot kind. The bound is selected at +// call time because a page mixes line slots and paragraph slots, and the two need +// very different amounts of room. +func (r slotRunner) numPredict(kind slotKind) int { + return slotNumPredict(r.cfg, kind) +} + +// outputReserve returns the context tokens a prompt payload must leave for +// generation: the largest bound any slot of this page may ask for, so sizing a +// payload against a line bound could leave a paragraph slot no room to answer. +func (r slotRunner) outputReserve() int { + bound := r.numPredict(slotLine) + if paragraph := r.numPredict(slotParagraph); paragraph > bound { + bound = paragraph + } + return bound +} + +// slotNumPredict returns the generation bound of a prose slot of the given kind. +// A line slot is trimmed by the code to its first sentence and to a 40 word budget, +// so it only needs room for that first sentence, while a paragraph slot keeps +// every sentence the model wrote and needs the larger bound. A lower explicit +// client-wide cap wins, so a user who lowers num_predict never gets a larger budget +// back from a slot default. +func slotNumPredict(cfg *config.Config, kind slotKind) int { + budget := cfg.ParagraphNumPredict + if kind == slotLine { + budget = cfg.SlotNumPredict + } + if budget <= 0 { + budget = slotNumPredictDefault(kind) + } + if cfg.NumPredict > 0 && cfg.NumPredict < budget { + return cfg.NumPredict + } + return budget +} + +// slotNumPredictDefault returns the shipped generation bound of a slot kind. +func slotNumPredictDefault(kind slotKind) int { + if kind == slotLine { + return config.SlotNumPredictDefault + } + return config.ParagraphNumPredictDefault +} + +// fill runs one slot micro-task and returns the normalized prose for it. +func (r slotRunner) fill(slot docSlot) (string, error) { + if err := r.ctx.Err(); err != nil { + return "", err + } + r.logEvent(EventStatus, slot.logMsg) + + res, err := r.client.CallLLMWithNumPredict(r.ctx, r.system, []Message{{Role: "user", Content: userMessage(r.style, slot.prompt)}}, r.numPredict(slot.kind)) + if err != nil { + return "", fmt.Errorf("%s: %w", slot.subject, err) + } + + var content string + if salvaged := slot.kind.salvage(res); salvaged != "" { + r.logEvent(EventStatus, fmt.Sprintf(" ↳ %s of an over-long answer: %s", slot.kind.salvagedSpan(), slot.subject)) + content = salvaged + } else { + content, err = requireCompleteContent(res, slot.subject) + if err != nil { + return "", err + } + } + + text := slot.kind.sanitize(content) + if text == "" { + return "", fmt.Errorf("%s: %w: the answer carried no prose once formatting was normalized", slot.subject, ErrEmptyOutput) + } + return text, nil +} + +// salvage returns the usable part of a slot answer that stopped on the generation +// limit, and an empty string for every other case. A line slot keeps its first +// complete sentence and a paragraph slot keeps everything up to its last complete +// sentence, because both prefixes are well-formed instances of what the slot asked +// for: aborting a whole document because the model wrote too much is the wrong +// failure mode. The call was bounded, so the salvage costs no more tokens than the +// answer it replaces. An answer with no complete sentence yields nothing, which +// sends the slot to the hard failure requireCompleteContent reports. +func (k slotKind) salvage(res LLMResult) string { + if res.DoneReason != doneReasonLength { + return "" + } + content := stripOuterMarkdownFence(res.Content) + if k == slotLine { + return firstCompleteSentence(content) + } + return lastCompleteSentence(content) +} + +// salvagedSpan names the prefix salvage kept, so the log line reads correctly for +// both slot kinds. +func (k slotKind) salvagedSpan() string { + if k == slotLine { + return "kept the first complete sentence" + } + return "kept the text up to the last complete sentence" +} + +// sanitize normalizes a slot answer to the shape its contract requires. +func (k slotKind) sanitize(content string) string { + if k == slotLine { + return sanitizeSlotLine(content) + } + return sanitizeSlotParagraph(content) +} + +// moduleComponent is one Go-owned entry of a module Components section: the identity +// the code assigns to a file or a child subsystem, plus the facts a slot call is +// allowed to read. A subsystem carries the rendered page of the child module. +type moduleComponent struct { + name string + facts string + subsystem bool +} + +// moduleComponentLine is a component identity together with its generated prose. +type moduleComponentLine struct { + name string + text string +} + +// modulePageParts is the full set of prose slots of a module page. +type modulePageParts struct { + responsibility string + components []moduleComponentLine + dataFlow string + errorHandling string +} + +// parseModuleComponents turns the collectComponents output into ordered components. +// The identities are kept exactly as collectComponents wrote them, which is what +// makes the rendered headings deterministic. +func parseModuleComponents(components []string) []moduleComponent { + parsed := make([]moduleComponent, 0, len(components)) + for _, raw := range components { + heading, facts, _ := strings.Cut(raw, "\n") + name := strings.TrimSpace(strings.TrimPrefix(strings.TrimSpace(heading), markdownHeadingPrefix)) + parsed = append(parsed, moduleComponent{ + name: name, + facts: strings.TrimSpace(facts), + subsystem: strings.HasPrefix(name, subsystemPrefix), + }) + } + return parsed +} + +// moduleShape reports whether a node has child subsystems. A module with children is +// documented at the boundary between them, which is a different question from the +// one a leaf module answers about its own files. +func moduleShape(components []moduleComponent) moduleNodeShape { + for _, c := range components { + if c.subsystem { + return shapeHierarchical + } + } + return shapeLeaf +} + +// moduleNodeShape selects the framing of the module page slots. +type moduleNodeShape int + +const ( + // shapeLeaf is a module made of files only. + shapeLeaf moduleNodeShape = iota + // shapeHierarchical is a module that owns child subsystems. + shapeHierarchical +) + +// readComponents returns the components as the slots of this page may read them. A +// child subsystem is reduced to the head of its own page: its Data Flow and Error +// Handling paragraphs describe the child's interior, which the child page already +// documents, and handing them to the parent is what makes a parent restate its +// children. File facts are untouched, so a leaf module reads exactly as before. +func readComponents(components []moduleComponent) []moduleComponent { + read := make([]moduleComponent, len(components)) + copy(read, components) + for i := range read { + if read[i].subsystem { + read[i].facts = childIdentityFacts(read[i].facts) + } + } + return read +} + +// childIdentityFacts keeps the part of a child module page a parent needs to reason +// about the child: its title, its responsibility and its component entries. +func childIdentityFacts(page string) string { + head, _, found := strings.Cut(page, "\n## "+headingDataFlow+"\n") + if !found { + return page + } + return strings.TrimSpace(head) +} + +// buildModulePage fills every prose slot of a module page with one bounded +// micro-task per slot and renders the result from the fixed skeleton. The caller +// must supply at least one component, which synthesizeNode guarantees. +func buildModulePage(ctx context.Context, p *pipelineState, nodePath string, components []moduleComponent) (string, error) { + runner := newSlotRunner(ctx, p.client, p.cfg, p.cfg.ModuleSynthesisPrompt, p.logEvent) + promptBudget := promptCharBudget(p.client.NumCtx(), runner.outputReserve(), p.cfg.CharsPerToken) + readable := readComponents(components) + identities := componentIdentities(readable) + digest := componentFactsDigest(readable, promptBudget) + dataFlowPrompt, errorHandlingPrompt := moduleParagraphPrompts(nodePath, identities, digest, moduleShape(components)) + + responsibility, err := runner.fill(docSlot{ + kind: slotLine, + subject: fmt.Sprintf("responsibility of module %s", nodePath), + prompt: fmt.Sprintf("Describe the single responsibility of this module in one line.\n\nModule: %s\nComponents:\n%s", + nodePath, identities), + logMsg: fmt.Sprintf("➜ Describing responsibility of %s", nodePath), + }) + if err != nil { + return "", err + } + + lines := make([]moduleComponentLine, 0, len(readable)) + for _, c := range readable { + text, err := runner.fill(docSlot{ + kind: slotLine, + subject: fmt.Sprintf("component %s of module %s", c.name, nodePath), + prompt: fmt.Sprintf("Describe this single component in one line.\n\nModule: %s\nComponent: %s\nFacts:\n%s", + nodePath, c.name, truncateRunes(c.facts, promptBudget)), + logMsg: fmt.Sprintf("➜ Describing component %s of %s", c.name, nodePath), + }) + if err != nil { + return "", err + } + lines = append(lines, moduleComponentLine{name: c.name, text: text}) + } + + dataFlow, err := runner.fill(docSlot{ + kind: slotParagraph, + subject: fmt.Sprintf("data flow of module %s", nodePath), + prompt: dataFlowPrompt, + logMsg: fmt.Sprintf("➜ Describing data flow of %s", nodePath), + }) + if err != nil { + return "", err + } + + errorHandling, err := runner.fill(docSlot{ + kind: slotParagraph, + subject: fmt.Sprintf("error handling of module %s", nodePath), + prompt: errorHandlingPrompt, + logMsg: fmt.Sprintf("➜ Describing error handling of %s", nodePath), + }) + if err != nil { + return "", err + } + + return renderModulePage(nodePath, modulePageParts{ + responsibility: responsibility, + components: lines, + dataFlow: dataFlow, + errorHandling: errorHandling, + }), nil +} + +// moduleParagraphPrompts returns the Data Flow and Error Handling instructions of a +// module page. A module that owns subsystems is documented across the boundary +// between them: what flows from one subsystem to the next, and how errors cross it. +// A leaf module has no such boundary and keeps the per-component framing. Both +// variants see the same identities and the same digest, so the choice is a framing +// decision and not a change of evidence. +func moduleParagraphPrompts(nodePath, identities, digest string, shape moduleNodeShape) (string, string) { + if shape == shapeHierarchical { + return fmt.Sprintf("Describe what flows between the components of this module and what each one hands to the next, in a short paragraph. Each component has its own page, so never restate what a component does internally.\n\nModule: %s\nComponents:\n%s\nFacts:\n%s", + nodePath, identities, digest), + fmt.Sprintf("Describe how errors cross the boundaries between the components of this module and how they reach the caller, in a short paragraph. Each component page covers its own error handling, so do not repeat it.\n\nModule: %s\nComponents:\n%s\nFacts:\n%s", + nodePath, identities, digest) + } + return fmt.Sprintf("Describe how data moves through this module in a short paragraph.\n\nModule: %s\nComponents:\n%s\nFacts:\n%s", + nodePath, identities, digest), + fmt.Sprintf("Describe how this module reports and handles errors in a short paragraph.\n\nModule: %s\nComponents:\n%s\nFacts:\n%s", + nodePath, identities, digest) +} + +// renderModulePage assembles a module page from the fixed skeleton. The title, the +// section headings and their order, and the component names and their order are all +// decided here and never by the model. +func renderModulePage(nodePath string, parts modulePageParts) string { + sections := []namedSection{ + {heading: headingResponsibility, text: parts.responsibility}, + {heading: headingComponents, text: renderComponentBlock(parts.components)}, + {heading: headingDataFlow, text: parts.dataFlow}, + {heading: headingErrorHandling, text: parts.errorHandling}, + } + return renderDocPage("Module: "+nodePath, sections) +} + +// renderComponentBlock renders the nested component entries of a module page. The +// leading newline separates the block from its section heading. +func renderComponentBlock(components []moduleComponentLine) string { + if len(components) == 0 { + return "" + } + entries := make([]string, 0, len(components)) + for _, c := range components { + entries = append(entries, fmt.Sprintf("%s%s\n%s", markdownHeadingPrefix, c.name, c.text)) + } + return "\n" + strings.Join(entries, "\n\n") +} + +// docSection is one prose slot of a standard documentation page. +type docSection struct { + heading string + instruction string +} + +// standardDocPage describes a standard documentation page: the title that heads +// it, the file it is written to, the sections it owns, and the label of the +// generated content its sections are derived from. +type standardDocPage struct { + title string + fileName string + sourceLabel string + sections []docSection +} + +// architecturePage and quickstartPage define the fixed section order of the two +// standard pages. Their headings come from here, never from the model, and the +// quickstart is derived from the architecture page rather than from the raw root +// module summary the architecture page was built from. +var architecturePage = standardDocPage{ + title: "Architecture", + fileName: "architecture.md", + sourceLabel: "Root module page:", + sections: []docSection{ + {heading: "Overview", instruction: "Describe the purpose of this project in a short paragraph."}, + {heading: "System Boundaries", instruction: "Describe what this project owns and what it leaves to external systems in a short paragraph."}, + {heading: "Module Interaction", instruction: "Describe how the modules of this project interact in a short paragraph."}, + }, +} + +var quickstartPage = standardDocPage{ + title: "Quickstart", + fileName: "quickstart.md", + sourceLabel: "Architecture page:", + sections: []docSection{ + {heading: "What This Project Does", instruction: "Describe what this project does for its users in a short paragraph."}, + {heading: "Project Layout", instruction: "Describe the repository layout and where a new developer should start reading, in a short paragraph."}, + {heading: "Common Workflows", instruction: "Describe the common developer workflows in a short paragraph."}, + }, +} + +// buildStandardDocPage fills every section of a standard page with one bounded +// micro-task per section and renders the page from the fixed skeleton. The source +// is the generated content the page is derived from, bounded to the prompt budget. +func buildStandardDocPage(ctx context.Context, c llmCaller, cfg *config.Config, page standardDocPage, source string, logEvent LogEventFunc) (string, error) { + runner := newSlotRunner(ctx, c, cfg, cfg.ArchitecturePrompt, logEvent) + budget := promptCharBudget(c.NumCtx(), runner.outputReserve(), cfg.CharsPerToken) + summaries := truncateRunes(source, budget) + + parts := make([]namedSection, 0, len(page.sections)) + for _, s := range page.sections { + text, err := runner.fill(docSlot{ + kind: slotParagraph, + subject: fmt.Sprintf("%s section of the %s page", s.heading, page.title), + prompt: fmt.Sprintf("%s\n\n%s\n%s", s.instruction, page.sourceLabel, summaries), + logMsg: fmt.Sprintf("➜ Writing the %s section of the %s page", s.heading, page.title), + }) + if err != nil { + return "", err + } + parts = append(parts, namedSection{heading: s.heading, text: text}) + } + return renderDocPage(page.title, parts), nil +} + +// renderDocPage assembles a documentation page from a fixed skeleton: the code owns +// the title, the headings and their order, the model only fills the prose. +func renderDocPage(title string, sections []namedSection) string { + var b strings.Builder + fmt.Fprintf(&b, "# %s\n", title) + for _, s := range sections { + fmt.Fprintf(&b, "\n## %s\n%s\n", s.heading, s.text) + } + return b.String() +} + +// componentIdentities renders the component names of a module, without their facts. +func componentIdentities(components []moduleComponent) string { + names := make([]string, 0, len(components)) + for _, c := range components { + names = append(names, "- "+c.name) + } + return strings.Join(names, "\n") +} + +// componentFactsDigest renders the component facts of a module into a payload of at +// most maxChars runes, splitting the budget evenly across components so a whole +// module slot never re-sends an entire subtree. +func componentFactsDigest(components []moduleComponent, maxChars int) string { + if len(components) == 0 { + return "" + } + share := maxChars / len(components) + digest := make([]string, 0, len(components)) + for _, c := range components { + digest = append(digest, fmt.Sprintf("### %s\n%s", c.name, truncateRunes(c.facts, share))) + } + return strings.Join(digest, "\n\n") +} + +// truncateRunes bounds a prompt payload to maxRunes runes. The cut is marked so the +// model never reads a truncated payload as the complete picture. +func truncateRunes(content string, maxRunes int) string { + if maxRunes <= 0 { + return "" + } + runes := []rune(content) + if len(runes) <= maxRunes { + return content + } + return strings.TrimSpace(string(runes[:maxRunes])) + "..." +} diff --git a/internal/engine/pages_test.go b/internal/engine/pages_test.go new file mode 100644 index 0000000..813c47b --- /dev/null +++ b/internal/engine/pages_test.go @@ -0,0 +1,666 @@ +package engine + +import ( + "context" + "errors" + "strings" + "testing" + + "github.com/arrase/code-reducer/internal/config" +) + +func TestRenderModulePageSkeleton(t *testing.T) { + page := renderModulePage("internal/engine", modulePageParts{ + responsibility: "Synthesizes module documentation", + components: []moduleComponentLine{ + {name: "File: client.go", text: "Talks to the Ollama API"}, + {name: "Subsystem: config", text: "Resolves configuration precedence"}, + }, + dataFlow: "File facts flow up the tree.", + errorHandling: "Failures abort the page.", + }) + + want := strings.Join([]string{ + "# Module: internal/engine", + "", + "## Responsibility", + "Synthesizes module documentation", + "", + "## Components", + "", + "### File: client.go", + "Talks to the Ollama API", + "", + "### Subsystem: config", + "Resolves configuration precedence", + "", + "## Data Flow", + "File facts flow up the tree.", + "", + "## Error Handling", + "Failures abort the page.", + "", + }, "\n") + + if page != want { + t.Fatalf("skeleton mismatch\n got:\n%s\nwant:\n%s", page, want) + } +} + +func TestRenderDocPageSkeleton(t *testing.T) { + page := renderDocPage("Architecture", []namedSection{ + {heading: "Overview", text: "Generates a wiki."}, + {heading: "System Boundaries", text: "Only talks to Ollama."}, + }) + + want := "# Architecture\n\n## Overview\nGenerates a wiki.\n\n## System Boundaries\nOnly talks to Ollama.\n" + if page != want { + t.Fatalf("skeleton mismatch\n got:\n%q\nwant:\n%q", page, want) + } +} + +func TestParseModuleComponents(t *testing.T) { + components := []string{ + "### File: client.go\n#### [API_SIGNATURES]\nfunc Call()", + "### Subsystem: config\n# Module: internal/config", + "### File: orphan.go", + } + + parsed := parseModuleComponents(components) + if len(parsed) != 3 { + t.Fatalf("expected 3 components, got %d", len(parsed)) + } + if parsed[0].name != "File: client.go" || parsed[0].facts != "#### [API_SIGNATURES]\nfunc Call()" { + t.Errorf("unexpected first component: %+v", parsed[0]) + } + if parsed[1].name != "Subsystem: config" || parsed[1].facts != "# Module: internal/config" { + t.Errorf("unexpected second component: %+v", parsed[1]) + } + if parsed[2].name != "File: orphan.go" || parsed[2].facts != "" { + t.Errorf("unexpected third component: %+v", parsed[2]) + } +} + +func TestSlotKindSalvage(t *testing.T) { + overLong := truncatedResult("Keeps the first sentence. Drops the rest that never end") + + if got := slotLine.salvage(overLong); got != "Keeps the first sentence." { + t.Errorf("expected the first complete sentence of a line slot, got %q", got) + } + if got := slotParagraph.salvage(overLong); got != "Keeps the first sentence." { + t.Errorf("expected a short paragraph answer to survive whole, got %q", got) + } + if got := slotLine.salvage(stopResult("A complete answer.")); got != "" { + t.Errorf("expected no salvage for a complete answer, got %q", got) + } + if got := slotLine.salvage(truncatedResult("")); got != "" { + t.Errorf("expected no salvage for an empty answer, got %q", got) + } + if got := slotLine.salvage(truncatedResult("Clipped before the first period")); got != "" { + t.Errorf("expected no salvage without a complete sentence, got %q", got) + } +} + +func TestSlotParagraphSalvageKeepsTextUpToTheLastSentence(t *testing.T) { + cases := []struct { + name string + in string + want string + }{ + { + name: "clipped mid paragraph", + in: "Facts flow up the tree. Failures travel back down. The rest of the paragraph is cut off mid", + want: "Facts flow up the tree. Failures travel back down.", + }, + { + name: "abbreviation does not end a sentence", + in: "The flags win, e.g. --slot-num-predict, over the file. The file is read last, i.e. from the repo root. And the tail never", + want: "The flags win, e.g. --slot-num-predict, over the file. The file is read last, i.e. from the repo root.", + }, + { + name: "clause punctuation does not end a sentence", + in: "The flags win, then the file, and the run starts. The tail of this answer never", + want: "The flags win, then the file, and the run starts.", + }, + { + name: "unterminated fence keeps its opening line", + in: "```\nThe flags win. The file is read last. The tail never", + want: "```\nThe flags win. The file is read last.", + }, + {name: "no complete sentence", in: "The tail of this answer never ends on a boundar", want: ""}, + {name: "empty answer", in: " ", want: ""}, + } + + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + if got := slotParagraph.salvage(truncatedResult(tc.in)); got != tc.want { + t.Fatalf("salvage(%q) = %q, want %q", tc.in, got, tc.want) + } + }) + } +} + +func TestSlotParagraphSalvageFeedsSanitizeAWholeParagraph(t *testing.T) { + res := truncatedResult("```\n- The flags win.\n- The file is read last.\n- The tail never") + + text := slotParagraph.sanitize(slotParagraph.salvage(res)) + if text != "The flags win. The file is read last." { + t.Fatalf("expected the salvaged prefix to survive normalization, got %q", text) + } +} + +func TestSlotKindSalvagedSpanNamesTheKeptPrefix(t *testing.T) { + if got := slotLine.salvagedSpan(); got != "kept the first complete sentence" { + t.Errorf("unexpected line salvage log: %q", got) + } + if got := slotParagraph.salvagedSpan(); got != "kept the text up to the last complete sentence" { + t.Errorf("unexpected paragraph salvage log: %q", got) + } +} + +func TestSanitizeSlotContent(t *testing.T) { + cases := []struct { + name string + kind slotKind + in string + want string + }{ + {name: "clean line", kind: slotLine, in: "Calls the Ollama API.", want: "Calls the Ollama API"}, + {name: "bullet line", kind: slotLine, in: "- Calls the Ollama API.", want: "Calls the Ollama API"}, + {name: "numbered line", kind: slotLine, in: "1. Calls the Ollama API", want: "Calls the Ollama API"}, + {name: "multi line", kind: slotLine, in: "Calls the Ollama API\nand caches the result", want: "Calls the Ollama API"}, + {name: "bulleted list", kind: slotLine, in: "- Calls the Ollama API\n- Caches the result\n- Streams the errors", want: "Calls the Ollama API"}, + {name: "multiple sentences", kind: slotLine, in: "Calls the Ollama API. Caches the result. Streams the errors.", want: "Calls the Ollama API"}, + {name: "no terminator", kind: slotLine, in: "Calls the Ollama API and caches the result", want: "Calls the Ollama API and caches the result"}, + {name: "abbreviation", kind: slotLine, in: "Resolves the config, e.g. num_ctx, then starts the run", want: "Resolves the config, e.g. num_ctx, then starts the run"}, + {name: "fenced line", kind: slotLine, in: "```go\nCall()\n```", want: "Call()"}, + {name: "heading line", kind: slotLine, in: "## Responsibility\nCalls the Ollama API", want: "Calls the Ollama API"}, + {name: "only formatting", kind: slotLine, in: "- \n- .\n```", want: ""}, + {name: "paragraph", kind: slotParagraph, in: "- first point\n- second point", want: "first point second point"}, + {name: "paragraph with fence", kind: slotParagraph, in: "```\nA wrapped answer.\n```", want: "A wrapped answer."}, + {name: "empty", kind: slotParagraph, in: " \n\t", want: ""}, + } + + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + if got := tc.kind.sanitize(tc.in); got != tc.want { + t.Fatalf("sanitize(%q) = %q, want %q", tc.in, got, tc.want) + } + }) + } +} + +func TestSanitizeSlotLineBoundsWordBudget(t *testing.T) { + t.Run("cuts at a clause boundary", func(t *testing.T) { + runOn := strings.TrimSpace(strings.Repeat("alpha ", 24) + "tail, " + strings.Repeat("beta ", 30)) + want := strings.TrimSpace(strings.Repeat("alpha ", 24) + "tail") + + if got := sanitizeSlotLine(runOn); got != want { + t.Fatalf("expected the cut to land on the clause boundary, got %q", got) + } + }) + + t.Run("cuts on a word boundary without punctuation", func(t *testing.T) { + words := strings.TrimSpace(strings.Repeat("gamma ", slotLineWordBudget+20)) + got := sanitizeSlotLine(words) + if n := len(strings.Fields(got)); n != slotLineWordBudget { + t.Fatalf("expected %d words, got %d in %q", slotLineWordBudget, n, got) + } + for _, w := range strings.Fields(got) { + if w != "gamma" { + t.Fatalf("expected whole words only, got %q", w) + } + } + }) + + t.Run("keeps a short line whole", func(t *testing.T) { + short := strings.TrimSpace(strings.Repeat("delta ", 10)) + if got := sanitizeSlotLine(short); got != short { + t.Fatalf("expected %q, got %q", short, got) + } + }) +} + +func TestBuildModulePageFillsOneSlotPerCall(t *testing.T) { + client := &scriptedLLM{numCtx: 2048, results: []LLMResult{ + stopResult("Holds every module in the repository"), + stopResult("Talks to the Ollama API"), + stopResult("Resolves configuration precedence"), + stopResult("Facts flow up the tree"), + stopResult("Failures abort the page"), + }} + p := newTestPipeline(t, client, &config.Config{ + DocsDir: "docs", + CharsPerToken: config.CharsPerTokenDefault, + SlotNumPredict: 96, + ParagraphNumPredict: 512, + SystemPrompt: "system", + ModuleSynthesisPrompt: "module", + }) + + page, err := buildModulePage(context.Background(), p, "internal", []moduleComponent{ + {name: "File: client.go", facts: "#### [API_SIGNATURES]\nCall()"}, + {name: "Subsystem: config", facts: "#### [API_SIGNATURES]\nResolve()"}, + }) + if err != nil { + t.Fatalf("unexpected error: %v", err) + } + + if client.calls != 5 { + t.Fatalf("expected 5 slot calls for 2 components, got %d", client.calls) + } + for i, system := range client.systems { + if system != "system" { + t.Errorf("call %d: expected the bare system prompt, got %q", i, system) + } + } + for i, bound := range client.bounds { + want := 96 + if i >= 3 { + want = 512 + } + if bound != want { + t.Errorf("call %d: expected a num_predict of %d, got %d", i, want, bound) + } + } + if !strings.Contains(page, "## Data Flow\nFacts flow up the tree") { + t.Errorf("expected the data flow slot in the page, got:\n%s", page) + } + + if !strings.Contains(client.prompts[0], "Components:\n- File: client.go\n- Subsystem: config") { + t.Errorf("expected the responsibility slot to see the component identities, got:\n%s", client.prompts[0]) + } + if !strings.HasPrefix(client.prompts[0], "module\n\n") { + t.Errorf("expected the module style prompt to open the user turn, got:\n%s", client.prompts[0]) + } + if strings.Contains(client.prompts[0], "Call()") || strings.Contains(client.prompts[0], "Resolve()") { + t.Errorf("expected the responsibility slot not to receive component facts, got:\n%s", client.prompts[0]) + } + if !strings.Contains(client.prompts[1], "Call()") || strings.Contains(client.prompts[1], "Resolve()") { + t.Errorf("expected the component slot to see only its own facts, got:\n%s", client.prompts[1]) + } +} + +func TestBuildModulePageFailsOnAnySlot(t *testing.T) { + cases := []struct { + name string + failing int + wantErr error + }{ + {name: "responsibility", failing: 0, wantErr: ErrTruncatedOutput}, + {name: "component", failing: 1, wantErr: ErrTruncatedOutput}, + {name: "data flow", failing: 2, wantErr: ErrEmptyOutput}, + {name: "error handling", failing: 3, wantErr: ErrTruncatedOutput}, + } + + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + results := []LLMResult{stopResult("ok"), stopResult("ok"), stopResult("ok"), stopResult("ok")} + switch tc.wantErr { + case ErrTruncatedOutput: + results[tc.failing] = truncatedResult("cut off") + default: + results[tc.failing] = stopResult("") + } + client := &scriptedLLM{numCtx: 2048, results: results} + p := newTestPipeline(t, client, &config.Config{ + DocsDir: "docs", + CharsPerToken: config.CharsPerTokenDefault, + SystemPrompt: "system", + ModuleSynthesisPrompt: "module", + }) + + page, err := buildModulePage(context.Background(), p, "internal", []moduleComponent{ + {name: "File: client.go", facts: "facts"}, + {name: "File: tree.go", facts: "facts"}, + }) + if !errors.Is(err, tc.wantErr) { + t.Fatalf("expected %v, got %v (page: %q)", tc.wantErr, err, page) + } + if page != "" { + t.Errorf("expected no page when a slot fails, got %q", page) + } + }) + } +} + +func TestBuildModulePageRejectsUnusableSlotOutput(t *testing.T) { + client := &scriptedLLM{numCtx: 2048, results: []LLMResult{stopResult("- \n- .\n```")}} + p := newTestPipeline(t, client, &config.Config{ + DocsDir: "docs", + CharsPerToken: config.CharsPerTokenDefault, + SystemPrompt: "system", + ModuleSynthesisPrompt: "module", + }) + + page, err := buildModulePage(context.Background(), p, "internal", []moduleComponent{{name: "File: client.go", facts: "facts"}}) + if !errors.Is(err, ErrEmptyOutput) { + t.Fatalf("expected ErrEmptyOutput, got %v (page: %q)", err, page) + } +} + +func TestBuildModulePageSalvagesTruncatedLineSlots(t *testing.T) { + client := &scriptedLLM{numCtx: 2048, results: []LLMResult{ + truncatedResult("Holds the CLI commands. It also writes a very long tail that the limit cuts off mid"), + truncatedResult("Registers the cobra commands. It also sets up the flags and"), + stopResult("Resolves configuration precedence"), + stopResult("Facts flow up the tree"), + stopResult("Failures abort the page"), + }} + p := newTestPipeline(t, client, &config.Config{ + DocsDir: "docs", + CharsPerToken: config.CharsPerTokenDefault, + SystemPrompt: "system", + ModuleSynthesisPrompt: "module", + }) + + page, err := buildModulePage(context.Background(), p, "cmd", []moduleComponent{ + {name: "File: root.go", facts: "facts"}, + {name: "File: setup.go", facts: "facts"}, + }) + if err != nil { + t.Fatalf("expected the page to complete, got %v", err) + } + if client.calls != 5 { + t.Fatalf("expected 5 slot calls, got %d", client.calls) + } + if !strings.Contains(page, "## Responsibility\nHolds the CLI commands\n") { + t.Errorf("expected the salvaged responsibility sentence, got:\n%s", page) + } + if !strings.Contains(page, "### File: root.go\nRegisters the cobra commands\n") { + t.Errorf("expected the salvaged component sentence, got:\n%s", page) + } + if strings.Contains(page, "cuts off mid") { + t.Errorf("expected the truncated tail to be dropped, got:\n%s", page) + } +} + +func TestBuildModulePageFailsOnTruncatedLineSlotWithoutSentence(t *testing.T) { + client := &scriptedLLM{numCtx: 2048, results: []LLMResult{ + truncatedResult("Holds the CLI commands and also writes a very long tail that the"), + }} + p := newTestPipeline(t, client, &config.Config{ + DocsDir: "docs", + CharsPerToken: config.CharsPerTokenDefault, + SystemPrompt: "system", + ModuleSynthesisPrompt: "module", + }) + + page, err := buildModulePage(context.Background(), p, "cmd", []moduleComponent{{name: "File: root.go", facts: "facts"}}) + if !errors.Is(err, ErrTruncatedOutput) { + t.Fatalf("expected ErrTruncatedOutput for a clipped sentence, got %v (page: %q)", err, page) + } +} + +func TestBuildModulePageSalvagesTruncatedParagraphSlot(t *testing.T) { + client := &scriptedLLM{numCtx: 2048, results: []LLMResult{ + stopResult("Holds the CLI commands"), + stopResult("Registers the cobra commands"), + truncatedResult("Facts flow up the tree. Failures travel back down the call chain. The rest of the paragraph is cut off mid"), + }} + p := newTestPipeline(t, client, &config.Config{ + DocsDir: "docs", + CharsPerToken: config.CharsPerTokenDefault, + SystemPrompt: "system", + ModuleSynthesisPrompt: "module", + }) + + page, err := buildModulePage(context.Background(), p, "cmd", []moduleComponent{{name: "File: root.go", facts: "facts"}}) + if err != nil { + t.Fatalf("expected the page to complete, got %v", err) + } + if !strings.Contains(page, "## Data Flow\nFacts flow up the tree. Failures travel back down the call chain.\n") { + t.Errorf("expected the text up to the last complete sentence, got:\n%s", page) + } + if strings.Contains(page, "cut off mid") { + t.Errorf("expected the clipped tail to be dropped, got:\n%s", page) + } +} + +func TestBuildModulePageFailsOnTruncatedParagraphSlotWithoutSentence(t *testing.T) { + client := &scriptedLLM{numCtx: 2048, results: []LLMResult{ + stopResult("Holds the CLI commands"), + stopResult("Registers the cobra commands"), + truncatedResult("Facts flow up the tree and then the rest of the paragraph never ends on a boundar"), + }} + p := newTestPipeline(t, client, &config.Config{ + DocsDir: "docs", + CharsPerToken: config.CharsPerTokenDefault, + SystemPrompt: "system", + ModuleSynthesisPrompt: "module", + }) + + page, err := buildModulePage(context.Background(), p, "cmd", []moduleComponent{{name: "File: root.go", facts: "facts"}}) + if !errors.Is(err, ErrTruncatedOutput) { + t.Fatalf("expected ErrTruncatedOutput for a clipped paragraph without a sentence, got %v (page: %q)", err, page) + } + if page != "" { + t.Errorf("expected no page when a slot fails, got %q", page) + } +} + +func TestBuildModulePageRendersOneBoundedLinePerSlot(t *testing.T) { + enumeration := strings.Join([]string{ + "- Talks to the Ollama API and streams the result", + "- Caches every response in the metadata file", + "- Retries a failed call once", + }, "\n") + client := &scriptedLLM{numCtx: 2048, results: []LLMResult{ + stopResult("Holds every module in the repository"), + stopResult(enumeration), + stopResult("Resolves configuration precedence"), + stopResult("Facts flow up the tree"), + stopResult("Failures abort the page"), + }} + p := newTestPipeline(t, client, &config.Config{ + DocsDir: "docs", + CharsPerToken: config.CharsPerTokenDefault, + SystemPrompt: "system", + ModuleSynthesisPrompt: "module", + }) + + page, err := buildModulePage(context.Background(), p, "cmd", []moduleComponent{{name: "File: client.go", facts: "facts"}}) + if err != nil { + t.Fatalf("unexpected error: %v", err) + } + if !strings.Contains(page, "### File: client.go\nTalks to the Ollama API and streams the result\n") { + t.Errorf("expected exactly one line for the component, got:\n%s", page) + } + if strings.Contains(page, "Caches every response") || strings.Contains(page, "Retries a failed call") { + t.Errorf("expected the trailing bullets to be dropped, got:\n%s", page) + } +} + +func TestBuildStandardDocPageFailsOnAnySection(t *testing.T) { + cfg := &config.Config{SystemPrompt: "system", ArchitecturePrompt: "architecture", CharsPerToken: config.CharsPerTokenDefault} + client := &scriptedLLM{numCtx: 2048, results: []LLMResult{ + stopResult("A generated wiki."), + truncatedResult("half a boundar"), + }} + + page, err := buildStandardDocPage(context.Background(), client, cfg, architecturePage, "# Module: .", func(t EventType, msg string) {}) + if !errors.Is(err, ErrTruncatedOutput) { + t.Fatalf("expected ErrTruncatedOutput, got %v (page: %q)", err, page) + } + if client.calls != 2 { + t.Fatalf("expected the page to stop at the failing section, got %d calls", client.calls) + } +} + +func TestSlotNumPredict(t *testing.T) { + cases := []struct { + name string + cfg config.Config + kind slotKind + want int + }{ + {name: "line default", cfg: config.Config{}, kind: slotLine, want: config.SlotNumPredictDefault}, + {name: "line override", cfg: config.Config{SlotNumPredict: 64}, kind: slotLine, want: 64}, + {name: "paragraph default", cfg: config.Config{}, kind: slotParagraph, want: config.ParagraphNumPredictDefault}, + {name: "paragraph override", cfg: config.Config{ParagraphNumPredict: 2048}, kind: slotParagraph, want: 2048}, + {name: "line bound ignores the paragraph field", cfg: config.Config{ParagraphNumPredict: 2048}, kind: slotLine, want: config.SlotNumPredictDefault}, + {name: "paragraph bound ignores the line field", cfg: config.Config{SlotNumPredict: 64}, kind: slotParagraph, want: config.ParagraphNumPredictDefault}, + {name: "global cap wins for a line slot", cfg: config.Config{SlotNumPredict: 256, NumPredict: 100}, kind: slotLine, want: 100}, + {name: "global cap wins for a paragraph slot", cfg: config.Config{ParagraphNumPredict: 2048, NumPredict: 100}, kind: slotParagraph, want: 100}, + {name: "global cap used when the slot is unset", cfg: config.Config{NumPredict: 100}, kind: slotLine, want: 100}, + {name: "a global cap above the slot bound never raises it", cfg: config.Config{NumPredict: 4096}, kind: slotParagraph, want: config.ParagraphNumPredictDefault}, + } + + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + if got := slotNumPredict(&tc.cfg, tc.kind); got != tc.want { + t.Fatalf("expected %d, got %d", tc.want, got) + } + }) + } +} + +func TestSlotRunnerOutputReserveTakesTheLargestBound(t *testing.T) { + cases := []struct { + name string + cfg config.Config + want int + }{ + {name: "defaults", cfg: config.Config{}, want: config.ParagraphNumPredictDefault}, + {name: "paragraph bound raised", cfg: config.Config{ParagraphNumPredict: 4096}, want: 4096}, + {name: "line bound only", cfg: config.Config{SlotNumPredict: 64}, want: config.ParagraphNumPredictDefault}, + {name: "both bounds lowered", cfg: config.Config{SlotNumPredict: 64, ParagraphNumPredict: 128}, want: 128}, + {name: "global cap below both", cfg: config.Config{NumPredict: 128}, want: 128}, + } + + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + runner := newSlotRunner(context.Background(), &scriptedLLM{}, &tc.cfg, "style", func(EventType, string) {}) + if got := runner.outputReserve(); got != tc.want { + t.Fatalf("expected %d, got %d", tc.want, got) + } + }) + } +} + +func TestComponentFactsDigestIsBounded(t *testing.T) { + components := []moduleComponent{ + {name: "File: client.go", facts: strings.Repeat("a", 500)}, + {name: "File: tree.go", facts: strings.Repeat("b", 500)}, + } + + digest := componentFactsDigest(components, 100) + if utf8Len := len([]rune(digest)); utf8Len > 100+len(components)*64 { + t.Fatalf("expected the digest to be bounded near the budget, got %d runes", utf8Len) + } + if !strings.Contains(digest, "### File: client.go") || !strings.Contains(digest, "### File: tree.go") { + t.Errorf("expected every component in the digest, got:\n%s", digest) + } + if !strings.HasSuffix(digest, "...") { + t.Errorf("expected a truncation marker on the last cut component, got:\n%s", digest) + } +} + +func TestChildIdentityFactsKeepsTheChildIdentity(t *testing.T) { + child := renderModulePage("internal/config", modulePageParts{ + responsibility: "Resolves the configuration", + components: []moduleComponentLine{{name: "File: resolve.go", text: "Merges flags, env and yaml"}}, + dataFlow: "The flags arrive first and the yaml last.", + errorHandling: "An invalid value fails the whole run.", + }) + + head := childIdentityFacts(child) + if !strings.Contains(head, "## Responsibility\nResolves the configuration") { + t.Errorf("expected the child responsibility in the head, got:\n%s", head) + } + if !strings.Contains(head, "### File: resolve.go\nMerges flags, env and yaml") { + t.Errorf("expected the child components in the head, got:\n%s", head) + } + if strings.Contains(head, "The flags arrive first") || strings.Contains(head, "An invalid value fails") { + t.Errorf("expected the child interior to be dropped, got:\n%s", head) + } + if got := childIdentityFacts("no skeleton here"); got != "no skeleton here" { + t.Errorf("expected an unskeletonized page to be kept whole, got %q", got) + } +} + +func TestModuleShapeSelectsParentFraming(t *testing.T) { + components := parseModuleComponents([]string{ + "### File: main.go\n#### [API_SIGNATURES]\nfunc main()", + "### Subsystem: config\n# Module: internal/config\n\n## Responsibility\nResolves the configuration", + }) + + if got := moduleShape(components); got != shapeHierarchical { + t.Fatalf("expected a module with a subsystem to be hierarchical, got %v", got) + } + dataFlow, errorHandling := moduleParagraphPrompts(".", componentIdentities(components), "digest", moduleShape(components)) + if !strings.Contains(dataFlow, "what flows between the components") { + t.Errorf("expected the parent data flow framing, got:\n%s", dataFlow) + } + if !strings.Contains(errorHandling, "how errors cross the boundaries") { + t.Errorf("expected the parent error framing, got:\n%s", errorHandling) + } + if !strings.Contains(errorHandling, "do not repeat it") { + t.Errorf("expected the parent error prompt to forbid restating the children, got:\n%s", errorHandling) + } + + leaf := parseModuleComponents([]string{"### File: main.go\nfunc main()"}) + if got := moduleShape(leaf); got != shapeLeaf { + t.Fatalf("expected a module without a subsystem to be a leaf, got %v", got) + } + dataFlow, errorHandling = moduleParagraphPrompts("cmd", componentIdentities(leaf), "digest", moduleShape(leaf)) + if !strings.Contains(dataFlow, "Describe how data moves through this module") { + t.Errorf("expected the leaf data flow framing, got:\n%s", dataFlow) + } + if !strings.Contains(errorHandling, "Describe how this module reports and handles errors") { + t.Errorf("expected the leaf error framing, got:\n%s", errorHandling) + } +} + +func TestBuildModulePageKeepsChildInteriorOutOfTheParentSlots(t *testing.T) { + child := renderModulePage("internal/config", modulePageParts{ + responsibility: "Resolves the configuration", + components: []moduleComponentLine{{name: "File: resolve.go", text: "Merges flags, env and yaml"}}, + dataFlow: "The flags arrive first and the yaml last.", + errorHandling: "An invalid value fails the whole run.", + }) + components := parseModuleComponents([]string{ + "### File: root.go\n#### [API_SIGNATURES]\nfunc main()", + markdownHeadingPrefix + subsystemPrefix + "config\n" + child, + }) + client := &scriptedLLM{numCtx: 2048, results: []LLMResult{ + stopResult("Holds the command layer"), + stopResult("Owns the entry point"), + stopResult("Owns the configuration"), + stopResult("The command layer hands the resolved config to the engine"), + stopResult("Every error travels back out through the command layer"), + }} + p := newTestPipeline(t, client, &config.Config{ + DocsDir: "docs", + CharsPerToken: config.CharsPerTokenDefault, + SystemPrompt: "system", + ModuleSynthesisPrompt: "module", + }) + + if _, err := buildModulePage(context.Background(), p, ".", components); err != nil { + t.Fatalf("unexpected error: %v", err) + } + for i, prompt := range client.prompts { + if strings.Contains(prompt, "The flags arrive first") || strings.Contains(prompt, "An invalid value fails") { + t.Errorf("call %d: expected the child interior to stay out of the parent slots, got:\n%s", i, prompt) + } + } + if !strings.Contains(client.prompts[2], "Resolves the configuration") { + t.Errorf("expected the child identity in the subsystem component slot, got:\n%s", client.prompts[2]) + } + if !strings.Contains(client.prompts[3], "what flows between the components") { + t.Errorf("expected the parent framing in the data flow slot, got:\n%s", client.prompts[3]) + } +} + +func TestTruncateRunes(t *testing.T) { + if got := truncateRunes("hello", 10); got != "hello" { + t.Errorf("expected no truncation, got %q", got) + } + if got := truncateRunes("hello", 0); got != "" { + t.Errorf("expected an empty payload for a zero budget, got %q", got) + } + if got := truncateRunes("héllo wörld", 5); got != "héllo..." { + t.Errorf("expected a rune-safe cut, got %q", got) + } +} diff --git a/internal/engine/runner.go b/internal/engine/runner.go index 10666af..7d70540 100644 --- a/internal/engine/runner.go +++ b/internal/engine/runner.go @@ -48,7 +48,7 @@ func (r *Runner) Run(ctx context.Context, repoRoot string, mode Mode, onEvent fu defer lock.Unlock() // 3. Instantiate LLM Client & Orchestrator - client := newLLMClient(r.cfg.ModelID, r.cfg.OllamaBaseURL, r.cfg.OllamaNumCtx) + client := newLLMClient(r.cfg) orch := &orchestrator{client: client} // 4. Run the documentation pipeline diff --git a/internal/engine/synthesize.go b/internal/engine/synthesize.go index 26cc713..2d5540a 100644 --- a/internal/engine/synthesize.go +++ b/internal/engine/synthesize.go @@ -69,13 +69,19 @@ func extractFileFacts(ctx context.Context, p *pipelineState, f string, nodePath } p.logEvent(EventStatus, fmt.Sprintf("➜ Extracting file (Step %d/%d - %s)%s: %s", i+1, len(p.cfg.ExtractionSteps), step.Name, chunkMsg, f)) - systemPrompt := p.cfg.SystemPrompt + "\n" + step.Prompt - userContent := fmt.Sprintf("File: %s%s inside Module: %s\n```\n%s\n```", filepath.Base(f), chunkMsg, nodePath, chunk) - res, err := p.client.CallLLM(ctx, systemPrompt, []Message{{Role: "user", Content: userContent}}, false) + userContent := userMessage( + fmt.Sprintf("File: %s%s inside Module: %s\n```\n%s\n```", filepath.Base(f), chunkMsg, nodePath, chunk), + step.Prompt, + ) + res, err := p.client.CallLLM(ctx, p.cfg.SystemPrompt, []Message{{Role: "user", Content: userContent}}, false) if err != nil { return "", fmt.Errorf("LLM error extracting %s for %s: %w", step.Name, f, err) } - stepFacts = append(stepFacts, stripOuterMarkdownFence(res)) + chunkFacts, err := requireCompleteContent(res, fmt.Sprintf("extracting %s for %s", step.Name, f)) + if err != nil { + return "", err + } + stepFacts = append(stepFacts, chunkFacts) } consolidatedFact, err := reduceFileFacts(ctx, p.client, f, step.Name, stepFacts, p.cfg, p.logEvent) @@ -96,11 +102,8 @@ func extractFileFacts(ctx context.Context, p *pipelineState, f string, nodePath return facts, nil } -func calculateFileLimit(numCtx int) int { - if numCtx < minNumCtxFloor { - numCtx = minNumCtxFloor - } - return int(float64(numCtx*4) * contextWindowAllocRatio) +func calculateFileLimit(numCtx int, cfg *config.Config) int { + return promptCharBudget(numCtx, cfg.OutputTokenReserve, cfg.CharsPerToken) } func synthesizeChildren(ctx context.Context, p *pipelineState, node *DirNode) (map[string]string, []string, error) { @@ -124,7 +127,7 @@ func synthesizeChildren(ctx context.Context, p *pipelineState, node *DirNode) (m } func collectComponents(ctx context.Context, p *pipelineState, node *DirNode, childSummaries map[string]string, childNames []string) ([]string, error) { - fileLimit := calculateFileLimit(p.client.NumCtx()) + fileLimit := calculateFileLimit(p.client.NumCtx(), p.cfg) var components []string for _, f := range node.Files { @@ -143,7 +146,7 @@ func collectComponents(ctx context.Context, p *pipelineState, node *DirNode, chi for _, childName := range childNames { if sum := childSummaries[childName]; sum != "" { - components = append(components, fmt.Sprintf("### Subsystem: %s\n%s", childName, sum)) + components = append(components, fmt.Sprintf("%s%s\n%s", markdownHeadingPrefix, subsystemPrefix+childName, sum)) } } return components, nil @@ -175,19 +178,19 @@ func synthesizeNode(ctx context.Context, p *pipelineState, node *DirNode) (strin return "", nil } - p.logEvent(EventStatus, fmt.Sprintf("➜ Synthesizing directory: %s (%d total components)", node.Path, len(components))) - finalSum, err := reduceInChunks(ctx, p.client, node.Path, components, p.cfg, p.logEvent) + p.logEvent(EventStatus, fmt.Sprintf("➜ Assembling module page: %s (%d total components)", node.Path, len(components))) + page, err := buildModulePage(ctx, p, node.Path, parseModuleComponents(components)) if err != nil { return "", err } // Update module cache - p.cache.Modules[node.Path] = finalSum + p.cache.Modules[node.Path] = page modulePath := filepath.Join(p.cfg.DocsDir, "modules", toSafeMarkdownFilename(node.Path)) - if err := tools.WriteFileSafely(p.repoRoot, modulePath, []byte(finalSum)); err != nil { + if err := tools.WriteFileSafely(p.repoRoot, modulePath, []byte(page)); err != nil { return "", fmt.Errorf("failed to write module documentation for %s: %w", node.Path, err) } - return finalSum, nil + return page, nil } diff --git a/internal/engine/synthesize_test.go b/internal/engine/synthesize_test.go index b23b7cf..e4645c0 100644 --- a/internal/engine/synthesize_test.go +++ b/internal/engine/synthesize_test.go @@ -2,6 +2,7 @@ package engine import ( "context" + "errors" "os" "path/filepath" "testing" @@ -9,19 +10,71 @@ import ( "github.com/arrase/code-reducer/internal/config" ) -type mockLLMCaller struct { - numCtx int +const mockLLMContent = "Mocked LLM summary response" + +// scriptedLLM is a fake llmCaller that replays one scripted result per call, so a +// test can let the extraction succeed and make a following slot fail. It also +// records the system message, the user message and the generation bound of every +// call. Calls beyond the script return generic content. +type scriptedLLM struct { + numCtx int + results []LLMResult + calls int + systems []string + prompts []string + bounds []int +} + +func (s *scriptedLLM) next() (LLMResult, error) { + if s.calls >= len(s.results) { + s.calls++ + return LLMResult{Content: mockLLMContent, DoneReason: "stop"}, nil + } + res := s.results[s.calls] + s.calls++ + return res, nil } -func (m *mockLLMCaller) CallLLM(ctx context.Context, systemPrompt string, messages []Message, jsonFormat bool) (string, error) { - return "Mocked LLM summary response", nil +func (s *scriptedLLM) CallLLM(ctx context.Context, systemPrompt string, messages []Message, jsonFormat bool) (LLMResult, error) { + s.systems = append(s.systems, systemPrompt) + s.prompts = append(s.prompts, messages[0].Content) + s.bounds = append(s.bounds, 0) + return s.next() } -func (m *mockLLMCaller) NumCtx() int { - if m.numCtx == 0 { +func (s *scriptedLLM) CallLLMWithNumPredict(ctx context.Context, systemPrompt string, messages []Message, numPredict int) (LLMResult, error) { + s.systems = append(s.systems, systemPrompt) + s.prompts = append(s.prompts, messages[0].Content) + s.bounds = append(s.bounds, numPredict) + return s.next() +} + +func (s *scriptedLLM) NumCtx() int { + if s.numCtx == 0 { return 2048 } - return m.numCtx + return s.numCtx +} + +func stopResult(content string) LLMResult { + return LLMResult{Content: content, DoneReason: "stop", EvalCount: 42, PromptCount: 100} +} + +func truncatedResult(content string) LLMResult { + return LLMResult{Content: content, DoneReason: "length", EvalCount: 1024, PromptCount: 19997} +} + +func newTestPipeline(t *testing.T, client llmCaller, cfg *config.Config) *pipelineState { + t.Helper() + return &pipelineState{ + client: client, + repoRoot: t.TempDir(), + cfg: cfg, + cache: newEmptyCache(), + affectedDirs: map[string]bool{}, + precalculatedHashes: map[string]string{}, + logEvent: func(t EventType, msg string) {}, + } } func TestSynthesizeNode(t *testing.T) { @@ -49,7 +102,7 @@ func TestSynthesizeNode(t *testing.T) { ctx := context.Background() p := &pipelineState{ - client: &mockLLMCaller{numCtx: 2048}, + client: &scriptedLLM{numCtx: 2048}, repoRoot: repoRoot, cfg: cfg, cache: cache, @@ -95,3 +148,103 @@ func TestSynthesizeNode(t *testing.T) { t.Fatalf("expected non-empty synthesis result") } } + +func TestExtractFileFactsRejectsIncompleteLLMOutput(t *testing.T) { + step := config.ExtractionStep{Name: "API_SIGNATURES", Prompt: "extract"} + + cases := []struct { + name string + res LLMResult + wantErr error + }{ + {name: "truncated by length", res: truncatedResult("partial facts"), wantErr: ErrTruncatedOutput}, + {name: "truncated after a complete sentence", res: truncatedResult("func Call() error. The rest of the list was cut"), wantErr: ErrTruncatedOutput}, + {name: "empty content on stop", res: stopResult(" \n\t "), wantErr: ErrEmptyOutput}, + {name: "empty content on length", res: truncatedResult(""), wantErr: ErrTruncatedOutput}, + {name: "empty fence on stop", res: stopResult("```json\n```"), wantErr: ErrEmptyOutput}, + } + + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + p := newTestPipeline(t, &scriptedLLM{results: []LLMResult{tc.res}}, + &config.Config{DocsDir: "docs", ExtractionSteps: []config.ExtractionStep{step}}) + + sampleFile := "main.go" + if err := os.WriteFile(filepath.Join(p.repoRoot, sampleFile), []byte("package main\n\nfunc main() {}\n"), 0644); err != nil { + t.Fatalf("failed to write sample file: %v", err) + } + + facts, err := extractFileFacts(context.Background(), p, sampleFile, ".", promptCharBudget(2048, 0, config.CharsPerTokenDefault)) + if !errors.Is(err, tc.wantErr) { + t.Fatalf("expected %v, got %v (facts: %q)", tc.wantErr, err, facts) + } + if facts != "" { + t.Errorf("expected no facts to be returned, got %q", facts) + } + if len(p.cache.Files) != 0 { + t.Errorf("expected no cache entry to be written, got %v", p.cache.Files) + } + }) + } +} + +func TestExtractFileFactsUnreadableFileIsNotAnLLMError(t *testing.T) { + p := newTestPipeline(t, &scriptedLLM{}, + &config.Config{DocsDir: "docs", ExtractionSteps: config.DefaultExtractionSteps}) + + facts, err := extractFileFacts(context.Background(), p, "does-not-exist.go", ".", 1024) + if err != nil { + t.Fatalf("expected nil error for unreadable file, got %v", err) + } + if facts != "" { + t.Errorf("expected empty facts, got %q", facts) + } + if len(p.cache.Files) != 0 { + t.Errorf("expected no cache entry, got %v", p.cache.Files) + } +} + +func TestSynthesizeNodeDoesNotPersistFailedSlot(t *testing.T) { + repoRoot := t.TempDir() + docsDir := "docs" + modulesDir := filepath.Join(repoRoot, docsDir, "modules") + if err := os.MkdirAll(modulesDir, 0755); err != nil { + t.Fatalf("failed to create modules dir: %v", err) + } + + sampleFile := "main.go" + if err := os.WriteFile(filepath.Join(repoRoot, sampleFile), []byte("package main\n\nfunc main() {}\n"), 0644); err != nil { + t.Fatalf("failed to write sample file: %v", err) + } + + step := config.ExtractionStep{Name: "API_SIGNATURES", Prompt: "extract"} + cache := newEmptyCache() + p := &pipelineState{ + client: &scriptedLLM{numCtx: 2048, results: []LLMResult{ + stopResult("extracted facts"), + truncatedResult("half a responsibility"), + }}, + repoRoot: repoRoot, + cfg: &config.Config{DocsDir: docsDir, ExtractionSteps: []config.ExtractionStep{step}}, + cache: cache, + affectedDirs: map[string]bool{"sub": true}, + precalculatedHashes: map[string]string{sampleFile: "hash123"}, + logEvent: func(t EventType, msg string) {}, + } + + node := &DirNode{Path: "sub", Files: []string{sampleFile}, Children: make(map[string]*DirNode)} + res, err := synthesizeNode(context.Background(), p, node) + if !errors.Is(err, ErrTruncatedOutput) { + t.Fatalf("expected ErrTruncatedOutput, got %v (res: %q)", err, res) + } + if _, ok := cache.Modules["sub"]; ok { + t.Error("expected no module cache entry when a slot fails") + } + moduleFile := filepath.Join(modulesDir, "sub", "README.md") + if _, err := os.Stat(moduleFile); !os.IsNotExist(err) { + t.Errorf("expected %s not to be written", moduleFile) + } + if len(cache.Files) != 1 { + t.Errorf("expected the successfully extracted file facts to be cached, got %v", cache.Files) + } +} diff --git a/internal/tools/file_tools.go b/internal/tools/file_tools.go index 5ba5e0a..9ea5980 100644 --- a/internal/tools/file_tools.go +++ b/internal/tools/file_tools.go @@ -4,6 +4,7 @@ import ( "bufio" "fmt" "os" + "path" "path/filepath" "strings" @@ -157,11 +158,53 @@ func ShouldIgnoreFile(relPath string, gitIgnore *ignore.GitIgnore) bool { return false } +// testFileSuffixes and testFilePrefixes hold the naming conventions the common +// languages use for test files. They are matched against the base name only, and +// case-sensitively, so a source file that merely ends in a similar word is kept. +var ( + testFileSuffixes = []string{ + // Go, Python, JavaScript, Ruby, Rust + "_test.go", "_test.py", "_test.js", ".test.js", ".spec.js", "_spec.rb", "_test.rb", "_test.rs", + // TypeScript + ".test.ts", ".spec.ts", + // JVM and .NET + "Test.java", "Tests.cs", "Tests.scala", + } + testFilePrefixes = []string{"test_"} +) + +// IsTestFile reports whether a repository-relative path is a test file by naming +// convention. Documentation of a test file mostly restates its assertions, so +// discovery drops these unless the user asks for them. +func IsTestFile(relPath string) bool { + name := path.Base(filepath.ToSlash(relPath)) + for _, suffix := range testFileSuffixes { + if strings.HasSuffix(name, suffix) { + return true + } + } + for _, prefix := range testFilePrefixes { + if strings.HasPrefix(name, prefix) { + return true + } + } + return false +} + +// DiscoveryOptions controls which repository files the discovery walk keeps. +type DiscoveryOptions struct { + // Ignores holds the gitignore-style patterns from the config and .gitignore. + Ignores []string + // IncludeTests keeps files that follow a test-file naming convention. + IncludeTests bool +} + // DiscoverCodeFiles recursively walks the codebase to find high-signal source files. -// It ignores build, dependency, and output files, as well as any paths in the custom ignores list. -func DiscoverCodeFiles(repoRoot string, ignores []string) ([]string, error) { +// It ignores build, dependency, and output files, as well as any paths in the custom +// ignores list, and drops test files unless IncludeTests is set. +func DiscoverCodeFiles(repoRoot string, opts DiscoveryOptions) ([]string, error) { var files []string - gitIgnore := ignore.CompileIgnoreLines(ignores...) + gitIgnore := ignore.CompileIgnoreLines(opts.Ignores...) err := filepath.WalkDir(repoRoot, func(path string, d os.DirEntry, err error) error { if err != nil { @@ -191,6 +234,10 @@ func DiscoverCodeFiles(repoRoot string, ignores []string) ([]string, error) { return nil } + if !opts.IncludeTests && IsTestFile(slashRel) { + return nil + } + files = append(files, slashRel) return nil }) diff --git a/internal/tools/file_tools_test.go b/internal/tools/file_tools_test.go index eee782c..3c4f529 100644 --- a/internal/tools/file_tools_test.go +++ b/internal/tools/file_tools_test.go @@ -4,6 +4,7 @@ import ( "os" "path/filepath" "reflect" + "sort" "testing" "github.com/arrase/code-reducer/internal/tools" @@ -80,6 +81,43 @@ func TestShouldIgnoreFile(t *testing.T) { } } +func TestIsTestFile(t *testing.T) { + tests := []struct { + relPath string + want bool + }{ + {"internal/engine/client.go", false}, + {"internal/engine/synthesize_test.go", true}, + {"pkg/parser_test.py", true}, + {"pkg/test_parser.py", true}, + {"src/parser_test.py", true}, + {"src/parser_test.js", true}, + {"src/parser.test.js", true}, + {"src/parser.spec.js", true}, + {"src/parser.test.ts", true}, + {"src/parser.spec.ts", true}, + {"src/OrderServiceTest.java", true}, + {"src/OrderRepositoryTests.cs", true}, + {"src/OrderSpec.scala", false}, + {"src/OrderServiceTests.scala", true}, + {"spec/orders_spec.rb", true}, + {"spec/test_orders.rb", true}, + {"src/orders_test.rs", true}, + {"src/contest.go", false}, + {"src/latest.java", false}, + {"docs/testing.md", false}, + {"test/main.go", false}, + } + + for _, tt := range tests { + t.Run(tt.relPath, func(t *testing.T) { + if got := tools.IsTestFile(tt.relPath); got != tt.want { + t.Errorf("IsTestFile(%q) = %v, want %v", tt.relPath, got, tt.want) + } + }) + } +} + func TestDiscoverCodeFiles(t *testing.T) { repoRoot := t.TempDir() @@ -98,6 +136,8 @@ func TestDiscoverCodeFiles(t *testing.T) { files := []string{ "src/main.go", "src/utils.go", + "src/utils_test.go", + "src/test_utils.py", "src/.hidden/secret.txt", "build/output.bin", "test.log", @@ -109,25 +149,33 @@ func TestDiscoverCodeFiles(t *testing.T) { } ignores := []string{"*.log", "build/"} - discovered, err := tools.DiscoverCodeFiles(repoRoot, ignores) + discovered, err := tools.DiscoverCodeFiles(repoRoot, tools.DiscoveryOptions{Ignores: ignores}) if err != nil { t.Fatalf("DiscoverCodeFiles() failed: %v", err) } expected := []string{"src/main.go", "src/utils.go"} - if len(discovered) != len(expected) { - t.Fatalf("Expected %d files, got %d: %v", len(expected), len(discovered), discovered) + assertDiscovered(t, discovered, expected) + + withTests, err := tools.DiscoverCodeFiles(repoRoot, tools.DiscoveryOptions{Ignores: ignores, IncludeTests: true}) + if err != nil { + t.Fatalf("DiscoverCodeFiles() with tests failed: %v", err) } + assertDiscovered(t, withTests, []string{"src/main.go", "src/test_utils.py", "src/utils.go", "src/utils_test.go"}) - // Order might vary depending on OS, but WalkDir is usually deterministic. - // We sort just in case or do a map check. - found := make(map[string]bool) - for _, f := range discovered { - found[f] = true + ignoresWithTests := append([]string{"**/*_test.go"}, ignores...) + explicitlyIgnored, err := tools.DiscoverCodeFiles(repoRoot, tools.DiscoveryOptions{Ignores: ignoresWithTests, IncludeTests: true}) + if err != nil { + t.Fatalf("DiscoverCodeFiles() with an explicit test ignore failed: %v", err) } - for _, f := range expected { - if !found[f] { - t.Errorf("Expected to find %s", f) - } + assertDiscovered(t, explicitlyIgnored, []string{"src/main.go", "src/test_utils.py", "src/utils.go"}) +} + +// assertDiscovered checks that discovery returned exactly the expected files. +func assertDiscovered(t *testing.T, got, want []string) { + t.Helper() + sort.Strings(got) + if !reflect.DeepEqual(got, want) { + t.Errorf("DiscoverCodeFiles() = %v, want %v", got, want) } }