diff --git a/CHANGELOG.md b/CHANGELOG.md index 546805e..dc6700a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,12 @@ and this project uses [Semantic Versioning](https://semver.org/). ## [Unreleased] +### Added +- `downshift report` shows the one-time analysis cost (eval model calls plus estimated judge + calls, at config prices) and the payback time. New `--audit-cost USD` option for an + assistant audit. `downshift export` adds the same numbers to `summary.json`. +- Web Overview shows the analysis cost and payback time. + ## [0.1.0] - 2026-09-26 First public release. diff --git a/bob_sessions/downshift_task12_payback_a.png b/bob_sessions/downshift_task12_payback_a.png new file mode 100644 index 0000000..af7be1f Binary files /dev/null and b/bob_sessions/downshift_task12_payback_a.png differ diff --git a/bob_sessions/downshift_task12_payback_b.png b/bob_sessions/downshift_task12_payback_b.png new file mode 100644 index 0000000..53b13b8 Binary files /dev/null and b/bob_sessions/downshift_task12_payback_b.png differ diff --git a/bob_sessions/downshift_task12_payback_c.png b/bob_sessions/downshift_task12_payback_c.png new file mode 100644 index 0000000..74f76f8 Binary files /dev/null and b/bob_sessions/downshift_task12_payback_c.png differ diff --git a/bob_sessions/downshift_task12_payback_d.png b/bob_sessions/downshift_task12_payback_d.png new file mode 100644 index 0000000..cc48dd3 Binary files /dev/null and b/bob_sessions/downshift_task12_payback_d.png differ diff --git a/bob_sessions/downshift_task12_payback_e.png b/bob_sessions/downshift_task12_payback_e.png new file mode 100644 index 0000000..c9d5bd3 Binary files /dev/null and b/bob_sessions/downshift_task12_payback_e.png differ diff --git a/bob_sessions/downshift_task12_payback_f.png b/bob_sessions/downshift_task12_payback_f.png new file mode 100644 index 0000000..8d005fb Binary files /dev/null and b/bob_sessions/downshift_task12_payback_f.png differ diff --git a/bob_sessions/downshift_task12_payback_g.png b/bob_sessions/downshift_task12_payback_g.png new file mode 100644 index 0000000..9b5667a Binary files /dev/null and b/bob_sessions/downshift_task12_payback_g.png differ diff --git a/bob_sessions/downshift_task12_payback_h.png b/bob_sessions/downshift_task12_payback_h.png new file mode 100644 index 0000000..9fcc49f Binary files /dev/null and b/bob_sessions/downshift_task12_payback_h.png differ diff --git a/bob_sessions/downshift_task12_payback_i.png b/bob_sessions/downshift_task12_payback_i.png new file mode 100644 index 0000000..87ad357 Binary files /dev/null and b/bob_sessions/downshift_task12_payback_i.png differ diff --git a/bob_sessions/downshift_task13_review_a.png b/bob_sessions/downshift_task13_review_a.png new file mode 100644 index 0000000..68276d2 Binary files /dev/null and b/bob_sessions/downshift_task13_review_a.png differ diff --git a/bob_sessions/downshift_task13_review_b.png b/bob_sessions/downshift_task13_review_b.png new file mode 100644 index 0000000..d4fe67a Binary files /dev/null and b/bob_sessions/downshift_task13_review_b.png differ diff --git a/bob_sessions/downshift_task13_review_c.png b/bob_sessions/downshift_task13_review_c.png new file mode 100644 index 0000000..0162d34 Binary files /dev/null and b/bob_sessions/downshift_task13_review_c.png differ diff --git a/bob_sessions/downshift_task13_review_d.png b/bob_sessions/downshift_task13_review_d.png new file mode 100644 index 0000000..f7b2d33 Binary files /dev/null and b/bob_sessions/downshift_task13_review_d.png differ diff --git a/bob_sessions/downshift_task13_review_e.png b/bob_sessions/downshift_task13_review_e.png new file mode 100644 index 0000000..029d2bd Binary files /dev/null and b/bob_sessions/downshift_task13_review_e.png differ diff --git a/bob_sessions/downshift_task13_review_f.png b/bob_sessions/downshift_task13_review_f.png new file mode 100644 index 0000000..37e3455 Binary files /dev/null and b/bob_sessions/downshift_task13_review_f.png differ diff --git a/bob_sessions/downshift_task13_review_g.png b/bob_sessions/downshift_task13_review_g.png new file mode 100644 index 0000000..ed6e6e5 Binary files /dev/null and b/bob_sessions/downshift_task13_review_g.png differ diff --git a/docs/specs/audit-payback-plan.md b/docs/specs/audit-payback-plan.md new file mode 100644 index 0000000..c29f21b --- /dev/null +++ b/docs/specs/audit-payback-plan.md @@ -0,0 +1,377 @@ +# Plan: Analysis Cost and Payback — `downshift report` + +## Overview + +Add a one-time "analysis cost" section to `downshift report` so users can see +what the evaluation run cost and how quickly projected savings will pay it back. +Scope is four files touched + two new files: + +| File | Change | +|---|---| +| `src/downshift/payback.py` | **NEW** — `AnalysisCost`, `Payback`, three functions | +| `src/downshift/report.py` | Add fields to `Report`; update `build_report`; update `render_markdown` | +| `src/downshift/cli.py` | Add `--audit-cost` to the `report` command | +| `src/downshift/export.py` | Add five keys to `summary_payload` | +| `tests/unit/test_payback.py` | **NEW** — unit tests for payback math and analysis_cost | +| `tests/unit/test_report.py` | Extend: section present/absent cases | +| `tests/cli/test_report_cli.py` | Extend: `--audit-cost` + negative rejection | +| `tests/unit/test_export.py` | Extend: new summary keys present | + +**Do NOT edit** the report snapshot file (`tests/snapshots/supportdesk_report.md` +or `examples/supportdesk/downshift.report.md`). The user regenerates it manually. + +--- + +## Sub-task 1 — `src/downshift/payback.py` (new module) + +**Status:** `[ ] pending` + +### Intent +Create a self-contained module for analysis-cost and payback calculations, +so `report.py` and `export.py` can import from it without circular dependencies. + +### Expected Outcomes +- `payback.py` exists with correct module docstring and `from __future__ import annotations`. +- Two `@dataclass(frozen=True)` classes: `AnalysisCost` and `Payback`. +- Three public functions: `analysis_cost`, `payback`, `format_payback`. +- All arithmetic matches the spec exactly (see Relevant Context). +- Module is importable; mypy passes on it. + +### Todo List + +1. Create `src/downshift/payback.py` with module docstring: + _"Analysis cost and payback for one downshift report run."_ + +2. Define module-level constants: + ``` + JUDGE_EXTRA_PROMPT_TOKENS = 150 + JUDGE_COMPLETION_TOKENS = 200 + ``` + +3. Define `AnalysisCost` (frozen dataclass): + - Fields: `model_calls: int`, `model_cost: float`, `judge_calls: int`, + `judge_cost: float`, `judge_model: str | None`, `judge_priced_as: str | None`, + `audit_cost: float = 0.0` + - Property `total -> float` = `model_cost + judge_cost + audit_cost` + +4. Define `Payback` (frozen dataclass): + - Field: `hours: float | None` + +5. Implement `analysis_cost(sites, results_dir, config, *, audit_cost=0.0) -> AnalysisCost`: + - Import `load_results` and `results_path` from `runner.py` — do not write a new parser. + - For each call site × each model: call `results_path(results_dir, site_id, model)`; + if it does not exist, skip. + - Call `load_results(path)` to get `dict[str, ResultRow]` (last row per case wins + because that is how `load_results` works). + - For each row (unique case_id → ResultRow): + - Skip if `row.error is not None`. + - Look up `config.pricing.get(row.model)` — skip silently if None (no price). + - Add `decide.call_cost(price, row.prompt_tokens, row.completion_tokens)` to + `model_cost`; increment `model_calls`. + - For each row with `row.judge_model` set (and not an error): + - Estimate prompt_tokens = `row.prompt_tokens + row.completion_tokens + JUDGE_EXTRA_PROMPT_TOKENS` + - Estimate completion = `JUDGE_COMPLETION_TOKENS` + - Look up judge price from `config.pricing.get(row.judge_model)`; + if None, fall back to `config.pricing.get(baseline_model)` — set + `judge_priced_as` to whichever model's price was used. + - Add `decide.call_cost(judge_price, prompt_est, JUDGE_COMPLETION_TOKENS)` to + `judge_cost`; increment `judge_calls`. + - `audit_cost` must be `>= 0`; raise `ValueError` otherwise. + - Return `AnalysisCost(...)`. + - The `judge_model` field on `AnalysisCost` is the first judge model seen in + any row (or `None` if no judge rows). + +6. Implement `payback(total_one_time: float, monthly_savings: float) -> Payback`: + - Import `HOURS_PER_MONTH` from `cost.py`. + - Return `Payback(hours=None)` if `monthly_savings <= 0`. + - Otherwise: `hours = total_one_time / (monthly_savings / HOURS_PER_MONTH)`. + +7. Implement `format_payback(p: Payback) -> str`: + - `None` → `"no payback (no projected savings)"` + - `< 1 hour` → `"N minutes"` (`math.ceil(hours * 60)`, minimum 1) + - `< 48 hours` → `"X.Y hours"` (one decimal, e.g. `"1.5 hours"`) + - `>= 48 hours` → `"N days"` (`math.ceil(hours / 24)`) + +### Relevant Context +- `runner.load_results(path)` → `dict[str, ResultRow]`; last row wins; already skips + unreadable lines. +- `runner.results_path(results_dir, site_id, model)` returns the JSONL path. +- `decide.call_cost(price, prompt_tokens, completion_tokens)` computes USD per call. +- `cost.HOURS_PER_MONTH = 730` — import directly, do not duplicate. +- `config.pricing` is a `Mapping[str, ModelPrice]`; `.get(model)` returns `None` + when the model has no price. +- Baseline model is `config.models.baseline`. + +--- + +## Sub-task 2 — `src/downshift/report.py` (edit) + +**Status:** `[ ] pending` + +### Intent +Thread `AnalysisCost | None` and `Payback | None` through `Report`, compute them +in `build_report`, and render the new section in `render_markdown`. + +### Expected Outcomes +- `Report` has two new optional fields; all existing tests still pass. +- `build_report` accepts `audit_cost: float = 0.0` and populates the new fields. +- `render_markdown` emits the "What this analysis cost" section right after + the Summary section, conditional on `report.analysis is not None`. + +### Todo List + +1. Add imports at the top of `report.py`: + ```python + from downshift.payback import AnalysisCost, Payback + from downshift.payback import analysis_cost as _analysis_cost + from downshift.payback import format_payback, payback as _payback + ``` + +2. Add two fields to the `Report` dataclass (after `min_pass_rate`): + ```python + analysis: AnalysisCost | None = None + payback: Payback | None = None + ``` + These are optional (default `None`) so all existing code that builds `Report` + directly (in tests) continues to work without changes. + +3. In `build_report`, add parameter `audit_cost: float = 0.0` (keyword-only after the + existing keyword args). + After `cost_summary(...)`: + ```python + ac = _analysis_cost(decided_sites, results_dir, config, audit_cost=audit_cost) + pb = _payback(ac.total, costs.savings) + ``` + Pass both to the `Report(...)` constructor. + +4. In `render_markdown`, insert the new section immediately after the blank line + that follows the Summary section (i.e., after the "Downgraded N of M" paragraph). + Only emit when `report.analysis is not None`. + + Section template: + ```markdown + ## What this analysis cost + + | | One-time cost | + |---|---:| + | Eval model calls (N) | $X | + | Judge calls, estimated (M) | $Y | <- omit when judge_calls == 0 + | Assistant audit | $Z | <- omit when audit_cost == 0 + | **Total** | **$T** | + + Pays back in **** of projected savings. + + > Priced at the same illustrative prices as the rest of the report. Judge tokens are not + > recorded, so each judge call is estimated as (case prompt + output + 150) tokens in and + > 200 out, priced as ``. Retries and warm-up calls are not counted. + > Local Ollama runs cost $0 in practice. + ``` + - Reuse the existing `_fmt_money` helper for all dollar values. + - Drop the judge row **and** the judge sentence in the blockquote when `judge_calls == 0`. + - The `judge_priced_as` model name goes in the backtick span in the blockquote. + +### Relevant Context +- `_fmt_money(x)` already exists in `report.py` and handles `None`. +- The summary section ends at the blank line after the "Downgraded N of M …" paragraph + (line 214 area in the current file). The new section is inserted there. +- Existing field order in `Report`: `sites`, `decisions`, `costs`, `missing_evals`, + `baseline`, `candidates`, `threshold`, `min_pass_rate`. New fields append after. + +--- + +## Sub-task 3 — `src/downshift/cli.py` (edit) + +**Status:** `[ ] pending` + +### Intent +Expose `--audit-cost` on the `report` command so users can pass the cost of an +AI-assisted audit session. + +### Expected Outcomes +- `downshift report --audit-cost 0.5` passes `audit_cost=0.5` to `build_report`. +- `downshift report --audit-cost -1` is rejected (typer `min=0.0`). +- The summary echo line in `--out` mode is unchanged. + +### Todo List + +1. Add to the `report` command's parameter list: + ```python + audit_cost: float = typer.Option( + 0.0, + "--audit-cost", + min=0.0, + help="One-time cost of an assistant audit, in USD, added to the analysis cost.", + ) + ``` + +2. Pass `audit_cost=audit_cost` to the `build_report(...)` call inside the `report` + command body. + +### Relevant Context +- `report` command is defined at line 668 in `cli.py`. +- The `build_report` call is at line 709. Add `audit_cost=audit_cost` as a + keyword argument. +- All other `build_report` call sites (`export` command) do not receive `--audit-cost`; + they use the default `0.0`. + +--- + +## Sub-task 4 — `src/downshift/export.py` (edit) + +**Status:** `[ ] pending` + +### Intent +Add the five new payback keys to the `summary.json` payload so the web app can +display them. + +### Expected Outcomes +- `summary_payload` returns a dict with five additional keys under a new + `"analysis"` sub-object (or flat, per spec — see Relevant Context). +- All existing `test_export.py` assertions still pass. + +### Todo List + +1. Add imports in `export.py`: + ```python + from downshift.payback import ( + AnalysisCost, + Payback, + analysis_cost as _analysis_cost, + payback as _payback, + ) + ``` + +2. Change `summary_payload` signature to accept the new fields. The cleanest + approach is to read them directly from `report.analysis` and `report.payback` + (both already `None`-safe). + +3. Add the following keys to the returned dict in `summary_payload` (flat, as the + spec says): + ```python + "analysis_cost_total": _round(report.analysis.total, 4) if report.analysis else None, + "analysis_model_cost": _round(report.analysis.model_cost, 4) if report.analysis else None, + "analysis_judge_cost": _round(report.analysis.judge_cost, 4) if report.analysis else None, + "analysis_calls": (report.analysis.model_calls + report.analysis.judge_calls) + if report.analysis else None, + "payback_hours": _round(report.payback.hours, 2) if report.payback else None, + ``` + +### Relevant Context +- `summary_payload` is at line 106 in `export.py`. +- `report.analysis` and `report.payback` are `None` if `build_report` was called + without `results_dir` having any results — treat as `None` gracefully. +- The spec says "float or null" for `payback_hours`; `Payback.hours` is already + `float | None`. + +--- + +## Sub-task 5 — `tests/unit/test_payback.py` (new) + +**Status:** `[ ] pending` + +### Intent +Unit-test `payback.py` in isolation: math correctness, edge cases, and the +`analysis_cost` function with a synthetic results folder. + +### Expected Outcomes +- All tests pass with no network access and no model calls. +- `analysis_cost` tests use `tmp_path` + `runner.append_row` to write real JSONL files. + +### Todo List + +1. Test `payback` math: + - Zero monthly savings → `Payback(hours=None)`. + - Negative monthly savings → `Payback(hours=None)`. + - Hand-computed case: `total=730.0, monthly=730.0` → `hours=730.0`. + - Check formula: `hours = total / (monthly / 730)`. + +2. Test `format_payback`: + - `Payback(hours=None)` → `"no payback (no projected savings)"`. + - `Payback(hours=0.2)` → `"12 minutes"` (ceil(0.2*60)=12). + - `Payback(hours=1.04)` → `"1.0 hours"`. + - `Payback(hours=47.9)` → `"47.9 hours"`. + - `Payback(hours=50.0)` → `"3 days"` (ceil(50/24)=3). + +3. Test `analysis_cost` with a tmp results folder: + - Setup: two models (`"big"`, `"mid"`), one call site (`"app.py::classify"`), + two eval cases. + - Write real JSONL rows via `runner.append_row`. + - Include one row with `error="boom"` → must be skipped. + - Include one row where model is not in `config.pricing` → must be skipped + (model_calls unchanged, no crash). + - Include one judge row whose `judge_model` has no price → falls back to + baseline price, sets `judge_priced_as` to baseline. + - Verify `model_calls`, `judge_calls`, `model_cost`, `judge_cost` with + hand-computed values. + +4. Test `audit_cost >= 0` constraint: + - `analysis_cost(..., audit_cost=-0.01)` raises `ValueError`. + - `analysis_cost(..., audit_cost=0.0)` is fine. + +5. Test `AnalysisCost.total`: + - Hand-verify `model_cost + judge_cost + audit_cost`. + +### Relevant Context +- `runner.append_row(path, row)` creates the file and parent dir automatically. +- `runner.ResultRow` constructor: `(case_id, model, prompt_tokens=…, completion_tokens=…, score=…, passed=…, judge_model=…, error=…)`. +- Use `Config(pricing={...}, models=ModelsConfig(baseline="big"))` for a minimal config. +- `decide.call_cost(price, p, c)` is the same function `analysis_cost` uses. + +--- + +## Sub-task 6 — Extend existing tests + +**Status:** `[ ] pending` + +### Intent +Extend `tests/unit/test_report.py`, `tests/cli/test_report_cli.py`, and +`tests/unit/test_export.py` to cover the new behaviour. + +### Expected Outcomes +- `render_markdown` tests verify section presence/absence. +- CLI test verifies `--audit-cost` and negative value rejection. +- Export test verifies five new keys exist in `summary_payload`. + +### Todo List + +#### `tests/unit/test_report.py` additions + +1. `test_analysis_cost_section_present` — build a report with a real (or mock) + `AnalysisCost` on `report.analysis`; call `render_markdown`; assert + `"## What this analysis cost"` is in the output. + +2. `test_judge_row_absent_when_no_judge_calls` — set `judge_calls=0` on + `AnalysisCost`; assert the judge row is not in the markdown. + +3. `test_audit_row_only_when_audit_cost_nonzero` — set `audit_cost=0.0`; + assert `"Assistant audit"` is not in the markdown. Then set `audit_cost=1.5`; + assert it is present. + +4. For `test_analysis_cost_section_absent_when_no_analysis` — set + `report.analysis = None`; assert section is absent. + +#### `tests/cli/test_report_cli.py` additions + +5. `test_audit_cost_flag_accepted` — invoke with `--audit-cost 0.5`; assert + exit code 0 and `"## What this analysis cost"` in output. + +6. `test_audit_cost_negative_rejected` — invoke with `--audit-cost -0.01`; + assert exit code 2. + +#### `tests/unit/test_export.py` additions + +7. `test_summary_analysis_keys_present` — using the existing `data` fixture, + assert the five keys (`analysis_cost_total`, `analysis_model_cost`, + `analysis_judge_cost`, `analysis_calls`, `payback_hours`) are all present + in the summary payload (value may be `None` or numeric — just assert key exists). + +### Relevant Context +- In `test_report.py`, the `_make_report` helper builds `Report` directly. + The new optional fields default to `None`, so existing helper calls stay unchanged. + Add overloads / direct construction only in the new tests. +- `test_analysis_cost_section_present` can construct `AnalysisCost` directly + (`from downshift.payback import AnalysisCost`) and set it on `report = Report(..., analysis=ac)`. +- CLI tests use the real supportdesk fixture in `EXAMPLE`, which has a real + `results/` folder, so `analysis` will be non-None. +- `test_audit_cost_flag_accepted` calls the real pipeline end-to-end; the + `"## What this analysis cost"` section must appear. diff --git a/docs/specs/audit-payback.md b/docs/specs/audit-payback.md new file mode 100644 index 0000000..d281e9c --- /dev/null +++ b/docs/specs/audit-payback.md @@ -0,0 +1,91 @@ +# Spec: analysis cost and payback in `downshift report` + +## Why +The report shows monthly savings but not what it cost to find them. Add a one-time +"analysis cost" (eval model calls + judge calls, priced from the config) and a payback +time, so users can see the tool pays for itself. + +## Scope (files) +- NEW `src/downshift/payback.py` +- EDIT `src/downshift/report.py` (Report field, build_report, render_markdown) +- EDIT `src/downshift/cli.py` (`report` gets `--audit-cost`) +- EDIT `src/downshift/export.py` (summary gets the new fields) +- NEW `tests/unit/test_payback.py`; extend existing report, export and CLI tests +- Do NOT edit the report snapshot file. The user regenerates it. + +## payback.py +Constants: +- `JUDGE_EXTRA_PROMPT_TOKENS = 150` (rubric + instructions around the case) +- `JUDGE_COMPLETION_TOKENS = 200` (matches the judge's max_tokens default in scorer.py) + +Dataclasses (frozen): +- `AnalysisCost`: `model_calls: int`, `model_cost: float`, `judge_calls: int`, + `judge_cost: float`, `judge_model: str | None`, `judge_priced_as: str | None`, + `audit_cost: float` (default 0.0). Property `total` = model_cost + judge_cost + audit_cost. +- `Payback`: `hours: float | None` (None when monthly savings <= 0). + +Functions: +- `analysis_cost(sites, results_dir, config, *, audit_cost=0.0) -> AnalysisCost` + - For every call site and every model with a results file, read rows with the EXISTING + results loader in `runner.py` (last row per case wins). Do not write a new JSONL parser. + - Model cost: for each row, `decide.call_cost(price, row.prompt_tokens, row.completion_tokens)` + with `config.price_for(row_model)`. Count one call per row. Skip models with no price + (do not crash), and skip rows with an error. + - Judge cost: for each row with `judge_model` set, estimate + prompt = row.prompt_tokens + row.completion_tokens + JUDGE_EXTRA_PROMPT_TOKENS, + completion = JUDGE_COMPLETION_TOKENS. Price with the judge model's price if it is in + `config.pricing`, otherwise with the baseline model's price, and set `judge_priced_as` + to the model whose price was used. + - `audit_cost` is a user-supplied one-time amount in USD (e.g. what an AI-assistant audit + cost). Must be >= 0. +- `payback(total_one_time: float, monthly_savings: float) -> Payback` + - hours = total / (monthly_savings / cost.HOURS_PER_MONTH). None if monthly_savings <= 0. +- `format_payback(p: Payback) -> str` + - None -> "no payback (no projected savings)" + - < 1 hour -> "N minutes" (round up, minimum 1) + - < 48 hours -> "X.Y hours" (one decimal) + - otherwise -> "N days" (round up) + +## report.py +- `Report` gets `analysis: AnalysisCost | None` and `payback: Payback | None`. +- `build_report` computes both from the same results folder and config it already uses, + and accepts `audit_cost: float = 0.0`. +- `render_markdown` adds this section right after the Summary section: + +``` +## What this analysis cost + +| | One-time cost | +|---|---:| +| Eval model calls (N) | $X | +| Judge calls, estimated (M) | $Y | +| Assistant audit | $Z | <- only when audit_cost > 0 +| **Total** | **$T** | + +Pays back in **** of projected savings. + +> Priced at the same illustrative prices as the rest of the report. Judge tokens are not +> recorded, so each judge call is estimated as (case prompt + output + 150) tokens in and +> 200 out, priced as ``. Retries and warm-up calls are not counted. +> Local Ollama runs cost $0 in practice. +``` +- Money uses the existing `_fmt_money`. With M = 0, drop the judge row and the judge sentence. + +## cli.py +- `downshift report --audit-cost USD` (float, min 0, default 0). Help: "One-time cost of an + assistant audit, in USD, added to the analysis cost." + +## export.py +- The summary JSON gets `analysis_cost_total`, `analysis_model_cost`, `analysis_judge_cost`, + `analysis_calls` (model + judge), `payback_hours` (float or null). + +## Tests +- payback math with hand-computed values; format_payback for None, 0.2 h, 1.04 h, 47.9 h, 50 h. +- analysis_cost with a tmp results folder: two models, one judge-graded site, one row with an + error (skipped), a judge model with no price (priced as baseline), a model with no price (skipped). +- render_markdown: section present, judge row absent when there are no judge calls, audit row + only when audit_cost > 0. +- CLI: `--audit-cost 0.5` shows the audit row; negative value is rejected. +- export: new keys present. +- Follow repo standards: ruff (E,F,I,B,UP,SIM, line length 100), mypy clean, `zip(strict=True)`, + no network, FakeLLMClient not needed. diff --git a/examples/supportdesk/downshift.report.md b/examples/supportdesk/downshift.report.md index bde827e..11b75e4 100644 --- a/examples/supportdesk/downshift.report.md +++ b/examples/supportdesk/downshift.report.md @@ -15,6 +15,18 @@ Downgraded **3 of 8** call sites. Rule: a cheaper model must keep at least 95% of the baseline pass rate and pass at least 80% of cases on its own. Decisions use pass rate, not mean score. +## What this analysis cost + +| | One-time cost | +|---|---:| +| Eval model calls (724) | $0.15 | +| Judge calls, estimated (176) | $0.50 | +| **Total** | **$0.65** | + +Pays back in **51 minutes** of projected savings. + +> Priced at the same illustrative prices as the rest of the report. Judge tokens are not recorded, so each judge call is estimated as (case prompt + output + 150) tokens in and 200 out, priced as `qwen2.5:7b`. Retries and warm-up calls are not counted. Local Ollama runs cost $0 in practice. + ## Decisions | Call site | Grading | Decision | Model | Pass rate | Before / month | After / month | diff --git a/src/downshift/cli.py b/src/downshift/cli.py index c4786ba..1ba1158 100644 --- a/src/downshift/cli.py +++ b/src/downshift/cli.py @@ -690,6 +690,12 @@ def report( help="Minimum pass rate override [0,1]. Default: from config.", ), out: Path | None = typer.Option(None, "--out", help="Write Markdown to this file."), + audit_cost: float = typer.Option( + 0.0, + "--audit-cost", + min=0.0, + help="One-time cost of an assistant audit, in USD, added to the analysis cost.", + ), ) -> None: """Render the cost and quality report.""" if threshold is not None and threshold <= 0: @@ -713,6 +719,7 @@ def report( results_dir, threshold=threshold, min_pass_rate=min_pass_rate, + audit_cost=audit_cost, ) except ReportError as exc: _fail(str(exc)) diff --git a/src/downshift/export.py b/src/downshift/export.py index 28b0be5..461d53b 100644 --- a/src/downshift/export.py +++ b/src/downshift/export.py @@ -150,6 +150,21 @@ def summary_payload(report: Report, config: Config, rows: Rows, *, project: str) }, "pricing": pricing, "disclaimer": DISCLAIMER, + "analysis_cost_total": ( + _round(report.analysis.total, 4) if report.analysis is not None else None + ), + "analysis_model_cost": ( + _round(report.analysis.model_cost, 4) if report.analysis is not None else None + ), + "analysis_judge_cost": ( + _round(report.analysis.judge_cost, 4) if report.analysis is not None else None + ), + "analysis_calls": ( + (report.analysis.model_calls + report.analysis.judge_calls) + if report.analysis is not None + else None + ), + "payback_hours": (_round(report.payback.hours, 2) if report.payback is not None else None), } diff --git a/src/downshift/payback.py b/src/downshift/payback.py new file mode 100644 index 0000000..3e92611 --- /dev/null +++ b/src/downshift/payback.py @@ -0,0 +1,172 @@ +"""Analysis cost and payback for one downshift report run. + +`analysis_cost` prices the eval-model calls and judge calls that were made +to produce a downshift report. `payback` converts a one-time cost plus a +monthly savings figure into a payback period. `format_payback` renders the +period as a human-readable string. +""" + +from __future__ import annotations + +import math +from collections import Counter +from collections.abc import Sequence +from dataclasses import dataclass +from pathlib import Path + +from downshift.config import Config +from downshift.cost import HOURS_PER_MONTH +from downshift.decide import call_cost +from downshift.runner import load_results, results_path +from downshift.schema import CallSite + +JUDGE_EXTRA_PROMPT_TOKENS = 150 +JUDGE_COMPLETION_TOKENS = 200 + + +@dataclass(frozen=True) +class AnalysisCost: + """One-time cost of the evaluation run that produced a downshift report.""" + + model_calls: int + model_cost: float + judge_calls: int + judge_cost: float + judge_model: str | None # most-frequent judge model seen in results + judge_priced_as: str | None # model whose price was used for judge calls + audit_cost: float = 0.0 + + @property + def total(self) -> float: + return self.model_cost + self.judge_cost + self.audit_cost + + +@dataclass(frozen=True) +class Payback: + """Payback period expressed in hours (None when monthly savings <= 0).""" + + hours: float | None + + +def analysis_cost( + sites: Sequence[CallSite], + results_dir: Path, + config: Config, + *, + audit_cost: float = 0.0, +) -> AnalysisCost: + """Compute the one-time cost of running evals for *sites*. + + For each call site × each model (baseline + candidates), the last result + row per case is used (that is how :func:`runner.load_results` works). Rows + with ``error`` set are skipped. Models without a price entry in + ``config.pricing`` are also skipped silently. + + The judge model is the most frequent ``judge_model`` value seen across all + non-error rows; ties are broken alphabetically. All judge rows are priced + with that model's price, or the baseline model's price when the judge model + has no price entry. + + *audit_cost* is a user-supplied one-time amount in USD (e.g. the cost of an + AI-assisted audit session). It must be >= 0. + """ + if audit_cost < 0: + raise ValueError(f"audit_cost must be >= 0, got {audit_cost}") + + models = list(config.models.all_models) + pricing = config.pricing + baseline = config.models.baseline + + model_calls = 0 + model_cost_total = 0.0 + judge_calls = 0 + judge_cost_total = 0.0 + judge_model_counter: Counter[str] = Counter() + + # First pass: accumulate model costs and count judge models. + # We also collect judge rows to price in a second pass once the dominant + # judge model is known. + judge_rows: list[tuple[int, int, str]] = [] # (prompt_tok, compl_tok, judge_model_name) + + for site in sites: + for model in models: + path = results_path(results_dir, site.id, model) + if not path.is_file(): + continue + rows = load_results(path) + for row in rows.values(): + if row.error is not None: + continue + # Model call cost + price = pricing.get(row.model) + if price is not None: + model_cost_total += call_cost(price, row.prompt_tokens, row.completion_tokens) + model_calls += 1 + # Judge call accounting + if row.judge_model is not None: + judge_model_counter[row.judge_model] += 1 + prompt_est = ( + row.prompt_tokens + row.completion_tokens + JUDGE_EXTRA_PROMPT_TOKENS + ) + judge_rows.append((prompt_est, JUDGE_COMPLETION_TOKENS, row.judge_model)) + + # Resolve dominant judge model (most frequent; ties → alphabetically first). + dominant_judge: str | None = None + judge_priced_as: str | None = None + if judge_model_counter: + dominant_judge = min( + judge_model_counter, + key=lambda m: (-judge_model_counter[m], m), + ) + # Determine which model's price to use. + if pricing.get(dominant_judge) is not None: + judge_priced_as = dominant_judge + elif pricing.get(baseline) is not None: + judge_priced_as = baseline + # else no price available at all; judge cost stays 0 + + if judge_priced_as is not None: + judge_price = pricing[judge_priced_as] + for prompt_est, compl_est, _jm in judge_rows: + judge_cost_total += call_cost(judge_price, prompt_est, compl_est) + judge_calls += 1 + + return AnalysisCost( + model_calls=model_calls, + model_cost=model_cost_total, + judge_calls=judge_calls, + judge_cost=judge_cost_total, + judge_model=dominant_judge, + judge_priced_as=judge_priced_as, + audit_cost=audit_cost, + ) + + +def payback(total_one_time: float, monthly_savings: float) -> Payback: + """Return the payback period for a one-time cost given a monthly saving. + + Returns ``Payback(hours=None)`` when *monthly_savings* <= 0. + """ + if monthly_savings <= 0: + return Payback(hours=None) + hours = total_one_time / (monthly_savings / HOURS_PER_MONTH) + return Payback(hours=hours) + + +def format_payback(p: Payback) -> str: + """Render a :class:`Payback` as a human-readable string. + + - ``None`` → ``"no payback (no projected savings)"`` + - < 1 hour → ``"N minutes"`` (ceil, minimum 1) + - < 48 hours → ``"X.Y hours"`` (one decimal) + - >= 48 hours → ``"N days"`` (ceil) + """ + if p.hours is None: + return "no payback (no projected savings)" + if p.hours < 1.0: + minutes = max(1, math.ceil(p.hours * 60)) + return f"{minutes} minute" if minutes == 1 else f"{minutes} minutes" + if p.hours < 48.0: + return f"{p.hours:.1f} hours" + days = math.ceil(p.hours / 24) + return f"{days} days" diff --git a/src/downshift/report.py b/src/downshift/report.py index b177deb..34cf2fd 100644 --- a/src/downshift/report.py +++ b/src/downshift/report.py @@ -22,6 +22,9 @@ load_site_stats, ) from downshift.evals import EVAL_SUFFIX, EvalError, load_eval_set, slug_for +from downshift.payback import AnalysisCost, Payback, format_payback +from downshift.payback import analysis_cost as _analysis_cost +from downshift.payback import payback as _payback from downshift.schema import CallSite, ScanResult NEAR_MISS_MARGIN = 0.05 @@ -43,6 +46,8 @@ class Report: candidates: tuple[str, ...] threshold: float min_pass_rate: float + analysis: AnalysisCost | None = None + payback: Payback | None = None @property def downgraded(self) -> tuple[Decision, ...]: @@ -89,6 +94,7 @@ def build_report( *, threshold: float | None = None, min_pass_rate: float | None = None, + audit_cost: float = 0.0, ) -> Report: """Assemble a Report from disk. @@ -131,6 +137,8 @@ def build_report( decisions.append(decision) costs = cost_summary(decisions, config.pricing, config.volume) + ac = _analysis_cost(decided_sites, results_dir, config, audit_cost=audit_cost) + pb = _payback(ac.total, costs.savings) return Report( sites=tuple(decided_sites), @@ -141,6 +149,8 @@ def build_report( candidates=config.models.candidates, threshold=eff_threshold, min_pass_rate=eff_min_pass_rate, + analysis=ac, + payback=pb, ) @@ -232,6 +242,40 @@ def render_markdown(report: Report) -> str: ) lines.append("") + # ------------------------------------------------------------------ analysis cost + if report.analysis is not None: + ac = report.analysis + lines.append("## What this analysis cost") + lines.append("") + lines.append("| | One-time cost |") + lines.append("|---|---:|") + lines.append(f"| Eval model calls ({ac.model_calls}) | {_fmt_money(ac.model_cost)} |") + if ac.judge_calls > 0: + lines.append( + f"| Judge calls, estimated ({ac.judge_calls}) | {_fmt_money(ac.judge_cost)} |" + ) + if ac.audit_cost > 0: + lines.append(f"| Assistant audit | {_fmt_money(ac.audit_cost)} |") + lines.append(f"| **Total** | **{_fmt_money(ac.total)}** |") + lines.append("") + pb_str = format_payback(report.payback) if report.payback is not None else "n/a" + lines.append(f"Pays back in **{pb_str}** of projected savings.") + lines.append("") + blockquote = "> Priced at the same illustrative prices as the rest of the report." + if ac.judge_calls > 0: + blockquote += ( + " Judge tokens are not recorded, so each judge call is estimated as" + " (case prompt + output + 150) tokens in and 200 out" + ) + if ac.judge_priced_as is not None: + blockquote += f", priced as `{ac.judge_priced_as}`" + blockquote += "." + blockquote += ( + " Retries and warm-up calls are not counted. Local Ollama runs cost $0 in practice." + ) + lines.append(blockquote) + lines.append("") + # ------------------------------------------------------------------ decisions table lines.append("## Decisions") lines.append("") diff --git a/tests/cli/test_report_cli.py b/tests/cli/test_report_cli.py index b1e4025..2aaa04c 100644 --- a/tests/cli/test_report_cli.py +++ b/tests/cli/test_report_cli.py @@ -113,3 +113,19 @@ def test_invalid_threshold_above_one_exits_2() -> None: def test_invalid_min_pass_rate_above_one_exits_2() -> None: result = invoke("--min-pass-rate", "1.1") assert result.exit_code == 2 + + +# --------------------------------------------------------------------------- +# --audit-cost +# --------------------------------------------------------------------------- + + +def test_audit_cost_flag_accepted() -> None: + result = invoke("--audit-cost", "0.5") + assert result.exit_code == 0, result.output + assert "## What this analysis cost" in result.output + + +def test_audit_cost_negative_rejected() -> None: + result = invoke("--audit-cost", "-0.01") + assert result.exit_code == 2 diff --git a/tests/snapshots/supportdesk_report.md b/tests/snapshots/supportdesk_report.md index bde827e..11b75e4 100644 --- a/tests/snapshots/supportdesk_report.md +++ b/tests/snapshots/supportdesk_report.md @@ -15,6 +15,18 @@ Downgraded **3 of 8** call sites. Rule: a cheaper model must keep at least 95% of the baseline pass rate and pass at least 80% of cases on its own. Decisions use pass rate, not mean score. +## What this analysis cost + +| | One-time cost | +|---|---:| +| Eval model calls (724) | $0.15 | +| Judge calls, estimated (176) | $0.50 | +| **Total** | **$0.65** | + +Pays back in **51 minutes** of projected savings. + +> Priced at the same illustrative prices as the rest of the report. Judge tokens are not recorded, so each judge call is estimated as (case prompt + output + 150) tokens in and 200 out, priced as `qwen2.5:7b`. Retries and warm-up calls are not counted. Local Ollama runs cost $0 in practice. + ## Decisions | Call site | Grading | Decision | Model | Pass rate | Before / month | After / month | diff --git a/tests/unit/test_export.py b/tests/unit/test_export.py index 4e3047b..b1de2d8 100644 --- a/tests/unit/test_export.py +++ b/tests/unit/test_export.py @@ -210,3 +210,19 @@ def test_write_export(data: dict[str, Any], tmp_path: Path) -> None: assert (out / REPORT_FILE).read_text(encoding="utf-8") == md committed = (SD / "downshift.report.md").read_text(encoding="utf-8") assert md == committed + + +def test_summary_analysis_keys_present(data: dict[str, Any]) -> None: + """The five new analysis/payback keys must be present in the summary payload.""" + s = data["payloads"][SUMMARY_FILE] + for key in ( + "analysis_cost_total", + "analysis_model_cost", + "analysis_judge_cost", + "analysis_calls", + "payback_hours", + ): + assert key in s, f"missing key: {key}" + # The supportdesk example has results, so values should be non-None + assert s["analysis_cost_total"] is not None + assert s["analysis_calls"] is not None diff --git a/tests/unit/test_payback.py b/tests/unit/test_payback.py new file mode 100644 index 0000000..b244db8 --- /dev/null +++ b/tests/unit/test_payback.py @@ -0,0 +1,363 @@ +"""Tests for payback.py: AnalysisCost, Payback, and the three public functions.""" + +from __future__ import annotations + +from pathlib import Path + +import pytest + +from downshift.config import Config, ModelPrice, ModelsConfig +from downshift.cost import HOURS_PER_MONTH +from downshift.decide import call_cost +from downshift.payback import ( + JUDGE_COMPLETION_TOKENS, + JUDGE_EXTRA_PROMPT_TOKENS, + AnalysisCost, + Payback, + analysis_cost, + format_payback, + payback, +) +from downshift.runner import ResultRow, append_row, results_path +from downshift.schema import CallSite, ModelRef + +# --------------------------------------------------------------------------- +# Helpers +# --------------------------------------------------------------------------- + +PRICES = { + "big": ModelPrice(2.50, 10.00), + "mid": ModelPrice(0.15, 0.60), +} + + +def _config( + baseline: str = "big", + candidates: tuple[str, ...] = ("mid",), + pricing: dict[str, ModelPrice] | None = None, +) -> Config: + return Config( + models=ModelsConfig(baseline=baseline, candidates=candidates), + pricing=pricing if pricing is not None else PRICES, + ) + + +def _site(site_id: str = "app.py::classify") -> CallSite: + return CallSite( + id=site_id, + file=site_id.split("::")[0], + line=1, + function=site_id.split("::")[-1], + api="openai", + model=ModelRef(value="big", source="literal", expression="big"), + ) + + +def _ok_row( + case_id: str, + model: str, + prompt_tokens: int = 100, + completion_tokens: int = 20, + judge_model: str | None = None, +) -> ResultRow: + return ResultRow( + case_id=case_id, + model=model, + prompt_tokens=prompt_tokens, + completion_tokens=completion_tokens, + score=1.0, + passed=True, + judge_model=judge_model, + ) + + +def _err_row(case_id: str, model: str) -> ResultRow: + return ResultRow(case_id=case_id, model=model, error="boom") + + +# --------------------------------------------------------------------------- +# payback() +# --------------------------------------------------------------------------- + + +def test_payback_zero_savings_gives_none() -> None: + assert payback(100.0, 0.0).hours is None + + +def test_payback_negative_savings_gives_none() -> None: + assert payback(100.0, -5.0).hours is None + + +def test_payback_formula() -> None: + # total=730, monthly=730 → hours = 730 / (730/730) = 730 + result = payback(730.0, 730.0) + assert result.hours == pytest.approx(HOURS_PER_MONTH) + + +def test_payback_formula_general() -> None: + # total=10, monthly=20 → hours = 10 / (20/730) = 10 * 730/20 = 365 + result = payback(10.0, 20.0) + assert result.hours == pytest.approx(10.0 / (20.0 / HOURS_PER_MONTH)) + + +# --------------------------------------------------------------------------- +# format_payback() +# --------------------------------------------------------------------------- + + +def test_format_payback_none_hours() -> None: + assert format_payback(Payback(hours=None)) == "no payback (no projected savings)" + + +def test_format_payback_minutes() -> None: + # 0.2 h → ceil(0.2*60)=12 minutes + assert format_payback(Payback(hours=0.2)) == "12 minutes" + + +def test_format_payback_minimum_one_minute() -> None: + # Very short but > 0 hours → at least 1 minute + result = format_payback(Payback(hours=0.001)) + assert result == "1 minute" + + +def test_format_payback_hours_one_decimal() -> None: + # 1.04 h < 48 h → "1.0 hours" + assert format_payback(Payback(hours=1.04)) == "1.0 hours" + + +def test_format_payback_just_below_48h() -> None: + # 47.9 h < 48 → hours format + assert format_payback(Payback(hours=47.9)) == "47.9 hours" + + +def test_format_payback_days() -> None: + # 50 h → ceil(50/24)=3 days + assert format_payback(Payback(hours=50.0)) == "3 days" + + +def test_format_payback_exactly_48h() -> None: + # 48.0 h → ceil(48/24)=2 days + assert format_payback(Payback(hours=48.0)) == "2 days" + + +# --------------------------------------------------------------------------- +# AnalysisCost.total +# --------------------------------------------------------------------------- + + +def test_analysis_cost_total_property() -> None: + ac = AnalysisCost( + model_calls=10, + model_cost=1.0, + judge_calls=5, + judge_cost=0.5, + judge_model=None, + judge_priced_as=None, + audit_cost=0.25, + ) + assert ac.total == pytest.approx(1.75) + + +def test_analysis_cost_total_default_audit_cost() -> None: + ac = AnalysisCost( + model_calls=1, + model_cost=2.0, + judge_calls=0, + judge_cost=0.0, + judge_model=None, + judge_priced_as=None, + ) + assert ac.total == pytest.approx(2.0) + + +# --------------------------------------------------------------------------- +# analysis_cost(): audit_cost validation +# --------------------------------------------------------------------------- + + +def test_analysis_cost_negative_audit_cost_raises(tmp_path: Path) -> None: + with pytest.raises(ValueError, match="audit_cost"): + analysis_cost([], tmp_path, _config(), audit_cost=-0.01) + + +def test_analysis_cost_zero_audit_cost_ok(tmp_path: Path) -> None: + ac = analysis_cost([], tmp_path, _config(), audit_cost=0.0) + assert ac.audit_cost == 0.0 + + +# --------------------------------------------------------------------------- +# analysis_cost(): main logic +# --------------------------------------------------------------------------- + + +def test_analysis_cost_empty_sites(tmp_path: Path) -> None: + ac = analysis_cost([], tmp_path, _config()) + assert ac.model_calls == 0 + assert ac.judge_calls == 0 + assert ac.model_cost == pytest.approx(0.0) + assert ac.judge_cost == pytest.approx(0.0) + assert ac.judge_model is None + assert ac.judge_priced_as is None + + +def test_analysis_cost_skips_missing_files(tmp_path: Path) -> None: + site = _site() + # No files written → all skipped + ac = analysis_cost([site], tmp_path, _config()) + assert ac.model_calls == 0 + + +def test_analysis_cost_basic_model_cost(tmp_path: Path) -> None: + """Two cases on 'big', no errors, no judge → model_cost should match hand calc.""" + site = _site() + cfg = _config() + path = results_path(tmp_path, site.id, "big") + row1 = _ok_row("c1", "big", prompt_tokens=100, completion_tokens=20) + row2 = _ok_row("c2", "big", prompt_tokens=200, completion_tokens=30) + append_row(path, row1) + append_row(path, row2) + + ac = analysis_cost([site], tmp_path, cfg) + price = PRICES["big"] + expected = call_cost(price, 100, 20) + call_cost(price, 200, 30) + assert ac.model_calls == 2 + assert ac.model_cost == pytest.approx(expected) + assert ac.judge_calls == 0 + assert ac.judge_cost == pytest.approx(0.0) + + +def test_analysis_cost_skips_error_rows(tmp_path: Path) -> None: + """Error rows must not contribute to model_calls or model_cost.""" + site = _site() + path = results_path(tmp_path, site.id, "big") + append_row(path, _ok_row("c1", "big", 100, 20)) + append_row(path, _err_row("c2", "big")) + + ac = analysis_cost([site], tmp_path, _config()) + assert ac.model_calls == 1 # only c1 + + +def test_analysis_cost_skips_model_with_no_price(tmp_path: Path) -> None: + """A model absent from config.pricing is silently skipped.""" + site = _site() + # 'mid' is in candidates but not in pricing + cfg = _config(pricing={"big": PRICES["big"]}) # 'mid' has no price + path_big = results_path(tmp_path, site.id, "big") + path_mid = results_path(tmp_path, site.id, "mid") + append_row(path_big, _ok_row("c1", "big", 100, 20)) + append_row(path_mid, _ok_row("c2", "mid", 50, 10)) + + ac = analysis_cost([site], tmp_path, cfg) + # 'mid' row skipped; only 'big' row counted + assert ac.model_calls == 1 + price = PRICES["big"] + assert ac.model_cost == pytest.approx(call_cost(price, 100, 20)) + + +def test_analysis_cost_judge_priced_at_judge_model(tmp_path: Path) -> None: + """Judge row is priced with judge model's price when available.""" + site = _site() + judge_model = "mid" + cfg = _config(pricing={"big": PRICES["big"], "mid": PRICES["mid"]}) + path_big = results_path(tmp_path, site.id, "big") + row = _ok_row("c1", "big", prompt_tokens=100, completion_tokens=20, judge_model=judge_model) + append_row(path_big, row) + + ac = analysis_cost([site], tmp_path, cfg) + assert ac.judge_calls == 1 + assert ac.judge_model == judge_model + assert ac.judge_priced_as == judge_model + + prompt_est = 100 + 20 + JUDGE_EXTRA_PROMPT_TOKENS + expected_judge_cost = call_cost(PRICES[judge_model], prompt_est, JUDGE_COMPLETION_TOKENS) + assert ac.judge_cost == pytest.approx(expected_judge_cost) + + +def test_analysis_cost_judge_falls_back_to_baseline_price(tmp_path: Path) -> None: + """When judge model has no price entry, fall back to baseline price.""" + site = _site() + judge_model = "judge-unknown" + # judge_model not in pricing → fall back to baseline "big" + cfg = _config(pricing={"big": PRICES["big"], "mid": PRICES["mid"]}) + path_big = results_path(tmp_path, site.id, "big") + row = _ok_row("c1", "big", prompt_tokens=100, completion_tokens=20, judge_model=judge_model) + append_row(path_big, row) + + ac = analysis_cost([site], tmp_path, cfg) + assert ac.judge_calls == 1 + assert ac.judge_model == judge_model + assert ac.judge_priced_as == "big" # baseline + + prompt_est = 100 + 20 + JUDGE_EXTRA_PROMPT_TOKENS + expected_judge_cost = call_cost(PRICES["big"], prompt_est, JUDGE_COMPLETION_TOKENS) + assert ac.judge_cost == pytest.approx(expected_judge_cost) + + +def test_analysis_cost_judge_error_rows_skipped(tmp_path: Path) -> None: + """Error rows with judge_model set are not counted as judge calls.""" + site = _site() + path_big = results_path(tmp_path, site.id, "big") + # ok row with judge + append_row(path_big, _ok_row("c1", "big", judge_model="mid")) + # error row with judge_model — should be skipped + err_with_judge = ResultRow( + case_id="c2", + model="big", + prompt_tokens=50, + completion_tokens=10, + judge_model="mid", + error="scoring failed: timeout", + ) + append_row(path_big, err_with_judge) + + ac = analysis_cost([site], tmp_path, _config()) + assert ac.judge_calls == 1 # only c1 + + +def test_analysis_cost_most_frequent_judge_model(tmp_path: Path) -> None: + """Most frequent judge_model wins; tie broken alphabetically.""" + site = _site() + cfg = _config( + pricing={"big": PRICES["big"], "mid": PRICES["mid"], "alpha": ModelPrice(1.0, 2.0)} + ) + path_big = results_path(tmp_path, site.id, "big") + # alpha appears once, mid appears twice → mid wins + append_row(path_big, _ok_row("c1", "big", judge_model="alpha")) + append_row(path_big, _ok_row("c2", "big", judge_model="mid")) + append_row(path_big, _ok_row("c3", "big", judge_model="mid")) + + ac = analysis_cost([site], tmp_path, cfg) + assert ac.judge_model == "mid" + + +def test_analysis_cost_most_frequent_tie_alphabetical(tmp_path: Path) -> None: + """Tie in judge model frequency → alphabetically first wins.""" + site = _site() + cfg = _config( + pricing={"big": PRICES["big"], "alpha": ModelPrice(1.0, 2.0), "zeta": ModelPrice(1.0, 2.0)} + ) + path_big = results_path(tmp_path, site.id, "big") + append_row(path_big, _ok_row("c1", "big", judge_model="zeta")) + append_row(path_big, _ok_row("c2", "big", judge_model="alpha")) + + ac = analysis_cost([site], tmp_path, cfg) + assert ac.judge_model == "alpha" + + +def test_analysis_cost_two_sites(tmp_path: Path) -> None: + """Rows across two sites are all accumulated.""" + site1 = _site("a.py::f1") + site2 = _site("a.py::f2") + cfg = _config() + for site in (site1, site2): + path = results_path(tmp_path, site.id, "big") + append_row(path, _ok_row("c1", "big", 100, 20)) + + ac = analysis_cost([site1, site2], tmp_path, cfg) + assert ac.model_calls == 2 + + +def test_analysis_cost_audit_cost_included_in_total(tmp_path: Path) -> None: + ac = analysis_cost([], tmp_path, _config(), audit_cost=5.0) + assert ac.audit_cost == pytest.approx(5.0) + assert ac.total == pytest.approx(5.0) diff --git a/tests/unit/test_report.py b/tests/unit/test_report.py index 24fb91d..cd76f0a 100644 --- a/tests/unit/test_report.py +++ b/tests/unit/test_report.py @@ -566,3 +566,147 @@ def test_build_report_real_supportdesk() -> None: md = render_markdown(report) assert md.endswith("\n") assert "# Downshift report" in md + + +# --------------------------------------------------------------------------- +# render_markdown: What this analysis cost section +# --------------------------------------------------------------------------- + + +def test_analysis_cost_section_present() -> None: + """Section appears when report.analysis is set.""" + from downshift.payback import AnalysisCost, Payback + + d = decide([mk("big", 22), mk("mid", 22), mk("small", 20)]) + ac = AnalysisCost( + model_calls=10, + model_cost=0.05, + judge_calls=5, + judge_cost=0.02, + judge_model="mid", + judge_priced_as="mid", + audit_cost=0.0, + ) + pb = Payback(hours=8.0) + report = _make_report([d]) + report = Report( + sites=report.sites, + decisions=report.decisions, + costs=report.costs, + missing_evals=report.missing_evals, + baseline=report.baseline, + candidates=report.candidates, + threshold=report.threshold, + min_pass_rate=report.min_pass_rate, + analysis=ac, + payback=pb, + ) + md = render_markdown(report) + assert "## What this analysis cost" in md + assert "Eval model calls (10)" in md + assert "Judge calls, estimated (5)" in md + assert "Pays back in" in md + + +def test_analysis_cost_section_absent_when_no_analysis() -> None: + """Section is absent when report.analysis is None.""" + d = decide([mk("big", 22), mk("mid", 22), mk("small", 20)]) + report = _make_report([d]) + # _make_report does not set analysis; it defaults to None + md = render_markdown(report) + assert "## What this analysis cost" not in md + + +def test_judge_row_absent_when_no_judge_calls() -> None: + """Judge row is omitted when judge_calls == 0.""" + from downshift.payback import AnalysisCost, Payback + + d = decide([mk("big", 22), mk("mid", 22), mk("small", 20)]) + ac = AnalysisCost( + model_calls=5, + model_cost=0.01, + judge_calls=0, + judge_cost=0.0, + judge_model=None, + judge_priced_as=None, + ) + report = _make_report([d]) + report = Report( + sites=report.sites, + decisions=report.decisions, + costs=report.costs, + missing_evals=report.missing_evals, + baseline=report.baseline, + candidates=report.candidates, + threshold=report.threshold, + min_pass_rate=report.min_pass_rate, + analysis=ac, + payback=Payback(hours=None), + ) + md = render_markdown(report) + assert "## What this analysis cost" in md + assert "Judge calls" not in md + assert "Judge tokens are not recorded" not in md + + +def test_audit_row_absent_when_audit_cost_zero() -> None: + """Assistant audit row is omitted when audit_cost == 0.""" + from downshift.payback import AnalysisCost, Payback + + d = decide([mk("big", 22), mk("mid", 22), mk("small", 20)]) + ac = AnalysisCost( + model_calls=5, + model_cost=0.01, + judge_calls=0, + judge_cost=0.0, + judge_model=None, + judge_priced_as=None, + audit_cost=0.0, + ) + report = _make_report([d]) + report = Report( + sites=report.sites, + decisions=report.decisions, + costs=report.costs, + missing_evals=report.missing_evals, + baseline=report.baseline, + candidates=report.candidates, + threshold=report.threshold, + min_pass_rate=report.min_pass_rate, + analysis=ac, + payback=Payback(hours=None), + ) + md = render_markdown(report) + assert "Assistant audit" not in md + + +def test_audit_row_present_when_audit_cost_nonzero() -> None: + """Assistant audit row appears when audit_cost > 0.""" + from downshift.payback import AnalysisCost, Payback + + d = decide([mk("big", 22), mk("mid", 22), mk("small", 20)]) + ac = AnalysisCost( + model_calls=5, + model_cost=0.01, + judge_calls=0, + judge_cost=0.0, + judge_model=None, + judge_priced_as=None, + audit_cost=1.50, + ) + report = _make_report([d]) + report = Report( + sites=report.sites, + decisions=report.decisions, + costs=report.costs, + missing_evals=report.missing_evals, + baseline=report.baseline, + candidates=report.candidates, + threshold=report.threshold, + min_pass_rate=report.min_pass_rate, + analysis=ac, + payback=Payback(hours=2.0), + ) + md = render_markdown(report) + assert "Assistant audit" in md + assert "$1.50" in md diff --git a/web/app/page.tsx b/web/app/page.tsx index 332f4de..93d8332 100644 --- a/web/app/page.tsx +++ b/web/app/page.tsx @@ -7,7 +7,7 @@ import { LinkButton } from "@/components/Button"; import { DecisionBadge } from "@/components/Badge"; import { summary, callsites, audit, totalEvalCases } from "@/lib/data"; import { auditRows, sortBySavings, chosenStats, baselineStats } from "@/lib/select"; -import { formatUsd, formatPct, formatPts, siteName, siteFile } from "@/lib/format"; +import { formatUsd, formatPct, formatPts, formatPayback, siteName, siteFile } from "@/lib/format"; import { REPO_URL, RESULT_NOTES, AUDIT_INTRO, CI_DEMO } from "@/lib/content"; const WORDS = ["zero", "one", "two", "three", "four", "five", "six", "seven", "eight", "nine", "ten"]; @@ -57,6 +57,15 @@ function CostBars() { + {summary.analysis_cost_total != null && summary.payback_hours != null && ( +

+ One-time analysis cost{" "} + + {formatUsd(summary.analysis_cost_total)} + {" "} + · pays back in {formatPayback(summary.payback_hours ?? null)} +

+ )} ); } diff --git a/web/lib/format.ts b/web/lib/format.ts index b6ed4af..666056c 100644 --- a/web/lib/format.ts +++ b/web/lib/format.ts @@ -60,3 +60,13 @@ export function shortModel(model: string): string { const i = model.lastIndexOf(":"); return i >= 0 ? model.slice(i + 1) : model; } + +export function formatPayback(hours: number | null): string { + if (hours === null) return "n/a"; + if (hours < 1) { + const m = Math.max(1, Math.ceil(hours * 60)); + return m === 1 ? "1 minute" : `${m} minutes`; + } + if (hours < 48) return `${hours.toFixed(1)} hours`; + return `${Math.ceil(hours / 24)} days`; +} diff --git a/web/lib/formatPayback.test.ts b/web/lib/formatPayback.test.ts new file mode 100644 index 0000000..cfb9e0d --- /dev/null +++ b/web/lib/formatPayback.test.ts @@ -0,0 +1,10 @@ +import { describe, expect, it } from "vitest"; +import { formatPayback } from "./format"; + +describe("formatPayback", () => { + it("handles null", () => expect(formatPayback(null)).toBe("n/a")); + it("rounds minutes up", () => expect(formatPayback(0.84)).toBe("51 minutes")); + it("has a one minute floor", () => expect(formatPayback(0.001)).toBe("1 minute")); + it("shows hours under 48", () => expect(formatPayback(1.04)).toBe("1.0 hours")); + it("shows days from 48 hours", () => expect(formatPayback(50)).toBe("3 days")); +}); diff --git a/web/lib/types.ts b/web/lib/types.ts index 31dd479..d617b3f 100644 --- a/web/lib/types.ts +++ b/web/lib/types.ts @@ -6,6 +6,11 @@ export interface Pricing { tier: string | null; } export interface Summary { + analysis_cost_total?: number | null; + analysis_model_cost?: number | null; + analysis_judge_cost?: number | null; + analysis_calls?: number | null; + payback_hours?: number | null; project: string; tool_version: string; baseline: string; diff --git a/web/public/data/report.md b/web/public/data/report.md index bde827e..11b75e4 100644 --- a/web/public/data/report.md +++ b/web/public/data/report.md @@ -15,6 +15,18 @@ Downgraded **3 of 8** call sites. Rule: a cheaper model must keep at least 95% of the baseline pass rate and pass at least 80% of cases on its own. Decisions use pass rate, not mean score. +## What this analysis cost + +| | One-time cost | +|---|---:| +| Eval model calls (724) | $0.15 | +| Judge calls, estimated (176) | $0.50 | +| **Total** | **$0.65** | + +Pays back in **51 minutes** of projected savings. + +> Priced at the same illustrative prices as the rest of the report. Judge tokens are not recorded, so each judge call is estimated as (case prompt + output + 150) tokens in and 200 out, priced as `qwen2.5:7b`. Retries and warm-up calls are not counted. Local Ollama runs cost $0 in practice. + ## Decisions | Call site | Grading | Decision | Model | Pass rate | Before / month | After / month | diff --git a/web/public/data/summary.json b/web/public/data/summary.json index d426329..fb38acf 100644 --- a/web/public/data/summary.json +++ b/web/public/data/summary.json @@ -71,5 +71,10 @@ "tier": "nano" } ], - "disclaimer": "Dollar figures are projections: measured token counts x illustrative per-model prices x an assumed call volume, all set in downshift.yaml. They are not a real bill." + "disclaimer": "Dollar figures are projections: measured token counts x illustrative per-model prices x an assumed call volume, all set in downshift.yaml. They are not a real bill.", + "analysis_cost_total": 0.6476, + "analysis_model_cost": 0.1518, + "analysis_judge_cost": 0.4957, + "analysis_calls": 900, + "payback_hours": 0.84 }