From d63eb361095fbe720938760f880b3f3fe2d66fb7 Mon Sep 17 00:00:00 2001 From: macblackstuff <148771651+macblackstuff@users.noreply.github.com> Date: Wed, 30 Sep 2026 01:32:34 +0200 Subject: [PATCH 1/7] feat(certify): certification gate with blocker/advisory model (U1) --certify LEDGER validates a review ledger against the input: exit 0 + record when complete, exit 3 naming blockers, exit 1 on bad ledger rows. Generation behavior and exit codes 0/1/2 unchanged. 15 new tests (99 total, green). Co-Authored-By: Claude Claude-Session: sess_c59696de-2f36-4d7c-99cd-ad4102b8cc4e --- .../scripts/interface_matrix.py | 258 ++++++++++++++++-- .../scripts/test_interface_matrix.py | 219 +++++++++++++++ 2 files changed, 451 insertions(+), 26 deletions(-) diff --git a/skills/interface-matrix/scripts/interface_matrix.py b/skills/interface-matrix/scripts/interface_matrix.py index 37cd0e4..7015f4d 100644 --- a/skills/interface-matrix/scripts/interface_matrix.py +++ b/skills/interface-matrix/scripts/interface_matrix.py @@ -4,6 +4,17 @@ Input: one Markdown file holding a Components table and an Interfaces table. Output: a deterministic Markdown report on stdout. Exit 1 on a bad input row. Stdlib only. + +With --certify LEDGER the same input is certified instead of reported. The +ledger is a Markdown file holding one disposition table, columns +`Kind | Finding | Disposition | Reason`, whose rows disposition the report's +findings: a missing-component candidate by input line (`| candidate | line 21 | +... |`), an interface gap by input line, a boundary finding by component name, +an unstated pair as `A -> B`, an uncited source span as `L7-9`. Certification +exits 0 with a certification record when every finding is dispositioned, or 3 +naming every finding that is not; a gap dispositioned open (e.g. `open-parked`) +is an advisory in the record, not a failure. A malformed or duplicate ledger +row is bad input: exit 1, like a bad input row. """ import argparse @@ -17,6 +28,8 @@ RULE_COLUMNS = ("producer class", "consumer class", "disposition", "reason") ATTRS = ("Flows", "Format", "Trigger", "Owner") DISPOSITIONS = ("none", "review") +LEDGER_COLUMNS = ("kind", "finding", "disposition", "reason") +LEDGER_KINDS = ("candidate", "gap", "boundary", "pair", "span") CITE = re.compile(r"(?EOF.""" + """Report lines and uncited spans for `--source`: which source lines nothing + cites. Returns (lines, spans as (first, last)); exit 1 on L>EOF.""" with open(path, encoding="utf-8", errors="replace") as fh: lines = fh.read().splitlines() bad = sorted({(n, ln) for ln, n, _ in cites if n > len(lines) or n < 1}) @@ -458,35 +472,29 @@ def coverage(cites, path): else: out.append("Every non-blank source line is cited.") out.append("") - return out + return out, spans -def report(names, external, specified, gaps, nones, candidates, retired, class_of, - rules=(), sample_n=20, cov=None, has_rules=False): +def span_label(span): + """`L7` or `L7-9`, the label the coverage section and a ledger row share.""" + a, b = span + return "L%d" % a if a == b else "L%d-%d" % (a, b) + + +def edges_of(specified, gaps, nones): + """Stated pairs and their matrix marks: `X` specified, `g` gap; a `none` stated.""" edge_of = {} for iface in specified: edge_of[(iface["producer"], iface["consumer"])] = "X" for iface in gaps: edge_of.setdefault((iface["producer"], iface["consumer"]), "g") stated = set(edge_of) | {(i["producer"], i["consumer"]) for i in nones} + return edge_of, stated - edges = sorted(k for k in edge_of if k[0] != k[1]) - selfdeps = sorted({a for a, b in edge_of if a == b}) - blocks = partition(names, edges) - order = [n for block in blocks for n in block] - loops = [b for b in blocks if len(b) > 1] - - ins = {n: 0 for n in names} - outs = {n: 0 for n in names} - for a, b in edges: - outs[a] += 1 - ins[b] += 1 - internal = [n for n in names if n not in external] - isolated = [n for n in internal if not ins[n] and not outs[n]] - unfed = [n for n in internal if not ins[n] and outs[n]] - unconsumed = [n for n in internal if not outs[n] and ins[n]] - unstated = [ +def unstated_pairs(names, external, stated): + """Pairs no row states: neither an interface, a gap nor an explicit `none`.""" + return [ (a, b) for a in names for b in names @@ -495,10 +503,33 @@ def report(names, external, specified, gaps, nones, candidates, retired, class_o and not (a in external and b in external) ] - active_rules = [r for r in rules if not superseded(r["status"])] - unclassed = [n for n in names if not class_of.get(n, "")] +def boundary(names, external, edges): + """Internal components by boundary state: (internal, unfed, unconsumed, isolated). + Nobody feeds an unfed component, nothing consumes an unconsumed one's output, + an isolated one has no interface at all. A self-dependency is not an edge + here: it neither feeds nor consumes anyone else. + """ + ins = {n: 0 for n in names} + outs = {n: 0 for n in names} + for a, b in edges: + outs[a] += 1 + ins[b] += 1 + internal = [n for n in names if n not in external] + return (internal, + [n for n in internal if not ins[n] and outs[n]], + [n for n in internal if not outs[n] and ins[n]], + [n for n in internal if not ins[n] and not outs[n]]) + + +def settle(pairs, edge_of, rules, class_of): + """Match unstated pairs against the active class rules, as report() renders them. + + Returns (active_rules, residue, matched, settled, by_rule, overrides); + `residue` is the pairs no `none` rule settled — the pairs a review must + disposition one by one. + """ def side(want, name): """`*` matches any classed component; a blank Class matches nothing.""" cls = class_of.get(name, "") @@ -507,10 +538,11 @@ def side(want, name): def hit(rule, a, b): return side(rule["producer"], a) and side(rule["consumer"], b) + active_rules = [r for r in rules if not superseded(r["status"])] matched = {r["line"]: 0 for r in active_rules} settled = {r["line"]: 0 for r in active_rules} residue, by_rule = [], 0 - for a, b in unstated: + for a, b in pairs: hits = [r for r in active_rules if hit(r, a, b)] for r in hits: matched[r["line"]] += 1 @@ -528,6 +560,27 @@ def hit(rule, a, b): for r in active_rules if r["disposition"] == "none" and hit(r, a, b) ] + return active_rules, residue, matched, settled, by_rule, overrides + + +def report(names, external, specified, gaps, nones, candidates, retired, class_of, + rules=(), sample_n=20, cov=None, has_rules=False): + edge_of, stated = edges_of(specified, gaps, nones) + + edges = sorted(k for k in edge_of if k[0] != k[1]) + selfdeps = sorted({a for a, b in edge_of if a == b}) + blocks = partition(names, edges) + order = [n for block in blocks for n in block] + loops = [b for b in blocks if len(b) > 1] + + internal, unfed, unconsumed, isolated = boundary(names, external, edges) + + unstated = unstated_pairs(names, external, stated) + + active_rules, residue, matched, settled, by_rule, overrides = settle( + unstated, edge_of, rules, class_of) + + unclassed = [n for n in names if not class_of.get(n, "")] out = [] w = out.append @@ -663,9 +716,151 @@ def hit(rule, a, b): return "\n".join(out).rstrip("\n") + "\n" +def read_ledger(path): + """Ledger disposition rows: (kind, finding, disposition, reason, line). + + The disposition table is the ledger's one table naming Kind and Finding. A + row of the wrong width, an unknown kind, or a row without a finding or a + disposition is bad input, exactly like a bad row of the input file. + """ + with open(path, encoding="utf-8") as fh: + lines = fh.read().splitlines() + rows = [] + seen = 0 + i = 0 + while i < len(lines): + if not is_header(lines, i): + i += 1 + continue + head = [c.lower() for c in cells(lines[i])] + if not {"kind", "finding"} <= set(head): + i += 1 + continue + if seen: + die("second disposition table at line %d (first at line %d)" % (i + 1, seen)) + seen = i + 1 + at = columns(head, LEDGER_COLUMNS, i + 1) + i += 2 + while i < len(lines) and lines[i].strip().startswith("|") and not is_header(lines, i): + row = cells(lines[i]) + if len(row) != len(head): + die("ledger row at line %d has %d cells, expected %d" + % (i + 1, len(row), len(head))) + kind = row[at["kind"]].strip().lower() + if kind not in LEDGER_KINDS: + die("ledger row at line %d: unknown kind %r (expected one of: %s)" + % (i + 1, kind, ", ".join(LEDGER_KINDS))) + if not row[at["finding"]].strip() or not row[at["disposition"]].strip(): + die("ledger row at line %d needs a finding and a disposition" % (i + 1)) + rows.append((kind, row[at["finding"]].strip(), + row[at["disposition"]].strip(), row[at["reason"]].strip(), i + 1)) + i += 1 + if not seen: + die("no disposition table in %s (expected columns: %s)" + % (path, ", ".join(LEDGER_COLUMNS))) + return rows + + +def certify(args, built, rules, spans): + """Judge the ledger against the findings the report derives from the input. + + Returns (record, refused): the record names every blocker — a finding with + no ledger disposition — and every advisory, a gap dispositioned open, which + may legitimately stay open; refused means exit 3. + """ + names, external, specified, gaps, nones, candidates, retired, class_of = built + rows = read_ledger(args.certify) + ledger = {} + for kind, finding, disposition, reason, ln in rows: + if (kind, finding) in ledger: + die("ledger rows at lines %d and %d both disposition %s %r" + % (ledger[(kind, finding)][2], ln, kind, finding)) + ledger[(kind, finding)] = (disposition, reason, ln) + + edge_of, stated = edges_of(specified, gaps, nones) + internal, unfed, unconsumed, isolated = boundary( + names, external, sorted(k for k in edge_of if k[0] != k[1])) + residue = settle(unstated_pairs(names, external, stated), edge_of, rules, + class_of)[1] + why = {} + for group, label in ((unfed, "nothing feeds it"), + (unconsumed, "nothing consumes its output"), + (isolated, "isolated")): + for n in group: + why[n] = label + + known = { + "candidate": {"line %d" % c["line"] for c in candidates}, + "gap": {"line %d" % g["line"] for g in gaps}, + "boundary": set(why), + "pair": {"%s -> %s" % (a, b) for a, b in residue}, + "span": {span_label(s) for s in spans}, + } + for kind, finding, disposition, reason, ln in rows: + if kind == "span" and not args.source: + die("ledger row at line %d dispositions source span %r but certification " + "was invoked without --source" % (ln, finding)) + if finding not in known[kind]: + die("ledger row at line %d: %s %r matches no %s finding in the input" + % (ln, kind, finding, kind)) + + blockers = [] + for c in sorted(candidates, key=lambda c: c["line"]): + if ("candidate", "line %d" % c["line"]) not in ledger: + blockers.append("missing-component candidate line %d (%s -> %s) is unresolved" + % (c["line"], c["producer"], c["consumer"])) + for g in sorted(gaps, key=lambda g: g["line"]): + if ("gap", "line %d" % g["line"]) not in ledger: + blockers.append("interface gap line %d (%s -> %s, missing %s) has no disposition" + % (g["line"], g["producer"], g["consumer"], + ", ".join(g["missing"]))) + for n in names: + if n in why and ("boundary", n) not in ledger: + blockers.append("boundary finding %s (%s) is unexplained" % (n, why[n])) + for a, b in residue: + if ("pair", "%s -> %s" % (a, b)) not in ledger: + blockers.append("unstated pair %s -> %s has no disposition" % (a, b)) + for s in spans: + if ("span", span_label(s)) not in ledger: + blockers.append("uncited span %s of %s is unread" % (span_label(s), args.source)) + + advisories = [] + for g in sorted(gaps, key=lambda g: g["line"]): + entry = ledger.get(("gap", "line %d" % g["line"])) + if entry and entry[0].lower().startswith("open"): + advisories.append("gap line %d (%s -> %s): %s" + % (g["line"], g["producer"], g["consumer"], + " — ".join(x for x in entry[:2] if x))) + + if blockers: + sys.stderr.write("error: certification refused: %d blocker(s), named in the " + "record\n" % len(blockers)) + return certification_record(args, blockers, advisories), bool(blockers) + + +def certification_record(args, blockers, advisories): + """The certification record: U1 prints it, U2 writes it to a file with hashes.""" + out = ["# Interface matrix certification", ""] + out.append("- input: %s" % args.input) + out.append("- ledger: %s" % args.certify) + out.append("- gate: %s" % ("refused" if blockers else "certified")) + out.append("- flags: --sample %d%s" % (args.sample, + " --source %s" % args.source if args.source else "")) + out.append("- blockers: %s" % (len(blockers) if blockers else "none")) + out.extend(" - %s" % b for b in blockers) + out.append("- advisories: %s" % (len(advisories) if advisories else "none")) + out.extend(" - %s" % a for a in advisories) + return "\n".join(out) + "\n" + + def main(argv=None): ap = argparse.ArgumentParser(description=__doc__) ap.add_argument("input", help="Markdown file with Components and Interfaces tables") + ap.add_argument("--certify", metavar="LEDGER", + help="review ledger to certify the input against instead of printing " + "the report: exit 0 with a certification record, exit 3 naming " + "every undispositioned finding, exit 1 on a bad ledger row " + "(ledger format in the module docstring)") ap.add_argument("--sample", type=nonneg, default=20, help="unstated pairs to print (0 = all)") ap.add_argument("--source", help="source file whose L citations to check for coverage") args = ap.parse_args(argv) @@ -680,9 +875,20 @@ def main(argv=None): with open(args.input, encoding="utf-8") as fh: text = fh.read() components, interfaces, rules, cites, has_rules = parse(text) - cov = coverage(cites, args.source) if args.source else None - sys.stdout.write(report(*build(components, interfaces, rules), rules=rules, - sample_n=args.sample, cov=cov, + if args.source: + cov, spans = coverage(cites, args.source) + else: + cov, spans = None, [] + built = build(components, interfaces, rules) + if args.certify: + # the gate certifies what the report shows: render it once so certification + # dies on the same invariant (exit 2), then judge the ledger against the + # same derivation — the record replaces the rendered report + report(*built, rules=rules, sample_n=args.sample, has_rules=has_rules) + record, refused = certify(args, built, rules, spans) + sys.stdout.write(record) + return 3 if refused else 0 + sys.stdout.write(report(*built, rules=rules, sample_n=args.sample, cov=cov, has_rules=has_rules)) return 0 diff --git a/skills/interface-matrix/scripts/test_interface_matrix.py b/skills/interface-matrix/scripts/test_interface_matrix.py index 7fc9a23..915abc0 100644 --- a/skills/interface-matrix/scripts/test_interface_matrix.py +++ b/skills/interface-matrix/scripts/test_interface_matrix.py @@ -154,6 +154,13 @@ def test_valid_input_exits_0(self): proc = run(doc("| Ingest | Store | rows | csv | cron | me | S:L1 |\n")) self.assertEqual(proc.returncode, 0, proc.stderr) + def test_generation_with_candidates_still_exits_0(self): + # a missing-component candidate is a report finding, not an input error: + # only --certify refuses to pass one + proc = run(doc("| ? | Scorer | rows | csv | cron | me | |\n")) + self.assertEqual(proc.returncode, 0, proc.stderr) + self.assertIn("missing-component candidates: 1", proc.stdout) + class TestParsing(unittest.TestCase): def test_escaped_pipe_is_one_cell(self): @@ -1000,5 +1007,217 @@ def test_report_survives_a_non_utf8_console(self): self.assertNotIn("UnicodeEncodeError", proc.stderr) +LEDGER_HEAD = ( + "# Review ledger\n" + "\n" + "| Kind | Finding | Disposition | Reason |\n" + "|---|---|---|---|\n" +) + + +def ledger(*rows): + """A ledger file: its disposition table under a heading.""" + return LEDGER_HEAD + "".join(rows) + + +def run_certify(text, ledger_text, *args): + """Certify `text` against `ledger_text` written to its own temp file.""" + with tempfile.NamedTemporaryFile("w", suffix=".md", delete=False) as fh: + fh.write(ledger_text) + path = fh.name + try: + return run(text, "--certify", path, *args) + finally: + os.unlink(path) + + +def run_certify_source(text, ledger_text, source_text): + """Certify with --source, as run_source() is to run().""" + with tempfile.NamedTemporaryFile("w", suffix=".txt", delete=False) as fh: + fh.write(source_text) + src = fh.name + try: + return run_certify(text, ledger_text, "--source", src) + finally: + os.unlink(src) + + +TWO_COMPONENTS = ( + "## Components\n\n" + "| Component | Kind | Notes |\n" + "|---|---|---|\n" + "| Ingest | | pulls |\n" + "| Store | | keeps |\n\n" +) + +# one finding of each kind the wildcard rule cannot settle: a candidate (line 22), +# a gap (line 23) and all three boundary findings; the rule settles every classed pair +CERT_INPUT = doc_rules( + "| Ingest | Store | rows | csv | cron | me | S:L1 |\n" + "| ? | Scorer | digest | csv | cron | me | |\n" + "| Ingest | Analyst | rows | | ? | me | |\n", + "| * | * | none | every classed pair is settled |\n", +) + +FULL_LEDGER = ledger( + "| candidate | line 22 | resolved | producer is Ingest, row fixed upstream |\n" + "| gap | line 23 | open-parked | blocked on the vendor's format doc |\n" + "| boundary | Ingest | explained | the pipeline's entry point |\n" + "| boundary | Scorer | explained | runs on a manual trigger |\n" + "| boundary | Store | explained | the terminal sink |\n" +) + + +class TestCertification(unittest.TestCase): + def test_fully_reviewed_input_certifies(self): + proc = run_certify(CERT_INPUT, FULL_LEDGER) + self.assertEqual(proc.returncode, 0, proc.stderr) + self.assertIn("# Interface matrix certification", proc.stdout) + self.assertIn("- input: ", proc.stdout) + self.assertIn("- ledger: ", proc.stdout) + self.assertIn("- gate: certified", proc.stdout) + self.assertIn("- flags: --sample 20\n", proc.stdout) + self.assertIn("- blockers: none", proc.stdout) + self.assertIn("- advisories: 1", proc.stdout) + self.assertIn("gap line 23 (Ingest -> Analyst): open-parked — blocked on the " + "vendor's format doc", proc.stdout) + + def test_open_gap_is_an_advisory_and_a_filled_gap_is_not(self): + # the gap is a finding either way; an open disposition is an advisory in + # the record, a non-open one simply satisfies the gate + text = doc( + "| Ingest | Store | rows | | cron | me | S:L1 |\n" + "| Store | Ingest | acks | csv | cron | me | S:L1 |\n", + components=TWO_COMPONENTS, + ) + gap = "line %d" % lineno(text, "| Ingest | Store | rows | | cron") + opened = run_certify(text, ledger( + "| gap | %s | open-parked | waiting on the vendor |\n" % gap)) + self.assertEqual(opened.returncode, 0, opened.stderr) + self.assertIn("- advisories: 1", opened.stdout) + self.assertIn("gap %s (Ingest -> Store): open-parked — waiting on the vendor" + % gap, opened.stdout) + filled = run_certify(text, ledger( + "| gap | %s | filled | the attrs live in the ADR |\n" % gap)) + self.assertEqual(filled.returncode, 0, filled.stderr) + self.assertIn("- advisories: none", filled.stdout) + + def test_unresolved_candidate_blocks(self): + proc = run_certify(CERT_INPUT, ledger( + "| gap | line 23 | open-parked | blocked |\n" + "| boundary | Ingest | explained | the entry point |\n" + "| boundary | Scorer | explained | the manual trigger |\n" + "| boundary | Store | explained | the terminal sink |\n", + )) + self.assertEqual(proc.returncode, 3, proc.stderr) + self.assertIn("- gate: refused", proc.stdout) + self.assertIn("- blockers: 1", proc.stdout) + self.assertIn("missing-component candidate line 22 (? -> Scorer) is unresolved", + proc.stdout) + + def test_unstated_pair_without_disposition_blocks(self): + text = doc("| Ingest | Store | rows | csv | cron | me | S:L1 |\n", + components=TWO_COMPONENTS) + proc = run_certify(text, ledger( + "| boundary | Ingest | explained | the entry point |\n" + "| boundary | Store | explained | the terminal sink |\n", + )) + self.assertEqual(proc.returncode, 3, proc.stderr) + self.assertIn("- blockers: 1", proc.stdout) + self.assertIn("unstated pair Store -> Ingest has no disposition", proc.stdout) + self.assertNotIn("boundary finding", proc.stdout) + + def test_uncited_span_blocks_under_source_until_dispositioned(self): + text = doc("| Ingest | Store | rows | csv | cron | me | S:L1 |\n", + components=TWO_COMPONENTS) + led = ledger( + "| boundary | Ingest | explained | the entry point |\n" + "| boundary | Store | explained | the terminal sink |\n" + "| pair | Store -> Ingest | none | nothing flows back |\n", + ) + proc = run_certify_source(text, led, SOURCE) + self.assertEqual(proc.returncode, 3, proc.stderr) + self.assertIn("uncited span L2-6", proc.stdout) + read = run_certify_source(text, led + "| span | L2-6 | read | narrative prose |\n", + SOURCE) + self.assertEqual(read.returncode, 0, read.stderr) + self.assertIn("- gate: certified", read.stdout) + + def test_bad_input_row_exits_1_in_certify_mode_too(self): + proc = run_certify(doc("| Ingest | Nope | rows | csv | cron | me | S:L1 |\n"), + FULL_LEDGER) + self.assertEqual(proc.returncode, 1, proc.stdout) + self.assertIn("Nope", proc.stderr) + + def test_malformed_ledger_row_exits_1_naming_the_line(self): + led = LEDGER_HEAD + "| candidate | line 22 | resolved |\n" + proc = run_certify(CERT_INPUT, led) + self.assertEqual(proc.returncode, 1, proc.stdout) + self.assertIn("ledger row at line %d" % lineno(led, "| candidate |"), proc.stderr) + self.assertIn("expected 4", proc.stderr) + + def test_unknown_ledger_kind_exits_1(self): + led = FULL_LEDGER + "| mystery | line 22 | resolved | no such kind |\n" + proc = run_certify(CERT_INPUT, led) + self.assertEqual(proc.returncode, 1, proc.stdout) + self.assertIn("unknown kind 'mystery'", proc.stderr) + + def test_ledger_row_for_an_unknown_finding_exits_1(self): + led = FULL_LEDGER + "| boundary | Ghost | explained | not a component |\n" + proc = run_certify(CERT_INPUT, led) + self.assertEqual(proc.returncode, 1, proc.stdout) + self.assertIn("Ghost", proc.stderr) + self.assertIn("matches no boundary finding", proc.stderr) + + def test_span_row_without_source_exits_1_naming_the_flag(self): + led = FULL_LEDGER + "| span | L2-6 | read | narrative prose |\n" + proc = run_certify(CERT_INPUT, led) + self.assertEqual(proc.returncode, 1, proc.stdout) + self.assertIn("L2-6", proc.stderr) + self.assertIn("--source", proc.stderr) + + def test_duplicate_ledger_row_exits_1_naming_both_lines(self): + led = FULL_LEDGER + "| boundary | Scorer | explained | twice |\n" + proc = run_certify(CERT_INPUT, led) + self.assertEqual(proc.returncode, 1, proc.stdout) + self.assertIn("both disposition boundary 'Scorer'", proc.stderr) + self.assertIn("lines %d and %d" % (lineno(led, "manual trigger"), + lineno(led, "twice")), proc.stderr) + + def test_header_only_ledger_lists_every_finding(self): + proc = run_certify(CERT_INPUT, LEDGER_HEAD) + self.assertEqual(proc.returncode, 3, proc.stdout) + self.assertIn("- gate: refused", proc.stdout) + self.assertIn("missing-component candidate line 22 (? -> Scorer) is unresolved", + proc.stdout) + self.assertIn("interface gap line 23 (Ingest -> Analyst, missing Format, Trigger)" + " has no disposition", proc.stdout) + self.assertIn("boundary finding Ingest (nothing feeds it) is unexplained", + proc.stdout) + self.assertIn("boundary finding Scorer (isolated) is unexplained", proc.stdout) + self.assertIn("boundary finding Store (nothing consumes its output) is unexplained", + proc.stdout) + self.assertIn("- blockers: 5", proc.stdout) + self.assertIn("refused", proc.stderr) + + def test_finding_free_input_certifies_with_an_empty_ledger(self): + text = doc( + "| Ingest | Store | rows | csv | cron | me | S:L1 |\n" + "| Store | Ingest | acks | csv | cron | me | S:L1 |\n", + components=TWO_COMPONENTS, + ) + proc = run_certify(text, LEDGER_HEAD) + self.assertEqual(proc.returncode, 0, proc.stderr) + self.assertIn("- gate: certified", proc.stdout) + self.assertIn("- blockers: none", proc.stdout) + self.assertIn("- advisories: none", proc.stdout) + + def test_certify_documented_in_help(self): + proc = run(doc(""), "--help") + self.assertEqual(proc.returncode, 0, proc.stderr) + self.assertIn("--certify", proc.stdout) + self.assertIn("certification record", proc.stdout) + + if __name__ == "__main__": unittest.main(verbosity=2) From e45f68556d600a49195f8cc571bb08ce4249dea6 Mon Sep 17 00:00:00 2001 From: macblackstuff <148771651+macblackstuff@users.noreply.github.com> Date: Wed, 30 Sep 2026 02:04:13 +0200 Subject: [PATCH 2/7] feat(certify): identity-keyed ledger, drift detection, record file (U2) Ledger entries keyed by finding-scoped identity (interface rows producer->consumer:flows, pairs, boundary names, spans by source sha+range) with disposition/reviewer/date/fingerprint columns. Certification record emitted as .cert.md binding input and report sha256, gate result, effective flags (replayed exactly on certify); source sha under --source. Drift re-opens exactly the touched rows; duplicate input identities exit 1. 30 certification tests (115 total, green); v0.3.0 example output byte-identical. Co-Authored-By: Claude Claude-Session: sess_c59696de-2f36-4d7c-99cd-ad4102b8cc4e --- .../scripts/interface_matrix.py | 393 +++++++++--- .../scripts/test_interface_matrix.py | 595 ++++++++++++++++-- 2 files changed, 856 insertions(+), 132 deletions(-) diff --git a/skills/interface-matrix/scripts/interface_matrix.py b/skills/interface-matrix/scripts/interface_matrix.py index 7015f4d..266c770 100644 --- a/skills/interface-matrix/scripts/interface_matrix.py +++ b/skills/interface-matrix/scripts/interface_matrix.py @@ -6,20 +6,30 @@ Stdlib only. With --certify LEDGER the same input is certified instead of reported. The -ledger is a Markdown file holding one disposition table, columns -`Kind | Finding | Disposition | Reason`, whose rows disposition the report's -findings: a missing-component candidate by input line (`| candidate | line 21 | -... |`), an interface gap by input line, a boundary finding by component name, -an unstated pair as `A -> B`, an uncited source span as `L7-9`. Certification -exits 0 with a certification record when every finding is dispositioned, or 3 -naming every finding that is not; a gap dispositioned open (e.g. `open-parked`) -is an advisory in the record, not a failure. A malformed or duplicate ledger -row is bad input: exit 1, like a bad input row. +ledger is a Markdown file kept beside the input and written during review: one +disposition table, columns `Kind | Finding | Disposition | Reason | Reviewer | +Date | Fingerprint`, keying every finding by stable identity rather than input +line — interface-row findings (candidates, gaps) as `producer -> consumer: +flows`, unstated pairs as `A -> B`, boundary findings by component name, +uncited source spans as `L7-9@` — with a content fingerprint +(the record's blocker lines carry current fingerprints to paste) pinning what +was dispositioned. Certification exits 0 when every finding is dispositioned; +3 naming every blocker, entries whose finding drifted from the input as +`drifted:` and findings no entry covers as `unreviewed:`; and 1 on a bad +ledger row, a duplicate identity in the ledger or the input, or an invocation +whose flags do not replay the ones the ledger's `## Certification record` +section declares. Every run writes a standalone certification record beside +the ledger (.cert.md) binding the input, report and (when --source +ran) source sha256, the gate result and the effective flags; a pass also +stamps the same record into the ledger as its `## Certification record` +section. """ import argparse import difflib import graphlib +import hashlib +import json import re import sys @@ -28,8 +38,23 @@ RULE_COLUMNS = ("producer class", "consumer class", "disposition", "reason") ATTRS = ("Flows", "Format", "Trigger", "Owner") DISPOSITIONS = ("none", "review") -LEDGER_COLUMNS = ("kind", "finding", "disposition", "reason") +LEDGER_COLUMNS = ("kind", "finding", "disposition", "reason", "reviewer", + "date", "fingerprint") LEDGER_KINDS = ("candidate", "gap", "boundary", "pair", "span") +KIND_NOUNS = { + "candidate": "missing-component candidate", + "gap": "interface gap", + "boundary": "boundary finding", + "pair": "unstated pair", + "span": "uncited span", +} +KIND_VERBS = { + "candidate": "is unresolved", + "gap": "has no disposition", + "boundary": "is unexplained", + "pair": "has no disposition", + "span": "is unread", +} CITE = re.compile(r"(? consumer: flows`, pairs `A -> B`, + boundary findings name the component, spans `L7-9@`. The + identity is what drift matching keys on; a cell that cannot be an identity + at all is a bad ledger row, not drift. + """ + if kind in ("candidate", "gap"): + if " -> " not in finding or ": " not in finding.split(" -> ", 1)[1]: + die("ledger row at line %d: %s finding %r must read " + "'producer -> consumer: flows'" % (ln, kind, finding)) + producer, rest = finding.split(" -> ", 1) + consumer, flows = rest.split(": ", 1) + return (producer.strip(), consumer.strip(), flows.strip()) + if kind == "pair": + if " -> " not in finding: + die("ledger row at line %d: pair finding %r must read 'A -> B'" + % (ln, finding)) + a, b = finding.split(" -> ", 1) + return (a.strip(), b.strip()) + if kind == "boundary": + return (finding,) + if not has_source: + die("ledger row at line %d dispositions source span %r but certification " + "was invoked without --source" % (ln, finding)) + label, _, sha = finding.rpartition("@") + if not sha or not re.match(r"^L\d+(-\d+)?$", label): + die("ledger row at line %d: span finding %r must read " + "'L7-9@'" % (ln, finding)) + first, _, last = label[1:].partition("-") + return (sha, int(first), int(last or first)) + + +def fingerprint(*parts): + """A stable content fingerprint: the sha256 of the parts as canonical JSON.""" + return hashlib.sha256(json.dumps(list(parts)).encode("utf-8")).hexdigest() + + +def row_fingerprint(iface): + """An interface row's full content: its identity cells plus format, + trigger, owner and source, so any edit to a dispositioned row is drift.""" + return fingerprint(iface["producer"], iface["consumer"], + *[iface["attrs"][a] for a in ATTRS], iface["source"]) + + +def sha256_file(path): + with open(path, "rb") as fh: + return hashlib.sha256(fh.read()).hexdigest() + + +def sha256_text(text): + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def record_path(ledger_path): + """The standalone record file lives beside the ledger.""" + return (ledger_path[:-3] if ledger_path.endswith(".md") else ledger_path) + ".cert.md" + + +def stamp_ledger_record(ledger_path, record): + """Write the record into the ledger's certification-record section, + replacing the section a previous pass stamped.""" + section = "## Certification record" + record[record.index("\n"):] + with open(ledger_path, encoding="utf-8") as fh: + lines = fh.read().splitlines() + out, i, stamped = [], 0, False + while i < len(lines): + if record_heading(lines[i]): + i += 1 + while i < len(lines) and not lines[i].lstrip().startswith("#"): + i += 1 + out.extend(section.rstrip("\n").split("\n")) + stamped = True + if i < len(lines): + out.append("") + continue + out.append(lines[i]) + i += 1 + if not stamped: + if out and out[-1].strip(): + out.append("") + out.extend(section.rstrip("\n").split("\n")) + with open(ledger_path, "w", encoding="utf-8") as fh: + fh.write("\n".join(out) + "\n") -def certify(args, built, rules, spans): +def certify(args, built, rules, spans, rendered): """Judge the ledger against the findings the report derives from the input. - Returns (record, refused): the record names every blocker — a finding with - no ledger disposition — and every advisory, a gap dispositioned open, which - may legitimately stay open; refused means exit 3. + Returns (record, refused): the record names every blocker — an entry whose + finding drifted (gone from the input, or changed since disposition) and + every finding no entry covers — plus every advisory, a gap dispositioned + open, which may legitimately stay open; refused means exit 3. """ names, external, specified, gaps, nones, candidates, retired, class_of = built - rows = read_ledger(args.certify) - ledger = {} - for kind, finding, disposition, reason, ln in rows: - if (kind, finding) in ledger: + + # the ledger keys interface-row findings by producer, consumer and flows, + # so two active input rows sharing that identity would be one ambiguous + # finding: an input error under --certify (generation is unchanged) + first_line = {} + for iface in sorted((i for group in (specified, gaps, nones, candidates) + for i in group), key=lambda i: i["line"]): + key = (iface["producer"], iface["consumer"], iface["attrs"]["Flows"]) + if key in first_line: + die("input rows at lines %d and %d share one interface identity " + "(%s -> %s: %s); the review ledger cannot tell them apart" + % (first_line[key], iface["line"], key[0], key[1], key[2])) + first_line[key] = iface["line"] + + rows, declared = read_ledger(args.certify) + if declared is not None: + check_flags(args, declared) + + entries, seen_keys = [], {} + for kind, finding, disposition, reason, reviewer, date, fp, ln in rows: + key = parse_finding(kind, finding, ln, bool(args.source)) + if (kind, key) in seen_keys: die("ledger rows at lines %d and %d both disposition %s %r" - % (ledger[(kind, finding)][2], ln, kind, finding)) - ledger[(kind, finding)] = (disposition, reason, ln) + % (seen_keys[(kind, key)], ln, kind, finding)) + seen_keys[(kind, key)] = ln + entries.append((kind, key, finding, fp, disposition, reason, ln)) edge_of, stated = edges_of(specified, gaps, nones) internal, unfed, unconsumed, isolated = boundary( @@ -788,64 +984,81 @@ def certify(args, built, rules, spans): (isolated, "isolated")): for n in group: why[n] = label + src_sha = sha256_file(args.source) if args.source else None - known = { - "candidate": {"line %d" % c["line"] for c in candidates}, - "gap": {"line %d" % g["line"] for g in gaps}, - "boundary": set(why), - "pair": {"%s -> %s" % (a, b) for a, b in residue}, - "span": {span_label(s) for s in spans}, - } - for kind, finding, disposition, reason, ln in rows: - if kind == "span" and not args.source: - die("ledger row at line %d dispositions source span %r but certification " - "was invoked without --source" % (ln, finding)) - if finding not in known[kind]: - die("ledger row at line %d: %s %r matches no %s finding in the input" - % (ln, kind, finding, kind)) - - blockers = [] + # every finding the gate can refuse on, keyed by stable identity with the + # content fingerprint the ledger pins — input line numbers appear only in + # the notes, never in the identity, so unrelated edits do not re-open rows + found = {} # (kind, identity) -> (label, fingerprint, note) for c in sorted(candidates, key=lambda c: c["line"]): - if ("candidate", "line %d" % c["line"]) not in ledger: - blockers.append("missing-component candidate line %d (%s -> %s) is unresolved" - % (c["line"], c["producer"], c["consumer"])) + key = (c["producer"], c["consumer"], c["attrs"]["Flows"]) + found[("candidate", key)] = ("%s -> %s: %s" % key, row_fingerprint(c), + "input line %d" % c["line"]) for g in sorted(gaps, key=lambda g: g["line"]): - if ("gap", "line %d" % g["line"]) not in ledger: - blockers.append("interface gap line %d (%s -> %s, missing %s) has no disposition" - % (g["line"], g["producer"], g["consumer"], - ", ".join(g["missing"]))) + key = (g["producer"], g["consumer"], g["attrs"]["Flows"]) + found[("gap", key)] = ("%s -> %s: %s" % key, row_fingerprint(g), + "input line %d, missing %s" + % (g["line"], ", ".join(g["missing"]))) for n in names: - if n in why and ("boundary", n) not in ledger: - blockers.append("boundary finding %s (%s) is unexplained" % (n, why[n])) + if n in why: + found[("boundary", (n,))] = (n, fingerprint(n, why[n]), why[n]) for a, b in residue: - if ("pair", "%s -> %s" % (a, b)) not in ledger: - blockers.append("unstated pair %s -> %s has no disposition" % (a, b)) + found[("pair", (a, b))] = ("%s -> %s" % (a, b), fingerprint(a, b), + "no row states it") for s in spans: - if ("span", span_label(s)) not in ledger: - blockers.append("uncited span %s of %s is unread" % (span_label(s), args.source)) - - advisories = [] - for g in sorted(gaps, key=lambda g: g["line"]): - entry = ledger.get(("gap", "line %d" % g["line"])) - if entry and entry[0].lower().startswith("open"): - advisories.append("gap line %d (%s -> %s): %s" - % (g["line"], g["producer"], g["consumer"], - " — ".join(x for x in entry[:2] if x))) - + found[("span", (src_sha, s[0], s[1]))] = ( + "%s@%s" % (span_label(s), src_sha), fingerprint(src_sha, s[0], s[1]), + "of %s" % args.source) + + drifted, unreviewed, advisories, covered = [], [], [], set() + for kind, key, label, fp, disposition, reason, ln in entries: + f = found.get((kind, key)) + if f is None: + drifted.append("drifted: %s %s (ledger line %d) no longer matches any " + "%s finding in the input" % (KIND_NOUNS[kind], label, ln, kind)) + continue + # the identity still exists: the finding is covered by this entry even + # when its content moved, so it is drift, never also unreviewed + covered.add((kind, key)) + if fp != f[1]: + drifted.append("drifted: %s %s (ledger line %d) changed since " + "disposition; current fingerprint %s" + % (KIND_NOUNS[kind], label, ln, f[1])) + elif kind == "gap" and disposition.lower().startswith("open"): + advisories.append("gap %s (%s): %s" + % (label, f[2], " — ".join( + x for x in (disposition, reason) if x))) + for (kind, key), (label, fp, note) in found.items(): + if (kind, key) not in covered: + unreviewed.append("unreviewed: %s %s (%s) %s (fingerprint %s)" + % (KIND_NOUNS[kind], label, note, KIND_VERBS[kind], fp)) + + blockers = drifted + unreviewed if blockers: - sys.stderr.write("error: certification refused: %d blocker(s), named in the " - "record\n" % len(blockers)) - return certification_record(args, blockers, advisories), bool(blockers) - - -def certification_record(args, blockers, advisories): - """The certification record: U1 prints it, U2 writes it to a file with hashes.""" + sys.stderr.write("error: certification refused: %d blocker(s) (%d drifted, " + "%d unreviewed), named in the record\n" + % (len(blockers), len(drifted), len(unreviewed))) + return (certification_record(args, blockers, advisories, + sha256_file(args.input), sha256_text(rendered), + src_sha), + bool(blockers)) + + +def certification_record(args, blockers, advisories, input_sha, report_sha, src_sha): + """The certification record: written beside the ledger as a standalone + file, printed, and — after a pass — stamped into the ledger itself. It + binds the exact input, report and (when --source ran) source file by + sha256, the gate result and the flags the review ran under; a later + --certify must replay those flags exactly.""" out = ["# Interface matrix certification", ""] - out.append("- input: %s" % args.input) + out.append("- input: %s (sha256 %s)" % (args.input, input_sha)) out.append("- ledger: %s" % args.certify) out.append("- gate: %s" % ("refused" if blockers else "certified")) + out.append("- report: sha256 %s" % report_sha) out.append("- flags: --sample %d%s" % (args.sample, " --source %s" % args.source if args.source else "")) + if src_sha: + out.append("- source: %s (sha256 %s)" % (args.source, src_sha)) out.append("- blockers: %s" % (len(blockers) if blockers else "none")) out.extend(" - %s" % b for b in blockers) out.append("- advisories: %s" % (len(advisories) if advisories else "none")) @@ -857,10 +1070,12 @@ def main(argv=None): ap = argparse.ArgumentParser(description=__doc__) ap.add_argument("input", help="Markdown file with Components and Interfaces tables") ap.add_argument("--certify", metavar="LEDGER", - help="review ledger to certify the input against instead of printing " - "the report: exit 0 with a certification record, exit 3 naming " - "every undispositioned finding, exit 1 on a bad ledger row " - "(ledger format in the module docstring)") + help="review ledger to certify the input against instead of " + "printing the report: exit 0 with a certification record, " + "exit 3 naming every blocker (drifted or unreviewed), " + "exit 1 on a bad ledger row, a duplicate identity, or " + "flags that do not replay the recorded review (ledger " + "format in the module docstring)") ap.add_argument("--sample", type=nonneg, default=20, help="unstated pairs to print (0 = all)") ap.add_argument("--source", help="source file whose L citations to check for coverage") args = ap.parse_args(argv) @@ -881,11 +1096,19 @@ def main(argv=None): cov, spans = None, [] built = build(components, interfaces, rules) if args.certify: - # the gate certifies what the report shows: render it once so certification - # dies on the same invariant (exit 2), then judge the ledger against the - # same derivation — the record replaces the rendered report - report(*built, rules=rules, sample_n=args.sample, has_rules=has_rules) - record, refused = certify(args, built, rules, spans) + # the gate certifies what the report shows: render it once — coverage + # section included when --source ran — so certification dies on the same + # invariant (exit 2) and the record can bind the report's sha256; then + # judge the ledger against the same derivation. The record replaces the + # rendered report on stdout, lands beside the ledger as a standalone + # file, and a pass also stamps it into the ledger itself. + rendered = report(*built, rules=rules, sample_n=args.sample, cov=cov, + has_rules=has_rules) + record, refused = certify(args, built, rules, spans, rendered) + with open(record_path(args.certify), "w", encoding="utf-8") as fh: + fh.write(record) + if not refused: + stamp_ledger_record(args.certify, record) sys.stdout.write(record) return 3 if refused else 0 sys.stdout.write(report(*built, rules=rules, sample_n=args.sample, cov=cov, diff --git a/skills/interface-matrix/scripts/test_interface_matrix.py b/skills/interface-matrix/scripts/test_interface_matrix.py index 915abc0..7bd1e77 100644 --- a/skills/interface-matrix/scripts/test_interface_matrix.py +++ b/skills/interface-matrix/scripts/test_interface_matrix.py @@ -1,5 +1,7 @@ """Tests for interface_matrix.py. Run: python3 scripts/test_interface_matrix.py""" +import hashlib +import json import os import re import subprocess @@ -1010,8 +1012,8 @@ def test_report_survives_a_non_utf8_console(self): LEDGER_HEAD = ( "# Review ledger\n" "\n" - "| Kind | Finding | Disposition | Reason |\n" - "|---|---|---|---|\n" + "| Kind | Finding | Disposition | Reason | Reviewer | Date | Fingerprint |\n" + "|---|---|---|---|---|---|---|\n" ) @@ -1020,6 +1022,28 @@ def ledger(*rows): return LEDGER_HEAD + "".join(rows) +def row(kind, finding, disposition, reason="", reviewer="reviewer", + date="2026-09-30", fingerprint=""): + """One disposition row, cells in table-column order.""" + return "| %s | %s | %s | %s | %s | %s | %s |\n" % ( + kind, finding, disposition, reason, reviewer, date, fingerprint) + + +def fp(*parts): + """The fingerprint the script derives from a finding's content.""" + return hashlib.sha256(json.dumps(list(parts)).encode("utf-8")).hexdigest() + + +def iface_fp(producer, consumer, flows, format="", trigger="", owner="", source=""): + """An interface row's fingerprint: every cell of the row.""" + return fp(producer, consumer, flows, format, trigger, owner, source) + + +def record_of(ledger_path): + """The standalone record file the script writes beside a ledger.""" + return (ledger_path[:-3] if ledger_path.endswith(".md") else ledger_path) + ".cert.md" + + def run_certify(text, ledger_text, *args): """Certify `text` against `ledger_text` written to its own temp file.""" with tempfile.NamedTemporaryFile("w", suffix=".md", delete=False) as fh: @@ -1028,12 +1052,17 @@ def run_certify(text, ledger_text, *args): try: return run(text, "--certify", path, *args) finally: - os.unlink(path) + for junk in (path, record_of(path)): + try: + os.unlink(junk) + except FileNotFoundError: + pass def run_certify_source(text, ledger_text, source_text): """Certify with --source, as run_source() is to run().""" - with tempfile.NamedTemporaryFile("w", suffix=".txt", delete=False) as fh: + with tempfile.NamedTemporaryFile("w", suffix=".txt", delete=False, + encoding="utf-8", newline="") as fh: fh.write(source_text) src = fh.name try: @@ -1042,6 +1071,40 @@ def run_certify_source(text, ledger_text, source_text): os.unlink(src) +def open_dir(text, ledger_text, source_text=None): + """A throwaway dir holding inventory.md, review.md and (optionally) + source.txt; multi-step tests re-run, edit and inspect the files in place, + then rm_dir the lot. Files are written with LF newlines so the sha256s + the script binds are predictable on every platform.""" + d = tempfile.mkdtemp() + for name, content in (("inventory.md", text), ("review.md", ledger_text), + ("source.txt", source_text)): + if content is not None: + with open(os.path.join(d, name), "w", encoding="utf-8", + newline="\n") as fh: + fh.write(content) + return d + + +def run_in(d, *args): + """Certify inventory.md against review.md inside `d`, extra flags after.""" + return subprocess.run( + [sys.executable, SCRIPT, os.path.join(d, "inventory.md"), + "--certify", os.path.join(d, "review.md"), *args], + capture_output=True, text=True) + + +def cat(d, name): + with open(os.path.join(d, name), encoding="utf-8") as fh: + return fh.read() + + +def rm_dir(d): + for name in os.listdir(d): + os.unlink(os.path.join(d, name)) + os.rmdir(d) + + TWO_COMPONENTS = ( "## Components\n\n" "| Component | Kind | Notes |\n" @@ -1059,12 +1122,25 @@ def run_certify_source(text, ledger_text, source_text): "| * | * | none | every classed pair is settled |\n", ) +CAND_FP = iface_fp("?", "Scorer", "digest", "csv", "cron", "me") +GAP_FP = iface_fp("Ingest", "Analyst", "rows", "", "?", "me") +BOUNDARY_FP = { + "Ingest": fp("Ingest", "nothing feeds it"), + "Scorer": fp("Scorer", "isolated"), + "Store": fp("Store", "nothing consumes its output"), +} + FULL_LEDGER = ledger( - "| candidate | line 22 | resolved | producer is Ingest, row fixed upstream |\n" - "| gap | line 23 | open-parked | blocked on the vendor's format doc |\n" - "| boundary | Ingest | explained | the pipeline's entry point |\n" - "| boundary | Scorer | explained | runs on a manual trigger |\n" - "| boundary | Store | explained | the terminal sink |\n" + row("candidate", "? -> Scorer: digest", "resolved", + "producer is Ingest, row fixed upstream", fingerprint=CAND_FP), + row("gap", "Ingest -> Analyst: rows", "open-parked", + "blocked on the vendor's format doc", fingerprint=GAP_FP), + row("boundary", "Ingest", "explained", "the pipeline's entry point", + fingerprint=BOUNDARY_FP["Ingest"]), + row("boundary", "Scorer", "explained", "runs on a manual trigger", + fingerprint=BOUNDARY_FP["Scorer"]), + row("boundary", "Store", "explained", "the terminal sink", + fingerprint=BOUNDARY_FP["Store"]), ) @@ -1074,13 +1150,16 @@ def test_fully_reviewed_input_certifies(self): self.assertEqual(proc.returncode, 0, proc.stderr) self.assertIn("# Interface matrix certification", proc.stdout) self.assertIn("- input: ", proc.stdout) + self.assertIn("(sha256 ", proc.stdout) self.assertIn("- ledger: ", proc.stdout) self.assertIn("- gate: certified", proc.stdout) + self.assertIn("- report: sha256 ", proc.stdout) self.assertIn("- flags: --sample 20\n", proc.stdout) self.assertIn("- blockers: none", proc.stdout) self.assertIn("- advisories: 1", proc.stdout) - self.assertIn("gap line 23 (Ingest -> Analyst): open-parked — blocked on the " - "vendor's format doc", proc.stdout) + self.assertIn("gap Ingest -> Analyst: rows (input line 23, missing Format, " + "Trigger): open-parked — blocked on the vendor's format doc", + proc.stdout) def test_open_gap_is_an_advisory_and_a_filled_gap_is_not(self): # the gap is a finding either way; an open disposition is an advisory in @@ -1090,56 +1169,76 @@ def test_open_gap_is_an_advisory_and_a_filled_gap_is_not(self): "| Store | Ingest | acks | csv | cron | me | S:L1 |\n", components=TWO_COMPONENTS, ) - gap = "line %d" % lineno(text, "| Ingest | Store | rows | | cron") + gap = "Ingest -> Store: rows" + line = lineno(text, "| Ingest | Store | rows | | cron") + fingerprint = iface_fp("Ingest", "Store", "rows", "", "cron", "me", "S:L1") opened = run_certify(text, ledger( - "| gap | %s | open-parked | waiting on the vendor |\n" % gap)) + row("gap", gap, "open-parked", "waiting on the vendor", + fingerprint=fingerprint))) self.assertEqual(opened.returncode, 0, opened.stderr) self.assertIn("- advisories: 1", opened.stdout) - self.assertIn("gap %s (Ingest -> Store): open-parked — waiting on the vendor" - % gap, opened.stdout) + self.assertIn("gap %s (input line %d, missing Format): open-parked — " + "waiting on the vendor" % (gap, line), opened.stdout) filled = run_certify(text, ledger( - "| gap | %s | filled | the attrs live in the ADR |\n" % gap)) + row("gap", gap, "filled", "the attrs live in the ADR", + fingerprint=fingerprint))) self.assertEqual(filled.returncode, 0, filled.stderr) self.assertIn("- advisories: none", filled.stdout) def test_unresolved_candidate_blocks(self): proc = run_certify(CERT_INPUT, ledger( - "| gap | line 23 | open-parked | blocked |\n" - "| boundary | Ingest | explained | the entry point |\n" - "| boundary | Scorer | explained | the manual trigger |\n" - "| boundary | Store | explained | the terminal sink |\n", + row("gap", "Ingest -> Analyst: rows", "open-parked", "blocked", + fingerprint=GAP_FP), + row("boundary", "Ingest", "explained", "the entry point", + fingerprint=BOUNDARY_FP["Ingest"]), + row("boundary", "Scorer", "explained", "the manual trigger", + fingerprint=BOUNDARY_FP["Scorer"]), + row("boundary", "Store", "explained", "the terminal sink", + fingerprint=BOUNDARY_FP["Store"]), )) self.assertEqual(proc.returncode, 3, proc.stderr) self.assertIn("- gate: refused", proc.stdout) self.assertIn("- blockers: 1", proc.stdout) - self.assertIn("missing-component candidate line 22 (? -> Scorer) is unresolved", + self.assertIn("unreviewed: missing-component candidate ? -> Scorer: digest " + "(input line 22) is unresolved (fingerprint %s)" % CAND_FP, proc.stdout) def test_unstated_pair_without_disposition_blocks(self): text = doc("| Ingest | Store | rows | csv | cron | me | S:L1 |\n", components=TWO_COMPONENTS) proc = run_certify(text, ledger( - "| boundary | Ingest | explained | the entry point |\n" - "| boundary | Store | explained | the terminal sink |\n", + row("boundary", "Ingest", "explained", "the entry point", + fingerprint=fp("Ingest", "nothing feeds it")), + row("boundary", "Store", "explained", "the terminal sink", + fingerprint=fp("Store", "nothing consumes its output")), )) self.assertEqual(proc.returncode, 3, proc.stderr) self.assertIn("- blockers: 1", proc.stdout) - self.assertIn("unstated pair Store -> Ingest has no disposition", proc.stdout) + self.assertIn("unreviewed: unstated pair Store -> Ingest (no row states it) " + "has no disposition (fingerprint %s)" % fp("Store", "Ingest"), + proc.stdout) self.assertNotIn("boundary finding", proc.stdout) def test_uncited_span_blocks_under_source_until_dispositioned(self): text = doc("| Ingest | Store | rows | csv | cron | me | S:L1 |\n", components=TWO_COMPONENTS) + sha = hashlib.sha256(SOURCE.encode("utf-8")).hexdigest() + span_fp = fp(sha, 2, 6) led = ledger( - "| boundary | Ingest | explained | the entry point |\n" - "| boundary | Store | explained | the terminal sink |\n" - "| pair | Store -> Ingest | none | nothing flows back |\n", + row("boundary", "Ingest", "explained", "the entry point", + fingerprint=fp("Ingest", "nothing feeds it")), + row("boundary", "Store", "explained", "the terminal sink", + fingerprint=fp("Store", "nothing consumes its output")), + row("pair", "Store -> Ingest", "none", "nothing flows back", + fingerprint=fp("Store", "Ingest")), ) proc = run_certify_source(text, led, SOURCE) self.assertEqual(proc.returncode, 3, proc.stderr) - self.assertIn("uncited span L2-6", proc.stdout) - read = run_certify_source(text, led + "| span | L2-6 | read | narrative prose |\n", - SOURCE) + self.assertIn("unreviewed: uncited span L2-6@%s (of " % sha, proc.stdout) + self.assertIn(") is unread (fingerprint %s)" % span_fp, proc.stdout) + read = run_certify_source( + text, led + row("span", "L2-6@%s" % sha, "read", "narrative prose", + fingerprint=span_fp), SOURCE) self.assertEqual(read.returncode, 0, read.stderr) self.assertIn("- gate: certified", read.stdout) @@ -1150,52 +1249,75 @@ def test_bad_input_row_exits_1_in_certify_mode_too(self): self.assertIn("Nope", proc.stderr) def test_malformed_ledger_row_exits_1_naming_the_line(self): - led = LEDGER_HEAD + "| candidate | line 22 | resolved |\n" + led = LEDGER_HEAD + "| candidate | ? -> Scorer: digest | resolved |\n" proc = run_certify(CERT_INPUT, led) self.assertEqual(proc.returncode, 1, proc.stdout) - self.assertIn("ledger row at line %d" % lineno(led, "| candidate |"), proc.stderr) - self.assertIn("expected 4", proc.stderr) + self.assertIn("ledger row at line %d" % lineno(led, "| candidate |"), + proc.stderr) + self.assertIn("expected 7", proc.stderr) def test_unknown_ledger_kind_exits_1(self): - led = FULL_LEDGER + "| mystery | line 22 | resolved | no such kind |\n" + led = FULL_LEDGER + row("mystery", "line 22", "resolved", "no such kind") proc = run_certify(CERT_INPUT, led) self.assertEqual(proc.returncode, 1, proc.stdout) self.assertIn("unknown kind 'mystery'", proc.stderr) - def test_ledger_row_for_an_unknown_finding_exits_1(self): - led = FULL_LEDGER + "| boundary | Ghost | explained | not a component |\n" + def test_ledger_row_for_an_unknown_finding_is_drift_not_an_error(self): + # U1 exited 1 here; R5's drift semantics supersede that: an entry whose + # finding matches nothing in the current input is drift, so the review + # re-opens (exit 3) instead of calling the ledger row malformed + led = FULL_LEDGER + row("boundary", "Ghost", "explained", "not a component", + fingerprint=fp("Ghost", "anything")) proc = run_certify(CERT_INPUT, led) - self.assertEqual(proc.returncode, 1, proc.stdout) - self.assertIn("Ghost", proc.stderr) - self.assertIn("matches no boundary finding", proc.stderr) + self.assertEqual(proc.returncode, 3, proc.stdout) + self.assertIn("drifted: boundary finding Ghost (ledger line %d) no longer " + "matches any boundary finding in the input" + % lineno(led, "not a component"), proc.stdout) def test_span_row_without_source_exits_1_naming_the_flag(self): - led = FULL_LEDGER + "| span | L2-6 | read | narrative prose |\n" + led = FULL_LEDGER + row("span", "L2-6@%s" % ("0" * 64), "read", + "narrative prose", fingerprint="0" * 64) proc = run_certify(CERT_INPUT, led) self.assertEqual(proc.returncode, 1, proc.stdout) self.assertIn("L2-6", proc.stderr) self.assertIn("--source", proc.stderr) def test_duplicate_ledger_row_exits_1_naming_both_lines(self): - led = FULL_LEDGER + "| boundary | Scorer | explained | twice |\n" + led = FULL_LEDGER + row("boundary", "Scorer", "explained", "twice", + fingerprint=BOUNDARY_FP["Scorer"]) proc = run_certify(CERT_INPUT, led) self.assertEqual(proc.returncode, 1, proc.stdout) self.assertIn("both disposition boundary 'Scorer'", proc.stderr) self.assertIn("lines %d and %d" % (lineno(led, "manual trigger"), lineno(led, "twice")), proc.stderr) + def test_duplicate_ledger_identity_exits_1_across_spellings(self): + # identities are parsed, not string-matched: two spellings of one + # identity are still one finding dispositioned twice + led = FULL_LEDGER + row("gap", "Ingest -> Analyst: rows", "filled", + "spelled differently", fingerprint=GAP_FP) + proc = run_certify(CERT_INPUT, led) + self.assertEqual(proc.returncode, 1, proc.stdout) + self.assertIn("both disposition gap", proc.stderr) + def test_header_only_ledger_lists_every_finding(self): proc = run_certify(CERT_INPUT, LEDGER_HEAD) self.assertEqual(proc.returncode, 3, proc.stdout) self.assertIn("- gate: refused", proc.stdout) - self.assertIn("missing-component candidate line 22 (? -> Scorer) is unresolved", + self.assertIn("unreviewed: missing-component candidate ? -> Scorer: digest " + "(input line 22) is unresolved (fingerprint %s)" % CAND_FP, proc.stdout) - self.assertIn("interface gap line 23 (Ingest -> Analyst, missing Format, Trigger)" - " has no disposition", proc.stdout) - self.assertIn("boundary finding Ingest (nothing feeds it) is unexplained", + self.assertIn("unreviewed: interface gap Ingest -> Analyst: rows (input " + "line 23, missing Format, Trigger) has no disposition " + "(fingerprint %s)" % GAP_FP, proc.stdout) + self.assertIn("unreviewed: boundary finding Ingest (nothing feeds it) is " + "unexplained (fingerprint %s)" % BOUNDARY_FP["Ingest"], proc.stdout) - self.assertIn("boundary finding Scorer (isolated) is unexplained", proc.stdout) - self.assertIn("boundary finding Store (nothing consumes its output) is unexplained", + self.assertIn("unreviewed: boundary finding Scorer (isolated) is " + "unexplained (fingerprint %s)" % BOUNDARY_FP["Scorer"], + proc.stdout) + self.assertIn("unreviewed: boundary finding Store (nothing consumes its " + "output) is unexplained (fingerprint %s)" % BOUNDARY_FP["Store"], proc.stdout) self.assertIn("- blockers: 5", proc.stdout) self.assertIn("refused", proc.stderr) @@ -1219,5 +1341,384 @@ def test_certify_documented_in_help(self): self.assertIn("certification record", proc.stdout) +class TestCertifyIdentity(unittest.TestCase): + def test_duplicate_interface_identity_in_the_input_exits_1_naming_both(self): + # two interface rows sharing producer, consumer and flows are one + # ambiguous finding under the ledger's identity scheme: an input error + # for certification only + text = doc( + "| Ingest | Store | rows | | cron | me | |\n" + "| Ingest | Store | rows | csv | cron | me | |\n", + components=TWO_COMPONENTS, + ) + proc = run_certify(text, LEDGER_HEAD) + self.assertEqual(proc.returncode, 1, proc.stdout) + self.assertIn("input rows at lines %d and %d" + % (lineno(text, "| Ingest | Store | rows | |"), + lineno(text, "| Ingest | Store | rows | csv")), + proc.stderr) + self.assertIn("Ingest -> Store: rows", proc.stderr) + # generation is unchanged: both rows are a valid report + plain = run(text) + self.assertEqual(plain.returncode, 0, plain.stderr) + self.assertIn("interfaces with gaps: 1", plain.stdout) + self.assertIn("specified interfaces: 1", plain.stdout) + + +def drift_doc(rows): + """The drift fixture: gap rows plus a wildcard rule that settles every + unstated classed pair, so the findings are exactly the gaps and boundary.""" + return doc_rules("".join(rows), + "| * | * | none | every classed pair is settled |\n") + + +DRIFT_ROWS = [ + "| Ingest | Store | rows | | ? | me | |\n", + "| Ingest | Scorer | events | csv | cron | me | |\n", + "| Scorer | Store | scores | | ? | me | |\n", +] + +DRIFT_LEDGER = ledger( + row("gap", "Ingest -> Store: rows", "filled", "attrs tracked in ADR-7", + fingerprint=iface_fp("Ingest", "Store", "rows", "", "?", "me")), + row("gap", "Scorer -> Store: scores", "filled", "attrs tracked in ADR-8", + fingerprint=iface_fp("Scorer", "Store", "scores", "", "?", "me")), + row("boundary", "Ingest", "explained", "the entry point", + fingerprint=fp("Ingest", "nothing feeds it")), + row("boundary", "Store", "explained", "the terminal sink", + fingerprint=fp("Store", "nothing consumes its output")), +) + + +class TestCertifyDrift(unittest.TestCase): + """R5: edits after review re-open exactly the rows they touch.""" + + def test_appended_unrelated_row_still_certifies(self): + # a new specified row on an already-stated pair changes no finding: + # identity keys survive unrelated edits + text = drift_doc( + DRIFT_ROWS + ["| Ingest | Store | batches | csv | cron | me | S:L1 |\n"]) + proc = run_certify(text, DRIFT_LEDGER) + self.assertEqual(proc.returncode, 0, proc.stderr) + self.assertIn("- gate: certified", proc.stdout) + self.assertIn("- blockers: none", proc.stdout) + + def test_unrelated_component_edit_still_certifies(self): + # a Notes edit touches no finding content: fingerprints hold + components = COMPONENTS_CLASS.replace("| ranks events | AGT |", + "| ranks events harder | AGT |") + text = doc_rules("".join(DRIFT_ROWS), + "| * | * | none | every classed pair is settled |\n", + components=components) + proc = run_certify(text, DRIFT_LEDGER) + self.assertEqual(proc.returncode, 0, proc.stderr) + self.assertIn("- gate: certified", proc.stdout) + + def test_edited_flows_cell_reopens_exactly_that_row(self): + # the flows cell is part of the identity: the old entry drifts away and + # the edited row arrives as a new finding; the sibling row is untouched + text = drift_doc([ + "| Ingest | Store | batches | | ? | me | |\n", + DRIFT_ROWS[1], + DRIFT_ROWS[2], + ]) + proc = run_certify(text, DRIFT_LEDGER) + self.assertEqual(proc.returncode, 3, proc.stderr) + self.assertIn("(1 drifted, 1 unreviewed)", proc.stderr) + self.assertIn("drifted: interface gap Ingest -> Store: rows (ledger line " + "%d) no longer matches any gap finding in the input" + % lineno(DRIFT_LEDGER, "ADR-7"), proc.stdout) + self.assertIn("unreviewed: interface gap Ingest -> Store: batches (input " + "line 21, missing Format, Trigger) has no disposition " + "(fingerprint %s)" + % iface_fp("Ingest", "Store", "batches", "", "?", "me"), + proc.stdout) + self.assertNotIn("Scorer -> Store: scores", proc.stdout) + self.assertIn("- blockers: 2", proc.stdout) + + def test_edited_format_cell_drifts_by_fingerprint(self): + # the identity holds (producer, consumer and flows unchanged) but the + # row's content moved: the entry's fingerprint no longer matches, and + # the finding is not unreviewed — it was reviewed, then edited + text = drift_doc([ + "| Ingest | Store | rows | csv | ? | me | |\n", + DRIFT_ROWS[1], + DRIFT_ROWS[2], + ]) + proc = run_certify(text, DRIFT_LEDGER) + self.assertEqual(proc.returncode, 3, proc.stderr) + self.assertIn("(1 drifted, 0 unreviewed)", proc.stderr) + self.assertIn("drifted: interface gap Ingest -> Store: rows (ledger line " + "%d) changed since disposition; current fingerprint %s" + % (lineno(DRIFT_LEDGER, "ADR-7"), + iface_fp("Ingest", "Store", "rows", "csv", "?", "me")), + proc.stdout) + self.assertIn("- blockers: 1", proc.stdout) + + def test_deleted_reviewed_row_drifts(self): + # the row vanished: its entry names a finding the input no longer has + text = drift_doc([DRIFT_ROWS[1], DRIFT_ROWS[2]]) + proc = run_certify(text, DRIFT_LEDGER) + self.assertEqual(proc.returncode, 3, proc.stderr) + self.assertIn("drifted: interface gap Ingest -> Store: rows (ledger line " + "%d) no longer matches any gap finding in the input" + % lineno(DRIFT_LEDGER, "ADR-7"), proc.stdout) + self.assertIn("- blockers: 1", proc.stdout) + + def test_edited_source_file_reopens_the_span_review(self): + # a span's identity carries the source file's sha256: editing the file + # re-opens the span review even when the line numbers still line up + text = doc("| Ingest | Store | rows | csv | cron | me | S:L1 |\n", + components=TWO_COMPONENTS) + old_sha = hashlib.sha256(SOURCE.encode("utf-8")).hexdigest() + led = ledger( + row("boundary", "Ingest", "explained", "the entry point", + fingerprint=fp("Ingest", "nothing feeds it")), + row("boundary", "Store", "explained", "the terminal sink", + fingerprint=fp("Store", "nothing consumes its output")), + row("pair", "Store -> Ingest", "none", "nothing flows back", + fingerprint=fp("Store", "Ingest")), + row("span", "L2-6@%s" % old_sha, "read", "narrative prose", + fingerprint=fp(old_sha, 2, 6)), + ) + d = open_dir(text, led, SOURCE) + try: + first = run_in(d, "--source", os.path.join(d, "source.txt")) + self.assertEqual(first.returncode, 0, first.stderr) + edited = SOURCE + "eta line\n" + with open(os.path.join(d, "source.txt"), "w", encoding="utf-8", + newline="\n") as fh: + fh.write(edited) + second = run_in(d, "--source", os.path.join(d, "source.txt")) + self.assertEqual(second.returncode, 3, second.stderr) + new_sha = hashlib.sha256(edited.encode("utf-8")).hexdigest() + self.assertIn("drifted: uncited span L2-6@%s (ledger line %d) no longer " + "matches any span finding in the input" + % (old_sha, lineno(led, "narrative prose")), + second.stdout) + self.assertIn("unreviewed: uncited span L2-7@%s (of " % new_sha, + second.stdout) + self.assertIn(") is unread (fingerprint %s)" % fp(new_sha, 2, 7), + second.stdout) + self.assertIn("(1 drifted, 1 unreviewed)", second.stderr) + finally: + rm_dir(d) + + +# finding-free under any flags: both pairs stated, both components fed and +# consumed, and every source line cited +FLAG_INPUT = doc( + "| Ingest | Store | rows | csv | cron | me | S:L1-6 |\n" + "| Store | Ingest | acks | csv | cron | me | S:L1-6 |\n", + components=TWO_COMPONENTS, +) + + +class TestCertifyFlags(unittest.TestCase): + """KTD3: --certify replays the recorded review exactly.""" + + def test_certify_without_the_declared_source_flag_exits_1(self): + d = open_dir(FLAG_INPUT, LEDGER_HEAD, SOURCE) + try: + src = os.path.join(d, "source.txt") + first = run_in(d, "--source", src) + self.assertEqual(first.returncode, 0, first.stderr) + self.assertIn("- flags: --sample 20 --source %s" % src, first.stdout) + again = run_in(d) + self.assertEqual(again.returncode, 1, again.stdout) + self.assertIn("--source", again.stderr) + self.assertIn("span checks", again.stderr) + finally: + rm_dir(d) + + def test_certify_with_the_wrong_sample_exits_1_and_the_recorded_one_passes(self): + d = open_dir(FLAG_INPUT, LEDGER_HEAD) + try: + self.assertEqual(run_in(d, "--sample", "0").returncode, 0) + wrong = run_in(d) # the default is 20, the record pins 0 + self.assertEqual(wrong.returncode, 1, wrong.stdout) + self.assertIn("--sample", wrong.stderr) + self.assertIn("replay", wrong.stderr) + right = run_in(d, "--sample", "0") + self.assertEqual(right.returncode, 0, right.stderr) + finally: + rm_dir(d) + + def test_undeclared_source_flag_exits_1(self): + d = open_dir(FLAG_INPUT, LEDGER_HEAD) + try: + self.assertEqual(run_in(d).returncode, 0) + with open(os.path.join(d, "source.txt"), "w", encoding="utf-8", + newline="\n") as fh: + fh.write(SOURCE) + extra = run_in(d, "--source", os.path.join(d, "source.txt")) + self.assertEqual(extra.returncode, 1, extra.stdout) + self.assertIn("no --source", extra.stderr) + finally: + rm_dir(d) + + def test_hand_written_record_section_pins_flags_before_any_pass(self): + led = LEDGER_HEAD + "\n## Certification record\n\n- flags: --sample 0\n" + d = open_dir(FLAG_INPUT, led) + try: + self.assertEqual(run_in(d).returncode, 1) + self.assertEqual(run_in(d, "--sample", "0").returncode, 0) + finally: + rm_dir(d) + + +class TestCertifyRecord(unittest.TestCase): + def test_success_writes_the_record_file_with_every_bound_field(self): + text = doc("| Ingest | Store | rows | csv | cron | me | S:L1 |\n", + components=TWO_COMPONENTS) + sha = hashlib.sha256(SOURCE.encode("utf-8")).hexdigest() + led = ledger( + row("boundary", "Ingest", "explained", "the entry point", + fingerprint=fp("Ingest", "nothing feeds it")), + row("boundary", "Store", "explained", "the terminal sink", + fingerprint=fp("Store", "nothing consumes its output")), + row("pair", "Store -> Ingest", "none", "nothing flows back", + fingerprint=fp("Store", "Ingest")), + row("span", "L2-6@%s" % sha, "read", "narrative prose", + fingerprint=fp(sha, 2, 6)), + ) + d = open_dir(text, led, SOURCE) + try: + src = os.path.join(d, "source.txt") + proc = run_in(d, "--source", src) + self.assertEqual(proc.returncode, 0, proc.stderr) + record = cat(d, "review.cert.md") + self.assertEqual(record, proc.stdout) # the file is what was printed + inp = os.path.join(d, "inventory.md") + with open(inp, "rb") as fh: + input_sha = hashlib.sha256(fh.read()).hexdigest() + with open(src, "rb") as fh: + src_sha = hashlib.sha256(fh.read()).hexdigest() + self.assertIn("- input: %s (sha256 %s)" % (inp, input_sha), record) + self.assertIn("- ledger: %s" % os.path.join(d, "review.md"), record) + self.assertIn("- gate: certified", record) + self.assertIn("- flags: --sample 20 --source %s" % src, record) + self.assertIn("- source: %s (sha256 %s)" % (src, src_sha), record) + self.assertIn("- blockers: none", record) + # the report hash binds what a plain run of the same flags prints + gen = run(text, "--source", src) + self.assertEqual(gen.returncode, 0, gen.stderr) + self.assertIn("- report: sha256 %s" + % hashlib.sha256(gen.stdout.encode("utf-8")).hexdigest(), + record) + finally: + rm_dir(d) + + def test_record_is_reproducible_and_the_ledger_section_stamped_once(self): + d = open_dir(FLAG_INPUT, LEDGER_HEAD) + try: + first = run_in(d) + self.assertEqual(first.returncode, 0, first.stderr) + one = cat(d, "review.cert.md") + stamped = cat(d, "review.md") + self.assertEqual(stamped.count("## Certification record"), 1) + self.assertIn("- gate: certified", stamped) + self.assertIn("- flags: --sample 20", stamped) + second = run_in(d) + self.assertEqual(second.returncode, 0, second.stderr) + self.assertEqual(second.stdout, first.stdout) + self.assertEqual(cat(d, "review.cert.md"), one) + self.assertEqual(cat(d, "review.md"), stamped) # idempotent stamp + finally: + rm_dir(d) + + def test_refused_certification_also_writes_its_record_but_not_the_ledger(self): + d = open_dir(CERT_INPUT, LEDGER_HEAD) + try: + proc = run_in(d) + self.assertEqual(proc.returncode, 3, proc.stderr) + record = cat(d, "review.cert.md") + self.assertEqual(record, proc.stdout) + self.assertIn("- gate: refused", record) + self.assertIn("- blockers: 5", record) + self.assertIn(" - unreviewed: ", record) + self.assertNotIn("## Certification record", cat(d, "review.md")) + finally: + rm_dir(d) + + +UNREVIEWED_LINE = re.compile( + r"^ - unreviewed: (missing-component candidate|interface gap|boundary " + r"finding|unstated pair|uncited span) (.*?) \(.*\) .* \(fingerprint " + r"([0-9a-f]{64})\)$") + +NOUN_KIND = { + "missing-component candidate": "candidate", + "interface gap": "gap", + "boundary finding": "boundary", + "unstated pair": "pair", + "uncited span": "span", +} + + +def harvest(stdout): + """Ledger rows for every unreviewed finding a refusal names: the + reviewer's paste-from-the-record step, done mechanically.""" + return [ + row(NOUN_KIND[m.group(1)], m.group(2), "resolved", "reviewed in cycle", + fingerprint=m.group(3)) + for m in map(UNREVIEWED_LINE.match, stdout.splitlines()) + if m + ] + + +class TestCertifyCycle(unittest.TestCase): + def test_generate_review_certify_drift_fix_recertify(self): + d = open_dir(drift_doc(DRIFT_ROWS), LEDGER_HEAD) + try: + # 1. generate: the report prints clean + gen = run(drift_doc(DRIFT_ROWS)) + self.assertEqual(gen.returncode, 0, gen.stderr) + self.assertIn("## 1. Summary", gen.stdout) + # 2. review: the refusal is the worksheet; harvest its findings + refuse = run_in(d) + self.assertEqual(refuse.returncode, 3, refuse.stderr) + entries = harvest(refuse.stdout) + self.assertEqual(len(entries), 4, entries) # two gaps, two boundary + with open(os.path.join(d, "review.md"), "w", encoding="utf-8", + newline="\n") as fh: + fh.write(ledger(*entries)) + # 3. certify: every finding dispositioned + ok = run_in(d) + self.assertEqual(ok.returncode, 0, ok.stderr) + self.assertIn("- gate: certified", ok.stdout) + # 4. edit a reviewed row's flows cell + with open(os.path.join(d, "inventory.md"), "w", encoding="utf-8", + newline="\n") as fh: + fh.write(drift_doc([ + "| Ingest | Store | batches | | ? | me | |\n", + DRIFT_ROWS[1], + DRIFT_ROWS[2], + ])) + # 5. certify: exactly that row re-opened, drifted and unreviewed + drift = run_in(d) + self.assertEqual(drift.returncode, 3, drift.stderr) + self.assertIn("(1 drifted, 1 unreviewed)", drift.stderr) + self.assertIn("drifted: interface gap Ingest -> Store: rows", + drift.stdout) + fixed = harvest(drift.stdout) + self.assertEqual(len(fixed), 1, fixed) + self.assertIn("Ingest -> Store: batches", fixed[0]) + # 6. fix the ledger: supersede the stale entry with the fresh one + stale = [ln for ln in cat(d, "review.md").splitlines() + if "Ingest -> Store: rows" in ln] + self.assertEqual(len(stale), 1, stale) + fixed_ledger = cat(d, "review.md").replace(stale[0] + "\n", fixed[0]) + with open(os.path.join(d, "review.md"), "w", encoding="utf-8", + newline="\n") as fh: + fh.write(fixed_ledger) + # 7. certify again: the review is whole once more + final = run_in(d) + self.assertEqual(final.returncode, 0, final.stderr) + self.assertIn("- gate: certified", final.stdout) + finally: + rm_dir(d) + + if __name__ == "__main__": unittest.main(verbosity=2) From 32406f99b53280e1a298f41eac7fa3aecddba6da Mon Sep 17 00:00:00 2001 From: macblackstuff <148771651+macblackstuff@users.noreply.github.com> Date: Wed, 30 Sep 2026 02:13:07 +0200 Subject: [PATCH 3/7] docs(skill): certification workflow, four-file deliverable, model pins (U3) Steps 3-5 become the review loop: review = writing the identity-keyed ledger (refusal record doubles as worksheet), gaps may be parked open, step 5 ends with --certify exit 0. Deliverable = report, record, input, ledger together; uncertified report is a draft. Anti-delegation rule becomes separation of duties (independent review by default, model reviewer only via explicit pin; arXiv 2312.04134 kept as rationale). Optional experimental model-pin keys documented under metadata. Co-Authored-By: Claude Claude-Session: sess_c59696de-2f36-4d7c-99cd-ad4102b8cc4e --- skills/interface-matrix/SKILL.md | 121 ++++++++++++++++++++++++++++--- 1 file changed, 109 insertions(+), 12 deletions(-) diff --git a/skills/interface-matrix/SKILL.md b/skills/interface-matrix/SKILL.md index 8f2f8f1..214e1da 100644 --- a/skills/interface-matrix/SKILL.md +++ b/skills/interface-matrix/SKILL.md @@ -6,6 +6,9 @@ compatibility: "Requires Python 3.9 or newer (Windows: py -3). Standard library metadata: author: macblackstuff version: 0.3.0 + # Optional model pins — experimental until adapters exist; see "Model pins": + # decision, thinker, reviewer, judge, each a model-name string, e.g. + # reviewer: "a review model you independently trust" --- # interface-matrix @@ -134,23 +137,78 @@ without `--source`. sample is drawn deterministically: round-robin across the producer rows that have unstated pairs, and spread evenly along each row so the picks sweep across columns. An unknown or duplicate component name exits 1 naming the input line, as does an active -row naming a superseded component; valid input exits 0. -Stdlib only, no install step. +row naming a superseded component. + +Exit codes: 0 a valid input — report printed, or under `--certify` certification +passed and the record printed; 1 bad input — a bad row of the input or of a ledger, a +duplicate identity, or a `--certify` whose flags do not replay the review the ledger's +record declares — every error naming its line; 2 the partition invariant tripping +while the report renders, and argparse usage errors; 3 certification refused, every +blocker named in the record. Stdlib only, no install step. + +`--certify LEDGER` certifies the input against a review ledger instead of printing +the report: + +```bash +python3 scripts/interface_matrix.py INPUT.md --certify INPUT.ledger.md +``` + +The ledger is a Markdown file kept beside the input and written during review (§3): +one disposition table, `Kind | Finding | Disposition | Reason | Reviewer | Date | +Fingerprint`, one row per finding, keyed by identity rather than input line. `Kind` is +one of `candidate`, `gap`, `boundary`, `pair`, `span`; the `Finding` cell carries the +identity — candidates and gaps read `producer -> consumer: flows`, unstated pairs +`A -> B`, boundary findings the component's name, uncited spans `L7-9@`. +`Reviewer` is the reviewer of record (§3) and `Date` when it reviewed; `Fingerprint` +pins the content dispositioned — the record's blocker lines carry current fingerprints +to paste. Every cell but `Reason` is required. + +An entry covers the finding whose identity it names when its fingerprint matches; the +disposition text is the reviewer's judgment. Certification exits 0 when every finding +the report derives from the input is dispositioned and none has drifted; 3 names +every blocker — an entry whose finding is gone from the input or changed since +disposition is `drifted:`, a finding no entry covers is `unreviewed:` — and lists +every gap dispositioned with a `Disposition` starting `open` as an advisory, not a +blocker. Under `--certify`, two active input rows sharing one +`producer -> consumer: flows` identity also exit 1 (the ledger cannot tell them +apart), as do the ledger's own bad rows: wrong width, unknown kind, a missing required +cell, a duplicate identity, a second disposition table. Every run writes a record +beside the ledger, `.cert.md`, and prints it instead of the report: the input, +report and (when `--source` ran) source file bound by sha256, the gate result, every +blocker and advisory, and the effective flags, which a later certification must replay +exactly. A pass also stamps the same record into the ledger as its +`## Certification record` section, replacing the section a previous pass stamped. Self-check: `python3 scripts/test_interface_matrix.py`. Operating it — health checks, every error message and its fix, rollback and escalation: `references/RUNBOOK.md`. -## 3. Human review is mandatory +## 3. Independent review is mandatory + +Not optional, and not a second pass by whatever drafted the input. An LLM asked to +generate a design structure matrix reproduced **357 of 462 entries — 77.3%** of a +published matrix ([arXiv 2312.04134](https://arxiv.org/abs/2312.04134)): roughly one +cell in four wrong or missing, and **false negatives dominate** — the interface that +was never written down is the one that hurts. The sparse form hides exactly that +error. That is the case for independent review, so review runs under separation of +duties: the reviewer of record — the `Reviewer` the ledger names — must be someone +other than whatever drafted the input. The default is an independent human reviewer; +a model may hold the role only when the user explicitly pinned one ("Model pins"), +and the ledger's `Reviewer` column records what actually reviewed either way. + +Review is writing the ledger. Start it as nothing but the header: -Not optional and not delegable to another model pass. An LLM asked to generate a -design structure matrix reproduced **357 of 462 entries — 77.3%** of a published -matrix ([arXiv 2312.04134](https://arxiv.org/abs/2312.04134)): roughly one cell in -four wrong or missing, and **false negatives dominate** — the interface that was never -written down is the one that hurts. The sparse form hides exactly that error. +```markdown +| Kind | Finding | Disposition | Reason | Reviewer | Date | Fingerprint | +|---|---|---|---|---|---|---| +``` -So review, cell by cell: +Certify once — with `--source` when the input cites one, or the uncited spans never +enter review — and every finding comes back an `unreviewed:` blocker, named by +identity and carrying its current fingerprint: the refusal record doubles as the +review worksheet. A passing run's flags become the ones every later certification +must replay. Work it cell by cell: 1. Every listed row: are the four attributes right, and does the source support them? Then, per row: can the named producer actually produce this flow, and can the named @@ -160,11 +218,19 @@ So review, cell by cell: the real endpoint is a component nobody declared yet). 2. The rules (section 9): is each `none` rule true of every pair it matched? A dead rule is wrong or premature; a rule that also matches an explicit interface contradicts it. -3. The residue (section 7): for each pair, is "no interface" actually true? Raise - `--sample` until you have looked at a share you can defend, or `--sample 0` for all. +3. The residue (section 7): for each pair, is "no interface" actually true? + Certification needs a ledger row for every pair in the residue, not just the + sampled ones — raise `--sample` until you have looked at a share you can defend, + or `--sample 0` for all. 4. The uncited spans (section 10): read each one. A span nothing cites is either irrelevant to the system or a component or interface nobody wrote down. +Each blocker ends one of two ways. Resolved: fix the input (§4), the finding leaves +the report — and any ledger row already written for it must go too, or certification +reports it as `drifted:`. Or dispositioned: a ledger row that leaves the finding in +place, covered, with the decision and its reason recorded. Rerun `--certify` after +each pass; it exits 0 only when every finding is resolved or dispositioned. + ## 4. Resolve the findings | Finding | Resolution | @@ -179,7 +245,10 @@ So review, cell by cell: Rerun until there are no missing-component candidates and no unexplained boundary findings. Gaps and loops may legitimately remain — candidates and silent boundary -findings may not. +findings may not. A gap you are not filling now is parked, not ignored: disposition +it in the ledger with a `Disposition` starting `open` — `open-parked` — and a reason +it stays open, and certification carries it as an advisory in the record, never a +blocker. ## 5. Record changes @@ -194,6 +263,34 @@ exits 1 — supersede or repoint those rows in the same pass. Section 1 reports A retired component may be re-declared under the same name in the addendum, and each name may have at most one active row. +Then certify, and ship everything together: + +```bash +python3 scripts/interface_matrix.py INPUT.md --certify INPUT.ledger.md +``` + +The matrix is not done until `--certify` exits 0 — a done-check that has not seen +exit 0 has not seen a finished matrix. The finished deliverable is four files shipped +together: the report, its certification record (`.cert.md`), the input, and +the ledger — enough for any consumer to re-run certification and check the record's +sha256 bindings against the files they were sent. Generate the report under the flags +the record declares, `--sample N` and `--source` as it names them, so its sha256 is +the one the record binds. A report without its certification record is a draft. + +## Model pins (optional, experimental) + +Four optional keys may live under `metadata:` in this file's frontmatter — +`decision`, `thinker`, `reviewer`, `judge` — each pinning that role to a model, as a +plain string value (`reviewer: "a review model you independently trust"`). They are +instructions to the agent executing the skill, not configuration: the Python script +reads no pins, only its flags. Experimental until adapters exist. Harness-specific +model settings (an agent's own `model` or `effort` fields) are non-portable and do +not belong here. Pins written into an installed copy are overwritten by a +`skills add` refresh, so persistent pinning means maintaining them in a fork or a +local override. A pin names an intended reviewer, still bound by §3's separation of +duties — distinct from whatever drafted the input; the ledger's `Reviewer` column +records what actually reviewed. + ## Reading the matrix Row feeds column. `X` specified, `g` gap, `-` explicit none, blank unstated, `S` self. From 7ac1e0d087749322e5ee239befa57a56050351a4 Mon Sep 17 00:00:00 2001 From: macblackstuff <148771651+macblackstuff@users.noreply.github.com> Date: Wed, 30 Sep 2026 02:21:00 +0200 Subject: [PATCH 4/7] docs: certification docs, certified example, release v0.4.0 (U4) README: certify usage, four-file deliverable, exit-code table (exit 3; exit 2 shared with argparse documented), test count 115. RUNBOOK: certify procedure, drift and flag-mismatch playbooks, updated escalation and health checks. CHANGELOG 0.4.0. New certified example ledger + record (exit 0, one parked-gap advisory). Version 0.4.0. Co-Authored-By: Claude Claude-Session: sess_c59696de-2f36-4d7c-99cd-ad4102b8cc4e --- CHANGELOG.md | 40 +++++++++++++++++ README.md | 38 +++++++++++++++- examples/example-ledger.cert.md | 10 +++++ examples/example-ledger.md | 38 ++++++++++++++++ skills/interface-matrix/SKILL.md | 2 +- skills/interface-matrix/references/RUNBOOK.md | 44 +++++++++++++++---- 6 files changed, 161 insertions(+), 11 deletions(-) create mode 100644 examples/example-ledger.cert.md create mode 100644 examples/example-ledger.md diff --git a/CHANGELOG.md b/CHANGELOG.md index 4b61d54..67adcf2 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,46 @@ (none) +## 0.4.0 — 2026-09-30 + +- Certification gate: `--certify LEDGER` judges a review ledger against the findings the + report derives from the input, instead of printing the report. Exit 0 — every finding + dispositioned, a certification record printed; exit 3 — refused, every blocker named in + the record. The ledger format and the gate's rules are SKILL.md §2–§3. +- The review ledger is identity-keyed, not line-keyed: one table + (`Kind | Finding | Disposition | Reason | Reviewer | Date | Fingerprint`), one row per + finding — candidates and gaps as `producer -> consumer: flows`, unstated pairs + `A -> B`, boundary findings by component name, uncited spans `L7-9@` — + each pinning the content it dispositioned by fingerprint, so an unrelated edit does not + re-open a row. +- Every certification run writes a record beside the ledger (`.cert.md`) binding + the input, the report and (when `--source` ran) the source file by sha256, plus the + effective flags; a passing run also stamps it into the ledger as its + `## Certification record` section, and a later run whose flags do not replay the + recorded ones exits 1 naming the flag. +- Drift detection: a ledger entry whose finding is gone from the input, or changed since + disposition, is a `drifted:` blocker — resolving a finding and dispositioning it are + both recorded, and the one can no longer masquerade as the other. +- Under `--certify`, two active input rows sharing one `producer -> consumer: flows` + identity exit 1: the ledger cannot tell them apart. +- SKILL.md workflow rewritten: review is writing the ledger (the first refusal record is + the worksheet), the finished deliverable is four files shipped together — report, + certification record, input, ledger — review runs under separation of duties (the + reviewer of record is someone other than whatever drafted the input), and a gap not + filled now is parked `open` and carried as an advisory, never a blocker. +- Optional, experimental model pins: `decision`, `thinker`, `reviewer` and `judge` keys + under `metadata:` pin a role to a model. They are instructions to the executing agent, + not configuration — the script reads no pins, only its flags. +- README and RUNBOOK document the certification flow: the exit codes (3 added; exit 2's + sharing with argparse usage errors was already true and is now written down), the + certify procedure, and the drift and flag-mismatch playbooks. +- The worked example is now certified: `examples/example-ledger.md` dispositions every + finding `examples/example.md` produces, and `examples/example-ledger.cert.md` is the + record its passing `--certify` run wrote. The example input and its report are + unchanged from 0.3.0. +- Standard-library additions: `hashlib` and `json` (fingerprints, record bindings). Still + no dependencies, no install step; the self-check now runs 115 tests. + ## 0.3.0 — 2026-09-29 - Windows is a supported platform: the script reconfigures stdout to UTF-8, so the report diff --git a/README.md b/README.md index 9155fc9..c74ab36 100644 --- a/README.md +++ b/README.md @@ -110,6 +110,11 @@ Three stated rows, and the report names one unowned digest producer, one interfa its format and trigger, a component nothing feeds, an output nothing consumes, and ten component pairs nobody has ruled in or out. +The example is also certified: [`examples/example-ledger.md`](examples/example-ledger.md) +is the review ledger that dispositions every finding it produces, and +[`examples/example-ledger.cert.md`](examples/example-ledger.cert.md) is the certification +record the passing `--certify` run wrote. + ## Install | Harness | Command | Notes | @@ -126,7 +131,7 @@ component pairs nobody has ruled in or out. Verify the install from inside the installed folder with [`scripts/test_interface_matrix.py`](skills/interface-matrix/scripts/test_interface_matrix.py): ```bash -python3 scripts/test_interface_matrix.py # Ran 84 tests ... OK +python3 scripts/test_interface_matrix.py # Ran 115 tests ... OK ``` On Windows the interpreter is `py -3` (`py -3 scripts/interface_matrix.py example.md`); @@ -136,7 +141,7 @@ every platform. ### Harnesses tested CI installs the skill with the [`skills` CLI](https://github.com/vercel-labs/skills) on every push -and pull request, once per agent in its own throwaway home, and runs the 84 tests from each +and pull request, once per agent in its own throwaway home, and runs the 115 tests from each installed copy. Every agent the CLI supports is covered — 79 at the time of writing (`skills` 1.7.0), of which 77 are installed and tested; the list is read from the CLI at run time. Two agents are excluded with reasons recorded in `.github/scripts/smoke-install.sh`: `eve` and @@ -167,6 +172,35 @@ On Windows use `py -3` in place of `python3`. `--sample N` sets how many unstated pairs are printed (default 20, `0` = all). `--source FILE` adds the coverage section over the document the inventory was read from. +### Certifying a reviewed matrix + +Findings are reviewed into a ledger — one table, +`Kind | Finding | Disposition | Reason | Reviewer | Date | Fingerprint`, one row per +finding, keyed by identity rather than input line. Start it as nothing but the header row +and certify once; every finding comes back an `unreviewed:` blocker carrying its current +fingerprint, so the refusal record doubles as the review worksheet: + +```bash +python3 scripts/interface_matrix.py INPUT.md --certify INPUT.ledger.md +``` + +Disposition each finding in the ledger — the fingerprints to paste are in the record — +and certify again. Exit 0 writes `.cert.md` beside the ledger: the certification +record, binding the input, the report and (under `--source`) the source file by sha256, +plus the flags the review ran under, which every later certification must replay exactly. +The finished deliverable is four files shipped together: the report, its certification +record, the input, and the ledger — enough for any consumer to re-run certification. The +ledger format, the review procedure and the optional experimental model pins (a model may +review only when one is explicitly pinned) are in +[`skills/interface-matrix/SKILL.md`](skills/interface-matrix/SKILL.md). + +| Exit | Meaning | +|---|---| +| 0 | Report written — or, under `--certify`, certification passed and the record printed. | +| 1 | Bad input row, of the input or of a ledger; a duplicate interface identity; or a `--certify` whose flags do not replay the recorded review. Every error names its line. | +| 2 | The partition invariant tripping while the report renders — and argparse usage errors, which have always shared it and are now documented. | +| 3 | Certification refused: every blocker (a drifted or unreviewed finding) is named in the record. | + ## How it works 1. Read the input file: the Components table, the Interfaces table and the optional Rules table; everything else is ignored. diff --git a/examples/example-ledger.cert.md b/examples/example-ledger.cert.md new file mode 100644 index 0000000..0bfd5ca --- /dev/null +++ b/examples/example-ledger.cert.md @@ -0,0 +1,10 @@ +# Interface matrix certification + +- input: ../../examples/example.md (sha256 d74e7e9eb77f183e25e1e80aef3d6c30eda65e5d01e648d20b67f491155a553b) +- ledger: ../../examples/example-ledger.md +- gate: certified +- report: sha256 1480d1f3872a8d603afa6ceaf17d868d4c225fe7a477dbd250b8312270cf224a +- flags: --sample 20 +- blockers: none +- advisories: 1 + - gap Store -> Scorer: event batches (input line 22, missing Format, Trigger): open-parked — format and trigger wait on the storage RFP, due before work packages are cut; gap register G-12 diff --git a/examples/example-ledger.md b/examples/example-ledger.md new file mode 100644 index 0000000..18597b8 --- /dev/null +++ b/examples/example-ledger.md @@ -0,0 +1,38 @@ +# Example review ledger for interface-matrix + +The review ledger for [`example.md`](example.md): one row per finding the report +derives from the input, keyed by identity rather than input line. It was started as +nothing but the header; the first `--certify` run refused with every finding named +`unreviewed:` and its current fingerprint, and the rows below disposition each one — +fingerprints copied from that refusal record, which is the intended move. Run from the +repository root: + + python3 skills/interface-matrix/scripts/interface_matrix.py examples/example.md --certify examples/example-ledger.md + +| Kind | Finding | Disposition | Reason | Reviewer | Date | Fingerprint | +|---|---|---|---|---|---|---| +| candidate | ? -> Analyst: weekly digest | accepted | the digest is written by the on-call engineer of the week, a person outside the boundary; no component to declare until the reporting pass names the real producer | J. Merrick | 2026-09-30 | 4e139b0f50474d7629fd7d55b8a86689520d171e5726eba1ff659ea72106ea6c | +| gap | Store -> Scorer: event batches | open-parked | format and trigger wait on the storage RFP, due before work packages are cut; gap register G-12 | J. Merrick | 2026-09-30 | 20c27832f848ac823ebf0c86955cfee63e322e7c6f89d44608158d600cb5a1be | +| boundary | Ingest | accepted | Ingest is the system's source: it reads the external event broker, which is outside the boundary | J. Merrick | 2026-09-30 | 7b2acd46b8fe5961f6e95a9f7a8c2a4d2bb2e847c76fd0b2188585051eafc68c | +| boundary | Scorer | accepted | scored events are read by the Analyst's ad-hoc queries at this stage; the digest interface will name the producer once the reporting pass lands | J. Merrick | 2026-09-30 | b07ad9ad99b47f37ce06a812f6e56d7b2c7cb966d3d130ecb7478c99bcd2418b | +| pair | Ingest -> Scorer | none | Scorer reads event batches from Store, never straight from Ingest | J. Merrick | 2026-09-30 | 6e54deada7063a257816a5093992b19828fd166f36db2f46fffc8bfc36129e10 | +| pair | Ingest -> Analyst | none | raw event rows never reach a human; the Analyst reads digests only | J. Merrick | 2026-09-30 | c37b55540d54122e0418eb7b66ce96402ae435efe482050e68c4a50dddcc0fca | +| pair | Store -> Ingest | none | the event store is write-only for Ingest; no read-back | J. Merrick | 2026-09-30 | 12b6a29a7004ff74da31cac8a1f73f48a4cddd4fbb0c720e72e654a98af04b02 | +| pair | Store -> Analyst | none | the Analyst reads the weekly digest, not the store directly | J. Merrick | 2026-09-30 | 949e4b23e7aa2c743995eb22b42f20691bd7919c57c567d0a2a4c6ebf7e32cae | +| pair | Scorer -> Ingest | none | scoring is downstream of ingest; nothing flows back | J. Merrick | 2026-09-30 | 20e66c9839d4f06130b5a4661c48289ffea19dc00c009b22f6235fc64143564c | +| pair | Scorer -> Store | none | scores are consumed by the Analyst's ad-hoc queries; nothing writes back to the store at this stage | J. Merrick | 2026-09-30 | 848530bc07bfa9cde490f6f6aeba4c506f3b026bea67f8c108829edfe810b686 | +| pair | Scorer -> Analyst | none | the digest is not produced by Scorer; its producer is the unresolved candidate above | J. Merrick | 2026-09-30 | e380fb91cf05a80a6b9d098c719976233f5f57e4d0504c47e9668132bbb8f661 | +| pair | Analyst -> Ingest | none | the Analyst is a read-only consumer; nothing flows into the pipeline | J. Merrick | 2026-09-30 | f1d6bd6de416fb04f3017c8d52d986f397b67d682888a9ee75a7761ca7888688 | +| pair | Analyst -> Store | none | read-only consumer, and external to the boundary | J. Merrick | 2026-09-30 | 0c3cc0de7a486c81adf21e5cf695d515b13be8386e6e75ee4458f4cfff02066f | +| pair | Analyst -> Scorer | none | read-only consumer, and external to the boundary | J. Merrick | 2026-09-30 | 13211bf01419e89a0a02da395a37ab66080baf52eee38caaffa4f127f4ef96fd | + +## Certification record + +- input: ../../examples/example.md (sha256 d74e7e9eb77f183e25e1e80aef3d6c30eda65e5d01e648d20b67f491155a553b) +- ledger: ../../examples/example-ledger.md +- gate: certified +- report: sha256 1480d1f3872a8d603afa6ceaf17d868d4c225fe7a477dbd250b8312270cf224a +- flags: --sample 20 +- blockers: none +- advisories: 1 + - gap Store -> Scorer: event batches (input line 22, missing Format, Trigger): open-parked — format and trigger wait on the storage RFP, due before work packages are cut; gap register G-12 diff --git a/skills/interface-matrix/SKILL.md b/skills/interface-matrix/SKILL.md index 214e1da..b6675f0 100644 --- a/skills/interface-matrix/SKILL.md +++ b/skills/interface-matrix/SKILL.md @@ -5,7 +5,7 @@ license: MIT compatibility: "Requires Python 3.9 or newer (Windows: py -3). Standard library only — no dependencies and no install step." metadata: author: macblackstuff - version: 0.3.0 + version: 0.4.0 # Optional model pins — experimental until adapters exist; see "Model pins": # decision, thinker, reviewer, judge, each a model-name string, e.g. # reviewer: "a review model you independently trust" diff --git a/skills/interface-matrix/references/RUNBOOK.md b/skills/interface-matrix/references/RUNBOOK.md index a6e1b3c..6e2cddc 100644 --- a/skills/interface-matrix/references/RUNBOOK.md +++ b/skills/interface-matrix/references/RUNBOOK.md @@ -13,13 +13,13 @@ skill's own directory. ```bash python3 scripts/test_interface_matrix.py ``` -Expected: `Ran 83 tests ... OK`, exit 0. Also run it under `python3 -O` — the partition +Expected: `Ran 115 tests ... OK`, exit 0. Also run it under `python3 -O` — the partition check must survive assertions being stripped. ```bash grep -E '^(import|from) ' scripts/interface_matrix.py ``` -Expected: only `argparse`, `difflib`, `graphlib`, `re`, `sys`. Any third-party import is a defect — the skill must stay dependency-free. +Expected: only `argparse`, `difflib`, `graphlib`, `hashlib`, `json`, `re`, `sys`. Any third-party import is a defect — the skill must stay dependency-free. ## Procedures @@ -43,7 +43,22 @@ Expected: only `argparse`, `difflib`, `graphlib`, `re`, `sys`. Any third-party i ``` A citation is any `L`, `L-` or `L7,11-12` in a cell of an active Components or Interfaces row. Not citations, but still range-checked: a Rules row's `Reason`, and any citation on a superseded row. Section 10 prints the source lines nothing cites as contiguous spans (blank lines ignored), the first 80 characters of each span, and the line/cited/uncited counts. Read every span: that is where an unmodelled component or interface hides. A reversed range such as `L9-7` exits 1 while parsing, with or without `--source`; a citation past the file's last line and `L0` are only detectable against a source file, so they exit 1 only under `--source`. -5. **Change the script.** Add or change a test in `scripts/test_interface_matrix.py` first and +5. **Certify a reviewed matrix.** Review writes a ledger beside the input — one table, + `Kind | Finding | Disposition | Reason | Reviewer | Date | Fingerprint`, started as + nothing but the header. Certify once: + ```bash + python3 scripts/interface_matrix.py INPUT.md --certify INPUT.ledger.md + ``` + Exit 3: every finding the report derives comes back an `unreviewed:` blocker carrying + its current fingerprint — the refusal record (written beside the ledger on every run) + doubles as the review worksheet. Disposition every finding it names, copying the + fingerprints from the record's blocker lines, then certify again: exit 0, a `certified` + record, and the record also stamped into the ledger as its `## Certification record` + section. Certify with `--source` when the input cites one and under the `--sample` the + review used; the finished deliverable is four files — report, certification record, + input, ledger (SKILL.md §3). + +6. **Change the script.** Add or change a test in `scripts/test_interface_matrix.py` first and watch it fail, then change `scripts/interface_matrix.py`, then rerun the health check. Keep the script standard-library only and Python 3.9-compatible. The `system-adoption-pipeline` skill vendors a byte-identical copy of this script and pins its sha256, so a change here is not live @@ -79,21 +94,34 @@ Expected: only `argparse`, `difflib`, `graphlib`, `re`, `sys`. Any third-party i | A cell splits into two, or a row is short | A literal pipe inside a cell was not escaped. A backslash escapes the next character and is dropped: `\|` is a literal pipe in the cell value (re-escaped in the report), and `\\` is a literal backslash that leaves the next `\|` a delimiter, so a cell ending in a Windows path needs `C:\\\| next`. Short rows are padded to the header width. | Escape literal pipes as `\|`, and double a trailing backslash. | | Report looks right but the matrix is mostly empty | Only a handful of pairs were declared; everything else is an unstated pair, not a "no". | This is a finding, not a fault. Declare `none` on the pairs that genuinely have no interface, and add the real interfaces. | | Everything lands in one giant feedback loop | Legitimate output for a densely coupled system. | Nothing to fix in the tool. A human decides what to assume to break the loop; the script deliberately does not tear. | +| `error: certification refused: N blocker(s) (... unreviewed)` (exit 3) | Findings the report derives from the input that no ledger row covers — a header-only ledger's first run, or a review not finished. | Open `.cert.md`: every blocker is named by identity, with the fingerprint to paste. Disposition each finding in the ledger, or fix the input so the finding leaves the report, and rerun. | +| `drifted:` blockers (exit 3) | The input moved under the review: the finding is gone from the input (`no longer matches any ... finding`), or its content changed since disposition (the record names the current fingerprint). | A finding fixed in the input leaves the report, and its ledger row must go too (SKILL.md §3). A finding still present but edited needs its row re-reviewed — new fingerprint, new date; input rows are superseded, never reworded (§5). | +| `error: the ledger's certification record declares --sample N but certification was invoked with --sample M` (exit 1; same wording for `--source`) | The record's flags pin the review, and `--certify` must replay them exactly — a run without `--source` cannot silently skip the span checks. | Rerun with the declared flags; the message names the flag and both values. To re-review under other flags, amend the ledger's certification-record section first, as the message says. | +| `error: input rows at lines N and M share one interface identity (P -> C: flows); the review ledger cannot tell them apart` (exit 1, under `--certify`) | Two active interface rows share one `producer -> consumer: flows` — the identity the ledger keys on. | Distinguish the flows, or supersede or merge one of the rows, then certify again. | +| `error: ledger row at line N ...` (exit 1) | A bad ledger row: wrong width, an unknown kind, a missing required cell, a duplicate identity — or a second disposition table in one ledger. | Fix the named row. The format — `Kind \| Finding \| Disposition \| Reason \| Reviewer \| Date \| Fingerprint`, kinds `candidate gap boundary pair span` — is SKILL.md §2. | ## Rollback and recovery -The skill holds no state and writes nothing outside the report you redirect to stdout, so rollback is a +The skill holds no state and writes nothing outside the report you redirect to stdout — +under `--certify`, also the record beside the ledger and the record section stamped into +the ledger itself — so rollback is a file revert in whatever repository carries the skill folder. A generated report is disposable: rerun the script against the input file. The input file is the artefact worth keeping, and a reviewed matrix is amended by a dated addendum rather than rewritten. ## Escalation -1. Bad input (exit 1): the author of the input file fixes it. No escalation. +1. Bad input (exit 1): the author of the input file fixes it. No escalation. A bad ledger + row or a duplicate identity is the same class — the ledger's author fixes it. 2. Script defect (exit 2, crash on valid input, wrong partition): open an issue with the input file attached, and fix it on a branch with a failing test first. -3. Method disputes — whether a loop is real, whether an unstated pair is truly `none`, which assumption +3. Certification refused (exit 3): not a fault — a review not finished. The reviewer of + record dispositions or resolves every blocker the record names; the matrix is not done + until `--certify` exits 0. +4. Method disputes — whether a loop is real, whether an unstated pair is truly `none`, which assumption breaks a coupled block — are human decisions and belong to the matrix's reviewer, not to the tool. An LLM-generated DSM reproduced only 357/462 entries of a published matrix - ([arXiv 2312.04134](https://arxiv.org/abs/2312.04134)); the human review pass is the control, and it - is not delegable. + ([arXiv 2312.04134](https://arxiv.org/abs/2312.04134)); review is the control, and it runs + under separation of duties — the reviewer of record must be someone other than whatever + drafted the input, with a model in that role only when the user explicitly pinned one + ("Model pins", SKILL.md). From 845c5143809d730f8156a92f98f51cad522c8aae Mon Sep 17 00:00:00 2001 From: macblackstuff <148771651+macblackstuff@users.noreply.github.com> Date: Wed, 30 Sep 2026 04:23:03 +0200 Subject: [PATCH 5/7] =?UTF-8?q?fix(review):=20apply=20validated=20review?= =?UTF-8?q?=20findings=20=E2=80=94=20close=20gate's=20silent-pass=20holes?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit P1: stamped record is now a verified replay anchor (input/source sha mismatch on re-run = drift blocker, R5); citing input certified without --source exits 1 (R3 spans); SKILL/docstring scope the certification claim to the five ledger kinds (loops, self-deps, rules audit are human-reviewed, not gated); dispositioned candidates and boundary findings list as record advisories so what ships is visible. P2/P3: record-section swallow dies; blank-flows identities round-trip ('?' placeholder); delimiter component names rejected under certify; five-file deliverable under --source documented; record-lifecycle docs corrected; CRLF-safe writers; ledger rows retire by deletion (no Status column); example regenerated from repo-root invocation (no source citations). Suite 115 → 125. Review run 20260930-030735-c0aa86ef: verdict Ready with fixes, 8/8 validator-confirmed findings addressed; #16 derive() deferred. Co-Authored-By: Claude Claude-Session: sess_c59696de-2f36-4d7c-99cd-ad4102b8cc4e --- CHANGELOG.md | 56 ++-- README.md | 16 +- examples/example-ledger.cert.md | 9 +- examples/example-ledger.md | 11 +- examples/example.md | 4 +- skills/interface-matrix/SKILL.md | 89 ++++--- skills/interface-matrix/references/RUNBOOK.md | 30 ++- .../scripts/interface_matrix.py | 233 ++++++++++++----- .../scripts/test_interface_matrix.py | 247 ++++++++++++++++-- 9 files changed, 532 insertions(+), 163 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 67adcf2..d8efb3b 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -8,41 +8,61 @@ - Certification gate: `--certify LEDGER` judges a review ledger against the findings the report derives from the input, instead of printing the report. Exit 0 — every finding - dispositioned, a certification record printed; exit 3 — refused, every blocker named in - the record. The ledger format and the gate's rules are SKILL.md §2–§3. + of the five ledger kinds (candidates, gaps, boundary findings, unstated pairs, uncited + spans) dispositioned, a certification record printed; exit 3 — refused, every blocker + named in the record. Feedback loops, self-dependencies and the class-rules audit stay + human-review findings the gate does not disposition. The ledger format and the gate's + rules are SKILL.md §2–§3. - The review ledger is identity-keyed, not line-keyed: one table (`Kind | Finding | Disposition | Reason | Reviewer | Date | Fingerprint`), one row per - finding — candidates and gaps as `producer -> consumer: flows`, unstated pairs + finding — candidates and gaps as `producer -> consumer: flows` (a blank Flows cell is + the placeholder `?`, which pastes back as the empty identity), unstated pairs `A -> B`, boundary findings by component name, uncited spans `L7-9@` — each pinning the content it dispositioned by fingerprint, so an unrelated edit does not - re-open a row. -- Every certification run writes a record beside the ledger (`.cert.md`) binding - the input, the report and (when `--source` ran) the source file by sha256, plus the - effective flags; a passing run also stamps it into the ledger as its - `## Certification record` section, and a later run whose flags do not replay the + re-open a row. The label is emitted by one helper beside the one parser, so the paste + contract cannot fork. +- A completed certification run (pass or refusal) writes a record beside the ledger + (`.cert.md`) binding the input, the report and (when `--source` ran) the + source file by sha256, plus the effective flags; an error (exit 1) writes nothing and + leaves the previous record in place. A passing run also stamps it into the ledger as + its `## Certification record` section — the last passing run and the replay anchor: + every later run re-derives the input and source sha256 it binds and refuses on + mismatch (`drifted: input changed since the last certified run`), so a post-review + edit of the input re-opens the whole review; a later run whose flags do not replay the recorded ones exits 1 naming the flag. - Drift detection: a ledger entry whose finding is gone from the input, or changed since - disposition, is a `drifted:` blocker — resolving a finding and dispositioning it are - both recorded, and the one can no longer masquerade as the other. + disposition, is a `drifted:` blocker. Dispositions are free text: a candidate or + boundary finding dispositioned rather than `resolved`, and a gap parked `open`, each + stay listed as an advisory in the certification record — what ships stays visible. +- An input that cites a source cannot be certified without `--source`: the first run is + the only window in which span review could be skipped, so the gate refuses it (exit 1) + rather than let a pass pin the hole into the record's flags. - Under `--certify`, two active input rows sharing one `producer -> consumer: flows` - identity exit 1: the ledger cannot tell them apart. + identity exit 1, as does a component name containing ` -> ` or `: ` (no Finding cell + can express it) and a disposition row placed inside the ledger's certification-record + section — content the record-section scan must not swallow. - SKILL.md workflow rewritten: review is writing the ledger (the first refusal record is the worksheet), the finished deliverable is four files shipped together — report, - certification record, input, ledger — review runs under separation of duties (the - reviewer of record is someone other than whatever drafted the input), and a gap not - filled now is parked `open` and carried as an advisory, never a blocker. + certification record, input, ledger — five under `--source` (the source file), with + the record's invocation-relative paths replayed verbatim; review runs under separation + of duties (the reviewer of record is someone other than whatever drafted the input), + and a gap not filled now is parked `open` and carried as an advisory, never a blocker. - Optional, experimental model pins: `decision`, `thinker`, `reviewer` and `judge` keys under `metadata:` pin a role to a model. They are instructions to the executing agent, not configuration — the script reads no pins, only its flags. - README and RUNBOOK document the certification flow: the exit codes (3 added; exit 2's sharing with argparse usage errors was already true and is now written down), the - certify procedure, and the drift and flag-mismatch playbooks. + certify procedure, and the drift, anchor, citing-input and flag-mismatch playbooks. - The worked example is now certified: `examples/example-ledger.md` dispositions every finding `examples/example.md` produces, and `examples/example-ledger.cert.md` is the - record its passing `--certify` run wrote. The example input and its report are - unchanged from 0.3.0. + record its passing `--certify` run wrote from the repository-root invocation the + ledger documents. The example input no longer cites `S:L42`/`S:L44` — a citing input + must certify with `--source`, and no source file ships — so its report is unchanged + from 0.3.0 (citations do not print in the report). +- The ledger and record writers pin LF newlines, so a Windows re-certification does not + rewrite the whole ledger as CRLF. - Standard-library additions: `hashlib` and `json` (fingerprints, record bindings). Still - no dependencies, no install step; the self-check now runs 115 tests. + no dependencies, no install step; the self-check now runs 125 tests. ## 0.3.0 — 2026-09-29 diff --git a/README.md b/README.md index c74ab36..ec8bcf8 100644 --- a/README.md +++ b/README.md @@ -55,8 +55,8 @@ report it produces is committed beside it as | Producer | Consumer | Flows | Format | Trigger | Owner | Source | Status | |---|---|---|---|---|---|---|---| -| Ingest | Store | raw event rows | ndjson file | nightly cron | platform | S:L42 | | -| Store | Scorer | event batches | ? | ? | platform | S:L44 | | +| Ingest | Store | raw event rows | ndjson file | nightly cron | platform | | | +| Store | Scorer | event batches | ? | ? | platform | | | | ? | Analyst | weekly digest | ? | ? | ? | | | ``` @@ -131,7 +131,7 @@ record the passing `--certify` run wrote. Verify the install from inside the installed folder with [`scripts/test_interface_matrix.py`](skills/interface-matrix/scripts/test_interface_matrix.py): ```bash -python3 scripts/test_interface_matrix.py # Ran 115 tests ... OK +python3 scripts/test_interface_matrix.py # Ran 125 tests ... OK ``` On Windows the interpreter is `py -3` (`py -3 scripts/interface_matrix.py example.md`); @@ -141,7 +141,7 @@ every platform. ### Harnesses tested CI installs the skill with the [`skills` CLI](https://github.com/vercel-labs/skills) on every push -and pull request, once per agent in its own throwaway home, and runs the 115 tests from each +and pull request, once per agent in its own throwaway home, and runs the 125 tests from each installed copy. Every agent the CLI supports is covered — 79 at the time of writing (`skills` 1.7.0), of which 77 are installed and tested; the list is read from the CLI at run time. Two agents are excluded with reasons recorded in `.github/scripts/smoke-install.sh`: `eve` and @@ -188,8 +188,12 @@ Disposition each finding in the ledger — the fingerprints to paste are in the and certify again. Exit 0 writes `.cert.md` beside the ledger: the certification record, binding the input, the report and (under `--source`) the source file by sha256, plus the flags the review ran under, which every later certification must replay exactly. -The finished deliverable is four files shipped together: the report, its certification -record, the input, and the ledger — enough for any consumer to re-run certification. The +A citing input must certify with `--source` — the gate refuses it otherwise — and the +record a pass stamps into the ledger anchors the input and source by sha256, so any +post-review edit re-opens the review. The finished deliverable is four files shipped +together: the report, its certification record, the input, and the ledger — five when +the review ran under `--source`, adding the source file. The record's paths are +invocation-relative and must be replayed verbatim. The ledger format, the review procedure and the optional experimental model pins (a model may review only when one is explicitly pinned) are in [`skills/interface-matrix/SKILL.md`](skills/interface-matrix/SKILL.md). diff --git a/examples/example-ledger.cert.md b/examples/example-ledger.cert.md index 0bfd5ca..5d02a03 100644 --- a/examples/example-ledger.cert.md +++ b/examples/example-ledger.cert.md @@ -1,10 +1,13 @@ # Interface matrix certification -- input: ../../examples/example.md (sha256 d74e7e9eb77f183e25e1e80aef3d6c30eda65e5d01e648d20b67f491155a553b) -- ledger: ../../examples/example-ledger.md +- input: examples/example.md (sha256 52ec2692280ee34a4c87124e7fe7117a1dbecb87b63dbf83562664b1daebd1a2) +- ledger: examples/example-ledger.md - gate: certified - report: sha256 1480d1f3872a8d603afa6ceaf17d868d4c225fe7a477dbd250b8312270cf224a - flags: --sample 20 - blockers: none -- advisories: 1 +- advisories: 4 + - candidate ? -> Analyst: weekly digest (input line 23): accepted — the digest is written by the on-call engineer of the week, a person outside the boundary; no component to declare until the reporting pass names the real producer - gap Store -> Scorer: event batches (input line 22, missing Format, Trigger): open-parked — format and trigger wait on the storage RFP, due before work packages are cut; gap register G-12 + - boundary Ingest (nothing feeds it): accepted — Ingest is the system's source: it reads the external event broker, which is outside the boundary + - boundary Scorer (nothing consumes its output): accepted — scored events are read by the Analyst's ad-hoc queries at this stage; the digest interface will name the producer once the reporting pass lands diff --git a/examples/example-ledger.md b/examples/example-ledger.md index 18597b8..6af19b3 100644 --- a/examples/example-ledger.md +++ b/examples/example-ledger.md @@ -12,7 +12,7 @@ repository root: | Kind | Finding | Disposition | Reason | Reviewer | Date | Fingerprint | |---|---|---|---|---|---|---| | candidate | ? -> Analyst: weekly digest | accepted | the digest is written by the on-call engineer of the week, a person outside the boundary; no component to declare until the reporting pass names the real producer | J. Merrick | 2026-09-30 | 4e139b0f50474d7629fd7d55b8a86689520d171e5726eba1ff659ea72106ea6c | -| gap | Store -> Scorer: event batches | open-parked | format and trigger wait on the storage RFP, due before work packages are cut; gap register G-12 | J. Merrick | 2026-09-30 | 20c27832f848ac823ebf0c86955cfee63e322e7c6f89d44608158d600cb5a1be | +| gap | Store -> Scorer: event batches | open-parked | format and trigger wait on the storage RFP, due before work packages are cut; gap register G-12 | J. Merrick | 2026-09-30 | efdae3966ddabbe36042c85f63e7235c49cc231accc8a6638ff20640cdd94a56 | | boundary | Ingest | accepted | Ingest is the system's source: it reads the external event broker, which is outside the boundary | J. Merrick | 2026-09-30 | 7b2acd46b8fe5961f6e95a9f7a8c2a4d2bb2e847c76fd0b2188585051eafc68c | | boundary | Scorer | accepted | scored events are read by the Analyst's ad-hoc queries at this stage; the digest interface will name the producer once the reporting pass lands | J. Merrick | 2026-09-30 | b07ad9ad99b47f37ce06a812f6e56d7b2c7cb966d3d130ecb7478c99bcd2418b | | pair | Ingest -> Scorer | none | Scorer reads event batches from Store, never straight from Ingest | J. Merrick | 2026-09-30 | 6e54deada7063a257816a5093992b19828fd166f36db2f46fffc8bfc36129e10 | @@ -28,11 +28,14 @@ repository root: ## Certification record -- input: ../../examples/example.md (sha256 d74e7e9eb77f183e25e1e80aef3d6c30eda65e5d01e648d20b67f491155a553b) -- ledger: ../../examples/example-ledger.md +- input: examples/example.md (sha256 52ec2692280ee34a4c87124e7fe7117a1dbecb87b63dbf83562664b1daebd1a2) +- ledger: examples/example-ledger.md - gate: certified - report: sha256 1480d1f3872a8d603afa6ceaf17d868d4c225fe7a477dbd250b8312270cf224a - flags: --sample 20 - blockers: none -- advisories: 1 +- advisories: 4 + - candidate ? -> Analyst: weekly digest (input line 23): accepted — the digest is written by the on-call engineer of the week, a person outside the boundary; no component to declare until the reporting pass names the real producer - gap Store -> Scorer: event batches (input line 22, missing Format, Trigger): open-parked — format and trigger wait on the storage RFP, due before work packages are cut; gap register G-12 + - boundary Ingest (nothing feeds it): accepted — Ingest is the system's source: it reads the external event broker, which is outside the boundary + - boundary Scorer (nothing consumes its output): accepted — scored events are read by the Analyst's ad-hoc queries at this stage; the digest interface will name the producer once the reporting pass lands diff --git a/examples/example.md b/examples/example.md index db55b9f..a1aac30 100644 --- a/examples/example.md +++ b/examples/example.md @@ -18,6 +18,6 @@ The input from the README's Example section, as a real file. Run from | Producer | Consumer | Flows | Format | Trigger | Owner | Source | Status | |---|---|---|---|---|---|---|---| -| Ingest | Store | raw event rows | ndjson file | nightly cron | platform | S:L42 | | -| Store | Scorer | event batches | ? | ? | platform | S:L44 | | +| Ingest | Store | raw event rows | ndjson file | nightly cron | platform | | | +| Store | Scorer | event batches | ? | ? | platform | | | | ? | Analyst | weekly digest | ? | ? | ? | | | diff --git a/skills/interface-matrix/SKILL.md b/skills/interface-matrix/SKILL.md index b6675f0..8906d32 100644 --- a/skills/interface-matrix/SKILL.md +++ b/skills/interface-matrix/SKILL.md @@ -157,7 +157,8 @@ The ledger is a Markdown file kept beside the input and written during review ( one disposition table, `Kind | Finding | Disposition | Reason | Reviewer | Date | Fingerprint`, one row per finding, keyed by identity rather than input line. `Kind` is one of `candidate`, `gap`, `boundary`, `pair`, `span`; the `Finding` cell carries the -identity — candidates and gaps read `producer -> consumer: flows`, unstated pairs +identity — candidates and gaps read `producer -> consumer: flows` (a blank Flows cell +reads `?`, and pastes back as the empty identity), unstated pairs `A -> B`, boundary findings the component's name, uncited spans `L7-9@`. `Reviewer` is the reviewer of record (§3) and `Date` when it reviewed; `Fingerprint` pins the content dispositioned — the record's blocker lines carry current fingerprints @@ -165,19 +166,31 @@ to paste. Every cell but `Reason` is required. An entry covers the finding whose identity it names when its fingerprint matches; the disposition text is the reviewer's judgment. Certification exits 0 when every finding -the report derives from the input is dispositioned and none has drifted; 3 names +of the five ledger kinds the report derives from the input — candidates, gaps, +boundary findings, unstated pairs and uncited spans — is dispositioned and none has +drifted; feedback loops, self-dependencies and class-rule audit findings are report +findings a human reviews (§4/§9) — the gate does not disposition them. Exit 3 names every blocker — an entry whose finding is gone from the input or changed since -disposition is `drifted:`, a finding no entry covers is `unreviewed:` — and lists -every gap dispositioned with a `Disposition` starting `open` as an advisory, not a -blocker. Under `--certify`, two active input rows sharing one -`producer -> consumer: flows` identity also exit 1 (the ledger cannot tell them -apart), as do the ledger's own bad rows: wrong width, unknown kind, a missing required -cell, a duplicate identity, a second disposition table. Every run writes a record -beside the ledger, `.cert.md`, and prints it instead of the report: the input, -report and (when `--source` ran) source file bound by sha256, the gate result, every -blocker and advisory, and the effective flags, which a later certification must replay -exactly. A pass also stamps the same record into the ledger as its -`## Certification record` section, replacing the section a previous pass stamped. +disposition is `drifted:`, a finding no entry covers is `unreviewed:` — and lists as +advisories, not blockers, every gap dispositioned with a `Disposition` starting `open` +and every candidate or boundary finding dispositioned with one not starting +`resolved`: what ships stays visible in the record. Under `--certify`, two active +input rows sharing one `producer -> consumer: flows` identity also exit 1 (the ledger +cannot tell them apart), as does a component name containing ` -> ` or `: ` (no +Finding cell can express it), an input that cites a source certified without +`--source` (the uncited spans would never enter review), and the ledger's own bad +rows: wrong width, unknown kind, a missing required cell, a duplicate identity, a +second disposition table, a disposition row placed inside the certification-record +section. A completed certification run — pass or refusal — writes a record beside the +ledger, `.cert.md`, and prints it instead of the report: the input, report and +(when `--source` ran) source file bound by sha256, the gate result, every blocker and +advisory, and the effective flags, which a later certification must replay exactly. +An error (exit 1) writes nothing and leaves the previous record in place. A pass also +stamps the same record into the ledger as its `## Certification record` section, +replacing the section a previous pass stamped: that section is the last passing run +and the replay anchor — a later run re-derives the input and source sha256 it binds, +and any mismatch is a `drifted:` blocker (`input changed since the last certified +run`), so any post-review edit of the input re-opens the whole review. Self-check: `python3 scripts/test_interface_matrix.py`. @@ -204,11 +217,12 @@ Review is writing the ledger. Start it as nothing but the header: |---|---|---|---|---|---|---| ``` -Certify once — with `--source` when the input cites one, or the uncited spans never -enter review — and every finding comes back an `unreviewed:` blocker, named by -identity and carrying its current fingerprint: the refusal record doubles as the -review worksheet. A passing run's flags become the ones every later certification -must replay. Work it cell by cell: +Certify once — with `--source` when the input cites one; the gate refuses a citing +input certified without it, so the uncited spans cannot be skipped — and every finding +of the five ledger kinds comes back an `unreviewed:` blocker, named by identity and +carrying its current fingerprint: the refusal record doubles as the review worksheet. +A passing run's flags become the ones every later certification must replay. Work it +cell by cell: 1. Every listed row: are the four attributes right, and does the source support them? Then, per row: can the named producer actually produce this flow, and can the named @@ -226,10 +240,15 @@ must replay. Work it cell by cell: irrelevant to the system or a component or interface nobody wrote down. Each blocker ends one of two ways. Resolved: fix the input (§4), the finding leaves -the report — and any ledger row already written for it must go too, or certification -reports it as `drifted:`. Or dispositioned: a ledger row that leaves the finding in -place, covered, with the decision and its reason recorded. Rerun `--certify` after -each pass; it exits 0 only when every finding is resolved or dispositioned. +the report — and any ledger row already written for it must go too (a ledger row is +retired by deleting it; the ledger has no Status column — §5's supersede rule governs +input rows), or certification reports it as `drifted:`. Or dispositioned: a ledger row +that leaves the finding in place, covered, with the decision and its reason recorded. +Rerun `--certify` after each pass; it exits 0 only when every finding of the five +ledger kinds is resolved or dispositioned — feedback loops, self-dependencies and +class-rule audit findings stay human-review findings (§4/§9) the gate does not +disposition. Editing the input after a passing run re-opens the whole review: the +ledger's stamped record section anchors the input by sha256. ## 4. Resolve the findings @@ -243,12 +262,14 @@ each pass; it exits 0 only when every finding is resolved or dispositioned. | Feedback loop | Keep it. A human decides what to assume to break it; the script does not tear. | | Self-dependency | Usually a retry or a state carry-over. Confirm it is intended. | -Rerun until there are no missing-component candidates and no unexplained boundary -findings. Gaps and loops may legitimately remain — candidates and silent boundary -findings may not. A gap you are not filling now is parked, not ignored: disposition -it in the ledger with a `Disposition` starting `open` — `open-parked` — and a reason -it stays open, and certification carries it as an advisory in the record, never a -blocker. +Each candidate and boundary finding ends one of two ways: resolved — fix the input +(§1), the row leaves the report — or dispositioned with its reason; a dispositioned +candidate or boundary finding stays visible in every report and is listed as an +advisory in the certification record until it is resolved, so a shipping candidate is +never silent. Gaps and loops may legitimately remain. A gap you are not filling now +is parked, not ignored: disposition it in the ledger with a `Disposition` starting +`open` — `open-parked` — and a reason it stays open, and certification carries it as +an advisory in the record, never a blocker. ## 5. Record changes @@ -272,10 +293,14 @@ python3 scripts/interface_matrix.py INPUT.md --certify INPUT.ledger.md The matrix is not done until `--certify` exits 0 — a done-check that has not seen exit 0 has not seen a finished matrix. The finished deliverable is four files shipped together: the report, its certification record (`.cert.md`), the input, and -the ledger — enough for any consumer to re-run certification and check the record's -sha256 bindings against the files they were sent. Generate the report under the flags -the record declares, `--sample N` and `--source` as it names them, so its sha256 is -the one the record binds. A report without its certification record is a draft. +the ledger — five when the review ran under `--source`, adding the source file, whose +sha256 the record binds and whose checks the pinned flags require — enough for any +consumer to re-run certification and check the record's sha256 bindings against the +files they were sent. The record's paths are invocation-relative — the input, ledger +and source paths exactly as the certified run named them — so a replay must use them +verbatim. Generate the report under the flags the record declares, `--sample N` and +`--source` as it names them, so its sha256 is the one the record binds. A report +without its certification record is a draft. ## Model pins (optional, experimental) diff --git a/skills/interface-matrix/references/RUNBOOK.md b/skills/interface-matrix/references/RUNBOOK.md index 6e2cddc..42f474b 100644 --- a/skills/interface-matrix/references/RUNBOOK.md +++ b/skills/interface-matrix/references/RUNBOOK.md @@ -13,7 +13,7 @@ skill's own directory. ```bash python3 scripts/test_interface_matrix.py ``` -Expected: `Ran 115 tests ... OK`, exit 0. Also run it under `python3 -O` — the partition +Expected: `Ran 125 tests ... OK`, exit 0. Also run it under `python3 -O` — the partition check must survive assertions being stripped. ```bash @@ -45,18 +45,24 @@ Expected: only `argparse`, `difflib`, `graphlib`, `hashlib`, `json`, `re`, `sys` 5. **Certify a reviewed matrix.** Review writes a ledger beside the input — one table, `Kind | Finding | Disposition | Reason | Reviewer | Date | Fingerprint`, started as - nothing but the header. Certify once: + nothing but the header. Certify once — with `--source` when the input cites one (the + gate refuses a citing input certified without it) and under the `--sample` the review + used: ```bash python3 scripts/interface_matrix.py INPUT.md --certify INPUT.ledger.md ``` - Exit 3: every finding the report derives comes back an `unreviewed:` blocker carrying - its current fingerprint — the refusal record (written beside the ledger on every run) - doubles as the review worksheet. Disposition every finding it names, copying the - fingerprints from the record's blocker lines, then certify again: exit 0, a `certified` - record, and the record also stamped into the ledger as its `## Certification record` - section. Certify with `--source` when the input cites one and under the `--sample` the - review used; the finished deliverable is four files — report, certification record, - input, ledger (SKILL.md §3). + Exit 3: every finding of the five ledger kinds comes back an `unreviewed:` blocker + carrying its current fingerprint — the refusal record doubles as the review worksheet. + Disposition every finding it names, copying the fingerprints from the record's blocker + lines, then certify again: exit 0, a `certified` record, and the record also stamped + into the ledger as its `## Certification record` section. A completed run (pass or + refusal) writes the record beside the ledger; an error (exit 1) writes nothing and + leaves the previous record in place. The section stamped inside the ledger is the + last passing run and the replay anchor — flags and input/source sha256 — while the + standalone record reflects the latest completed run; editing the input or source + after a pass re-opens the whole review. The finished deliverable is four files — + report, certification record, input, ledger — five under `--source`, whose paths the + record names as the run typed them (SKILL.md §3/§5). 6. **Change the script.** Add or change a test in `scripts/test_interface_matrix.py` first and watch it fail, then change `scripts/interface_matrix.py`, then rerun the health check. Keep the @@ -96,6 +102,10 @@ Expected: only `argparse`, `difflib`, `graphlib`, `hashlib`, `json`, `re`, `sys` | Everything lands in one giant feedback loop | Legitimate output for a densely coupled system. | Nothing to fix in the tool. A human decides what to assume to break the loop; the script deliberately does not tear. | | `error: certification refused: N blocker(s) (... unreviewed)` (exit 3) | Findings the report derives from the input that no ledger row covers — a header-only ledger's first run, or a review not finished. | Open `.cert.md`: every blocker is named by identity, with the fingerprint to paste. Disposition each finding in the ledger, or fix the input so the finding leaves the report, and rerun. | | `drifted:` blockers (exit 3) | The input moved under the review: the finding is gone from the input (`no longer matches any ... finding`), or its content changed since disposition (the record names the current fingerprint). | A finding fixed in the input leaves the report, and its ledger row must go too (SKILL.md §3). A finding still present but edited needs its row re-reviewed — new fingerprint, new date; input rows are superseded, never reworded (§5). | +| `drifted: input changed since the last certified run` (exit 3; same shape for `source`) | The ledger's stamped record section binds the sha256 of the input (or source file) of the last passing run, and the current file hashes differently — any edit re-opens the whole review, stated and explicit-`none` rows included, because those rows are never fingerprinted per-finding. | Re-review: the per-finding `drifted:` lines in the same record name what else moved; fix those, delete the stale `## Certification record` section, and certify again — a pass re-stamps a fresh anchor. | +| `error: the input cites N source line(s) ...; certify with --source FILE so the uncited spans are reviewed` (exit 1) | The input cites `L` source lines but `--certify` ran without `--source`: the first run is the only window in which span review can be skipped, and a pass would pin the hole into the record's flags. | Re-run with `--source FILE`, the file the citations were written against. | +| `error: ledger line N: a disposition row cannot live inside a certification record` (exit 1) | A disposition table or row was placed after the `## Certification record` heading, where the record-section scan would silently swallow it. | Move those rows into the ledger's disposition table, above the record section. | +| `error: component name 'A -> B' at line N: component names cannot contain ' -> ' or ': ' under certification` (exit 1) | The ledger's `Finding` identities are parsed out of those delimiters; a component name containing one cannot be pasted into a cell and parsed back — every retry would add false `drifted:` blockers. | Rename the component in the input and re-run. Generation is unaffected; only certification refuses the name. | | `error: the ledger's certification record declares --sample N but certification was invoked with --sample M` (exit 1; same wording for `--source`) | The record's flags pin the review, and `--certify` must replay them exactly — a run without `--source` cannot silently skip the span checks. | Rerun with the declared flags; the message names the flag and both values. To re-review under other flags, amend the ledger's certification-record section first, as the message says. | | `error: input rows at lines N and M share one interface identity (P -> C: flows); the review ledger cannot tell them apart` (exit 1, under `--certify`) | Two active interface rows share one `producer -> consumer: flows` — the identity the ledger keys on. | Distinguish the flows, or supersede or merge one of the rows, then certify again. | | `error: ledger row at line N ...` (exit 1) | A bad ledger row: wrong width, an unknown kind, a missing required cell, a duplicate identity — or a second disposition table in one ledger. | Fix the named row. The format — `Kind \| Finding \| Disposition \| Reason \| Reviewer \| Date \| Fingerprint`, kinds `candidate gap boundary pair span` — is SKILL.md §2. | diff --git a/skills/interface-matrix/scripts/interface_matrix.py b/skills/interface-matrix/scripts/interface_matrix.py index 266c770..07c6110 100644 --- a/skills/interface-matrix/scripts/interface_matrix.py +++ b/skills/interface-matrix/scripts/interface_matrix.py @@ -10,19 +10,26 @@ disposition table, columns `Kind | Finding | Disposition | Reason | Reviewer | Date | Fingerprint`, keying every finding by stable identity rather than input line — interface-row findings (candidates, gaps) as `producer -> consumer: -flows`, unstated pairs as `A -> B`, boundary findings by component name, +flows` (a blank Flows cell renders as `?` and pastes back as the empty +identity), unstated pairs as `A -> B`, boundary findings by component name, uncited source spans as `L7-9@` — with a content fingerprint (the record's blocker lines carry current fingerprints to paste) pinning what -was dispositioned. Certification exits 0 when every finding is dispositioned; -3 naming every blocker, entries whose finding drifted from the input as -`drifted:` and findings no entry covers as `unreviewed:`; and 1 on a bad -ledger row, a duplicate identity in the ledger or the input, or an invocation -whose flags do not replay the ones the ledger's `## Certification record` -section declares. Every run writes a standalone certification record beside -the ledger (.cert.md) binding the input, report and (when --source -ran) source sha256, the gate result and the effective flags; a pass also -stamps the same record into the ledger as its `## Certification record` -section. +was dispositioned. Certification exits 0 when every finding of the five ledger +kinds is dispositioned and none has drifted — feedback loops, self-dependencies +and the class-rules audit are report findings a human reviews; the gate does +not disposition them; 3 naming every blocker, entries whose finding drifted +from the input as `drifted:` and findings no entry covers as `unreviewed:`; +and 1 on a bad ledger row, a duplicate identity in the ledger or the input, a +citing input certified without --source, a component name containing the +identity delimiters ` -> ` or `: `, or an invocation whose flags do not replay +the ones the ledger's `## Certification record` section declares. A completed +certification run (pass or refusal) writes a standalone certification record +beside the ledger (.cert.md) binding the input, report and (when +--source ran) source sha256, the gate result and the effective flags; an +error (exit 1) writes nothing and leaves the previous record in place. A pass +also stamps the same record into the ledger as its `## Certification record` +section — the last passing run, whose flag and sha256 bindings every later +run re-derives and compares, so any input or source edit re-opens the review. """ import argparse @@ -492,7 +499,7 @@ def coverage(cites, path): out.append("| lines | first line |") out.append("|---|---|") for a, b in spans: - label = "L%d" % a if a == b else "L%d-%d" % (a, b) + label = span_label((a, b)) out.append("| %s | %s |" % (label, md(lines[a - 1].strip()[:80]))) else: out.append("Every non-blank source line is cited.") @@ -588,11 +595,16 @@ def hit(rule, a, b): return active_rules, residue, matched, settled, by_rule, overrides +def directed(edge_of): + """The stated edges as a sorted list, self-dependencies left out.""" + return sorted(k for k in edge_of if k[0] != k[1]) + + def report(names, external, specified, gaps, nones, candidates, retired, class_of, rules=(), sample_n=20, cov=None, has_rules=False): edge_of, stated = edges_of(specified, gaps, nones) - edges = sorted(k for k in edge_of if k[0] != k[1]) + edges = directed(edge_of) selfdeps = sorted({a for a, b in edge_of if a == b}) blocks = partition(names, edges) order = [n for block in blocks for n in block] @@ -756,16 +768,28 @@ def parse_declared_flags(value, ln): return (int(m.group(1)), m.group(2)) +def record_sha(value, ln, what): + """The sha256 a record section's `- input:`/`- source:` line binds.""" + m = re.search(r"\(sha256 ([^)]+)\)\s*$", value) + if not m: + die("certification record at line %d: the %s line must read " + "'- %s: PATH (sha256 HEX)'" % (ln, what, what)) + return m.group(1) + + def read_ledger(path): - """Ledger disposition rows and the flags its certification record pins. - - Returns (rows, declared): rows are (kind, finding, disposition, reason, - reviewer, date, fingerprint, line) from the ledger's one disposition table - — the one table naming Kind and Finding; declared is the (--sample N, - --source PATH) the ledger's `## Certification record` section declares, or - None when the ledger pins none. A row of the wrong width, an unknown kind, - or a row without a finding, disposition, reviewer, date and fingerprint is - bad input, exactly like a bad row of the input file. + """Ledger disposition rows, and what its certification record pins. + + Returns (rows, declared, bound): rows are (kind, finding, disposition, + reason, reviewer, date, fingerprint, line) from the ledger's one + disposition table — the one table naming Kind and Finding; declared is the + (--sample N, --source PATH) the ledger's `## Certification record` section + declares, or None when the ledger pins none; bound maps 'input' and + 'source' to the (sha256, line) the section binds, for a later run to + re-derive and compare. A row of the wrong width, an unknown kind, or a row + without a finding, disposition, reviewer, date and fingerprint is bad + input, exactly like a bad row of the input file — and so is a disposition + row placed inside the record section. """ with open(path, encoding="utf-8") as fh: lines = fh.read().splitlines() @@ -773,6 +797,7 @@ def read_ledger(path): seen = 0 rec_seen = 0 declared = None # (flags, line) once a `- flags:` line is read + bound = {} # 'input'/'source' -> (sha256, line) the record section binds i = 0 while i < len(lines): if record_heading(lines[i]): @@ -782,6 +807,9 @@ def read_ledger(path): rec_seen = i + 1 i += 1 while i < len(lines) and not lines[i].lstrip().startswith("#"): + if lines[i].lstrip().startswith("|"): + die("ledger line %d: a disposition row cannot live inside " + "a certification record" % (i + 1)) m = re.match(r"^- flags: (.+)$", lines[i]) if m: if declared is not None: @@ -789,6 +817,15 @@ def read_ledger(path): "%d and %d" % (declared[1], i + 1)) declared = (parse_declared_flags(m.group(1).strip(), i + 1), i + 1) + for what in ("input", "source"): + m = re.match(r"^- %s: (.+)$" % what, lines[i]) + if m: + if what in bound: + die("certification record binds the %s sha twice, " + "at lines %d and %d" + % (what, bound[what][1], i + 1)) + bound[what] = (record_sha(m.group(1), i + 1, what), + i + 1) i += 1 continue if not is_header(lines, i): @@ -816,14 +853,13 @@ def read_ledger(path): ("finding", "disposition", "reviewer", "date", "fingerprint")): die("ledger row at line %d needs a finding, a disposition, a " "reviewer, a date and a fingerprint" % (i + 1)) - rows.append(tuple(row[at[c]].strip() for c in - ("kind", "finding", "disposition", "reason", - "reviewer", "date", "fingerprint")) + (i + 1,)) + rows.append(tuple(row[at[c]].strip() for c in LEDGER_COLUMNS) + + (i + 1,)) i += 1 if not seen: die("no disposition table in %s (expected columns: %s)" % (path, ", ".join(LEDGER_COLUMNS))) - return rows, declared[0] if declared else None + return rows, declared[0] if declared else None, bound def check_flags(args, declared): @@ -850,6 +886,32 @@ def check_flags(args, declared): "review exactly" % (source, args.source)) +def flows_key(iface): + """The Flows cell as an identity component: a blank or `?` cell is one + empty identity (the label renders `?`), so the two spellings of a missing + flows cell cannot fork into findings no paste can tell apart.""" + cell = iface["attrs"]["Flows"] + return "" if cell in ("", "?") else cell + + +def finding_label(kind, key): + """The Finding cell the record prints and the reviewer pastes back. + + The one emitter beside the one parser (parse_finding): candidate and gap + identities read `producer -> consumer: flows` — an empty flows renders as + the `?` placeholder, because a stripped ledger cell cannot hold the + trailing space an empty label would end with — pairs read `A -> B`, + boundary findings name the component, spans `L7-9@`. + """ + if kind == "pair": + return "%s -> %s" % key + if kind == "span": + return "%s@%s" % (span_label((key[1], key[2])), key[0]) + if kind == "boundary": + return key[0] + return "%s -> %s: %s" % (key[0], key[1], key[2] or "?") + + def parse_finding(kind, finding, ln, has_source): """A ledger Finding cell back into its identity key, by kind. @@ -864,7 +926,9 @@ def parse_finding(kind, finding, ln, has_source): "'producer -> consumer: flows'" % (ln, kind, finding)) producer, rest = finding.split(" -> ", 1) consumer, flows = rest.split(": ", 1) - return (producer.strip(), consumer.strip(), flows.strip()) + flows = flows.strip() + return (producer.strip(), consumer.strip(), + "" if flows == "?" else flows) if kind == "pair": if " -> " not in finding: die("ledger row at line %d: pair finding %r must read 'A -> B'" @@ -914,6 +978,7 @@ def stamp_ledger_record(ledger_path, record): """Write the record into the ledger's certification-record section, replacing the section a previous pass stamped.""" section = "## Certification record" + record[record.index("\n"):] + section_lines = section.rstrip("\n").split("\n") with open(ledger_path, encoding="utf-8") as fh: lines = fh.read().splitlines() out, i, stamped = [], 0, False @@ -922,7 +987,7 @@ def stamp_ledger_record(ledger_path, record): i += 1 while i < len(lines) and not lines[i].lstrip().startswith("#"): i += 1 - out.extend(section.rstrip("\n").split("\n")) + out.extend(section_lines) stamped = True if i < len(lines): out.append("") @@ -932,20 +997,23 @@ def stamp_ledger_record(ledger_path, record): if not stamped: if out and out[-1].strip(): out.append("") - out.extend(section.rstrip("\n").split("\n")) - with open(ledger_path, "w", encoding="utf-8") as fh: + out.extend(section_lines) + with open(ledger_path, "w", encoding="utf-8", newline="\n") as fh: fh.write("\n".join(out) + "\n") -def certify(args, built, rules, spans, rendered): +def certify(args, built, rules, spans, rendered, input_sha, cites): """Judge the ledger against the findings the report derives from the input. Returns (record, refused): the record names every blocker — an entry whose - finding drifted (gone from the input, or changed since disposition) and - every finding no entry covers — plus every advisory, a gap dispositioned - open, which may legitimately stay open; refused means exit 3. + finding drifted (gone from the input, or changed since disposition), a + finding no entry covers, or an input or source file that changed since the + run the ledger's record section stamps — plus every advisory: a gap + dispositioned open, or a candidate or boundary finding dispositioned + rather than resolved, each of which may legitimately stay, visibly. + refused means exit 3. """ - names, external, specified, gaps, nones, candidates, retired, class_of = built + names, external, specified, gaps, nones, candidates, _retired, class_of = built # the ledger keys interface-row findings by producer, consumer and flows, # so two active input rows sharing that identity would be one ambiguous @@ -953,17 +1021,27 @@ def certify(args, built, rules, spans, rendered): first_line = {} for iface in sorted((i for group in (specified, gaps, nones, candidates) for i in group), key=lambda i: i["line"]): - key = (iface["producer"], iface["consumer"], iface["attrs"]["Flows"]) + key = (iface["producer"], iface["consumer"], flows_key(iface)) if key in first_line: die("input rows at lines %d and %d share one interface identity " - "(%s -> %s: %s); the review ledger cannot tell them apart" - % (first_line[key], iface["line"], key[0], key[1], key[2])) + "(%s); the review ledger cannot tell them apart" + % (first_line[key], iface["line"], finding_label("gap", key))) first_line[key] = iface["line"] - rows, declared = read_ledger(args.certify) + rows, declared, bound = read_ledger(args.certify) if declared is not None: check_flags(args, declared) + # the first certify run is the only window in which span review can be + # skipped: an input that cites a source must certify with --source, or + # the uncited spans never exist to be reviewed — and a pass would pin + # the hole into the record's flags + if cites and not args.source: + first = cites[0] + die("the input cites %d source line%s (first L%d, at input line %d); " + "certify with --source FILE so the uncited spans are reviewed" + % (len(cites), "" if len(cites) == 1 else "s", first[1], first[0])) + entries, seen_keys = [], {} for kind, finding, disposition, reason, reviewer, date, fp, ln in rows: key = parse_finding(kind, finding, ln, bool(args.source)) @@ -974,8 +1052,8 @@ def certify(args, built, rules, spans, rendered): entries.append((kind, key, finding, fp, disposition, reason, ln)) edge_of, stated = edges_of(specified, gaps, nones) - internal, unfed, unconsumed, isolated = boundary( - names, external, sorted(k for k in edge_of if k[0] != k[1])) + _internal, unfed, unconsumed, isolated = boundary( + names, external, directed(edge_of)) residue = settle(unstated_pairs(names, external, stated), edge_of, rules, class_of)[1] why = {} @@ -991,26 +1069,39 @@ def certify(args, built, rules, spans, rendered): # the notes, never in the identity, so unrelated edits do not re-open rows found = {} # (kind, identity) -> (label, fingerprint, note) for c in sorted(candidates, key=lambda c: c["line"]): - key = (c["producer"], c["consumer"], c["attrs"]["Flows"]) - found[("candidate", key)] = ("%s -> %s: %s" % key, row_fingerprint(c), - "input line %d" % c["line"]) + key = (c["producer"], c["consumer"], flows_key(c)) + found[("candidate", key)] = (finding_label("candidate", key), + row_fingerprint(c), "input line %d" % c["line"]) for g in sorted(gaps, key=lambda g: g["line"]): - key = (g["producer"], g["consumer"], g["attrs"]["Flows"]) - found[("gap", key)] = ("%s -> %s: %s" % key, row_fingerprint(g), + key = (g["producer"], g["consumer"], flows_key(g)) + found[("gap", key)] = (finding_label("gap", key), row_fingerprint(g), "input line %d, missing %s" % (g["line"], ", ".join(g["missing"]))) for n in names: if n in why: - found[("boundary", (n,))] = (n, fingerprint(n, why[n]), why[n]) + found[("boundary", (n,))] = (finding_label("boundary", (n,)), + fingerprint(n, why[n]), why[n]) for a, b in residue: - found[("pair", (a, b))] = ("%s -> %s" % (a, b), fingerprint(a, b), - "no row states it") + found[("pair", (a, b))] = (finding_label("pair", (a, b)), + fingerprint(a, b), "no row states it") for s in spans: - found[("span", (src_sha, s[0], s[1]))] = ( - "%s@%s" % (span_label(s), src_sha), fingerprint(src_sha, s[0], s[1]), - "of %s" % args.source) + key = (src_sha, s[0], s[1]) + found[("span", key)] = (finding_label("span", key), + fingerprint(src_sha, s[0], s[1]), + "of %s" % args.source) drifted, unreviewed, advisories, covered = [], [], [], set() + # the record section stamped by the last pass anchors the files it bound: + # any input or source edit since re-opens the whole review, exactly as a + # flag mismatch refuses the run — heavier than per-finding drift, but it + # covers rows the ledger never fingerprints (specified and explicit-none) + for what, current in (("input", input_sha), ("source", src_sha)): + if what in bound and current is not None and bound[what][0] != current: + sha, ln = bound[what] + drifted.append("drifted: %s changed since the last certified run " + "(record line %d binds sha256 %s; the current %s " + "hashes %s)" + % (what, ln, sha, what, current)) for kind, key, label, fp, disposition, reason, ln in entries: f = found.get((kind, key)) if f is None: @@ -1028,6 +1119,14 @@ def certify(args, built, rules, spans, rendered): advisories.append("gap %s (%s): %s" % (label, f[2], " — ".join( x for x in (disposition, reason) if x))) + elif (kind in ("candidate", "boundary") + and not disposition.lower().startswith("resolved")): + # a dispositioned candidate or boundary finding stays in the + # matrix: the record says so, so a shipping candidate is never + # silent + advisories.append("%s %s (%s): %s" + % (kind, label, f[2], " — ".join( + x for x in (disposition, reason) if x))) for (kind, key), (label, fp, note) in found.items(): if (kind, key) not in covered: unreviewed.append("unreviewed: %s %s (%s) %s (fingerprint %s)" @@ -1039,7 +1138,7 @@ def certify(args, built, rules, spans, rendered): "%d unreviewed), named in the record\n" % (len(blockers), len(drifted), len(unreviewed))) return (certification_record(args, blockers, advisories, - sha256_file(args.input), sha256_text(rendered), + input_sha, sha256_text(rendered), src_sha), bool(blockers)) @@ -1087,32 +1186,42 @@ def main(argv=None): sys.stdout.reconfigure(encoding="utf-8") except (AttributeError, ValueError, OSError): pass - with open(args.input, encoding="utf-8") as fh: - text = fh.read() - components, interfaces, rules, cites, has_rules = parse(text) + with open(args.input, "rb") as fh: + raw = fh.read() + components, interfaces, rules, cites, has_rules = parse(raw.decode("utf-8")) if args.source: cov, spans = coverage(cites, args.source) else: cov, spans = None, [] built = build(components, interfaces, rules) + rendered = report(*built, rules=rules, sample_n=args.sample, cov=cov, + has_rules=has_rules) if args.certify: + # a component name carrying the identity delimiters cannot be pasted + # into a ledger Finding cell and parsed back — the review would brick + # with false drift on every retry — so refuse the name up front + # (generation is unchanged) + for name, _kind, _cls, state, lineno in components: + if not superseded(state) and (" -> " in name or ": " in name): + die("component name %r at line %d: component names cannot " + "contain ' -> ' or ': ' under certification — rename the " + "component" % (name, lineno)) # the gate certifies what the report shows: render it once — coverage # section included when --source ran — so certification dies on the same # invariant (exit 2) and the record can bind the report's sha256; then # judge the ledger against the same derivation. The record replaces the # rendered report on stdout, lands beside the ledger as a standalone # file, and a pass also stamps it into the ledger itself. - rendered = report(*built, rules=rules, sample_n=args.sample, cov=cov, - has_rules=has_rules) - record, refused = certify(args, built, rules, spans, rendered) - with open(record_path(args.certify), "w", encoding="utf-8") as fh: + record, refused = certify(args, built, rules, spans, rendered, + hashlib.sha256(raw).hexdigest(), cites) + with open(record_path(args.certify), "w", encoding="utf-8", + newline="\n") as fh: fh.write(record) if not refused: stamp_ledger_record(args.certify, record) sys.stdout.write(record) return 3 if refused else 0 - sys.stdout.write(report(*built, rules=rules, sample_n=args.sample, cov=cov, - has_rules=has_rules)) + sys.stdout.write(rendered) return 0 diff --git a/skills/interface-matrix/scripts/test_interface_matrix.py b/skills/interface-matrix/scripts/test_interface_matrix.py index 7bd1e77..daa0d3e 100644 --- a/skills/interface-matrix/scripts/test_interface_matrix.py +++ b/skills/interface-matrix/scripts/test_interface_matrix.py @@ -1114,9 +1114,10 @@ def rm_dir(d): ) # one finding of each kind the wildcard rule cannot settle: a candidate (line 22), -# a gap (line 23) and all three boundary findings; the rule settles every classed pair +# a gap (line 23) and all three boundary findings; the rule settles every classed pair. +# No row cites a source: a citing input cannot be certified without --source CERT_INPUT = doc_rules( - "| Ingest | Store | rows | csv | cron | me | S:L1 |\n" + "| Ingest | Store | rows | csv | cron | me | |\n" "| ? | Scorer | digest | csv | cron | me | |\n" "| Ingest | Analyst | rows | | ? | me | |\n", "| * | * | none | every classed pair is settled |\n", @@ -1156,22 +1157,26 @@ def test_fully_reviewed_input_certifies(self): self.assertIn("- report: sha256 ", proc.stdout) self.assertIn("- flags: --sample 20\n", proc.stdout) self.assertIn("- blockers: none", proc.stdout) - self.assertIn("- advisories: 1", proc.stdout) + # the open gap plus every boundary finding dispositioned rather than + # resolved is carried as an advisory: what shipped stays visible + self.assertIn("- advisories: 4", proc.stdout) self.assertIn("gap Ingest -> Analyst: rows (input line 23, missing Format, " "Trigger): open-parked — blocked on the vendor's format doc", proc.stdout) + self.assertIn("boundary Ingest (nothing feeds it): explained — the " + "pipeline's entry point", proc.stdout) def test_open_gap_is_an_advisory_and_a_filled_gap_is_not(self): # the gap is a finding either way; an open disposition is an advisory in # the record, a non-open one simply satisfies the gate text = doc( - "| Ingest | Store | rows | | cron | me | S:L1 |\n" - "| Store | Ingest | acks | csv | cron | me | S:L1 |\n", + "| Ingest | Store | rows | | cron | me | |\n" + "| Store | Ingest | acks | csv | cron | me | |\n", components=TWO_COMPONENTS, ) gap = "Ingest -> Store: rows" line = lineno(text, "| Ingest | Store | rows | | cron") - fingerprint = iface_fp("Ingest", "Store", "rows", "", "cron", "me", "S:L1") + fingerprint = iface_fp("Ingest", "Store", "rows", "", "cron", "me") opened = run_certify(text, ledger( row("gap", gap, "open-parked", "waiting on the vendor", fingerprint=fingerprint))) @@ -1185,6 +1190,51 @@ def test_open_gap_is_an_advisory_and_a_filled_gap_is_not(self): self.assertEqual(filled.returncode, 0, filled.stderr) self.assertIn("- advisories: none", filled.stdout) + def test_dispositioned_candidate_and_boundary_findings_are_advisories(self): + # a candidate or boundary finding may ship dispositioned, never + # silently: the record lists it, kind-named, until it is resolved + led = ledger( + row("candidate", "? -> Scorer: digest", "accepted", + "scope cut to the nightly digest run", fingerprint=CAND_FP), + row("gap", "Ingest -> Analyst: rows", "open-parked", + "blocked on the vendor's format doc", fingerprint=GAP_FP), + row("boundary", "Ingest", "explained", "the pipeline's entry point", + fingerprint=BOUNDARY_FP["Ingest"]), + row("boundary", "Scorer", "explained", "runs on a manual trigger", + fingerprint=BOUNDARY_FP["Scorer"]), + row("boundary", "Store", "explained", "the terminal sink", + fingerprint=BOUNDARY_FP["Store"]), + ) + proc = run_certify(CERT_INPUT, led) + self.assertEqual(proc.returncode, 0, proc.stderr) + self.assertIn("- advisories: 5", proc.stdout) + self.assertIn("candidate ? -> Scorer: digest (input line 22): accepted — " + "scope cut to the nightly digest run", proc.stdout) + self.assertIn("boundary Store (nothing consumes its output): explained — " + "the terminal sink", proc.stdout) + + def test_blank_flows_gap_round_trips_through_the_ledger(self): + # an empty Flows cell is a legitimate gap; its identity label is + # `producer -> consumer: ?` — a string a stripped ledger cell can hold + # and parse_finding maps back to the empty identity + text = doc( + "| Ingest | Store | | ? | cron | me | |\n" + "| Store | Ingest | acks | csv | cron | me | |\n", + components=TWO_COMPONENTS, + ) + line = lineno(text, "| Ingest | Store | |") + fingerprint = iface_fp("Ingest", "Store", "", "?", "cron", "me") + refused = run_certify(text, LEDGER_HEAD) + self.assertEqual(refused.returncode, 3, refused.stderr) + self.assertIn("unreviewed: interface gap Ingest -> Store: ? (input line " + "%d, missing Flows, Format) has no disposition " + "(fingerprint %s)" % (line, fingerprint), refused.stdout) + ok = run_certify(text, ledger( + row("gap", "Ingest -> Store: ?", "filled", + "the flows cell is the ack payload", fingerprint=fingerprint))) + self.assertEqual(ok.returncode, 0, ok.stderr) + self.assertIn("- gate: certified", ok.stdout) + def test_unresolved_candidate_blocks(self): proc = run_certify(CERT_INPUT, ledger( row("gap", "Ingest -> Analyst: rows", "open-parked", "blocked", @@ -1204,7 +1254,7 @@ def test_unresolved_candidate_blocks(self): proc.stdout) def test_unstated_pair_without_disposition_blocks(self): - text = doc("| Ingest | Store | rows | csv | cron | me | S:L1 |\n", + text = doc("| Ingest | Store | rows | csv | cron | me | |\n", components=TWO_COMPONENTS) proc = run_certify(text, ledger( row("boundary", "Ingest", "explained", "the entry point", @@ -1263,9 +1313,8 @@ def test_unknown_ledger_kind_exits_1(self): self.assertIn("unknown kind 'mystery'", proc.stderr) def test_ledger_row_for_an_unknown_finding_is_drift_not_an_error(self): - # U1 exited 1 here; R5's drift semantics supersede that: an entry whose - # finding matches nothing in the current input is drift, so the review - # re-opens (exit 3) instead of calling the ledger row malformed + # an entry whose finding matches nothing in the current input is drift: + # the review re-opens (exit 3) instead of a malformed-row error led = FULL_LEDGER + row("boundary", "Ghost", "explained", "not a component", fingerprint=fp("Ghost", "anything")) proc = run_certify(CERT_INPUT, led) @@ -1282,6 +1331,28 @@ def test_span_row_without_source_exits_1_naming_the_flag(self): self.assertIn("L2-6", proc.stderr) self.assertIn("--source", proc.stderr) + def test_citing_input_certified_without_source_exits_1(self): + # the first certify run is the only window in which span review can be + # skipped: an input that cites a source must certify with --source + text = doc("| Ingest | Store | rows | csv | cron | me | S:L1 |\n", + components=TWO_COMPONENTS) + proc = run_certify(text, LEDGER_HEAD) + self.assertEqual(proc.returncode, 1, proc.stdout) + self.assertIn("cites", proc.stderr) + self.assertIn("--source", proc.stderr) + self.assertIn("uncited spans", proc.stderr) + self.assertIn("L1", proc.stderr) + + def test_citing_input_certified_with_source_is_a_normal_refusal(self): + # with --source the same input proceeds to the gate: the refusal names + # the uncited span, not the guard + text = doc("| Ingest | Store | rows | csv | cron | me | S:L1 |\n", + components=TWO_COMPONENTS) + proc = run_certify_source(text, LEDGER_HEAD, SOURCE) + self.assertEqual(proc.returncode, 3, proc.stderr) + self.assertIn("unreviewed: uncited span L2-6@", proc.stdout) + self.assertNotIn("uncited spans are reviewed", proc.stderr) + def test_duplicate_ledger_row_exits_1_naming_both_lines(self): led = FULL_LEDGER + row("boundary", "Scorer", "explained", "twice", fingerprint=BOUNDARY_FP["Scorer"]) @@ -1300,6 +1371,29 @@ def test_duplicate_ledger_identity_exits_1_across_spellings(self): self.assertEqual(proc.returncode, 1, proc.stdout) self.assertIn("both disposition gap", proc.stderr) + def test_table_after_the_record_section_exits_1(self): + # a disposition table placed after a certification record is content + # the record-section scan must not swallow + led = (FULL_LEDGER.rstrip("\n") + "\n\n## Certification record\n\n" + + "- flags: --sample 20\n\n" + + "| Kind | Finding | Disposition | Reason | Reviewer | Date | Fingerprint |\n" + + "|---|---|---|---|---|---|---|\n" + + row("gap", "Ghost -> Nobody: x", "filled", "a second table")) + proc = run_certify(CERT_INPUT, led) + self.assertEqual(proc.returncode, 1, proc.stdout) + self.assertIn("cannot live inside a certification record", proc.stderr) + self.assertIn("line %d" % (led[:led.rindex("| Kind |")].count("\n") + 1), + proc.stderr) + + def test_row_after_the_record_section_exits_1(self): + led = (FULL_LEDGER.rstrip("\n") + "\n\n## Certification record\n\n" + + "- flags: --sample 20\n\n" + + row("mystery", "line 22", "resolved", "no such kind")) + proc = run_certify(CERT_INPUT, led) + self.assertEqual(proc.returncode, 1, proc.stdout) + self.assertIn("cannot live inside a certification record", proc.stderr) + self.assertIn("line %d" % lineno(led, "| mystery |"), proc.stderr) + def test_header_only_ledger_lists_every_finding(self): proc = run_certify(CERT_INPUT, LEDGER_HEAD) self.assertEqual(proc.returncode, 3, proc.stdout) @@ -1324,8 +1418,8 @@ def test_header_only_ledger_lists_every_finding(self): def test_finding_free_input_certifies_with_an_empty_ledger(self): text = doc( - "| Ingest | Store | rows | csv | cron | me | S:L1 |\n" - "| Store | Ingest | acks | csv | cron | me | S:L1 |\n", + "| Ingest | Store | rows | csv | cron | me | |\n" + "| Store | Ingest | acks | csv | cron | me | |\n", components=TWO_COMPONENTS, ) proc = run_certify(text, LEDGER_HEAD) @@ -1364,6 +1458,21 @@ def test_duplicate_interface_identity_in_the_input_exits_1_naming_both(self): self.assertIn("interfaces with gaps: 1", plain.stdout) self.assertIn("specified interfaces: 1", plain.stdout) + def test_component_name_with_identity_delimiters_exits_1_under_certify(self): + # `A -> B` cannot be expressed as a Finding identity no paste can + # satisfy; certification refuses the name, generation is unchanged + components = ("| Component | Kind | Notes |\n|---|---|---|\n" + "| A -> B | | quirky |\n| C | | plain |\n\n") + text = doc("| A -> B | C | flows | csv | cron | me | |\n", + components=components) + proc = run_certify(text, LEDGER_HEAD) + self.assertEqual(proc.returncode, 1, proc.stdout) + self.assertIn("component names cannot contain", proc.stderr) + self.assertIn("line %d" % lineno(text, "| A -> B |"), proc.stderr) + plain = run(text) + self.assertEqual(plain.returncode, 0, plain.stderr) + self.assertIn("specified interfaces: 1", plain.stdout) + def drift_doc(rows): """The drift fixture: gap rows plus a wildcard rule that settles every @@ -1391,13 +1500,13 @@ def drift_doc(rows): class TestCertifyDrift(unittest.TestCase): - """R5: edits after review re-open exactly the rows they touch.""" + """Edits after review re-open exactly the rows they touch.""" def test_appended_unrelated_row_still_certifies(self): # a new specified row on an already-stated pair changes no finding: # identity keys survive unrelated edits text = drift_doc( - DRIFT_ROWS + ["| Ingest | Store | batches | csv | cron | me | S:L1 |\n"]) + DRIFT_ROWS + ["| Ingest | Store | batches | csv | cron | me | |\n"]) proc = run_certify(text, DRIFT_LEDGER) self.assertEqual(proc.returncode, 0, proc.stderr) self.assertIn("- gate: certified", proc.stdout) @@ -1496,26 +1605,38 @@ def test_edited_source_file_reopens_the_span_review(self): "matches any span finding in the input" % (old_sha, lineno(led, "narrative prose")), second.stdout) + # the stamped record also binds the source file itself: editing it + # re-opens the review even before any span finding is judged + self.assertIn("drifted: source changed since the last certified run", + second.stdout) self.assertIn("unreviewed: uncited span L2-7@%s (of " % new_sha, second.stdout) self.assertIn(") is unread (fingerprint %s)" % fp(new_sha, 2, 7), second.stdout) - self.assertIn("(1 drifted, 1 unreviewed)", second.stderr) + self.assertIn("(2 drifted, 1 unreviewed)", second.stderr) finally: rm_dir(d) # finding-free under any flags: both pairs stated, both components fed and -# consumed, and every source line cited +# consumed, and every source line cited — so the --source runs pass too FLAG_INPUT = doc( "| Ingest | Store | rows | csv | cron | me | S:L1-6 |\n" "| Store | Ingest | acks | csv | cron | me | S:L1-6 |\n", components=TWO_COMPONENTS, ) +# the same input without citations, for flag tests that certify without +# --source (a citing input cannot be certified without it) +FLAG_INPUT_NOCITE = doc( + "| Ingest | Store | rows | csv | cron | me | |\n" + "| Store | Ingest | acks | csv | cron | me | |\n", + components=TWO_COMPONENTS, +) + class TestCertifyFlags(unittest.TestCase): - """KTD3: --certify replays the recorded review exactly.""" + """--certify replays the recorded review exactly.""" def test_certify_without_the_declared_source_flag_exits_1(self): d = open_dir(FLAG_INPUT, LEDGER_HEAD, SOURCE) @@ -1532,7 +1653,7 @@ def test_certify_without_the_declared_source_flag_exits_1(self): rm_dir(d) def test_certify_with_the_wrong_sample_exits_1_and_the_recorded_one_passes(self): - d = open_dir(FLAG_INPUT, LEDGER_HEAD) + d = open_dir(FLAG_INPUT_NOCITE, LEDGER_HEAD) try: self.assertEqual(run_in(d, "--sample", "0").returncode, 0) wrong = run_in(d) # the default is 20, the record pins 0 @@ -1545,12 +1666,11 @@ def test_certify_with_the_wrong_sample_exits_1_and_the_recorded_one_passes(self) rm_dir(d) def test_undeclared_source_flag_exits_1(self): - d = open_dir(FLAG_INPUT, LEDGER_HEAD) + # a hand-pinned record section declares no --source (a citing input + # can no longer stamp such a section: the certify guard refuses it) + led = LEDGER_HEAD + "\n## Certification record\n\n- flags: --sample 20\n" + d = open_dir(FLAG_INPUT_NOCITE, led, SOURCE) try: - self.assertEqual(run_in(d).returncode, 0) - with open(os.path.join(d, "source.txt"), "w", encoding="utf-8", - newline="\n") as fh: - fh.write(SOURCE) extra = run_in(d, "--source", os.path.join(d, "source.txt")) self.assertEqual(extra.returncode, 1, extra.stdout) self.assertIn("no --source", extra.stderr) @@ -1559,7 +1679,7 @@ def test_undeclared_source_flag_exits_1(self): def test_hand_written_record_section_pins_flags_before_any_pass(self): led = LEDGER_HEAD + "\n## Certification record\n\n- flags: --sample 0\n" - d = open_dir(FLAG_INPUT, led) + d = open_dir(FLAG_INPUT_NOCITE, led) try: self.assertEqual(run_in(d).returncode, 1) self.assertEqual(run_in(d, "--sample", "0").returncode, 0) @@ -1567,6 +1687,60 @@ def test_hand_written_record_section_pins_flags_before_any_pass(self): rm_dir(d) +class TestCertifyAnchor(unittest.TestCase): + """The stamped record is a replay anchor: the input and source sha256 it + binds are re-derived and compared on every later certification run.""" + + def test_post_review_edit_of_a_stated_row_reopens_the_review(self): + # a fully-specified input has no finding to fingerprint; the stamped + # input sha is what re-opens the review when the input is edited + d = open_dir(FLAG_INPUT_NOCITE, LEDGER_HEAD) + try: + first = run_in(d) + self.assertEqual(first.returncode, 0, first.stderr) + with open(os.path.join(d, "inventory.md"), "w", encoding="utf-8", + newline="\n") as fh: + fh.write(FLAG_INPUT_NOCITE.replace("rows | csv", "rows | xml")) + second = run_in(d) + self.assertEqual(second.returncode, 3, second.stdout) + self.assertIn("(1 drifted, 0 unreviewed)", second.stderr) + self.assertIn("drifted: input changed since the last certified run", + second.stdout) + self.assertIn("- blockers: 1", second.stdout) + # the ledger keeps its stamped section — the last passing run — + # while the standalone record is the refused latest run + self.assertIn("- gate: certified", cat(d, "review.md")) + self.assertIn("- gate: refused", cat(d, "review.cert.md")) + finally: + rm_dir(d) + + def test_tampered_record_input_sha_refuses(self): + d = open_dir(FLAG_INPUT_NOCITE, LEDGER_HEAD) + try: + self.assertEqual(run_in(d).returncode, 0) + stamped = cat(d, "review.md") + bad = re.sub(r"^- input: (.*) \(sha256 [0-9a-f]+\)$", + r"- input: \1 (sha256 %s)" % ("0" * 64), stamped, + flags=re.M) + self.assertNotEqual(bad, stamped) + with open(os.path.join(d, "review.md"), "w", encoding="utf-8", + newline="\n") as fh: + fh.write(bad) + proc = run_in(d) + self.assertEqual(proc.returncode, 3, proc.stdout) + self.assertIn("drifted: input changed since the last certified run", + proc.stdout) + # a line that cannot be a binding at all is a bad record section + broken = re.sub(r"^- input: .*$", "- input: not-a-binding", + stamped, flags=re.M) + with open(os.path.join(d, "review.md"), "w", encoding="utf-8", + newline="\n") as fh: + fh.write(broken) + self.assertEqual(run_in(d).returncode, 1) + finally: + rm_dir(d) + + class TestCertifyRecord(unittest.TestCase): def test_success_writes_the_record_file_with_every_bound_field(self): text = doc("| Ingest | Store | rows | csv | cron | me | S:L1 |\n", @@ -1610,7 +1784,7 @@ def test_success_writes_the_record_file_with_every_bound_field(self): rm_dir(d) def test_record_is_reproducible_and_the_ledger_section_stamped_once(self): - d = open_dir(FLAG_INPUT, LEDGER_HEAD) + d = open_dir(FLAG_INPUT_NOCITE, LEDGER_HEAD) try: first = run_in(d) self.assertEqual(first.returncode, 0, first.stderr) @@ -1641,6 +1815,23 @@ def test_refused_certification_also_writes_its_record_but_not_the_ledger(self): finally: rm_dir(d) + def test_after_a_refusal_the_record_and_the_ledger_section_disagree(self): + # the section stamped inside the ledger is the last PASSING run — the + # replay anchor; the standalone record reflects the latest completed + # run, refused here + d = open_dir(FLAG_INPUT_NOCITE, LEDGER_HEAD) + try: + self.assertEqual(run_in(d).returncode, 0) + with open(os.path.join(d, "inventory.md"), "w", encoding="utf-8", + newline="\n") as fh: + fh.write(FLAG_INPUT_NOCITE.replace("rows | csv", "rows | xml")) + refused = run_in(d) + self.assertEqual(refused.returncode, 3, refused.stderr) + self.assertIn("- gate: refused", cat(d, "review.cert.md")) + self.assertIn("- gate: certified", cat(d, "review.md")) + finally: + rm_dir(d) + UNREVIEWED_LINE = re.compile( r"^ - unreviewed: (missing-component candidate|interface gap|boundary " @@ -1687,7 +1878,8 @@ def test_generate_review_certify_drift_fix_recertify(self): ok = run_in(d) self.assertEqual(ok.returncode, 0, ok.stderr) self.assertIn("- gate: certified", ok.stdout) - # 4. edit a reviewed row's flows cell + # 4. edit a reviewed row's flows cell; the stamped record section + # anchored that pass, so it goes with the input it bound with open(os.path.join(d, "inventory.md"), "w", encoding="utf-8", newline="\n") as fh: fh.write(drift_doc([ @@ -1695,6 +1887,9 @@ def test_generate_review_certify_drift_fix_recertify(self): DRIFT_ROWS[1], DRIFT_ROWS[2], ])) + with open(os.path.join(d, "review.md"), "w", encoding="utf-8", + newline="\n") as fh: + fh.write(ledger(*entries)) # 5. certify: exactly that row re-opened, drifted and unreviewed drift = run_in(d) self.assertEqual(drift.returncode, 3, drift.stderr) From 4409261d03f4c5f52974d1b39f5898ba3a2553e5 Mon Sep 17 00:00:00 2001 From: macblackstuff <148771651+macblackstuff@users.noreply.github.com> Date: Wed, 30 Sep 2026 04:48:38 +0200 Subject: [PATCH 6/7] fix(tests): decode subprocess output as UTF-8 on every platform The script reconfigures stdout to UTF-8, but the test helpers decoded with the locale default; on a US-Windows runner (cp437) the em-dashes in certification record lines mojibake and four advisory/record assertions fail. Decode the captured streams explicitly. Co-Authored-By: Claude Claude-Session: sess_c59696de-2f36-4d7c-99cd-ad4102b8cc4e --- skills/interface-matrix/scripts/test_interface_matrix.py | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/skills/interface-matrix/scripts/test_interface_matrix.py b/skills/interface-matrix/scripts/test_interface_matrix.py index daa0d3e..9f64cdf 100644 --- a/skills/interface-matrix/scripts/test_interface_matrix.py +++ b/skills/interface-matrix/scripts/test_interface_matrix.py @@ -69,6 +69,11 @@ def run(text, *args): [sys.executable, SCRIPT, path, *args], capture_output=True, text=True, + # the script reconfigures stdout to UTF-8; decode the same way on + # every platform or the em-dashes in record lines mojibake on + # legacy Windows codepages (cp437 has none). + encoding="utf-8", + errors="replace", ) finally: os.unlink(path) @@ -1091,7 +1096,7 @@ def run_in(d, *args): return subprocess.run( [sys.executable, SCRIPT, os.path.join(d, "inventory.md"), "--certify", os.path.join(d, "review.md"), *args], - capture_output=True, text=True) + capture_output=True, text=True, encoding="utf-8", errors="replace") def cat(d, name): From 8889f82155ac58a21bbdf04c0c6b71d7a0ec7f37 Mon Sep 17 00:00:00 2001 From: macblackstuff <148771651+macblackstuff@users.noreply.github.com> Date: Wed, 30 Sep 2026 15:01:11 +0200 Subject: [PATCH 7/7] fix: docstring line started with 'from', tripping SAP's import lint MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit SAP's CI greps ^(import|from) over the vendored script; a wrapped docstring sentence beginning 'from the input as…' read as a third-party import. Reworded; no code change. Co-Authored-By: Claude Claude-Session: sess_c59696de-2f36-4d7c-99cd-ad4102b8cc4e --- skills/interface-matrix/scripts/interface_matrix.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/skills/interface-matrix/scripts/interface_matrix.py b/skills/interface-matrix/scripts/interface_matrix.py index 07c6110..a447f24 100644 --- a/skills/interface-matrix/scripts/interface_matrix.py +++ b/skills/interface-matrix/scripts/interface_matrix.py @@ -17,8 +17,8 @@ was dispositioned. Certification exits 0 when every finding of the five ledger kinds is dispositioned and none has drifted — feedback loops, self-dependencies and the class-rules audit are report findings a human reviews; the gate does -not disposition them; 3 naming every blocker, entries whose finding drifted -from the input as `drifted:` and findings no entry covers as `unreviewed:`; +not disposition them; 3 naming every blocker — entries whose finding drifted +out of the input as `drifted:` and findings no entry covers as `unreviewed:`; and 1 on a bad ledger row, a duplicate identity in the ledger or the input, a citing input certified without --source, a component name containing the identity delimiters ` -> ` or `: `, or an invocation whose flags do not replay