143 lines
6.1 KiB
TypeScript
143 lines
6.1 KiB
TypeScript
import * as React from "react";
|
|
import { GatePipelineDiagram } from "@/src/components/notes/GatePipelineDiagram";
|
|
import {
|
|
NoteCode,
|
|
NoteFigure,
|
|
NoteList,
|
|
NoteParagraph,
|
|
NoteSection,
|
|
} from "@/src/components/notes/NoteProse";
|
|
import type { Note } from "./types";
|
|
|
|
const Body: React.FC = () => (
|
|
<>
|
|
<NoteParagraph>
|
|
A workflow that depends on discipline lasts until the first deadline. In
|
|
my Rust systems the workflow is enforced by scripts instead: a change that
|
|
skips the spec, drops coverage or leaves a weak test does not get past the
|
|
gates, no matter who wrote it or how it was generated.
|
|
</NoteParagraph>
|
|
|
|
<NoteSection>The gates</NoteSection>
|
|
<NoteParagraph>
|
|
There are four, and each one answers a different question.
|
|
</NoteParagraph>
|
|
<NoteList>
|
|
<li>
|
|
<strong>TDD gate.</strong> Was the behavior specified as a test before
|
|
it was implemented? Skipping the spec fails the gate.
|
|
</li>
|
|
<li>
|
|
<strong>Functional-style gate.</strong> It forbids methods, traits,{" "}
|
|
<NoteCode>&mut</NoteCode> and <NoteCode>unwrap</NoteCode> in the
|
|
domain code. Pure functions over immutable data are easier to test and
|
|
to mutate, so the style is a rule, not a preference. The gate is itself
|
|
covered by a self-test script, because a checker that silently stops
|
|
checking is worse than none.
|
|
</li>
|
|
<li>
|
|
<strong>Mutation gate.</strong> The tool changes the code in small ways
|
|
and the specs must notice. The policy is zero survivors. A surviving
|
|
mutant means either a weak assertion or dead code, and both get fixed.
|
|
</li>
|
|
<li>
|
|
<strong>Coverage ratchet.</strong> Coverage may rise but may not fall.
|
|
The gate fails when the number drops below the last recorded one.
|
|
</li>
|
|
</NoteList>
|
|
|
|
<NoteFigure caption="Four gates in sequence, run at three scopes.">
|
|
<GatePipelineDiagram />
|
|
</NoteFigure>
|
|
|
|
<NoteSection>Three speeds</NoteSection>
|
|
<NoteParagraph>
|
|
Mutation testing is slow, so the same gates run at three scopes. In the
|
|
dev loop they look only at changed lines, which keeps feedback fast enough
|
|
to use constantly. Before a merge they run on the changed crates. The
|
|
merge gate runs everything. Nothing is skipped on the way to main; a
|
|
cheaper check just runs earlier and more often.
|
|
</NoteParagraph>
|
|
<NoteParagraph>
|
|
The mutation gate also keeps a proof ledger. When the code bytes and the
|
|
toolchain are unchanged since a previous run, the mutants already caught
|
|
are carried over instead of being recomputed. That keeps a zero-survivor
|
|
policy affordable on a large codebase, and it stays honest because any
|
|
change to the code or the toolchain invalidates the entry.
|
|
</NoteParagraph>
|
|
|
|
<NoteSection>Claims drift, measurements do not</NoteSection>
|
|
<NoteParagraph>
|
|
The most useful lesson came from reading my own documentation. A README
|
|
and a Makefile said "100% coverage at all times", while the
|
|
measured badge said 87.5%. Nobody lied; the sentence was true once and
|
|
then the code moved. Prose does not fail a build.
|
|
</NoteParagraph>
|
|
<NoteParagraph>
|
|
So I quote the measured number and enforce a ratchet instead of promising
|
|
an absolute. A ratchet is a claim the machine checks on every run: the
|
|
number may only go up. It is a weaker sentence than "always
|
|
100%" and a far more reliable one.
|
|
</NoteParagraph>
|
|
|
|
<NoteSection>A story from this website</NoteSection>
|
|
<NoteParagraph>
|
|
This site has a small domain core with the same setup: Vitest for the
|
|
specs and Stryker for mutation testing. At one point Stryker reported a
|
|
mutation score of 14%. That looked like terrible specs, but the specs were
|
|
fine. The mutation tool was silently running zero tests per mutant under
|
|
Vitest 5, so every mutant looked like it had survived.
|
|
</NoteParagraph>
|
|
<NoteParagraph>
|
|
The cause was a version mismatch between the Stryker Vitest runner and
|
|
Vitest itself. Pinning Vitest to 4.1.x fixed it. The first real run then
|
|
found 9 genuine gaps in the specs, the kind a green test suite had been
|
|
hiding. Rewriting one date check also removed redundant conditions that no
|
|
test could ever distinguish from the simpler version.
|
|
</NoteParagraph>
|
|
<NoteParagraph>
|
|
The pin is now written down in the project instructions, with the reason,
|
|
so the next dependency bump does not quietly undo it.
|
|
</NoteParagraph>
|
|
|
|
<NoteSection>What I take from it</NoteSection>
|
|
<NoteList>
|
|
<li>
|
|
<strong>Verify that your verification runs.</strong> A gate that
|
|
executes zero tests passes or fails for the wrong reasons. Treat "0
|
|
tests ran" as a failure, and look at a surprising score before
|
|
explaining it away.
|
|
</li>
|
|
<li>
|
|
<strong>Enforce, do not remind.</strong> If a rule matters, a script
|
|
should fail when it is broken. Rules that live in a README drift.
|
|
</li>
|
|
<li>
|
|
<strong>Quote measurements.</strong> Write the number the tool printed,
|
|
with the command that printed it, and let a ratchet protect it.
|
|
</li>
|
|
<li>
|
|
<strong>Make the cheap check run first.</strong> Three speeds let the
|
|
slow, thorough gates exist without slowing the loop that people use
|
|
every few minutes.
|
|
</li>
|
|
</NoteList>
|
|
<NoteParagraph>
|
|
None of this is specific to Rust. The same shape works for a TypeScript
|
|
project: a spec before the code, mutation testing on the changed files,
|
|
and a coverage number that is only allowed to go up.
|
|
</NoteParagraph>
|
|
</>
|
|
);
|
|
|
|
export const specFirstGates: Note = {
|
|
slug: "how-i-enforce-spec-first-mechanically",
|
|
title: "How I enforce spec-first mechanically",
|
|
summary:
|
|
"Four gates, three speeds, and one lesson: claims drift, so measure and ratchet instead of promising.",
|
|
publishedOn: "2026-10-03",
|
|
readingMinutes: 4,
|
|
tags: ["testing", "mutation testing", "process"],
|
|
Body,
|
|
};
|