docs: OpenStream benchmark reports, published per release (#150)

Adds a reports section to the OpenStream spec so its claims rest on a
reproducible measurement rather than an assertion. Each report is a run of
the envelope over a defined corpus on real hardware: proof that
decompression restores every byte, that an incompressible input costs only
the framing overhead, that a compressible one saves what it claims against
the complete wire size, and how long each codec takes.

- docs/openstream/reports/ holds a machine-readable <id>.json (canonical,
  with a versioned schema) and a rendered <id>.md per report, plus a README
  on the shape and on submitting one. The seed report is nixamp 0.17.1 over
  the synthetic corpus, labelled synthetic so no one reads a padded-fixture
  number as production.
- The site renders them at /docs/openstream/reports (index) and
  /docs/openstream/reports/<id> (one report), under the dynamic /docs/[slug]
  tree so the reports routes never shadow a spec's own doc page. A small
  lib/reports.ts reads the JSON at build time; REPORTED_SPECS keeps the
  route surface explicit. sitemap includes the index and every report.
- The spec doc gains a Benchmark reports section linking there, and repeats
  the honest caveats: OpenStream frames Zstandard and gzip rather than being
  a new algorithm, synthetic padding flatters a codec, an efficient real
  feed saves little, and round-trip exactness is the one pass/fail.

The report format is produced by `nixamp compression benchmark` (in the
nixamp repo); a release runs it and commits the two files here.


Claude-Session: https://claude.ai/code/session_01MxNif5tsYq4LczgG7aE8Jp

Co-authored-by: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
Anthony Ettinger 2026-09-12 06:23:46 -07:00 • committed by GitHub
parent 692bfa0a2a
commit 9f42ce222a
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
8 changed files with 1145 additions and 1 deletions

View file

@ -0,0 +1,116 @@
import { existsSync, readdirSync, readFileSync } from "node:fs";
import { resolve } from "node:path";
// Benchmark reports published alongside a spec, at docs/<spec>/reports/.
// Each report is a machine-readable <id>.json (the canonical artifact,
// produced by a reference implementation and committed on a release) and a
// rendered <id>.md that this site displays. Read at build time, so the
// deployed image has no runtime filesystem dependency.
const DOCS_DIR = resolve(process.cwd(), "../../docs");
// Specs that carry a reports section. Kept explicit so a stray directory
// never becomes a route.
export const REPORTED_SPECS = ["openstream"] as const;
export type ReportedSpec = (typeof REPORTED_SPECS)[number];
export function hasReports(spec: string): spec is ReportedSpec {
return (REPORTED_SPECS as readonly string[]).includes(spec);
}
function reportsDir(spec: string): string {
return resolve(DOCS_DIR, spec, "reports");
}
const ID = /^[a-z0-9][a-z0-9._-]{0,80}$/;
/** The report ids published for a spec, newest first by filename. */
export function reportIds(spec: string): string[] {
if (!hasReports(spec)) return [];
const dir = reportsDir(spec);
if (!existsSync(dir)) return [];
const ids = new Set<string>();
for (const name of readdirSync(dir)) {
const m = /^(.+)\.json$/.exec(name);
if (m && ID.test(m[1] as string)) ids.add(m[1] as string);
}
return [...ids].sort().reverse();
}
export interface ReportEnvironment {
runtime: string;
zstd: string;
zlib: string;
os: string;
arch: string;
cpu: string;
cores: number;
memoryGiB: number;
}
export interface ReportModeSummary {
mode: string;
level: number;
wireBytes: number;
savingsPercent: number;
roundTrip: boolean;
encodeMs: number;
decodeMs: number;
}
export interface Report {
schema: number;
spec: string;
specVersion: string;
generatedAt: string;
implementation: { name: string; version: string };
environment: ReportEnvironment;
summary: { corpusBytes: number; byMode: ReportModeSummary[] };
caveats: string[];
// Other fields (samples, policy, envelope) are present in the JSON but not
// needed for the listing; the detail page renders the committed Markdown.
[key: string]: unknown;
}
export function readReportJson(spec: string, id: string): Report | null {
if (!hasReports(spec) || !ID.test(id)) return null;
try {
return JSON.parse(readFileSync(resolve(reportsDir(spec), `${id}.json`), "utf8")) as Report;
} catch {
return null;
}
}
export function readReportMarkdown(spec: string, id: string): string | null {
if (!hasReports(spec) || !ID.test(id)) return null;
try {
return readFileSync(resolve(reportsDir(spec), `${id}.md`), "utf8");
} catch {
return null;
}
}
export interface ReportSummary {
id: string;
generatedAt: string;
implementation: string;
/** The best round-tripping saving on the corpus, for the listing line. */
headline: string;
}
export function listReports(spec: string): ReportSummary[] {
const out: ReportSummary[] = [];
for (const id of reportIds(spec)) {
const r = readReportJson(spec, id);
if (!r) continue;
const best = r.summary?.byMode
?.filter((m) => m.roundTrip && m.mode !== "stored")
.sort((a, b) => b.savingsPercent - a.savingsPercent)[0];
out.push({
id,
generatedAt: r.generatedAt,
implementation: `${r.implementation.name} ${r.implementation.version}`,
headline: best ? `${best.mode} saved ${best.savingsPercent}% across the corpus` : "no codec beat stored on this corpus",
});
}
return out;
}