Files
atomic-design-poc/docs/project/archive/refactor-backlog-setup/refactor-backlog/tools/baseline-scan.mjs
T
ehoandClaude Opus 5 12f17d9d73 docs: archive the finished backlogs (RD-30)
Two backlog trees are complete: `docs/project/backlog/` (75 files, every
WP done) and `docs/project/refactor-backlog-setup/` (the arc before it).
Move both under `docs/project/archive/` with `git mv`, so history stays
intact through `git log --follow`. `SHOWCASE-ROADMAP.md` moves with them,
because it points at the now-archived backlog README.

Add `docs/project/archive/README.md`. It states that these trees are
historical and names the two directories that are still live.

Repoint every inbound reference named in RD-30's Files table: CLAUDE.md,
the root README, both backend READMEs, `LetterHtml.cs`, `a11y.mdx`, the
`document-feature` and `new-ssp` skills, and the readable-codebase PLAN,
README, and RD-19 ticket. Fix two upward-relative links inside the moved
WP files (WP-68, WP-69) that gained a directory level and would otherwise
break. Repoint `.prettierignore`'s two agent-prompt exclusions to their
new path, so prettier keeps leaving those files' exact wording alone.

Mark RD-30 done and check off its acceptance criteria; flip its README
row to done.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-08 23:00:38 +02:00

256 lines
9.9 KiB
JavaScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env node
// Baseline scanner for 00-baseline.md: duplication % and C# cyclomatic complexity.
// Both are the same crude text scan, so they share one file.
//
// ponytail: line-window hashing, not token-based like jscpd, and regex method
// detection that does not understand C# expression-bodied members or nested
// lambdas. Deterministic and zero-install, which is what a *baseline* needs.
// Upgrade path: jscpd for duplication, a Roslyn analyzer for C# complexity —
// only if a ticket's before/after needs more precision than "did it move".
//
// Usage: node baseline-scan.mjs [--dup] [--complexity] (default: both)
import { readFileSync, readdirSync, statSync } from 'node:fs';
import { createHash } from 'node:crypto';
import { join, relative } from 'node:path';
const ROOT = new URL('../../../../..', import.meta.url).pathname.replace(/\/$/, '');
const SKIP_DIR =
/(^|\/)(node_modules|dist|coverage|bin|obj|\.angular|\.git|storybook-static.*|TestResults|public|openzaak)$/;
const SKIP_FILE = /(api-client\.ts|\.Designer\.cs|AppDbContextModelSnapshot\.cs)$/;
const SCAN = ['apps', 'libs', 'backend/src', 'backend/tests', 'e2e'];
const WINDOW = 6; // duplicate = >= 6 consecutive normalized lines seen elsewhere
function walk(dir, out = []) {
for (const e of readdirSync(dir)) {
const p = join(dir, e);
if (SKIP_DIR.test(p)) continue;
if (statSync(p).isDirectory()) walk(p, out);
else if (/\.(ts|cs)$/.test(p) && !SKIP_FILE.test(p)) out.push(p);
}
return out;
}
// Module attribution: the unit the baseline table reports on.
function moduleOf(rel) {
let m;
if ((m = rel.match(/^apps\/(ssp|behandelportal)\/src\/app\/([^/]+)\//)))
return `${m[1] === 'ssp' ? 'ssp' : 'bhp'}/${m[2]}`;
if (rel.startsWith('apps/ssp/')) return 'ssp/root';
if (rel.startsWith('apps/behandelportal/')) return 'bhp/root';
if ((m = rel.match(/^libs\/(shared|beheer)\/src\/([^/]+)\//))) return `libs/${m[1]}/${m[2]}`;
if (rel.startsWith('libs/')) return rel.split('/').slice(0, 2).join('/');
if (rel === 'backend/src/BigRegister.Api/Program.cs') return 'backend/Program.cs';
if ((m = rel.match(/^backend\/src\/BigRegister\.Api\/([^/]+)\//))) return `backend/${m[1]}`;
if (rel.startsWith('backend/tests/')) return 'backend/tests';
if (rel.startsWith('e2e/')) return 'e2e';
return 'other';
}
const files = SCAN.flatMap((d) => {
try {
return walk(join(ROOT, d));
} catch {
return [];
}
}).map((p) => ({ abs: p, rel: relative(ROOT, p) }));
// --- Duplication ------------------------------------------------------------
// Normalize away formatting/comments, hash every WINDOW-line run, and mark a run
// duplicated when its hash occurs in more than one place.
function duplication() {
const norm = new Map(); // rel -> [{line, text}]
for (const f of files) {
const kept = [];
readFileSync(f.abs, 'utf8')
.split('\n')
.forEach((raw, i) => {
const t = raw
.replace(/\/\/.*$/, '')
.replace(/\s+/g, ' ')
.trim();
if (t.length > 3) kept.push({ line: i + 1, text: t });
});
norm.set(f.rel, kept);
}
const seen = new Map(); // hash -> [{rel, idx}]
for (const [rel, lines] of norm)
for (let i = 0; i + WINDOW <= lines.length; i++) {
const h = createHash('sha1')
.update(
lines
.slice(i, i + WINDOW)
.map((l) => l.text)
.join('\n'),
)
.digest('hex');
(seen.get(h) ?? seen.set(h, []).get(h)).push({ rel, idx: i });
}
const dupLines = new Map(); // rel -> Set(line)
const clones = new Map(); // "relA|relB" -> count
for (const occ of seen.values()) {
if (occ.length < 2) continue;
for (const { rel, idx } of occ) {
const set = dupLines.get(rel) ?? dupLines.set(rel, new Set()).get(rel);
for (let k = 0; k < WINDOW; k++) set.add(norm.get(rel)[idx + k].line);
}
const pair = [...new Set(occ.map((o) => o.rel))].sort();
if (pair.length > 1) {
const key = pair.slice(0, 2).join(' | ');
clones.set(key, (clones.get(key) ?? 0) + 1);
}
}
const per = new Map(); // module -> {total, dup}
for (const [rel, lines] of norm) {
const mod = moduleOf(rel);
const e = per.get(mod) ?? per.set(mod, { total: 0, dup: 0 }).get(mod);
e.total += lines.length;
e.dup += dupLines.get(rel)?.size ?? 0;
}
console.log('## Duplication (normalized lines, window=' + WINDOW + ')\n');
console.log('| Module | Sig. lines | Duplicated | % |');
console.log('|---|---:|---:|---:|');
let T = 0,
D = 0;
for (const [mod, e] of [...per].sort((a, b) => b[1].dup / b[1].total - a[1].dup / a[1].total)) {
T += e.total;
D += e.dup;
console.log(`| ${mod} | ${e.total} | ${e.dup} | ${((100 * e.dup) / e.total).toFixed(1)}% |`);
}
console.log(`| **TOTAL** | **${T}** | **${D}** | **${((100 * D) / T).toFixed(1)}%** |`);
console.log('\n### Top clone pairs (distinct duplicated windows)\n');
for (const [pair, n] of [...clones].sort((a, b) => b[1] - a[1]).slice(0, 15))
console.log(`- ${n} ×${pair}`);
}
// --- C# cyclomatic complexity (approximate) ---------------------------------
// Two numbers per file. FILE CC (sum of decision points) is exact enough to trust.
// Per-METHOD CC uses depth-aware regex detection and is the approximate one.
const BRANCH =
/\bif\s*\(|\bwhile\s*\(|\bfor\s*\(|\bforeach\s*\(|\bcase\s+|\bcatch\s*[({]|\?\?|&&|\|\||\?\.|\bwhen\s+/g;
const TYPE_DECL =
/^\s*(?:\[[^\]]*\]\s*)*(?:public|private|internal|protected|static|sealed|abstract|partial|file|\s)*\b(?:class|record|struct|interface|enum|namespace)\b/;
const NOT_A_CALL =
/^(if|for|foreach|while|switch|catch|using|lock|return|throw|new|await|yield|else|do|fixed|checked)$/;
// A member: optional attrs/modifiers, a return type, a name, then `(`.
const SIG =
/^\s*(?:\[[^\]]*\]\s*)*(?:(?:public|private|internal|protected|static|async|override|virtual|sealed|partial|extern|new|unsafe)\s+)*[\w<>,\[\]?.]+\s+(\w+)\s*(?:<[^>()]*>)?\s*\(/;
function branchesIn(line) {
return (line.replace(/\/\/.*$/, '').match(BRANCH) ?? []).length;
}
function csComplexity() {
const rows = [];
const fileCc = [];
for (const f of files.filter((f) => f.rel.endsWith('.cs'))) {
const lines = readFileSync(f.abs, 'utf8').split('\n');
let depth = 0,
cur = null,
total = 1,
code = 0;
for (let i = 0; i < lines.length; i++) {
const l = lines[i];
const bare = l.replace(/\/\/.*$/, '');
const b = branchesIn(l);
total += b;
if (bare.trim().length > 1 && !/^\s*(\/\/|\/\*|\*)/.test(l)) code++;
const opens = (bare.match(/{/g) ?? []).length;
const closes = (bare.match(/}/g) ?? []).length;
if (cur) {
cur.lines++;
cur.cc += b;
// Expression-bodied member: `=> expr;` with no block of its own.
if (cur.depth === null && /=>/.test(bare)) {
if (/;\s*$/.test(bare) && opens === closes) {
rows.push(cur);
cur = null;
} else if (opens > closes) cur.depth = depth;
} else if (cur.depth === null && opens > closes) cur.depth = depth;
else if (cur.depth !== null && depth + opens - closes <= cur.depth) {
rows.push(cur);
cur = null;
}
} else if (!TYPE_DECL.test(l)) {
const m = SIG.exec(bare);
// `Name(` must not be a call/keyword, and the line must not be a statement.
if (
m &&
!NOT_A_CALL.test(m[1]) &&
!/^\s*(var|return|await)\b/.test(bare) &&
!/;\s*$/.test(bare.replace(/=>.*/, ''))
)
cur = {
file: f.rel,
name: m[1],
line: i + 1,
cc: 1 + b,
lines: 1,
depth: opens > closes ? depth : null,
};
}
depth += opens - closes;
}
if (cur) rows.push(cur);
fileCc.push({ file: f.rel, cc: total, code });
}
const p = (arr, q) =>
arr.slice().sort((a, b) => a - b)[Math.min(arr.length - 1, Math.floor(q * arr.length))] ?? 0;
const perFile = new Map();
for (const r of fileCc) {
const mod = moduleOf(r.file);
(perFile.get(mod) ?? perFile.set(mod, []).get(mod)).push(r);
}
console.log('\n\n## C# complexity — per module\n');
console.log(
'| Module | Files | Σ file CC | max file CC | Methods | max method CC | p90 method CC | CC>10 |',
);
console.log('|---|---:|---:|---:|---:|---:|---:|---:|');
const perMethod = new Map();
for (const r of rows) {
const mod = moduleOf(r.file);
(perMethod.get(mod) ?? perMethod.set(mod, []).get(mod)).push(r);
}
for (const [mod, fs] of [...perFile].sort(
(a, b) => Math.max(...b[1].map((r) => r.cc)) - Math.max(...a[1].map((r) => r.cc)),
)) {
const ms = perMethod.get(mod) ?? [];
const ccs = ms.map((r) => r.cc);
console.log(
`| ${mod} | ${fs.length} | ${fs.reduce((a, r) => a + r.cc, 0)} | ${Math.max(...fs.map((r) => r.cc))} | ` +
`${ms.length} | ${ccs.length ? Math.max(...ccs) : 0} | ${p(ccs, 0.9)} | ${ms.filter((r) => r.cc > 10).length} |`,
);
}
console.log('\n### C# files by CC (top 12)\n');
console.log('| File CC | Code lines | File |');
console.log('|---:|---:|---|');
for (const r of fileCc.sort((a, b) => b.cc - a.cc).slice(0, 12))
console.log(`| ${r.cc} | ${r.code} | ${r.file} |`);
console.log('\n### C# methods over CC 10 (approximate detection)\n');
console.log('| CC | Lines | Method | Location |');
console.log('|---:|---:|---|---|');
for (const r of rows.filter((r) => r.cc > 10).sort((a, b) => b.cc - a.cc))
console.log(`| ${r.cc} | ${r.lines} | \`${r.name}\` | ${r.file}:${r.line} |`);
const all = rows.map((r) => r.lines);
console.log(
`\nMethod-length distribution (n=${rows.length}): p50 ${p(all, 0.5)}, p90 ${p(all, 0.9)}, p99 ${p(all, 0.99)}, max ${Math.max(...all)}`,
);
}
const args = process.argv.slice(2);
const all = args.length === 0;
if (all || args.includes('--dup')) duplication();
if (all || args.includes('--complexity')) csComplexity();