From 890c93e126eec700ca66558023d74d49958c0405 Mon Sep 17 00:00:00 2001 From: Bryan Van Deusen Date: Tue, 6 Oct 2026 21:24:05 -0400 Subject: [PATCH] feat(scripts): measure_duplication - the DRY close-out measure: before/after duplicate share, and the copies that exist only after (#4745) A multi-pass DRY audit could not answer whether it created duplication: a pass's own shortening can leave two statements identical to a third, and no per-pass scan sees a copy that did not exist when it ran. The Librarian retrospective improvised this measure in a scratchpad; this is it as a tool the DRY Pass process can name. 6-line windows over significant lines (comments, blanks and bare punctuation dropped, strings folded), grouped by glob or extension. Revisions are read through git archive, so nothing is checked out. Stdlib only, so any project can run a scratch copy. Co-Authored-By: Claude Opus 5.5 --- scripts/measure_duplication.py | 272 ++++++++++++++++++++++++++++++ tests/test_measure_duplication.py | 128 ++++++++++++++ 2 files changed, 400 insertions(+) create mode 100644 scripts/measure_duplication.py create mode 100644 tests/test_measure_duplication.py diff --git a/scripts/measure_duplication.py b/scripts/measure_duplication.py new file mode 100644 index 00000000..34c51c00 --- /dev/null +++ b/scripts/measure_duplication.py @@ -0,0 +1,272 @@ +#!/usr/bin/env python3 +"""Measure near-verbatim duplication in a tree, and what a change added to it. + +WHY THIS EXISTS + +A DRY audit made of several passes ends with a question no single pass can +answer: did the audit as a whole reduce duplication, and did it create any? The +second half is the one that matters. A pass that shortens two statements can +leave them byte-identical to a third, and the new copy is invisible to that +pass's own scan because it did not exist when the scan ran (#4745: a retro over +a 17-pass audit found exactly this, and folded it into a view). This script is +the close-out measure the DRY Pass process names, so the measure is a tool and +not something each audit improvises. + +WHAT IT MEASURES + +Each file is reduced to its significant lines: blank lines, comment-only lines +and lines of bare punctuation (`}`, `});`, `],`) are dropped; whitespace is +collapsed; string literals are folded to "S", so two copies that differ only in +a message or a key still match. A *window* is N consecutive significant lines +(default 6). A window whose text occurs at two or more places in the same group +is duplicated, and every line it covers is a duplicated line. + +Per group it reports significant lines, duplicated windows (distinct texts), +duplicated lines and their share. With `--before REV` it measures that revision +too and lists the duplicated windows that exist ONLY after — not duplicated +before, either because the text was absent or because it occurred once. Those +are grouped by the set of files they appear in, which is the list to read. + +WHAT IT DOES NOT SEE + +Structural or semantic copies — two components with the same shape and +different names, the same predicate written two ways. It catches near-verbatim +text only, so a fold of structural copies shows here less than it counts. Read +the numbers as a floor and the after-only list as the finding. + +USAGE + + measure_duplication.py [--repo DIR] [--before REV] [--after REV] + [--group NAME=GLOB[,GLOB...]]... [--exclude GLOB]... + [--window N] [--show N] [--json] + +The after tree is the working tree's tracked files unless `--after REV` names a +revision. Revisions are read through `git archive`, so nothing is checked out. +Globs are fnmatch patterns over repo-relative paths, where `*` crosses `/`. +Groups are tried in order and the first match claims a file, so list +`--group 'go tests=*_test.go'` before `--group 'go=*.go'`. With no `--group`, +files are grouped by extension among common source types. + +Stdlib only, so it runs from a scratch copy in any project: +`get_snippet` the recorded snippet, write its code to a file, run it there. +""" +from __future__ import annotations + +import argparse +import fnmatch +import hashlib +import io +import json +import re +import subprocess +import sys +import tarfile +from collections import defaultdict +from pathlib import Path + +DEFAULT_EXTS = { + "c", "cc", "cpp", "cs", "css", "dart", "go", "h", "hpp", "java", "js", + "jsx", "kt", "kts", "php", "py", "rb", "rs", "scss", "sh", "sql", + "svelte", "swift", "ts", "tsx", "vue", +} + +_COMMENT = re.compile(r"^(//|/\*|\*|--|