diff --git a/scripts/measure_duplication.py b/scripts/measure_duplication.py new file mode 100644 index 00000000..34c51c00 --- /dev/null +++ b/scripts/measure_duplication.py @@ -0,0 +1,272 @@ +#!/usr/bin/env python3 +"""Measure near-verbatim duplication in a tree, and what a change added to it. + +WHY THIS EXISTS + +A DRY audit made of several passes ends with a question no single pass can +answer: did the audit as a whole reduce duplication, and did it create any? The +second half is the one that matters. A pass that shortens two statements can +leave them byte-identical to a third, and the new copy is invisible to that +pass's own scan because it did not exist when the scan ran (#4745: a retro over +a 17-pass audit found exactly this, and folded it into a view). This script is +the close-out measure the DRY Pass process names, so the measure is a tool and +not something each audit improvises. + +WHAT IT MEASURES + +Each file is reduced to its significant lines: blank lines, comment-only lines +and lines of bare punctuation (`}`, `});`, `],`) are dropped; whitespace is +collapsed; string literals are folded to "S", so two copies that differ only in +a message or a key still match. A *window* is N consecutive significant lines +(default 6). A window whose text occurs at two or more places in the same group +is duplicated, and every line it covers is a duplicated line. + +Per group it reports significant lines, duplicated windows (distinct texts), +duplicated lines and their share. With `--before REV` it measures that revision +too and lists the duplicated windows that exist ONLY after — not duplicated +before, either because the text was absent or because it occurred once. Those +are grouped by the set of files they appear in, which is the list to read. + +WHAT IT DOES NOT SEE + +Structural or semantic copies — two components with the same shape and +different names, the same predicate written two ways. It catches near-verbatim +text only, so a fold of structural copies shows here less than it counts. Read +the numbers as a floor and the after-only list as the finding. + +USAGE + + measure_duplication.py [--repo DIR] [--before REV] [--after REV] + [--group NAME=GLOB[,GLOB...]]... [--exclude GLOB]... + [--window N] [--show N] [--json] + +The after tree is the working tree's tracked files unless `--after REV` names a +revision. Revisions are read through `git archive`, so nothing is checked out. +Globs are fnmatch patterns over repo-relative paths, where `*` crosses `/`. +Groups are tried in order and the first match claims a file, so list +`--group 'go tests=*_test.go'` before `--group 'go=*.go'`. With no `--group`, +files are grouped by extension among common source types. + +Stdlib only, so it runs from a scratch copy in any project: +`get_snippet` the recorded snippet, write its code to a file, run it there. +""" +from __future__ import annotations + +import argparse +import fnmatch +import hashlib +import io +import json +import re +import subprocess +import sys +import tarfile +from collections import defaultdict +from pathlib import Path + +DEFAULT_EXTS = { + "c", "cc", "cpp", "cs", "css", "dart", "go", "h", "hpp", "java", "js", + "jsx", "kt", "kts", "php", "py", "rb", "rs", "scss", "sh", "sql", + "svelte", "swift", "ts", "tsx", "vue", +} + +_COMMENT = re.compile(r"^(//|/\*|\*|--|