CI & Build / Plugin hooks (push) Successful in 13s
CI & Build / Python lint (push) Successful in 3s
CI & Build / TypeScript typecheck (push) Successful in 57s
CI & Build / integration (push) Successful in 2m0s
CI & Build / Python tests (push) Successful in 2m40s
CI & Build / Build & push image (push) Successful in 25s
A multi-pass DRY audit could not answer whether it created duplication: a pass's own shortening can leave two statements identical to a third, and no per-pass scan sees a copy that did not exist when it ran. The Librarian retrospective improvised this measure in a scratchpad; this is it as a tool the DRY Pass process can name. 6-line windows over significant lines (comments, blanks and bare punctuation dropped, strings folded), grouped by glob or extension. Revisions are read through git archive, so nothing is checked out. Stdlib only, so any project can run a scratch copy. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
129 lines
4.5 KiB
Python
129 lines
4.5 KiB
Python
"""The DRY close-out measure (#4745).
|
|
|
|
Stdlib-only and kept in scripts/ so any project can run a scratch copy of it,
|
|
so these tests import it by path rather than as a package.
|
|
|
|
What matters is the after-only list: it is the part of the measure that finds
|
|
something a pass could not, so it must name a copy the change CREATED and stay
|
|
quiet about one that was already there. A list that repeats old duplication
|
|
every time is one nobody reads.
|
|
"""
|
|
import importlib.util
|
|
import pathlib
|
|
import shutil
|
|
import subprocess
|
|
|
|
import pytest
|
|
|
|
_PATH = pathlib.Path(__file__).resolve().parents[1] / "scripts" / "measure_duplication.py"
|
|
_spec = importlib.util.spec_from_file_location("measure_duplication", _PATH)
|
|
dup = importlib.util.module_from_spec(_spec)
|
|
_spec.loader.exec_module(dup)
|
|
|
|
|
|
BODY = """
|
|
def load(user_id, note_id):
|
|
# a comment the measure ignores
|
|
if not allowed(user_id, note_id):
|
|
raise ValueError("note {} not found".format(note_id))
|
|
with session() as s:
|
|
row = s.get(Row, note_id)
|
|
if row is None:
|
|
raise ValueError("not a row")
|
|
return row
|
|
"""
|
|
|
|
|
|
def _files(**texts):
|
|
return [(name, text.encode()) for name, text in texts.items()]
|
|
|
|
|
|
def test_comments_blanks_and_bare_punctuation_are_not_code():
|
|
lines = dup.significant_lines('x = 1\n\n// note\n# note\n });\n-- sql note\ny = "two"\n')
|
|
assert [t for _, t in lines] == ["x = 1", 'y = "S"']
|
|
assert [n for n, _ in lines] == [1, 7]
|
|
|
|
|
|
def test_a_preprocessor_line_is_code_and_a_hash_comment_is_not():
|
|
lines = dup.significant_lines("#include <x.h>\n# a comment\n#!/bin/sh\n")
|
|
assert [t for _, t in lines] == ["#include <x.h>"]
|
|
|
|
|
|
def test_copies_that_differ_only_in_their_strings_are_one_copy():
|
|
other = BODY.replace('"not a row"', '"no such row"')
|
|
result = dup.measure(_files(**{"a.py": BODY, "b.py": other}), [], [], window=6)
|
|
py = result["py"]
|
|
assert py["dup_windows"] > 0
|
|
assert py["dup_lines"] == py["lines"]
|
|
assert py["share"] == 1.0
|
|
|
|
|
|
def test_unrelated_files_measure_zero():
|
|
other = "\n".join(f"v{i} = compute({i})" for i in range(20))
|
|
result = dup.measure(_files(**{"a.py": BODY, "b.py": other}), [], [], window=6)
|
|
assert result["py"]["dup_lines"] == 0
|
|
|
|
|
|
def test_the_first_matching_group_claims_a_file():
|
|
groups = [("tests", ["tests/*"]), ("py", ["*.py"])]
|
|
assert dup.group_of("tests/test_x.py", groups) == "tests"
|
|
assert dup.group_of("src/x.py", groups) == "py"
|
|
assert dup.group_of("README.md", groups) is None
|
|
|
|
|
|
def test_without_groups_only_source_extensions_are_measured():
|
|
assert dup.group_of("src/x.go", []) == "go"
|
|
assert dup.group_of("docs/x.md", []) is None
|
|
assert dup.group_of("Makefile", []) is None
|
|
|
|
|
|
def test_excluded_paths_are_not_measured():
|
|
result = dup.measure(
|
|
_files(**{"a.py": BODY, "gen/b.py": BODY}), [], ["gen/*"], window=6,
|
|
)
|
|
assert result["py"]["dup_lines"] == 0
|
|
|
|
|
|
def test_a_copy_the_change_created_is_listed_after_only():
|
|
before = dup.measure(_files(**{"a.py": BODY, "c.py": "z = 1\n"}), [], [], window=6)
|
|
after = dup.measure(_files(**{"a.py": BODY, "c.py": BODY}), [], [], window=6)
|
|
fresh = dup.only_after(before, after)
|
|
assert [e["files"] for e in fresh["py"]] == [["a.py", "c.py"]]
|
|
assert fresh["py"][0]["at"] == ["a.py:2", "c.py:2"]
|
|
|
|
|
|
def test_a_copy_that_was_already_there_is_not_listed():
|
|
both = _files(**{"a.py": BODY, "b.py": BODY})
|
|
before = dup.measure(both, [], [], window=6)
|
|
after = dup.measure(both, [], [], window=6)
|
|
assert dup.only_after(before, after) == {}
|
|
|
|
|
|
def test_the_report_shows_before_and_after_side_by_side():
|
|
before = dup.measure(_files(**{"a.py": BODY}), [], [], window=6)
|
|
after = dup.measure(_files(**{"a.py": BODY, "b.py": BODY}), [], [], window=6)
|
|
text = dup.report(before, after, dup.only_after(before, after), show=5)
|
|
assert "0.0% → 100.0%" in text
|
|
assert "a.py:2, b.py:2" in text
|
|
|
|
|
|
@pytest.mark.skipif(shutil.which("git") is None, reason="needs git")
|
|
def test_a_revision_is_read_without_a_checkout(tmp_path, capsys):
|
|
def git(*args):
|
|
subprocess.run(
|
|
["git", "-C", str(tmp_path), "-c", "user.name=t", "-c", "user.email=t@t", *args],
|
|
check=True, capture_output=True,
|
|
)
|
|
|
|
git("init", "-q")
|
|
(tmp_path / "a.py").write_text(BODY)
|
|
git("add", "a.py")
|
|
git("commit", "-qm", "one copy")
|
|
(tmp_path / "b.py").write_text(BODY)
|
|
git("add", "b.py")
|
|
|
|
assert dup.main(["--repo", str(tmp_path), "--before", "HEAD"]) == 0
|
|
out = capsys.readouterr().out
|
|
assert "Duplicated only after: 1 file set(s)" in out
|
|
assert "a.py:2, b.py:2" in out
|