"""The DRY close-out measure (#4745). Stdlib-only and kept in scripts/ so any project can run a scratch copy of it, so these tests import it by path rather than as a package. What matters is the after-only list: it is the part of the measure that finds something a pass could not, so it must name a copy the change CREATED and stay quiet about one that was already there. A list that repeats old duplication every time is one nobody reads. """ import importlib.util import pathlib import shutil import subprocess import pytest _PATH = pathlib.Path(__file__).resolve().parents[1] / "scripts" / "measure_duplication.py" _spec = importlib.util.spec_from_file_location("measure_duplication", _PATH) dup = importlib.util.module_from_spec(_spec) _spec.loader.exec_module(dup) BODY = """ def load(user_id, note_id): # a comment the measure ignores if not allowed(user_id, note_id): raise ValueError("note {} not found".format(note_id)) with session() as s: row = s.get(Row, note_id) if row is None: raise ValueError("not a row") return row """ def _files(**texts): return [(name, text.encode()) for name, text in texts.items()] def test_comments_blanks_and_bare_punctuation_are_not_code(): lines = dup.significant_lines('x = 1\n\n// note\n# note\n });\n-- sql note\ny = "two"\n') assert [t for _, t in lines] == ["x = 1", 'y = "S"'] assert [n for n, _ in lines] == [1, 7] def test_a_preprocessor_line_is_code_and_a_hash_comment_is_not(): lines = dup.significant_lines("#include \n# a comment\n#!/bin/sh\n") assert [t for _, t in lines] == ["#include "] def test_copies_that_differ_only_in_their_strings_are_one_copy(): other = BODY.replace('"not a row"', '"no such row"') result = dup.measure(_files(**{"a.py": BODY, "b.py": other}), [], [], window=6) py = result["py"] assert py["dup_windows"] > 0 assert py["dup_lines"] == py["lines"] assert py["share"] == 1.0 def test_unrelated_files_measure_zero(): other = "\n".join(f"v{i} = compute({i})" for i in range(20)) result = dup.measure(_files(**{"a.py": BODY, "b.py": other}), [], [], window=6) assert result["py"]["dup_lines"] == 0 def test_the_first_matching_group_claims_a_file(): groups = [("tests", ["tests/*"]), ("py", ["*.py"])] assert dup.group_of("tests/test_x.py", groups) == "tests" assert dup.group_of("src/x.py", groups) == "py" assert dup.group_of("README.md", groups) is None def test_without_groups_only_source_extensions_are_measured(): assert dup.group_of("src/x.go", []) == "go" assert dup.group_of("docs/x.md", []) is None assert dup.group_of("Makefile", []) is None def test_excluded_paths_are_not_measured(): result = dup.measure( _files(**{"a.py": BODY, "gen/b.py": BODY}), [], ["gen/*"], window=6, ) assert result["py"]["dup_lines"] == 0 def test_a_copy_the_change_created_is_listed_after_only(): before = dup.measure(_files(**{"a.py": BODY, "c.py": "z = 1\n"}), [], [], window=6) after = dup.measure(_files(**{"a.py": BODY, "c.py": BODY}), [], [], window=6) fresh = dup.only_after(before, after) assert [e["files"] for e in fresh["py"]] == [["a.py", "c.py"]] assert fresh["py"][0]["at"] == ["a.py:2", "c.py:2"] def test_a_copy_that_was_already_there_is_not_listed(): both = _files(**{"a.py": BODY, "b.py": BODY}) before = dup.measure(both, [], [], window=6) after = dup.measure(both, [], [], window=6) assert dup.only_after(before, after) == {} def test_the_report_shows_before_and_after_side_by_side(): before = dup.measure(_files(**{"a.py": BODY}), [], [], window=6) after = dup.measure(_files(**{"a.py": BODY, "b.py": BODY}), [], [], window=6) text = dup.report(before, after, dup.only_after(before, after), show=5) assert "0.0% → 100.0%" in text assert "a.py:2, b.py:2" in text @pytest.mark.skipif(shutil.which("git") is None, reason="needs git") def test_a_revision_is_read_without_a_checkout(tmp_path, capsys): def git(*args): subprocess.run( ["git", "-C", str(tmp_path), "-c", "user.name=t", "-c", "user.email=t@t", *args], check=True, capture_output=True, ) git("init", "-q") (tmp_path / "a.py").write_text(BODY) git("add", "a.py") git("commit", "-qm", "one copy") (tmp_path / "b.py").write_text(BODY) git("add", "b.py") assert dup.main(["--repo", str(tmp_path), "--before", "HEAD"]) == 0 out = capsys.readouterr().out assert "Duplicated only after: 1 file set(s)" in out assert "a.py:2, b.py:2" in out