diff --git a/plugin/.claude-plugin/plugin.json b/plugin/.claude-plugin/plugin.json index fe082589..aeb17858 100644 --- a/plugin/.claude-plugin/plugin.json +++ b/plugin/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "scribe", "description": "Scribe for Claude Code: connects the scribe MCP server, adds the hooks that deliver live project state and relevant records at the right moment, ships the shared client-neutral Scribe skills (using-scribe, writing-plans, reporting-back, systematic-debugging, verification, brainstorming, reusing-code, shape-accounting, family-canon), and syncs your saved Scribe Processes as skills (/scribe:sync).", - "version": "2026.10.09.1845", + "version": "2026.10.09.1850", "author": { "name": "Bryan Van Deusen" }, diff --git a/plugin/hooks/scribe_defs.sh b/plugin/hooks/scribe_defs.sh index 9019adb1..af43fb5e 100644 --- a/plugin/hooks/scribe_defs.sh +++ b/plugin/hooks/scribe_defs.sh @@ -205,8 +205,26 @@ scribe_json_len() { } # Raw JSON string bodies on stdin → text. One line in, one value out. +# +# A `\uXXXX` ESCAPE IS ENCODED AS UTF-8 HERE, BYTE BY BYTE, UNDER LC_ALL=C. +# The server escapes every non-ASCII character this way (the JSON default), +# and `sprintf("%c", n)` for n > 127 means a different thing in every awk and +# locale: gawk in a UTF-8 locale writes the character, gawk in the C locale +# and mawk write the single byte n % 256. So "·" (U+00B7) arrived as a lone +# 0xB7 and "—" (U+2014) as 0x14 wherever the hook ran without a UTF-8 locale — +# found by a test that ran the hook with a bare environment (milestone 500). +# Under LC_ALL=C every awk writes exactly the byte asked for, so the encoding +# below is the only one that happens. scribe_json_unescape() { - awk ' + LC_ALL=C awk ' + function utf8(n) { + if (n < 128) return sprintf("%c", n) + if (n < 2048) return sprintf("%c%c", 192 + int(n / 64), 128 + n % 64) + if (n < 65536) return sprintf("%c%c%c", 224 + int(n / 4096), + 128 + int(n / 64) % 64, 128 + n % 64) + return sprintf("%c%c%c%c", 240 + int(n / 262144), 128 + int(n / 4096) % 64, + 128 + int(n / 64) % 64, 128 + n % 64) + } function hex4(h, i, c, d, v) { v = 0 for (i = 1; i <= 4; i++) { @@ -246,7 +264,7 @@ scribe_json_unescape() { i += 6 } } - o = o sprintf("%c", hi) + o = o utf8(hi) } else o = o d } diff --git a/tests/test_hook_json_reader.py b/tests/test_hook_json_reader.py index c94b16c7..c5b09387 100644 --- a/tests/test_hook_json_reader.py +++ b/tests/test_hook_json_reader.py @@ -21,6 +21,7 @@ guard here can fail (rule 167). from __future__ import annotations import json +import os import re import subprocess from pathlib import Path @@ -156,6 +157,33 @@ def test_a_value_survives_the_round_trip_the_hooks_actually_make(value): assert got.rstrip("\n") == value.rstrip("\n") +ESCAPED = ["a · b", "an em dash — here", "a face 😀 here", "plain ascii"] + + +@pytest.mark.parametrize("locale", [None, "C", "C.UTF-8"]) +@pytest.mark.parametrize("value", ESCAPED) +def test_an_escaped_character_decodes_to_utf8_in_any_locale(value, locale): + """The server's JSON escapes every non-ASCII character (`\\u00b7`), and a + hook may run with no locale at all. The round trip above sends raw UTF-8 + and never reaches the escape path; this one sends what the server sends, + from an environment with nothing but PATH.""" + need_tools("bash", "awk") + env = {"PATH": os.environ["PATH"]} + if locale: + env["LC_ALL"] = locale + body = json.dumps(value)[1:-1] + r = subprocess.run(["bash", "-c", f'. "{DEFS}"; scribe_json_unescape'], + input=body.encode(), capture_output=True, env=env) + assert r.returncode == 0, r.stderr.decode() + assert r.stdout == (value + "\n").encode() + + +def test_the_utf8_guard_can_fail(): + """A lone byte where the encoded character belongs is what the old + decoder wrote; the comparison above has to tell them apart.""" + assert "·\n".encode() != b"\xb7\n" + + # -------------------------------------------------------------------------- # Reading: the shell side.