fix(500): the hooks decode a \u escape to UTF-8 in any locale and any awk
CI & Build / Python lint (push) Successful in 2s
CI & Build / Plugin hooks (push) Successful in 11s
CI & Build / TypeScript typecheck (push) Successful in 52s
CI & Build / integration (push) Successful in 1m1s
CI & Build / Python tests (push) Successful in 1m57s
CI & Build / Build & push image (push) Successful in 25s
CI & Build / Python lint (push) Successful in 2s
CI & Build / Plugin hooks (push) Successful in 11s
CI & Build / TypeScript typecheck (push) Successful in 52s
CI & Build / integration (push) Successful in 1m1s
CI & Build / Python tests (push) Successful in 1m57s
CI & Build / Build & push image (push) Successful in 25s
scribe_json_unescape wrote each escaped code point with sprintf("%c", n),
which writes the character only under gawk in a UTF-8 locale; gawk in the C
locale and mawk in any locale write the single byte n % 256. The server
escapes every non-ASCII character, so "·" reached the session as 0xB7, "—"
as 0x14 and an emoji as a NUL wherever a hook ran without a UTF-8 locale.
The decoder now encodes UTF-8 itself under LC_ALL=C. A new test sends what
the server sends from a bare environment, in three locales (#5495).
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -1,7 +1,7 @@
|
|||||||
{
|
{
|
||||||
"name": "scribe",
|
"name": "scribe",
|
||||||
"description": "Scribe for Claude Code: connects the scribe MCP server, adds the hooks that deliver live project state and relevant records at the right moment, ships the shared client-neutral Scribe skills (using-scribe, writing-plans, reporting-back, systematic-debugging, verification, brainstorming, reusing-code, shape-accounting, family-canon), and syncs your saved Scribe Processes as skills (/scribe:sync).",
|
"description": "Scribe for Claude Code: connects the scribe MCP server, adds the hooks that deliver live project state and relevant records at the right moment, ships the shared client-neutral Scribe skills (using-scribe, writing-plans, reporting-back, systematic-debugging, verification, brainstorming, reusing-code, shape-accounting, family-canon), and syncs your saved Scribe Processes as skills (/scribe:sync).",
|
||||||
"version": "2026.10.09.1845",
|
"version": "2026.10.09.1850",
|
||||||
"author": {
|
"author": {
|
||||||
"name": "Bryan Van Deusen"
|
"name": "Bryan Van Deusen"
|
||||||
},
|
},
|
||||||
|
|||||||
@@ -205,8 +205,26 @@ scribe_json_len() {
|
|||||||
}
|
}
|
||||||
|
|
||||||
# Raw JSON string bodies on stdin → text. One line in, one value out.
|
# Raw JSON string bodies on stdin → text. One line in, one value out.
|
||||||
|
#
|
||||||
|
# A `\uXXXX` ESCAPE IS ENCODED AS UTF-8 HERE, BYTE BY BYTE, UNDER LC_ALL=C.
|
||||||
|
# The server escapes every non-ASCII character this way (the JSON default),
|
||||||
|
# and `sprintf("%c", n)` for n > 127 means a different thing in every awk and
|
||||||
|
# locale: gawk in a UTF-8 locale writes the character, gawk in the C locale
|
||||||
|
# and mawk write the single byte n % 256. So "·" (U+00B7) arrived as a lone
|
||||||
|
# 0xB7 and "—" (U+2014) as 0x14 wherever the hook ran without a UTF-8 locale —
|
||||||
|
# found by a test that ran the hook with a bare environment (milestone 500).
|
||||||
|
# Under LC_ALL=C every awk writes exactly the byte asked for, so the encoding
|
||||||
|
# below is the only one that happens.
|
||||||
scribe_json_unescape() {
|
scribe_json_unescape() {
|
||||||
awk '
|
LC_ALL=C awk '
|
||||||
|
function utf8(n) {
|
||||||
|
if (n < 128) return sprintf("%c", n)
|
||||||
|
if (n < 2048) return sprintf("%c%c", 192 + int(n / 64), 128 + n % 64)
|
||||||
|
if (n < 65536) return sprintf("%c%c%c", 224 + int(n / 4096),
|
||||||
|
128 + int(n / 64) % 64, 128 + n % 64)
|
||||||
|
return sprintf("%c%c%c%c", 240 + int(n / 262144), 128 + int(n / 4096) % 64,
|
||||||
|
128 + int(n / 64) % 64, 128 + n % 64)
|
||||||
|
}
|
||||||
function hex4(h, i, c, d, v) {
|
function hex4(h, i, c, d, v) {
|
||||||
v = 0
|
v = 0
|
||||||
for (i = 1; i <= 4; i++) {
|
for (i = 1; i <= 4; i++) {
|
||||||
@@ -246,7 +264,7 @@ scribe_json_unescape() {
|
|||||||
i += 6
|
i += 6
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
o = o sprintf("%c", hi)
|
o = o utf8(hi)
|
||||||
}
|
}
|
||||||
else o = o d
|
else o = o d
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -21,6 +21,7 @@ guard here can fail (rule 167).
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import json
|
import json
|
||||||
|
import os
|
||||||
import re
|
import re
|
||||||
import subprocess
|
import subprocess
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
@@ -156,6 +157,33 @@ def test_a_value_survives_the_round_trip_the_hooks_actually_make(value):
|
|||||||
assert got.rstrip("\n") == value.rstrip("\n")
|
assert got.rstrip("\n") == value.rstrip("\n")
|
||||||
|
|
||||||
|
|
||||||
|
ESCAPED = ["a · b", "an em dash — here", "a face 😀 here", "plain ascii"]
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize("locale", [None, "C", "C.UTF-8"])
|
||||||
|
@pytest.mark.parametrize("value", ESCAPED)
|
||||||
|
def test_an_escaped_character_decodes_to_utf8_in_any_locale(value, locale):
|
||||||
|
"""The server's JSON escapes every non-ASCII character (`\\u00b7`), and a
|
||||||
|
hook may run with no locale at all. The round trip above sends raw UTF-8
|
||||||
|
and never reaches the escape path; this one sends what the server sends,
|
||||||
|
from an environment with nothing but PATH."""
|
||||||
|
need_tools("bash", "awk")
|
||||||
|
env = {"PATH": os.environ["PATH"]}
|
||||||
|
if locale:
|
||||||
|
env["LC_ALL"] = locale
|
||||||
|
body = json.dumps(value)[1:-1]
|
||||||
|
r = subprocess.run(["bash", "-c", f'. "{DEFS}"; scribe_json_unescape'],
|
||||||
|
input=body.encode(), capture_output=True, env=env)
|
||||||
|
assert r.returncode == 0, r.stderr.decode()
|
||||||
|
assert r.stdout == (value + "\n").encode()
|
||||||
|
|
||||||
|
|
||||||
|
def test_the_utf8_guard_can_fail():
|
||||||
|
"""A lone byte where the encoded character belongs is what the old
|
||||||
|
decoder wrote; the comparison above has to tell them apart."""
|
||||||
|
assert "·\n".encode() != b"\xb7\n"
|
||||||
|
|
||||||
|
|
||||||
# --------------------------------------------------------------------------
|
# --------------------------------------------------------------------------
|
||||||
# Reading: the shell side.
|
# Reading: the shell side.
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user