grammar: a #tag starts after whitespace and with a letter, on every surface
CI & Build / Python lint (push) Successful in 2s
CI & Build / Build now, or wait for Android? (push) Successful in 2s
Android / Build, or is the channel already serving this? (push) Successful in 4s
Desktop (Tauri) / Build, or is the channel already serving this? (push) Successful in 2s
CI & Build / Web typecheck and unit tests (push) Successful in 8s
CI & Build / Python tests (push) Successful in 11s
CI & Build / integration (push) Successful in 38s
CI & Build / Build & push image (push) Skipped
Desktop (Tauri) / Web tests, clippy, Rust tests and rustfmt (push) Successful in 2m7s
Desktop (Tauri) / Windows installer (cross-compiled) (push) Successful in 3m26s
Desktop (Tauri) / Tauri desktop (Linux) (push) Successful in 4m38s
Desktop (Tauri) / Update manifest (push) Successful in 4s
Android / Kotlin + Rust (APK) (push) Successful in 10m35s

The shared fixture went red on the server (run 8468: 5 failed), because the three
tag rules disagreed:
- core and web: any non-tag character counts as a boundary, so `(#todo)`
  and `end.#tag` are tags, and so is the `/#section` of a pasted URL;
- server: only whitespace counts, but `#1st` and `#_x` are tags.

A note's labels could therefore change every time it synced.

All three now share the strict rule: start of line or whitespace, then a
letter, then letters, digits, `_` and `-`. Nothing becomes a tag that wasn't
already one everywhere, and URL anchors stop becoming labels on desktop and
Android. The server's existing `http://x/#nope` test already expected this.

derive.rs's boundary, markdown.ts's lookbehind and tags.py's regex change
together; `_is_tag` goes because the regex now requires the letter. The
fixture flips `(#todo)`, `end.#tag` and its lift case, and adds the URL case.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
2026-10-06 23:22:53 -04:00
co-authored by Claude Opus 5.5
parent 535331c5b2
commit 38da5160a0
4 changed files with 33 additions and 21 deletions
+9 -3
View File
@@ -2,8 +2,10 @@
//! computes on save. Pure string scanning (no regex dependency), kept in lockstep
//! with the frontend's inline rules (see frontend notes/markdown.ts):
//!
//! - `#tag`: `#` at a word boundary followed by tag characters (letter first).
//! On save these become labels attached with `via_tag = true`.
//! - `#tag`: `#` at the start of a line or after whitespace, then a letter, then
//! letters, digits, `_` and `-`. On save these become labels attached with
//! `via_tag = true`. The server and the web run the same cases
//! (core/testdata/grammar.json).
//! - `- [ ] item`: a checklist item. The body IS the checklist (M304) — there is no
//! table of items beside it, so a list can sit between two paragraphs instead of
//! only after them.
@@ -29,7 +31,11 @@ fn line_tags(chars: &[char]) -> Vec<(usize, usize, String)> {
let mut i = 0;
while i < chars.len() {
if chars[i] == '#' {
let boundary = i == 0 || (!is_tag_char(chars[i - 1]) && chars[i - 1] != '#');
// Start of line or after whitespace, and nothing else. Any non-tag
// character used to count, which made `(#todo)` a tag here and plain
// text on the server, and made the `/#section` of a pasted URL a label.
// Whitespace is the rule all three implementations now share (#5166).
let boundary = i == 0 || chars[i - 1].is_whitespace();
// A tag must start with a letter (so "#1" or a bare "#" is not a tag).
if boundary && i + 1 < chars.len() && chars[i + 1].is_alphabetic() {
let mut j = i + 1;
+4 -3
View File
@@ -43,8 +43,9 @@
{ "body": "#tag, and #more.", "tags": ["tag", "more"] },
{ "body": "#x-y_z9", "tags": ["x-y_z9"] },
{ "body": "#café #über", "tags": ["café", "über"] },
{ "body": "(#todo)", "tags": ["todo"] },
{ "body": "end.#tag", "tags": ["tag"] },
{ "body": "(#todo)", "tags": [] },
{ "body": "end.#tag", "tags": [] },
{ "body": "https://x.com/#section", "tags": [] },
{ "body": "#tag#more", "tags": ["tag"] },
{ "body": "a#b", "tags": [] },
{ "body": "x-#tag", "tags": [] },
@@ -64,7 +65,7 @@
{ "body": "#todo\ncall #todo later", "standalone": [], "inline": ["todo"], "lifted": "call #todo later" },
{ "body": "#a #b", "standalone": [], "inline": ["a", "b"], "lifted": "#a #b" },
{ "body": "```\n# comment #todo\n```", "standalone": [], "inline": ["todo"], "lifted": "```\n# comment #todo\n```" },
{ "body": "(#todo)\nnotes", "standalone": [], "inline": ["todo"], "lifted": "(#todo)\nnotes" }
{ "body": "(#todo)\nnotes", "standalone": [], "inline": [], "lifted": "(#todo)\nnotes" }
],
"tint": {
+5 -4
View File
@@ -46,15 +46,16 @@ export interface TaskMeta {
// chance to start inside `**bold #x**`.
//
// The grammar MIRRORS `line_tags` in core/src/local/derive.rs, which is the definition:
// a `#` at a word boundary (the preceding character is neither a tag character nor
// another `#`, so `a#b` and `##x` are not tags), a letter immediately after it, then
// a `#` at the start of the text or after whitespace (so `a#b`, `##x`, `(#x)` and the
// `/#section` of a URL are not tags), a letter immediately after it, then
// alphanumerics, `_` and `-`. Rust's `is_alphanumeric` is `Alphabetic | N`, hence the
// property escapes rather than `\w` — and hence the `u` flag.
// property escapes rather than `\w` — and hence the `u` flag. All three
// implementations run core/testdata/grammar.json (notes/grammar.test.ts here).
//
// A heading cannot collide with this: `parseMarkdown` requires a space after the `#`s,
// which `#tag` by definition does not have.
const INLINE_RE =
/(`[^`]+`)|(\*\*[^*]+\*\*)|(\*[^*]+\*)|(_[^_]+_)|((?<![\p{Alphabetic}\p{N}_#-])#\p{Alphabetic}[\p{Alphabetic}\p{N}_-]*)/gu;
/(`[^`]+`)|(\*\*[^*]+\*\*)|(\*[^*]+\*)|(_[^_]+_)|((?<!\S)#\p{Alphabetic}[\p{Alphabetic}\p{N}_-]*)/gu;
export function parseInline(text: string): InlineToken[] {
const tokens: InlineToken[] = [];
+15 -11
View File
@@ -22,10 +22,19 @@ from ..models.label import Label, NoteLabel
from ..models.note import Note
from .helpers import derive_display_title
# A #tag: `#` at the start of the body or after whitespace, then a word char and
# word chars/hyphens. A URL fragment (foo#bar) or mid-word `#` is not preceded by
# whitespace, so it won't match.
_TAG_RE = re.compile(r"(?:^|(?<=\s))#(\w[\w-]*)")
# A #tag: `#` at the start of a line or after whitespace, then a LETTER, then
# letters, digits, `_` and `-`. The same rule as `line_tags` in
# core/src/local/derive.rs and the web's markdown.ts, and all three run the cases in
# core/testdata/grammar.json.
#
# The three used to differ. The core took a tag after any punctuation, so `(#todo)`
# was a label on the phone and plain text here, and the `/#section` of a pasted URL
# became a label too. This side let a tag start with a digit or `_` (`#1st`). So a
# note's labels changed every time it synced. The shared rule is the strict one:
# nothing is a tag now that was not already one everywhere (#5166).
#
# `[^\W\d_]` is "a word character that is not a digit or `_`", which is a letter.
_TAG_RE = re.compile(r"(?<!\S)#([^\W\d_][\w-]*)")
# A fence opens or closes a code block. A `#tag` inside one is CODE — the shell
@@ -35,11 +44,6 @@ _TAG_RE = re.compile(r"(?:^|(?<=\s))#(\w[\w-]*)")
_FENCE_RE = re.compile(r"^\s*(?:```|~~~)")
def _is_tag(name: str) -> bool:
"""A tag must contain a letter, so #2024 and #_ are ignored (avoids numeric noise)."""
return any(c.isalpha() for c in name)
def _dedupe(names: list[str]) -> list[str]:
"""First-seen order, deduped case-insensitively — tags are case-insensitive."""
out: list[str] = []
@@ -56,7 +60,7 @@ def parse_tags(body: str | None) -> list[str]:
"""Distinct #hashtags from a note body, in order, deduped case-insensitively."""
if not body:
return []
return _dedupe([m.group(1) for m in _TAG_RE.finditer(body) if _is_tag(m.group(1))])
return _dedupe([m.group(1) for m in _TAG_RE.finditer(body)])
def split_body_tags(body: str | None) -> tuple[list[str], list[str], str]:
@@ -90,7 +94,7 @@ def split_body_tags(body: str | None) -> tuple[list[str], list[str], str]:
in_fence = not in_fence
kept.append(line)
continue
matches = [m for m in _TAG_RE.finditer(line) if _is_tag(m.group(1))]
matches = list(_TAG_RE.finditer(line))
names = [m.group(1) for m in matches]
# Cutting the tags out and finding nothing left is what "standalone" means.
remainder = line