diff --git a/core/src/local/derive.rs b/core/src/local/derive.rs index 92af128..81f7946 100644 --- a/core/src/local/derive.rs +++ b/core/src/local/derive.rs @@ -2,8 +2,10 @@ //! computes on save. Pure string scanning (no regex dependency), kept in lockstep //! with the frontend's inline rules (see frontend notes/markdown.ts): //! -//! - `#tag`: `#` at a word boundary followed by tag characters (letter first). -//! On save these become labels attached with `via_tag = true`. +//! - `#tag`: `#` at the start of a line or after whitespace, then a letter, then +//! letters, digits, `_` and `-`. On save these become labels attached with +//! `via_tag = true`. The server and the web run the same cases +//! (core/testdata/grammar.json). //! - `- [ ] item`: a checklist item. The body IS the checklist (M304) — there is no //! table of items beside it, so a list can sit between two paragraphs instead of //! only after them. @@ -29,7 +31,11 @@ fn line_tags(chars: &[char]) -> Vec<(usize, usize, String)> { let mut i = 0; while i < chars.len() { if chars[i] == '#' { - let boundary = i == 0 || (!is_tag_char(chars[i - 1]) && chars[i - 1] != '#'); + // Start of line or after whitespace, and nothing else. Any non-tag + // character used to count, which made `(#todo)` a tag here and plain + // text on the server, and made the `/#section` of a pasted URL a label. + // Whitespace is the rule all three implementations now share (#5166). + let boundary = i == 0 || chars[i - 1].is_whitespace(); // A tag must start with a letter (so "#1" or a bare "#" is not a tag). if boundary && i + 1 < chars.len() && chars[i + 1].is_alphabetic() { let mut j = i + 1; diff --git a/core/testdata/grammar.json b/core/testdata/grammar.json index 8e83faa..98270fe 100644 --- a/core/testdata/grammar.json +++ b/core/testdata/grammar.json @@ -43,8 +43,9 @@ { "body": "#tag, and #more.", "tags": ["tag", "more"] }, { "body": "#x-y_z9", "tags": ["x-y_z9"] }, { "body": "#café #über", "tags": ["café", "über"] }, - { "body": "(#todo)", "tags": ["todo"] }, - { "body": "end.#tag", "tags": ["tag"] }, + { "body": "(#todo)", "tags": [] }, + { "body": "end.#tag", "tags": [] }, + { "body": "https://x.com/#section", "tags": [] }, { "body": "#tag#more", "tags": ["tag"] }, { "body": "a#b", "tags": [] }, { "body": "x-#tag", "tags": [] }, @@ -64,7 +65,7 @@ { "body": "#todo\ncall #todo later", "standalone": [], "inline": ["todo"], "lifted": "call #todo later" }, { "body": "#a #b", "standalone": [], "inline": ["a", "b"], "lifted": "#a #b" }, { "body": "```\n# comment #todo\n```", "standalone": [], "inline": ["todo"], "lifted": "```\n# comment #todo\n```" }, - { "body": "(#todo)\nnotes", "standalone": [], "inline": ["todo"], "lifted": "(#todo)\nnotes" } + { "body": "(#todo)\nnotes", "standalone": [], "inline": [], "lifted": "(#todo)\nnotes" } ], "tint": { diff --git a/frontend/src/notes/markdown.ts b/frontend/src/notes/markdown.ts index d9e841a..934fec5 100644 --- a/frontend/src/notes/markdown.ts +++ b/frontend/src/notes/markdown.ts @@ -46,15 +46,16 @@ export interface TaskMeta { // chance to start inside `**bold #x**`. // // The grammar MIRRORS `line_tags` in core/src/local/derive.rs, which is the definition: -// a `#` at a word boundary (the preceding character is neither a tag character nor -// another `#`, so `a#b` and `##x` are not tags), a letter immediately after it, then +// a `#` at the start of the text or after whitespace (so `a#b`, `##x`, `(#x)` and the +// `/#section` of a URL are not tags), a letter immediately after it, then // alphanumerics, `_` and `-`. Rust's `is_alphanumeric` is `Alphabetic | N`, hence the -// property escapes rather than `\w` — and hence the `u` flag. +// property escapes rather than `\w` — and hence the `u` flag. All three +// implementations run core/testdata/grammar.json (notes/grammar.test.ts here). // // A heading cannot collide with this: `parseMarkdown` requires a space after the `#`s, // which `#tag` by definition does not have. const INLINE_RE = - /(`[^`]+`)|(\*\*[^*]+\*\*)|(\*[^*]+\*)|(_[^_]+_)|((? bool: - """A tag must contain a letter, so #2024 and #_ are ignored (avoids numeric noise).""" - return any(c.isalpha() for c in name) - - def _dedupe(names: list[str]) -> list[str]: """First-seen order, deduped case-insensitively — tags are case-insensitive.""" out: list[str] = [] @@ -56,7 +60,7 @@ def parse_tags(body: str | None) -> list[str]: """Distinct #hashtags from a note body, in order, deduped case-insensitively.""" if not body: return [] - return _dedupe([m.group(1) for m in _TAG_RE.finditer(body) if _is_tag(m.group(1))]) + return _dedupe([m.group(1) for m in _TAG_RE.finditer(body)]) def split_body_tags(body: str | None) -> tuple[list[str], list[str], str]: @@ -90,7 +94,7 @@ def split_body_tags(body: str | None) -> tuple[list[str], list[str], str]: in_fence = not in_fence kept.append(line) continue - matches = [m for m in _TAG_RE.finditer(line) if _is_tag(m.group(1))] + matches = list(_TAG_RE.finditer(line)) names = [m.group(1) for m in matches] # Cutting the tags out and finding nothing left is what "standalone" means. remainder = line