grammar: a #tag starts after whitespace and with a letter, on every surface
CI & Build / Python lint (push) Successful in 2s
CI & Build / Build now, or wait for Android? (push) Successful in 2s
Android / Build, or is the channel already serving this? (push) Successful in 4s
Desktop (Tauri) / Build, or is the channel already serving this? (push) Successful in 2s
CI & Build / Web typecheck and unit tests (push) Successful in 8s
CI & Build / Python tests (push) Successful in 11s
CI & Build / integration (push) Successful in 38s
CI & Build / Build & push image (push) Skipped
Desktop (Tauri) / Web tests, clippy, Rust tests and rustfmt (push) Successful in 2m7s
Desktop (Tauri) / Windows installer (cross-compiled) (push) Successful in 3m26s
Desktop (Tauri) / Tauri desktop (Linux) (push) Successful in 4m38s
Desktop (Tauri) / Update manifest (push) Successful in 4s
Android / Kotlin + Rust (APK) (push) Successful in 10m35s
CI & Build / Python lint (push) Successful in 2s
CI & Build / Build now, or wait for Android? (push) Successful in 2s
Android / Build, or is the channel already serving this? (push) Successful in 4s
Desktop (Tauri) / Build, or is the channel already serving this? (push) Successful in 2s
CI & Build / Web typecheck and unit tests (push) Successful in 8s
CI & Build / Python tests (push) Successful in 11s
CI & Build / integration (push) Successful in 38s
CI & Build / Build & push image (push) Skipped
Desktop (Tauri) / Web tests, clippy, Rust tests and rustfmt (push) Successful in 2m7s
Desktop (Tauri) / Windows installer (cross-compiled) (push) Successful in 3m26s
Desktop (Tauri) / Tauri desktop (Linux) (push) Successful in 4m38s
Desktop (Tauri) / Update manifest (push) Successful in 4s
Android / Kotlin + Rust (APK) (push) Successful in 10m35s
The shared fixture went red on the server (run 8468: 5 failed), because the three tag rules disagreed: - core and web: any non-tag character counts as a boundary, so `(#todo)` and `end.#tag` are tags, and so is the `/#section` of a pasted URL; - server: only whitespace counts, but `#1st` and `#_x` are tags. A note's labels could therefore change every time it synced. All three now share the strict rule: start of line or whitespace, then a letter, then letters, digits, `_` and `-`. Nothing becomes a tag that wasn't already one everywhere, and URL anchors stop becoming labels on desktop and Android. The server's existing `http://x/#nope` test already expected this. derive.rs's boundary, markdown.ts's lookbehind and tags.py's regex change together; `_is_tag` goes because the regex now requires the letter. The fixture flips `(#todo)`, `end.#tag` and its lift case, and adds the URL case. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -2,8 +2,10 @@
|
||||
//! computes on save. Pure string scanning (no regex dependency), kept in lockstep
|
||||
//! with the frontend's inline rules (see frontend notes/markdown.ts):
|
||||
//!
|
||||
//! - `#tag`: `#` at a word boundary followed by tag characters (letter first).
|
||||
//! On save these become labels attached with `via_tag = true`.
|
||||
//! - `#tag`: `#` at the start of a line or after whitespace, then a letter, then
|
||||
//! letters, digits, `_` and `-`. On save these become labels attached with
|
||||
//! `via_tag = true`. The server and the web run the same cases
|
||||
//! (core/testdata/grammar.json).
|
||||
//! - `- [ ] item`: a checklist item. The body IS the checklist (M304) — there is no
|
||||
//! table of items beside it, so a list can sit between two paragraphs instead of
|
||||
//! only after them.
|
||||
@@ -29,7 +31,11 @@ fn line_tags(chars: &[char]) -> Vec<(usize, usize, String)> {
|
||||
let mut i = 0;
|
||||
while i < chars.len() {
|
||||
if chars[i] == '#' {
|
||||
let boundary = i == 0 || (!is_tag_char(chars[i - 1]) && chars[i - 1] != '#');
|
||||
// Start of line or after whitespace, and nothing else. Any non-tag
|
||||
// character used to count, which made `(#todo)` a tag here and plain
|
||||
// text on the server, and made the `/#section` of a pasted URL a label.
|
||||
// Whitespace is the rule all three implementations now share (#5166).
|
||||
let boundary = i == 0 || chars[i - 1].is_whitespace();
|
||||
// A tag must start with a letter (so "#1" or a bare "#" is not a tag).
|
||||
if boundary && i + 1 < chars.len() && chars[i + 1].is_alphabetic() {
|
||||
let mut j = i + 1;
|
||||
|
||||
Vendored
+4
-3
@@ -43,8 +43,9 @@
|
||||
{ "body": "#tag, and #more.", "tags": ["tag", "more"] },
|
||||
{ "body": "#x-y_z9", "tags": ["x-y_z9"] },
|
||||
{ "body": "#café #über", "tags": ["café", "über"] },
|
||||
{ "body": "(#todo)", "tags": ["todo"] },
|
||||
{ "body": "end.#tag", "tags": ["tag"] },
|
||||
{ "body": "(#todo)", "tags": [] },
|
||||
{ "body": "end.#tag", "tags": [] },
|
||||
{ "body": "https://x.com/#section", "tags": [] },
|
||||
{ "body": "#tag#more", "tags": ["tag"] },
|
||||
{ "body": "a#b", "tags": [] },
|
||||
{ "body": "x-#tag", "tags": [] },
|
||||
@@ -64,7 +65,7 @@
|
||||
{ "body": "#todo\ncall #todo later", "standalone": [], "inline": ["todo"], "lifted": "call #todo later" },
|
||||
{ "body": "#a #b", "standalone": [], "inline": ["a", "b"], "lifted": "#a #b" },
|
||||
{ "body": "```\n# comment #todo\n```", "standalone": [], "inline": ["todo"], "lifted": "```\n# comment #todo\n```" },
|
||||
{ "body": "(#todo)\nnotes", "standalone": [], "inline": ["todo"], "lifted": "(#todo)\nnotes" }
|
||||
{ "body": "(#todo)\nnotes", "standalone": [], "inline": [], "lifted": "(#todo)\nnotes" }
|
||||
],
|
||||
|
||||
"tint": {
|
||||
|
||||
@@ -46,15 +46,16 @@ export interface TaskMeta {
|
||||
// chance to start inside `**bold #x**`.
|
||||
//
|
||||
// The grammar MIRRORS `line_tags` in core/src/local/derive.rs, which is the definition:
|
||||
// a `#` at a word boundary (the preceding character is neither a tag character nor
|
||||
// another `#`, so `a#b` and `##x` are not tags), a letter immediately after it, then
|
||||
// a `#` at the start of the text or after whitespace (so `a#b`, `##x`, `(#x)` and the
|
||||
// `/#section` of a URL are not tags), a letter immediately after it, then
|
||||
// alphanumerics, `_` and `-`. Rust's `is_alphanumeric` is `Alphabetic | N`, hence the
|
||||
// property escapes rather than `\w` — and hence the `u` flag.
|
||||
// property escapes rather than `\w` — and hence the `u` flag. All three
|
||||
// implementations run core/testdata/grammar.json (notes/grammar.test.ts here).
|
||||
//
|
||||
// A heading cannot collide with this: `parseMarkdown` requires a space after the `#`s,
|
||||
// which `#tag` by definition does not have.
|
||||
const INLINE_RE =
|
||||
/(`[^`]+`)|(\*\*[^*]+\*\*)|(\*[^*]+\*)|(_[^_]+_)|((?<![\p{Alphabetic}\p{N}_#-])#\p{Alphabetic}[\p{Alphabetic}\p{N}_-]*)/gu;
|
||||
/(`[^`]+`)|(\*\*[^*]+\*\*)|(\*[^*]+\*)|(_[^_]+_)|((?<!\S)#\p{Alphabetic}[\p{Alphabetic}\p{N}_-]*)/gu;
|
||||
|
||||
export function parseInline(text: string): InlineToken[] {
|
||||
const tokens: InlineToken[] = [];
|
||||
|
||||
+15
-11
@@ -22,10 +22,19 @@ from ..models.label import Label, NoteLabel
|
||||
from ..models.note import Note
|
||||
from .helpers import derive_display_title
|
||||
|
||||
# A #tag: `#` at the start of the body or after whitespace, then a word char and
|
||||
# word chars/hyphens. A URL fragment (foo#bar) or mid-word `#` is not preceded by
|
||||
# whitespace, so it won't match.
|
||||
_TAG_RE = re.compile(r"(?:^|(?<=\s))#(\w[\w-]*)")
|
||||
# A #tag: `#` at the start of a line or after whitespace, then a LETTER, then
|
||||
# letters, digits, `_` and `-`. The same rule as `line_tags` in
|
||||
# core/src/local/derive.rs and the web's markdown.ts, and all three run the cases in
|
||||
# core/testdata/grammar.json.
|
||||
#
|
||||
# The three used to differ. The core took a tag after any punctuation, so `(#todo)`
|
||||
# was a label on the phone and plain text here, and the `/#section` of a pasted URL
|
||||
# became a label too. This side let a tag start with a digit or `_` (`#1st`). So a
|
||||
# note's labels changed every time it synced. The shared rule is the strict one:
|
||||
# nothing is a tag now that was not already one everywhere (#5166).
|
||||
#
|
||||
# `[^\W\d_]` is "a word character that is not a digit or `_`", which is a letter.
|
||||
_TAG_RE = re.compile(r"(?<!\S)#([^\W\d_][\w-]*)")
|
||||
|
||||
|
||||
# A fence opens or closes a code block. A `#tag` inside one is CODE — the shell
|
||||
@@ -35,11 +44,6 @@ _TAG_RE = re.compile(r"(?:^|(?<=\s))#(\w[\w-]*)")
|
||||
_FENCE_RE = re.compile(r"^\s*(?:```|~~~)")
|
||||
|
||||
|
||||
def _is_tag(name: str) -> bool:
|
||||
"""A tag must contain a letter, so #2024 and #_ are ignored (avoids numeric noise)."""
|
||||
return any(c.isalpha() for c in name)
|
||||
|
||||
|
||||
def _dedupe(names: list[str]) -> list[str]:
|
||||
"""First-seen order, deduped case-insensitively — tags are case-insensitive."""
|
||||
out: list[str] = []
|
||||
@@ -56,7 +60,7 @@ def parse_tags(body: str | None) -> list[str]:
|
||||
"""Distinct #hashtags from a note body, in order, deduped case-insensitively."""
|
||||
if not body:
|
||||
return []
|
||||
return _dedupe([m.group(1) for m in _TAG_RE.finditer(body) if _is_tag(m.group(1))])
|
||||
return _dedupe([m.group(1) for m in _TAG_RE.finditer(body)])
|
||||
|
||||
|
||||
def split_body_tags(body: str | None) -> tuple[list[str], list[str], str]:
|
||||
@@ -90,7 +94,7 @@ def split_body_tags(body: str | None) -> tuple[list[str], list[str], str]:
|
||||
in_fence = not in_fence
|
||||
kept.append(line)
|
||||
continue
|
||||
matches = [m for m in _TAG_RE.finditer(line) if _is_tag(m.group(1))]
|
||||
matches = list(_TAG_RE.finditer(line))
|
||||
names = [m.group(1) for m in matches]
|
||||
# Cutting the tags out and finding nothing left is what "standalone" means.
|
||||
remainder = line
|
||||
|
||||
Reference in New Issue
Block a user