grammar: a #tag starts after whitespace and with a letter, on every surface
CI & Build / Python lint (push) Successful in 2s
CI & Build / Build now, or wait for Android? (push) Successful in 2s
Android / Build, or is the channel already serving this? (push) Successful in 4s
Desktop (Tauri) / Build, or is the channel already serving this? (push) Successful in 2s
CI & Build / Web typecheck and unit tests (push) Successful in 8s
CI & Build / Python tests (push) Successful in 11s
CI & Build / integration (push) Successful in 38s
CI & Build / Build & push image (push) Skipped
Desktop (Tauri) / Web tests, clippy, Rust tests and rustfmt (push) Successful in 2m7s
Desktop (Tauri) / Windows installer (cross-compiled) (push) Successful in 3m26s
Desktop (Tauri) / Tauri desktop (Linux) (push) Successful in 4m38s
Desktop (Tauri) / Update manifest (push) Successful in 4s
Android / Kotlin + Rust (APK) (push) Successful in 10m35s
CI & Build / Python lint (push) Successful in 2s
CI & Build / Build now, or wait for Android? (push) Successful in 2s
Android / Build, or is the channel already serving this? (push) Successful in 4s
Desktop (Tauri) / Build, or is the channel already serving this? (push) Successful in 2s
CI & Build / Web typecheck and unit tests (push) Successful in 8s
CI & Build / Python tests (push) Successful in 11s
CI & Build / integration (push) Successful in 38s
CI & Build / Build & push image (push) Skipped
Desktop (Tauri) / Web tests, clippy, Rust tests and rustfmt (push) Successful in 2m7s
Desktop (Tauri) / Windows installer (cross-compiled) (push) Successful in 3m26s
Desktop (Tauri) / Tauri desktop (Linux) (push) Successful in 4m38s
Desktop (Tauri) / Update manifest (push) Successful in 4s
Android / Kotlin + Rust (APK) (push) Successful in 10m35s
The shared fixture went red on the server (run 8468: 5 failed), because the three tag rules disagreed: - core and web: any non-tag character counts as a boundary, so `(#todo)` and `end.#tag` are tags, and so is the `/#section` of a pasted URL; - server: only whitespace counts, but `#1st` and `#_x` are tags. A note's labels could therefore change every time it synced. All three now share the strict rule: start of line or whitespace, then a letter, then letters, digits, `_` and `-`. Nothing becomes a tag that wasn't already one everywhere, and URL anchors stop becoming labels on desktop and Android. The server's existing `http://x/#nope` test already expected this. derive.rs's boundary, markdown.ts's lookbehind and tags.py's regex change together; `_is_tag` goes because the regex now requires the letter. The fixture flips `(#todo)`, `end.#tag` and its lift case, and adds the URL case. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -2,8 +2,10 @@
|
|||||||
//! computes on save. Pure string scanning (no regex dependency), kept in lockstep
|
//! computes on save. Pure string scanning (no regex dependency), kept in lockstep
|
||||||
//! with the frontend's inline rules (see frontend notes/markdown.ts):
|
//! with the frontend's inline rules (see frontend notes/markdown.ts):
|
||||||
//!
|
//!
|
||||||
//! - `#tag`: `#` at a word boundary followed by tag characters (letter first).
|
//! - `#tag`: `#` at the start of a line or after whitespace, then a letter, then
|
||||||
//! On save these become labels attached with `via_tag = true`.
|
//! letters, digits, `_` and `-`. On save these become labels attached with
|
||||||
|
//! `via_tag = true`. The server and the web run the same cases
|
||||||
|
//! (core/testdata/grammar.json).
|
||||||
//! - `- [ ] item`: a checklist item. The body IS the checklist (M304) — there is no
|
//! - `- [ ] item`: a checklist item. The body IS the checklist (M304) — there is no
|
||||||
//! table of items beside it, so a list can sit between two paragraphs instead of
|
//! table of items beside it, so a list can sit between two paragraphs instead of
|
||||||
//! only after them.
|
//! only after them.
|
||||||
@@ -29,7 +31,11 @@ fn line_tags(chars: &[char]) -> Vec<(usize, usize, String)> {
|
|||||||
let mut i = 0;
|
let mut i = 0;
|
||||||
while i < chars.len() {
|
while i < chars.len() {
|
||||||
if chars[i] == '#' {
|
if chars[i] == '#' {
|
||||||
let boundary = i == 0 || (!is_tag_char(chars[i - 1]) && chars[i - 1] != '#');
|
// Start of line or after whitespace, and nothing else. Any non-tag
|
||||||
|
// character used to count, which made `(#todo)` a tag here and plain
|
||||||
|
// text on the server, and made the `/#section` of a pasted URL a label.
|
||||||
|
// Whitespace is the rule all three implementations now share (#5166).
|
||||||
|
let boundary = i == 0 || chars[i - 1].is_whitespace();
|
||||||
// A tag must start with a letter (so "#1" or a bare "#" is not a tag).
|
// A tag must start with a letter (so "#1" or a bare "#" is not a tag).
|
||||||
if boundary && i + 1 < chars.len() && chars[i + 1].is_alphabetic() {
|
if boundary && i + 1 < chars.len() && chars[i + 1].is_alphabetic() {
|
||||||
let mut j = i + 1;
|
let mut j = i + 1;
|
||||||
|
|||||||
Vendored
+4
-3
@@ -43,8 +43,9 @@
|
|||||||
{ "body": "#tag, and #more.", "tags": ["tag", "more"] },
|
{ "body": "#tag, and #more.", "tags": ["tag", "more"] },
|
||||||
{ "body": "#x-y_z9", "tags": ["x-y_z9"] },
|
{ "body": "#x-y_z9", "tags": ["x-y_z9"] },
|
||||||
{ "body": "#café #über", "tags": ["café", "über"] },
|
{ "body": "#café #über", "tags": ["café", "über"] },
|
||||||
{ "body": "(#todo)", "tags": ["todo"] },
|
{ "body": "(#todo)", "tags": [] },
|
||||||
{ "body": "end.#tag", "tags": ["tag"] },
|
{ "body": "end.#tag", "tags": [] },
|
||||||
|
{ "body": "https://x.com/#section", "tags": [] },
|
||||||
{ "body": "#tag#more", "tags": ["tag"] },
|
{ "body": "#tag#more", "tags": ["tag"] },
|
||||||
{ "body": "a#b", "tags": [] },
|
{ "body": "a#b", "tags": [] },
|
||||||
{ "body": "x-#tag", "tags": [] },
|
{ "body": "x-#tag", "tags": [] },
|
||||||
@@ -64,7 +65,7 @@
|
|||||||
{ "body": "#todo\ncall #todo later", "standalone": [], "inline": ["todo"], "lifted": "call #todo later" },
|
{ "body": "#todo\ncall #todo later", "standalone": [], "inline": ["todo"], "lifted": "call #todo later" },
|
||||||
{ "body": "#a #b", "standalone": [], "inline": ["a", "b"], "lifted": "#a #b" },
|
{ "body": "#a #b", "standalone": [], "inline": ["a", "b"], "lifted": "#a #b" },
|
||||||
{ "body": "```\n# comment #todo\n```", "standalone": [], "inline": ["todo"], "lifted": "```\n# comment #todo\n```" },
|
{ "body": "```\n# comment #todo\n```", "standalone": [], "inline": ["todo"], "lifted": "```\n# comment #todo\n```" },
|
||||||
{ "body": "(#todo)\nnotes", "standalone": [], "inline": ["todo"], "lifted": "(#todo)\nnotes" }
|
{ "body": "(#todo)\nnotes", "standalone": [], "inline": [], "lifted": "(#todo)\nnotes" }
|
||||||
],
|
],
|
||||||
|
|
||||||
"tint": {
|
"tint": {
|
||||||
|
|||||||
@@ -46,15 +46,16 @@ export interface TaskMeta {
|
|||||||
// chance to start inside `**bold #x**`.
|
// chance to start inside `**bold #x**`.
|
||||||
//
|
//
|
||||||
// The grammar MIRRORS `line_tags` in core/src/local/derive.rs, which is the definition:
|
// The grammar MIRRORS `line_tags` in core/src/local/derive.rs, which is the definition:
|
||||||
// a `#` at a word boundary (the preceding character is neither a tag character nor
|
// a `#` at the start of the text or after whitespace (so `a#b`, `##x`, `(#x)` and the
|
||||||
// another `#`, so `a#b` and `##x` are not tags), a letter immediately after it, then
|
// `/#section` of a URL are not tags), a letter immediately after it, then
|
||||||
// alphanumerics, `_` and `-`. Rust's `is_alphanumeric` is `Alphabetic | N`, hence the
|
// alphanumerics, `_` and `-`. Rust's `is_alphanumeric` is `Alphabetic | N`, hence the
|
||||||
// property escapes rather than `\w` — and hence the `u` flag.
|
// property escapes rather than `\w` — and hence the `u` flag. All three
|
||||||
|
// implementations run core/testdata/grammar.json (notes/grammar.test.ts here).
|
||||||
//
|
//
|
||||||
// A heading cannot collide with this: `parseMarkdown` requires a space after the `#`s,
|
// A heading cannot collide with this: `parseMarkdown` requires a space after the `#`s,
|
||||||
// which `#tag` by definition does not have.
|
// which `#tag` by definition does not have.
|
||||||
const INLINE_RE =
|
const INLINE_RE =
|
||||||
/(`[^`]+`)|(\*\*[^*]+\*\*)|(\*[^*]+\*)|(_[^_]+_)|((?<![\p{Alphabetic}\p{N}_#-])#\p{Alphabetic}[\p{Alphabetic}\p{N}_-]*)/gu;
|
/(`[^`]+`)|(\*\*[^*]+\*\*)|(\*[^*]+\*)|(_[^_]+_)|((?<!\S)#\p{Alphabetic}[\p{Alphabetic}\p{N}_-]*)/gu;
|
||||||
|
|
||||||
export function parseInline(text: string): InlineToken[] {
|
export function parseInline(text: string): InlineToken[] {
|
||||||
const tokens: InlineToken[] = [];
|
const tokens: InlineToken[] = [];
|
||||||
|
|||||||
+15
-11
@@ -22,10 +22,19 @@ from ..models.label import Label, NoteLabel
|
|||||||
from ..models.note import Note
|
from ..models.note import Note
|
||||||
from .helpers import derive_display_title
|
from .helpers import derive_display_title
|
||||||
|
|
||||||
# A #tag: `#` at the start of the body or after whitespace, then a word char and
|
# A #tag: `#` at the start of a line or after whitespace, then a LETTER, then
|
||||||
# word chars/hyphens. A URL fragment (foo#bar) or mid-word `#` is not preceded by
|
# letters, digits, `_` and `-`. The same rule as `line_tags` in
|
||||||
# whitespace, so it won't match.
|
# core/src/local/derive.rs and the web's markdown.ts, and all three run the cases in
|
||||||
_TAG_RE = re.compile(r"(?:^|(?<=\s))#(\w[\w-]*)")
|
# core/testdata/grammar.json.
|
||||||
|
#
|
||||||
|
# The three used to differ. The core took a tag after any punctuation, so `(#todo)`
|
||||||
|
# was a label on the phone and plain text here, and the `/#section` of a pasted URL
|
||||||
|
# became a label too. This side let a tag start with a digit or `_` (`#1st`). So a
|
||||||
|
# note's labels changed every time it synced. The shared rule is the strict one:
|
||||||
|
# nothing is a tag now that was not already one everywhere (#5166).
|
||||||
|
#
|
||||||
|
# `[^\W\d_]` is "a word character that is not a digit or `_`", which is a letter.
|
||||||
|
_TAG_RE = re.compile(r"(?<!\S)#([^\W\d_][\w-]*)")
|
||||||
|
|
||||||
|
|
||||||
# A fence opens or closes a code block. A `#tag` inside one is CODE — the shell
|
# A fence opens or closes a code block. A `#tag` inside one is CODE — the shell
|
||||||
@@ -35,11 +44,6 @@ _TAG_RE = re.compile(r"(?:^|(?<=\s))#(\w[\w-]*)")
|
|||||||
_FENCE_RE = re.compile(r"^\s*(?:```|~~~)")
|
_FENCE_RE = re.compile(r"^\s*(?:```|~~~)")
|
||||||
|
|
||||||
|
|
||||||
def _is_tag(name: str) -> bool:
|
|
||||||
"""A tag must contain a letter, so #2024 and #_ are ignored (avoids numeric noise)."""
|
|
||||||
return any(c.isalpha() for c in name)
|
|
||||||
|
|
||||||
|
|
||||||
def _dedupe(names: list[str]) -> list[str]:
|
def _dedupe(names: list[str]) -> list[str]:
|
||||||
"""First-seen order, deduped case-insensitively — tags are case-insensitive."""
|
"""First-seen order, deduped case-insensitively — tags are case-insensitive."""
|
||||||
out: list[str] = []
|
out: list[str] = []
|
||||||
@@ -56,7 +60,7 @@ def parse_tags(body: str | None) -> list[str]:
|
|||||||
"""Distinct #hashtags from a note body, in order, deduped case-insensitively."""
|
"""Distinct #hashtags from a note body, in order, deduped case-insensitively."""
|
||||||
if not body:
|
if not body:
|
||||||
return []
|
return []
|
||||||
return _dedupe([m.group(1) for m in _TAG_RE.finditer(body) if _is_tag(m.group(1))])
|
return _dedupe([m.group(1) for m in _TAG_RE.finditer(body)])
|
||||||
|
|
||||||
|
|
||||||
def split_body_tags(body: str | None) -> tuple[list[str], list[str], str]:
|
def split_body_tags(body: str | None) -> tuple[list[str], list[str], str]:
|
||||||
@@ -90,7 +94,7 @@ def split_body_tags(body: str | None) -> tuple[list[str], list[str], str]:
|
|||||||
in_fence = not in_fence
|
in_fence = not in_fence
|
||||||
kept.append(line)
|
kept.append(line)
|
||||||
continue
|
continue
|
||||||
matches = [m for m in _TAG_RE.finditer(line) if _is_tag(m.group(1))]
|
matches = list(_TAG_RE.finditer(line))
|
||||||
names = [m.group(1) for m in matches]
|
names = [m.group(1) for m in matches]
|
||||||
# Cutting the tags out and finding nothing left is what "standalone" means.
|
# Cutting the tags out and finding nothing left is what "standalone" means.
|
||||||
remainder = line
|
remainder = line
|
||||||
|
|||||||
Reference in New Issue
Block a user