//! Deriving structure from a note's body — the local mirror of what the server //! computes on save. Pure string scanning (no regex dependency), kept in lockstep //! with the frontend's inline rules (see frontend notes/markdown.ts): //! //! - `#tag`: `#` at the start of a line or after whitespace, then a letter, then //! letters, digits, `_` and `-`. On save these become labels attached with //! `via_tag = true`. The server and the web run the same cases //! (core/testdata/grammar.json). //! - `- [ ] item`: a checklist item. The body IS the checklist (M304) — there is no //! table of items beside it, so a list can sit between two paragraphs instead of //! only after them. //! //! The two are the same idea at different strengths. Tags MATERIALISE into label //! rows, because the board queries by label. Items materialise into nothing, //! because nothing queries them: their only readers are the card, the editor and //! `display_title`. So `extract_items` is the whole storage layer for a checklist, //! and the rewriters below are how one is edited. //! //! Dedupes case-insensitively, preserving first-seen order. //! //! Also derived `[[wiki-links]]` until they were removed (note 2897) — this is a //! capture-and-recall surface, and a linking system is organization. /// Every `#tag` in ONE line, as `(start, end, name)` in char indices. /// /// Char indices rather than byte offsets so the spans can be used to cut the tags /// back out of the line without ever landing mid-codepoint — see /// [`lift_standalone_tags`], which is the only reason the spans exist. fn line_tags(chars: &[char]) -> Vec<(usize, usize, String)> { let mut out: Vec<(usize, usize, String)> = Vec::new(); let mut i = 0; while i < chars.len() { if chars[i] == '#' { // Start of line or after whitespace, and nothing else. Any non-tag // character used to count, which made `(#todo)` a tag here and plain // text on the server, and made the `/#section` of a pasted URL a label. // Whitespace is the rule all three implementations now share (#5166). let boundary = i == 0 || chars[i - 1].is_whitespace(); // A tag must start with a letter (so "#1" or a bare "#" is not a tag). if boundary && i + 1 < chars.len() && chars[i + 1].is_alphabetic() { let mut j = i + 1; while j < chars.len() && is_tag_char(chars[j]) { j += 1; } out.push((i, j, chars[i + 1..j].iter().collect())); i = j; continue; } } i += 1; } out } /// One `#tag` and exactly where it sits, for a renderer drawing the body itself. /// /// The card no longer prints a chip for a tag whose text is still in the note — it /// colours the token where it was typed instead. To do that a renderer needs the /// SPAN, not just the name, and asking it to find the name again would be a second /// grammar quietly disagreeing with this one about what `##a` or `#1` is. #[derive(Debug, Clone, PartialEq, Eq)] pub struct DerivedTag { /// Which body line it sits on, like [`DerivedItem::line`]. pub line: u32, /// Offsets into that line, in UTF-16 code units — INCLUDING the leading `#`. /// /// UTF-16 rather than chars or bytes because the two languages that consume this /// both index strings that way: Kotlin's `AnnotatedString` and JavaScript. A char /// index is right up until somebody puts an emoji before a tag, and then it lands /// mid-token with no error anywhere. pub start: u32, pub end: u32, pub name: String, } /// Every `#tag` in `body` with its position — the scan [`lift_standalone_tags`] does, /// keeping the spans instead of throwing them away. /// /// Not deduped: two mentions of `#todo` are two pieces of text to colour. Fences are /// not skipped either, and that is deliberate — a `#tag` inside a code block stays an /// inline label on the note, and a renderer that left it plain would be the only /// surface disagreeing. pub fn extract_tag_spans(body: &str) -> Vec { let mut out = Vec::new(); for (n, line) in body.split('\n').enumerate() { let chars: Vec = line.chars().collect(); let spans = line_tags(&chars); if spans.is_empty() { continue; } // Prefix sums, built once per tagged line: char index -> UTF-16 offset. let mut units: Vec = Vec::with_capacity(chars.len() + 1); let mut total: u32 = 0; units.push(0); for c in &chars { total += c.len_utf16() as u32; units.push(total); } for (start, end, name) in spans { out.push(DerivedTag { line: n as u32, start: units[start], end: units[end], name, }); } } out } /// Whether a line opens or closes a fenced code block. fn is_fence(line: &str) -> bool { let trimmed = line.trim_start(); trimmed.starts_with("```") || trimmed.starts_with("~~~") } /// Runs of three or more newlines become two, and the ends are trimmed. /// /// Removing a line must not leave a hole where it was. fn collapse_blank_runs(text: &str) -> String { let mut out = String::with_capacity(text.len()); let mut run = 0; for c in text.chars() { if c == '\n' { run += 1; if run <= 2 { out.push(c); } } else { run = 0; out.push(c); } } out.trim_matches('\n').to_string() } /// Split a body's tags by whether the text around them can be taken away. /// /// Returns `(standalone, inline, lifted_body)`. /// /// THE RULE: a line containing nothing but tags and whitespace is removed. Anything /// else is left exactly as written. /// /// The MIRROR of `split_body_tags` in the server's `notes/tags.py`, and it has to stay /// one: a note lifted differently here than there would change under the operator the /// moment it synced. Same discipline, and the same reason, as `DerivedTint`. /// /// The conservative reading of "standalone" is deliberate. A trailing tag is /// ambiguous and the text does not say which it is — `buy milk #grocery` is filing, /// `remember to call #mom` is the sentence's object, and lifting the second leaves /// "remember to call". A tag sharing a line with words keeps its words. /// /// `standalone` tags become ORDINARY labels (`via_tag = 0`): nothing is left to derive /// them from, so the row becomes the record and the chip's × becomes the way to remove /// one. `inline` tags stay derived exactly as before. That is what `via_tag` means from /// here on — backed by text still in the body. pub fn lift_standalone_tags(body: &str) -> (Vec, Vec, String) { let mut standalone: Vec = Vec::new(); let mut inline: Vec = Vec::new(); let mut kept: Vec<&str> = Vec::new(); let mut in_fence = false; for line in body.split('\n') { if is_fence(line) { in_fence = !in_fence; kept.push(line); continue; } let chars: Vec = line.chars().collect(); let spans = line_tags(&chars); // Cut the tags out and see whether anything is left. That is what // "standalone" means, and it is the whole rule. let mut remainder = String::new(); let mut pos = 0; for (start, end, _) in &spans { remainder.extend(chars[pos..*start].iter()); pos = *end; } remainder.extend(chars[pos..].iter()); // A fence's contents are CODE: a `#tag` there is a shell comment in somebody's // snippet, and deleting the line would eat part of their example. if in_fence || spans.is_empty() || !remainder.trim().is_empty() { for (_, _, name) in &spans { push_unique(&mut inline, name); } kept.push(line); } else { for (_, _, name) in &spans { push_unique(&mut standalone, name); } } } let lifted = collapse_blank_runs(&kept.join("\n")); if !body.trim().is_empty() && lifted.trim().is_empty() { // The note was NOTHING but tags. Lifting would leave a blank card, which is a // worse outcome than a duplicated chip — so leave it alone. let mut all = standalone; for name in &inline { push_unique(&mut all, name); } return (Vec::new(), all, body.to_string()); } // A tag that ALSO appears in prose stays derived: the prose copy still backs it, // so deleting that copy should still detach the label. let inline_lower: Vec = inline.iter().map(|n| n.to_lowercase()).collect(); let standalone = standalone .into_iter() .filter(|n| !inline_lower.contains(&n.to_lowercase())) .collect(); (standalone, inline, lifted) } fn is_tag_char(c: char) -> bool { c.is_alphanumeric() || c == '_' || c == '-' } fn push_unique(out: &mut Vec, candidate: &str) { if !out.iter().any(|x| x.eq_ignore_ascii_case(candidate)) { out.push(candidate.to_string()); } } // ── checklist items ───────────────────────────────────────────────────────── // // The grammar, in one place, because three languages implement it (here, // `notes/checklist.py`, `notes/markdown.ts`) and a difference between any two of // them is a checklist that changes shape when it syncs: // // optional indent, `-` or `*`, one-or-more spaces, `[ ]`/`[x]`/`[X]`, // then either end-of-line or one-or-more spaces and the text. // // `*` is accepted because markdown.ts already accepts it for a plain bullet, and a // grammar that takes `* item` but not `* [ ] item` would be a rule with no reason // anyone could guess. `- [ ]` with nothing after it IS an item with empty text: // that is exactly what pressing Enter on a list leaves behind, and refusing to // parse it would make a half-typed list stop being a list. /// A checklist item, as found in the body. Its position in the returned vector is /// its identity — the same thing `position` meant when these were rows, and all the /// wire ever carried (`push.rs` sent text and checked, never an id). #[derive(Debug, Clone, PartialEq, Eq)] pub struct DerivedItem { pub text: String, pub checked: bool, /// Which body line it sits on. /// /// Carried here rather than offered as a second function, because every renderer /// that walks a body line by line — the Android card, the block editor — needs the /// text, the state AND the position together, and asking for them separately is /// how two calls come to disagree about a body that changed between them. pub line: u32, } /// One parsed task line, holding enough to put it back exactly as it was found. struct TaskLine<'a> { indent: &'a str, /// Preserved rather than normalised to `-`: rewriting someone's `*` bullets /// because they ticked a box would be an edit they did not ask for. bullet: char, checked: bool, text: &'a str, } fn parse_task_line(line: &str) -> Option> { let indent_len = line.len() - line.trim_start().len(); let (indent, rest) = line.split_at(indent_len); let bullet = rest.chars().next()?; if bullet != '-' && bullet != '*' { return None; } // At least one space after the bullet. `-[ ] x` is not a list item in any // markdown either, so it stays prose here too. let rest = &rest[bullet.len_utf8()..]; let gap = rest.len() - rest.trim_start_matches(' ').len(); if gap == 0 { return None; } let rest = &rest[gap..]; let mut chars = rest.chars(); if chars.next()? != '[' { return None; } let mark = chars.next()?; if chars.next()? != ']' { return None; } // Decided BEFORE the slice below, which is what guarantees `mark` is one byte // and `[?]` is exactly three. let checked = match mark { ' ' => false, 'x' | 'X' => true, _ => return None, }; let rest = &rest[3..]; let text = if rest.is_empty() { // "- [ ]" — an empty item, which is what an unfinished list line is. rest } else { let gap = rest.len() - rest.trim_start_matches(' ').len(); // "- [ ]x" is prose: without the space this is not a marker, it is a // sentence that happens to start with brackets. if gap == 0 { return None; } &rest[gap..] }; Some(TaskLine { indent, bullet, checked, text, }) } /// One item as the line that stores it, in canonical form. /// /// Public because a block editor has to write a line back after someone edits it in a /// widget that never showed them the marker. Rendering is trivial where PARSING is /// not, but it still belongs here: this is the file that decides what canonical looks /// like, and a caller inventing its own `- [x] ` would be a fourth opinion on it. pub fn render_item(text: &str, checked: bool) -> String { render_task_line("", '-', checked, text) } fn render_task_line(indent: &str, bullet: char, checked: bool, text: &str) -> String { // Always lowercase `x`, whatever was parsed: one canonical output is what makes // a round trip stable, so `- [X]` normalises the first time it is touched and // never again. let mark = if checked { 'x' } else { ' ' }; if text.is_empty() { format!("{indent}{bullet} [{mark}]") } else { format!("{indent}{bullet} [{mark}] {text}") } } /// The text of a line with its task marker removed, or the line as it was. /// /// For naming a note: a list-only note is named by its first item, and calling one /// "- [ ] milk" would be showing someone the storage instead of the note. pub fn strip_marker(line: &str) -> &str { match parse_task_line(line) { Some(t) => t.text, None => line, } } /// Every checklist item in `body`, in the order they appear. pub fn extract_items(body: &str) -> Vec { let mut out = Vec::new(); for (n, line) in body.split('\n').enumerate() { if let Some(t) = parse_task_line(line) { out.push(DerivedItem { text: t.text.to_string(), checked: t.checked, line: n as u32, }); } } out } /// Tick or untick the `index`-th item, keeping its indent, bullet and text. /// /// The only edit to an item that isn't typing in the body: adding, rewording and /// deleting one are all text edits, which every client makes in its editor. /// /// A body with fewer task lines than that is returned UNCHANGED rather than /// panicking: the index comes from a UI that may be a moment behind the store, and /// a stale tap should do nothing rather than take the app down. pub fn set_item_checked(body: &str, index: usize, checked: bool) -> String { let mut lines: Vec = body.split('\n').map(str::to_string).collect(); let Some(line) = lines .iter_mut() .filter(|l| parse_task_line(l.as_str()).is_some()) .nth(index) else { return body.to_string(); }; let Some(t) = parse_task_line(line.as_str()) else { return body.to_string(); }; let ticked = render_task_line(t.indent, t.bullet, checked, t.text); *line = ticked; lines.join("\n") } /// Add an item at the end of the body. /// /// Spaced exactly as `import_export.py:_note_markdown` writes a list — a blank line /// between prose and the list, and nothing between consecutive items. That is not /// cosmetic: the server migration folds existing rows into bodies using the same /// layout, so an export taken before the migration and one taken after have to /// agree byte for byte. /// /// `checked` is a parameter rather than always false because the two migrations that /// fold existing rows into bodies have to carry the state those rows were in. A new /// item from the UI passes false. pub fn append_item(body: &str, text: &str, checked: bool) -> String { let line = render_task_line("", '-', checked, text.trim()); let trimmed = body.trim_end_matches('\n'); if trimmed.trim().is_empty() { return line; } let follows_a_list = trimmed .split('\n') .next_back() .is_some_and(|l| parse_task_line(l).is_some()); if follows_a_list { format!("{trimmed}\n{line}") } else { format!("{trimmed}\n\n{line}") } } #[cfg(test)] mod tests { use super::*; /// The tag names in `body`, once each — what the shared fixture lists. fn tag_names(body: &str) -> Vec { let mut out = Vec::new(); for t in extract_tag_spans(body) { push_unique(&mut out, &t.name); } out } #[test] fn tags_basic() { assert_eq!( tag_names("a #todo and #Work-item_2 here"), vec!["todo", "Work-item_2"] ); } #[test] fn tags_require_letter_start_and_boundary() { // "#1" (digit) and an in-word "#" (email-ish) are not tags. assert_eq!(tag_names("#1 nope a#b no but #Yes ##no"), vec!["Yes"]); } #[test] fn empty_body() { assert!(extract_tag_spans("").is_empty()); } // ── tag spans, for the renderer that draws them in place ───────────────── #[test] fn tag_spans_carry_the_hash_and_the_line() { let spans = extract_tag_spans("buy milk #grocery\nand call #mom about #mom"); assert_eq!(spans.len(), 3); assert_eq!((spans[0].line, spans[0].start, spans[0].end), (0, 9, 17)); assert_eq!(spans[0].name, "grocery"); // Not deduped: two mentions are two pieces of text to colour. assert_eq!(spans[1].line, 1); assert_eq!(spans[2].name, "mom"); assert_eq!((spans[2].start, spans[2].end), (20, 24)); } #[test] fn tag_spans_are_utf16_offsets_not_char_indices() { // The emoji is ONE char and TWO UTF-16 code units. Kotlin and JS both index // the second way, so a char index would highlight one character too early. let spans = extract_tag_spans("🎁 #gift"); assert_eq!(spans.len(), 1); assert_eq!((spans[0].start, spans[0].end), (3, 8)); } // ── lifting standalone tags ────────────────────────────────────────────── // // The MIRROR of `split_body_tags` in the server's notes/tags.py, case for case. // A note lifted differently here than there would change under the operator the // moment it synced, so these are the cases that file agrees to. #[test] fn lifts_a_line_that_is_nothing_but_tags() { let (standalone, inline, body) = lift_standalone_tags("#todo\nreorganize the homepage"); assert_eq!(standalone, vec!["todo"]); assert!(inline.is_empty()); assert_eq!(body, "reorganize the homepage"); let (standalone, _, body) = lift_standalone_tags("needs a tauri app\n#todo"); assert_eq!(standalone, vec!["todo"]); assert_eq!(body, "needs a tauri app"); let (standalone, _, body) = lift_standalone_tags("#todo #work\nreal text"); assert_eq!(standalone, vec!["todo", "work"]); assert_eq!(body, "real text"); } /// The cases that must come back byte-identical. Getting any of these wrong /// destroys somebody's words, which is why the rule is the conservative one: /// a trailing tag is ambiguous and the text does not say which kind it is. #[test] fn leaves_a_tag_that_shares_its_line_with_words() { for prose in [ "remember to call #mom tomorrow", "buy milk #grocery", "#2024\nreal", ] { let (standalone, _, body) = lift_standalone_tags(prose); assert!(standalone.is_empty(), "{prose}"); assert_eq!(body, prose, "{prose}"); } } #[test] fn removing_a_line_leaves_no_hole() { let (_, _, body) = lift_standalone_tags("foo\n\n#todo\n\nbar"); assert_eq!(body, "foo\n\nbar"); } /// A `#tag` in a fence is a shell comment in somebody's snippet. It still becomes /// a label — it always has — but the line is never touched. #[test] fn never_touches_a_fenced_line() { let fenced = "code:\n```\n#!/bin/sh\n#deploy\n```\ndone"; let (standalone, inline, body) = lift_standalone_tags(fenced); assert!(standalone.is_empty()); assert_eq!(inline, vec!["deploy"]); assert_eq!(body, fenced); } /// Lifting would leave a blank card, which is worse than the duplication this /// removes. So the note keeps its text and its tags stay derived. #[test] fn will_not_blank_a_note_that_is_only_tags() { let (standalone, inline, body) = lift_standalone_tags("#todo"); assert!(standalone.is_empty()); assert_eq!(inline, vec!["todo"]); assert_eq!(body, "#todo"); } /// Appearing on its own line does NOT lift a tag also written in a sentence — the /// sentence still backs it, so deleting the sentence should still detach it. #[test] fn a_tag_still_in_prose_stays_derived() { let (standalone, inline, body) = lift_standalone_tags("#todo\nremember the #todo list"); assert!(standalone.is_empty()); assert_eq!(inline, vec!["todo"]); assert_eq!(body, "remember the #todo list"); } #[test] fn lifting_an_empty_body_is_a_no_op() { let (standalone, inline, body) = lift_standalone_tags(""); assert!(standalone.is_empty()); assert!(inline.is_empty()); assert_eq!(body, ""); } // ── checklist items ───────────────────────────────────────────────────── fn item(text: &str, checked: bool, line: u32) -> DerivedItem { DerivedItem { text: text.to_string(), checked, line, } } #[test] fn items_basic() { let body = "shopping\n\n- [ ] milk\n- [x] eggs"; assert_eq!( extract_items(body), vec![item("milk", false, 2), item("eggs", true, 3)] ); } #[test] fn items_may_sit_between_paragraphs() { // The whole reason the body owns the list: a table of rows could only ever // render after the prose. let body = "before\n- [ ] middle\nafter"; assert_eq!(extract_items(body), vec![item("middle", false, 1)]); } #[test] fn items_reject_near_misses() { // Each of these is prose, and each has been someone's bug report somewhere. for body in [ "-[ ] no space after the dash", "- [] empty brackets", "- [ ]no space after the brackets", "- [y] not a mark", "a [ ] mid sentence", "[ ] no bullet at all", ] { assert!(extract_items(body).is_empty(), "should be prose: {body}"); } } #[test] fn items_accept_star_bullets_and_indentation() { // `*` because markdown.ts already takes it for a plain bullet. let body = "* [ ] star\n - [x] indented"; assert_eq!( extract_items(body), vec![item("star", false, 0), item("indented", true, 1)] ); } #[test] fn an_empty_item_is_still_an_item() { // What pressing Enter on a list leaves behind. assert_eq!(extract_items("- [ ]"), vec![item("", false, 0)]); assert_eq!(extract_items("- [ ] "), vec![item("", false, 0)]); } #[test] fn uppercase_x_parses_and_normalises_on_rewrite() { assert_eq!(extract_items("- [X] done"), vec![item("done", true, 0)]); // Touching it once canonicalises it, and never again. assert_eq!(set_item_checked("- [X] done", 0, true), "- [x] done"); } #[test] fn checking_preserves_indent_bullet_and_text() { assert_eq!(set_item_checked(" * [ ] milk", 0, true), " * [x] milk"); assert_eq!(set_item_checked("- [x] milk", 0, false), "- [ ] milk"); } #[test] fn checking_addresses_items_not_lines() { let body = "note\n- [ ] a\nprose\n- [ ] b"; assert_eq!( set_item_checked(body, 1, true), "note\n- [ ] a\nprose\n- [x] b" ); } #[test] fn append_spaces_like_the_exporter() { // Prose then a blank line then the list — byte-for-byte what // import_export.py:_note_markdown writes, which is what the server // migration will fold existing rows into. assert_eq!(append_item("a note", "milk", false), "a note\n\n- [ ] milk"); // Nothing between consecutive items. let one = "a note\n\n- [ ] milk"; assert_eq!( append_item(one, "eggs", false), format!("{one}\n- [ ] eggs") ); // A list-only note starts at the first line. assert_eq!(append_item("", "milk", false), "- [ ] milk"); assert_eq!(append_item("\n\n", "milk", false), "- [ ] milk"); // Carries state, which is what the two migrations need of it. assert_eq!(append_item("", "done", true), "- [x] done"); } #[test] fn strip_marker_names_a_list_only_note() { assert_eq!(strip_marker("- [x] milk"), "milk"); assert_eq!(strip_marker("just prose"), "just prose"); } #[test] fn render_item_is_what_extract_reads_back() { assert_eq!(render_item("milk", false), "- [ ] milk"); assert_eq!(render_item("done", true), "- [x] done"); // An empty item has no trailing space, so a round trip does not grow it. assert_eq!(render_item("", false), "- [ ]"); let line = render_item("milk", true); assert_eq!(extract_items(&line), vec![item("milk", true, 0)]); } #[test] fn items_carry_the_line_they_sit_on() { let found = extract_items("a\n- [ ] x\nb\n- [x] y"); assert_eq!(found.iter().map(|i| i.line).collect::>(), vec![1, 3]); } #[test] fn a_stale_index_does_nothing() { // The index comes from a UI that may be a moment behind the store. A tap // that arrives late should be inert, not fatal. let body = "- [ ] only"; assert_eq!(set_item_checked(body, 7, true), body); } #[test] fn a_plain_body_is_returned_byte_identical() { let body = "just prose\nwith two lines"; assert_eq!(set_item_checked(body, 0, true), body); } #[test] fn round_trip_is_stable() { let body = "- [ ] a\n- [x] b\n- [ ] c"; let items = extract_items(body); // Ticking and unticking returns the original bytes. let touched = set_item_checked(&set_item_checked(body, 0, true), 0, false); assert_eq!(touched, body); assert_eq!(extract_items(&touched), items); } // ── the shared fixture ─────────────────────────────────────────────────── // // core/testdata/grammar.json is the one set of cases the server (pytest), the web // (vitest) and this file all run. The cases above stay as this file's own // reasoning; these are the ones the other languages have agreed to. fn fixture() -> serde_json::Value { serde_json::from_str(include_str!("../../testdata/grammar.json")) .expect("grammar.json parses") } fn strings(v: &serde_json::Value) -> Vec { v.as_array() .expect("an array") .iter() .map(|s| s.as_str().expect("a string").to_string()) .collect() } #[test] fn fixture_task_lines() { for case in fixture()["task_lines"].as_array().unwrap() { let line = case["line"].as_str().unwrap(); let got: Vec<(String, bool)> = extract_items(line) .into_iter() .map(|i| (i.text, i.checked)) .collect(); let want: Vec<(String, bool)> = match &case["item"] { serde_json::Value::Null => Vec::new(), item => vec![( item["text"].as_str().unwrap().to_string(), item["checked"].as_bool().unwrap(), )], }; assert_eq!(got, want, "line {line:?}"); } } #[test] fn fixture_rendered_items() { for case in fixture()["rendered_items"].as_array().unwrap() { let text = case["text"].as_str().unwrap(); let checked = case["checked"].as_bool().unwrap(); assert_eq!(render_item(text, checked), case["line"].as_str().unwrap()); } } #[test] fn fixture_tags() { for case in fixture()["tags"].as_array().unwrap() { let body = case["body"].as_str().unwrap(); assert_eq!(tag_names(body), strings(&case["tags"]), "body {body:?}"); } } #[test] fn fixture_lifts() { for case in fixture()["lifts"].as_array().unwrap() { let body = case["body"].as_str().unwrap(); let (standalone, inline, lifted) = lift_standalone_tags(body); assert_eq!( standalone, strings(&case["standalone"]), "standalone, body {body:?}" ); assert_eq!(inline, strings(&case["inline"]), "inline, body {body:?}"); assert_eq!( lifted, case["lifted"].as_str().unwrap(), "lifted, body {body:?}" ); } } }