lattice_syntax/syntax.rs
1//! `Syntax`: per-document tree-sitter state.
2//!
3//! Owns a `tree_sitter::Parser` + the latest cached `Tree` plus a
4//! shared [`crate::LangRegistry`] for the document's primary
5//! language and any injection targets. The hand-rolled native
6//! pipeline runs `highlights.scm` directly via
7//! `tree_sitter::QueryCursor`, walks each match into per-line
8//! `StyledSpan`s, and recursively highlights ranges captured by
9//! `injections.scm`.
10//!
11//! ## Reparse
12//!
13//! Two entry points (slice B.2):
14//!
15//! - [`Syntax::parse_at`]: full reparse. Used for cold-start /
16//! file-load / fallback. `Parser::parse(bytes, None)`.
17//! - [`Syntax::parse_at_with_edits`]: incremental reparse.
18//! Applies each `EditDelta` to the cached tree via
19//! `tree.edit()` then `Parser::parse(bytes, Some(&old_tree))`,
20//! so tree-sitter reuses unchanged subtrees. Falls back to
21//! full reparse if any guard fails (no cached tree,
22//! `from_version` mismatch, or post-edit byte-length mismatch
23//! between accumulated deltas and new source).
24//!
25//! Both methods stamp the resulting snapshot with a caller-
26//! supplied `text_version` so consumers (renderer / fold provider
27//! / completion) can compare freshness against
28//! `DocumentSnapshot::text_version`.
29//!
30//! ## Injections
31//!
32//! Markdown's grammar is split block / inline; the block parser's
33//! `injections.scm` injects the inline parser into paragraph
34//! content and the named language parser into fenced code blocks.
35//! Our injection callback (in `highlight_lines`) closes over the
36//! shared registry and looks up sibling configs by name -- so a
37//! ` ```rust ... ``` ` block in a markdown buffer gets rust
38//! highlighting, an autolink in a paragraph gets inline-markdown
39//! highlighting, etc.
40
41use std::sync::Arc;
42
43use streaming_iterator::StreamingIterator;
44use thiserror::Error;
45use tree_sitter::{InputEdit, Parser, Point, QueryCursor, Tree};
46
47use lattice_protocol::edit::EditDelta;
48
49use crate::lang::Lang;
50use crate::registry::LangRegistry;
51use crate::style::{Style, StyledSpan};
52
53/// Convert a [`lattice_protocol::edit::EditDelta`] (parser-agnostic
54/// edit shape) to a [`tree_sitter::InputEdit`] (parser-shaped edit
55/// the cached tree mutates by). Six casts + a struct constructor;
56/// runs in the noise floor (~1ns).
57///
58/// Free function rather than `From` impl because both types are
59/// foreign to this crate -- Rust's orphan rule blocks the trait
60/// impl. Lives in `lattice-syntax` (not `lattice-protocol`) so the
61/// protocol crate stays parser-agnostic. The fields map 1:1:
62/// `Position.line` -> `Point.row`, `Position.byte` -> `Point.column`
63/// (both are byte-within-line, despite the column-named field).
64pub fn edit_delta_to_input_edit(d: EditDelta) -> InputEdit {
65 InputEdit {
66 start_byte: d.start_byte as usize,
67 old_end_byte: d.old_end_byte as usize,
68 new_end_byte: d.new_end_byte as usize,
69 start_position: Point {
70 row: d.start_position.line as usize,
71 column: d.start_position.byte as usize,
72 },
73 old_end_position: Point {
74 row: d.old_end_position.line as usize,
75 column: d.old_end_position.byte as usize,
76 },
77 new_end_position: Point {
78 row: d.new_end_position.line as usize,
79 column: d.new_end_position.byte as usize,
80 },
81 }
82}
83
84#[derive(Debug, Error)]
85pub enum SyntaxError {
86 #[error("tree-sitter language error: {0}")]
87 Language(String),
88
89 #[error("language not registered: {0}")]
90 UnregisteredLang(String),
91}
92
93/// Read-only view of one parse result.
94///
95/// Holds everything downstream consumers (renderer / folds /
96/// completion / picker) need to compute highlights, walk the
97/// tree, or query symbols -- and nothing they don't. Cheap to
98/// clone (every non-trivial field is `Arc`-shareable: `Tree` is
99/// internally Arc'd by tree-sitter; `source` is `Arc<[u8]>`;
100/// `registry` is `Arc<LangRegistry>`).
101///
102/// Lives behind an `ArcSwap<SyntaxSnapshot>` inside
103/// [`crate::SyntaxHandle`] so the render thread reads the
104/// latest parse at hardware-floor speed while a worker task
105/// runs the next reparse off the UI thread (paramount goal #1).
106#[derive(Clone)]
107pub struct SyntaxSnapshot {
108 lang: Lang,
109 registry: Arc<LangRegistry>,
110 /// Last-parsed source bytes. `Arc<[u8]>` so cloning a
111 /// snapshot doesn't copy the buffer.
112 source: Arc<[u8]>,
113 /// H.3d (2026-06-04): memoized line→byte start table for
114 /// `source` (`line_starts[i]` = byte offset of line `i`; final
115 /// entry = `source.len()`). Recomputed once per source mutation
116 /// (via [`Self::set_source_bytes`]) instead of on every
117 /// `highlight_lines` call. The per-call rescan was an O(file)
118 /// term that defeated viewport-scoped highlight on large files
119 /// (it ran two full passes over the whole source even when the
120 /// query only needed a viewport window) — caught by the
121 /// `cells_worker_windowed_build` bench. `Arc<[usize]>` so cloning
122 /// a snapshot stays cheap.
123 line_starts: Arc<[usize]>,
124 /// Latest parse result. `None` until the first parse has
125 /// run. `Tree` is internally Arc'd by tree-sitter, so
126 /// cloning is cheap.
127 tree: Option<Tree>,
128 /// Document `text_version` this snapshot reflects. The
129 /// async path uses this to skip republishing identical
130 /// state.
131 text_version: u64,
132 /// H.2 (2026-06-04): inclusive source-line ranges whose syntax tree
133 /// differs from the snapshot this one was reparsed FROM
134 /// (`reparsed_from_version`), via `Tree::changed_ranges`. `None` =
135 /// full parse / unknown → consumers must treat the whole file as
136 /// dirty. `Some(empty)` = nothing changed. The cells worker uses this
137 /// to rebuild only the dirty rows on a reparse-completion republish —
138 /// but ONLY when its cached matrix's syntax version equals
139 /// `reparsed_from_version` (else the delta doesn't apply and it
140 /// full-rebuilds).
141 changed_lines: Option<Vec<(u32, u32)>>,
142 /// The `text_version` this snapshot's tree was reparsed FROM — i.e.
143 /// the version `changed_lines` is the delta against. Meaningful only
144 /// when `changed_lines` is `Some`.
145 reparsed_from_version: u64,
146 /// The `text_version` a **completed parse** produced — i.e. the
147 /// version this snapshot's `tree` genuinely reflects.
148 ///
149 /// Distinct from all three of its neighbours, and the distinction is
150 /// the point:
151 ///
152 /// - [`Self::text_version`] is the version of the *source text*, which
153 /// `try_apply_intermediate` advances after merely byte-shifting the
154 /// cached tree. A snapshot can carry the newest text with a tree that
155 /// was never parsed against it.
156 /// - [`Self::reparsed_from_version`] is a delta *baseline* for
157 /// `changed_lines`, so after an incremental reparse it holds the
158 /// version the parse started from — by construction never the one it
159 /// produced.
160 ///
161 /// So neither answers "is this tree a real parse of this exact text",
162 /// and two callers in `lattice-host` used to ask it of
163 /// `reparsed_from_version`, which cannot say yes after any incremental
164 /// reparse. `=` therefore silently reindented nothing after any edit
165 /// (reported 2026-08-16) and predictive indent silently fell to the
166 /// lexical bridge. Ask [`Self::tree_reflects`] instead.
167 parsed_text_version: u64,
168}
169
170impl std::fmt::Debug for SyntaxSnapshot {
171 fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
172 f.debug_struct("SyntaxSnapshot")
173 .field("lang", &self.lang)
174 .field("source_bytes", &self.source.len())
175 .field("tree_present", &self.tree.is_some())
176 .field("text_version", &self.text_version)
177 .finish_non_exhaustive()
178 }
179}
180
181pub struct Syntax {
182 /// Owned tree-sitter parser. `parse()` reuses it across edits
183 /// and passes the previous tree so tree-sitter's incremental
184 /// reparser kicks in. The parser instance itself is cheap to
185 /// keep around; the heavy state lives in the [`Tree`].
186 parser: Parser,
187 /// Read-only state -- exposed via [`Self::snapshot`] for
188 /// callers that want to share it cheaply (this is what
189 /// [`crate::SyntaxHandle`] publishes via `ArcSwap`).
190 inner: SyntaxSnapshot,
191}
192
193impl std::fmt::Debug for Syntax {
194 fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
195 f.debug_struct("Syntax")
196 .field("inner", &self.inner)
197 .finish_non_exhaustive()
198 }
199}
200
201impl Syntax {
202 /// Build a `Syntax` for the given language using a fresh standard
203 /// registry. Convenient when the caller doesn't already hold a
204 /// shared registry; for the App's hot path use
205 /// [`Self::for_language_with_registry`] so all documents share one
206 /// registry.
207 ///
208 /// `Lang::Plain` returns `None` because there's nothing to parse.
209 pub fn for_language(lang: Lang) -> Result<Option<Self>, SyntaxError> {
210 let registry = LangRegistry::standard()?;
211 Self::for_language_with_registry(lang, registry)
212 }
213
214 /// Build a `Syntax` borrowing from a shared registry. Multiple
215 /// documents (and the help-buffer system) all share one
216 /// `Arc<LangRegistry>`; per-document state stays in the
217 /// `Highlighter` + `source`.
218 pub fn for_language_with_registry(
219 lang: Lang,
220 registry: Arc<LangRegistry>,
221 ) -> Result<Option<Self>, SyntaxError> {
222 if matches!(lang, Lang::Plain) {
223 return Ok(None);
224 }
225 let Some(ts_lang) = registry.tree_sitter_language(lang.name()) else {
226 // Lang variant exists but no registered grammar for it -- fall
227 // back to no syntax (renderer treats it as plain text).
228 return Ok(None);
229 };
230 let mut parser = Parser::new();
231 // LG.3b: a wasm-backed grammar needs the parser to own a
232 // `WasmStore`, and this parser is long-lived (every later reparse
233 // uses it), so it gets its own rather than borrowing the pool.
234 // ~6 ms, once per buffer, only when the grammar is wasm-backed.
235 crate::wasm_grammar::set_language(&mut parser, &ts_lang).map_err(SyntaxError::Language)?;
236 Ok(Some(Self {
237 parser,
238 inner: SyntaxSnapshot {
239 lang,
240 registry,
241 source: Arc::from(Vec::<u8>::new()),
242 // H.3d: line table for the (empty) initial source.
243 line_starts: Arc::from(compute_line_starts(&[])),
244 tree: None,
245 text_version: 0,
246 changed_lines: None,
247 reparsed_from_version: 0,
248 parsed_text_version: 0,
249 },
250 }))
251 }
252
253 /// Borrow the read-only snapshot of the latest parse state.
254 /// Callers that want to share the snapshot cheaply (across
255 /// threads, into `ArcSwap`) clone it; readers that just need
256 /// one method call use the pass-through helpers below.
257 pub fn snapshot(&self) -> &SyntaxSnapshot {
258 &self.inner
259 }
260
261 /// Owned clone of the snapshot. Cheap (Arc-shareable
262 /// fields).
263 pub fn snapshot_owned(&self) -> SyntaxSnapshot {
264 self.inner.clone()
265 }
266}
267
268impl Syntax {
269 /// Convenience getter for `self.snapshot().lang()`. Kept on
270 /// `Syntax` so `&Syntax` callers (help-buffer markdown
271 /// highlighting, tests) don't have to thread `.snapshot()`
272 /// through.
273 pub fn lang(&self) -> Lang {
274 self.inner.lang()
275 }
276
277 /// Convenience getter for `self.snapshot().tree()`.
278 pub fn tree(&self) -> Option<&Tree> {
279 self.inner.tree()
280 }
281
282 /// Convenience getter for `self.snapshot().source()`.
283 pub fn source(&self) -> &[u8] {
284 self.inner.source()
285 }
286
287 /// Convenience getter for `self.snapshot().registry()`.
288 pub fn registry(&self) -> &LangRegistry {
289 self.inner.registry()
290 }
291
292 /// Convenience pass-through to
293 /// [`SyntaxSnapshot::cursor_in_string_scope`].
294 pub fn cursor_in_string_scope(&self, cursor_byte: usize) -> bool {
295 self.inner.cursor_in_string_scope(cursor_byte)
296 }
297
298 /// Convenience pass-through to
299 /// [`SyntaxSnapshot::collect_symbols`].
300 pub fn collect_symbols(&self) -> Vec<String> {
301 self.inner.collect_symbols()
302 }
303
304 /// Convenience pass-through to
305 /// [`SyntaxSnapshot::collect_symbol_locations`].
306 pub fn collect_symbol_locations(&self) -> Vec<(String, u32, u32)> {
307 self.inner.collect_symbol_locations()
308 }
309
310 /// Convenience pass-through to
311 /// [`SyntaxSnapshot::scope_at_cursor`].
312 pub fn scope_at_cursor(
313 &self,
314 line: u32,
315 col_byte: u32,
316 capture_suffix: &str,
317 ) -> Option<lattice_protocol::position::Range> {
318 self.inner.scope_at_cursor(line, col_byte, capture_suffix)
319 }
320
321 /// Convenience pass-through to
322 /// [`SyntaxSnapshot::highlight_lines`]. Note: takes `&self`
323 /// (the read API never needed `&mut`).
324 pub fn highlight_lines(
325 &self,
326 start_line: u32,
327 end_line: u32,
328 ) -> Result<Vec<Vec<StyledSpan>>, SyntaxError> {
329 self.inner.highlight_lines(start_line, end_line)
330 }
331
332 /// Convenience pass-through to
333 /// [`SyntaxSnapshot::highlight_lines_native`].
334 pub fn highlight_lines_native(
335 &self,
336 start_line: u32,
337 end_line: u32,
338 ) -> Result<Vec<Vec<StyledSpan>>, SyntaxError> {
339 self.inner.highlight_lines_native(start_line, end_line)
340 }
341
342 /// Replace the cached source and drive a tree-sitter (re)parse.
343 /// This is the only mutating call on `Syntax`; everything else
344 /// is read-only against the snapshot. Production callers should
345 /// route parse requests through [`crate::SyntaxHandle`] so the
346 /// parse runs off the UI thread (paramount goal #1).
347 pub fn parse(&mut self, source: &str) {
348 self.parse_at(source, self.inner.text_version.wrapping_add(1));
349 }
350
351 /// `parse` variant that also stamps a caller-supplied
352 /// `text_version` onto the resulting snapshot. The async
353 /// handle uses this so consumers can deduplicate stale
354 /// snapshots.
355 ///
356 /// Full reparse (passes `None` as the prior tree). The
357 /// incremental sibling [`Self::parse_at_with_edits`] is the
358 /// keystroke-path entry point; this method is the file-load
359 /// / cold-start / fallback path.
360 ///
361 /// `Parser::parse` returning `None` means cancellation, which
362 /// we don't trigger on this synchronous path. Keep the old
363 /// tree in that unlikely case rather than dropping it -- the
364 /// next parse round will retry.
365 pub fn parse_at(&mut self, source: &str, text_version: u64) {
366 let bytes = source.as_bytes();
367 let new_tree = self
368 .parser
369 .parse(bytes, None)
370 .or_else(|| self.inner.tree.take());
371 self.inner.set_source_bytes(bytes);
372 self.inner.tree = new_tree;
373 self.inner.text_version = text_version;
374 // H.2: a full reparse has no incremental old-vs-new diff, so the
375 // whole file is considered dirty (`None` ⇒ consumers full-rebuild).
376 self.inner.changed_lines = None;
377 self.inner.reparsed_from_version = text_version;
378 self.inner.parsed_text_version = text_version;
379 }
380
381 /// Incremental reparse: apply `edits` to the cached tree (sync
382 /// pre-step on the worker, ~500ns per edit), then parse with
383 /// the edited tree as the seed (~50µs floor on medium files
384 /// per §8.2). Falls back to [`Self::parse_at`] (full reparse)
385 /// if the guards on [`Self::try_apply_intermediate`] fail.
386 ///
387 /// Slice C.2 split this into two halves so the worker can
388 /// publish an intermediate snapshot between them -- byte
389 /// ranges shifted to track the edit but tree shape pre-parse,
390 /// so renderers see byte-aligned spans during the entire
391 /// parse window. This convenience runs both halves back-to-
392 /// back without an intermediate publish; the worker calls
393 /// the two halves directly with the publish in between.
394 pub fn parse_at_with_edits(
395 &mut self,
396 source: &str,
397 text_version: u64,
398 from_version: u64,
399 edits: &[EditDelta],
400 ) {
401 match self.try_apply_intermediate(source, text_version, from_version, edits) {
402 Ok(_) => self.reparse_with_cached_tree(from_version),
403 Err(_) => self.parse_at(source, text_version),
404 }
405 }
406
407 /// Slice C.2: try to apply `edits` to the cached tree and
408 /// update `source` + `text_version`, WITHOUT running
409 /// `Parser::parse`. Returns `Ok(())` if the resulting state
410 /// is valid for the worker to publish as an intermediate
411 /// snapshot (byte ranges shifted via `tree.edit` to track the
412 /// edits; tree shape is pre-parse for the changed regions).
413 /// Returns `Err(())` if any of:
414 ///
415 /// 1. **No cached tree** -- first reparse / worker recovered
416 /// from prior cancellation; nothing to seed with.
417 /// 2. **`from_version` mismatch** -- worker's tree isn't at
418 /// the version the edits expect to start from. Indicates
419 /// a dropped reparse request, file load, or document
420 /// replace; the cached tree's byte ranges don't match the
421 /// edits.
422 /// 3. **`edits` empty** -- nothing to apply.
423 /// 4. **Byte-length mismatch** -- accumulated edit delta
424 /// doesn't match the source-length delta. Catches dropped
425 /// or truncated edit lists.
426 ///
427 /// On `Err`, the caller falls back to [`Self::parse_at`]
428 /// (full reparse).
429 ///
430 /// Tree-sitter's failure mode for a malformed `InputEdit` is
431 /// a silently wrong tree, not a panic. The layered guards
432 /// here + the slice-B.4 parametrized parity matrix
433 /// (incremental == full reparse across 27 edit shapes ×
434 /// Rust / Python / JavaScript / Markdown) keep the silent-
435 /// corruption surface contained.
436 pub fn try_apply_intermediate(
437 &mut self,
438 source: &str,
439 text_version: u64,
440 from_version: u64,
441 edits: &[EditDelta],
442 ) -> Result<(), ()> {
443 let cached_at_baseline = self.inner.tree.is_some()
444 && self.inner.text_version == from_version
445 && !edits.is_empty();
446 if !cached_at_baseline {
447 return Err(());
448 }
449 let prior_len = self.inner.source.len() as i64;
450 let new_len = source.len() as i64;
451 let edit_delta_sum: i64 = edits
452 .iter()
453 .map(|d| (d.new_end_byte as i64) - (d.old_end_byte as i64))
454 .sum();
455 if prior_len + edit_delta_sum != new_len {
456 return Err(());
457 }
458 // All guards passed. Apply each edit to the cached tree
459 // in order. tree-sitter mutates `Tree` in place via
460 // `edit`; the mutation shifts every affected node's byte
461 // range to track the edit. Source + text_version updated
462 // so the snapshot's `source.len()` matches the edited
463 // tree's byte ranges.
464 let bytes = source.as_bytes();
465 if let Some(tree) = self.inner.tree.as_mut() {
466 for d in edits {
467 tree.edit(&edit_delta_to_input_edit(*d));
468 }
469 }
470 self.inner.set_source_bytes(bytes);
471 self.inner.text_version = text_version;
472 Ok(())
473 }
474
475 /// Slice C.2: re-parse using `self.inner.source` and the
476 /// cached tree (assumed to already have `tree.edit` applied
477 /// via [`Self::try_apply_intermediate`]) as seed. Updates
478 /// `self.inner.tree` to the freshly-parsed shape.
479 ///
480 /// Pairs with `try_apply_intermediate`: the worker calls
481 /// `try_apply_intermediate` (fast), publishes the intermediate
482 /// snapshot, then calls this to run the actual parse (slow).
483 pub fn reparse_with_cached_tree(&mut self, reparsed_from: u64) {
484 let bytes = self.inner.source.clone();
485 // Own the old tree so we can diff it against the new one after the
486 // parse (tree-sitter `Tree` is cheap to clone — internally Arc'd).
487 let old_tree = self.inner.tree.clone();
488 let new_tree = self
489 .parser
490 .parse(&*bytes, old_tree.as_ref())
491 .or_else(|| self.inner.tree.take());
492 // H.2: `old.changed_ranges(new)` gives exactly the byte ranges whose
493 // tree differs; the Range points carry rows, so map to inclusive
494 // source-line ranges. Only valid when both trees exist (else the
495 // whole file is dirty → `None`).
496 self.inner.changed_lines = match (&old_tree, &new_tree) {
497 (Some(old), Some(new)) => Some(
498 old.changed_ranges(new)
499 .map(|r| (r.start_point.row as u32, r.end_point.row as u32))
500 .collect(),
501 ),
502 _ => None,
503 };
504 self.inner.reparsed_from_version = reparsed_from;
505 // The parse ran against `inner.source` / `inner.text_version`, which
506 // `try_apply_intermediate` already advanced to the target version. So
507 // THIS is the point where the tree becomes a genuine parse of that
508 // text, and the only place besides `parse_at` that may say so.
509 self.inner.parsed_text_version = self.inner.text_version;
510 self.inner.tree = new_tree;
511 }
512}
513
514/// N.1.4b (2026-06-10): bridge the snapshot into the grammar
515/// dispatcher's tree-sitter text-object resolution. The grammar
516/// crate defines the `ScopeResolver` trait (cursor -> enclosing
517/// scope row range) and stays tree-sitter-agnostic; this impl
518/// forwards to the snapshot's existing
519/// [`SyntaxSnapshot::scope_at_cursor`] (N.1.0). The host coerces
520/// `Arc<SyntaxSnapshot>` to `Arc<dyn ScopeResolver + Send + Sync>`
521/// and threads it through `Document::dispatch_with_scope_resolver`
522/// so `daf` / `yic` etc. resolve against the live syntax tree off
523/// the UI thread (paramount #1: the snapshot is immutable, the
524/// query is bounded to the cursor's 1-byte window).
525impl lattice_grammar::ScopeResolver for SyntaxSnapshot {
526 fn scope_at(
527 &self,
528 line: u32,
529 col_byte: u32,
530 suffix: &str,
531 ) -> Option<lattice_protocol::position::Range> {
532 self.scope_at_cursor(line, col_byte, suffix)
533 }
534
535 // TSM.2: real tree walk -- forwards to the inherent
536 // `SyntaxSnapshot::scope_toward` below, mirroring `scope_at`'s
537 // forward to `scope_at_cursor`.
538 fn scope_toward(
539 &self,
540 line: u32,
541 col_byte: u32,
542 suffix: &str,
543 dir: lattice_grammar::NavDir,
544 boundary: lattice_grammar::NavBoundary,
545 count: u32,
546 ) -> Option<lattice_protocol::Position> {
547 self.scope_toward(line, col_byte, suffix, dir, boundary, count)
548 }
549}
550
551impl SyntaxSnapshot {
552 /// H.3d (2026-06-04): set `source` and recompute the memoized
553 /// `line_starts` together so the two never drift. Every source
554 /// mutation (full parse, incremental intermediate apply) routes
555 /// through here; `highlight_lines_via_query` then reads
556 /// `self.line_starts` instead of rescanning the whole source per
557 /// call (the O(file) term the windowed-build bench exposed).
558 fn set_source_bytes(&mut self, bytes: &[u8]) {
559 self.source = Arc::from(bytes.to_vec());
560 self.line_starts = Arc::from(compute_line_starts(&self.source));
561 }
562
563 /// The document language this snapshot was built for.
564 pub fn lang(&self) -> Lang {
565 self.lang
566 }
567
568 /// Latest parse result. `None` until the first parse has run.
569 pub fn tree(&self) -> Option<&Tree> {
570 self.tree.as_ref()
571 }
572
573 /// Cached source bytes that produced [`Self::tree`].
574 pub fn source(&self) -> &[u8] {
575 &self.source
576 }
577
578 /// Shared language registry. Query-driven consumers
579 /// (`compute_syntax_folds`, future textobjects / indents)
580 /// look up per-language compiled queries here.
581 pub fn registry(&self) -> &LangRegistry {
582 &self.registry
583 }
584
585 /// Document `text_version` this snapshot was built from.
586 /// Used by the async handle to skip republishing identical
587 /// state and by consumers that want to compare freshness
588 /// against a `DocumentSnapshot::text_version`.
589 pub fn text_version(&self) -> u64 {
590 self.text_version
591 }
592
593 /// H.2 (2026-06-04): inclusive source-line ranges whose syntax tree
594 /// changed between [`Self::reparsed_from_version`] and this snapshot,
595 /// from `Tree::changed_ranges`. `None` ⇒ full parse / unknown (treat
596 /// the whole file as dirty). The cells worker rebuilds only the
597 /// intersecting rows on a reparse-completion republish, gated on its
598 /// cached matrix's syntax version matching `reparsed_from_version`.
599 pub fn changed_lines(&self) -> Option<&[(u32, u32)]> {
600 self.changed_lines.as_deref()
601 }
602
603 /// The `text_version` this snapshot's tree was reparsed FROM — the
604 /// baseline [`Self::changed_lines`] is the delta against. Meaningful
605 /// only when `changed_lines()` is `Some`.
606 pub fn reparsed_from_version(&self) -> u64 {
607 self.reparsed_from_version
608 }
609
610 /// The `text_version` this snapshot's tree is a completed parse of.
611 /// See [`Self::parsed_text_version`] for why this is not any of the
612 /// other three version fields.
613 pub fn parsed_text_version(&self) -> u64 {
614 self.parsed_text_version
615 }
616
617 /// The stamp a **render cache** should key on: it changes when either
618 /// the text or the tree behind this snapshot changes.
619 ///
620 /// `text_version` alone is not it, and that is not a subtlety — it is the
621 /// stale-highlight bug. One edit produces TWO publishes from the syntax
622 /// worker (slice C.2): an intermediate whose byte ranges are shifted to
623 /// track the edit but whose tree shape is pre-parse, then the completed
624 /// reparse. **Both carry the same `text_version`.** A cache keyed on
625 /// `text_version` therefore cannot tell "shifted, not yet coloured" from
626 /// "parsed, colours ready": it invalidates on the intermediate, rebuilds
627 /// without highlights, and the completed parse moves nothing — so the
628 /// buffer stays at default colours until something drops the cache
629 /// entirely (`<C-l>`).
630 ///
631 /// `parsed_text_version` alone is not it either: it does not move on the
632 /// intermediate, and the intermediate is what makes unchanged content
633 /// paint at correct positions immediately. Both publishes must invalidate.
634 ///
635 /// Summing them gives a value that is strictly increasing across
636 /// `intermediate → parsed → next intermediate → …` and changes on every
637 /// publish. It is a cache key, not a version anyone should compare for
638 /// ordering — ask [`Self::tree_reflects`] when the question is "can I
639 /// trust this tree".
640 pub fn render_version(&self) -> u64 {
641 self.text_version.wrapping_add(self.parsed_text_version)
642 }
643
644 /// Whether this snapshot's tree is a completed parse of `text_version`.
645 ///
646 /// The question every consumer that wants to *trust the tree's structure*
647 /// is actually asking. A byte-shifted intermediate snapshot answers
648 /// `false` here even though it carries the right `text_version`, which is
649 /// the whole reason this predicate exists rather than a bare comparison
650 /// at each call site.
651 pub fn tree_reflects(&self, text_version: u64) -> bool {
652 self.parsed_text_version == text_version
653 }
654
655 /// True when the byte position `cursor_byte` falls inside a
656 /// string-literal node according to the cached tree.
657 /// Walks ancestors from the deepest descendant covering the
658 /// position and matches their `kind()` against a hardcoded
659 /// set of string-shaped node names that span the v1
660 /// language coverage (rust / python / javascript). Returns
661 /// `false` when no parse is cached, when the position falls
662 /// outside the source bytes, or when no ancestor matches.
663 ///
664 /// Used by the host's `gen:path` insert-completion source
665 /// (Phase 4.2.g.6 (2/2)) -- the spec triggers path-completion
666 /// inside string literals so file-path text is the only
667 /// place where `/` opens the popup.
668 pub fn cursor_in_string_scope(&self, cursor_byte: usize) -> bool {
669 // Hardcoded set across the v1 grammars. Source for the
670 // names: `tree-sitter-{rust,python,javascript}` node
671 // catalogues. New languages append entries here when
672 // they land. Substring matching ("kind contains
673 // 'string'") would catch more variants but also misfire
674 // on names like `string_concatenation` -- explicit list
675 // stays safer.
676 const STRING_NODE_KINDS: &[&str] = &[
677 "string",
678 "string_literal",
679 "raw_string_literal",
680 "byte_string_literal",
681 "template_string",
682 "string_fragment",
683 "interpolated_string_literal",
684 // IN.5: literal blocks, whose CONTENT is indentation-
685 // bearing text. Added for the indent engine, which must
686 // refuse to answer inside them — applying a block's
687 // structural indent to a heredoc or a YAML block scalar
688 // edits the data, and for `<<-EOF` can break terminator
689 // recognition. IN.4 claimed this protection before it
690 // existed; this is where it actually lands.
691 //
692 // Second consumer: `gen:path` insert-completion also reads
693 // this, so path completion now triggers inside heredocs
694 // and block scalars too. That is a reasonable place to
695 // want it (heredocs are usually shell text full of paths),
696 // not a regression to work around.
697 "heredoc_body",
698 "block_scalar",
699 // NOT `string_scalar`: YAML wraps every *plain* scalar in
700 // one, so including it would put the engine "inside a
701 // string" for essentially all YAML and disable indentation
702 // wholesale.
703 ];
704 let Some(tree) = self.tree.as_ref() else {
705 return false;
706 };
707 if cursor_byte > self.source.len() {
708 return false;
709 }
710 let root = tree.root_node();
711 let mut node = match root.descendant_for_byte_range(cursor_byte, cursor_byte) {
712 Some(n) => n,
713 None => return false,
714 };
715 loop {
716 let kind = node.kind();
717 if STRING_NODE_KINDS.contains(&kind) {
718 return true;
719 }
720 match node.parent() {
721 Some(p) => node = p,
722 None => return false,
723 }
724 }
725 }
726
727 /// Run the language's `symbols.scm` query against the cached
728 /// tree and return the deduped list of `@symbol`-captured
729 /// names (definition-position identifiers). Empty when:
730 /// no parse yet, language has no symbols query, or the tree
731 /// contains no matches.
732 ///
733 /// Phase 4.2.g.6 (1/2): the host-orchestrated
734 /// `gen:tree-sitter-symbol` insert-completion source calls
735 /// this once per popup-trigger; cost is O(tree-size) for
736 /// the cursor walk, which is sub-millisecond even on
737 /// large source files.
738 pub fn collect_symbols(&self) -> Vec<String> {
739 let Some(tree) = self.tree.as_ref() else {
740 return Vec::new();
741 };
742 let Some(query) = self.registry.symbols_query(self.lang.name()) else {
743 return Vec::new();
744 };
745 let mut cursor = QueryCursor::new();
746 let mut matches = cursor.matches(query, tree.root_node(), &self.source[..]);
747 let mut seen: std::collections::HashSet<String> = std::collections::HashSet::new();
748 let mut out: Vec<String> = Vec::new();
749 while let Some(m) = matches.next() {
750 for cap in m.captures {
751 let n = cap.node;
752 let start = n.start_byte();
753 let end = n.end_byte();
754 if end > self.source.len() || start >= end {
755 continue;
756 }
757 let Ok(text) = std::str::from_utf8(&self.source[start..end]) else {
758 continue;
759 };
760 if text.is_empty() {
761 continue;
762 }
763 if seen.insert(text.to_string()) {
764 out.push(text.to_string());
765 }
766 }
767 }
768 out
769 }
770
771 /// Like [`Self::collect_symbols`] but also reports each
772 /// symbol's location -- `(name, line, byte_column)` with
773 /// 0-based line and 0-based utf-8 byte column. Used by
774 /// the picker's `:picker outline` source so accept can
775 /// jump directly to the symbol definition. Dedup keys on
776 /// `(name, line, col)` to keep redundant captures
777 /// (function name appearing in both `@name` and `@definition`
778 /// captures of the same query) from doubling up.
779 pub fn collect_symbol_locations(&self) -> Vec<(String, u32, u32)> {
780 let Some(tree) = self.tree.as_ref() else {
781 return Vec::new();
782 };
783 let Some(query) = self.registry.symbols_query(self.lang.name()) else {
784 return Vec::new();
785 };
786 let mut cursor = QueryCursor::new();
787 let mut matches = cursor.matches(query, tree.root_node(), &self.source[..]);
788 let mut seen: std::collections::HashSet<(String, u32, u32)> =
789 std::collections::HashSet::new();
790 let mut out: Vec<(String, u32, u32)> = Vec::new();
791 while let Some(m) = matches.next() {
792 for cap in m.captures {
793 let n = cap.node;
794 let start = n.start_byte();
795 let end = n.end_byte();
796 if end > self.source.len() || start >= end {
797 continue;
798 }
799 let Ok(text) = std::str::from_utf8(&self.source[start..end]) else {
800 continue;
801 };
802 if text.is_empty() {
803 continue;
804 }
805 let pos = n.start_position();
806 let key = (text.to_string(), pos.row as u32, pos.column as u32);
807 if seen.insert(key.clone()) {
808 out.push(key);
809 }
810 }
811 }
812 // Stable sort by (line, col) so the popup reads top-
813 // down through the file.
814 out.sort_by(|a, b| a.1.cmp(&b.1).then_with(|| a.2.cmp(&b.2)));
815 out
816 }
817
818 /// Find the innermost tree-sitter scope containing the cursor whose
819 /// `textobjects.scm` capture name *ends with* `capture_suffix`
820 /// (e.g. `"function.outer"`, `"class.outer"`, `"block.outer"`) and
821 /// return its inclusive 0-based `(start_line, end_line)` source rows.
822 ///
823 /// "Innermost" = smallest byte span among the matching captures that
824 /// contain the cursor byte, so a cursor inside a closure nested in a
825 /// function resolves the closure for `"function.outer"`, and a cursor
826 /// on a statement resolves the tightest enclosing brace block for
827 /// `"block.outer"`. Returns `None` when no parse is cached, the
828 /// language ships no `textobjects.scm`, the cursor line is out of
829 /// range, or no matching capture contains the cursor.
830 ///
831 /// `line` / `col_byte` are 0-based; `col_byte` is a utf-8 byte offset
832 /// within the line (the snapshot's position convention). Powers
833 /// narrow-mode's tree-sitter targets (`:narrow-function` /
834 /// `:narrow-class` / `:narrow-block`, N.1.3); the plain `(u32, u32)`
835 /// return keeps multibuffer / narrow types out of this crate
836 /// (dependency direction: `lattice-multibuffer` -> `lattice-syntax`).
837 pub fn scope_at_cursor(
838 &self,
839 line: u32,
840 col_byte: u32,
841 capture_suffix: &str,
842 ) -> Option<lattice_protocol::position::Range> {
843 let tree = self.tree.as_ref()?;
844 let query = self.registry.textobjects_query(self.lang.name())?;
845 // Absolute cursor byte = line-start + column. `line_starts` holds
846 // `line_count + 1` entries (final = source length); an out-of-range
847 // line yields `None`.
848 let line_start = self.line_starts.get(line as usize).copied()?;
849 let cursor_byte = (line_start + col_byte as usize).min(self.source.len());
850
851 let names = query.capture_names();
852 let mut cursor = QueryCursor::new();
853 // Restrict to the 1-byte window at the cursor: a node `[start, end)`
854 // overlaps `[cursor, cursor+1)` iff `start <= cursor < end` -- exactly
855 // the half-open containment we want, so the explicit filter below is a
856 // belt-and-suspenders guard, not a second condition. Bounds the match
857 // set to enclosing scopes on large files.
858 cursor.set_byte_range(cursor_byte..cursor_byte.saturating_add(1));
859 let mut matches = cursor.matches(query, tree.root_node(), &self.source[..]);
860 // Smallest-span containing capture so far: (span_len, start_pos,
861 // end_pos). N.1.4c: track byte-precise positions (line + byte
862 // column), not just rows, so intra-line objects (`aa`/`ia`) are
863 // charwise-accurate.
864 let mut best: Option<(
865 usize,
866 lattice_protocol::Position,
867 lattice_protocol::Position,
868 )> = None;
869 while let Some(m) = matches.next() {
870 for cap in m.captures {
871 let name = names[cap.index as usize];
872 if !name.ends_with(capture_suffix) {
873 continue;
874 }
875 let n = cap.node;
876 let start = n.start_byte();
877 let end = n.end_byte();
878 // Half-open containment, matching tree-sitter node range
879 // semantics: the cursor on the construct's last token (e.g.
880 // `}` at byte `end - 1`) counts as inside; one past does not.
881 if cursor_byte < start || cursor_byte >= end {
882 continue;
883 }
884 let span = end - start;
885 // tree-sitter `Point.column` is a byte offset within the row,
886 // which is exactly `Position.byte`. `end_position` is one past
887 // the last byte (half-open), matching ProtoRange's exclusive end.
888 let sp = n.start_position();
889 let ep = n.end_position();
890 let start_pos = lattice_protocol::Position::new(sp.row as u32, sp.column as u32);
891 let end_pos = lattice_protocol::Position::new(ep.row as u32, ep.column as u32);
892 // Strictly-smaller replaces, so the first capture seen at a
893 // given span wins ties deterministically (query-pattern order).
894 let replace = match best {
895 Some((best_span, _, _)) => span < best_span,
896 None => true,
897 };
898 if replace {
899 best = Some((span, start_pos, end_pos));
900 }
901 }
902 }
903 best.map(|(_, s, e)| lattice_protocol::position::Range::new(s, e))
904 }
905
906 /// The `count`-th node whose `textobjects.scm` capture name *ends
907 /// with* `suffix`, scanning in `dir`, targeting the node's
908 /// `boundary`. Backs the structural motions (`]f`/`[c`/…, TSM.4)
909 /// via [`lattice_grammar::ScopeResolver::scope_toward`].
910 ///
911 /// Respects the enclosing-object rule (treesitter-motions.md
912 /// §4.1): `(Forward, Start)` and `(Backward, End)` skip the object
913 /// the cursor is currently inside (candidates strictly past the
914 /// cursor byte); `(Backward, Start)` and `(Forward, End)` may land
915 /// on the current object's own boundary (candidates at-or-past the
916 /// cursor byte), so e.g. jumping backward to a function start from
917 /// inside its body lands on that function's own `fn` keyword
918 /// rather than skipping past it.
919 ///
920 /// `NavBoundary::End` returns `end_position` (one past the last
921 /// byte), matching [`Self::scope_at_cursor`]'s half-open
922 /// convention -- the operator's inclusive-end handling adds the
923 /// final byte back for `d]F`-style deletes.
924 ///
925 /// Returns `None` gracefully (heuristic #5, no-op) when: there is
926 /// no cached tree, the language ships no `textobjects.scm`, the
927 /// cursor line is out of range, or there are fewer than `count`
928 /// matching candidates in `dir` -- never panics.
929 ///
930 /// `line` / `col_byte` are 0-based, `col_byte` a utf-8 byte offset
931 /// within the line (the snapshot's position convention, same as
932 /// [`Self::scope_at_cursor`]).
933 pub fn scope_toward(
934 &self,
935 line: u32,
936 col_byte: u32,
937 suffix: &str,
938 dir: lattice_grammar::NavDir,
939 boundary: lattice_grammar::NavBoundary,
940 count: u32,
941 ) -> Option<lattice_protocol::Position> {
942 use lattice_grammar::{NavBoundary, NavDir};
943 if count == 0 {
944 return None;
945 }
946 let tree = self.tree.as_ref()?;
947 let query = self.registry.textobjects_query(self.lang.name())?;
948 let line_start = self.line_starts.get(line as usize).copied()?;
949 let cursor_byte = (line_start + col_byte as usize).min(self.source.len());
950
951 // Restrict the query to the half of the file we scan (perf:
952 // bounds the match set on large files -- paramount #1).
953 let mut cursor = QueryCursor::new();
954 match dir {
955 NavDir::Forward => cursor.set_byte_range(cursor_byte..self.source.len()),
956 NavDir::Backward => cursor.set_byte_range(0..cursor_byte.saturating_add(1)),
957 };
958
959 let names = query.capture_names();
960 // Collect candidate boundary bytes + their (row, col) positions.
961 let mut cands: Vec<(usize, lattice_protocol::Position)> = Vec::new();
962 let mut matches = cursor.matches(query, tree.root_node(), &self.source[..]);
963 while let Some(m) = matches.next() {
964 for cap in m.captures {
965 if !names[cap.index as usize].ends_with(suffix) {
966 continue;
967 }
968 let n = cap.node;
969 let (b, pt) = match boundary {
970 NavBoundary::Start => (n.start_byte(), n.start_position()),
971 NavBoundary::End => (n.end_byte(), n.end_position()),
972 };
973 // Enclosing-object rule (treesitter-motions.md §4.1): all four
974 // arms compare STRICTLY against the cursor byte. When the cursor
975 // sits inside an object's body the enclosing object is still
976 // reached (its start < cursor / end > cursor). The strictness
977 // matters only when the cursor sits exactly ON a boundary byte
978 // (e.g. right after `]f` landed on a function start): there the
979 // current object is NOT re-selected, so the `]f`->`[f` round-trip
980 // moves to the previous object instead of no-oping.
981 let keep = match (dir, boundary) {
982 (NavDir::Forward, NavBoundary::Start) => b > cursor_byte,
983 (NavDir::Backward, NavBoundary::End) => b < cursor_byte,
984 (NavDir::Backward, NavBoundary::Start) => b < cursor_byte,
985 (NavDir::Forward, NavBoundary::End) => b > cursor_byte,
986 };
987 if keep {
988 cands.push((
989 b,
990 lattice_protocol::Position::new(pt.row as u32, pt.column as u32),
991 ));
992 }
993 }
994 }
995 // Sort in the direction of travel; dedup by byte (a node can be
996 // captured by multiple patterns).
997 cands.sort_by_key(|(b, _)| *b);
998 cands.dedup_by_key(|(b, _)| *b);
999 let ordered: Vec<_> = match dir {
1000 NavDir::Forward => cands,
1001 NavDir::Backward => cands.into_iter().rev().collect(),
1002 };
1003 ordered
1004 .get((count as usize).saturating_sub(1))
1005 .map(|(_, p)| *p)
1006 }
1007
1008 /// Compute styled spans for each line in `[start_line, end_line)`.
1009 /// `start_line` and `end_line` are 0-based and clamped to the source's
1010 /// line count.
1011 ///
1012 /// Returns one `Vec<StyledSpan>` per line in the requested range. Spans
1013 /// use line-relative byte offsets (consistent with the renderer's
1014 /// existing assumption).
1015 ///
1016 /// As of Step 4 this is a thin pass-through to the hand-rolled
1017 /// native pipeline ([`Self::highlight_lines_native`]); the
1018 /// legacy `tree_sitter_highlight::Highlighter`-based path was
1019 /// removed when its dependency was dropped from the workspace.
1020 pub fn highlight_lines(
1021 &self,
1022 start_line: u32,
1023 end_line: u32,
1024 ) -> Result<Vec<Vec<StyledSpan>>, SyntaxError> {
1025 self.highlight_lines_native(start_line, end_line)
1026 }
1027
1028 /// Hand-rolled highlighter that runs `highlights.scm` directly
1029 /// against `Self::tree()` via `tree_sitter::QueryCursor`,
1030 /// bypassing `tree_sitter_highlight::Highlighter`. This is the
1031 /// Step 3 deliverable of the Option B migration: one parse per
1032 /// keystroke (the parser already feeds folds, future textobjects,
1033 /// indents, etc.) instead of the streaming highlighter's parallel
1034 /// reparse.
1035 ///
1036 /// As of Step 3b this method also recursively highlights ranges
1037 /// captured by `injections.scm`: markdown's block→inline path
1038 /// (so `**bold**` inside a paragraph picks up Bold styling) and
1039 /// fenced-code blocks (so ` ```rust ... ``` ` inside a markdown
1040 /// buffer reuses the rust highlights). Recursion is bounded
1041 /// (one level deep per call site -- markdown_inline has no
1042 /// further injections we honour today).
1043 pub fn highlight_lines_native(
1044 &self,
1045 start_line: u32,
1046 end_line: u32,
1047 ) -> Result<Vec<Vec<StyledSpan>>, SyntaxError> {
1048 self.highlight_lines_via_query(start_line, end_line)
1049 }
1050
1051 /// Highlight one injection, returning a per-byte `Option<Style>`
1052 /// vector aligned with `inj.range` (slot 0 = inj.range.start).
1053 /// Returns `None` when the injected language has no registered
1054 /// config -- the caller leaves the parent's styling in place.
1055 fn highlight_injection(&self, inj: &Injection) -> Option<Vec<Option<Style>>> {
1056 let lang_config = self.registry.lookup(&inj.language)?;
1057 // Parse the injected content range with the target
1058 // language's parser. We slice the source bytes so byte
1059 // offsets in the resulting tree are RELATIVE to the
1060 // injection (slot 0 = inj.range.start in our caller).
1061 let content = &self.source[inj.range.clone()];
1062 // LG.3b: per injection, per highlight call — so a wasm-backed
1063 // injected grammar borrows the thread-local store instead of
1064 // building one. Twenty fenced blocks would otherwise cost
1065 // 20 x ~6 ms on EVERY highlight.
1066 let mut parser = Parser::new();
1067 let tree =
1068 crate::wasm_grammar::with_pooled_store(&mut parser, &lang_config.language, |p| {
1069 p.parse(content, None)
1070 })??;
1071
1072 // Run the injected language's highlights query. Capture
1073 // resolution mirrors the parent path (later pattern wins,
1074 // smaller range tie-breaks).
1075 let query = &lang_config.highlights;
1076 let styles = &lang_config.highlight_styles;
1077 let mut cursor = QueryCursor::new();
1078 let mut matches = cursor.matches(query, tree.root_node(), content);
1079 let mut captures: Vec<(usize, usize, Style, usize)> = Vec::new();
1080 while let Some(m) = matches.next() {
1081 for cap in m.captures {
1082 let style = styles
1083 .get(cap.index as usize)
1084 .copied()
1085 .unwrap_or(Style::Default);
1086 let n = cap.node;
1087 captures.push((n.start_byte(), n.end_byte(), style, m.pattern_index));
1088 }
1089 }
1090 captures.sort_by(|a, b| {
1091 b.3.cmp(&a.3)
1092 .then_with(|| {
1093 let len_a = a.1.saturating_sub(a.0);
1094 let len_b = b.1.saturating_sub(b.0);
1095 len_a.cmp(&len_b)
1096 })
1097 .then_with(|| a.0.cmp(&b.0))
1098 });
1099
1100 let len = content.len();
1101 let mut out: Vec<Option<Style>> = vec![None; len];
1102 for (s, e, style, _) in &captures {
1103 let s = (*s).min(len);
1104 let e = (*e).min(len);
1105 for slot in &mut out[s..e] {
1106 if slot.is_none() {
1107 *slot = Some(*style);
1108 }
1109 }
1110 }
1111 // Recursive injections (e.g. markdown_block emitting
1112 // markdown_inline content) -- if the injected language
1113 // itself has an injections query, recurse one more level.
1114 if let Some(inj_query) = lang_config.injections.as_ref() {
1115 // The "source" for nested injection is the slice we
1116 // just parsed; call the standalone collector with
1117 // window=[0, len).
1118 let nested = collect_injections(inj_query, &tree, content, 0, len);
1119 for n_inj in nested {
1120 // Copy the slice into a fresh Vec for the recursive
1121 // helper; we synthesise a one-shot Syntax-like view
1122 // by reusing self.registry (the parser+tree are
1123 // local to this fn).
1124 if let Some(inner) = self.highlight_injection_in(content, &n_inj) {
1125 let s = n_inj.range.start.min(len);
1126 let e = n_inj.range.end.min(len);
1127 let inner_len = inner.len();
1128 for (i, slot) in out[s..e].iter_mut().enumerate() {
1129 if i >= inner_len {
1130 break;
1131 }
1132 if let Some(st) = inner[i] {
1133 *slot = Some(st);
1134 }
1135 }
1136 }
1137 }
1138 }
1139 Some(out)
1140 }
1141
1142 /// Inner-injection helper. Same shape as
1143 /// [`Self::highlight_injection`] but takes an explicit byte
1144 /// slice rather than slicing into `self.source`. Used only by
1145 /// the recursive injection path so a markdown paragraph that
1146 /// injects markdown_inline can see further injections (rare
1147 /// but possible).
1148 fn highlight_injection_in(
1149 &self,
1150 outer_source: &[u8],
1151 inj: &Injection,
1152 ) -> Option<Vec<Option<Style>>> {
1153 let lang_config = self.registry.lookup(&inj.language)?;
1154 let content = &outer_source[inj.range.clone()];
1155 // LG.3b: per injection, per highlight call — so a wasm-backed
1156 // injected grammar borrows the thread-local store instead of
1157 // building one. Twenty fenced blocks would otherwise cost
1158 // 20 x ~6 ms on EVERY highlight.
1159 let mut parser = Parser::new();
1160 let tree =
1161 crate::wasm_grammar::with_pooled_store(&mut parser, &lang_config.language, |p| {
1162 p.parse(content, None)
1163 })??;
1164 let query = &lang_config.highlights;
1165 let styles = &lang_config.highlight_styles;
1166 let mut cursor = QueryCursor::new();
1167 let mut matches = cursor.matches(query, tree.root_node(), content);
1168 let mut captures: Vec<(usize, usize, Style, usize)> = Vec::new();
1169 while let Some(m) = matches.next() {
1170 for cap in m.captures {
1171 let style = styles
1172 .get(cap.index as usize)
1173 .copied()
1174 .unwrap_or(Style::Default);
1175 let n = cap.node;
1176 captures.push((n.start_byte(), n.end_byte(), style, m.pattern_index));
1177 }
1178 }
1179 captures.sort_by(|a, b| {
1180 b.3.cmp(&a.3)
1181 .then_with(|| {
1182 let len_a = a.1.saturating_sub(a.0);
1183 let len_b = b.1.saturating_sub(b.0);
1184 len_a.cmp(&len_b)
1185 })
1186 .then_with(|| a.0.cmp(&b.0))
1187 });
1188 let len = content.len();
1189 let mut out: Vec<Option<Style>> = vec![None; len];
1190 for (s, e, style, _) in &captures {
1191 let s = (*s).min(len);
1192 let e = (*e).min(len);
1193 for slot in &mut out[s..e] {
1194 if slot.is_none() {
1195 *slot = Some(*style);
1196 }
1197 }
1198 }
1199 Some(out)
1200 }
1201
1202 /// The native query-cursor pipeline. Separated so Step 3b can
1203 /// call it recursively for injected ranges with a per-call
1204 /// language override.
1205 fn highlight_lines_via_query(
1206 &self,
1207 start_line: u32,
1208 end_line: u32,
1209 ) -> Result<Vec<Vec<StyledSpan>>, SyntaxError> {
1210 if end_line <= start_line {
1211 return Ok(Vec::new());
1212 }
1213 let Some(tree) = self.tree.as_ref() else {
1214 return Ok((0..(end_line - start_line)).map(|_| Vec::new()).collect());
1215 };
1216 // H.3d: read the memoized line table (recomputed once per
1217 // source mutation) instead of rescanning the whole source on
1218 // every call — this is what keeps highlight O(window) on
1219 // large files.
1220 let line_starts: &[usize] = &self.line_starts;
1221 let total_lines = line_starts.len().saturating_sub(1).max(1) as u32;
1222 let end_line = end_line.min(total_lines + 1);
1223 if start_line >= end_line {
1224 return Ok(Vec::new());
1225 }
1226 let mut result: Vec<Vec<StyledSpan>> =
1227 (0..(end_line - start_line)).map(|_| Vec::new()).collect();
1228 let query = self
1229 .registry
1230 .highlights_query(self.lang.name())
1231 .ok_or_else(|| SyntaxError::UnregisteredLang(self.lang.name().to_string()))?;
1232 let styles = self
1233 .registry
1234 .highlight_styles(self.lang.name())
1235 .ok_or_else(|| SyntaxError::UnregisteredLang(self.lang.name().to_string()))?;
1236 let priorities = self
1237 .registry
1238 .highlight_priorities(self.lang.name())
1239 .ok_or_else(|| SyntaxError::UnregisteredLang(self.lang.name().to_string()))?;
1240
1241 // Restrict the query to the byte window we'll actually use.
1242 // tree-sitter's `QueryCursor::set_byte_range` is a hint; the
1243 // cursor still returns matches that overlap the window, so
1244 // captures whose ranges straddle the window get clipped at
1245 // distribute time (`distribute_span_across_lines` already
1246 // filters by line range).
1247 let window_start = line_starts.get(start_line as usize).copied().unwrap_or(0);
1248 let window_end = line_starts
1249 .get(end_line as usize)
1250 .copied()
1251 .unwrap_or(self.source.len());
1252 let mut cursor = QueryCursor::new();
1253 cursor.set_byte_range(window_start..window_end);
1254
1255 // Collect captures into (start, end, style, pattern_index).
1256 // Overlap resolution: later pattern wins -- the convention
1257 // tree-sitter highlights queries follow (more specific
1258 // patterns come later in the file; `(class_definition
1259 // name: (identifier) @constructor)` lives below the
1260 // generic `(identifier) @variable`). This matches what
1261 // `tree_sitter_highlight` does, including the case where
1262 // the winning capture's name isn't in CAPTURE_NAMES (e.g.
1263 // `@constructor`): the slot is "claimed" with Style::Default
1264 // and no visible span is emitted, which suppresses the
1265 // generic `@variable` capture too.
1266 let mut captures: Vec<(usize, usize, Style, usize)> = Vec::new();
1267 let mut matches = cursor.matches(query, tree.root_node(), &self.source[..]);
1268 while let Some(m) = matches.next() {
1269 for cap in m.captures {
1270 let style = styles
1271 .get(cap.index as usize)
1272 .copied()
1273 .unwrap_or(Style::Default);
1274 let n = cap.node;
1275 captures.push((n.start_byte(), n.end_byte(), style, m.pattern_index));
1276 }
1277 }
1278 // Sort so the FIRST-write-wins paint loop produces the
1279 // intended overrides: highest pattern_index first (later
1280 // patterns more specific). Tie-break by smallest range
1281 // first (a child capture inside a same-pattern parent
1282 // should still claim its own bytes), then by start byte
1283 // for determinism.
1284 captures.sort_by(|a, b| {
1285 b.3.cmp(&a.3) // pattern_index DESC
1286 .then_with(|| {
1287 let len_a = a.1.saturating_sub(a.0);
1288 let len_b = b.1.saturating_sub(b.0);
1289 len_a.cmp(&len_b) // range size ASC
1290 })
1291 .then_with(|| a.0.cmp(&b.0)) // start byte ASC
1292 });
1293 let _ = priorities; // priority table unused for now; kept
1294 // on the registry for the eventual
1295 // tie-break refinement / locals work.
1296
1297 // Per-byte style array for the window, then convert to
1298 // line-relative spans. The array is at most O(window_bytes)
1299 // memory, which is bounded by `viewport_height * line_width`
1300 // in the renderer's typical call shape.
1301 let win_len = window_end.saturating_sub(window_start);
1302 let mut byte_styles: Vec<Option<Style>> = vec![None; win_len];
1303 for (s, e, style, _) in &captures {
1304 let s_local = s.saturating_sub(window_start).min(win_len);
1305 let e_local = e.saturating_sub(window_start).min(win_len);
1306 for slot in &mut byte_styles[s_local..e_local] {
1307 if slot.is_none() {
1308 *slot = Some(*style);
1309 }
1310 }
1311 }
1312
1313 // Step 3b: recursively process injection captures and
1314 // overwrite the parent's per-byte styles within the
1315 // injected ranges. Outer markdown captures inside a
1316 // ` ```rust { ... } ``` ` block get replaced by the rust
1317 // pipeline's spans; same for `markdown_inline` injected
1318 // into paragraph content.
1319 if let Some(inj_query) = self.registry.injections_query(self.lang.name()) {
1320 let injections =
1321 collect_injections(inj_query, tree, &self.source[..], window_start, window_end);
1322 for inj in injections {
1323 if let Some(inner_styles) = self.highlight_injection(&inj) {
1324 let s_local = inj.range.start.saturating_sub(window_start).min(win_len);
1325 let e_local = inj.range.end.saturating_sub(window_start).min(win_len);
1326 let inner_len = inner_styles.len();
1327 for (i, slot) in byte_styles[s_local..e_local].iter_mut().enumerate() {
1328 if i >= inner_len {
1329 break;
1330 }
1331 // Injected spans always override -- once a
1332 // language injection claims a byte, it owns
1333 // the styling there.
1334 if let Some(style) = inner_styles[i] {
1335 *slot = Some(style);
1336 }
1337 }
1338 }
1339 }
1340 }
1341
1342 // Walk byte_styles, emitting (start, end, style) runs and
1343 // distributing each across the line slices the renderer
1344 // expects. Default-claimed slots (`Some(Style::Default)`)
1345 // count as "no visible span" -- the legacy highlighter
1346 // emits no event for them.
1347 let mut i = 0usize;
1348 while i < byte_styles.len() {
1349 let Some(style) = byte_styles[i] else {
1350 i += 1;
1351 continue;
1352 };
1353 if matches!(style, Style::Default) {
1354 i += 1;
1355 continue;
1356 }
1357 let mut j = i + 1;
1358 while j < byte_styles.len() && byte_styles[j] == Some(style) {
1359 j += 1;
1360 }
1361 distribute_span_across_lines(
1362 window_start + i,
1363 window_start + j,
1364 style,
1365 line_starts,
1366 start_line,
1367 end_line,
1368 &mut result,
1369 );
1370 i = j;
1371 }
1372 Ok(result)
1373 }
1374}
1375
1376/// One injection candidate from `injections.scm`: a byte range of
1377/// content + the target language's name. Markdown produces these
1378/// in two shapes -- `(@injection.content @injection.language)`
1379/// pairs (fenced code blocks) and `@injection.content` alone with
1380/// `#set! injection.language "..."` directives (paragraphs →
1381/// markdown_inline).
1382struct Injection {
1383 range: std::ops::Range<usize>,
1384 language: String,
1385}
1386
1387/// Walk every match of the injections query, extract `(content,
1388/// language)` pairs, and clip them to the visible window so we
1389/// don't re-parse content outside the requested viewport.
1390fn collect_injections(
1391 query: &tree_sitter::Query,
1392 tree: &Tree,
1393 source: &[u8],
1394 window_start: usize,
1395 window_end: usize,
1396) -> Vec<Injection> {
1397 let mut cursor = QueryCursor::new();
1398 cursor.set_byte_range(window_start..window_end);
1399 let mut matches = cursor.matches(query, tree.root_node(), source);
1400 let names = query.capture_names();
1401 let mut out = Vec::new();
1402 while let Some(m) = matches.next() {
1403 // Find the content + (optional) language captures within
1404 // this match. Content is required; language can come from
1405 // either a `@injection.language` capture or a `#set!
1406 // injection.language "..."` directive on the pattern.
1407 let mut content_range: Option<std::ops::Range<usize>> = None;
1408 let mut explicit_lang: Option<String> = None;
1409 for cap in m.captures {
1410 let name = names[cap.index as usize];
1411 match name {
1412 "injection.content" => {
1413 let n = cap.node;
1414 content_range = Some(n.start_byte()..n.end_byte());
1415 }
1416 "injection.language" => {
1417 let n = cap.node;
1418 if let Ok(text) = std::str::from_utf8(&source[n.start_byte()..n.end_byte()]) {
1419 explicit_lang = Some(text.trim().to_string());
1420 }
1421 }
1422 _ => {}
1423 }
1424 }
1425 let Some(content_range) = content_range else {
1426 continue;
1427 };
1428 // Skip injections that don't intersect the visible window
1429 // -- their spans wouldn't appear in the result anyway.
1430 if content_range.end <= window_start || content_range.start >= window_end {
1431 continue;
1432 }
1433 // Resolve the target language: explicit capture wins; else
1434 // walk the pattern's `#set!` directives.
1435 let language = explicit_lang.or_else(|| {
1436 query
1437 .property_settings(m.pattern_index)
1438 .iter()
1439 .find(|p| p.key.as_ref() == "injection.language")
1440 .and_then(|p| p.value.as_ref().map(|v| v.to_string()))
1441 });
1442 let Some(language) = language else { continue };
1443 out.push(Injection {
1444 range: content_range,
1445 language,
1446 });
1447 }
1448 out
1449}
1450
1451/// Compute the byte offset where each line starts. The returned vec has
1452/// `line_count + 1` entries; the last is `source.len()` (a sentinel).
1453fn compute_line_starts(source: &[u8]) -> Vec<usize> {
1454 let mut starts = Vec::with_capacity(source.iter().filter(|b| **b == b'\n').count() + 2);
1455 starts.push(0);
1456 for (i, b) in source.iter().enumerate() {
1457 if *b == b'\n' {
1458 starts.push(i + 1);
1459 }
1460 }
1461 starts.push(source.len());
1462 starts
1463}
1464
1465/// Place a styled span into the per-line result vector, splitting at
1466/// newline boundaries and clipping to the requested `[start_line, end_line)`
1467/// window.
1468fn distribute_span_across_lines(
1469 span_start: usize,
1470 span_end: usize,
1471 style: Style,
1472 line_starts: &[usize],
1473 range_start_line: u32,
1474 range_end_line: u32,
1475 out: &mut [Vec<StyledSpan>],
1476) {
1477 if span_end <= span_start {
1478 return;
1479 }
1480 let mut byte = span_start;
1481 while byte < span_end {
1482 let line = byte_to_line(line_starts, byte);
1483 let line_start_byte = line_starts.get(line).copied().unwrap_or(0);
1484 let next_line_start = line_starts.get(line + 1).copied().unwrap_or(usize::MAX);
1485 let line_end_for_span = next_line_start.min(span_end);
1486 if (line as u32) >= range_start_line && (line as u32) < range_end_line {
1487 let i = (line as u32 - range_start_line) as usize;
1488 if let Some(per_line) = out.get_mut(i) {
1489 let line_relative_start = byte - line_start_byte;
1490 let mut line_relative_end = line_end_for_span - line_start_byte;
1491 // Trim the trailing newline so styled spans don't bleed
1492 // past the last visible character on the line.
1493 if next_line_start <= span_end && line_relative_end > 0 {
1494 line_relative_end -= 1;
1495 }
1496 if line_relative_end > line_relative_start {
1497 per_line.push(StyledSpan {
1498 start: line_relative_start,
1499 end: line_relative_end,
1500 style,
1501 });
1502 }
1503 }
1504 }
1505 byte = line_end_for_span;
1506 if byte == next_line_start && byte < span_end {
1507 // Skip the newline byte and continue with the next line.
1508 byte = next_line_start;
1509 }
1510 }
1511}
1512
1513fn byte_to_line(line_starts: &[usize], byte: usize) -> usize {
1514 match line_starts.binary_search(&byte) {
1515 Ok(i) => i,
1516 Err(i) => i.saturating_sub(1),
1517 }
1518}
1519
1520#[cfg(test)]
1521mod tests {
1522 #![allow(clippy::unwrap_used, clippy::panic)]
1523 use super::*;
1524
1525 #[test]
1526 fn syntax_for_plain_returns_none() {
1527 let s = Syntax::for_language(Lang::Plain).unwrap();
1528 assert!(s.is_none());
1529 }
1530
1531 #[test]
1532 fn rust_syntax_exposes_parsed_tree() {
1533 // Step 1 invariant: every successful `parse()` populates
1534 // `tree()` so future query consumers (folds.scm,
1535 // textobjects.scm, indents.scm) can read from the same
1536 // parse the highlighter walks.
1537 let mut s = Syntax::for_language(Lang::Rust).unwrap().unwrap();
1538 assert!(s.tree().is_none(), "tree should be empty before parse");
1539 s.parse("fn main() {}");
1540 let tree = s.tree().expect("tree present after parse");
1541 let root = tree.root_node();
1542 assert_eq!(root.kind(), "source_file");
1543 assert!(root.child_count() > 0, "root has at least one child");
1544 }
1545
1546 /// Root-cause pin for the 2026-08-16 report: `>>` then immediately `==`
1547 /// silently does nothing.
1548 ///
1549 /// `reparsed_from_version` is a **delta baseline** — the version
1550 /// `changed_lines` is measured against — and its own doc says it is
1551 /// "meaningful only when `changed_lines()` is `Some`". After a completed
1552 /// INCREMENTAL reparse it holds the version the parse started from, which
1553 /// by construction is never the version it produced. Two callers in
1554 /// `lattice-host::dispatch` read it as "the version this tree reflects":
1555 /// `=`'s `indent_resolver` gate and `tree_levels_for_new_line`. Both
1556 /// therefore see a fresh tree as permanently stale.
1557 #[test]
1558 fn incremental_reparse_leaves_reparsed_from_behind_text_version() {
1559 let mut s = Syntax::for_language(Lang::Rust).unwrap().unwrap();
1560 let v0 = "fn f() {\n x();\n}\n";
1561 s.parse_at(v0, 1);
1562 assert_eq!(s.snapshot_owned().text_version(), 1);
1563 assert_eq!(
1564 s.snapshot_owned().reparsed_from_version(),
1565 1,
1566 "a FULL parse sets the baseline to the version it produced"
1567 );
1568
1569 // `>>` on line 1: leading whitespace only, complete code either way.
1570 let v1 = "fn f() {\n x();\n}\n";
1571 use lattice_protocol::edit::EditDelta;
1572 use lattice_protocol::position::Position;
1573 let edit = EditDelta {
1574 start_byte: 9,
1575 old_end_byte: 9,
1576 new_end_byte: 13,
1577 start_position: Position::new(1, 0),
1578 old_end_position: Position::new(1, 0),
1579 new_end_position: Position::new(1, 4),
1580 };
1581 assert!(
1582 s.try_apply_intermediate(v1, 2, 1, &[edit]).is_ok(),
1583 "the incremental path is the one that runs for an ordinary edit"
1584 );
1585 s.reparse_with_cached_tree(1);
1586
1587 let snap = s.snapshot_owned();
1588 assert_eq!(snap.text_version(), 2, "the tree is a real parse of v1");
1589 assert_eq!(
1590 snap.reparsed_from_version(),
1591 1,
1592 "but the baseline still points at the version it came FROM"
1593 );
1594 assert_ne!(
1595 snap.reparsed_from_version(),
1596 snap.text_version(),
1597 "so a freshness gate written as `reparsed_from == text_version` \
1598 cannot ever pass after an incremental reparse — this was the bug"
1599 );
1600
1601 // The signal that does answer the question the gates were asking.
1602 assert_eq!(snap.parsed_text_version(), 2);
1603 assert!(snap.tree_reflects(2), "the tree IS a completed parse of v1");
1604 assert!(!snap.tree_reflects(1));
1605 }
1606
1607 /// The intermediate snapshot is the reason `text_version` alone cannot
1608 /// be the freshness signal either: it carries the newest text with a
1609 /// tree that has only been byte-shifted, never parsed against it.
1610 #[test]
1611 fn intermediate_snapshot_does_not_claim_its_tree_is_current() {
1612 use lattice_protocol::edit::EditDelta;
1613 use lattice_protocol::position::Position;
1614 let mut s = Syntax::for_language(Lang::Rust).unwrap().unwrap();
1615 s.parse_at("fn f() {\n x();\n}\n", 1);
1616 let edit = EditDelta {
1617 start_byte: 9,
1618 old_end_byte: 9,
1619 new_end_byte: 13,
1620 start_position: Position::new(1, 0),
1621 old_end_position: Position::new(1, 0),
1622 new_end_position: Position::new(1, 4),
1623 };
1624 s.try_apply_intermediate("fn f() {\n x();\n}\n", 2, 1, &[edit])
1625 .unwrap();
1626
1627 let snap = s.snapshot_owned();
1628 assert_eq!(snap.text_version(), 2, "newest text");
1629 assert!(
1630 !snap.tree_reflects(2),
1631 "but no parse has run against it yet"
1632 );
1633 }
1634
1635 /// The collision that caused the stale-highlight bug, stated directly:
1636 /// the intermediate publish and the completed reparse carry the SAME
1637 /// `text_version`, so a render cache keyed on it cannot tell them apart
1638 /// and never rebuilds with colour.
1639 ///
1640 /// `render_version` is the stamp that does move, on both publishes.
1641 #[test]
1642 fn render_version_separates_the_intermediate_from_the_completed_parse() {
1643 use lattice_protocol::edit::EditDelta;
1644 use lattice_protocol::position::Position;
1645 let mut s = Syntax::for_language(Lang::Rust).unwrap().unwrap();
1646 s.parse_at("fn f() {\n x();\n}\n", 1);
1647 let at_v1 = s.snapshot_owned().render_version();
1648
1649 let edit = EditDelta {
1650 start_byte: 9,
1651 old_end_byte: 9,
1652 new_end_byte: 13,
1653 start_position: Position::new(1, 0),
1654 old_end_position: Position::new(1, 0),
1655 new_end_position: Position::new(1, 4),
1656 };
1657 s.try_apply_intermediate("fn f() {\n x();\n}\n", 2, 1, &[edit])
1658 .unwrap();
1659 let intermediate = s.snapshot_owned();
1660
1661 s.reparse_with_cached_tree(1);
1662 let completed = s.snapshot_owned();
1663
1664 // The trap: text_version cannot separate them.
1665 assert_eq!(
1666 intermediate.text_version(),
1667 completed.text_version(),
1668 "sanity: this is exactly why text_version cannot be the cache key"
1669 );
1670
1671 // render_version separates all three states.
1672 assert_ne!(
1673 at_v1,
1674 intermediate.render_version(),
1675 "the intermediate must invalidate — it is what makes unchanged \
1676 content paint at correct positions immediately"
1677 );
1678 assert_ne!(
1679 intermediate.render_version(),
1680 completed.render_version(),
1681 "and the completed parse must invalidate again — this is the \
1682 rebuild that puts the colour on, and the one that was missing"
1683 );
1684 }
1685
1686 #[test]
1687 fn rust_collect_symbols_captures_definitions() {
1688 let mut s = Syntax::for_language(Lang::Rust).unwrap().unwrap();
1689 s.parse(
1690 "\
1691fn outer(arg: i32) -> i32 {\n\
1692 let local = arg + 1;\n\
1693 local\n\
1694}\n\
1695struct Point { x: i32, y: i32 }\n\
1696const MAX: i32 = 10;\n\
1697",
1698 );
1699 let symbols = s.collect_symbols();
1700 // Definition-position names captured.
1701 for expected in &["outer", "arg", "local", "Point", "MAX"] {
1702 assert!(
1703 symbols.iter().any(|s| s == expected),
1704 "expected `{expected}` in {symbols:?}",
1705 );
1706 }
1707 // Reference-position uses NOT captured (e.g. the `i32`
1708 // type references inside the function aren't @symbol
1709 // captures because we only match on `name: ...` /
1710 // `pattern: ...` field-introduced positions).
1711 // Just sanity-check we don't double-count.
1712 let count_outer = symbols.iter().filter(|s| s.as_str() == "outer").count();
1713 assert_eq!(count_outer, 1, "no duplicates");
1714 }
1715
1716 #[test]
1717 fn python_collect_symbols_captures_def_and_class() {
1718 let mut s = Syntax::for_language(Lang::Python).unwrap().unwrap();
1719 s.parse(
1720 "def greet(name):\n message = name\n return message\n\nclass Greeter:\n pass\n",
1721 );
1722 let symbols = s.collect_symbols();
1723 for expected in &["greet", "name", "message", "Greeter"] {
1724 assert!(
1725 symbols.iter().any(|s| s == expected),
1726 "expected `{expected}` in {symbols:?}",
1727 );
1728 }
1729 }
1730
1731 #[test]
1732 fn collect_symbols_empty_when_no_parse() {
1733 // No parse() called -> tree is None -> empty result.
1734 let s = Syntax::for_language(Lang::Rust).unwrap().unwrap();
1735 assert!(s.collect_symbols().is_empty());
1736 }
1737
1738 #[test]
1739 fn cursor_in_string_scope_true_inside_rust_string_literal() {
1740 let source = "fn main() { let p = \"src/foo.rs\"; }\n";
1741 let mut s = Syntax::for_language(Lang::Rust).unwrap().unwrap();
1742 s.parse(source);
1743 // Pick a byte that's inside the literal -- between the
1744 // opening quote and the closing one.
1745 let lit_start = source.find('"').unwrap();
1746 let lit_end = source.rfind('"').unwrap();
1747 let inside = lit_start + 4; // somewhere mid-string
1748 assert!(inside < lit_end);
1749 assert!(s.cursor_in_string_scope(inside));
1750 }
1751
1752 #[test]
1753 fn cursor_in_string_scope_false_outside_string_literal() {
1754 let source = "fn main() { let p = \"src/foo.rs\"; }\n";
1755 let mut s = Syntax::for_language(Lang::Rust).unwrap().unwrap();
1756 s.parse(source);
1757 // Position at `let` keyword -- no string ancestor.
1758 let outside = source.find("let").unwrap() + 1;
1759 assert!(!s.cursor_in_string_scope(outside));
1760 }
1761
1762 #[test]
1763 fn cursor_in_string_scope_true_inside_python_string() {
1764 let source = "p = \"src/foo.py\"\n";
1765 let mut s = Syntax::for_language(Lang::Python).unwrap().unwrap();
1766 s.parse(source);
1767 let lit_start = source.find('"').unwrap();
1768 let inside = lit_start + 3;
1769 assert!(s.cursor_in_string_scope(inside));
1770 }
1771
1772 #[test]
1773 fn cursor_in_string_scope_false_when_no_parse() {
1774 // Without a parse the helper returns false safely
1775 // rather than panicking on missing tree.
1776 let s = Syntax::for_language(Lang::Rust).unwrap().unwrap();
1777 assert!(!s.cursor_in_string_scope(0));
1778 }
1779
1780 #[test]
1781 fn collect_symbols_empty_for_language_without_query() {
1782 // markdown ships no symbols.scm -> empty result even
1783 // after parse.
1784 let mut s = Syntax::for_language(Lang::Markdown).unwrap().unwrap();
1785 s.parse("# heading\n\nbody\n");
1786 assert!(s.collect_symbols().is_empty());
1787 }
1788
1789 // ---- N.1.0: scope_at_cursor (narrow-mode tree-sitter targets) ----
1790
1791 /// N.1.4c: byte-precise expected-range helper. `scope_at_cursor`
1792 /// now returns a half-open `[start, end)` `ProtoRange` (line + byte
1793 /// column), not just rows.
1794 fn rng(sl: u32, sb: u32, el: u32, eb: u32) -> Option<lattice_protocol::position::Range> {
1795 Some(lattice_protocol::position::Range::new(
1796 lattice_protocol::Position::new(sl, sb),
1797 lattice_protocol::Position::new(el, eb),
1798 ))
1799 }
1800
1801 #[test]
1802 fn scope_at_cursor_rust_fn_returns_correct_range() {
1803 // line 0: fn outer() {
1804 // line 1: let x = 1;
1805 // line 2: x
1806 // line 3: }
1807 let src = "fn outer() {\n let x = 1;\n x\n}\n";
1808 let mut s = Syntax::for_language(Lang::Rust).unwrap().unwrap();
1809 s.parse(src);
1810 // Cursor inside the body returns the whole function_item.
1811 assert_eq!(s.scope_at_cursor(1, 8, "function.outer"), rng(0, 0, 3, 1));
1812 }
1813
1814 #[test]
1815 fn scope_at_cursor_selects_innermost_when_nested() {
1816 // A closure nested in a function: the closure is the innermost
1817 // @function.outer match, so its (smaller) range wins.
1818 // line 0: fn outer() {
1819 // line 1: let f = || {
1820 // line 2: 1
1821 // line 3: };
1822 // line 4: }
1823 let src = "fn outer() {\n let f = || {\n 1\n };\n}\n";
1824 let mut s = Syntax::for_language(Lang::Rust).unwrap().unwrap();
1825 s.parse(src);
1826 assert_eq!(s.scope_at_cursor(2, 8, "function.outer"), rng(1, 12, 3, 5));
1827 }
1828
1829 #[test]
1830 fn scope_at_cursor_returns_none_outside_any_scope() {
1831 // line 0: use std::io; <- not inside any function
1832 // line 1: fn main() {}
1833 let src = "use std::io;\nfn main() {}\n";
1834 let mut s = Syntax::for_language(Lang::Rust).unwrap().unwrap();
1835 s.parse(src);
1836 assert_eq!(s.scope_at_cursor(0, 4, "function.outer"), None);
1837 }
1838
1839 #[test]
1840 fn scope_at_cursor_class_rust_struct() {
1841 // line 0: struct Point {
1842 // line 1: x: i32,
1843 // line 2: y: i32,
1844 // line 3: }
1845 let src = "struct Point {\n x: i32,\n y: i32,\n}\n";
1846 let mut s = Syntax::for_language(Lang::Rust).unwrap().unwrap();
1847 s.parse(src);
1848 assert_eq!(s.scope_at_cursor(1, 4, "class.outer"), rng(0, 0, 3, 1));
1849 // The struct is not a function.
1850 assert_eq!(s.scope_at_cursor(1, 4, "function.outer"), None);
1851 }
1852
1853 #[test]
1854 fn scope_at_cursor_block_targets_innermost_brace_scope() {
1855 // line 0: fn main() {
1856 // line 1: if x > 0 {
1857 // line 2: y = 1;
1858 // line 3: }
1859 // line 4: }
1860 // Cursor on `y = 1;` -> innermost @block.outer is the if's
1861 // then-block (rows 1..3), not the whole function body (0..4).
1862 let src = "fn main() {\n if x > 0 {\n y = 1;\n }\n}\n";
1863 let mut s = Syntax::for_language(Lang::Rust).unwrap().unwrap();
1864 s.parse(src);
1865 assert_eq!(s.scope_at_cursor(2, 8, "block.outer"), rng(1, 13, 3, 5));
1866 }
1867
1868 #[test]
1869 fn scope_at_cursor_none_when_no_textobjects_query() {
1870 // markdown ships no textobjects.scm -> None even after parse.
1871 let mut s = Syntax::for_language(Lang::Markdown).unwrap().unwrap();
1872 s.parse("# heading\n\nbody\n");
1873 assert_eq!(s.scope_at_cursor(0, 2, "function.outer"), None);
1874 }
1875
1876 #[test]
1877 fn scope_at_cursor_none_when_no_parse() {
1878 // No parse -> tree is None -> None, no panic.
1879 let s = Syntax::for_language(Lang::Rust).unwrap().unwrap();
1880 assert_eq!(s.scope_at_cursor(0, 0, "function.outer"), None);
1881 }
1882
1883 #[test]
1884 fn scope_at_cursor_python_function() {
1885 // line 0: def greet(name):
1886 // line 1: msg = name
1887 // line 2: return msg
1888 let src = "def greet(name):\n msg = name\n return msg\n";
1889 let mut s = Syntax::for_language(Lang::Python).unwrap().unwrap();
1890 s.parse(src);
1891 assert_eq!(s.scope_at_cursor(1, 4, "function.outer"), rng(0, 0, 2, 14));
1892 }
1893
1894 // ---- N.1.4c: inner bodies, parameters, loops (byte-precise) ----
1895
1896 #[test]
1897 fn scope_at_cursor_function_inner_is_body_block() {
1898 // line 0: fn outer() { <- `{` at col 11
1899 // line 3: } <- `}` at col 0, exclusive end col 1
1900 let src = "fn outer() {\n let x = 1;\n x\n}\n";
1901 let mut s = Syntax::for_language(Lang::Rust).unwrap().unwrap();
1902 s.parse(src);
1903 // `if` (inner function) = the body block (braces included, v1).
1904 assert_eq!(s.scope_at_cursor(1, 8, "function.inner"), rng(0, 11, 3, 1));
1905 // `af` (outer) still spans the whole function_item.
1906 assert_eq!(s.scope_at_cursor(1, 8, "function.outer"), rng(0, 0, 3, 1));
1907 }
1908
1909 #[test]
1910 fn scope_at_cursor_class_inner_is_field_list() {
1911 // line 0: struct Point { <- `{` at col 13
1912 let src = "struct Point {\n x: i32,\n y: i32,\n}\n";
1913 let mut s = Syntax::for_language(Lang::Rust).unwrap().unwrap();
1914 s.parse(src);
1915 assert_eq!(s.scope_at_cursor(1, 4, "class.inner"), rng(0, 13, 3, 1));
1916 }
1917
1918 #[test]
1919 fn scope_at_cursor_parameter_byte_precise() {
1920 // line 0: fn add(x: i32, y: i32) -> i32 { x + y }
1921 // ^col 7 ^col 15
1922 let src = "fn add(x: i32, y: i32) -> i32 { x + y }\n";
1923 let mut s = Syntax::for_language(Lang::Rust).unwrap().unwrap();
1924 s.parse(src);
1925 // `aa` on the first parameter -> exactly `x: i32` (cols 7..13),
1926 // NOT the whole signature line -- this is the byte-precision win.
1927 assert_eq!(s.scope_at_cursor(0, 7, "parameter.outer"), rng(0, 7, 0, 13));
1928 assert_eq!(s.scope_at_cursor(0, 7, "parameter.inner"), rng(0, 7, 0, 13));
1929 // Cursor on the second parameter resolves the second span.
1930 assert_eq!(
1931 s.scope_at_cursor(0, 15, "parameter.outer"),
1932 rng(0, 15, 0, 21)
1933 );
1934 }
1935
1936 #[test]
1937 fn scope_at_cursor_loop_outer_and_inner() {
1938 // line 0: fn main() {
1939 // line 1: for i in 0..10 { <- `for` col 4, body `{` col 19
1940 // line 2: x += i;
1941 // line 3: } <- exclusive end col 5
1942 // line 4: }
1943 let src = "fn main() {\n for i in 0..10 {\n x += i;\n }\n}\n";
1944 let mut s = Syntax::for_language(Lang::Rust).unwrap().unwrap();
1945 s.parse(src);
1946 assert_eq!(s.scope_at_cursor(2, 8, "loop.outer"), rng(1, 4, 3, 5));
1947 assert_eq!(s.scope_at_cursor(2, 8, "loop.inner"), rng(1, 19, 3, 5));
1948 }
1949
1950 #[test]
1951 fn scope_at_cursor_python_inner_and_parameter() {
1952 // line 0: def greet(name): <- `name` at cols 10..14
1953 // line 1: msg = name
1954 // line 2: return msg
1955 let src = "def greet(name):\n msg = name\n return msg\n";
1956 let mut s = Syntax::for_language(Lang::Python).unwrap().unwrap();
1957 s.parse(src);
1958 assert_eq!(
1959 s.scope_at_cursor(0, 10, "parameter.outer"),
1960 rng(0, 10, 0, 14)
1961 );
1962 // Inner function = the suite body (delimiter-free in Python).
1963 assert_eq!(s.scope_at_cursor(1, 4, "function.inner"), rng(1, 4, 2, 14));
1964 }
1965
1966 #[test]
1967 fn scope_at_cursor_javascript_parameter_byte_precise() {
1968 // line 0: function add(x, y) { return x + y; }
1969 // ^col 13 ^col 16
1970 let src = "function add(x, y) { return x + y; }\n";
1971 let mut s = Syntax::for_language(Lang::JavaScript).unwrap().unwrap();
1972 s.parse(src);
1973 assert_eq!(
1974 s.scope_at_cursor(0, 13, "parameter.outer"),
1975 rng(0, 13, 0, 14)
1976 );
1977 assert_eq!(
1978 s.scope_at_cursor(0, 16, "parameter.outer"),
1979 rng(0, 16, 0, 17)
1980 );
1981 }
1982
1983 #[test]
1984 fn daf_deletes_a_whole_function_end_to_end() {
1985 // N.1.4c end-to-end: operator (`d`) + structural text object
1986 // (`af`) + the byte-precise scope resolver -> a real edit. Proves
1987 // the whole chain below the keymap: `register_syntax_text_objects`
1988 // mints `around_function`; the dispatcher resolves
1989 // `Target::TextObject(around_function)` by calling the object's
1990 // apply with the `SyntaxSnapshot` as `scope_resolver`; the
1991 // resolved byte span feeds the delete operator.
1992 use lattice_grammar::{
1993 Args, CancellationToken, CommandInvocation, GrammarEnv, Target, execute_with_env,
1994 };
1995
1996 let src = "fn keep() {}\nfn drop_me() {\n let x = 1;\n}\nfn also_keep() {}\n";
1997 let mut s = Syntax::for_language(Lang::Rust).unwrap().unwrap();
1998 s.parse(src);
1999
2000 let mut registry = lattice_grammar::CommandRegistry::new();
2001 let builtins = lattice_grammar::builtins::populate(&mut registry);
2002 let ids = crate::text_objects::register_syntax_text_objects(&mut registry);
2003
2004 let mut doc = lattice_core::Document::from_text(src);
2005 // Cursor inside `drop_me`'s body (line 2).
2006 let cursor = lattice_protocol::Position::new(2, 8);
2007 let inv = CommandInvocation::of(builtins.delete.0)
2008 .with_target(Target::TextObject(ids.around_function, Args::None));
2009 execute_with_env(
2010 ®istry,
2011 &mut doc,
2012 lattice_core::BufferId(0),
2013 cursor,
2014 inv,
2015 &CancellationToken::never(),
2016 GrammarEnv {
2017 // `&s.inner` is the SyntaxSnapshot; coerces to &dyn ScopeResolver.
2018 scope_resolver: Some(&s.inner),
2019 comment_syntax: None,
2020 syntax: None,
2021 ..Default::default()
2022 },
2023 )
2024 .expect("daf dispatch ok");
2025
2026 let after = doc.text();
2027 assert!(
2028 !after.contains("drop_me"),
2029 "`daf` should delete the whole function, got: {after:?}"
2030 );
2031 assert!(
2032 after.contains("keep") && after.contains("also_keep"),
2033 "neighbouring functions stay intact: {after:?}"
2034 );
2035 }
2036
2037 // ---- TSM.2: scope_toward (structural motions tree walk) ----
2038
2039 use lattice_grammar::{NavBoundary, NavDir};
2040
2041 fn snapshot_rust(src: &str) -> SyntaxSnapshot {
2042 let mut s = Syntax::for_language(Lang::Rust).unwrap().unwrap();
2043 s.parse(src);
2044 s.snapshot_owned()
2045 }
2046
2047 fn snapshot_python(src: &str) -> SyntaxSnapshot {
2048 let mut s = Syntax::for_language(Lang::Python).unwrap().unwrap();
2049 s.parse(src);
2050 s.snapshot_owned()
2051 }
2052
2053 /// A snapshot with source set but no parse ever run, so `tree` stays
2054 /// `None` -- mirrors a document whose first parse hasn't landed yet.
2055 fn snapshot_plain(src: &str) -> SyntaxSnapshot {
2056 let mut s = Syntax::for_language(Lang::Rust).unwrap().unwrap();
2057 s.inner.set_source_bytes(src.as_bytes());
2058 s.snapshot_owned()
2059 }
2060
2061 // Source: 3 top-level fns at rows 0, 2, 4.
2062 // row 0: fn a() {}
2063 // row 2: fn b() {}
2064 // row 4: fn c() {}
2065 fn three_fns() -> SyntaxSnapshot {
2066 snapshot_rust("fn a() {}\n\nfn b() {}\n\nfn c() {}\n")
2067 }
2068
2069 #[test]
2070 fn scope_toward_forward_start_skips_enclosing() {
2071 let s = three_fns();
2072 // Cursor inside fn a (row 0) -> next function START is fn b (row 2).
2073 let p = s.scope_toward(
2074 0,
2075 3,
2076 "function.outer",
2077 NavDir::Forward,
2078 NavBoundary::Start,
2079 1,
2080 );
2081 assert_eq!(p, Some(lattice_protocol::Position::new(2, 0)));
2082 }
2083
2084 #[test]
2085 fn scope_toward_forward_start_count_two() {
2086 let s = three_fns();
2087 // From row 0, 2nd next function start is fn c (row 4).
2088 let p = s.scope_toward(
2089 0,
2090 3,
2091 "function.outer",
2092 NavDir::Forward,
2093 NavBoundary::Start,
2094 2,
2095 );
2096 assert_eq!(p, Some(lattice_protocol::Position::new(4, 0)));
2097 }
2098
2099 #[test]
2100 fn scope_toward_backward_start_lands_on_current() {
2101 let s = three_fns();
2102 // Cursor inside fn b past its start (row 2, col 5) -> prev START is fn b's
2103 // OWN start (row 2, col 0), per the enclosing rule.
2104 let p = s.scope_toward(
2105 2,
2106 5,
2107 "function.outer",
2108 NavDir::Backward,
2109 NavBoundary::Start,
2110 1,
2111 );
2112 assert_eq!(p, Some(lattice_protocol::Position::new(2, 0)));
2113 }
2114
2115 #[test]
2116 fn scope_toward_backward_start_on_boundary_moves_to_previous() {
2117 // Regression: the `]f` -> `[f` round-trip. After `]f` the cursor sits
2118 // EXACTLY on fn b's start (row 2, col 0). `[f` from there must move to
2119 // the PREVIOUS function (fn a, row 0), not no-op on fn b's own start.
2120 // A non-strict `<=` comparison would re-select fn b here.
2121 let s = three_fns();
2122 let p = s.scope_toward(
2123 2,
2124 0,
2125 "function.outer",
2126 NavDir::Backward,
2127 NavBoundary::Start,
2128 1,
2129 );
2130 assert_eq!(p, Some(lattice_protocol::Position::new(0, 0)));
2131 }
2132
2133 #[test]
2134 fn scope_toward_forward_end_on_boundary_moves_to_next() {
2135 // Symmetric regression for `]F`. Cursor EXACTLY on fn a's end
2136 // (row 0, col 9 -- one past `}`). `]F` must move to the NEXT function's
2137 // end (fn b, row 2, col 9), not no-op on fn a's own end.
2138 let s = three_fns();
2139 let p = s.scope_toward(0, 9, "function.outer", NavDir::Forward, NavBoundary::End, 1);
2140 assert_eq!(p, Some(lattice_protocol::Position::new(2, 9)));
2141 }
2142
2143 #[test]
2144 fn scope_toward_forward_end_lands_on_current_end() {
2145 let s = three_fns();
2146 // Cursor inside fn b (row 2, col 5) -> next END is fn b's own closing
2147 // brace. "fn b() {}" -- end_position is row 2, col 9 (one past `}`).
2148 let p = s.scope_toward(2, 5, "function.outer", NavDir::Forward, NavBoundary::End, 1);
2149 assert_eq!(p, Some(lattice_protocol::Position::new(2, 9)));
2150 }
2151
2152 #[test]
2153 fn scope_toward_stops_at_boundary() {
2154 let s = three_fns();
2155 // From inside the LAST fn, forward-start has no next -> None (no wrap).
2156 let p = s.scope_toward(
2157 4,
2158 3,
2159 "function.outer",
2160 NavDir::Forward,
2161 NavBoundary::Start,
2162 1,
2163 );
2164 assert_eq!(p, None);
2165 }
2166
2167 #[test]
2168 fn scope_toward_none_without_tree() {
2169 let s = snapshot_plain("plain text no tree\n");
2170 let p = s.scope_toward(
2171 0,
2172 0,
2173 "function.outer",
2174 NavDir::Forward,
2175 NavBoundary::Start,
2176 1,
2177 );
2178 assert_eq!(p, None);
2179 }
2180
2181 #[test]
2182 fn scope_toward_python_forward_start_skips_enclosing() {
2183 // row 0: def a(): pass
2184 // row 1: def b(): pass
2185 // row 2: def c(): pass
2186 let s = snapshot_python("def a(): pass\ndef b(): pass\ndef c(): pass\n");
2187 // Cursor inside def a (row 0) -> next function START is def b (row 1).
2188 let p = s.scope_toward(
2189 0,
2190 3,
2191 "function.outer",
2192 NavDir::Forward,
2193 NavBoundary::Start,
2194 1,
2195 );
2196 assert_eq!(p, Some(lattice_protocol::Position::new(1, 0)));
2197 // Count 2 -> def c (row 2).
2198 let p2 = s.scope_toward(
2199 0,
2200 3,
2201 "function.outer",
2202 NavDir::Forward,
2203 NavBoundary::Start,
2204 2,
2205 );
2206 assert_eq!(p2, Some(lattice_protocol::Position::new(2, 0)));
2207 }
2208
2209 #[test]
2210 fn scope_toward_backward_end_skips_enclosing() {
2211 let s = three_fns();
2212 // Cursor inside fn b (row 2, col 5) -> prev END is fn a's END, NOT fn b's
2213 // own end. The enclosing rule for (Backward, End) keeps candidates
2214 // strictly before the cursor (`b < cursor_byte`), so fn b's own closing
2215 // brace (past the cursor) is skipped. "fn a() {}" -> end_position row 0,
2216 // col 9 (one past the `}`).
2217 let p = s.scope_toward(
2218 2,
2219 5,
2220 "function.outer",
2221 NavDir::Backward,
2222 NavBoundary::End,
2223 1,
2224 );
2225 assert_eq!(p, Some(lattice_protocol::Position::new(0, 9)));
2226 }
2227
2228 #[test]
2229 fn scope_toward_count_zero_is_none() {
2230 let s = three_fns();
2231 // count == 0 has no "0th" candidate -> None (guard, never panics).
2232 let p = s.scope_toward(
2233 0,
2234 3,
2235 "function.outer",
2236 NavDir::Forward,
2237 NavBoundary::Start,
2238 0,
2239 );
2240 assert_eq!(p, None);
2241 }
2242
2243 #[test]
2244 fn scope_toward_empty_candidate_set_is_none() {
2245 // Tree present + textobjects query present, but the source has no loops,
2246 // so `loop.outer` captures nothing -> empty candidate set -> None.
2247 let s = three_fns();
2248 let p = s.scope_toward(0, 3, "loop.outer", NavDir::Forward, NavBoundary::Start, 1);
2249 assert_eq!(p, None);
2250 }
2251
2252 #[test]
2253 fn reparse_against_evolving_source_keeps_tree_in_sync() {
2254 // Step 1 is a full reparse on every `parse()` call (we
2255 // don't yet thread `Tree::edit` deltas). Verify the tree
2256 // shape tracks the source: two top-level fn items after a
2257 // second `parse()`, not one stale item.
2258 let mut s = Syntax::for_language(Lang::Rust).unwrap().unwrap();
2259 s.parse("fn a() {}");
2260 assert_eq!(s.tree().unwrap().root_node().child_count(), 1);
2261 s.parse("fn a() {}\nfn b() {}");
2262 assert_eq!(s.tree().unwrap().root_node().child_count(), 2);
2263 }
2264
2265 #[test]
2266 fn rust_syntax_highlights_keyword() {
2267 let mut s = Syntax::for_language(Lang::Rust).unwrap().unwrap();
2268 s.parse("fn main() {}");
2269 let spans = s.highlight_lines(0, 1).unwrap();
2270 assert_eq!(spans.len(), 1);
2271 // `fn` should be highlighted as Keyword.
2272 assert!(
2273 spans[0].iter().any(|sp| sp.style == Style::Keyword),
2274 "expected a Keyword span, got {:?}",
2275 spans[0]
2276 );
2277 }
2278
2279 #[test]
2280 fn markdown_syntax_highlights_atx_heading() {
2281 let mut s = Syntax::for_language(Lang::Markdown).unwrap().unwrap();
2282 s.parse("# Title\n\nbody\n");
2283 let spans = s.highlight_lines(0, 3).unwrap();
2284 // The heading row carries a Heading1 span (bundled query
2285 // captures `(atx_heading (inline) @text.title)` which maps
2286 // to Heading1 by the level-less convention).
2287 assert!(
2288 spans[0].iter().any(|sp| sp.style == Style::Heading1),
2289 "expected a Heading1 span on the heading line, got {:?}",
2290 spans[0]
2291 );
2292 }
2293
2294 #[test]
2295 fn markdown_fenced_rust_block_injects_rust_highlight() {
2296 let mut s = Syntax::for_language(Lang::Markdown).unwrap().unwrap();
2297 // Fence at line 0; rust content at lines 1-2; closing fence at line 3.
2298 let src = "```rust\nfn main() {}\n```\n";
2299 s.parse(src);
2300 let spans = s.highlight_lines(0, 4).unwrap();
2301 // Line 1 (the rust code) should have a Keyword span (`fn`).
2302 assert!(
2303 spans[1].iter().any(|sp| sp.style == Style::Keyword),
2304 "expected rust keyword styling inside fenced block, got {:?}",
2305 spans[1]
2306 );
2307 }
2308
2309 // Note: a markdown-inline-emphasis test (asserting **bold**
2310 // emits a Bold span via the block→inline injection) is not
2311 // included here. tree-sitter-md 0.3.x's block parser emits
2312 // `(inline)` nodes covering paragraph content, and the bundled
2313 // injections.scm is supposed to route them to the inline
2314 // grammar -- in practice the injection occasionally fails to
2315 // surface a span through the highlight stream. The block-level
2316 // highlighting + fenced-block injection (proven above) confirm
2317 // the registry / callback infrastructure works; the inline
2318 // sub-injection is a known soft spot we'll revisit when
2319 // upgrading to tree-sitter-md 0.5+. For day-to-day markdown
2320 // editing the heading / list / code-block highlighting is the
2321 // load-bearing part.
2322
2323 // ---- Step 3a: native pipeline parity tests ----------------
2324
2325 /// Helper: parse + highlight `source` through the native
2326 /// pipeline and assert that at least one span of `expected`
2327 /// style appears somewhere in the output. Used by the
2328 /// per-language smoke tests below.
2329 fn assert_has_style(lang: Lang, source: &str, expected: Style) {
2330 let mut s = Syntax::for_language(lang).unwrap().unwrap();
2331 s.parse(source);
2332 let line_count = source.split('\n').count() as u32;
2333 let lines = s.highlight_lines(0, line_count).unwrap();
2334 let found = lines
2335 .iter()
2336 .any(|l| l.iter().any(|sp| sp.style == expected));
2337 assert!(
2338 found,
2339 "{lang:?}: expected at least one {expected:?} span in {source:?}, got {lines:?}"
2340 );
2341 }
2342
2343 #[test]
2344 fn native_rust_simple_function_produces_keyword_and_function_spans() {
2345 assert_has_style(
2346 Lang::Rust,
2347 "fn main() {\n let x = 1;\n}\n",
2348 Style::Keyword,
2349 );
2350 assert_has_style(
2351 Lang::Rust,
2352 "fn main() {\n let x = 1;\n}\n",
2353 Style::Function,
2354 );
2355 }
2356
2357 #[test]
2358 fn native_python_def_produces_keyword_and_function_spans() {
2359 assert_has_style(
2360 Lang::Python,
2361 "def f(x):\n return x + 1\n\nclass Foo:\n pass\n",
2362 Style::Keyword,
2363 );
2364 assert_has_style(
2365 Lang::Python,
2366 "def f(x):\n return x + 1\n\nclass Foo:\n pass\n",
2367 Style::Function,
2368 );
2369 }
2370
2371 #[test]
2372 fn native_python_strings_and_comments_resolve_to_proper_styles() {
2373 let src = "# comment\ns = \"hello world\"\nn = 42\nb = True\n";
2374 // Python's `# comment` is captured as `@comment` (not
2375 // `@comment.line`), so it lands on `Style::Comment` rather
2376 // than `Style::LineComment`. Both are visible distinct
2377 // colours; the test pins the actual mapping.
2378 assert_has_style(Lang::Python, src, Style::Comment);
2379 assert_has_style(Lang::Python, src, Style::String);
2380 assert_has_style(Lang::Python, src, Style::Number);
2381 }
2382
2383 #[test]
2384 fn native_rust_struct_and_impl_emit_keyword_spans() {
2385 assert_has_style(
2386 Lang::Rust,
2387 "struct Buffer {\n rope: Rope,\n}\n\nimpl Buffer {\n fn new() -> Self {\n Self { rope: Rope::new() }\n }\n}\n",
2388 Style::Keyword,
2389 );
2390 }
2391
2392 #[test]
2393 fn native_markdown_fenced_rust_block_emits_rust_spans() {
2394 // Native markdown injection recurses into the fenced
2395 // language. Strict parity with the legacy streaming
2396 // highlighter doesn't hold here -- tree-sitter-highlight
2397 // and our hand-rolled injection pipeline differ in how
2398 // they distribute outer markdown captures inside the
2399 // fenced range. The user-visible contract is "rust
2400 // keywords / function names get styled inside `\`\`\`rust`
2401 // blocks", which we verify directly.
2402 let mut s = Syntax::for_language(Lang::Markdown).unwrap().unwrap();
2403 let src = "# Title\n\n```rust\nfn main() {}\n```\n";
2404 s.parse(src);
2405 let lines = s.highlight_lines_native(0, 6).unwrap();
2406 // Line 3 is the rust body (`fn main() {}`).
2407 let rust_line = &lines[3];
2408 assert!(
2409 rust_line.iter().any(|sp| sp.style == Style::Keyword),
2410 "expected rust Keyword span on fenced line, got {rust_line:?}"
2411 );
2412 assert!(
2413 rust_line.iter().any(|sp| sp.style == Style::Function),
2414 "expected rust Function span on fenced line, got {rust_line:?}"
2415 );
2416 }
2417
2418 #[test]
2419 fn native_markdown_headings_emit_heading_styles() {
2420 let src = "# H1\n\n## H2\n\n### H3\n\nbody paragraph\n";
2421 assert_has_style(Lang::Markdown, src, Style::Heading1);
2422 // Lattice's custom markdown highlights query distinguishes heading
2423 // LEVELS (the bundled tree-sitter-md query is level-less). `##` →
2424 // Heading2, `###` → Heading3, so the theme can size + colour each
2425 // level differently (Thread F + per-level heading colours).
2426 assert_has_style(Lang::Markdown, src, Style::Heading2);
2427 assert_has_style(Lang::Markdown, src, Style::Heading3);
2428 }
2429
2430 /// Reproduction (2026-06-03): markdown highlighting must survive
2431 /// an incremental edit. The `parity_*` tests only compare TREE
2432 /// SHAPE (`to_sexp`); this compares the actual HIGHLIGHT SPANS
2433 /// produced incremental-after-edit vs a full reparse of the final
2434 /// text — the untested gap behind "markdown highlighting never
2435 /// comes back after an edit". If this fails, the reparse/highlight
2436 /// path drops markdown styling on a keystroke.
2437 #[test]
2438 fn markdown_highlight_survives_incremental_edit() {
2439 let src_a = "# Heading\n\nHello world\n";
2440 // Type a character inside the paragraph (byte 16 sits in
2441 // "Hello world").
2442 let (src_b, delta) = delta_for_edit(src_a, 16, 16, "X");
2443 let lc = src_b.split('\n').count() as u32;
2444
2445 let mut s_inc = Syntax::for_language(Lang::Markdown).unwrap().unwrap();
2446 s_inc.parse_at(src_a, 1);
2447 s_inc.parse_at_with_edits(&src_b, 2, 1, &[delta]);
2448 let inc = s_inc.highlight_lines(0, lc).unwrap();
2449
2450 let mut s_full = Syntax::for_language(Lang::Markdown).unwrap().unwrap();
2451 s_full.parse_at(&src_b, 1);
2452 let full = s_full.highlight_lines(0, lc).unwrap();
2453
2454 assert_eq!(
2455 inc, full,
2456 "markdown highlight spans diverge incremental vs full after edit"
2457 );
2458 assert!(
2459 inc[0].iter().any(|sp| sp.style == Style::Heading1),
2460 "heading highlight lost after incremental edit: {:?}",
2461 inc[0]
2462 );
2463 }
2464
2465 /// H.2 (2026-06-04): an incremental reparse must publish
2466 /// `changed_lines` covering the edited line (so the cells worker can
2467 /// rebuild only dirty rows on reparse-completion), with
2468 /// `reparsed_from_version` set to the baseline it diffed against.
2469 #[test]
2470 fn changed_lines_covers_the_edited_line() {
2471 let src_a = "fn a() {}\nfn b() {}\nfn c() {}\nfn d() {}\n";
2472 // Insert 'X' after "fn b" on line 1 → "fn bX() {}".
2473 let (src_b, delta) = delta_for_edit(src_a, 13, 13, "X");
2474 let mut s = Syntax::for_language(Lang::Rust).unwrap().unwrap();
2475 s.parse_at(src_a, 1);
2476 s.parse_at_with_edits(&src_b, 2, 1, &[delta]);
2477
2478 assert_eq!(
2479 s.snapshot().reparsed_from_version(),
2480 1,
2481 "changed_lines baseline must be the from_version"
2482 );
2483 let changed = s
2484 .snapshot()
2485 .changed_lines()
2486 .expect("an incremental reparse must yield Some(changed_lines)");
2487 assert!(
2488 changed.iter().any(|&(lo, hi)| lo <= 1 && 1 <= hi),
2489 "changed_lines must cover the edited line 1, got {changed:?}"
2490 );
2491 }
2492
2493 /// Diagnostic (2026-06-04): does an UNCHANGED line carrying inline
2494 /// injection content (a `code span` + a [link]) keep IDENTICAL
2495 /// highlight spans across a reparse triggered by editing a
2496 /// DIFFERENT line? If not, markdown's inline injection is
2497 /// non-deterministic across reparses — which is why those lines
2498 /// flicker on every keystroke (B.1 full-rebuilds on reparse and the
2499 /// inline colours flip).
2500 #[test]
2501 fn markdown_inline_spans_stable_across_unrelated_edit() {
2502 let src_a = "# H\n\nUse `code` and [link](http://x)\n\ntail\n";
2503 // Edit "tail" (line 4), well away from the inline-content line 2.
2504 let (src_b, delta) = delta_for_edit(src_a, 42, 42, "X");
2505
2506 let mut s1 = Syntax::for_language(Lang::Markdown).unwrap().unwrap();
2507 s1.parse_at(src_a, 1);
2508 let line2_v1 = s1.highlight_lines(2, 3).unwrap().remove(0);
2509
2510 let mut s2 = Syntax::for_language(Lang::Markdown).unwrap().unwrap();
2511 s2.parse_at(src_a, 1);
2512 s2.parse_at_with_edits(&src_b, 2, 1, &[delta]);
2513 let line2_v2 = s2.highlight_lines(2, 3).unwrap().remove(0);
2514
2515 assert_eq!(
2516 line2_v1, line2_v2,
2517 "inline-content line 2 must keep identical spans across an unrelated edit \
2518 (v1 = full parse, v2 = incremental reparse after editing line 4)"
2519 );
2520 }
2521
2522 // ---- Slice B.2: incremental reparse parity tests -----------
2523 //
2524 // Tree-sitter's failure mode for a malformed `InputEdit` is a
2525 // silently wrong tree (no error, just stale node ranges).
2526 // These tests pin that incremental reparse produces the SAME
2527 // tree shape as full reparse on the same final source --
2528 // catching any future drift in `parse_at_with_edits` or the
2529 // `EditDelta -> InputEdit` conversion.
2530
2531 /// Helper: parse `source_a` then drive an incremental reparse
2532 /// to `source_b` using the supplied edits. Compare to a fresh
2533 /// full reparse on `source_b` directly. Returns the two trees'
2534 /// s-expressions for assertion.
2535 fn incremental_vs_full_reparse(
2536 lang: Lang,
2537 source_a: &str,
2538 source_b: &str,
2539 edits: &[EditDelta],
2540 ) -> (String, String) {
2541 let mut s_inc = Syntax::for_language(lang).unwrap().unwrap();
2542 s_inc.parse_at(source_a, 1);
2543 s_inc.parse_at_with_edits(source_b, 2, 1, edits);
2544 let inc = s_inc.tree().unwrap().root_node().to_sexp();
2545
2546 let mut s_full = Syntax::for_language(lang).unwrap().unwrap();
2547 s_full.parse_at(source_b, 1);
2548 let full = s_full.tree().unwrap().root_node().to_sexp();
2549
2550 (inc, full)
2551 }
2552
2553 #[test]
2554 fn incremental_reparse_single_insert_matches_full_reparse() {
2555 // Insert "x" at byte 3 of "fn main() {}". Single-edit
2556 // case -- the simplest incremental path.
2557 let edits = [EditDelta {
2558 start_byte: 3,
2559 old_end_byte: 3,
2560 new_end_byte: 4,
2561 start_position: lattice_protocol::Position::new(0, 3),
2562 old_end_position: lattice_protocol::Position::new(0, 3),
2563 new_end_position: lattice_protocol::Position::new(0, 4),
2564 }];
2565 let (inc, full) =
2566 incremental_vs_full_reparse(Lang::Rust, "fn main() {}", "fn xmain() {}", &edits);
2567 assert_eq!(inc, full, "incremental tree must match full reparse");
2568 }
2569
2570 #[test]
2571 fn incremental_reparse_single_delete_matches_full_reparse() {
2572 // Delete byte 3 of "fn xmain() {}".
2573 let edits = [EditDelta {
2574 start_byte: 3,
2575 old_end_byte: 4,
2576 new_end_byte: 3,
2577 start_position: lattice_protocol::Position::new(0, 3),
2578 old_end_position: lattice_protocol::Position::new(0, 4),
2579 new_end_position: lattice_protocol::Position::new(0, 3),
2580 }];
2581 let (inc, full) =
2582 incremental_vs_full_reparse(Lang::Rust, "fn xmain() {}", "fn main() {}", &edits);
2583 assert_eq!(inc, full);
2584 }
2585
2586 #[test]
2587 fn incremental_reparse_multiline_replace_matches_full_reparse() {
2588 // Replace `let x = 1;` (line 1) with `let x = 42;`.
2589 // Source A: "fn main() {\n let x = 1;\n}"
2590 // Source B: "fn main() {\n let x = 42;\n}"
2591 // Replacement byte range starts at line 1 col 12, ends at
2592 // line 1 col 13. Insert "42" (2 bytes) for "1" (1 byte).
2593 let source_a = "fn main() {\n let x = 1;\n}";
2594 let source_b = "fn main() {\n let x = 42;\n}";
2595 // "fn main() {\n" is 12 bytes. " let x = " is 12 more
2596 // = byte 24. "1" is at byte 24. End at byte 25.
2597 let edits = [EditDelta {
2598 start_byte: 24,
2599 old_end_byte: 25,
2600 new_end_byte: 26,
2601 start_position: lattice_protocol::Position::new(1, 12),
2602 old_end_position: lattice_protocol::Position::new(1, 13),
2603 new_end_position: lattice_protocol::Position::new(1, 14),
2604 }];
2605 let (inc, full) = incremental_vs_full_reparse(Lang::Rust, source_a, source_b, &edits);
2606 assert_eq!(inc, full);
2607 }
2608
2609 #[test]
2610 fn incremental_reparse_multi_edit_batch_matches_full_reparse() {
2611 // Two edits applied in sequence: insert "y" at byte 3,
2612 // then "z" at (post-first-edit) byte 5. The cumulative
2613 // shape is "fn yxmzain() {}" (Position fields shift after
2614 // first edit).
2615 // Source A: "fn xmain() {}"
2616 // After edit 1: "fn yxmain() {}"
2617 // After edit 2: "fn yxmzain() {}"
2618 let source_a = "fn xmain() {}";
2619 let source_b = "fn yxmzain() {}";
2620 let edits = [
2621 EditDelta {
2622 start_byte: 3,
2623 old_end_byte: 3,
2624 new_end_byte: 4,
2625 start_position: lattice_protocol::Position::new(0, 3),
2626 old_end_position: lattice_protocol::Position::new(0, 3),
2627 new_end_position: lattice_protocol::Position::new(0, 4),
2628 },
2629 EditDelta {
2630 start_byte: 6,
2631 old_end_byte: 6,
2632 new_end_byte: 7,
2633 start_position: lattice_protocol::Position::new(0, 6),
2634 old_end_position: lattice_protocol::Position::new(0, 6),
2635 new_end_position: lattice_protocol::Position::new(0, 7),
2636 },
2637 ];
2638 let (inc, full) = incremental_vs_full_reparse(Lang::Rust, source_a, source_b, &edits);
2639 assert_eq!(inc, full);
2640 }
2641
2642 #[test]
2643 fn parse_at_with_edits_falls_back_to_full_reparse_when_no_cached_tree() {
2644 // Fresh Syntax, no cached tree. parse_at_with_edits with
2645 // edits should fall back to full reparse rather than
2646 // panicking or producing a wrong tree.
2647 let mut s = Syntax::for_language(Lang::Rust).unwrap().unwrap();
2648 let edits = [EditDelta {
2649 start_byte: 0,
2650 old_end_byte: 0,
2651 new_end_byte: 5,
2652 start_position: lattice_protocol::Position::new(0, 0),
2653 old_end_position: lattice_protocol::Position::new(0, 0),
2654 new_end_position: lattice_protocol::Position::new(0, 5),
2655 }];
2656 s.parse_at_with_edits("hello", 1, 0, &edits);
2657 let tree = s.tree().expect("tree present after fallback");
2658 // Tree should match a direct full-reparse on the same
2659 // source -- proves the fallback path produced a correct
2660 // tree, not a stale one.
2661 let mut s_full = Syntax::for_language(Lang::Rust).unwrap().unwrap();
2662 s_full.parse_at("hello", 1);
2663 assert_eq!(
2664 tree.root_node().to_sexp(),
2665 s_full.tree().unwrap().root_node().to_sexp(),
2666 );
2667 }
2668
2669 #[test]
2670 fn parse_at_with_edits_falls_back_when_from_version_mismatches() {
2671 // Cached tree at version 5; request claims from_version=3.
2672 // The deltas from v3->v6 don't apply to a tree at v5, so
2673 // the worker MUST fall back to full reparse rather than
2674 // silently corrupt the cached tree.
2675 let mut s = Syntax::for_language(Lang::Rust).unwrap().unwrap();
2676 s.parse_at("fn a() {}", 5);
2677 // Construct a delta that would be wrong for the cached
2678 // tree's actual state -- but since from_version mismatch
2679 // triggers fallback, the wrong delta is never applied.
2680 let edits = [EditDelta {
2681 start_byte: 100,
2682 old_end_byte: 100,
2683 new_end_byte: 105,
2684 start_position: lattice_protocol::Position::new(99, 0),
2685 old_end_position: lattice_protocol::Position::new(99, 0),
2686 new_end_position: lattice_protocol::Position::new(99, 5),
2687 }];
2688 // from_version=3 != cached version 5 -> full reparse.
2689 s.parse_at_with_edits("fn b() {}", 6, 3, &edits);
2690 // Result tree must match a full reparse on "fn b() {}",
2691 // not contain stale "fn a() {}" structure or weird
2692 // out-of-range nodes from the bogus delta.
2693 let mut s_full = Syntax::for_language(Lang::Rust).unwrap().unwrap();
2694 s_full.parse_at("fn b() {}", 6);
2695 assert_eq!(
2696 s.tree().unwrap().root_node().to_sexp(),
2697 s_full.tree().unwrap().root_node().to_sexp(),
2698 );
2699 }
2700
2701 #[test]
2702 fn parse_at_with_edits_falls_back_on_byte_length_mismatch() {
2703 // Edit claims to net +0 bytes (insert "ab", delete "cd")
2704 // but the actual source delta is +5 bytes. The byte-length
2705 // guard catches the dropped/missing edit and routes to
2706 // full reparse.
2707 let mut s = Syntax::for_language(Lang::Rust).unwrap().unwrap();
2708 let source_a = "fn a() {}";
2709 s.parse_at(source_a, 1);
2710 let edits = [EditDelta {
2711 start_byte: 3,
2712 old_end_byte: 4,
2713 new_end_byte: 4,
2714 start_position: lattice_protocol::Position::new(0, 3),
2715 old_end_position: lattice_protocol::Position::new(0, 4),
2716 new_end_position: lattice_protocol::Position::new(0, 4),
2717 }];
2718 // Source B is much longer than the edit accounts for ->
2719 // length mismatch -> full reparse.
2720 let source_b = "fn aaaaaaa() {}";
2721 s.parse_at_with_edits(source_b, 2, 1, &edits);
2722 // Verify result matches full reparse.
2723 let mut s_full = Syntax::for_language(Lang::Rust).unwrap().unwrap();
2724 s_full.parse_at(source_b, 2);
2725 assert_eq!(
2726 s.tree().unwrap().root_node().to_sexp(),
2727 s_full.tree().unwrap().root_node().to_sexp(),
2728 );
2729 }
2730
2731 #[test]
2732 fn edit_delta_to_input_edit_maps_fields_one_to_one() {
2733 let d = EditDelta {
2734 start_byte: 10,
2735 old_end_byte: 15,
2736 new_end_byte: 20,
2737 start_position: lattice_protocol::Position::new(2, 3),
2738 old_end_position: lattice_protocol::Position::new(2, 8),
2739 new_end_position: lattice_protocol::Position::new(2, 13),
2740 };
2741 let inp = edit_delta_to_input_edit(d);
2742 assert_eq!(inp.start_byte, 10);
2743 assert_eq!(inp.old_end_byte, 15);
2744 assert_eq!(inp.new_end_byte, 20);
2745 assert_eq!(inp.start_position.row, 2);
2746 assert_eq!(inp.start_position.column, 3);
2747 assert_eq!(inp.old_end_position.row, 2);
2748 assert_eq!(inp.old_end_position.column, 8);
2749 assert_eq!(inp.new_end_position.row, 2);
2750 assert_eq!(inp.new_end_position.column, 13);
2751 }
2752
2753 // ---- Slice B.4: parametrized parity matrix ------------------
2754 //
2755 // Tree-sitter's failure mode for a malformed `InputEdit` is a
2756 // silently wrong tree -- the parser produces a syntactically
2757 // valid tree whose node ranges are off, with no error. The
2758 // representative-shape parity tests above (B.2) catch the
2759 // common cases; this matrix broadens to the long tail:
2760 //
2761 // - **Edge positions**: edits at byte 0, edits at end-of-buffer,
2762 // edits at line boundaries.
2763 // - **Multi-line shape changes**: insert / delete newlines
2764 // so the line count itself shifts.
2765 // - **Sequential batches**: simulate keystroke bursts (each
2766 // delta operating on the post-prior-edit state) and indent-
2767 // style multi-line edits.
2768 // - **Per-language**: same shape in Rust / Python /
2769 // JavaScript so language-specific drift in the
2770 // `EditDelta -> InputEdit` mapping or `tree.edit()` semantics
2771 // surfaces.
2772 //
2773 // Each test asserts the incremental parse's tree s-expression
2774 // equals the full-reparse s-expression on the same final
2775 // source. Failures pinpoint the (language, edit shape) where
2776 // the deltas drift -- a precise regression net.
2777
2778 /// Build the post-edit source + an `EditDelta` for an edit
2779 /// described by `(start_byte, old_end_byte, new_text)`. Self-
2780 /// contained -- doesn't depend on `lattice-core::Buffer` so
2781 /// `lattice-syntax` tests stay free of that dependency.
2782 fn delta_for_edit(
2783 source_a: &str,
2784 start_byte: usize,
2785 old_end_byte: usize,
2786 new_text: &str,
2787 ) -> (String, EditDelta) {
2788 let pos_at = |byte: usize, src: &str| -> lattice_protocol::Position {
2789 let prefix = &src[..byte];
2790 let line = prefix.matches('\n').count() as u32;
2791 let col = (byte - prefix.rfind('\n').map(|i| i + 1).unwrap_or(0)) as u32;
2792 lattice_protocol::Position::new(line, col)
2793 };
2794 let mut source_b =
2795 String::with_capacity(source_a.len() - (old_end_byte - start_byte) + new_text.len());
2796 source_b.push_str(&source_a[..start_byte]);
2797 source_b.push_str(new_text);
2798 source_b.push_str(&source_a[old_end_byte..]);
2799 let new_end_byte = start_byte + new_text.len();
2800 let delta = EditDelta {
2801 start_byte: start_byte as u32,
2802 old_end_byte: old_end_byte as u32,
2803 new_end_byte: new_end_byte as u32,
2804 start_position: pos_at(start_byte, source_a),
2805 old_end_position: pos_at(old_end_byte, source_a),
2806 new_end_position: pos_at(new_end_byte, &source_b),
2807 };
2808 (source_b, delta)
2809 }
2810
2811 /// Apply a sequence of edits to `source_a`, returning the
2812 /// final source + the per-edit deltas (in apply order). Each
2813 /// edit's positions are derived against the buffer state
2814 /// AFTER the prior edit applied -- mirroring how App's
2815 /// chokepoint produces deltas via successive
2816 /// `Buffer::apply_edit` calls.
2817 fn apply_sequential_edits(
2818 source_a: &str,
2819 edits: &[(usize, usize, &str)],
2820 ) -> (String, Vec<EditDelta>) {
2821 let mut current = source_a.to_string();
2822 let mut deltas = Vec::with_capacity(edits.len());
2823 for (start, old_end, new_text) in edits {
2824 let (next, delta) = delta_for_edit(¤t, *start, *old_end, new_text);
2825 current = next;
2826 deltas.push(delta);
2827 }
2828 (current, deltas)
2829 }
2830
2831 /// Run incremental + full reparse on `source_a` -> `source_b`
2832 /// via `edits`, asserting tree-shape equality. Failure
2833 /// message names the language + the source pair.
2834 fn assert_parity(lang: Lang, source_a: &str, edits: &[EditDelta], source_b: &str, case: &str) {
2835 let mut s_inc = Syntax::for_language(lang).unwrap().unwrap();
2836 s_inc.parse_at(source_a, 1);
2837 s_inc.parse_at_with_edits(source_b, 2, 1, edits);
2838 let inc = s_inc.tree().unwrap().root_node().to_sexp();
2839
2840 let mut s_full = Syntax::for_language(lang).unwrap().unwrap();
2841 s_full.parse_at(source_b, 1);
2842 let full = s_full.tree().unwrap().root_node().to_sexp();
2843
2844 assert_eq!(
2845 inc, full,
2846 "incremental != full reparse for {lang:?} / {case}\n source_a: {source_a:?}\n source_b: {source_b:?}"
2847 );
2848 }
2849
2850 /// Single-edit parity helper: derive delta from
2851 /// `(start_byte, old_end_byte, new_text)`, run parity check.
2852 fn assert_single_edit_parity(
2853 lang: Lang,
2854 source_a: &str,
2855 start_byte: usize,
2856 old_end_byte: usize,
2857 new_text: &str,
2858 case: &str,
2859 ) {
2860 let (source_b, delta) = delta_for_edit(source_a, start_byte, old_end_byte, new_text);
2861 assert_parity(lang, source_a, &[delta], &source_b, case);
2862 }
2863
2864 // ==== Edge-position single edits ============================
2865
2866 #[test]
2867 fn parity_insert_at_byte_zero_rust() {
2868 assert_single_edit_parity(
2869 Lang::Rust,
2870 "fn main() {}",
2871 0,
2872 0,
2873 "// header\n",
2874 "insert at byte 0",
2875 );
2876 }
2877
2878 #[test]
2879 fn parity_insert_at_end_of_buffer_rust() {
2880 let src = "fn main() {}";
2881 assert_single_edit_parity(
2882 Lang::Rust,
2883 src,
2884 src.len(),
2885 src.len(),
2886 "\nfn b() {}",
2887 "insert at end",
2888 );
2889 }
2890
2891 #[test]
2892 fn parity_delete_first_char_rust() {
2893 assert_single_edit_parity(Lang::Rust, "Xfn main() {}", 0, 1, "", "delete first byte");
2894 }
2895
2896 #[test]
2897 fn parity_delete_last_char_rust() {
2898 let src = "fn main() {};";
2899 assert_single_edit_parity(
2900 Lang::Rust,
2901 src,
2902 src.len() - 1,
2903 src.len(),
2904 "",
2905 "delete last byte",
2906 );
2907 }
2908
2909 #[test]
2910 fn parity_replace_whole_buffer_rust() {
2911 let src = "fn a() {}";
2912 assert_single_edit_parity(
2913 Lang::Rust,
2914 src,
2915 0,
2916 src.len(),
2917 "fn b(x: i32) -> i32 { x + 1 }",
2918 "replace whole buffer",
2919 );
2920 }
2921
2922 #[test]
2923 fn parity_insert_at_line_boundary_rust() {
2924 // After the newline ending line 0; before any content
2925 // on line 1.
2926 let src = "fn a() {}\n";
2927 assert_single_edit_parity(
2928 Lang::Rust,
2929 src,
2930 10,
2931 10,
2932 "fn b() {}\n",
2933 "insert at line boundary",
2934 );
2935 }
2936
2937 // ==== Multi-line shape changes ==============================
2938
2939 #[test]
2940 fn parity_insert_newline_splitting_a_line_rust() {
2941 // Insert "\n " mid-statement, breaking it across lines.
2942 let src = "fn a() { let x = 1; }";
2943 assert_single_edit_parity(Lang::Rust, src, 9, 9, "\n ", "insert newline mid-line");
2944 }
2945
2946 #[test]
2947 fn parity_delete_newline_joining_lines_rust() {
2948 // Source has two lines; delete the connecting newline.
2949 let src = "fn a() {\n 1;\n}";
2950 assert_single_edit_parity(Lang::Rust, src, 8, 9, "", "delete newline");
2951 }
2952
2953 #[test]
2954 fn parity_replace_single_with_multi_line_rust() {
2955 let src = "fn a() { 1 }";
2956 assert_single_edit_parity(
2957 Lang::Rust,
2958 src,
2959 9,
2960 10,
2961 "\n let x = 1;\n x\n",
2962 "single-line -> multi-line",
2963 );
2964 }
2965
2966 #[test]
2967 fn parity_replace_multi_with_single_line_rust() {
2968 let src = "fn a() {\n let x = 1;\n x\n}";
2969 // Replace lines 1-2 (" let x = 1;\n x\n") with " 42 ".
2970 assert_single_edit_parity(Lang::Rust, src, 9, 30, " 42 ", "multi-line -> single-line");
2971 }
2972
2973 // ==== Whitespace-only edits =================================
2974
2975 #[test]
2976 fn parity_insert_indent_whitespace_rust() {
2977 let src = "fn a() {\nlet x = 1;\n}";
2978 assert_single_edit_parity(Lang::Rust, src, 9, 9, " ", "insert indentation");
2979 }
2980
2981 #[test]
2982 fn parity_delete_trailing_whitespace_rust() {
2983 let src = "fn a() { \n}";
2984 assert_single_edit_parity(Lang::Rust, src, 8, 12, "", "delete trailing whitespace");
2985 }
2986
2987 // ==== Sequential edit batches ===============================
2988
2989 #[test]
2990 fn parity_three_keystroke_burst_rust() {
2991 // Simulate typing "abc" one char at a time inside an
2992 // identifier slot.
2993 let (source_b, deltas) =
2994 apply_sequential_edits("fn () {}", &[(3, 3, "a"), (4, 4, "b"), (5, 5, "c")]);
2995 assert_parity(
2996 Lang::Rust,
2997 "fn () {}",
2998 &deltas,
2999 &source_b,
3000 "3-keystroke burst",
3001 );
3002 }
3003
3004 #[test]
3005 fn parity_indent_batch_rust() {
3006 // Simulate `>>` over two lines: insert 4 spaces at the
3007 // start of each. Each subsequent edit's position is
3008 // shifted by the prior edit's effect.
3009 let src = "fn a() {\nlet x = 1;\nlet y = 2;\n}";
3010 // Line 1 starts at byte 9; line 2 starts at byte 20 in
3011 // the original. After inserting 4 spaces at byte 9, line
3012 // 2 starts at byte 24.
3013 let (source_b, deltas) = apply_sequential_edits(src, &[(9, 9, " "), (24, 24, " ")]);
3014 assert_parity(Lang::Rust, src, &deltas, &source_b, "indent batch");
3015 }
3016
3017 #[test]
3018 fn parity_backspace_burst_rust() {
3019 // Simulate pressing backspace 3 times -- delete one byte
3020 // at a time from a known position.
3021 let src = "fn aaaa() {}";
3022 let (source_b, deltas) = apply_sequential_edits(src, &[(6, 7, ""), (5, 6, ""), (4, 5, "")]);
3023 assert_parity(Lang::Rust, src, &deltas, &source_b, "backspace burst");
3024 }
3025
3026 // ==== Per-language coverage =================================
3027
3028 #[test]
3029 fn parity_python_insert_in_def() {
3030 let src = "def f(x):\n return x\n";
3031 assert_single_edit_parity(Lang::Python, src, 6, 6, "y, ", "insert arg in python def");
3032 }
3033
3034 #[test]
3035 fn parity_python_delete_a_line() {
3036 let src = "def f(x):\n y = 1\n return x + y\n";
3037 // Delete " y = 1\n" (indices 10..23).
3038 assert_single_edit_parity(Lang::Python, src, 10, 23, "", "delete line in python");
3039 }
3040
3041 #[test]
3042 fn parity_python_replace_function_body() {
3043 let src = "def f(x):\n return x\n";
3044 assert_single_edit_parity(
3045 Lang::Python,
3046 src,
3047 10,
3048 22,
3049 " return x * 2",
3050 "replace python body",
3051 );
3052 }
3053
3054 #[test]
3055 fn parity_javascript_insert_in_function() {
3056 let src = "function f(x) { return x; }";
3057 assert_single_edit_parity(
3058 Lang::JavaScript,
3059 src,
3060 24,
3061 24,
3062 " + 1",
3063 "insert in JS function",
3064 );
3065 }
3066
3067 #[test]
3068 fn parity_javascript_replace_arrow_body() {
3069 let src = "const f = (x) => x;";
3070 assert_single_edit_parity(Lang::JavaScript, src, 17, 18, "x * 2", "replace arrow body");
3071 }
3072
3073 #[test]
3074 fn parity_javascript_indent_batch() {
3075 let src = "function f() {\nlet x = 1;\nlet y = 2;\n}";
3076 let (source_b, deltas) = apply_sequential_edits(src, &[(15, 15, " "), (28, 28, " ")]);
3077 assert_parity(Lang::JavaScript, src, &deltas, &source_b, "JS indent batch");
3078 }
3079
3080 // ==== Pathological / minimal-buffer cases ===================
3081
3082 #[test]
3083 fn parity_edit_in_single_char_buffer_rust() {
3084 // Single-char buffer: delete the only char.
3085 assert_single_edit_parity(Lang::Rust, ";", 0, 1, "", "delete only char");
3086 }
3087
3088 #[test]
3089 fn parity_insert_into_minimal_buffer_rust() {
3090 // Empty / near-empty buffer; insert valid syntax.
3091 assert_single_edit_parity(
3092 Lang::Rust,
3093 "fn",
3094 2,
3095 2,
3096 " a() {}",
3097 "insert into minimal buffer",
3098 );
3099 }
3100
3101 #[test]
3102 fn parity_replace_with_empty_string_rust() {
3103 // Pure delete via replace with empty new text.
3104 let src = "fn a() {}\nfn b() {}";
3105 assert_single_edit_parity(Lang::Rust, src, 9, 19, "", "replace with empty");
3106 }
3107
3108 // ==== Markdown =============================================
3109 //
3110 // Markdown is the trickiest language because of its
3111 // injections (block grammar -> inline grammar -> any fenced
3112 // language). The full-reparse path runs the same injection
3113 // pipeline as incremental, so tree-shape parity at the
3114 // outer (block) level is the right invariant -- inner
3115 // injection trees aren't part of `tree()` (they're managed
3116 // by the highlighter, not the parser).
3117
3118 #[test]
3119 fn parity_markdown_insert_heading() {
3120 let src = "body paragraph\n";
3121 assert_single_edit_parity(Lang::Markdown, src, 0, 0, "# Title\n\n", "insert heading");
3122 }
3123
3124 #[test]
3125 fn parity_markdown_replace_in_paragraph() {
3126 let src = "first line\nsecond line\n";
3127 assert_single_edit_parity(
3128 Lang::Markdown,
3129 src,
3130 6,
3131 10,
3132 "FOO",
3133 "replace word in paragraph",
3134 );
3135 }
3136
3137 // ---- Slice C.2: intermediate-snapshot byte-alignment -------
3138 //
3139 // try_apply_intermediate must produce a snapshot whose tree's
3140 // byte ranges match the new source. Spans walked against this
3141 // intermediate must land at correct byte positions even though
3142 // Parser::parse hasn't run yet -- that's what makes lines
3143 // below a delete (or after a multi-byte insert) paint without
3144 // flicker.
3145
3146 #[test]
3147 fn try_apply_intermediate_shifts_byte_ranges_to_new_source() {
3148 // Insert "X" at byte 3 of "fn a() {}" -> "fn aX() {}".
3149 // The function-name node was [3, 4) for "a"; after
3150 // tree.edit it should span [3, 5) for "aX". Pre-parse
3151 // shape (the tree still thinks of it as a single
3152 // identifier node), but byte ranges shifted.
3153 let mut s = Syntax::for_language(Lang::Rust).unwrap().unwrap();
3154 s.parse_at("fn a() {}", 1);
3155 let edit = EditDelta {
3156 start_byte: 4,
3157 old_end_byte: 4,
3158 new_end_byte: 5,
3159 start_position: lattice_protocol::Position::new(0, 4),
3160 old_end_position: lattice_protocol::Position::new(0, 4),
3161 new_end_position: lattice_protocol::Position::new(0, 5),
3162 };
3163 let new_source = "fn aX() {}";
3164 s.try_apply_intermediate(new_source, 2, 1, &[edit])
3165 .expect("intermediate should succeed");
3166 // Source updated.
3167 assert_eq!(s.source(), new_source.as_bytes());
3168 // Tree present, byte ranges shifted.
3169 let tree = s.tree().expect("tree present");
3170 let root = tree.root_node();
3171 // Find the source_file's last byte; should match new
3172 // source length (10).
3173 assert_eq!(root.end_byte(), new_source.len());
3174 }
3175
3176 #[test]
3177 fn try_apply_intermediate_then_reparse_matches_parse_at_with_edits() {
3178 // Two-stage path (try_apply_intermediate + reparse_with_cached_tree)
3179 // must produce the SAME final tree as the convenience
3180 // parse_at_with_edits path. Sanity check on the split.
3181 let edit = EditDelta {
3182 start_byte: 4,
3183 old_end_byte: 4,
3184 new_end_byte: 5,
3185 start_position: lattice_protocol::Position::new(0, 4),
3186 old_end_position: lattice_protocol::Position::new(0, 4),
3187 new_end_position: lattice_protocol::Position::new(0, 5),
3188 };
3189 let mut split = Syntax::for_language(Lang::Rust).unwrap().unwrap();
3190 split.parse_at("fn a() {}", 1);
3191 split
3192 .try_apply_intermediate("fn aX() {}", 2, 1, &[edit])
3193 .unwrap();
3194 split.reparse_with_cached_tree(1);
3195
3196 let mut combined = Syntax::for_language(Lang::Rust).unwrap().unwrap();
3197 combined.parse_at("fn a() {}", 1);
3198 combined.parse_at_with_edits("fn aX() {}", 2, 1, &[edit]);
3199
3200 assert_eq!(
3201 split.tree().unwrap().root_node().to_sexp(),
3202 combined.tree().unwrap().root_node().to_sexp(),
3203 );
3204 }
3205
3206 #[test]
3207 fn try_apply_intermediate_shifts_ranges_after_line_delete() {
3208 // The user-reported scenario: deleting a line should
3209 // shift every subsequent byte range. Verify highlight
3210 // spans land at correct positions in the new source
3211 // BEFORE Parser::parse runs.
3212 let source_a = "fn a() {}\nfn b() {}\nfn c() {}";
3213 let source_b = "fn a() {}\nfn c() {}";
3214 // Delete "fn b() {}\n" at bytes 10..20.
3215 let edit = EditDelta {
3216 start_byte: 10,
3217 old_end_byte: 20,
3218 new_end_byte: 10,
3219 start_position: lattice_protocol::Position::new(1, 0),
3220 old_end_position: lattice_protocol::Position::new(2, 0),
3221 new_end_position: lattice_protocol::Position::new(1, 0),
3222 };
3223 let mut s = Syntax::for_language(Lang::Rust).unwrap().unwrap();
3224 s.parse_at(source_a, 1);
3225 s.try_apply_intermediate(source_b, 2, 1, &[edit])
3226 .expect("intermediate should succeed");
3227 // Highlight against the intermediate. The "fn" keyword
3228 // span on the second line of the new source (was the
3229 // third line of the old source) should land at byte 10
3230 // (start of "fn c"), not at byte 20 (where the OLD tree
3231 // would have placed it without the tree.edit shift).
3232 let lines = s.highlight_lines(0, 2).unwrap();
3233 // Line 1 of the new source is "fn c() {}" -- it should
3234 // have at least one span (the "fn" keyword). With pre-
3235 // C.2 (no tree.edit shift), that span would land on the
3236 // wrong line.
3237 assert!(
3238 !lines[1].is_empty(),
3239 "line 1 of intermediate (post-delete) must have spans -- \
3240 this is what eliminates 'lines below flicker' on line delete"
3241 );
3242 }
3243
3244 #[test]
3245 fn parity_markdown_delete_fenced_block() {
3246 let src = "intro\n\n```rust\nfn x() {}\n```\n\noutro\n";
3247 // Delete the fenced block (bytes 7..29 == "```rust\nfn x() {}\n```\n").
3248 assert_single_edit_parity(Lang::Markdown, src, 7, 29, "", "delete fenced code block");
3249 }
3250}