Skip to main content

lattice_lsp/
position.rs

1//! Position-encoding conversion (LSP 3.17 §3.17 / §General).
2//!
3//! Lattice uses UTF-8 byte offsets internally
4//! (`lattice_protocol::Position { line, byte }`). LSP sends:
5//!
6//! - **utf-8** (`PositionEncodingKind::UTF8`): column is the
7//!   UTF-8 byte offset within the line. Identical to ours -- no
8//!   conversion needed.
9//! - **utf-16** (`PositionEncodingKind::UTF16`, the LSP 3.16
10//!   default): column is the UTF-16 code-unit offset within the
11//!   line. We convert when the negotiated encoding is utf-16.
12//! - **utf-32** (`PositionEncodingKind::UTF32`): column is the
13//!   Unicode codepoint count. Practically unused; we don't
14//!   advertise support, but the converter is here in case a
15//!   server demands it.
16//!
17//! The converters work line-by-line: callers pass the line text
18//! plus an offset and get back the offset in the target
19//! encoding. Lattice's `Position::byte` is always within a
20//! single line, so we never have to walk multiple lines.
21//!
22//! ## Performance
23//!
24//! `utf8_byte_to_utf16_column` is `O(byte)` -- it walks the
25//! prefix and counts UTF-16 code units. For ASCII lines this
26//! collapses to `byte` (no multi-byte chars). The bench
27//! `lsp::position::utf8_to_utf16` measures the worst case
28//! (line of CJK glyphs) at sub-microsecond.
29
30use lsp_types::PositionEncodingKind;
31
32/// Convert a UTF-8 byte offset within `line` to the offset in
33/// the negotiated `encoding`. Used when constructing LSP
34/// `Position::character` from lattice's `Position::byte`.
35///
36/// Bounds: `byte` may equal `line.len()` (one-past-the-end);
37/// callers passing larger values get the count for the whole
38/// line plus their over-shoot, matching Rust's `&str` slicing
39/// semantics.
40pub fn byte_to_lsp_character(line: &str, byte: u32, encoding: &PositionEncodingKind) -> u32 {
41    if encoding == &PositionEncodingKind::UTF8 {
42        return byte;
43    }
44    if encoding == &PositionEncodingKind::UTF32 {
45        return utf8_byte_to_utf32_column(line, byte);
46    }
47    // utf-16 is the spec default and our fallback.
48    utf8_byte_to_utf16_column(line, byte)
49}
50
51/// Convert an LSP `character` value (in the negotiated encoding)
52/// to a UTF-8 byte offset within `line`. Used for ranges that
53/// arrive FROM the server (definitions, diagnostics, etc.).
54pub fn lsp_character_to_byte(line: &str, character: u32, encoding: &PositionEncodingKind) -> u32 {
55    if encoding == &PositionEncodingKind::UTF8 {
56        // Clamp to line length so a server reporting an offset
57        // past EOL doesn't yield an out-of-bounds byte index.
58        return character.min(line.len() as u32);
59    }
60    if encoding == &PositionEncodingKind::UTF32 {
61        return utf32_column_to_utf8_byte(line, character);
62    }
63    utf16_column_to_utf8_byte(line, character)
64}
65
66/// UTF-8 byte offset → UTF-16 code-unit offset within `line`.
67/// `byte` is treated as a position within the line text;
68/// if it's past the end the function returns the line's full
69/// utf-16 length plus the over-shoot in bytes (a useful approximation
70/// for clamped callers, but production paths shouldn't pass
71/// beyond `line.len()`).
72pub fn utf8_byte_to_utf16_column(line: &str, byte: u32) -> u32 {
73    let cap = line.len() as u32;
74    if byte == 0 {
75        return 0;
76    }
77    if byte >= cap {
78        // Whole line fits.
79        return line.encode_utf16().count() as u32 + byte.saturating_sub(cap);
80    }
81    // SAFETY: callers pass a byte offset that lies on a UTF-8
82    // boundary -- guaranteed by `lattice_core::Buffer` (Position
83    // .byte is always at a char boundary). If a buggy caller
84    // breaks this, `is_char_boundary` catches it and we fall
85    // back to a safe approximation: walk char-by-char.
86    if !line.is_char_boundary(byte as usize) {
87        return line
88            .char_indices()
89            .take_while(|(i, _)| (*i as u32) < byte)
90            .fold(0u32, |acc, (_, c)| acc + c.len_utf16() as u32);
91    }
92    let prefix = &line[..byte as usize];
93    prefix.encode_utf16().count() as u32
94}
95
96/// UTF-16 code-unit offset → UTF-8 byte offset within `line`.
97/// Walks chars accumulating utf-16 units; stops when the running
98/// count reaches `character`.
99pub fn utf16_column_to_utf8_byte(line: &str, character: u32) -> u32 {
100    let mut units_seen: u32 = 0;
101    let mut byte: u32 = 0;
102    for c in line.chars() {
103        if units_seen >= character {
104            return byte;
105        }
106        units_seen = units_seen.saturating_add(c.len_utf16() as u32);
107        byte = byte.saturating_add(c.len_utf8() as u32);
108    }
109    // Past end of line: clamp to the line's byte length.
110    byte
111}
112
113/// UTF-8 byte offset → UTF-32 codepoint offset within `line`.
114pub fn utf8_byte_to_utf32_column(line: &str, byte: u32) -> u32 {
115    let cap = line.len() as u32;
116    if byte == 0 {
117        return 0;
118    }
119    if byte >= cap {
120        return line.chars().count() as u32 + byte.saturating_sub(cap);
121    }
122    if !line.is_char_boundary(byte as usize) {
123        return line
124            .char_indices()
125            .take_while(|(i, _)| (*i as u32) < byte)
126            .count() as u32;
127    }
128    line[..byte as usize].chars().count() as u32
129}
130
131/// UTF-32 codepoint offset → UTF-8 byte offset.
132pub fn utf32_column_to_utf8_byte(line: &str, character: u32) -> u32 {
133    let mut chars_seen: u32 = 0;
134    let mut byte: u32 = 0;
135    for c in line.chars() {
136        if chars_seen >= character {
137            return byte;
138        }
139        chars_seen = chars_seen.saturating_add(1);
140        byte = byte.saturating_add(c.len_utf8() as u32);
141    }
142    byte
143}
144
145#[cfg(test)]
146mod tests {
147    use super::*;
148
149    #[test]
150    fn ascii_byte_equals_utf16_column() {
151        let line = "hello world";
152        assert_eq!(utf8_byte_to_utf16_column(line, 0), 0);
153        assert_eq!(utf8_byte_to_utf16_column(line, 5), 5);
154        assert_eq!(utf8_byte_to_utf16_column(line, 11), 11);
155    }
156
157    #[test]
158    fn latin1_two_byte_one_utf16_unit() {
159        // `é` is U+00E9, 2 bytes in UTF-8, 1 UTF-16 code unit.
160        let line = "café";
161        assert_eq!(line.len(), 5); // "caf" + 2 bytes for é
162        // byte 4 is start of é, byte 5 is past it.
163        assert_eq!(utf8_byte_to_utf16_column(line, 3), 3); // "caf"
164        assert_eq!(utf8_byte_to_utf16_column(line, 5), 4); // "café"
165    }
166
167    #[test]
168    fn cjk_three_byte_one_utf16_unit() {
169        // CJK ideograph U+4E2D '中': 3 UTF-8 bytes, 1 UTF-16 unit.
170        let line = "中文";
171        assert_eq!(line.len(), 6);
172        assert_eq!(utf8_byte_to_utf16_column(line, 0), 0);
173        assert_eq!(utf8_byte_to_utf16_column(line, 3), 1); // after '中'
174        assert_eq!(utf8_byte_to_utf16_column(line, 6), 2); // after '文'
175    }
176
177    #[test]
178    fn supplementary_plane_four_bytes_two_utf16_units() {
179        // U+1F600 '😀': 4 UTF-8 bytes, 2 UTF-16 code units (surrogate pair).
180        let line = "x😀y";
181        assert_eq!(line.len(), 6); // 1 + 4 + 1
182        assert_eq!(utf8_byte_to_utf16_column(line, 1), 1); // after 'x'
183        assert_eq!(utf8_byte_to_utf16_column(line, 5), 3); // after '😀' = 1 + 2
184        assert_eq!(utf8_byte_to_utf16_column(line, 6), 4); // after 'y'
185    }
186
187    #[test]
188    fn utf16_to_byte_round_trips_ascii() {
189        let line = "hello";
190        for byte in 0..=line.len() as u32 {
191            let col = utf8_byte_to_utf16_column(line, byte);
192            assert_eq!(utf16_column_to_utf8_byte(line, col), byte);
193        }
194    }
195
196    #[test]
197    fn utf16_to_byte_round_trips_unicode() {
198        let line = "x😀y中z";
199        for byte in [0, 1, 5, 6, 9, 10] {
200            let col = utf8_byte_to_utf16_column(line, byte);
201            assert_eq!(utf16_column_to_utf8_byte(line, col), byte, "byte={byte}");
202        }
203    }
204
205    #[test]
206    fn utf16_to_byte_clamps_past_eol() {
207        // Server says character=999; we clamp to line length.
208        let line = "abc";
209        assert_eq!(utf16_column_to_utf8_byte(line, 999), 3);
210    }
211
212    #[test]
213    fn utf32_byte_round_trip() {
214        let line = "x😀y";
215        // 😀 is one codepoint in utf-32.
216        // After x: byte=1, col=1
217        // After 😀: byte=5, col=2
218        // After y: byte=6, col=3
219        assert_eq!(utf8_byte_to_utf32_column(line, 0), 0);
220        assert_eq!(utf8_byte_to_utf32_column(line, 1), 1);
221        assert_eq!(utf8_byte_to_utf32_column(line, 5), 2);
222        assert_eq!(utf8_byte_to_utf32_column(line, 6), 3);
223
224        for col in 0..=3 {
225            let byte = utf32_column_to_utf8_byte(line, col);
226            assert_eq!(utf8_byte_to_utf32_column(line, byte), col);
227        }
228    }
229
230    #[test]
231    fn dispatch_utf8_short_circuits() {
232        let line = "x😀y";
233        assert_eq!(
234            byte_to_lsp_character(line, 5, &PositionEncodingKind::UTF8),
235            5,
236            "utf-8 mode preserves byte offset"
237        );
238    }
239
240    #[test]
241    fn dispatch_utf16_routes_to_utf16_converter() {
242        let line = "x😀y";
243        assert_eq!(
244            byte_to_lsp_character(line, 5, &PositionEncodingKind::UTF16),
245            3
246        );
247    }
248
249    #[test]
250    fn dispatch_utf32_routes_to_utf32_converter() {
251        let line = "x😀y";
252        assert_eq!(
253            byte_to_lsp_character(line, 5, &PositionEncodingKind::UTF32),
254            2
255        );
256    }
257
258    #[test]
259    fn one_past_end_is_handled() {
260        let line = "ab";
261        // byte=2 is one-past-the-end of "ab".
262        assert_eq!(utf8_byte_to_utf16_column(line, 2), 2);
263        // Past the end -- approximation.
264        assert_eq!(utf8_byte_to_utf16_column(line, 5), 5);
265    }
266
267    #[test]
268    fn empty_line_returns_zero() {
269        assert_eq!(utf8_byte_to_utf16_column("", 0), 0);
270        assert_eq!(utf16_column_to_utf8_byte("", 0), 0);
271    }
272
273    #[test]
274    fn non_char_boundary_falls_back_to_safe_walk() {
275        // byte=2 inside the multi-byte CJK char '中' (3 bytes).
276        // Production code shouldn't pass this, but if it does we
277        // mustn't panic. The fallback walks char_indices and
278        // counts every char whose start index is < 2; '中' starts
279        // at 0 so it counts -> 1 utf-16 unit.
280        let line = "中";
281        assert_eq!(utf8_byte_to_utf16_column(line, 2), 1);
282    }
283}