1use lsp_types::PositionEncodingKind;
31
32pub fn byte_to_lsp_character(line: &str, byte: u32, encoding: &PositionEncodingKind) -> u32 {
41 if encoding == &PositionEncodingKind::UTF8 {
42 return byte;
43 }
44 if encoding == &PositionEncodingKind::UTF32 {
45 return utf8_byte_to_utf32_column(line, byte);
46 }
47 utf8_byte_to_utf16_column(line, byte)
49}
50
51pub fn lsp_character_to_byte(line: &str, character: u32, encoding: &PositionEncodingKind) -> u32 {
55 if encoding == &PositionEncodingKind::UTF8 {
56 return character.min(line.len() as u32);
59 }
60 if encoding == &PositionEncodingKind::UTF32 {
61 return utf32_column_to_utf8_byte(line, character);
62 }
63 utf16_column_to_utf8_byte(line, character)
64}
65
66pub fn utf8_byte_to_utf16_column(line: &str, byte: u32) -> u32 {
73 let cap = line.len() as u32;
74 if byte == 0 {
75 return 0;
76 }
77 if byte >= cap {
78 return line.encode_utf16().count() as u32 + byte.saturating_sub(cap);
80 }
81 if !line.is_char_boundary(byte as usize) {
87 return line
88 .char_indices()
89 .take_while(|(i, _)| (*i as u32) < byte)
90 .fold(0u32, |acc, (_, c)| acc + c.len_utf16() as u32);
91 }
92 let prefix = &line[..byte as usize];
93 prefix.encode_utf16().count() as u32
94}
95
96pub fn utf16_column_to_utf8_byte(line: &str, character: u32) -> u32 {
100 let mut units_seen: u32 = 0;
101 let mut byte: u32 = 0;
102 for c in line.chars() {
103 if units_seen >= character {
104 return byte;
105 }
106 units_seen = units_seen.saturating_add(c.len_utf16() as u32);
107 byte = byte.saturating_add(c.len_utf8() as u32);
108 }
109 byte
111}
112
113pub fn utf8_byte_to_utf32_column(line: &str, byte: u32) -> u32 {
115 let cap = line.len() as u32;
116 if byte == 0 {
117 return 0;
118 }
119 if byte >= cap {
120 return line.chars().count() as u32 + byte.saturating_sub(cap);
121 }
122 if !line.is_char_boundary(byte as usize) {
123 return line
124 .char_indices()
125 .take_while(|(i, _)| (*i as u32) < byte)
126 .count() as u32;
127 }
128 line[..byte as usize].chars().count() as u32
129}
130
131pub fn utf32_column_to_utf8_byte(line: &str, character: u32) -> u32 {
133 let mut chars_seen: u32 = 0;
134 let mut byte: u32 = 0;
135 for c in line.chars() {
136 if chars_seen >= character {
137 return byte;
138 }
139 chars_seen = chars_seen.saturating_add(1);
140 byte = byte.saturating_add(c.len_utf8() as u32);
141 }
142 byte
143}
144
145#[cfg(test)]
146mod tests {
147 use super::*;
148
149 #[test]
150 fn ascii_byte_equals_utf16_column() {
151 let line = "hello world";
152 assert_eq!(utf8_byte_to_utf16_column(line, 0), 0);
153 assert_eq!(utf8_byte_to_utf16_column(line, 5), 5);
154 assert_eq!(utf8_byte_to_utf16_column(line, 11), 11);
155 }
156
157 #[test]
158 fn latin1_two_byte_one_utf16_unit() {
159 let line = "café";
161 assert_eq!(line.len(), 5); assert_eq!(utf8_byte_to_utf16_column(line, 3), 3); assert_eq!(utf8_byte_to_utf16_column(line, 5), 4); }
166
167 #[test]
168 fn cjk_three_byte_one_utf16_unit() {
169 let line = "中文";
171 assert_eq!(line.len(), 6);
172 assert_eq!(utf8_byte_to_utf16_column(line, 0), 0);
173 assert_eq!(utf8_byte_to_utf16_column(line, 3), 1); assert_eq!(utf8_byte_to_utf16_column(line, 6), 2); }
176
177 #[test]
178 fn supplementary_plane_four_bytes_two_utf16_units() {
179 let line = "x😀y";
181 assert_eq!(line.len(), 6); assert_eq!(utf8_byte_to_utf16_column(line, 1), 1); assert_eq!(utf8_byte_to_utf16_column(line, 5), 3); assert_eq!(utf8_byte_to_utf16_column(line, 6), 4); }
186
187 #[test]
188 fn utf16_to_byte_round_trips_ascii() {
189 let line = "hello";
190 for byte in 0..=line.len() as u32 {
191 let col = utf8_byte_to_utf16_column(line, byte);
192 assert_eq!(utf16_column_to_utf8_byte(line, col), byte);
193 }
194 }
195
196 #[test]
197 fn utf16_to_byte_round_trips_unicode() {
198 let line = "x😀y中z";
199 for byte in [0, 1, 5, 6, 9, 10] {
200 let col = utf8_byte_to_utf16_column(line, byte);
201 assert_eq!(utf16_column_to_utf8_byte(line, col), byte, "byte={byte}");
202 }
203 }
204
205 #[test]
206 fn utf16_to_byte_clamps_past_eol() {
207 let line = "abc";
209 assert_eq!(utf16_column_to_utf8_byte(line, 999), 3);
210 }
211
212 #[test]
213 fn utf32_byte_round_trip() {
214 let line = "x😀y";
215 assert_eq!(utf8_byte_to_utf32_column(line, 0), 0);
220 assert_eq!(utf8_byte_to_utf32_column(line, 1), 1);
221 assert_eq!(utf8_byte_to_utf32_column(line, 5), 2);
222 assert_eq!(utf8_byte_to_utf32_column(line, 6), 3);
223
224 for col in 0..=3 {
225 let byte = utf32_column_to_utf8_byte(line, col);
226 assert_eq!(utf8_byte_to_utf32_column(line, byte), col);
227 }
228 }
229
230 #[test]
231 fn dispatch_utf8_short_circuits() {
232 let line = "x😀y";
233 assert_eq!(
234 byte_to_lsp_character(line, 5, &PositionEncodingKind::UTF8),
235 5,
236 "utf-8 mode preserves byte offset"
237 );
238 }
239
240 #[test]
241 fn dispatch_utf16_routes_to_utf16_converter() {
242 let line = "x😀y";
243 assert_eq!(
244 byte_to_lsp_character(line, 5, &PositionEncodingKind::UTF16),
245 3
246 );
247 }
248
249 #[test]
250 fn dispatch_utf32_routes_to_utf32_converter() {
251 let line = "x😀y";
252 assert_eq!(
253 byte_to_lsp_character(line, 5, &PositionEncodingKind::UTF32),
254 2
255 );
256 }
257
258 #[test]
259 fn one_past_end_is_handled() {
260 let line = "ab";
261 assert_eq!(utf8_byte_to_utf16_column(line, 2), 2);
263 assert_eq!(utf8_byte_to_utf16_column(line, 5), 5);
265 }
266
267 #[test]
268 fn empty_line_returns_zero() {
269 assert_eq!(utf8_byte_to_utf16_column("", 0), 0);
270 assert_eq!(utf16_column_to_utf8_byte("", 0), 0);
271 }
272
273 #[test]
274 fn non_char_boundary_falls_back_to_safe_walk() {
275 let line = "中";
281 assert_eq!(utf8_byte_to_utf16_column(line, 2), 1);
282 }
283}