turbopack_core/source_pos.rs
1use bincode::{Decode, Encode};
2use serde::Serialize;
3use turbo_tasks_hash::DeterministicHash;
4
5/// LINE FEED (LF), one of the basic JS line terminators.
6const U8_LF: u8 = 0x0A;
7/// CARRIAGE RETURN (CR), one of the basic JS line terminators.
8const U8_CR: u8 = 0x0D;
9
10#[turbo_tasks::task_input]
11#[derive(
12 Default,
13 Debug,
14 PartialEq,
15 Eq,
16 Copy,
17 Clone,
18 Hash,
19 PartialOrd,
20 Ord,
21 Serialize,
22 DeterministicHash,
23 Encode,
24 Decode,
25)]
26pub struct SourcePos {
27 /// The line, 0-indexed.
28 pub line: u32,
29 /// The byte index of the column, 0-indexed.
30 pub column: u32,
31}
32
33impl SourcePos {
34 pub fn new(start_line: u32) -> Self {
35 Self {
36 line: start_line,
37 column: 0,
38 }
39 }
40
41 pub fn max() -> Self {
42 Self {
43 line: u32::MAX,
44 column: u32::MAX,
45 }
46 }
47
48 /// Increments the line/column position to account for new source code.
49 /// Line terminators are the classic "\n", "\r", "\r\n" (which counts as
50 /// a single terminator), and JSON LINE/PARAGRAPH SEPARATORs.
51 ///
52 /// See <https://tc39.es/ecma262/multipage/ecmascript-language-lexical-grammar.html#sec-line-terminators>
53 pub fn update(&mut self, code: &[u8]) {
54 // JS source text is interpreted as UCS-2, which is basically UTF-16 with less
55 // restrictions. We cannot iterate UTF-8 bytes here, 2-byte UTF-8 octets
56 // should count as a 1 char and not 2.
57 let &mut SourcePos {
58 mut line,
59 mut column,
60 } = self;
61
62 let mut i = 0;
63 while i < code.len() {
64 // This is not a UTF-8 validator, but it's likely close enough. It's assumed
65 // that the input is valid (and if it isn't than what are you doing trying to
66 // embed it into source code anyways?). The important part is that we process in
67 // order, and use the first octet's bit pattern to decode the octet length of
68 // the char.
69 match code[i] {
70 U8_LF => {
71 i += 1;
72 line += 1;
73 column = 0;
74 }
75 U8_CR => {
76 // Count "\r\n" as a single terminator.
77 if code.get(i + 1) == Some(&U8_LF) {
78 i += 2;
79 } else {
80 i += 1;
81 }
82 line += 1;
83 column = 0;
84 }
85
86 // 1 octet chars do not have the high bit set. If it's not a LF or CR, then it's
87 // just a regular ASCII.
88 b if b & 0b10000000 == 0 => {
89 i += 1;
90 column += 1;
91 }
92
93 // 2 octet chars have a leading `110` bit pattern. None are considered line
94 // terminators.
95 b if b & 0b11100000 == 0b11000000 => {
96 // eat this byte and the next.
97 i += 2;
98 column += 1;
99 }
100
101 // 3 octet chars have a leading `1110` bit pattern. Both the LINE/PARAGRAPH
102 // SEPARATOR exist in 3 octets.
103 b if b & 0b11110000 == 0b11100000 => {
104 // The LINE and PARAGRAPH have the bits `11100010 10000000 1010100X`, with the X
105 // denoting either line or paragraph.
106 let mut separator = false;
107 if b == 0b11100010 && code.get(i + 1) == Some(&0b10000000) {
108 let last = code.get(i + 2).cloned().unwrap_or_default();
109 separator = (last & 0b11111110) == 0b10101000
110 }
111
112 // eat this byte and the next 2.
113 i += 3;
114 if separator {
115 line += 1;
116 column = 0;
117 } else {
118 column += 1;
119 }
120 }
121
122 // 4 octet chars have a leading `11110` pattern, but we don't need to check because
123 // none of the other patterns matched.
124 _ => {
125 // eat this byte and the next 3.
126 i += 4;
127 column += 1;
128 }
129 }
130 }
131 self.line = line;
132 self.column = column;
133 }
134}
135
136impl std::cmp::PartialEq<(u32, u32)> for SourcePos {
137 fn eq(&self, other: &(u32, u32)) -> bool {
138 &(self.line, self.column) == other
139 }
140}