Skip to main content

turbopack_core/
source_pos.rs

1use bincode::{Decode, Encode};
2use serde::Serialize;
3use turbo_tasks_hash::DeterministicHash;
4
5/// LINE FEED (LF), one of the basic JS line terminators.
6const U8_LF: u8 = 0x0A;
7/// CARRIAGE RETURN (CR), one of the basic JS line terminators.
8const U8_CR: u8 = 0x0D;
9
10#[turbo_tasks::task_input]
11#[derive(
12    Default,
13    Debug,
14    PartialEq,
15    Eq,
16    Copy,
17    Clone,
18    Hash,
19    PartialOrd,
20    Ord,
21    Serialize,
22    DeterministicHash,
23    Encode,
24    Decode,
25)]
26pub struct SourcePos {
27    /// The line, 0-indexed.
28    pub line: u32,
29    /// The byte index of the column, 0-indexed.
30    pub column: u32,
31}
32
33impl SourcePos {
34    pub fn new(start_line: u32) -> Self {
35        Self {
36            line: start_line,
37            column: 0,
38        }
39    }
40
41    pub fn max() -> Self {
42        Self {
43            line: u32::MAX,
44            column: u32::MAX,
45        }
46    }
47
48    /// Increments the line/column position to account for new source code.
49    /// Line terminators are the classic "\n", "\r", "\r\n" (which counts as
50    /// a single terminator), and JSON LINE/PARAGRAPH SEPARATORs.
51    ///
52    /// See <https://tc39.es/ecma262/multipage/ecmascript-language-lexical-grammar.html#sec-line-terminators>
53    pub fn update(&mut self, code: &[u8]) {
54        // JS source text is interpreted as UCS-2, which is basically UTF-16 with less
55        // restrictions. We cannot iterate UTF-8 bytes here, 2-byte UTF-8 octets
56        // should count as a 1 char and not 2.
57        let &mut SourcePos {
58            mut line,
59            mut column,
60        } = self;
61
62        let mut i = 0;
63        while i < code.len() {
64            // This is not a UTF-8 validator, but it's likely close enough. It's assumed
65            // that the input is valid (and if it isn't than what are you doing trying to
66            // embed it into source code anyways?). The important part is that we process in
67            // order, and use the first octet's bit pattern to decode the octet length of
68            // the char.
69            match code[i] {
70                U8_LF => {
71                    i += 1;
72                    line += 1;
73                    column = 0;
74                }
75                U8_CR => {
76                    // Count "\r\n" as a single terminator.
77                    if code.get(i + 1) == Some(&U8_LF) {
78                        i += 2;
79                    } else {
80                        i += 1;
81                    }
82                    line += 1;
83                    column = 0;
84                }
85
86                // 1 octet chars do not have the high bit set. If it's not a LF or CR, then it's
87                // just a regular ASCII.
88                b if b & 0b10000000 == 0 => {
89                    i += 1;
90                    column += 1;
91                }
92
93                // 2 octet chars have a leading `110` bit pattern. None are considered line
94                // terminators.
95                b if b & 0b11100000 == 0b11000000 => {
96                    // eat this byte and the next.
97                    i += 2;
98                    column += 1;
99                }
100
101                // 3 octet chars have a leading `1110` bit pattern. Both the LINE/PARAGRAPH
102                // SEPARATOR exist in 3 octets.
103                b if b & 0b11110000 == 0b11100000 => {
104                    // The LINE and PARAGRAPH have the bits `11100010 10000000 1010100X`, with the X
105                    // denoting either line or paragraph.
106                    let mut separator = false;
107                    if b == 0b11100010 && code.get(i + 1) == Some(&0b10000000) {
108                        let last = code.get(i + 2).cloned().unwrap_or_default();
109                        separator = (last & 0b11111110) == 0b10101000
110                    }
111
112                    // eat this byte and the next 2.
113                    i += 3;
114                    if separator {
115                        line += 1;
116                        column = 0;
117                    } else {
118                        column += 1;
119                    }
120                }
121
122                // 4 octet chars have a leading `11110` pattern, but we don't need to check because
123                // none of the other patterns matched.
124                _ => {
125                    // eat this byte and the next 3.
126                    i += 4;
127                    column += 1;
128                }
129            }
130        }
131        self.line = line;
132        self.column = column;
133    }
134}
135
136impl std::cmp::PartialEq<(u32, u32)> for SourcePos {
137    fn eq(&self, other: &(u32, u32)) -> bool {
138        &(self.line, self.column) == other
139    }
140}