Repository navigation
Expand file tree
/
Copy pathtextutil.rs
More file actions
240 lines (224 loc) · 8.01 KB
/
Copy pathtextutil.rs
File metadata and controls
240 lines (224 loc) · 8.01 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
//! Shared utilities for the TS/JS walker: compiled regexes, UTF-16 position
//! conversion, generated-file detection, and small text helpers — each
//! mirroring a specific helper in src/extraction/tree-sitter.ts (noted inline).
use regex::Regex;
use std::sync::OnceLock;
macro_rules! re {
($name:ident, $pat:expr) => {
pub fn $name() -> &'static Regex {
static RE: OnceLock<Regex> = OnceLock::new();
RE.get_or_init(|| Regex::new($pat).expect(concat!("regex ", stringify!($name))))
}
};
}
// RTK_HOOK_NAME_RE (tree-sitter.ts)
re!(rtk_hook_name, r"^use[A-Z][A-Za-z0-9]*(?:Query|Mutation)$");
// reactComponentHoc's styled test
re!(styled_callee, r"^styled\b");
// PascalCase component gate (#841)
re!(pascal_case, r"^[A-Z]");
// extractCall parenthesized-conversion normalization
re!(paren_conversion, r"^\(\s*\*?\s*([A-Za-z_][\w.]*)\s*\)$");
// flushFnRefCandidates SIMPLE_NAME
re!(simple_name, r"^[A-Za-z_$][A-Za-z0-9_$]*$");
// flushFnRefCandidates QUALIFIED_IMPORT
re!(qualified_import, r"^[A-Za-z_$][A-Za-z0-9_$.\\]*[.\\]([A-Za-z_$][A-Za-z0-9_$]*)$");
// captureFnRefCandidates rhs param-storage skip — trailing identifier of LHS
re!(lhs_last_name, r"([A-Za-z_$][A-Za-z0-9_$]*)\s*$");
// extractTsTupleContractNames identifier test
re!(ident_dollar, r"^[A-Za-z_$][A-Za-z0-9_$]*$");
// looksLikeVueStoreFile signal (VUE_STORE_FILE_SIGNAL)
re!(
vue_store_signal,
r"\bdefineStore\b|\bcreateStore\b|\bVuex\b|\bmutations\b|\bactions\b|\bgetters\b|\bnamespaced\b"
);
// value-ref target-name distinctiveness: /[A-Z_]/
re!(has_upper_or_underscore, r"[A-Z_]");
/// isGeneratedFile (src/extraction/generated-detection.ts) — full pattern list
/// ported so future language walkers share it.
pub fn is_generated_file(file_path: &str) -> bool {
static RES: OnceLock<Vec<Regex>> = OnceLock::new();
let patterns = RES.get_or_init(|| {
[
r"\.pb\.go$",
r"\.pulsar\.go$",
r"_grpc\.pb\.go$",
r"_mock\.go$",
r"_mocks\.go$",
r"^mock_[^/]+\.go$",
r"\.generated\.[jt]sx?$",
r"\.gen\.[jt]sx?$",
r"\.pb\.[jt]s$",
r"_pb\.[jt]s$",
r"_grpc_pb\.[jt]s$",
r"[.-]min\.m?js$",
r"_pb2(_grpc)?\.py$",
r"_pb2\.pyi$",
r"\.pb\.(cc|h)$",
r"\.g\.cs$",
r"Grpc\.cs$",
r"OuterClass\.java$",
r"Grpc\.java$",
r"\.pb\.swift$",
r"\.g\.dart$",
r"\.freezed\.dart$",
r"\.pb\.dart$",
r"\.pbgrpc\.dart$",
r"\.chopper\.dart$",
r"\.generated\.rs$",
]
.iter()
.map(|p| Regex::new(p).expect("generated pattern"))
.collect()
});
patterns.iter().any(|p| p.is_match(file_path))
}
// isMinifiedContent's WEBPACK_RUNTIME (`\b` ASCII, as in JS).
re!(webpack_runtime, r"(?-u:\b)function __webpack_require__\s*\(");
/// isMinifiedContent (src/extraction/generated-detection.ts): a `.js` / `.mjs`
/// / `.cjs` file of at least 4000 UTF-16 units whose text sits mostly on lines
/// of 1000 units or more, dense with `;{}(),` — or that carries a webpack
/// runtime. Lengths are counted in UTF-16 units, as the JS string ops count
/// them, so both extractors agree on every file.
pub fn is_minified_content(file_path: &str, content: &str) -> bool {
const LINE: usize = 1000;
let lower = file_path.to_ascii_lowercase();
if !(lower.ends_with(".js") || lower.ends_with(".mjs") || lower.ends_with(".cjs")) {
return false;
}
// UTF-16 units never outnumber UTF-8 bytes: a short file is out at once.
if content.len() < 4 * LINE {
return false;
}
let total = content.encode_utf16().count();
if total < 4 * LINE {
return false;
}
if webpack_runtime().is_match(content) {
return true;
}
let (mut long, mut punctuation) = (0usize, 0usize);
let (mut line_len, mut line_punct) = (0usize, 0usize);
for unit in content.encode_utf16() {
if unit == 10 {
if line_len >= LINE {
long += line_len;
punctuation += line_punct;
}
line_len = 0;
line_punct = 0;
continue;
}
line_len += 1;
// ; { } ( ) ,
if matches!(unit, 59 | 123 | 125 | 40 | 41 | 44) {
line_punct += 1;
}
}
if line_len >= LINE {
long += line_len;
punctuation += line_punct;
}
long as f64 >= total as f64 * 0.5 && punctuation as f64 >= long as f64 * 0.03
}
/// Byte offsets of each line start, for UTF-16 column conversion.
pub fn line_starts(src: &str) -> Vec<usize> {
let mut out = vec![0usize];
for (i, b) in src.bytes().enumerate() {
if b == b'\n' {
out.push(i + 1);
}
}
out
}
/// UTF-16 code units in `s` — what web-tree-sitter (and JS string ops)
/// count, so kernel-emitted columns are byte-identical to the wasm path's.
pub fn utf16_len(s: &str) -> usize {
s.chars().map(|c| c.len_utf16()).sum()
}
/// Column (UTF-16 units) of `byte_pos` on line `row`, given `line_starts`.
pub fn col16(src: &str, starts: &[usize], row: usize, byte_pos: usize) -> u32 {
let ls = starts.get(row).copied().unwrap_or(0);
if byte_pos <= ls {
return 0;
}
utf16_len(&src[ls..byte_pos]) as u32
}
/// JS `String.prototype.slice(0, n)` in UTF-16 units, without splitting a
/// surrogate pair (when the cut would split one, we stop one code unit short —
/// a lone surrogate isn't representable in Rust and never round-trips through
/// SQLite anyway). Returns (sliced, was_truncated_at_or_beyond_n).
pub fn slice_utf16(s: &str, n: usize) -> (String, bool) {
let mut used = 0usize;
let mut out = String::new();
for c in s.chars() {
let w = c.len_utf16();
if used + w > n {
return (out, true);
}
used += w;
out.push(c);
if used == n {
// Exactly at the limit: truncated iff any source remains.
let truncated = out.len() < s.len();
return (out, truncated);
}
}
(out, false)
}
/// objectKeyName (tree-sitter.ts): strip ONE leading and ONE trailing quote
/// character (`'`, `"`, or backtick).
pub fn object_key_name(s: &str) -> String {
let mut out = s;
if let Some(first) = out.chars().next() {
if first == '\'' || first == '"' || first == '`' {
out = &out[first.len_utf8()..];
}
}
if let Some(last) = out.chars().last() {
if last == '\'' || last == '"' || last == '`' {
out = &out[..out.len() - last.len_utf8()];
}
}
out.to_string()
}
/// The `= <first 100 UTF-16 units>[...]` initializer signature used by
/// extractVariable (its `.length >= 100` check fires exactly when the slice
/// hit the cap).
pub fn init_signature(value_text: &str) -> String {
let (sliced, _) = slice_utf16(value_text, 100);
if utf16_len(&sliced) >= 100 {
format!("= {sliced}...")
} else {
format!("= {sliced}")
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn utf16_cols() {
let src = "aé😀b";
// 'a'=1, 'é'=1, '😀'=2 utf16 units; bytes: a=1, é=2, 😀=4
assert_eq!(utf16_len(src), 5);
let starts = line_starts(src);
assert_eq!(col16(src, &starts, 0, 1), 1); // after 'a'
assert_eq!(col16(src, &starts, 0, 3), 2); // after 'é'
assert_eq!(col16(src, &starts, 0, 7), 4); // after '😀'
}
#[test]
fn init_sig_short_and_long() {
assert_eq!(init_signature("[1, 2]"), "= [1, 2]");
let long = "x".repeat(150);
let sig = init_signature(&long);
assert!(sig.starts_with("= "));
assert!(sig.ends_with("..."));
assert_eq!(utf16_len(&sig[2..sig.len() - 3]), 100);
}
#[test]
fn generated_patterns() {
assert!(is_generated_file("src/api.generated.ts"));
assert!(is_generated_file("vendor/jquery.min.js"));
assert!(!is_generated_file("src/app.ts"));
}
}