Repository navigation
Expand file tree
/
Copy pathutils.ts
More file actions
471 lines (433 loc) · 17.3 KB
/
Copy pathutils.ts
File metadata and controls
471 lines (433 loc) · 17.3 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
// Shared types, regex constants, and utility functions used across lint rule modules.
// Nothing in this file should emit findings — it only parses and extracts.
import { Parser } from "htmlparser2";
export type OpenTag = {
raw: string;
name: string;
attrs: string;
index: number;
closeIndex?: number;
endIndex?: number;
};
export type ExtractedBlock = {
attrs: string;
content: string;
raw: string;
index: number;
};
const COMPOSITION_ID_IN_CSS_PATTERN = /\[data-composition-id=["']([^"']+)["']\]/g;
export const TIMELINE_REGISTRY_INIT_PATTERN =
/window\.__timelines\s*=\s*window\.__timelines\s*\|\|\s*\{\}|window\.__timelines\s*=\s*\{\}|window\.__timelines\s*\?\?=\s*\{\}/i;
// Object-literal registration that assigns at least one `key: value` entry inline,
// e.g. `window.__timelines = { main: tl }` or `window.__timelines = { "comp-1": tl }`.
// Distinct from the empty-init form (`= {}`) — requires a key followed by `:`.
export const TIMELINE_REGISTRY_OBJECT_LITERAL_PATTERN =
/window\.__timelines\s*=\s*\{\s*(?:["'][^"']+["']|[A-Za-z_$][\w$]*)\s*:/i;
export const TIMELINE_REGISTRY_ASSIGN_PATTERN =
/window\.__timelines(?:\[[^\]]+\]|\.[A-Za-z_$][\w$]*)\s*=/i;
// The bracket branch accepts either a quoted string key (`["root"]`) or a
// computed key (`[spec.id]`, `[id]`) — a bare-identifier-only bracket branch
// missed `window.__timelines[spec.id] = tl`, a pattern the shipped
// code-particle-assemble/code-3d-extrude registry blocks actually use,
// making gsap_timeline_not_registered false-fire on correctly registered
// timelines. The computed-key alternative is deliberately non-capturing:
// its text isn't a literal composition id, so callers reading group 1/2
// (readRegisteredTimelineCompositionId) must keep falling back to null for it.
export const WINDOW_TIMELINE_ASSIGN_PATTERN =
/window\.__timelines(?:\[\s*(?:["']([^"']+)["']|[A-Za-z_$][\w$.]*)\s*\]|\.\s*([A-Za-z_$][\w$]*))\s*=\s*([A-Za-z_$][\w$]*)/i;
export const INVALID_SCRIPT_CLOSE_PATTERN = /<script[^>]*>[\s\S]*?<\s*\/\s*script(?!>)/i;
const TIMELINE_REGISTRY_KEY_PATTERN =
/window\.__timelines(?:\[\s*["']([^"']+)["']\s*\]|\.\s*([A-Za-z_$][\w$]*))\s*=/g;
// The `window.__timelines = { ... }` object-literal body (group 1), captured so its
// `key: value` entries can be scanned for registered keys.
// Locates the START of a `window.__timelines = { ... }` literal. Deliberately does
// not try to match the closing brace: see readTimelineRegistryObjectBody, which walks
// braces instead. A regex cannot tell the registry's own `}` from the `}` of an
// inlined options object.
const TIMELINE_REGISTRY_OBJECT_OPEN_PATTERN = /window\.__timelines\s*=\s*\{/i;
// A single object-literal entry whose value is an identifier (real timeline registration),
// e.g. `main: tl` or `"comp-1": tl`. Captures the key in group 1 (quoted) or 2 (bare).
const TIMELINE_REGISTRY_OBJECT_ENTRY_PATTERN =
/(?:["']([^"']+)["']|([A-Za-z_$][\w$]*))\s*:\s*[A-Za-z_$][\w$]*/g;
export function parseHtmlStructure(source: string): {
tags: OpenTag[];
scripts: ExtractedBlock[];
styles: ExtractedBlock[];
} {
const tags: OpenTag[] = [];
const blocks = { script: [] as ExtractedBlock[], style: [] as ExtractedBlock[] };
const openTagsByName = new Map<string, OpenTag[]>();
const openBlocks: Array<{
name: "script" | "style";
attrs: string;
contentStart: number;
index: number;
}> = [];
const parser: Parser = new Parser(
{
onopentag(name) {
const index = parser.startIndex;
const raw = source.slice(index, parser.endIndex + 1);
const attrs = raw.slice(name.length + 1, -1).replace(/\s*\/$/, "");
const tag = { raw, name, attrs, index };
tags.push(tag);
const sameNameStack = openTagsByName.get(name) ?? [];
sameNameStack.push(tag);
openTagsByName.set(name, sameNameStack);
if (name === "script" || name === "style") {
openBlocks.push({ name, attrs, contentStart: parser.endIndex + 1, index });
}
},
onclosetag(name) {
const tag = openTagsByName.get(name)?.pop();
if (tag) {
tag.closeIndex = parser.startIndex;
tag.endIndex = parser.endIndex + 1;
}
if (name !== "script" && name !== "style") return;
const block = openBlocks.pop();
if (!block || block.name !== name) return;
blocks[name].push({
attrs: block.attrs,
content: source.slice(block.contentStart, parser.startIndex),
raw: source.slice(block.index, parser.endIndex + 1),
index: block.index,
});
},
},
{ decodeEntities: false, lowerCaseAttributeNames: false, lowerCaseTags: true },
);
parser.end(source);
return { tags, scripts: blocks.script, styles: blocks.style };
}
/**
* Find the `<html>` open tag in the source. Distinct from `findRootTag`,
* which returns the first element inside `<body>` — the latter is "the
* composition's visible root", whereas `<html>` is where document-level
* metadata like `data-composition-variables` lives.
*/
export function findHtmlTag(tags: readonly OpenTag[]): OpenTag | null {
return tags.find((tag) => tag.name === "html") ?? null;
}
// fallow-ignore-next-line complexity
export function findRootTag(source: string, parsedTags?: readonly OpenTag[]): OpenTag | null {
const tags = parsedTags ?? parseHtmlStructure(source).tags;
const bodyTag = tags.find((tag) => tag.name === "body");
if (
bodyTag &&
(readDecodedAttr(bodyTag.raw, "data-composition-id") ||
readAttr(bodyTag.raw, "data-width") ||
readAttr(bodyTag.raw, "data-height"))
) {
return bodyTag;
}
const bodyStart = bodyTag ? bodyTag.index + bodyTag.raw.length : 0;
const bodyEnd = bodyTag?.closeIndex ?? source.length;
const bodyTags = tags.filter((tag) => tag.index >= bodyStart && tag.index < bodyEnd);
// Set when a leading <svg> defs block is skipped (see below) — extractOpenTags
// is a flat, nesting-unaware scan, so without this the very next tag it
// returns is the svg's own nested child (<defs>, <filter>, ...), not the
// sibling that follows the closed </svg>.
let skipBefore = -1;
for (const tag of bodyTags) {
if (tag.index < skipBefore) continue;
if (["script", "style", "meta", "link", "title"].includes(tag.name)) continue;
// A leading <svg> block (icon/gradient/filter <defs>, referenced by url(#id)
// from elsewhere in the document) is shared visual plumbing, not the
// composition root — two independent reports of this being mistaken for
// the root, manufacturing root_missing_composition_id/root_missing_dimensions
// on an otherwise-correct composition. Only skip it when it carries none of
// the composition markers itself, so an intentionally SVG-rooted composition
// (data-composition-id/data-width/data-height directly on the <svg>) is
// still eligible as the root.
if (
tag.name === "svg" &&
!readDecodedAttr(tag.raw, "data-composition-id") &&
!readAttr(tag.raw, "data-width") &&
!readAttr(tag.raw, "data-height")
) {
// No closing tag found (malformed HTML) — skip everything rather than
// risk returning one of the svg's own children as the root.
skipBefore = tag.endIndex ?? Infinity;
continue;
}
return tag;
}
return null;
}
export function readAttr(tagSource: string, attr: string): string | null {
if (!tagSource) return null;
const escaped = attr.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
// `(?<![\w-])` not `\b`: a plain `\b` boundary treats the hyphen in a longer
// attribute as a word break, so reading "id" would wrongly match the trailing
// `id="…"` inside `data-hf-id="…"` (and "width" inside `data-width`, etc.).
// The lookbehind requires the match to start a fresh attribute name.
const match = tagSource.match(new RegExp(`(?<![\\w-])${escaped}\\s*=\\s*["']([^"']+)["']`, "i"));
return match?.[1] || null;
}
/** Read an HTML attribute using browser-equivalent character-reference decoding. */
export function readDecodedAttr(tagSource: string, attr: string): string | null {
if (!tagSource) return null;
let value: string | null = null;
const parser = new Parser(
{
onattribute(name, decodedValue) {
if (value === null && name.toLowerCase() === attr.toLowerCase()) value = decodedValue;
},
},
{ decodeEntities: true, lowerCaseAttributeNames: false, lowerCaseTags: true },
);
parser.end(tagSource);
return value;
}
/**
* Read a JSON-bearing attribute with browser-equivalent character-reference
* decoding. Imported or formatter-serialized HTML commonly stores JSON quotes
* as `"`; lint must inspect the same decoded value that `getAttribute()`
* exposes at runtime.
*/
export function readJsonAttr(tagSource: string, attr: string): string | null {
return readDecodedAttr(tagSource, attr);
}
export function collectCompositionIds(tags: OpenTag[]): Set<string> {
const ids = new Set<string>();
for (const tag of tags) {
const compId = readDecodedAttr(tag.raw, "data-composition-id");
if (compId) ids.add(compId);
}
return ids;
}
export function extractCompositionIdsFromCss(css: string): string[] {
const ids = new Set<string>();
let match: RegExpExecArray | null;
const pattern = new RegExp(
COMPOSITION_ID_IN_CSS_PATTERN.source,
COMPOSITION_ID_IN_CSS_PATTERN.flags,
);
while ((match = pattern.exec(css)) !== null) {
if (match[1]) ids.add(match[1]);
}
return [...ids];
}
export function extractTimelineRegistryKeys(source: string): string[] {
const keys = new Set<string>();
let match: RegExpExecArray | null;
const pattern = new RegExp(
TIMELINE_REGISTRY_KEY_PATTERN.source,
TIMELINE_REGISTRY_KEY_PATTERN.flags,
);
while ((match = pattern.exec(source)) !== null) {
const key = match[1] ?? match[2];
if (key) keys.add(key);
}
for (const entry of readTimelineRegistryTopLevelKeys(source)) keys.add(entry);
return [...keys];
}
/**
* Top-level keys of a `window.__timelines = { ... }` literal.
*
* Walks brace depth rather than regex-matching the body. The previous non-greedy
* body match stopped at the first `}` it saw, which for the legal one-liner
*
* window.__timelines = { main: gsap.timeline({ paused: true }) };
*
* was the brace of the INLINED OPTIONS OBJECT. The entry scanner then harvested
* `paused` as a composition id and timeline_id_mismatch reported a timeline
* "registered as paused" — a registration that does not exist, so its fixHint
* could never be applied. Hoisting the timeline to a variable was the only escape,
* and nothing said so.
*/
/** Index of the brace that closes the group opened just before `bodyStart`. */
function findMatchingBrace(source: string, bodyStart: number): number {
let depth = 1;
for (let i = bodyStart; i < source.length; i += 1) {
if (source[i] === "{") depth += 1;
else if (source[i] === "}" && (depth -= 1) === 0) return i;
}
return source.length;
}
/** Replace every nested brace group with spaces so only depth-0 text remains. */
function blankNestedBraceGroups(body: string): string {
let out = "";
let depth = 0;
for (const ch of body) {
if (ch === "{") depth += 1;
else if (ch === "}") depth = Math.max(0, depth - 1);
else if (depth === 0) {
out += ch;
continue;
}
out += " ";
}
return out;
}
function readTimelineRegistryTopLevelKeys(source: string): string[] {
const open = TIMELINE_REGISTRY_OBJECT_OPEN_PATTERN.exec(source);
if (!open) return [];
const bodyStart = open.index + open[0].length;
const body = source.slice(bodyStart, findMatchingBrace(source, bodyStart));
const flattened = blankNestedBraceGroups(body);
const keys: string[] = [];
const entryPattern = new RegExp(
TIMELINE_REGISTRY_OBJECT_ENTRY_PATTERN.source,
TIMELINE_REGISTRY_OBJECT_ENTRY_PATTERN.flags,
);
let entry: RegExpExecArray | null;
while ((entry = entryPattern.exec(flattened)) !== null) {
const key = entry[1] ?? entry[2];
if (key) keys.push(key);
}
return keys;
}
export function getInlineScriptSyntaxError(source: string): string | null {
if (!source.trim()) return null;
try {
// eslint-disable-next-line no-new-func
new Function(source);
return null;
} catch (error) {
if (error instanceof Error) return error.message;
return String(error);
}
}
// fallow-ignore-next-line complexity
/**
* Blank the contents of every `'...'` and `"..."` literal, keeping the quotes so
* the source stays the same shape.
*
* Needed because a composition that *displays* source code carries things like
* `Math.random()` inside a string it never executes. Scanning raw script text for
* non-determinism reported those compositions as non-deterministic, and no edit
* could clear it while keeping the displayed snippet intact.
*
* Template literals are deliberately left alone: `${Math.random()}` inside one IS
* executed, and blanking it would hide real non-determinism. A snippet stored in a
* backtick string therefore still reports — a narrower gap than the one this closes.
*/
export function stripStringLiterals(source: string): string {
return source.replace(
/(['"])(?:\\.|(?!\1)[^\\\n])*\1?/g,
(literal) => literal[0] + " ".repeat(Math.max(0, literal.length - 1)),
);
}
// fallow-ignore-next-line complexity
export function stripJsComments(source: string): string {
let out = "";
let i = 0;
let quote: "'" | '"' | "`" | null = null;
let escaped = false;
while (i < source.length) {
const ch = source[i] ?? "";
const next = source[i + 1] ?? "";
if (quote) {
out += ch;
if (escaped) {
escaped = false;
} else if (ch === "\\") {
escaped = true;
} else if (ch === quote) {
quote = null;
}
i += 1;
continue;
}
if (ch === "'" || ch === '"' || ch === "`") {
quote = ch;
out += ch;
i += 1;
continue;
}
if (ch === "/" && next === "/") {
out += " ";
i += 2;
while (i < source.length && source[i] !== "\n" && source[i] !== "\r") {
out += " ";
i += 1;
}
continue;
}
if (ch === "/" && next === "*") {
out += " ";
i += 2;
while (i < source.length) {
const blockCh = source[i] ?? "";
const blockNext = source[i + 1] ?? "";
if (blockCh === "*" && blockNext === "/") {
out += " ";
i += 2;
break;
}
out += blockCh === "\n" || blockCh === "\r" ? blockCh : " ";
i += 1;
}
continue;
}
out += ch;
i += 1;
}
return out;
}
// One linear pass that drops every `<!-- … -->` region. Uses indexOf, not a
// `/<!--[\s\S]*?-->/` regex: that pattern backtracks O(n²) on inputs with many
// unterminated "<!--" (CodeQL js/polynomial-redos). An unterminated "<!--" with
// no closing "-->" is kept verbatim, matching the prior regex's no-match behavior.
function stripHtmlCommentsOnce(source: string): string {
let out = "";
let i = 0;
for (;;) {
const start = source.indexOf("<!--", i);
if (start < 0) return out + source.slice(i);
const end = source.indexOf("-->", start + 4);
if (end < 0) return out + source.slice(i);
out += source.slice(i, start);
i = end + 3;
}
}
// Strip HTML comments to a fixpoint. A single pass is not enough: deleting one
// comment can splice adjacent markers into a fresh, complete <!-- … --> (e.g.
// "<<!-- -->!-- … -->" → "<!-- … -->"), which would otherwise survive and let a
// commented-out <template>/tag hijack the linter's tag scan.
export function stripHtmlComments(source: string): string {
let out = source;
for (let prev = ""; prev !== out; ) {
prev = out;
out = stripHtmlCommentsOnce(out);
}
return out;
}
export function extractScriptTextsAndSrcs(scripts: ExtractedBlock[]): {
texts: string[];
srcs: string[];
} {
const texts = scripts.filter((s) => !/\bsrc\s*=/.test(s.attrs)).map((s) => s.content);
const srcs = scripts.map((s) => readAttr(`<script ${s.attrs}>`, "src") || "").filter(Boolean);
return { texts, srcs };
}
export function isMediaTag(tagName: string): boolean {
return tagName === "video" || tagName === "audio" || tagName === "img";
}
// Whether any <style> block in the composition defines caption group/word
// classes (`.caption-group`, `.caption_word`, etc.) — the signal several
// caption-specific rules use to skip non-caption compositions entirely.
export function hasCaptionStyles(styles: ExtractedBlock[]): boolean {
return styles.some((s) => /\.caption[-_]?(?:group|word)/i.test(s.content));
}
export function truncateSnippet(value: string, maxLength = 220): string | undefined {
const normalized = value.replace(/\s+/g, " ").trim();
if (!normalized) return undefined;
if (normalized.length <= maxLength) return normalized;
return `${normalized.slice(0, maxLength - 3)}...`;
}
/**
* Matches a media tag carrying a real `src` attribute, capturing the tag name in
* group 1 and the src value in group 2.
*
* The leading whitespace before `src` is load-bearing: `\bsrc\s*=` also matches
* the tail of `data-var-src="bg"` (a hyphen/`s` boundary is a word boundary), and
* since `[^>]*` is greedy it wins over a real `src` earlier in the same tag. Every
* element using a variable binding was therefore reported as referencing a missing
* file named after the variable id.
*/
export function mediaSrcTagRe(tagAlternation: string): RegExp {
return new RegExp(`<(${tagAlternation})\\b[^>]*\\ssrc\\s*=\\s*["']([^"']+)["'][^>]*>`, "gi");
}