mirror of
https://github.com/THU-MAIC/OpenMAIC.git
synced 2026-10-02 09:24:43 +08:00
fix: tolerate malformed authored CSS in classroom exports (#1422)
* fix: tolerate malformed authored CSS in classroom exports
Interactive-scene HTML authored by LLMs can carry browser-tolerated CSS
syntax errors (e.g. `-- crop-green: #22c55e`), which browsers silently
drop. The strict postcss pass used by resource-pack / classroom-zip
export threw on them, failing the whole export while PPTX (which never
parses interactive HTML) kept working.
Inlining is an enhancement, not a precondition: on strict-parse failure
the stylesheet/style attribute/SVG attribute is now kept byte-for-byte
with a warning, and export continues. Remote dependencies inside
unparseable CSS (url() / @import, absolute or relative, quoted with
parens) are still surfaced via a comment- and string-aware fallback
scanner so zip exporters warn and offline validation is not silently
bypassed.
* fix: stop the css fallback scanner at bad-string tokens
Per CSS syntax, an unescaped \n/\r/\f inside a quoted string produces a
bad-string token and the browser recovers right there. The fallback
scanner previously ran unterminated strings to EOF, swallowing every
following rule and hiding browser-active url()/@import dependencies from
dependency reporting and residual validation.
Both string-scanning loops now terminate at an unescaped newline and
resume scanning from it; an escaped newline (\ + \n, \ + \r\n, \ + \r,
\ + \f) continues the string, and escaped newlines are stripped from
captured URLs — in the fallback scanner and in the strict-path
collectors (cssUrlReferences / cssImportReference), which previously
kept the escape in the URL for valid CSS too.
* fix: keep EOF-truncated css strings as valid tokens
Per CSS Syntax 4.3.5, only an unescaped \n/\r/\f produces a bad-string
token; a quoted string truncated at EOF is a valid string token that
browsers keep (with the function auto-closed and a trailing backslash
consumed). The previous round wrongly dropped EOF-truncated url(".../
@import "... values, hiding real remote dependencies again — the exact
shape of an LLM-authored <style> cut off mid-generation.
Both string paths now share scanCssString with three-way termination
(quote / unescaped newline / EOF), and the collectors were aligned with
CSS token semantics on both the strict and fallback paths, with parity
tests: escaped newlines (\ + LF/CRLF/CR/FF) are stripped escape-aware
from quoted values; an unquoted url() containing a backslash-newline is
a bad-url and no dependency; non-whitespace (comments allowed) around a
quoted payload makes a bad-url, and a bad url consumes through its ')'
so nested url(...) text is not re-scanned; an escaped ')' belongs to an
unquoted url value.
---------
Co-authored-by: wyuc <wang-yc24@mails.tsinghua.edu.cn>
This commit is contained in:
+241
-12
@@ -7,16 +7,87 @@ export interface CssUrlReference {
|
||||
end: number;
|
||||
}
|
||||
|
||||
/** CSS drops an escaped newline from a token value: url("a\<LF>b") is "ab". */
|
||||
const ESCAPED_NEWLINE_RE = /\\\r\n|\\\n|\\\r|\\\f/;
|
||||
/**
|
||||
* Escape-aware removal of escaped newlines from a quoted-string value: `\` +
|
||||
* newline (LF / CRLF / CR / FF) is dropped; other escapes are kept verbatim so
|
||||
* `\\` + newline leaves a raw newline behind (a bad-string token) for the
|
||||
* caller to reject.
|
||||
*/
|
||||
function stripEscapedNewlines(value: string): string {
|
||||
let out = '';
|
||||
for (let i = 0; i < value.length; i++) {
|
||||
const c = value[i];
|
||||
if (c !== '\\') {
|
||||
out += c;
|
||||
continue;
|
||||
}
|
||||
const next = value[i + 1];
|
||||
if (next === undefined) break;
|
||||
if (next === '\n' || next === '\r' || next === '\f') {
|
||||
i += next === '\r' && value[i + 2] === '\n' ? 2 : 1;
|
||||
continue;
|
||||
}
|
||||
out += c + next;
|
||||
i++;
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
type CssStringEnd = { kind: 'quote' | 'newline' | 'eof'; end: number };
|
||||
|
||||
/**
|
||||
* Scan a CSS quoted string starting after its opening quote. Per CSS Syntax
|
||||
* §4.3.5, an unescaped \n/\r/\f yields a bad-string token (browsers recover at
|
||||
* that character), while EOF yields a valid string token that browsers keep;
|
||||
* a backslash at EOF is consumed.
|
||||
*/
|
||||
function scanCssString(css: string, start: number, quote: string): CssStringEnd {
|
||||
const n = css.length;
|
||||
let j = start;
|
||||
while (j < n) {
|
||||
if (css[j] === '\\') {
|
||||
// An escaped newline continues the string (\r\n counts as one).
|
||||
if (j + 1 >= n) return { kind: 'eof', end: j };
|
||||
j += css[j + 1] === '\r' && css[j + 2] === '\n' ? 3 : 2;
|
||||
continue;
|
||||
}
|
||||
if (css[j] === quote) return { kind: 'quote', end: j };
|
||||
if (css[j] === '\n' || css[j] === '\r' || css[j] === '\f') return { kind: 'newline', end: j };
|
||||
j++;
|
||||
}
|
||||
return { kind: 'eof', end: n };
|
||||
}
|
||||
|
||||
export function cssUrlReferences(value: string): CssUrlReference[] {
|
||||
const refs: CssUrlReference[] = [];
|
||||
valueParser(value).walk((node: CssValueNode) => {
|
||||
if (node.type !== 'function' || node.value.toLowerCase() !== 'url') return;
|
||||
const inner = node.nodes?.[0];
|
||||
if (inner && inner.type === 'string') {
|
||||
// Non-whitespace (comments are fine) after the closing quote makes a
|
||||
// bad-url token.
|
||||
if (node.nodes.slice(1).some((part) => part.type !== 'space' && part.type !== 'comment'))
|
||||
return false;
|
||||
// Quoted url() (including an EOF-truncated one, which CSS treats as a
|
||||
// valid string token — stringify() would drop the closing quote).
|
||||
let content = (inner as { unclosed?: boolean; value: string }).value;
|
||||
if ((inner as { unclosed?: boolean }).unclosed && content.endsWith('\\'))
|
||||
content = content.slice(0, -1);
|
||||
content = stripEscapedNewlines(content);
|
||||
// A raw (unescaped) newline inside the string is a bad-string token;
|
||||
// browsers drop it, so it is no dependency — same as the fallback scanner.
|
||||
if (/[\n\r\f]/.test(content)) return false;
|
||||
refs.push({ raw: content, start: node.sourceIndex, end: node.sourceEndIndex });
|
||||
return false;
|
||||
}
|
||||
const raw = valueParser.stringify(node.nodes).trim();
|
||||
const unquoted =
|
||||
(raw.startsWith('"') && raw.endsWith('"')) || (raw.startsWith("'") && raw.endsWith("'"))
|
||||
? raw.slice(1, -1)
|
||||
: raw;
|
||||
refs.push({ raw: unquoted, start: node.sourceIndex, end: node.sourceEndIndex });
|
||||
if (!raw) return false;
|
||||
// In an unquoted url() a backslash newline is a bad-url token instead,
|
||||
// and a bad url is not a dependency: browsers drop the declaration.
|
||||
if (ESCAPED_NEWLINE_RE.test(raw)) return false;
|
||||
refs.push({ raw, start: node.sourceIndex, end: node.sourceEndIndex });
|
||||
return false;
|
||||
});
|
||||
return refs;
|
||||
@@ -43,15 +114,38 @@ export function cssImportReference(rule: AtRule): { url: string; conditions: str
|
||||
);
|
||||
if (!node) return null;
|
||||
let url: string | null = null;
|
||||
if (node.type === 'string') url = node.value;
|
||||
let quoted = false;
|
||||
let unclosed = false;
|
||||
if (node.type === 'string') {
|
||||
url = (node as { unclosed?: boolean; value: string }).value;
|
||||
quoted = true;
|
||||
unclosed = (node as { unclosed?: boolean }).unclosed === true;
|
||||
}
|
||||
if (node.type === 'function' && node.value.toLowerCase() === 'url') {
|
||||
const raw = valueParser.stringify(node.nodes).trim();
|
||||
url =
|
||||
(raw.startsWith('"') && raw.endsWith('"')) || (raw.startsWith("'") && raw.endsWith("'"))
|
||||
? raw.slice(1, -1)
|
||||
: raw;
|
||||
const inner = (node.nodes ?? [])[0];
|
||||
if (inner && inner.type === 'string') {
|
||||
// Non-whitespace (comments are fine) after the closing quote makes a
|
||||
// bad-url token.
|
||||
if (
|
||||
(node.nodes ?? []).slice(1).some((part) => part.type !== 'space' && part.type !== 'comment')
|
||||
)
|
||||
return null;
|
||||
url = (inner as { unclosed?: boolean; value: string }).value;
|
||||
quoted = true;
|
||||
unclosed = (inner as { unclosed?: boolean }).unclosed === true;
|
||||
} else {
|
||||
const raw = valueParser.stringify(node.nodes).trim();
|
||||
if (!raw || ESCAPED_NEWLINE_RE.test(raw)) return null;
|
||||
url = raw;
|
||||
}
|
||||
}
|
||||
if (!url) return null;
|
||||
if (quoted) {
|
||||
if (unclosed && url.endsWith('\\')) url = url.slice(0, -1);
|
||||
url = stripEscapedNewlines(url);
|
||||
// A raw newline inside a quoted string is a bad-string token.
|
||||
if (/[\n\r\f]/.test(url)) return null;
|
||||
}
|
||||
return { url, conditions: rule.params.slice(node.sourceEndIndex).trim() };
|
||||
}
|
||||
|
||||
@@ -65,11 +159,146 @@ export function parseCss(css: string, cssUrl: string): Root {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Best-effort fallback for `collectCssAssetReferences` when strict parsing
|
||||
* fails. A tiny token scanner (not a regex pile) that is comment- and
|
||||
* string-aware, so textual `content: "url(...)"`, commented-out imports and
|
||||
* lookalike function names (`myurl(`) are not mistaken for dependencies, while
|
||||
* real remote refs in malformed CSS stay visible to residual validation (e.g.
|
||||
* the video export's offline-completeness check).
|
||||
*/
|
||||
export function collectCssAssetReferencesByRegex(
|
||||
css: string,
|
||||
): Array<{ kind: 'css-url' | 'css-import'; url: string }> {
|
||||
const refs: Array<{ kind: 'css-url' | 'css-import'; url: string }> = [];
|
||||
const n = css.length;
|
||||
let i = 0;
|
||||
let importNext = false;
|
||||
const isWordChar = (c: string) => /[\w-]/.test(c);
|
||||
while (i < n) {
|
||||
const ch = css[i];
|
||||
if (ch === '/' && css[i + 1] === '*') {
|
||||
// Unterminated comments run to end-of-stylesheet, like browsers.
|
||||
// Comments count as whitespace: `@import /* c */ "x.css"` stays intact.
|
||||
const end = css.indexOf('*/', i + 2);
|
||||
i = end === -1 ? n : end + 2;
|
||||
continue;
|
||||
}
|
||||
if (ch === '"' || ch === "'") {
|
||||
const scan = scanCssString(css, i + 1, ch);
|
||||
// EOF yields a valid string token (browsers keep the value); only an
|
||||
// unescaped newline is a bad string whose value must not be taken.
|
||||
if (scan.kind !== 'newline' && importNext)
|
||||
refs.push({ kind: 'css-import', url: stripEscapedNewlines(css.slice(i + 1, scan.end)) });
|
||||
importNext = false;
|
||||
i = scan.kind === 'quote' ? scan.end + 1 : scan.end;
|
||||
continue;
|
||||
}
|
||||
if (ch === '@') {
|
||||
i++;
|
||||
continue;
|
||||
}
|
||||
if (!isWordChar(ch)) {
|
||||
if (!/\s/.test(ch)) importNext = false;
|
||||
i++;
|
||||
continue;
|
||||
}
|
||||
let j = i;
|
||||
while (j < n && isWordChar(css[j])) j++;
|
||||
const word = css.slice(i, j).toLowerCase();
|
||||
if (word === 'import') {
|
||||
importNext = true;
|
||||
i = j;
|
||||
continue;
|
||||
}
|
||||
if (word === 'url' && css[j] === '(') {
|
||||
let k = j + 1;
|
||||
// Whitespace and comments may precede the url payload.
|
||||
for (;;) {
|
||||
if (css[k] === '/' && css[k + 1] === '*') {
|
||||
const endc = css.indexOf('*/', k + 2);
|
||||
k = endc === -1 ? n : endc + 2;
|
||||
continue;
|
||||
}
|
||||
if (/\s/.test(css[k] ?? '')) {
|
||||
k++;
|
||||
continue;
|
||||
}
|
||||
break;
|
||||
}
|
||||
const quote = css[k] === '"' || css[k] === "'" ? css[k] : null;
|
||||
if (quote) {
|
||||
const scan = scanCssString(css, k + 1, quote);
|
||||
// Non-whitespace between the closing quote and ')' is a bad-url token.
|
||||
let after = scan.kind === 'quote' ? scan.end + 1 : scan.end;
|
||||
// Whitespace and comments may sit between the closing quote and ')'.
|
||||
for (;;) {
|
||||
const cch = css[after];
|
||||
if (cch === '/' && css[after + 1] === '*') {
|
||||
const endc = css.indexOf('*/', after + 2);
|
||||
after = endc === -1 ? n : endc + 2;
|
||||
continue;
|
||||
}
|
||||
if (/\s/.test(cch ?? '')) {
|
||||
after++;
|
||||
continue;
|
||||
}
|
||||
break;
|
||||
}
|
||||
// EOF auto-closes the function; a newline is a bad string; otherwise
|
||||
// only whitespace may sit between the closing quote and ')'.
|
||||
const closedCleanly =
|
||||
scan.kind === 'eof' || (scan.kind === 'quote' && (after >= n || css[after] === ')'));
|
||||
if (closedCleanly) {
|
||||
refs.push({
|
||||
kind: importNext ? 'css-import' : 'css-url',
|
||||
url: stripEscapedNewlines(css.slice(k + 1, scan.end)),
|
||||
});
|
||||
}
|
||||
// A bad url consumes everything up to its ')' like the browser does,
|
||||
// so nested url(...) text inside it is not re-scanned.
|
||||
let close = scan.kind === 'quote' ? scan.end + 1 : scan.end;
|
||||
while (close < n && css[close] !== ')') close++;
|
||||
i = close + 1;
|
||||
} else {
|
||||
let m = k;
|
||||
while (m < n) {
|
||||
if (css[m] === '\\') {
|
||||
m += 2;
|
||||
continue;
|
||||
}
|
||||
if (css[m] === ')') break;
|
||||
m++;
|
||||
}
|
||||
// In an unquoted url() a backslash newline is a bad-url token
|
||||
// (§4.3.6): browsers drop the declaration, so it is no dependency.
|
||||
const raw = css.slice(k, m).trim();
|
||||
if (raw && !ESCAPED_NEWLINE_RE.test(raw))
|
||||
refs.push({ kind: importNext ? 'css-import' : 'css-url', url: raw });
|
||||
i = m + 1;
|
||||
}
|
||||
importNext = false;
|
||||
continue;
|
||||
}
|
||||
importNext = false;
|
||||
i = j;
|
||||
}
|
||||
return refs;
|
||||
}
|
||||
|
||||
export function collectCssAssetReferences(
|
||||
css: string,
|
||||
context: 'stylesheet' | 'declaration-list' = 'stylesheet',
|
||||
): Array<{ kind: 'css-url' | 'css-import'; url: string }> {
|
||||
const root = parseCss(context === 'stylesheet' ? css : `.x{${css}}`, 'inline-css');
|
||||
// Authored CSS (e.g. LLM-generated interactive scenes) can carry browser-
|
||||
// tolerated syntax errors like `-- name: value`. Strict parsing here would
|
||||
// abort the whole export, so collection is best-effort instead.
|
||||
let root: Root;
|
||||
try {
|
||||
root = parseCss(context === 'stylesheet' ? css : `.x{${css}}`, 'inline-css');
|
||||
} catch {
|
||||
return collectCssAssetReferencesByRegex(css);
|
||||
}
|
||||
const refs: Array<{ kind: 'css-url' | 'css-import'; url: string }> = [];
|
||||
root.walkAtRules((rule) => {
|
||||
if (rule.name.toLowerCase() !== 'import') return;
|
||||
|
||||
@@ -21,11 +21,25 @@ import {
|
||||
import parseSrcset, { type SrcsetCandidate } from 'parse-srcset';
|
||||
import type { AtRule, Root } from 'postcss';
|
||||
import {
|
||||
collectCssAssetReferencesByRegex,
|
||||
cssImportReference,
|
||||
cssUrlReferences,
|
||||
parseCss,
|
||||
rewriteCssValue,
|
||||
} from './css-asset-parser';
|
||||
import { createLogger } from '@/lib/logger';
|
||||
|
||||
const log = createLogger('InlineAssets');
|
||||
|
||||
/** Best-effort parse for enhancement paths; null means "leave the CSS as-is". */
|
||||
function tryParseCss(css: string, cssUrl: string): Root | null {
|
||||
try {
|
||||
return parseCss(css, cssUrl);
|
||||
} catch (error) {
|
||||
log.warn(`CSS parse failed for ${cssUrl}; leaving value untouched`, error);
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
export { toDataUri } from './inline-assets-shared';
|
||||
export type { InlineReport, InlineOptions, FetchAsset } from './inline-assets-shared';
|
||||
@@ -262,7 +276,26 @@ export async function inlineCssUrls(
|
||||
): Promise<{ css: string; failed: { url: string; reason: string }[]; inlined: string[] }> {
|
||||
const failed: { url: string; reason: string }[] = [];
|
||||
const inlined: string[] = [];
|
||||
const root = parseCss(css, cssUrl);
|
||||
// Authored CSS can carry browser-tolerated syntax errors (e.g. `-- name`).
|
||||
// Inlining is an enhancement, not a precondition: on a parse failure keep the
|
||||
// stylesheet byte-for-byte and continue instead of failing the whole export.
|
||||
let root: Root;
|
||||
try {
|
||||
root = parseCss(css, cssUrl);
|
||||
} catch (error) {
|
||||
log.warn(`CSS parse failed while inlining ${cssUrl}; leaving stylesheet untouched`, error);
|
||||
// Zip exporters only consume `failed`, so remote refs inside the
|
||||
// unparseable stylesheet must be surfaced here, not silently packaged.
|
||||
for (const ref of collectCssAssetReferencesByRegex(css)) {
|
||||
if (/^(?:data:|blob:|about:|#)/i.test(ref.url)) continue;
|
||||
try {
|
||||
failed.push({ url: new URL(ref.url, cssUrl).href, reason: 'css parse failed' });
|
||||
} catch {
|
||||
// Unresolvable values stay untouched, as elsewhere.
|
||||
}
|
||||
}
|
||||
return { css, failed, inlined };
|
||||
}
|
||||
const imports: AtRule[] = [];
|
||||
root.walkAtRules((rule) => {
|
||||
if (rule.name.toLowerCase() === 'import') imports.push(rule);
|
||||
@@ -295,10 +328,19 @@ export async function inlineCssUrls(
|
||||
fetchAsset,
|
||||
new Set(activeCssUrls).add(abs),
|
||||
);
|
||||
inlined.push(abs, ...nested.inlined);
|
||||
failed.push(...nested.failed);
|
||||
const wrapped = wrapImportedCss(nested.css, parseCssImportConditions(reference.conditions));
|
||||
rule.replaceWith(...parseCss(wrapped, abs).nodes);
|
||||
// nested.css is raw whenever the imported stylesheet failed strict parsing;
|
||||
// re-parsing it here would rethrow. Keep the original @import rule then. The
|
||||
// remote dependency stays in the output, so report it as a failure instead
|
||||
// of claiming the import (and its nested assets) were inlined.
|
||||
const wrappedRoot = tryParseCss(wrapped, abs);
|
||||
if (wrappedRoot) {
|
||||
rule.replaceWith(...wrappedRoot.nodes);
|
||||
inlined.push(abs, ...nested.inlined);
|
||||
} else {
|
||||
failed.push({ url: abs, reason: 'css parse failed' });
|
||||
}
|
||||
}
|
||||
|
||||
// 1. Find @font-face blocks; build dropRefs (non-woff2 fonts in blocks that have a woff2).
|
||||
@@ -323,11 +365,19 @@ export async function inlineCssUrls(
|
||||
return { css: root.toString(), failed, inlined };
|
||||
}
|
||||
|
||||
/** Absolute http(s) refs inside unparseable CSS; surfaced as export failures. */
|
||||
function unresolvedCssRefs(css: string): { url: string; reason: string }[] {
|
||||
return collectCssAssetReferencesByRegex(css)
|
||||
.filter((ref) => /^https?:\/\//i.test(ref.url))
|
||||
.map((ref) => ({ url: ref.url, reason: 'css parse failed' }));
|
||||
}
|
||||
|
||||
async function inlineStyleAttributeUrls(
|
||||
css: string,
|
||||
fetchAsset: FetchAsset,
|
||||
): Promise<{ css: string; failed: { url: string; reason: string }[]; inlined: string[] }> {
|
||||
const root = parseCss(`.openmaic-style{${css}}`, 'style-attribute');
|
||||
const root = tryParseCss(`.openmaic-style{${css}}`, 'style-attribute');
|
||||
if (!root) return { css, failed: unresolvedCssRefs(css), inlined: [] };
|
||||
const result = await inlineParsedCssUrls(root, 'about:blank', fetchAsset, new Set());
|
||||
const serialized = root.toString();
|
||||
return {
|
||||
@@ -341,10 +391,11 @@ async function inlineSvgPresentationAttributeUrls(
|
||||
cssValue: string,
|
||||
fetchAsset: FetchAsset,
|
||||
): Promise<{ cssValue: string; failed: { url: string; reason: string }[]; inlined: string[] }> {
|
||||
const root = parseCss(
|
||||
const root = tryParseCss(
|
||||
`.openmaic-svg{${attributeName}:${cssValue}}`,
|
||||
'svg-presentation-attribute',
|
||||
);
|
||||
if (!root) return { cssValue, failed: unresolvedCssRefs(cssValue), inlined: [] };
|
||||
let declarationValue = cssValue;
|
||||
const result = await inlineParsedCssUrls(root, 'about:blank', fetchAsset, new Set());
|
||||
root.walkDecls((declaration) => {
|
||||
|
||||
@@ -894,3 +894,285 @@ describe('inlineHtmlAssets', () => {
|
||||
expect(out).toContain('src="data:text/javascript;base64,');
|
||||
});
|
||||
});
|
||||
|
||||
describe('malformed authored CSS tolerance', () => {
|
||||
// Regression for the stage-LHSa3TDAYh export failures: an LLM-authored
|
||||
// <style> carried `-- crop-green: #22c55e;` (space after --). Browsers drop
|
||||
// such declarations, but strict postcss parsing used to abort the whole
|
||||
// resource-pack / classroom-zip export.
|
||||
const malformedCss = `:root {\n -- crop-green: #22c55e;\n}\nbody { color: red; }`;
|
||||
|
||||
it('collectAssetRefs does not throw on an unparsable <style> block', () => {
|
||||
expect(() => collectAssetRefs(`<style>${malformedCss}</style>`)).not.toThrow();
|
||||
});
|
||||
|
||||
it('inlineHtmlAssets keeps a malformed <style> block byte-for-byte and still succeeds', async () => {
|
||||
const { html: out, report } = await inlineHtmlAssets(
|
||||
`<style>${malformedCss}</style><img src="https://cdn.test/ok.png">`,
|
||||
{
|
||||
fetcher: async () => ({
|
||||
bytes: new Uint8Array([1]),
|
||||
contentType: 'image/png',
|
||||
}),
|
||||
},
|
||||
);
|
||||
expect(out).toContain(`-- crop-green: #22c55e;`);
|
||||
expect(out).toContain('<img src="data:image/png;base64,');
|
||||
expect(report.failed).toHaveLength(0);
|
||||
});
|
||||
|
||||
it('inlineCssUrls returns the stylesheet untouched when it cannot parse', async () => {
|
||||
const fetcher = async () => null as unknown as Awaited<ReturnType<typeof fetch>>;
|
||||
const { css, failed, inlined } = await inlineCssUrls(
|
||||
malformedCss,
|
||||
'about:blank',
|
||||
fetcher as never,
|
||||
);
|
||||
expect(css).toBe(malformedCss);
|
||||
expect(failed).toHaveLength(0);
|
||||
expect(inlined).toHaveLength(0);
|
||||
});
|
||||
|
||||
it('collectCssAssetReferences still surfaces url() refs in malformed CSS', async () => {
|
||||
const { collectCssAssetReferences } = await import('@/lib/export/css-asset-parser');
|
||||
expect(
|
||||
collectCssAssetReferences(
|
||||
':root { -- crop-green: #22c55e; } .a { background: url(https://x/a.png); }',
|
||||
),
|
||||
).toEqual([{ kind: 'css-url', url: 'https://x/a.png' }]);
|
||||
});
|
||||
|
||||
it('ignores url() text inside quoted strings but keeps parenthesized quoted urls', async () => {
|
||||
const { collectCssAssetReferencesByRegex } = await import('@/lib/export/css-asset-parser');
|
||||
expect(
|
||||
collectCssAssetReferencesByRegex(
|
||||
'.a { content: "url(https://x/text-only.png)"; } .b { src: url("https://x/font(foo).woff2"); }',
|
||||
),
|
||||
).toEqual([{ kind: 'css-url', url: 'https://x/font(foo).woff2' }]);
|
||||
});
|
||||
|
||||
it('reports a malformed @import url() only once', async () => {
|
||||
const { collectCssAssetReferencesByRegex } = await import('@/lib/export/css-asset-parser');
|
||||
expect(collectCssAssetReferencesByRegex('@import url(https://x/a.css);')).toEqual([
|
||||
{ kind: 'css-import', url: 'https://x/a.css' },
|
||||
]);
|
||||
});
|
||||
|
||||
it('keeps collecting urls after a malformed unterminated @import', async () => {
|
||||
const { collectCssAssetReferencesByRegex } = await import('@/lib/export/css-asset-parser');
|
||||
expect(
|
||||
collectCssAssetReferencesByRegex(
|
||||
'@import url(https://x/a.css) .a { background: url(https://x/b.png); }',
|
||||
),
|
||||
).toEqual([
|
||||
{ kind: 'css-import', url: 'https://x/a.css' },
|
||||
{ kind: 'css-url', url: 'https://x/b.png' },
|
||||
]);
|
||||
});
|
||||
|
||||
it('ignores unterminated comments and non-url function names', async () => {
|
||||
const { collectCssAssetReferencesByRegex } = await import('@/lib/export/css-asset-parser');
|
||||
expect(
|
||||
collectCssAssetReferencesByRegex(
|
||||
'.a { background: myurl(https://x/fake.png); } /* url(https://x/gone.png)',
|
||||
),
|
||||
).toEqual([]);
|
||||
});
|
||||
|
||||
it('keeps an @import target across an interleaved comment', async () => {
|
||||
const { collectCssAssetReferencesByRegex } = await import('@/lib/export/css-asset-parser');
|
||||
expect(collectCssAssetReferencesByRegex('@import /* x */ "https://x/a.css";')).toEqual([
|
||||
{ kind: 'css-import', url: 'https://x/a.css' },
|
||||
]);
|
||||
});
|
||||
|
||||
it('recovers at a bad-string token and still collects later urls', async () => {
|
||||
const { collectCssAssetReferencesByRegex } = await import('@/lib/export/css-asset-parser');
|
||||
// An unescaped newline ends the string; the browser applies .ok below.
|
||||
expect(
|
||||
collectCssAssetReferencesByRegex(
|
||||
'.bad { content: "oops\n;}\n.ok { background: url(https://x/ok.png) }',
|
||||
),
|
||||
).toEqual([{ kind: 'css-url', url: 'https://x/ok.png' }]);
|
||||
// Same for a quoted url( whose string is unterminated: the url function
|
||||
// consumes everything up to the first ')', so .ok inside it is gone too.
|
||||
expect(
|
||||
collectCssAssetReferencesByRegex(
|
||||
'.bad { src: url("oops\n;}\n.ok { background: url(https://x/ok.png) }',
|
||||
),
|
||||
).toEqual([]);
|
||||
// With the ')' close by, scanning recovers and the later url is collected.
|
||||
expect(
|
||||
collectCssAssetReferencesByRegex(
|
||||
'.bad { src: url("oops\n) }\n.ok { background: url(https://x/ok.png) }',
|
||||
),
|
||||
).toEqual([{ kind: 'css-url', url: 'https://x/ok.png' }]);
|
||||
});
|
||||
|
||||
it('handles escaped and unescaped newlines per CSS string semantics', async () => {
|
||||
const { collectCssAssetReferencesByRegex } = await import('@/lib/export/css-asset-parser');
|
||||
const tail = '.ok { background: url(https://x/ok.png) }';
|
||||
// Escaped \n / \r\n continue the string; the url inside is taken.
|
||||
expect(
|
||||
collectCssAssetReferencesByRegex(`.a { content: "esc\\
|
||||
line" }
|
||||
.ok2 { src: url(https://x/a.png) }`),
|
||||
).toContainEqual({ kind: 'css-url', url: 'https://x/a.png' });
|
||||
// Backslash + CR + LF is one escaped newline in CSS: string continues.
|
||||
expect(
|
||||
collectCssAssetReferencesByRegex('.a { content: "crlf\\\r\nend" }\n' + tail),
|
||||
).toContainEqual({ kind: 'css-url', url: 'https://x/ok.png' });
|
||||
// Unescaped \r and \f are bad-string terminators like \n.
|
||||
expect(
|
||||
collectCssAssetReferencesByRegex(`.bad { content: "oops
|
||||
}
|
||||
${tail}`),
|
||||
).toEqual([{ kind: 'css-url', url: 'https://x/ok.png' }]);
|
||||
expect(
|
||||
collectCssAssetReferencesByRegex(`.bad { content: "oops}
|
||||
${tail}`),
|
||||
).toEqual([{ kind: 'css-url', url: 'https://x/ok.png' }]);
|
||||
// Unterminated at EOF swallows nothing real afterwards (there is none).
|
||||
expect(collectCssAssetReferencesByRegex('.bad { content: "oops')).toEqual([]);
|
||||
});
|
||||
|
||||
it('keeps strings truncated at EOF, like browsers (CSS Syntax 4.3.5)', async () => {
|
||||
const { collectCssAssetReferencesByRegex } = await import('@/lib/export/css-asset-parser');
|
||||
// EOF yields a valid string token: the dependency stays reportable.
|
||||
expect(collectCssAssetReferencesByRegex('.a { background: url("https://x/a.png')).toEqual([
|
||||
{ kind: 'css-url', url: 'https://x/a.png' },
|
||||
]);
|
||||
expect(collectCssAssetReferencesByRegex('@import "https://x/a.css')).toEqual([
|
||||
{ kind: 'css-import', url: 'https://x/a.css' },
|
||||
]);
|
||||
// A backslash at EOF is consumed: url("a\ at EOF is the string "a".
|
||||
expect(collectCssAssetReferencesByRegex('.a { background: url("a\\')).toEqual([
|
||||
{ kind: 'css-url', url: 'a' },
|
||||
]);
|
||||
// The strict-path collectors agree on EOF-truncated quoted urls.
|
||||
const { cssUrlReferences } = await import('@/lib/export/css-asset-parser');
|
||||
expect(cssUrlReferences('url("https://x/a.png')).toEqual([
|
||||
{ raw: 'https://x/a.png', start: 0, end: 21 },
|
||||
]);
|
||||
expect(cssUrlReferences('url("a\\')).toEqual([{ raw: 'a', start: 0, end: 8 }]);
|
||||
});
|
||||
|
||||
it('continues a quoted url() string across escaped newlines', async () => {
|
||||
const { collectCssAssetReferencesByRegex } = await import('@/lib/export/css-asset-parser');
|
||||
// Escaped LF inside a double-quoted url(): CSS drops the escape, so the
|
||||
// dependency URL is the concatenation across the line break.
|
||||
expect(collectCssAssetReferencesByRegex('.a { src: url("https://x/a\\\nb.png") }')).toEqual([
|
||||
{ kind: 'css-url', url: 'https://x/ab.png' },
|
||||
]);
|
||||
// Escaped CRLF inside a single-quoted url(): skip-3 continues the string.
|
||||
expect(collectCssAssetReferencesByRegex(".a { src: url('https://x/c\\\r\nd.png') }")).toEqual([
|
||||
{ kind: 'css-url', url: 'https://x/cd.png' },
|
||||
]);
|
||||
// Unescaped newline inside url("...") is a bad string: no URL taken,
|
||||
// scanning resumes and the later dependency is still collected.
|
||||
expect(
|
||||
collectCssAssetReferencesByRegex(
|
||||
'.bad { src: url("oops\n) }\n.ok { src: url(https://x/ok.png) }',
|
||||
),
|
||||
).toEqual([{ kind: 'css-url', url: 'https://x/ok.png' }]);
|
||||
// Unescaped \r / \f inside a quoted url(...) are bad strings too.
|
||||
const tail = '.ok { src: url(https://x/ok.png) }';
|
||||
expect(collectCssAssetReferencesByRegex('.bad { src: url("oops\r) }\n' + tail)).toEqual([
|
||||
{ kind: 'css-url', url: 'https://x/ok.png' },
|
||||
]);
|
||||
expect(collectCssAssetReferencesByRegex('.bad { src: url("oops\f) }\n' + tail)).toEqual([
|
||||
{ kind: 'css-url', url: 'https://x/ok.png' },
|
||||
]);
|
||||
});
|
||||
|
||||
it('strict and fallback paths agree on newlines in url tokens', async () => {
|
||||
const { cssUrlReferences, collectCssAssetReferencesByRegex } =
|
||||
await import('@/lib/export/css-asset-parser');
|
||||
// Quoted with escaped newline: escape dropped, same value both paths.
|
||||
const cssQuoted = 'url("https://x/a\\\nb.png")';
|
||||
expect(cssUrlReferences(cssQuoted)).toEqual([{ raw: 'https://x/ab.png', start: 0, end: 25 }]);
|
||||
expect(collectCssAssetReferencesByRegex(`.a { src: ${cssQuoted} }`)).toEqual([
|
||||
{ kind: 'css-url', url: 'https://x/ab.png' },
|
||||
]);
|
||||
// Multiple escaped newlines in one quoted url: all escapes dropped.
|
||||
expect(cssUrlReferences('url("https://x/a\\\nb\\\nc.png")')).toEqual([
|
||||
{ raw: 'https://x/abc.png', start: 0, end: 28 },
|
||||
]);
|
||||
// Unquoted with backslash-newline: bad-url token, no dependency either path.
|
||||
const cssUnquoted = 'url(https://x/a\\\nb.png)';
|
||||
expect(cssUrlReferences(cssUnquoted)).toEqual([]);
|
||||
expect(collectCssAssetReferencesByRegex(`.a { src: ${cssUnquoted} }`)).toEqual([]);
|
||||
// Quoted string containing a raw newline: bad-string, dropped both paths.
|
||||
const cssRawNewline = 'url("https://x/a\nb.png")';
|
||||
expect(cssUrlReferences(cssRawNewline)).toEqual([]);
|
||||
expect(collectCssAssetReferencesByRegex(`.a { src: ${cssRawNewline} }`)).toEqual([]);
|
||||
// Non-whitespace after the closing quote is a bad-url token: no
|
||||
// dependency on either path.
|
||||
expect(cssUrlReferences('url("https://x/a.png" foo)')).toEqual([]);
|
||||
expect(collectCssAssetReferencesByRegex('.a { src: url("https://x/a.png" foo) }')).toEqual([]);
|
||||
// Comments before the quoted payload are fine in the fallback scanner.
|
||||
// (The strict path's value-parser lumps `/*c*/ "..."` into one word node;
|
||||
// a pre-existing limitation of that parser, out of scope here.)
|
||||
expect(collectCssAssetReferencesByRegex('.a { src: url(/*c*/ "https://x/a.png") }')).toEqual([
|
||||
{ kind: 'css-url', url: 'https://x/a.png' },
|
||||
]);
|
||||
// An escaped ')' inside an unquoted url is part of the URL.
|
||||
expect(collectCssAssetReferencesByRegex('.a { src: url(https://x/a\\)b.png) }')).toEqual([
|
||||
{ kind: 'css-url', url: 'https://x/a' + String.fromCharCode(92) + ')b.png' },
|
||||
]);
|
||||
// Comments between the closing quote and ')' are fine, not a bad url.
|
||||
expect(cssUrlReferences('url("https://x/a.png" /*c*/)')).toEqual([
|
||||
{ raw: 'https://x/a.png', start: 0, end: 28 },
|
||||
]);
|
||||
expect(collectCssAssetReferencesByRegex('.a { src: url("https://x/a.png" /*c*/) }')).toEqual([
|
||||
{ kind: 'css-url', url: 'https://x/a.png' },
|
||||
]);
|
||||
// A bad url consumes through its ')': a nested url inside it is not a
|
||||
// dependency on either path.
|
||||
expect(cssUrlReferences('url("x" url(https://x/nested.png))')).toEqual([]);
|
||||
expect(
|
||||
collectCssAssetReferencesByRegex('.a { src: url("x" url(https://x/nested.png)) }'),
|
||||
).toEqual([]);
|
||||
// An escaped backslash before the newline leaves it unescaped: still a
|
||||
// bad-string token, dropped on both paths.
|
||||
const cssEscapedBackslash = 'url("https://x/a\\\\\nb.png")';
|
||||
expect(cssUrlReferences(cssEscapedBackslash)).toEqual([]);
|
||||
expect(collectCssAssetReferencesByRegex(`.a { src: ${cssEscapedBackslash} }`)).toEqual([]);
|
||||
});
|
||||
|
||||
it('surfaces absolute and relative url() refs of unparseable CSS as failures', async () => {
|
||||
const { css, failed } = await inlineCssUrls(
|
||||
':root { -- crop-green: #22c55e; } .a { background: url(https://x/a.png) ; } .b { background: url(imgs/b.png); } /* url(https://x/comment.png) */',
|
||||
'https://cdn.test/page.html',
|
||||
(async () => null) as never,
|
||||
);
|
||||
expect(css).toContain('-- crop-green');
|
||||
expect(failed).toContainEqual({ url: 'https://x/a.png', reason: 'css parse failed' });
|
||||
expect(failed).toContainEqual({
|
||||
url: 'https://cdn.test/imgs/b.png',
|
||||
reason: 'css parse failed',
|
||||
});
|
||||
expect(failed).not.toContainEqual({
|
||||
url: 'https://x/comment.png',
|
||||
reason: 'css parse failed',
|
||||
});
|
||||
});
|
||||
|
||||
it('does not rethrow when an @import pulls malformed nested CSS', async () => {
|
||||
const imported = ':root { -- crop-green: #22c55e; }';
|
||||
const { css, inlined, failed } = await inlineCssUrls(
|
||||
'@import url(https://cdn.test/bad.css); body { color: red; }',
|
||||
'https://cdn.test/a.css',
|
||||
(async (url: string) =>
|
||||
url === 'https://cdn.test/bad.css'
|
||||
? { bytes: new TextEncoder().encode(imported), contentType: 'text/css' }
|
||||
: null) as never,
|
||||
);
|
||||
expect(css).toContain('@import url(https://cdn.test/bad.css)');
|
||||
expect(inlined).not.toContain('https://cdn.test/bad.css');
|
||||
expect(failed).toContainEqual({
|
||||
url: 'https://cdn.test/bad.css',
|
||||
reason: 'css parse failed',
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user