From 9fd8288d4c1d51e241185926def4e1ac4f7505fa Mon Sep 17 00:00:00 2001 From: Kristofer Baxter Date: Mon, 31 Aug 2026 14:33:47 -0700 Subject: [PATCH 1/2] Recover from unterminated at-rule preludes and function calls An unclosed parenthesis in an at-rule prelude or a function call leaks its scope to the end of the stylesheet. `@media (min-width: 40em{` scopes every remaining line of the file as `meta.at-rule.media.header.css`, so the rest of the stylesheet stops being highlighted as CSS. A single missing character while typing takes the whole file with it. The `end` pattern of each affected region gains one alternative: bail out at a `{`. In these regions a brace cannot be part of the construct, so the test is a single lookahead that reads one character. The closing parenthesis moves to capture group 1 so that the bail-out, which consumes nothing, does not claim the punctuation scope. Applied at 16 sites: `@supports` conditions, media features, `@document` argument functions, `layer()` in `@import`, `calc()`, the gradient, shape, timing-function, transform and misc value functions, `url()`, the colour functions, and the functional pseudo-classes. The `@media` condition is the one place a brace can be legal, because `` is `( ? )` and `` admits a balanced curly block. There the bail is narrowed to a `{` that is not closed before the next brace, so `@media (a: {b}) {` still parses while the unterminated form still recovers. The scan it needs is bounded by the distance to that brace rather than by the length of the line. Two regions are deliberately left alone. A `var()` fallback and a custom function argument are declaration values, and `` admits a balanced curly block that legally spans lines, as in `--x: --foo({ ... })`. The legal and the malformed forms are indistinguishable within one line, so recovering there would cost legal CSS. Both are commented in place. A string in an `@media` or `@supports` condition is now parsed by a condition-local copy of the string rule rather than the shared one. Without it, a legal brace inside a general-enclosed string, `@media (a: "x{")`, opens the body early. The copy is local because the shared newline-escape rule is used by every string in CSS; changing it altered tokenization in 12 of 15 measured contexts. Verification: the 217 existing tests pass unchanged, and every character of 1.02M of Bootstrap, Bulma and normalize.css keeps the scope stack it has on `main`. A broken-input matrix covers all ten regions that can hold an unclosed parenthesis. The guard costs nothing on well-formed CSS: `@media (min-width: 40em) {` followed by 5000 rules takes 260ms here and 277ms on `main`, which is within the noise between runs. On a file whose prelude is unterminated the cost does change, because this grammar highlights the rest of the file where `main` does not: 251ms against 26ms. That is the price of the fix. It is linear in the size of the file, and it is the same work that highlighting a well-formed file of that size already costs. --- grammars/css.cson | 219 ++++++++++++++---- spec/css-spec.mjs | 508 ++++++++++++++++++++++++++++++++++++++++++ testing-util/test.mjs | 16 ++ 3 files changed, 703 insertions(+), 40 deletions(-) diff --git a/grammars/css.cson b/grammars/css.cson index 039f54b..ef627c1 100644 --- a/grammars/css.cson +++ b/grammars/css.cson @@ -711,9 +711,9 @@ 'name': 'support.function.layer.css' '2': 'name': 'punctuation.section.function.begin.bracket.round.css' - 'end': '\\)' + 'end': '(\\))|(?=\\{)' 'endCaptures': - '0': + '1': 'name': 'punctuation.section.function.end.bracket.round.css' 'name': 'meta.function.layer.css' 'patterns': [ @@ -853,6 +853,69 @@ 'name': 'constant.character.escape.css' } ] + # A string inside an at-rule condition. Identical to `#string` except + # that it does not enter the multi-line newline-escape region: a + # condition region's `end` cannot be tested while that region is + # active, so a legal `"a{\` continuation would hold the prelude open + # for the rest of the stylesheet. Without it, a trailing backslash + # simply leaves the string open to its real closing quote. + 'condition-string': + 'patterns': [ + { + 'begin': '"' + 'beginCaptures': + '0': + 'name': 'punctuation.definition.string.begin.css' + 'end': '"|(?, which may + # legally contain a balanced curly block, so they keep the unguarded + # ending and stay scoped exactly as they are today. + { + 'begin': '(?i)(?` admits a balanced curly + # block that legally spans lines, as in `--x: --foo({ ... });`. The legal + # `--foo({` and the malformed `--foo(a{` are indistinguishable within a + # single line, so recovering here would cost legal CSS. + 'end': '(\\))' 'endCaptures': - '0': + '1': 'name': 'punctuation.section.function.end.bracket.round.css' 'name': 'meta.function.custom.css' 'patterns': [ @@ -1188,9 +1291,12 @@ 'name': 'support.function.misc.css' '2': 'name': 'punctuation.section.function.begin.bracket.round.css' - 'end': '\\)' + # No `{` bail-out here: a `var()` fallback may legally contain a + # balanced curly block spanning lines (`var(--x, { color: red })`), + # and `main` already recovers an unclosed `var(` no better than this. + 'end': '(\\))' 'endCaptures': - '0': + '1': 'name': 'punctuation.section.function.end.bracket.round.css' 'name': 'meta.function.variable.css' 'patterns': [ @@ -1215,9 +1321,9 @@ 'beginCaptures': '0': 'name': 'punctuation.definition.function.begin.bracket.round.css' - 'end': '\\)' + 'end': '(\\))|(?=\\{)' 'endCaptures': - '0': + '1': 'name': 'punctuation.definition.function.end.bracket.round.css' 'patterns': [ { @@ -1242,9 +1348,9 @@ 'name': 'punctuation.definition.entity.css' '3': 'name': 'punctuation.section.function.begin.bracket.round.css' - 'end': '\\)' + 'end': '(\\))|(?=\\{)' 'endCaptures': - '0': + '1': 'name': 'punctuation.section.function.end.bracket.round.css' 'patterns': [ { @@ -1272,9 +1378,9 @@ 'name': 'punctuation.definition.entity.css' '3': 'name': 'punctuation.section.function.begin.bracket.round.css' - 'end': '\\)' + 'end': '(\\))|(?=\\{)' 'endCaptures': - '0': + '1': 'name': 'punctuation.section.function.end.bracket.round.css' 'patterns': [ { @@ -1286,7 +1392,7 @@ 'beginCaptures': '0': 'name': 'punctuation.definition.string.begin.css' - 'end': '"' + 'end': '"|(?` and stays, an + # unclosed one is not. Testing for a balanced block rather than for a + # `)` further along the line keeps this from re-scanning the tail at + # every brace. + 'end': '(\\))|(?=\\{(?![^{}]*\\}))' 'endCaptures': - '0': + '1': 'name': 'punctuation.definition.parameters.end.bracket.round.css' 'patterns': [ + { + 'include': '#condition-string' + } { 'include': '#media-features' } diff --git a/spec/css-spec.mjs b/spec/css-spec.mjs index c769608..4207d9b 100644 --- a/spec/css-spec.mjs +++ b/spec/css-spec.mjs @@ -3631,6 +3631,20 @@ describe('CSS grammar', function () { }); describe("performance regressions", function () { + it("tokenizes long function-like container names in linear time", function () { + var start; + start = Date.now(); + testGrammar.tokenizeLine('@container ' + 'a'.repeat(20000) + '() {'); + assert.ok(Date.now() - start < 5000); + }); + + it("tokenizes long function-like color arguments in linear time", function () { + var start; + start = Date.now(); + testGrammar.tokenizeLine('a { color: rgb(' + 'a'.repeat(20000) + '() }'); + assert.ok(Date.now() - start < 5000); + }); + it("does not hang when tokenizing invalid input preceding an equals sign", function () { var start; start = Date.now(); @@ -3864,4 +3878,498 @@ describe('CSS grammar', function () { assert.deepStrictEqual(tokens[22], { scopes: ['source.css', 'meta.property-list.css', 'meta.property-value.css', 'meta.function.misc.css', 'punctuation.section.function.end.bracket.round.css'], value: ')' }); }); }); + describe('unterminated prelude and function recovery', function () { + it('recovers from an unclosed parenthesis in a prelude', function () { + // An unclosed parenthesis used to leak the at-rule header scope to + // the end of the stylesheet. The `end` pattern now also bails out at + // a `{`. That is only sound where a `{` cannot appear in legal CSS, + // so `@scope` preludes, value functions and functional + // pseudo-classes carry the bail-out. `@media`, `@supports` and + // `@container` conditions deliberately do not: `` + // is `( ? )`, and `` admits a balanced + // `{...}` block, so bailing there would mis-scope legal CSS. + [ + "@scope (.a{", + "@scope (:is(.b){", + "@scope (.a) to (.b{", + "a:is(.b{", + "a:nth-child(2n{" + ].forEach(function (prelude) { + var lines = testGrammar.tokenizeLines(prelude + '\n.after { color: red; }'); + var token = lines[1].find(t => t.value === 'after'); + assert.ok(token, prelude + ' -> .after missing'); + token.scopes.forEach(function (scope) { + assert.ok(!['header', 'meta.function', 'scope.limit'].some(l => scope.includes(l)), + prelude + ' -> leaked ' + scope); + }); + }); + }); + + it('still scopes a well-formed prelude and its closing parenthesis', function () { + // The bail-out must not fire on legal CSS. Each of these closes its + // own parenthesis, so the closing `)` keeps its punctuation scope + // and the following rule is still a selector. + [ + '@media (min-width: 40em) { .a { color: red; } }', + '@supports (display: grid) { .a { color: red; } }', + '@media screen and (min-width: 40em) { .a { color: red; } }', + '@supports not (display: grid) { .a { color: red; } }' + ].forEach(function (source) { + var tokens = testGrammar.tokenizeLines(source)[0]; + var close = tokens.find(t => t.value === ')'); + assert.ok(close, source + ' -> no closing parenthesis token'); + assert.ok(close.scopes.some(s => s.startsWith('punctuation.definition.') + || s.startsWith('punctuation.section.')), + source + ' -> closing parenthesis lost its scope'); + assert.ok(tokens.some(t => t.value === 'red' + && t.scopes.some(s => s.includes('support.constant.color'))), + source + ' -> body declaration not scoped'); + tokens.forEach(function (t) { + if (t.value !== 'red') return; + t.scopes.forEach(function (scope) { + assert.ok(!scope.includes('header'), source + ' -> header leaked to body'); + }); + }); + }); + }); + + it('keeps a var() fallback that legally contains a curly block intact', function () { + // `` admits a balanced block, so `var()` is + // deliberately left without the `{` bail-out: protecting this legal + // value is worth losing recovery on `var(--y{`, which `origin/main` + // does not recover either. + var lines = testGrammar.tokenizeLines('a {\n --t: var(--fb, {\n color: red;\n });\n}'); + assert.deepStrictEqual(lines[1].find(t => t.value === '{').scopes, [ + 'source.css', + 'meta.property-list.css', + 'meta.property-value.css', + 'meta.function.variable.css', + 'punctuation.section.group.begin.bracket.curly.css' + ]); + assert.deepStrictEqual(lines[3].find(t => t.value === ')').scopes, [ + 'source.css', + 'meta.property-list.css', + 'meta.property-value.css', + 'meta.function.variable.css', + 'punctuation.section.function.end.bracket.round.css' + ]); + }); + + it('keeps a custom function argument that legally spans a curly block intact', function () { + // Same trade-off as `var()`. A custom function call is a declaration + // value, so a balanced block inside it may legally span lines. + // `origin/main` scopes this correctly and the guard had regressed it. + var lines = testGrammar.tokenizeLines('.foo {\n --x: --foo({\n color: red;\n });\n}'); + assert.deepStrictEqual(lines[1].find(t => t.value === '{').scopes, [ + 'source.css', + 'meta.property-list.css', + 'meta.property-value.css', + 'meta.function.custom.css', + 'punctuation.section.group.begin.bracket.curly.css' + ]); + assert.deepStrictEqual(lines[3].find(t => t.value === ')').scopes, [ + 'source.css', + 'meta.property-list.css', + 'meta.property-value.css', + 'meta.function.custom.css', + 'punctuation.section.function.end.bracket.round.css' + ]); + }); + + it('keeps a brace inside a legal general-enclosed string out of the body', function () { + // `` accepts ``, so a quoted string + // containing `{` is legal. The bail-out only inspects text from + // the candidate brace onwards, so without a string rule active it + // read the closing quote as an opening one, took the real `)` for + // shielded, and opened the body at the brace inside the string. + [ + ['@media (future: "a{b") {', 'meta.at-rule.media.header.css'], + ['@supports (future "a{b") {', 'meta.at-rule.supports.header.css'] + ].forEach(function (pair) { + var line = pair[0], header = pair[1]; + var tokens = testGrammar.tokenizeLine(line).tokens; + var index = line.indexOf('{'); + var offset = 0, token = null; + tokens.forEach(function (t) { + if (token === null && offset + t.value.length > index) { + token = t; + } + offset += t.value.length; + }); + assert.ok(token.scopes.includes(header), + line + ' -> brace inside string left ' + header); + assert.ok(token.scopes.some(function (s) { return s.includes('string.quoted'); }), + line + ' -> brace inside string is not scoped as a string'); + }); + }); + + it('releases a line-continuation string in an at-rule condition', function () { + // A legal `` string may end a line with an escaped + // newline. The shared `#string` rule handles that by entering a + // multi-line `constant.character.escape.newline.css` region whose + // `end` is `^(? no `color` token on the trailing rule'); + assert.ok(colour.scopes.includes('support.type.property-name.css'), + src + ' -> continuation escaped the condition: ' + colour.scopes.join(' ')); + assert.ok(!last.some(function (t) { + return t.scopes.some(function (sc) { + return sc.includes('escape.newline') || sc.includes('feature-query'); + }); + }), src + ' -> escape/feature-query region still open on the trailing rule'); + }); + }); + + it('bounds an unterminated layer() in an @import prelude', function () { + // `@import` legitimately runs to its `;`, as on `origin/main`, but + // the `layer()` region must not carry its own function scopes past + // the brace. + var lines = testGrammar.tokenizeLines('@import layer(base{\n.after { color: red; }'); + lines[1].forEach(function (token) { + token.scopes.forEach(function (scope) { + assert.ok(!scope.includes('meta.function'), + JSON.stringify(token.value) + ' leaked ' + scope); + }); + }); + }); + }); + describe('feature coverage', function () { + it('does not match scroll-state feature name prefixes', function () { + ['stuckand', 'snappedor'].forEach(function (identifier) { + var tokens = testGrammar.tokenizeLine('@container scroll-state(' + identifier + ') {').tokens; + assert.ok(!tokens.find(x => x.scopes.includes('support.type.property-name.container.css')), identifier); + }); + }); + it('does not tokenize reserved container names', function () { + var tokens = testGrammar.tokenizeLine('@container none (width > 10px) {').tokens; + assert.ok(!tokens.find(x => x.value === 'none' && x.scopes.includes('variable.parameter.container-name.css'))); + }); + it('recovers from semicolon-terminated block at-rules', function () { + [ + '@container (width > 1px);', + '@container;', + '@scope (.x);', + '@scope;', + '@starting-style ;', + '@starting-style;', + '@property --theme;', + '@property;' + ].forEach(function (atRule) { + var lines = testGrammar.tokenizeLines(atRule + '\n.after { color: red; }'); + assert.deepStrictEqual(lines[1].find(x => x.value === 'after').scopes, ['source.css', 'meta.selector.css', 'entity.other.attribute-name.class.css'], atRule); + }); + }); + it('recovers from an unclosed brace nested one level deeper than the function call', function () { + // These two sites survived an earlier mutation campaign only + // because nothing exercised a bare parenthesised group inside a + // function. They are load-bearing, not redundant. + [ + '@container (width > calc((10px{', + 'a { background: -webkit-gradient(linear, left top, from(red{' + ].forEach(function (prelude) { + var lines = testGrammar.tokenizeLines(prelude + '\n.after { color: red; }'); + lines[1].forEach(function (token) { + token.scopes.forEach(function (scope) { + assert.ok(!scope.includes('meta.function'), + prelude + ' -> ' + JSON.stringify(token.value) + ' leaked ' + scope); + }); + }); + }); + }); + it('recovers from an unclosed condition in every region the base grammar recovers in', function () { + // The base grammar has no dedicated @container rule, so its generic + // at-rule header ends at the brace and the following rule is scoped + // normally. Adding a dedicated rule must not lose that. + [ + '@container (width > {', + '@container card (min-width: 100px {', + '@container scroll-state(stuck: {', + '@container scroll-state((stuck: {', + '@media (foo {', + '@media screen and (min-width: {' + ].forEach(function (prelude) { + var lines = testGrammar.tokenizeLines(prelude + '\n.after { color: red; }'); + var token = lines[1].find(t => t.value === 'red'); + assert.ok(token, prelude + ' -> `red` missing, so the line was swallowed whole'); + assert.ok(token.scopes.some(s => s.includes('meta.property-value')), + prelude + ' -> `red` is not a declaration value: ' + token.scopes.join(' ')); + }); + }); + it('does not let an unclosed prelude swallow the rest of the stylesheet', function () { + var rules = Array.from({ length: 48 }, function (_, i) { return '.r' + i + ' { color: red; }'; }); + // Every parenthesised region, against every way the prelude can be + // left unterminated. Covering only one terminator hid leaks in the + // others, and covering only a distant rule hid a leak that consumed + // the first rule whole. + [ + ['@scope (:is(.a)', ['{', '', ';']] + ].forEach(function (probe) { + var open = probe[0]; + probe[1].forEach(function (terminator) { + var prelude = open + terminator; + var lines = testGrammar.tokenizeLines([prelude].concat(rules).join('\n')); + var leaked = ['header', 'meta.function', 'property-value']; + // A bare or `;`-terminated prelude legitimately continues + // onto the next line -- a prelude may span lines until `{`, + // and `;` is legal inside one -- so only a line-final `{` + // guarantees the very next rule is already outside it. + var probes = terminator === '{' ? [[1, 'r0'], [48, 'r47']] : [[48, 'r47']]; + probes.forEach(function (at) { + var token = lines[at[0]].find(t => t.value === at[1]); + assert.ok(token, prelude + ' -> ' + at[1] + ' missing'); + token.scopes.forEach(function (scope) { + assert.ok(!leaked.some(l => scope.includes(l)), prelude + ' -> ' + at[1] + ' leaked ' + scope); + }); + // The rule's own brace must open a property list, not be + // misread as the prelude's own body brace. A `{` + // terminator legitimately opens a container body, so the + // property list is nested rather than top level. + var brace = lines[at[0]].find(t => t.value === '{'); + assert.deepStrictEqual(brace.scopes.slice(-2), ['meta.property-list.css', 'punctuation.section.property-list.begin.bracket.curly.css'], prelude + ' -> ' + at[1] + ' brace'); + brace.scopes.forEach(function (scope) { + assert.ok(!leaked.some(l => scope.includes(l)), prelude + ' -> ' + at[1] + ' brace leaked ' + scope); + }); + }); + }); + }); + }); + it('recovers from env() arguments that begin with a block delimiter', function () { + // A `;` inside a function block and a matched `{}` pair are both + // allowed in a ``, so these are valid custom + // property declarations, not just mid-edit garbage. Either way the + // grammar must not leak past the closing bracket. + [ + 'a { padding: env(;) }', + 'a { padding: env(}) }', + 'a { padding: env(;); }', + 'a { padding: env(});}', + 'a{padding:env(;)}', + 'a { --x: env(;); }', + 'a { --gap: env(;) }', + 'a { --x: env({}); }', + 'a { --x: env({ color: red; }); }' + ].forEach(function (declaration) { + var source = declaration + '\n.after > p { color: red; }'; + var lines = testGrammar.tokenizeLines(source); + assert.deepStrictEqual(lines[1].find(t => t.value === 'after').scopes, ['source.css', 'meta.selector.css', 'entity.other.attribute-name.class.css'], declaration); + assert.ok(!lines[1].find(t => t.scopes.includes('meta.function.env.css')), declaration); + assert.deepStrictEqual(testGrammar.scopeStackAtEnd(source), ['source.css'], declaration); + }); + }); + it('recovers for every region that carries the brace bail-out', function () { + // One case per region carrying the bail-out, so that dropping it from + // any one of them fails a test. Value functions are reached through an + // ordinary declaration and selectors through `@scope`, because the + // container and media conditions that used to reach them no longer + // bail out at a brace. + [ + 'a { color: rgb(255{', + 'a { color: oklch(50%{', + 'a { background: linear-gradient(red{', + 'a { background: -webkit-gradient(linear{', + 'a { clip-path: polygon(0 0{', + 'a { transition-timing-function: cubic-bezier(0,0{', + 'a { transition-timing-function: steps(2{', + 'a { top: anchor(top{', + 'a { width: anchor-size(width{', + 'a { width: clamp(1px{', + 'a { transform: translate(1px{', + 'a { width: calc(1px{', + 'a { width: calc((1px{' + ].forEach(function (declaration) { + var lines = testGrammar.tokenizeLines(declaration + '\n.after { color: red; }'); + var red = lines[1].find(t => t.value === 'red'); + assert.ok(red, declaration + ' -> red missing'); + red.scopes.forEach(function (scope) { + assert.ok(!scope.includes('meta.function'), + declaration + ' -> leaked ' + scope); + }); + }); + [ + '@scope (:dir(ltr{', + '@scope (:lang(en{', + '@scope (:state(foo{', + '@scope (::part(box{', + '@scope (::highlight(h{', + '@scope (::view-transition-group(g{', + '@scope (::scroll-button(up{', + '@scope (::slotted(a{', + '@scope (:is(.a{', + '@scope (:nth-child(2 of .special{' + ].forEach(function (prelude) { + var lines = testGrammar.tokenizeLines(prelude + '\n.after { color: red; }'); + var token = lines[1].find(t => t.value === 'after'); + assert.ok(token, prelude + ' -> .after missing'); + token.scopes.forEach(function (scope) { + assert.ok(!['header', 'meta.function', 'scope.limit'].some(l => scope.includes(l)), + prelude + ' -> leaked ' + scope); + }); + }); + }); + it('does not scope the slash separator inside color functions', function () { + var tokens; + tokens = testGrammar.tokenizeLine('a { color: rgb(0 0 0 / 50%); }').tokens; + var slash = tokens.find(t => t.value.trim() === '/'); + assert.deepStrictEqual(slash.scopes, ['source.css', 'meta.property-list.css', 'meta.property-value.css', 'meta.function.color.css']); + }); + it('does not recognize functions dropped from the css-values-5 draft', function () { + ['media-progress', 'container-progress'].forEach(function (fn) { + var tokens = testGrammar.tokenizeLine('a { width: ' + fn + '(width, 0px, 100px); }').tokens; + assert.ok(!tokens.find(x => x.scopes.includes('support.function.misc.css')), fn); + }); + }); + it('does not recognize toggle(), which the draft renamed to cycle()', function () { + var tokens = testGrammar.tokenizeLine('a { width: toggle(1px, 2px); }').tokens; + assert.ok(!tokens.find(x => x.value === 'toggle' && x.scopes.includes('support.function.misc.css'))); + }); + it('recovers when the unclosed parenthesis is a nested one', function () { + // The bail-out on an outer prelude region cannot fire while an inner + // parenthesised rule is still active, so every region reachable + // inside a prelude needs it too. Covering only outer parentheses + // hid this: each of these leaked the rest of the stylesheet into + // the at-rule header while the outer-paren cases all passed. + [ + '@scope (:is(.a{', + '@scope (.a) to (:where(.b{', + '@scope (:nth-child(2 of .special{' + ].forEach(function (prelude) { + var lines = testGrammar.tokenizeLines(prelude + '\n.after { color: red; }'); + var token = lines[1].find(t => t.value === 'after'); + assert.ok(token, prelude + ' -> .after missing'); + token.scopes.forEach(function (scope) { + assert.ok(!['header', 'meta.function', 'scope.limit'].some(l => scope.includes(l)), + prelude + ' -> leaked ' + scope); + }); + assert.deepStrictEqual( + lines[1].find(t => t.value === '{').scopes.slice(-2), + ['meta.property-list.css', 'punctuation.section.property-list.begin.bracket.curly.css'], + prelude + ' -> brace'); + }); + }); + it('bounds an unterminated language range to its own line', function () { + // `:lang()` carries its own local string rules; without the + // end-of-line fallback the shared `#string` rule already uses, an + // unterminated range stayed open and scoped every following line + // as string content to the end of the file. (`:dir()` takes only + // `ltr`/`rtl` and has no string region, so it is unaffected.) + ['"', "'"].forEach(function (quote) { + var lines = testGrammar.tokenizeLines('@scope (:lang(' + quote + 'en\n.after { color: red; }'); + lines[1].forEach(function (token) { + token.scopes.forEach(function (scope) { + assert.ok(!scope.includes('string.quoted'), + quote + ' -> ' + JSON.stringify(token.value) + ' leaked ' + scope); + }); + }); + assert.deepStrictEqual( + lines[1].find(t => t.value === 'color').scopes.slice(-1), + ['support.type.property-name.css'], + quote + ' -> declaration not recognised on the following line'); + }); + }); + it('recovers regardless of a closing parenthesis later on the line', function () { + // The bail-out is unconditional: it no longer scans the rest of the + // line for a `)`, so a `)` inside a comment, string or escape can + // neither suppress nor trigger it. That scan was quadratic on a + // legal line holding many balanced blocks. + [ + '@scope (.a{ /* ) */', + '@scope (.a{ \\)', + '@scope (.a{ "x)"', + "@scope (.a{ 'x)'", + '@scope (:nth-child(2 of .x{ /* ) */' + ].forEach(function (prelude) { + var lines = testGrammar.tokenizeLines(prelude + '\n.after { color: red; }'); + var token = lines[1].find(t => t.value === 'after'); + assert.ok(token, prelude + ' -> .after missing'); + token.scopes.forEach(function (scope) { + assert.ok(!['header', 'meta.function', 'scope.limit'].some(l => scope.includes(l)), + prelude + ' -> leaked ' + scope); + }); + }); + }); + it('recovers when the prelude line ends in a line-continuation backslash', function () { + // A trailing backslash is not a complete escape, so it matched no + // alternative in the bail-out. The newline-escape rule then stayed + // active and held the prelude open across the line boundary. + [ + '@scope (:nth-child(2 of .x{ \\', + // The same incomplete escape inside an unterminated string. + '@scope (.a{ "x\\', + '@scope (.a{ \'x\\' + ].forEach(function (prelude) { + var lines = testGrammar.tokenizeLines(prelude + '\n.after { color: red; }'); + lines[1].forEach(function (token) { + token.scopes.forEach(function (scope) { + assert.ok(!['header', 'meta.function', 'scope.limit'].some(l => scope.includes(l)), + prelude + ' -> ' + JSON.stringify(token.value) + ' leaked ' + scope); + }); + }); + }); + }); + it('keeps scoping property names that moved within the property list', function () { + // These four sit in a different branch of the property-name + // alternation than they used to. The scope is the same either way, + // so this is a guard against the move dropping one, not a change in + // output: it passes on the unmodified grammar too. + ['ruby-overhang', 'text-wrap', 'white-space-collapse', 'scroll-snap-stop'].forEach(function (prop) { + var tokens = testGrammar.tokenizeLine('a { ' + prop + ': inherit; }').tokens; + var t = tokens.find(x => x.value === prop); + assert.deepStrictEqual(t.scopes, ['source.css', 'meta.property-list.css', 'meta.property-name.css', 'support.type.property-name.css'], prop); + }); + }); + it('does not recognize a bare shape property', function () { + var tokens = testGrammar.tokenizeLine('a { shape: none; }').tokens; + assert.ok(!tokens.find(x => x.value === 'shape' && x.scopes.includes('support.type.property-name.css'))); + }); + it('does not recognize the renamed inset-area property', function () { + var tokens = testGrammar.tokenizeLine('a { inset-area: top; }').tokens; + assert.ok(!tokens.find(x => x.value === 'inset-area' && x.scopes.includes('support.type.property-name.css'))); + }); + it('does not recognize functional-only pseudo-elements without arguments', function () { + ['scroll-button', 'view-transition-group', 'view-transition-image-pair', 'view-transition-new', 'view-transition-old'].forEach(function (pseudoElement) { + var tokens = testGrammar.tokenizeLine('a::' + pseudoElement + ' {}').tokens; + assert.ok(!tokens.find(x => x.value === pseudoElement && x.scopes.includes('entity.other.attribute-name.pseudo-element.css')), pseudoElement); + }); + }); + it('only treats :current() as a functional time pseudo-class', function () { + ['past', 'future'].forEach(function (pseudoClass) { + var tokens = testGrammar.tokenizeLine('a:' + pseudoClass + '(.x) {}').tokens; + assert.ok(!tokens.find(x => x.value === '(' && x.scopes.includes('punctuation.section.function.begin.bracket.round.css')), pseudoClass); + }); + }); + it('does not accept an `of` clause in :nth-of-type()', function () { + // Selectors 4 gives the `of S` syntax to :nth-child()/:nth-last-child() + // only; :nth-of-type() takes An+B alone. + ['a:nth-of-type(2 of .x) {}', 'a:nth-last-of-type(2 of .x) {}'].forEach(function (css) { + var tokens = testGrammar.tokenizeLine(css).tokens; + assert.ok(!tokens.find(t => t.value === 'of' && t.scopes.includes('keyword.operator.logical.of.css')), css); + }); + var tokens = testGrammar.tokenizeLine('a:nth-last-of-type(2n) {}').tokens; + assert.deepStrictEqual(tokens.find(t => t.value === 'nth-last-of-type').scopes, ['source.css', 'meta.selector.css', 'entity.other.attribute-name.pseudo-class.css']); + assert.deepStrictEqual(tokens.find(t => t.value === '2n').scopes, ['source.css', 'meta.selector.css', 'constant.numeric.css']); + }); + it('does not recognize pseudo-classes removed from Selectors 4', function () { + var tokens = testGrammar.tokenizeLine('a:target-within {}').tokens; + assert.ok(!tokens.find(x => x.value === 'target-within' && x.scopes.includes('entity.other.attribute-name.pseudo-class.css'))); + }); + it('does not recognize removed media features', function () { + ['display-capabilities', 'shape'].forEach(function (feature) { + var tokens = testGrammar.tokenizeLine('@media (' + feature + ': round) {').tokens; + assert.ok(!tokens.find(x => x.value === feature && x.scopes.includes('support.type.property-name.media.css')), feature); + }); + }); + }); }); diff --git a/testing-util/test.mjs b/testing-util/test.mjs index 26abebf..c22a849 100644 --- a/testing-util/test.mjs +++ b/testing-util/test.mjs @@ -65,5 +65,21 @@ export default { } return tokenizedLines; + }, + // The scope names left on the tokenizer's rule stack once the whole text + // has been tokenized. A well-behaved grammar closes everything it opens, so + // this is `['source.css']` unless a rule leaked past the end of the input. + scopeStackAtEnd: function scopeStackAtEnd(text) { + const lines = text.split(/\r\n|\r|\n/g); + + let ruleStack = vsctm.INITIAL; + for (let i = 0; i < lines.length; i++) { + ruleStack = grammar.tokenizeLine(lines[i], ruleStack).ruleStack; + } + + // The scopes an editor would report for the next character, rather than + // one entry per rule frame: an unnamed frame inherits its ancestor's + // name, so walking frames reports duplicates that no token ever carries. + return ruleStack.contentNameScopesList.getScopeNames(); } } From 893fca19004be986bccdab6b6b2a4e32a6501b8a Mon Sep 17 00:00:00 2001 From: Kristofer Baxter Date: Mon, 31 Aug 2026 17:20:36 -0700 Subject: [PATCH 2/2] Test the recovery guards that mutation showed were unpinned Six of the guards this branch adds could be removed without any test failing, and two of the deliberate exemptions could be guarded without any test failing. The existing recovery test only rejected scopes containing `header`, `meta.function` or `scope.limit`, and a leaked functional pseudo-class region is still `meta.selector.css`, so it passed either way. Pin the exact scopes instead, and cover: - recovery for `:dir()`, `:lang()`, the `:is()` family and `:nth-of-type()`, none of which had a distinguishing test - recovery for a transform function in an at-rule condition - both sides of the narrowed general-enclosed bail-out, so that `@media (a: {b}) {` stays legal and `@media (a: {b {` recovers - the declaration-value exemption, so that adding a guard to `attr()`/`if()`/`style()`/`cycle()` fails - a url token containing a brace, in `url()` and `url-prefix()` Also replace the `of` clause assertion, which checked for the absence of a scope name no rule can produce and so passed on main as well, with one that pins the tokens. --- spec/css-spec.mjs | 176 +++++++++++++++++++++++++++++++++++++++++++++- 1 file changed, 175 insertions(+), 1 deletion(-) diff --git a/spec/css-spec.mjs b/spec/css-spec.mjs index 4207d9b..d5d14c0 100644 --- a/spec/css-spec.mjs +++ b/spec/css-spec.mjs @@ -3976,6 +3976,166 @@ describe('CSS grammar', function () { ]); }); + it('recovers from an unclosed functional pseudo-class', function () { + // These four sites carry the `{` bail-out. The assertion pins the + // exact scopes of `.after` instead of rejecting a few scope + // fragments: a leaked pseudo-class region is still + // `meta.selector.css`, which a fragment check cannot tell apart + // from a recovered selector, so a weaker assertion passes even + // when the guard is removed. + [ + 'a:dir(ltr{', + 'a:lang(en{', + 'a:is(.b{', + 'a:nth-of-type(2{' + ].forEach(function (prelude) { + var lines = testGrammar.tokenizeLines(prelude + '\n.after { color: red; }'); + assert.deepStrictEqual(lines[1][1], { + scopes: [ + 'source.css', + 'meta.property-list.css', + 'meta.selector.css', + 'entity.other.attribute-name.class.css' + ], + value: 'after' + }, prelude); + }); + }); + + it('recovers from an unclosed transform function in a prelude', function () { + // The transform function rule is reachable from an at-rule + // condition, where losing the bail-out leaks + // `meta.at-rule.media.header.css` over the rest of the file. + var lines = testGrammar.tokenizeLines('@media (x: translate(1px{\n.after { color: red; }'); + assert.deepStrictEqual(lines[1][1], { + scopes: [ + 'source.css', + 'meta.at-rule.media.body.css', + 'meta.selector.css', + 'entity.other.attribute-name.class.css' + ], + value: 'after' + }); + }); + + it('keeps a general-enclosed condition holding a balanced curly block intact', function () { + // `` is `( ? )` and `` + // admits a balanced `{...}` block, so this is legal CSS. The + // condition therefore uses a narrowed bail-out that only fires on + // a `{` which is not closed before the next brace. + var lines = testGrammar.tokenizeLines('@media (a: {b}) {\n.x { color: red; }\n}'); + assert.deepStrictEqual(lines[0][6], { + scopes: ['source.css', 'meta.at-rule.media.header.css'], + value: ' {b}' + }); + assert.deepStrictEqual(lines[0][7], { + scopes: [ + 'source.css', + 'meta.at-rule.media.header.css', + 'punctuation.definition.parameters.end.bracket.round.css' + ], + value: ')' + }); + assert.deepStrictEqual(lines[1][1], { + scopes: [ + 'source.css', + 'meta.at-rule.media.body.css', + 'meta.selector.css', + 'entity.other.attribute-name.class.css' + ], + value: 'x' + }); + }); + + it('recovers from a general-enclosed condition whose curly block is unbalanced', function () { + // The other side of the narrowed bail-out. Here the `{` is not + // closed before the next brace, so the condition releases and the + // body opens rather than swallowing the rest of the file. + var lines = testGrammar.tokenizeLines('@media (a: {b {\n.after { color: red; }'); + assert.deepStrictEqual(lines[0][7], { + scopes: [ + 'source.css', + 'meta.at-rule.media.body.css', + 'punctuation.section.media.begin.bracket.curly.css' + ], + value: '{' + }); + assert.deepStrictEqual(lines[1][1], { + scopes: [ + 'source.css', + 'meta.at-rule.media.body.css', + 'meta.property-list.css', + 'meta.selector.css', + 'entity.other.attribute-name.class.css' + ], + value: 'after' + }); + }); + + it('keeps a declaration-value function holding a curly block intact', function () { + // `attr()`, `if()`, `style()` and `cycle()` share one rule and are + // deliberately left without the bail-out, because a + // `` may legally contain a balanced block that + // spans lines. Adding a guard here would mis-scope legal CSS, so + // this test exists to fail if one is ever added. + var lines = testGrammar.tokenizeLines('a { --x: attr(data-x, {\n color: red;\n}); }'); + assert.deepStrictEqual(lines[0][12], { + scopes: [ + 'source.css', + 'meta.property-list.css', + 'meta.property-value.css', + 'meta.function.misc.css', + 'punctuation.section.group.begin.bracket.curly.css' + ], + value: '{' + }); + assert.deepStrictEqual(lines[2][1], { + scopes: [ + 'source.css', + 'meta.property-list.css', + 'meta.property-value.css', + 'meta.function.misc.css', + 'punctuation.section.function.end.bracket.round.css' + ], + value: ')' + }); + }); + + it('keeps a url token containing a curly brace intact', function () { + // A url token may legally contain `{`, so neither `url()` nor + // `@document url-prefix()` carries the bail-out. + var tokens = testGrammar.tokenizeLine('a { background: url(foo{bar); }').tokens; + assert.deepStrictEqual(tokens[9], { + scopes: [ + 'source.css', + 'meta.property-list.css', + 'meta.property-value.css', + 'meta.function.url.css', + 'variable.parameter.url.css' + ], + value: 'foo{bar' + }); + var lines = testGrammar.tokenizeLines('@document url-prefix(https://example.test/{path) {\n.x { color: red; }\n}'); + assert.deepStrictEqual(lines[0][5], { + scopes: [ + 'source.css', + 'meta.at-rule.document.header.css', + 'meta.function.document-rule.css', + 'variable.parameter.document-rule.css' + ], + value: 'https://example.test/{path' + }); + assert.deepStrictEqual(lines[1][1], { + scopes: [ + 'source.css', + 'meta.at-rule.document.body.css', + 'meta.selector.css', + 'entity.other.attribute-name.class.css' + ], + value: 'x' + }); + }); + it('keeps a brace inside a legal general-enclosed string out of the body', function () { // `` accepts ``, so a quoted string // containing `{` is legal. The bail-out only inspects text from @@ -4353,9 +4513,23 @@ describe('CSS grammar', function () { it('does not accept an `of` clause in :nth-of-type()', function () { // Selectors 4 gives the `of S` syntax to :nth-child()/:nth-last-child() // only; :nth-of-type() takes An+B alone. + // Pin the tokens rather than asserting the absence of a scope name, + // which no rule can produce yet and which therefore passes on + // `origin/main` too. The clause has to stay unrecognised text. ['a:nth-of-type(2 of .x) {}', 'a:nth-last-of-type(2 of .x) {}'].forEach(function (css) { var tokens = testGrammar.tokenizeLine(css).tokens; - assert.ok(!tokens.find(t => t.value === 'of' && t.scopes.includes('keyword.operator.logical.of.css')), css); + assert.deepStrictEqual(tokens[5], { + scopes: ['source.css', 'meta.selector.css'], + value: ' of .x' + }, css); + assert.deepStrictEqual(tokens[6], { + scopes: [ + 'source.css', + 'meta.selector.css', + 'punctuation.section.function.end.bracket.round.css' + ], + value: ')' + }, css); }); var tokens = testGrammar.tokenizeLine('a:nth-last-of-type(2n) {}').tokens; assert.deepStrictEqual(tokens.find(t => t.value === 'nth-last-of-type').scopes, ['source.css', 'meta.selector.css', 'entity.other.attribute-name.pseudo-class.css']);