From 9746e8e875536b3da45ee12a87d65ea98a3b692a Mon Sep 17 00:00:00 2001 From: Ajay Ghanwat Date: Sat, 22 Aug 2026 06:05:09 +0530 Subject: [PATCH] fix(syntax): do not let a regex literal unbalance a block The brace scanner knew about strings and comments but had no case for regex literals. A quote inside one opened a phantom string that swallowed every brace until the next quote; a lone `{` or `}` inside one miscounted block depth. Both failed the component with "Unbalanced braces" pointing at the block's first line. `/-/g` parsed fine, which is why this went unnoticed -- it needs a quote or a brace inside the pattern to bite. Regex-vs-division is decided by scanning back to the last significant character, erring towards division: mistaking division for a regex would swallow code to the next `/` and lose any braces between. A regex cannot span a newline, so an unterminated one on the line is treated as "not a regex", which is what keeps a bare URL in view text intact. Co-Authored-By: Claude Opus 5 --- packages/syntax/src/tokenizer.ts | 86 +++++++++++++++++++++++++++++ packages/syntax/test/syntax.test.ts | 51 +++++++++++++++++ 2 files changed, 137 insertions(+) diff --git a/packages/syntax/src/tokenizer.ts b/packages/syntax/src/tokenizer.ts index a8e2dd76..52554d69 100644 --- a/packages/syntax/src/tokenizer.ts +++ b/packages/syntax/src/tokenizer.ts @@ -48,6 +48,85 @@ export const isIdentPart = (c: string) => /[A-Za-z0-9_]/.test(c); * reimplementing it — a second hand-rolled scanner is how apostrophes in * prose used to swallow braces. */ +/** + * Identifiers that can precede a `/` without ending an expression, so the `/` + * opens a regex rather than dividing. + */ +const REGEX_PRECEDING_KEYWORDS = new Set([ + "return", + "typeof", + "instanceof", + "in", + "of", + "new", + "delete", + "void", + "do", + "else", + "yield", + "await", + "case", +]); + +/** + * Decide whether the `/` at `i` opens a regex literal or is a division sign. + * + * Scans backwards for the last significant character. `a / b` divides; `(/a/)`, + * `= /a/` and `return /a/` do not. Erring towards division is the safe + * direction -- mistaking division for a regex would swallow everything to the + * next `/` and lose any braces in between. + */ +function opensRegex(src: string, i: number): boolean { + let j = i - 1; + while (j >= 0 && (src[j] === " " || src[j] === "\t" || src[j] === "\r" || src[j] === "\n")) j--; + if (j < 0) return true; + const prev = src[j]!; + if (/[A-Za-z0-9_$]/.test(prev)) { + // An identifier ends an expression, so `/` divides -- unless it is a + // keyword that cannot end one, like `return`. + let k = j; + while (k >= 0 && /[A-Za-z0-9_$]/.test(src[k]!)) k--; + return REGEX_PRECEDING_KEYWORDS.has(src.slice(k + 1, j + 1)); + } + // `)` and `]` close an expression, `.` continues one, and a quote ends a + // literal; anything else leaves us in a position where a regex may start. + return ( + prev !== ")" && prev !== "]" && prev !== "." && prev !== '"' && prev !== "'" && prev !== "`" + ); +} + +/** + * Scan a regex literal starting at `i`, returning the index just past its + * closing `/` and flags, or null when this is not in fact a regex. + * + * A regex literal cannot span a newline, so an unterminated one is treated as + * "not a regex" rather than swallowing the rest of the file. That is what keeps + * a bare URL in view text (`https://example.com/a//b`) intact. + */ +function skipRegex(src: string, i: number): number | null { + let j = i + 1; + let inClass = false; + while (j < src.length) { + const c = src[j]!; + if (c === "\n") return null; + if (c === "\\") { + j += 2; + continue; + } + if (inClass) { + if (c === "]") inClass = false; + } else if (c === "[") { + inClass = true; + } else if (c === "/") { + j++; + while (j < src.length && /[a-z]/.test(src[j]!)) j++; + return j; + } + j++; + } + return null; +} + export function skipLiteralOrComment(src: string, i: number, atLineStart: boolean): number | null { const c = src[i]; if (c === "/" && src[i + 1] === "*") { @@ -70,6 +149,13 @@ export function skipLiteralOrComment(src: string, i: number, atLineStart: boolea } return src.length; } + // A regex literal is neither a string nor a brace pair, but it can contain + // both. Without this, a quote inside one opened a phantom string that + // swallowed every brace to the next quote, and a lone `{`/`}` miscounted + // block depth. + if (c === "/" && opensRegex(src, i)) { + return skipRegex(src, i); + } return null; } diff --git a/packages/syntax/test/syntax.test.ts b/packages/syntax/test/syntax.test.ts index bddff1b7..8115628c 100644 --- a/packages/syntax/test/syntax.test.ts +++ b/packages/syntax/test/syntax.test.ts @@ -235,3 +235,54 @@ test("comments may contain apostrophes without unbalancing a block", () => { // The URL in view text must survive: `//` is only a comment at line start. expect(JSON.stringify(ast.view)).toContain("https://example.com/a//b"); }); + +// A regex literal is not a string and not a pair of braces. The brace scanner +// knew about quotes and comments but had no case for regexes, so a quote inside +// one opened a phantom string that swallowed every brace until the next quote, +// and a lone `{` or `}` inside one miscounted depth. Both failed the whole +// component -- with a green build in the reported case, because the damage +// landed in generated output rather than at parse time. +test("regex literals do not unbalance a block", () => { + const mk = (body: string) => + `page P { + load server { +${body} + return { x }; + } + + view {
{x}
} +} +`; + // A quote inside a regex used to open a string that ran to the next quote. + expect(parse(mk(` const x = /it's/.test("its");`)).name).toBe("P"); + expect(parse(mk(` const x = /"/.test("q");`)).name).toBe("P"); + // A brace inside a regex used to be counted as block depth. + expect(parse(mk(` const x = /\{/.test("{");`)).name).toBe("P"); + expect(parse(mk(` const x = /}/.test("}");`)).name).toBe("P"); + // A brace quantifier is balanced, but must not be counted either. + expect(parse(mk(` const x = /^a{2,3}$/.test("aa");`)).name).toBe("P"); + // A `/` inside a character class does not close the regex. + expect(parse(mk(` const x = /[/'"{]/.test("/");`)).name).toBe("P"); + // The case that already worked must keep working. + expect(parse(mk(` const x = "a-b".replace(/-/g, " ");`)).name).toBe("P"); +}); + +// Division must not be mistaken for a regex, or the scanner would swallow code +// from the `/` to the next one and lose any braces in between. +test("division is not treated as a regex literal", () => { + const source = `page P { + load server { + const half = 10 / 2; + const ratio = (a + b) / 2; + const each = items[0] / total; + if (half > 1) { + return { half }; + } + return { half: 0 }; + } + + view {
{half}
} +} +`; + expect(parse(source).name).toBe("P"); +});