From 005c8c1002e10e71ba2b96c31f9263bf9e85b397 Mon Sep 17 00:00:00 2001 From: Jie Date: Sun, 13 Sep 2026 18:07:19 +0800 Subject: [PATCH 1/3] Let comments and strings compete for the text by position No order of rules can render both "https://jsray.org" as a string and // don't stop, won't stop as a comment: strings first cuts the comment at its first apostrophe, comments first cuts the string at its //. Every grammar had chosen one, and the choice was written beside its rules as though it were the fix. Measured on beta.4, 27 grammars cut a line comment holding two quotes, and JavaScript, the C family and PHP took /* x */ inside a string for a comment. A rule may now carry group. Adjacent rules sharing one run as a single pass in which the match beginning earliest wins, and listed order only breaks a tie. Positions are compared where the whole match begins, lookbehind prefix included, so a heredoc opened by << on a line that also carries a quoted redirect is not lost to the quoted name. Comments, strings, heredocs, regex literals, JSON keys, C preprocessor lines and JavaScript parameter lists are spans in every grammar that has them. The parameter list had to be one: ahead of the comments it read a signature inside a doc comment as code, and behind the strings it could not match a list holding a string default. The field is optional and Grammar is still GrammarRule[]; a rule without it runs exactly as before. --- CHANGELOG.md | 32 +++ CONTRIBUTING.md | 7 +- dist/jsray.js | 455 +++++++++++++++++++++++++------------- docs/development.md | 8 +- docs/development.zh-CN.md | 2 +- integrity.json | 2 +- src/jsray.js | 455 +++++++++++++++++++++++++------------- tests/constructs.test.mjs | 121 ++++++++++ types/jsray.d.ts | 7 + 9 files changed, 765 insertions(+), 324 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 4dc4862..540df8c 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,38 @@ versioning follows [SemVer](https://semver.org/). ## [Unreleased] +### Fixed + +- **A comment may hold a quote, and a string a comment marker, in the same + grammar.** Rules ran one after another, and no order renders both: strings + first read `// don't stop, won't stop` as a comment holding the string + `'t stop, won'`; comments first read `"https://jsray.org"` as a string + holding a comment. Every grammar had picked one. Measured on beta.4, 27 + grammars cut a line comment holding two quotes — JavaScript and TypeScript, + Python, PHP, shell, Ruby, SQL, YAML and the eleven C-family grammars among + them — and JavaScript, the C family and PHP read `"/* x */"` inside a string + as a comment. SQL fared worst: its strings may span lines, so `-- don't` + opened a literal that ran on to the next apostrophe anywhere below it. + + A rule may now carry `group`, and adjacent rules sharing one compete by + position: whichever begins first owns the text to its own end, and listed + order only breaks a tie at the same index. Comments, strings, heredocs, regex + literals, JSON keys, C preprocessor lines and JavaScript parameter lists are + spans wherever they exist. The field is optional — `Grammar` is still + `GrammarRule[]`, and a rule without it behaves exactly as before. + + Three forms added in beta.4 were part of the problem. Ruby's `%w[…]`, Perl's + `q{…}` and Elixir's `~r/…/` sat ahead of the comment rule, so + `# prefer %w[a b]` lost the rest of its comment; and a heredoc opening written + inside a comment, such as `# cat < s.replace(/[&<>"]/g, (c) => escapeMap[c]); /** - * Apply grammar rules to a string in order. Rule array order = priority (first-match wins). + * Apply grammar rules to a string in order. Rule array order = priority + * (first-match wins), except among adjacent rules sharing a group. * Each rule: * { cls: 'tk-xxx', * pattern: /re/, // must be globalizable; the 'g' flag is forced internally * inside?: rules, // nested grammar (recursively tokenize captured text) * lookbehind?: true, // capture group 1 is consumed as prefix but not colored - * close?: fn } // pattern matches the opening only; fn finds the end + * close?: fn, // pattern matches the opening only; fn finds the end + * group?: 'name' } // adjacent rules sharing a name compete by position * * `close(match, text, from) -> index | -1` exists for the forms whose end is * not knowable when the rule is written. A heredoc ends at the word its own @@ -34,47 +36,130 @@ * Returning -1 means "no terminator here": the opening is left to the rules * behind this one rather than swallowing the rest of the file, because a * false opening is likelier than a genuinely unterminated literal. + * + * `group` exists because order cannot decide between a string and a + * comment. Strings first reads `// don't stop, won't stop` as a comment + * holding the string `'t stop, won'`; comments first reads + * `"https://jsray.org"` as a string holding a comment. Every grammar here had + * picked one of those two failures and written the choice down beside its + * rules as though it were the fix. The language decides by position — + * whichever opens first owns the text up to its own end — so the rules of a + * group run as one pass in which the earliest match wins at every step, and + * listed order only breaks a tie at the same index (which is how `/**` still + * beats `/*`). Positions are compared where the whole match begins, lookbehind + * prefix included: a heredoc's body starts on the next line, and measured + * from there `cat < "out.txt"` would lose its `<<` to the quoted name. */ function tokenize(code, rules) { let stream = [code]; - for (const rule of rules) { - const next = []; - // Compile once per rule and cache on the rule object — the old code - // built a fresh RegExp per stream piece, which dominated tokenize time - // on fragmented streams. lastIndex is reset per piece instead. - const re = rule._re || (rule._re = new RegExp( - rule.pattern.source, - (rule.pattern.flags || '').replace('g', '') + 'g' - )); - for (const piece of stream) { - if (typeof piece !== 'string') { next.push(piece); continue; } - re.lastIndex = 0; - let last = 0, m; - while ((m = re.exec(piece)) !== null) { - const lbLen = rule.lookbehind && m[1] ? m[1].length : 0; - const start = m.index + lbLen; - let text = m[0].slice(lbLen); - if (rule.close) { - const end = rule.close(m, piece, m.index + m[0].length); - if (end < 0) { re.lastIndex = m.index + 1; continue; } - text = piece.slice(start, end); + for (let r = 0; r < rules.length; ) { + const rule = rules[r]; + if (!rule.group) { + stream = applyRule(stream, rule); + r++; + continue; + } + let end = r + 1; + while (end < rules.length && rules[end].group === rule.group) end++; + stream = applyGroup(stream, rules.slice(r, end)); + r = end; + } + return stream; + } + + function compile(rule) { + // Compile once per rule and cache on the rule object — the old code + // built a fresh RegExp per stream piece, which dominated tokenize time + // on fragmented streams. lastIndex is reset per piece instead. + return rule._re || (rule._re = new RegExp( + rule.pattern.source, + (rule.pattern.flags || '').replace('g', '') + 'g' + )); + } + + function applyRule(stream, rule) { + const next = []; + const re = compile(rule); + for (const piece of stream) { + if (typeof piece !== 'string') { next.push(piece); continue; } + re.lastIndex = 0; + let last = 0, m; + while ((m = re.exec(piece)) !== null) { + const lbLen = rule.lookbehind && m[1] ? m[1].length : 0; + const start = m.index + lbLen; + let text = m[0].slice(lbLen); + if (rule.close) { + const end = rule.close(m, piece, m.index + m[0].length); + if (end < 0) { re.lastIndex = m.index + 1; continue; } + text = piece.slice(start, end); + } + if (!text) { re.lastIndex++; continue; } + if (start > last) next.push(piece.slice(last, start)); + next.push({ + type: rule.cls, + content: rule.inside ? tokenize(text, rule.inside) : text, + }); + last = start + text.length; + // The body of a close-delimited form has already been consumed; + // resuming inside it would re-match its own contents. + if (rule.close) re.lastIndex = last; + } + if (last < piece.length) next.push(piece.slice(last)); + } + return next; + } + + /** The first acceptable match of `rule` at or after `from`, or null. */ + function locate(rule, piece, from) { + const re = compile(rule); + re.lastIndex = from; + let m; + while ((m = re.exec(piece)) !== null) { + const lbLen = rule.lookbehind && m[1] ? m[1].length : 0; + const start = m.index + lbLen; + let text = m[0].slice(lbLen); + if (rule.close) { + const end = rule.close(m, piece, m.index + m[0].length); + if (end < 0) { re.lastIndex = m.index + 1; continue; } + text = piece.slice(start, end); + } + if (!text) { re.lastIndex = m.index + 1; continue; } + return { index: m.index, start, text }; + } + return null; + } + + function applyGroup(stream, group) { + const next = []; + for (const piece of stream) { + if (typeof piece !== 'string') { next.push(piece); continue; } + // One cursor per rule, kept until the text it points into has been + // claimed by another rule. Cursors only move forward, so a pass costs + // about what running the same rules one after another did. + const hits = new Array(group.length); + let pos = 0; + for (;;) { + let win = -1; + for (let i = 0; i < group.length; i++) { + const hit = hits[i]; + if (hit === undefined || (hit !== null && hit.index < pos)) { + hits[i] = locate(group[i], piece, pos); } - if (!text) { re.lastIndex++; continue; } - if (start > last) next.push(piece.slice(last, start)); - next.push({ - type: rule.cls, - content: rule.inside ? tokenize(text, rule.inside) : text, - }); - last = start + text.length; - // The body of a close-delimited form has already been consumed; - // resuming inside it would re-match its own contents. - if (rule.close) re.lastIndex = last; + if (hits[i] && (win < 0 || hits[i].index < hits[win].index)) win = i; } - if (last < piece.length) next.push(piece.slice(last)); + if (win < 0) break; + const { start, text } = hits[win]; + const rule = group[win]; + if (start > pos) next.push(piece.slice(pos, start)); + next.push({ + type: rule.cls, + content: rule.inside ? tokenize(text, rule.inside) : text, + }); + pos = start + text.length; } - stream = next; + if (pos < piece.length) next.push(piece.slice(pos)); } - return stream; + return next; } function render(stream) { @@ -183,30 +268,41 @@ ).split(' '); // Parameter-list sub-grammar · colors n / name in (n: number, name = "x") as var-param + // + // A parameter list is claimed where it opens, ahead of any comment or string + // default inside it, so this is the first grammar to see those. They are + // taken before punctuation can split them at a comma. const jsParamInside = [ + { cls: 'tk-comment', pattern: /\/\*[\s\S]*?\*\/|\/\/.*/, group: 'span' }, + { cls: 'tk-string', pattern: /"(?:\\.|[^"\\])*"|'(?:\\.|[^'\\])*'/, group: 'span' }, { cls: 'tk-punct', pattern: /[(),]/ }, { cls: 'tk-type', pattern: /(:\s*)[A-Z][\w$]*/, lookbehind: true }, { cls: 'tk-keyword', pattern: /\b(?:number|string|boolean|void|null|undefined|any|unknown|never|true|false)\b/ }, - { cls: 'tk-string', pattern: /"(?:\\.|[^"\\])*"|'(?:\\.|[^'\\])*'/ }, { cls: 'tk-number', pattern: /\b\d[\d_]*(?:\.\d+)?\b/ }, { cls: 'tk-var-param', pattern: /\b[A-Za-z_$][\w$]*\b/ }, { cls: 'tk-operator', pattern: /[=?:.]/ }, ]; G.javascript = [ - // Doc comments must come before plain block comments. Block comments stay - // before strings; LINE comments come after strings (see below) so - // "https://..." inside a string never becomes a comment. - { cls: 'tk-doc', pattern: /\/\*\*[\s\S]*?\*\// }, - { cls: 'tk-comment', pattern: /\/\*[\s\S]*?\*\// }, + // Everything that opens a span — doc and block comments, parameter lists, + // strings, line comments, regex literals — competes by position (see + // `group` in tokenize). Listed order only breaks ties at one index: `/**` + // before `/*`, and `//` before a regex, which cannot begin with `//`. + // + // A parameter list is a span too. Ahead of the comments it would read + // `function f(a, b)` inside a doc comment as a real signature; behind the + // strings it could not match `(a = "x")` once the default was split off. + // Competing, it is claimed only where it opens first. + { cls: 'tk-doc', pattern: /\/\*\*[\s\S]*?\*\//, group: 'span' }, + { cls: 'tk-comment', pattern: /\/\*[\s\S]*?\*\//, group: 'span' }, // Parameter lists · function foo(...) / (...) => / async (...) => { cls: 'tk-scope', pattern: /(\bfunction\s*[\w$]*\s*)\([^()]*\)/, lookbehind: true, - inside: jsParamInside }, + inside: jsParamInside, group: 'span' }, { cls: 'tk-scope', pattern: /\([^()]*\)(?=\s*=>)/, - inside: jsParamInside }, + inside: jsParamInside, group: 'span' }, // Template strings (with inline ${...}) // @@ -216,22 +312,24 @@ // engine try every combination: 26 placeholders took 8.7s to fail. Every // interpolating grammar below keeps its fallback and its interpolation // branch disjoint for the same reason. - { cls: 'tk-string', pattern: /`(?:\\.|\$\{[^}]*\}|\$(?!\{)|[^`\\$])*`/, inside: [ + { cls: 'tk-string', pattern: /`(?:\\.|\$\{[^}]*\}|\$(?!\{)|[^`\\$])*`/, group: 'span', inside: [ { cls: 'tk-operator', pattern: /\$\{[^}]*\}/, inside: [ { cls: 'tk-punct', pattern: /^\$\{|\}$/ }, // No JS recursion here; minimal coloring avoids rule cross-talk { cls: 'tk-var', pattern: /[A-Za-z_$][\w$]*/ }, ]}, ]}, - { cls: 'tk-string', pattern: RX.string1 }, - { cls: 'tk-string', pattern: RX.string2 }, - { cls: 'tk-comment', pattern: /\/\/.*/ }, + { cls: 'tk-string', pattern: RX.string1, group: 'span' }, + { cls: 'tk-string', pattern: RX.string2, group: 'span' }, + { cls: 'tk-comment', pattern: /\/\/.*/, group: 'span' }, // Regex: only recognize after =/(/,/!/keyword to avoid eating division // Capture prefix whitespace as lookbehind so it stays outside the regex token. + // In the span group because a regex may hold a quote: `s.split(/"/)` read + // as the start of a string took the rest of the line with it. { cls: 'tk-regex', pattern: /(^|[=(,!&|?:;{}\[\]]\s*|\breturn\s*)\/(?![*\/])(?:\\.|\[(?:\\.|[^\]\\\n])*\]|[^\/\\\n])+\/[gimsuy]*/, - lookbehind: true }, + lookbehind: true, group: 'span' }, // Decorators { cls: 'tk-decorator', pattern: /@[A-Za-z_$][\w$]*/ }, @@ -305,9 +403,11 @@ ]; G.python = [ - // Triple-quoted strings (with f/r/b prefixes). All strings before - // comments so # inside "..." never becomes a comment. - { cls: 'tk-string', pattern: /(?:[rRbBuUfF]{0,2})("""[\s\S]*?"""|'''[\s\S]*?''')/ }, + // Strings and comments compete by position (see `group` in tokenize), so + // `#` inside "..." stays text and a quote inside a comment stays comment. + // Triple-quoted strings (with f/r/b prefixes) are listed first: at the + // same index they must win over `""` followed by a lone quote. + { cls: 'tk-string', pattern: /(?:[rRbBuUfF]{0,2})("""[\s\S]*?"""|'''[\s\S]*?''')/, group: 'span' }, // PEP 701 lets a replacement field carry the same quote that delimits the // string: `f"{a["k"]}"` is valid from Python 3.12. The general rule below // stops at the first inner quote, which split one string into two tokens @@ -320,9 +420,9 @@ // is deliberately left to fall through to the general rule: covering it // needs a nested quantifier, which is the ambiguous shape that caused the // beta.4 denial of service. - { cls: 'tk-string', pattern: /(?:[rRbB][fF]|[fF][rRbB]?)("(?:\{[^{}]*\}|\\.|[^"\\\n{])*"|'(?:\{[^{}]*\}|\\.|[^'\\\n{])*')/ }, - { cls: 'tk-string', pattern: /(?:[rRbBuUfF]{0,2})("(?:\\.|[^"\\\n])*"|'(?:\\.|[^'\\\n])*')/ }, - { cls: 'tk-comment', pattern: /#.*/ }, + { cls: 'tk-string', pattern: /(?:[rRbB][fF]|[fF][rRbB]?)("(?:\{[^{}]*\}|\\.|[^"\\\n{])*"|'(?:\{[^{}]*\}|\\.|[^'\\\n{])*')/, group: 'span' }, + { cls: 'tk-string', pattern: /(?:[rRbBuUfF]{0,2})("(?:\\.|[^"\\\n])*"|'(?:\\.|[^'\\\n])*')/, group: 'span' }, + { cls: 'tk-comment', pattern: /#.*/, group: 'span' }, // Parameter list · def foo(self, n: int = 0) { cls: 'tk-scope', @@ -440,8 +540,13 @@ { cls: 'tk-punct', pattern: /[{}[\]:,]/ }, ]; G.jsonc = [ - { cls: 'tk-comment', pattern: /\/\*[\s\S]*?\*\/|\/\/.*/ }, - ...G.json, + // Comments and strings compete by position (see `group` in tokenize). The + // old order put comments first, so a value such as "https://jsray.org" — + // everywhere in editor settings and tsconfig files — was cut at its `//`. + { cls: 'tk-comment', pattern: /\/\*[\s\S]*?\*\/|\/\/.*/, group: 'span' }, + { cls: 'tk-type', pattern: /"(?:\\.|[^"\\])*"(?=\s*:)/, group: 'span' }, + { cls: 'tk-string', pattern: /"(?:\\.|[^"\\])*"/, group: 'span' }, + ...G.json.filter((rule) => rule.cls !== 'tk-type' && rule.cls !== 'tk-string'), ]; // ============================================================ @@ -465,17 +570,20 @@ { cls: 'tk-string', pattern: /(<<(-?)[ \t]*(['"]?)([A-Za-z_]\w*)\3[^\n]*\n)/, lookbehind: true, - close: heredocEnd(4, 2) }, + close: heredocEnd(4, 2), + group: 'span' }, - // Strings before comments, else # inside "..." is eaten as a comment. + // Strings, heredocs and comments compete by position (see `group` in + // tokenize): `#` inside "..." stays text, a quote in a comment stays + // comment, and `# cat </i }, - { cls: 'tk-string', pattern: /"(?:\\.|\$[A-Za-z_]\w*|[^"\\$\n])*"/, inside: [ + close: heredocEnd(3, true, '\\b'), + group: 'span' }, + + // Comments and strings compete by position (see `group` in tokenize), so + // "https://..." and "#anchor" stay strings, "/* x */" stays a string, and + // a quote inside a comment stays comment. + { cls: 'tk-comment', pattern: /\/\*[\s\S]*?\*\//, group: 'span' }, + { cls: 'tk-string', pattern: /"(?:\\.|\$[A-Za-z_]\w*|[^"\\$\n])*"/, group: 'span', inside: [ { cls: 'tk-var', pattern: /\$[A-Za-z_]\w*/ }, ]}, - { cls: 'tk-string', pattern: /'(?:\\.|[^'\\\n])*'/ }, - { cls: 'tk-comment', pattern: /\/\/.*|#.*/ }, + { cls: 'tk-string', pattern: /'(?:\\.|[^'\\\n])*'/, group: 'span' }, + { cls: 'tk-comment', pattern: /\/\/.*|#.*/, group: 'span' }, + // Open and close tags after the spans, so "?>" inside a string is text. + { cls: 'tk-decorator', pattern: /<\?(?:php|=)?|\?>/i }, { cls: 'tk-var', pattern: /\$[A-Za-z_]\w*/ }, { cls: 'tk-var-const', pattern: /\b[A-Z][A-Z0-9_]{2,}\b/ }, { cls: 'tk-type', pattern: /(\b(?:class|interface|trait|enum|extends|implements|new)\s+)[A-Za-z_]\w*/, lookbehind: true }, @@ -556,16 +667,23 @@ function cLikeGrammar(keywords, builtins, options) { const opts = options || {}; const rules = [ - // Block comments stay before strings (license headers quote freely); - // LINE comments come after strings so "https://..." never becomes a + // Comments, strings and preprocessor lines compete by position (see + // `group` in tokenize): a license header can quote freely, a string can + // hold `/* … */` or `https://`, and a comment can hold `don't` twice. + // The option blocks below splice their literals in among these, in the + // same group. + // + // A preprocessor line is a span because it runs to the end of its line: + // `#include "a.h"` keeps its path, and a commented-out `#define` stays a // comment. - { cls: 'tk-doc', pattern: /\/\*\*[\s\S]*?\*\// }, - { cls: 'tk-comment', pattern: /\/\*[\s\S]*?\*\// }, - { cls: 'tk-decorator', pattern: /^\s*#\s*[A-Za-z_]\w*.*/m }, + { cls: 'tk-doc', pattern: /\/\*\*[\s\S]*?\*\//, group: 'span' }, + { cls: 'tk-comment', pattern: /\/\*[\s\S]*?\*\//, group: 'span' }, + { cls: 'tk-decorator', pattern: /^\s*#\s*[A-Za-z_]\w*.*/m, group: 'span' }, + { cls: 'tk-string', pattern: /"(?:\\.|[^"\\\n])*"/, group: 'span' }, + { cls: 'tk-string', pattern: /'(?:\\.|[^'\\\n])*'/, group: 'span' }, + { cls: 'tk-comment', pattern: /\/\/.*/, group: 'span' }, + // Annotations come after the spans, so `// see @Override` stays a comment. { cls: 'tk-decorator', pattern: /@[A-Za-z_]\w*/ }, - { cls: 'tk-string', pattern: /"(?:\\.|[^"\\\n])*"/ }, - { cls: 'tk-string', pattern: /'(?:\\.|[^'\\\n])*'/ }, - { cls: 'tk-comment', pattern: /\/\/.*/ }, { cls: 'tk-var-const', pattern: /\b[A-Z][A-Z0-9_]{2,}\b/ }, { cls: 'tk-type', pattern: /(\b(?:class|struct|interface|enum|trait|extends|implements|namespace|using|new|object|protocol|extension|mixin|record|actor)\s+)[A-Za-z_]\w*/, lookbehind: true }, { cls: 'tk-fn-decl', pattern: new RegExp('\\b(?!(?:' + CLIKE_DECL_SKIP + ')\\b)[A-Za-z_]\\w*(?=\\s*\\([^;{}]*\\)\\s*(?:const\\s*)?(?:->\\s*[A-Za-z_:][\\w:<>]*)?\\{)') }, @@ -604,6 +722,7 @@ rules.splice(rules.findIndex((r) => r.cls === 'tk-string'), 0, { cls: 'tk-string', pattern: /`[^`]*`/, + group: 'span', }); } @@ -623,6 +742,7 @@ rules.splice(rules.findIndex((r) => r.cls === 'tk-string'), 0, { cls: 'tk-string', pattern: /\b(?:[uU]8?|[LU])?R"([^\s()\\]{0,16})\([\s\S]*?\)\1"/, + group: 'span', }); } @@ -630,6 +750,7 @@ rules.splice(rules.findIndex((r) => r.cls === 'tk-string'), 0, { cls: 'tk-string', pattern: /\b(?:b?r)(#{0,16})"[\s\S]*?"\1/, + group: 'span', }); } @@ -642,6 +763,7 @@ rules.splice(rules.findIndex((r) => r.cls === 'tk-string'), 0, { cls: 'tk-string', pattern: /"""[\s\S]*?"""|'''[\s\S]*?'''/, + group: 'span', }); } @@ -654,8 +776,8 @@ // preceded: leaving it in place would keep one pattern around that can // still span from any apostrophe to any later one. if (opts.lifetimes) { - // The character literal stays where the string rule was, because it is a - // string and line comments deliberately come after strings. + // The character literal takes the string rule's place in the span group, + // because it is a string. const quoted = rules.findIndex( (r) => r.cls === 'tk-string' && r.pattern.source[0] === "'" ); @@ -663,6 +785,7 @@ // Exactly one character or one escape, then the closing quote. cls: 'tk-string', pattern: /'(?:\\(?:u\{[\da-fA-F]{1,6}\}|.)|[^'\\\n])'/, + group: 'span', }); // The lifetime goes AFTER the line-comment rule, and the distance between @@ -830,14 +953,14 @@ { cls: 'tk-string', pattern: /(<<([-~]?)(['"]?)([A-Z_]\w*)\3[^\n]*\n)/, lookbehind: true, - close: heredocEnd(4, 2) }, + close: heredocEnd(4, 2), + group: 'span' }, - // `=begin` / `=end` blocks come before the strings that come before - // everything else. The markers are only special at column zero, so the - // anchors here are load-bearing rather than decorative. Without this rule - // a documentation block was read as ordinary code — the body's words came - // out coloured as function calls and keywords. - { cls: 'tk-comment', pattern: /^=begin\b[\s\S]*?^=end.*$/m }, + // `=begin` / `=end` blocks. The markers are only special at column zero, + // so the anchors here are load-bearing rather than decorative. Without + // this rule a documentation block was read as ordinary code — the body's + // words came out coloured as function calls and keywords. + { cls: 'tk-comment', pattern: /^=begin\b[\s\S]*?^=end.*$/m, group: 'span' }, // %w[…] %i(…) %q{…} %Q<…>: the delimiter is picked at the call site, so // the closer is only knowable once the opener has been read, and bracket @@ -845,19 +968,24 @@ // out on purpose — it cannot be told from the modulo operator without // parsing, and it is rare enough not to be worth mistaking `a %(b)` for a // literal. - { cls: 'tk-regex', pattern: /%r([([{<|!\/])/, close: pairedEnd(1) }, - { cls: 'tk-string', pattern: /%[wWiIqQsx]([([{<|!\/])/, close: pairedEnd(1) }, - - // Strings must come before comments, else # inside "..." (incl. #{} interpolation) - // is eaten as a comment. String bodies stay single-line so an unpaired quote - // in a comment can't swallow following lines. + // + // These sat ahead of the comment rule in beta.4, which is how + // `# prefer %w[a b]` lost the rest of its comment. In the span group they + // are claimed only where they open before a `#` does. + { cls: 'tk-regex', pattern: /%r([([{<|!\/])/, close: pairedEnd(1), group: 'span' }, + { cls: 'tk-string', pattern: /%[wWiIqQsx]([([{<|!\/])/, close: pairedEnd(1), group: 'span' }, + + // Strings and comments compete by position (see `group` in tokenize), so + // `#` inside "..." (incl. #{} interpolation) stays text and a quote inside + // a comment stays comment. String bodies stay single-line so an unpaired + // quote can't swallow following lines. // Fallback excludes `#`; a bare `#` is admitted only when no `{` follows, // so `#{...}` has exactly one parse (see the JS template-string note). - { cls: 'tk-string', pattern: /"(?:\\.|#\{[^}\n]*\}|#(?!\{)|[^"\\\n#])*"/, inside: [ + { cls: 'tk-string', pattern: /"(?:\\.|#\{[^}\n]*\}|#(?!\{)|[^"\\\n#])*"/, group: 'span', inside: [ { cls: 'tk-operator', pattern: /#\{[^}\n]*\}/ }, ]}, - { cls: 'tk-string', pattern: /'(?:\\.|[^'\\\n])*'/ }, - { cls: 'tk-comment', pattern: /#.*/ }, + { cls: 'tk-string', pattern: /'(?:\\.|[^'\\\n])*'/, group: 'span' }, + { cls: 'tk-comment', pattern: /#.*/, group: 'span' }, { cls: 'tk-var-builtin', pattern: /[@$]{1,2}[A-Za-z_]\w*|\bself\b/ }, { cls: 'tk-var-const', pattern: /\b[A-Z][A-Z0-9_]{2,}\b/ }, { cls: 'tk-type', pattern: /(\b(?:class|module)\s+)[A-Z]\w*/, lookbehind: true }, @@ -884,13 +1012,14 @@ ).split(' '); G.lua = [ - // Block comments before strings; line comments after strings so - // "not -- a comment" stays a string. - { cls: 'tk-comment', pattern: /--\[\[[\s\S]*?\]\]/ }, - { cls: 'tk-string', pattern: /\[\[[\s\S]*?\]\]/ }, - { cls: 'tk-string', pattern: /"(?:\\.|[^"\\\n])*"/ }, - { cls: 'tk-string', pattern: /'(?:\\.|[^'\\\n])*'/ }, - { cls: 'tk-comment', pattern: /--.*/ }, + // Comments and strings compete by position (see `group` in tokenize), so + // "not -- a comment" stays a string and `-- don't` stays a comment. The + // long comment is listed ahead of the line comment: both open at `--`. + { cls: 'tk-comment', pattern: /--\[\[[\s\S]*?\]\]/, group: 'span' }, + { cls: 'tk-string', pattern: /\[\[[\s\S]*?\]\]/, group: 'span' }, + { cls: 'tk-string', pattern: /"(?:\\.|[^"\\\n])*"/, group: 'span' }, + { cls: 'tk-string', pattern: /'(?:\\.|[^'\\\n])*'/, group: 'span' }, + { cls: 'tk-comment', pattern: /--.*/, group: 'span' }, { cls: 'tk-var-builtin', pattern: /\b(?:self|_G|_VERSION)\b/ }, { cls: 'tk-fn-decl', pattern: /(\bfunction\s+)[A-Za-z_]\w*(?:[.:][A-Za-z_]\w*)?/, @@ -917,11 +1046,13 @@ const SQL_BUILTINS = 'count sum avg min max coalesce nullif lower upper substr substring now date'.split(' '); G.sql = [ - // Block comments before strings; line comments after strings so - // 'not -- a comment' stays a string. - { cls: 'tk-comment', pattern: /\/\*[\s\S]*?\*\// }, - { cls: 'tk-string', pattern: /'(?:''|[^'])*'|"(?:\\"|[^"])*"/ }, - { cls: 'tk-comment', pattern: /--.*/ }, + // Comments and strings compete by position (see `group` in tokenize), so + // 'not -- a comment' stays a string. This grammar's strings may span lines, + // which made the old order worse than elsewhere: `-- don't` opened a string + // that ran on until the next apostrophe anywhere below it. + { cls: 'tk-comment', pattern: /\/\*[\s\S]*?\*\//, group: 'span' }, + { cls: 'tk-string', pattern: /'(?:''|[^'])*'|"(?:\\"|[^"])*"/, group: 'span' }, + { cls: 'tk-comment', pattern: /--.*/, group: 'span' }, { cls: 'tk-keyword', pattern: new RegExp('\\b(?:' + SQL_KEYWORDS.join('|') + ')\\b', 'i') }, { cls: 'tk-fn-builtin', pattern: new RegExp('\\b(?:' + SQL_BUILTINS.join('|') + ')\\b', 'i') }, { cls: 'tk-number', pattern: /-?\b\d+(?:\.\d+)?\b/ }, @@ -940,11 +1071,14 @@ // and stopping early beats swallowing the rest of the document. { cls: 'tk-string', pattern: /(:[ \t]*)[|>][-+]?\d*[ \t]*(?:\n[ \t]+.*)*/, - lookbehind: true }, + lookbehind: true, + group: 'span' }, - // Strings before comments, else # inside "..." is eaten as a comment - { cls: 'tk-string', pattern: /"(?:\\.|[^"\\\n])*"|'(?:''|[^'\n])*'/ }, - { cls: 'tk-comment', pattern: /#.*/ }, + // Strings, block scalars and comments compete by position (see `group` in + // tokenize): `#` inside "..." stays text, and a `key: |` written inside a + // comment opens nothing. + { cls: 'tk-string', pattern: /"(?:\\.|[^"\\\n])*"|'(?:''|[^'\n])*'/, group: 'span' }, + { cls: 'tk-comment', pattern: /#.*/, group: 'span' }, { cls: 'tk-type', pattern: /^(\s*)[A-Za-z_][\w.-]*(?=\s*:)/m, lookbehind: true }, { cls: 'tk-decorator', pattern: /[&*][A-Za-z_][\w-]*/ }, { cls: 'tk-keyword', pattern: /\b(?:true|false|null|yes|no|on|off)\b/i }, @@ -997,9 +1131,9 @@ ).split(' '); G.r = [ - // Strings before comments, else # inside "..." is eaten as a comment - { cls: 'tk-string', pattern: /"(?:\\.|[^"\\\n])*"|'(?:\\.|[^'\\\n])*'/ }, - { cls: 'tk-comment', pattern: /#.*/ }, + // Strings and comments compete by position (see `group` in tokenize) + { cls: 'tk-string', pattern: /"(?:\\.|[^"\\\n])*"|'(?:\\.|[^'\\\n])*'/, group: 'span' }, + { cls: 'tk-comment', pattern: /#.*/, group: 'span' }, { cls: 'tk-fn-decl', pattern: /\b[A-Za-z._][\w.]*(?=\s*(?:<-|=)\s*function\b)/ }, { cls: 'tk-keyword', pattern: wordPattern(R_KEYWORDS) }, { cls: 'tk-fn-builtin', @@ -1025,21 +1159,22 @@ ).split(' '); G.perl = [ - { cls: 'tk-doc', pattern: /^=\w+[\s\S]*?^=cut\s*$/m }, - // Strings before comments, else # inside "..." is eaten as a comment - { cls: 'tk-string', pattern: /"(?:\\.|[^"\\\n])*"/, inside: [ + // POD, strings, quoting operators, comments and bound regexes compete by + // position (see `group` in tokenize). `#` is ordinary text inside a string, + // a `q{…}` or a `/#/`; a quote or a `q{` inside a comment is comment. + { cls: 'tk-doc', pattern: /^=\w+[\s\S]*?^=cut\s*$/m, group: 'span' }, + { cls: 'tk-string', pattern: /"(?:\\.|[^"\\\n])*"/, group: 'span', inside: [ { cls: 'tk-var', pattern: /[$@][A-Za-z_]\w*/ }, ]}, - { cls: 'tk-string', pattern: /'(?:\\.|[^'\\\n])*'/ }, + { cls: 'tk-string', pattern: /'(?:\\.|[^'\\\n])*'/, group: 'span' }, - // q{…} qq{…} qw{…} qr{…} — ahead of the comment rule, because `#` is - // ordinary text inside one. `/` is excluded as a delimiter for the - // quoting forms: after a bare word it is far more often division. - { cls: 'tk-regex', pattern: /\bqr[ \t]*([([{<|!\/])/, close: pairedEnd(1) }, - { cls: 'tk-string', pattern: /\b(?:qq|qw|q)[ \t]*([([{<|!])/, close: pairedEnd(1) }, + // q{…} qq{…} qw{…} qr{…}. `/` is excluded as a delimiter for the quoting + // forms: after a bare word it is far more often division. + { cls: 'tk-regex', pattern: /\bqr[ \t]*([([{<|!\/])/, close: pairedEnd(1), group: 'span' }, + { cls: 'tk-string', pattern: /\b(?:qq|qw|q)[ \t]*([([{<|!])/, close: pairedEnd(1), group: 'span' }, - { cls: 'tk-comment', pattern: /#.*/ }, - { cls: 'tk-regex', pattern: /((?:=~|!~)\s*)(?:m|s|tr|y)?\/(?:\\.|[^/\n])*\/[a-z]*/, lookbehind: true }, + { cls: 'tk-comment', pattern: /#.*/, group: 'span' }, + { cls: 'tk-regex', pattern: /((?:=~|!~)\s*)(?:m|s|tr|y)?\/(?:\\.|[^/\n])*\/[a-z]*/, lookbehind: true, group: 'span' }, { cls: 'tk-var-builtin', pattern: /\$[_0-9&`'+^!]|\$\^\w|\@ARGV\b|\%ENV\b|\$0\b/ }, { cls: 'tk-var', pattern: /\$#?[A-Za-z_]\w*|[@%][A-Za-z_]\w*|\$\{[^}]+\}/ }, { cls: 'tk-fn-decl', pattern: /(\bsub\s+)[A-Za-z_]\w*/, lookbehind: true }, @@ -1061,14 +1196,15 @@ ).split(' '); G.powershell = [ - // Block comments before strings; line comments after strings so - // "not # a comment" stays a string. - { cls: 'tk-comment', pattern: /<#[\s\S]*?#>/ }, - { cls: 'tk-string', pattern: /"(?:`.|\$\w+|\$\{[^}]*\}|[^"`$\n])*"/, inside: [ + // Comments and strings compete by position (see `group` in tokenize), so + // "not # a comment" stays a string. The block comment is listed ahead of + // the line comment so `<#` wins the tie over the `#` inside it. + { cls: 'tk-comment', pattern: /<#[\s\S]*?#>/, group: 'span' }, + { cls: 'tk-string', pattern: /"(?:`.|\$\w+|\$\{[^}]*\}|[^"`$\n])*"/, group: 'span', inside: [ { cls: 'tk-var-builtin', pattern: /\$\{[^}]+\}|\$\w+/ }, ]}, - { cls: 'tk-string', pattern: /'[^'\n]*'/ }, - { cls: 'tk-comment', pattern: /#.*/ }, + { cls: 'tk-string', pattern: /'[^'\n]*'/, group: 'span' }, + { cls: 'tk-comment', pattern: /#.*/, group: 'span' }, { cls: 'tk-decorator', pattern: /\[[A-Za-z][\w.]*(?:\(\)|\[\])?\]/ }, { cls: 'tk-var-builtin', pattern: /\$(?:_|PSItem|PSScriptRoot|PSCommandPath|args|input|this|null|true|false|error|home|host|profile|pid|pwd)\b|\$env:\w+/i }, { cls: 'tk-var', pattern: /\$\{[^}]+\}|\$\w+/ }, @@ -1093,22 +1229,23 @@ ).split(' '); G.elixir = [ - { cls: 'tk-doc', pattern: /@(?:moduledoc|doc)\s+"""[\s\S]*?"""/ }, - // Strings before comments, else # inside "..." (incl. #{} interpolation) is eaten + // Doc attributes, strings, sigils and comments compete by position (see + // `group` in tokenize): `#` inside "..." (incl. #{} interpolation) stays + // text, and a quote or a `~r/…/` inside a comment stays comment. + { cls: 'tk-doc', pattern: /@(?:moduledoc|doc)\s+"""[\s\S]*?"""/, group: 'span' }, // Fallback and interpolation branch kept disjoint (see the JS template-string note). - { cls: 'tk-string', pattern: /"""[\s\S]*?"""|"(?:\\.|#\{[^}\n]*\}|#(?!\{)|[^"\\\n#])*"/, inside: [ + { cls: 'tk-string', pattern: /"""[\s\S]*?"""|"(?:\\.|#\{[^}\n]*\}|#(?!\{)|[^"\\\n#])*"/, group: 'span', inside: [ { cls: 'tk-operator', pattern: /#\{[^}\n]*\}/ }, ]}, - { cls: 'tk-string', pattern: /'(?:\\.|[^'\\\n])*'/ }, + { cls: 'tk-string', pattern: /'(?:\\.|[^'\\\n])*'/, group: 'span' }, - // Sigils ~s{…} ~w[…] ~r/…/, ahead of the comment rule for the same reason - // the strings are. A `"` delimiter is not accepted here: it would end - // `~s"""…"""` at the second quote, and the triple-quote rule above - // already renders that form correctly. - { cls: 'tk-regex', pattern: /~[rR]([([{<|\/'])/, close: pairedEnd(1) }, - { cls: 'tk-string', pattern: /~[a-zA-Z]([([{<|\/'])/, close: pairedEnd(1) }, + // Sigils ~s{…} ~w[…] ~r/…/. A `"` delimiter is not accepted here: it + // would end `~s"""…"""` at the second quote, and the triple-quote rule + // above already renders that form correctly. + { cls: 'tk-regex', pattern: /~[rR]([([{<|\/'])/, close: pairedEnd(1), group: 'span' }, + { cls: 'tk-string', pattern: /~[a-zA-Z]([([{<|\/'])/, close: pairedEnd(1), group: 'span' }, - { cls: 'tk-comment', pattern: /#.*/ }, + { cls: 'tk-comment', pattern: /#.*/, group: 'span' }, { cls: 'tk-decorator', pattern: /@[a-z_]\w*/ }, { cls: 'tk-var-const', pattern: /:[a-z_]\w*[?!]?/ }, { cls: 'tk-fn-decl', pattern: /(\b(?:defp?|defmacrop?|defguard|defdelegate)\s+)[a-z_]\w*[?!]?/, lookbehind: true }, @@ -1134,11 +1271,11 @@ ).split(' '); G.haskell = [ - // Block comments before strings; line comments after strings so - // "not -- a comment" stays a string. - { cls: 'tk-comment', pattern: /\{-[\s\S]*?-\}/ }, - { cls: 'tk-string', pattern: /"(?:\\.|[^"\\\n])*"/ }, - { cls: 'tk-comment', pattern: /--.*/ }, + // Comments and strings compete by position (see `group` in tokenize), so + // "not -- a comment" stays a string and a quote in a comment stays comment. + { cls: 'tk-comment', pattern: /\{-[\s\S]*?-\}/, group: 'span' }, + { cls: 'tk-string', pattern: /"(?:\\.|[^"\\\n])*"/, group: 'span' }, + { cls: 'tk-comment', pattern: /--.*/, group: 'span' }, { cls: 'tk-fn-decl', pattern: /^[a-z_][\w']*(?=\s*::)/m }, { cls: 'tk-keyword', pattern: wordPattern(HS_KEYWORDS) }, { cls: 'tk-type', pattern: /\b[A-Z][\w']*/ }, @@ -1152,9 +1289,9 @@ // GraphQL // ============================================================ G.graphql = [ - // Strings before comments so "not # a comment" stays a string. - { cls: 'tk-string', pattern: /"""[\s\S]*?"""|"(?:\\.|[^"\\\n])*"/ }, - { cls: 'tk-comment', pattern: /#.*/ }, + // Strings and comments compete by position (see `group` in tokenize). + { cls: 'tk-string', pattern: /"""[\s\S]*?"""|"(?:\\.|[^"\\\n])*"/, group: 'span' }, + { cls: 'tk-comment', pattern: /#.*/, group: 'span' }, { cls: 'tk-decorator', pattern: /@[A-Za-z_]\w*/ }, { cls: 'tk-var-param', pattern: /\$[A-Za-z_]\w*/ }, { cls: 'tk-keyword', pattern: /\b(?:query|mutation|subscription|fragment|on|type|interface|union|enum|input|scalar|schema|directive|extend|implements|repeatable|true|false|null)\b/ }, @@ -1168,11 +1305,11 @@ // TOML / INI // ============================================================ G.toml = [ - // Table headers first (they may contain quoted keys), then strings before - // comments, else # inside "..." is eaten as a comment + // Table headers first (they may contain quoted keys), then strings and + // comments competing by position (see `group` in tokenize). { cls: 'tk-tag', pattern: /^[ \t]*\[\[?[^\]\n]+\]\]?/m }, - { cls: 'tk-string', pattern: /"""[\s\S]*?"""|'''[\s\S]*?'''|"(?:\\.|[^"\\\n])*"|'[^'\n]*'/ }, - { cls: 'tk-comment', pattern: /#.*/ }, + { cls: 'tk-string', pattern: /"""[\s\S]*?"""|'''[\s\S]*?'''|"(?:\\.|[^"\\\n])*"|'[^'\n]*'/, group: 'span' }, + { cls: 'tk-comment', pattern: /#.*/, group: 'span' }, { cls: 'tk-type', pattern: /^(\s*)[A-Za-z0-9_.-]+(?=\s*=)/m, lookbehind: true }, { cls: 'tk-keyword', pattern: /\b(?:true|false)\b/ }, { cls: 'tk-number', pattern: /\d{4}-\d{2}-\d{2}(?:[T ][\d:.]+(?:Z|[+-]\d{2}:\d{2})?)?|[+-]?\b(?:0[xX][\da-fA-F_]+|0[oO][0-7_]+|0[bB][01_]+|\d[\d_]*(?:\.\d[\d_]*)?(?:[eE][+-]?\d+)?|inf|nan)\b/ }, @@ -1194,9 +1331,9 @@ // Dockerfile // ============================================================ G.dockerfile = [ - // Strings before comments so "not # a comment" stays a string. - { cls: 'tk-string', pattern: /"(?:\\.|[^"\\\n])*"|'[^'\n]*'/ }, - { cls: 'tk-comment', pattern: /#.*/ }, + // Strings and comments compete by position (see `group` in tokenize). + { cls: 'tk-string', pattern: /"(?:\\.|[^"\\\n])*"|'[^'\n]*'/, group: 'span' }, + { cls: 'tk-comment', pattern: /#.*/, group: 'span' }, { cls: 'tk-keyword', pattern: /^\s*(?:FROM|RUN|CMD|LABEL|MAINTAINER|EXPOSE|ENV|ADD|COPY|ENTRYPOINT|VOLUME|USER|WORKDIR|ARG|ONBUILD|STOPSIGNAL|HEALTHCHECK|SHELL)\b|\bAS\b/m }, { cls: 'tk-var-builtin', pattern: /\$\{[^}]+\}|\$\w+/ }, { cls: 'tk-decorator', pattern: /(^|\s)--[\w-]+(?==|\s|$)/, lookbehind: true }, diff --git a/docs/development.md b/docs/development.md index 01a4aba..86f4228 100644 --- a/docs/development.md +++ b/docs/development.md @@ -72,9 +72,11 @@ code string ──tokenize(code, rules)──▶ token stream ──renderer─ `tokenize`, `render`, `applyTheme`, `detectLanguage`, `normalizeLanguage`, `languages`. UMD-ish export: CommonJS `module.exports` + `global.JSRay`. -**Grammar rule ordering matters.** Strings must be matched before comments -(a `#` or `//` inside a string must not start a comment), declaration rules -before keyword rules (otherwise `function`/`def` consume the name). +**Grammar rule ordering matters — except among spans.** Comments, strings and +other span openers share `group: 'span'` and compete by position, so a `#` or +`//` inside a string stays text and a quote inside a comment stays comment; no +fixed order gets both right. Everywhere else order is priority: declaration +rules before keyword rules (otherwise `function`/`def` consume the name). See CONTRIBUTING.md for the full checklist. --- diff --git a/docs/development.zh-CN.md b/docs/development.zh-CN.md index 4b7dde0..5bb3ccc 100644 --- a/docs/development.zh-CN.md +++ b/docs/development.zh-CN.md @@ -41,7 +41,7 @@ JSRay 生态的工程参考:全部官方仓库的架构、契约、约定与工 5. **主题运行时** —— `applyTheme(themeBlock, root)` 写入 `--jr-*` CSS 变量。默认目标是携带 `data-theme` 的元素(通常是 ``):主题样式表通过 `[data-theme]` 选择器把同名变量定义在那里,写到祖先节点的内联变量会被遮蔽。 6. **公开 API** —— `highlight`、`highlightElement`、`highlightAll`、`tokenize`、`render`、`applyTheme`、`detectLanguage`、`normalizeLanguage`、`languages`。UMD 式导出:CommonJS `module.exports` + `global.JSRay`。 -**语法规则顺序至关重要。** 字符串必须先于注释匹配(字符串里的 `#` 或 `//` 不能触发注释),声明规则先于关键字规则(否则 `function`/`def` 会吃掉声明名)。完整清单见 CONTRIBUTING.md。 +**语法规则顺序至关重要 —— 跨度类规则除外。** 注释、字符串以及其它会开启一段跨度的规则共用 `group: 'span'`,按起始位置竞争:字符串里的 `#` 或 `//` 保持为文本,注释里的引号保持为注释 —— 任何固定顺序都无法两边都对。其余规则仍按顺序定优先级:声明规则先于关键字规则(否则 `function`/`def` 会吃掉声明名)。完整清单见 CONTRIBUTING.md。 --- diff --git a/integrity.json b/integrity.json index 69b7a31..00e6074 100644 --- a/integrity.json +++ b/integrity.json @@ -4,7 +4,7 @@ "algorithm": "sha256", "note": "Base64 SHA-256 digests of the released Core assets. Integrations copy the relevant digest at sync time and verify their bundled snapshot against it.", "files": { - "dist/jsray.js": "sha256-ZU8tfz1KrOvaewDqhJC9D5rzbItkXfreHJF+zkQ7EUU=", + "dist/jsray.js": "sha256-/MfXR8QVmR8xPViGqX6OPy2R3T5JHmwhx0OY11LL/uk=", "dist/jsray.css": "sha256-6Fuva+aZwCcmltNi2oN+1yWIJoKE+1rLpUxZvD9iutc=", "dist/themes/aurora.css": "sha256-S8X+R8XZNC8WRdHOen839zMkPxVFol6PPTrpZibbeJA=", "dist/themes/default.css": "sha256-vENOigRjwn1hW6wPjdM4tbZ58Ja44FMsa+akavPiMSI=", diff --git a/src/jsray.js b/src/jsray.js index d9edea5..9ceecb2 100644 --- a/src/jsray.js +++ b/src/jsray.js @@ -18,13 +18,15 @@ const escapeHtml = (s) => s.replace(/[&<>"]/g, (c) => escapeMap[c]); /** - * Apply grammar rules to a string in order. Rule array order = priority (first-match wins). + * Apply grammar rules to a string in order. Rule array order = priority + * (first-match wins), except among adjacent rules sharing a group. * Each rule: * { cls: 'tk-xxx', * pattern: /re/, // must be globalizable; the 'g' flag is forced internally * inside?: rules, // nested grammar (recursively tokenize captured text) * lookbehind?: true, // capture group 1 is consumed as prefix but not colored - * close?: fn } // pattern matches the opening only; fn finds the end + * close?: fn, // pattern matches the opening only; fn finds the end + * group?: 'name' } // adjacent rules sharing a name compete by position * * `close(match, text, from) -> index | -1` exists for the forms whose end is * not knowable when the rule is written. A heredoc ends at the word its own @@ -34,47 +36,130 @@ * Returning -1 means "no terminator here": the opening is left to the rules * behind this one rather than swallowing the rest of the file, because a * false opening is likelier than a genuinely unterminated literal. + * + * `group` exists because order cannot decide between a string and a + * comment. Strings first reads `// don't stop, won't stop` as a comment + * holding the string `'t stop, won'`; comments first reads + * `"https://jsray.org"` as a string holding a comment. Every grammar here had + * picked one of those two failures and written the choice down beside its + * rules as though it were the fix. The language decides by position — + * whichever opens first owns the text up to its own end — so the rules of a + * group run as one pass in which the earliest match wins at every step, and + * listed order only breaks a tie at the same index (which is how `/**` still + * beats `/*`). Positions are compared where the whole match begins, lookbehind + * prefix included: a heredoc's body starts on the next line, and measured + * from there `cat < "out.txt"` would lose its `<<` to the quoted name. */ function tokenize(code, rules) { let stream = [code]; - for (const rule of rules) { - const next = []; - // Compile once per rule and cache on the rule object — the old code - // built a fresh RegExp per stream piece, which dominated tokenize time - // on fragmented streams. lastIndex is reset per piece instead. - const re = rule._re || (rule._re = new RegExp( - rule.pattern.source, - (rule.pattern.flags || '').replace('g', '') + 'g' - )); - for (const piece of stream) { - if (typeof piece !== 'string') { next.push(piece); continue; } - re.lastIndex = 0; - let last = 0, m; - while ((m = re.exec(piece)) !== null) { - const lbLen = rule.lookbehind && m[1] ? m[1].length : 0; - const start = m.index + lbLen; - let text = m[0].slice(lbLen); - if (rule.close) { - const end = rule.close(m, piece, m.index + m[0].length); - if (end < 0) { re.lastIndex = m.index + 1; continue; } - text = piece.slice(start, end); + for (let r = 0; r < rules.length; ) { + const rule = rules[r]; + if (!rule.group) { + stream = applyRule(stream, rule); + r++; + continue; + } + let end = r + 1; + while (end < rules.length && rules[end].group === rule.group) end++; + stream = applyGroup(stream, rules.slice(r, end)); + r = end; + } + return stream; + } + + function compile(rule) { + // Compile once per rule and cache on the rule object — the old code + // built a fresh RegExp per stream piece, which dominated tokenize time + // on fragmented streams. lastIndex is reset per piece instead. + return rule._re || (rule._re = new RegExp( + rule.pattern.source, + (rule.pattern.flags || '').replace('g', '') + 'g' + )); + } + + function applyRule(stream, rule) { + const next = []; + const re = compile(rule); + for (const piece of stream) { + if (typeof piece !== 'string') { next.push(piece); continue; } + re.lastIndex = 0; + let last = 0, m; + while ((m = re.exec(piece)) !== null) { + const lbLen = rule.lookbehind && m[1] ? m[1].length : 0; + const start = m.index + lbLen; + let text = m[0].slice(lbLen); + if (rule.close) { + const end = rule.close(m, piece, m.index + m[0].length); + if (end < 0) { re.lastIndex = m.index + 1; continue; } + text = piece.slice(start, end); + } + if (!text) { re.lastIndex++; continue; } + if (start > last) next.push(piece.slice(last, start)); + next.push({ + type: rule.cls, + content: rule.inside ? tokenize(text, rule.inside) : text, + }); + last = start + text.length; + // The body of a close-delimited form has already been consumed; + // resuming inside it would re-match its own contents. + if (rule.close) re.lastIndex = last; + } + if (last < piece.length) next.push(piece.slice(last)); + } + return next; + } + + /** The first acceptable match of `rule` at or after `from`, or null. */ + function locate(rule, piece, from) { + const re = compile(rule); + re.lastIndex = from; + let m; + while ((m = re.exec(piece)) !== null) { + const lbLen = rule.lookbehind && m[1] ? m[1].length : 0; + const start = m.index + lbLen; + let text = m[0].slice(lbLen); + if (rule.close) { + const end = rule.close(m, piece, m.index + m[0].length); + if (end < 0) { re.lastIndex = m.index + 1; continue; } + text = piece.slice(start, end); + } + if (!text) { re.lastIndex = m.index + 1; continue; } + return { index: m.index, start, text }; + } + return null; + } + + function applyGroup(stream, group) { + const next = []; + for (const piece of stream) { + if (typeof piece !== 'string') { next.push(piece); continue; } + // One cursor per rule, kept until the text it points into has been + // claimed by another rule. Cursors only move forward, so a pass costs + // about what running the same rules one after another did. + const hits = new Array(group.length); + let pos = 0; + for (;;) { + let win = -1; + for (let i = 0; i < group.length; i++) { + const hit = hits[i]; + if (hit === undefined || (hit !== null && hit.index < pos)) { + hits[i] = locate(group[i], piece, pos); } - if (!text) { re.lastIndex++; continue; } - if (start > last) next.push(piece.slice(last, start)); - next.push({ - type: rule.cls, - content: rule.inside ? tokenize(text, rule.inside) : text, - }); - last = start + text.length; - // The body of a close-delimited form has already been consumed; - // resuming inside it would re-match its own contents. - if (rule.close) re.lastIndex = last; + if (hits[i] && (win < 0 || hits[i].index < hits[win].index)) win = i; } - if (last < piece.length) next.push(piece.slice(last)); + if (win < 0) break; + const { start, text } = hits[win]; + const rule = group[win]; + if (start > pos) next.push(piece.slice(pos, start)); + next.push({ + type: rule.cls, + content: rule.inside ? tokenize(text, rule.inside) : text, + }); + pos = start + text.length; } - stream = next; + if (pos < piece.length) next.push(piece.slice(pos)); } - return stream; + return next; } function render(stream) { @@ -183,30 +268,41 @@ ).split(' '); // Parameter-list sub-grammar · colors n / name in (n: number, name = "x") as var-param + // + // A parameter list is claimed where it opens, ahead of any comment or string + // default inside it, so this is the first grammar to see those. They are + // taken before punctuation can split them at a comma. const jsParamInside = [ + { cls: 'tk-comment', pattern: /\/\*[\s\S]*?\*\/|\/\/.*/, group: 'span' }, + { cls: 'tk-string', pattern: /"(?:\\.|[^"\\])*"|'(?:\\.|[^'\\])*'/, group: 'span' }, { cls: 'tk-punct', pattern: /[(),]/ }, { cls: 'tk-type', pattern: /(:\s*)[A-Z][\w$]*/, lookbehind: true }, { cls: 'tk-keyword', pattern: /\b(?:number|string|boolean|void|null|undefined|any|unknown|never|true|false)\b/ }, - { cls: 'tk-string', pattern: /"(?:\\.|[^"\\])*"|'(?:\\.|[^'\\])*'/ }, { cls: 'tk-number', pattern: /\b\d[\d_]*(?:\.\d+)?\b/ }, { cls: 'tk-var-param', pattern: /\b[A-Za-z_$][\w$]*\b/ }, { cls: 'tk-operator', pattern: /[=?:.]/ }, ]; G.javascript = [ - // Doc comments must come before plain block comments. Block comments stay - // before strings; LINE comments come after strings (see below) so - // "https://..." inside a string never becomes a comment. - { cls: 'tk-doc', pattern: /\/\*\*[\s\S]*?\*\// }, - { cls: 'tk-comment', pattern: /\/\*[\s\S]*?\*\// }, + // Everything that opens a span — doc and block comments, parameter lists, + // strings, line comments, regex literals — competes by position (see + // `group` in tokenize). Listed order only breaks ties at one index: `/**` + // before `/*`, and `//` before a regex, which cannot begin with `//`. + // + // A parameter list is a span too. Ahead of the comments it would read + // `function f(a, b)` inside a doc comment as a real signature; behind the + // strings it could not match `(a = "x")` once the default was split off. + // Competing, it is claimed only where it opens first. + { cls: 'tk-doc', pattern: /\/\*\*[\s\S]*?\*\//, group: 'span' }, + { cls: 'tk-comment', pattern: /\/\*[\s\S]*?\*\//, group: 'span' }, // Parameter lists · function foo(...) / (...) => / async (...) => { cls: 'tk-scope', pattern: /(\bfunction\s*[\w$]*\s*)\([^()]*\)/, lookbehind: true, - inside: jsParamInside }, + inside: jsParamInside, group: 'span' }, { cls: 'tk-scope', pattern: /\([^()]*\)(?=\s*=>)/, - inside: jsParamInside }, + inside: jsParamInside, group: 'span' }, // Template strings (with inline ${...}) // @@ -216,22 +312,24 @@ // engine try every combination: 26 placeholders took 8.7s to fail. Every // interpolating grammar below keeps its fallback and its interpolation // branch disjoint for the same reason. - { cls: 'tk-string', pattern: /`(?:\\.|\$\{[^}]*\}|\$(?!\{)|[^`\\$])*`/, inside: [ + { cls: 'tk-string', pattern: /`(?:\\.|\$\{[^}]*\}|\$(?!\{)|[^`\\$])*`/, group: 'span', inside: [ { cls: 'tk-operator', pattern: /\$\{[^}]*\}/, inside: [ { cls: 'tk-punct', pattern: /^\$\{|\}$/ }, // No JS recursion here; minimal coloring avoids rule cross-talk { cls: 'tk-var', pattern: /[A-Za-z_$][\w$]*/ }, ]}, ]}, - { cls: 'tk-string', pattern: RX.string1 }, - { cls: 'tk-string', pattern: RX.string2 }, - { cls: 'tk-comment', pattern: /\/\/.*/ }, + { cls: 'tk-string', pattern: RX.string1, group: 'span' }, + { cls: 'tk-string', pattern: RX.string2, group: 'span' }, + { cls: 'tk-comment', pattern: /\/\/.*/, group: 'span' }, // Regex: only recognize after =/(/,/!/keyword to avoid eating division // Capture prefix whitespace as lookbehind so it stays outside the regex token. + // In the span group because a regex may hold a quote: `s.split(/"/)` read + // as the start of a string took the rest of the line with it. { cls: 'tk-regex', pattern: /(^|[=(,!&|?:;{}\[\]]\s*|\breturn\s*)\/(?![*\/])(?:\\.|\[(?:\\.|[^\]\\\n])*\]|[^\/\\\n])+\/[gimsuy]*/, - lookbehind: true }, + lookbehind: true, group: 'span' }, // Decorators { cls: 'tk-decorator', pattern: /@[A-Za-z_$][\w$]*/ }, @@ -305,9 +403,11 @@ ]; G.python = [ - // Triple-quoted strings (with f/r/b prefixes). All strings before - // comments so # inside "..." never becomes a comment. - { cls: 'tk-string', pattern: /(?:[rRbBuUfF]{0,2})("""[\s\S]*?"""|'''[\s\S]*?''')/ }, + // Strings and comments compete by position (see `group` in tokenize), so + // `#` inside "..." stays text and a quote inside a comment stays comment. + // Triple-quoted strings (with f/r/b prefixes) are listed first: at the + // same index they must win over `""` followed by a lone quote. + { cls: 'tk-string', pattern: /(?:[rRbBuUfF]{0,2})("""[\s\S]*?"""|'''[\s\S]*?''')/, group: 'span' }, // PEP 701 lets a replacement field carry the same quote that delimits the // string: `f"{a["k"]}"` is valid from Python 3.12. The general rule below // stops at the first inner quote, which split one string into two tokens @@ -320,9 +420,9 @@ // is deliberately left to fall through to the general rule: covering it // needs a nested quantifier, which is the ambiguous shape that caused the // beta.4 denial of service. - { cls: 'tk-string', pattern: /(?:[rRbB][fF]|[fF][rRbB]?)("(?:\{[^{}]*\}|\\.|[^"\\\n{])*"|'(?:\{[^{}]*\}|\\.|[^'\\\n{])*')/ }, - { cls: 'tk-string', pattern: /(?:[rRbBuUfF]{0,2})("(?:\\.|[^"\\\n])*"|'(?:\\.|[^'\\\n])*')/ }, - { cls: 'tk-comment', pattern: /#.*/ }, + { cls: 'tk-string', pattern: /(?:[rRbB][fF]|[fF][rRbB]?)("(?:\{[^{}]*\}|\\.|[^"\\\n{])*"|'(?:\{[^{}]*\}|\\.|[^'\\\n{])*')/, group: 'span' }, + { cls: 'tk-string', pattern: /(?:[rRbBuUfF]{0,2})("(?:\\.|[^"\\\n])*"|'(?:\\.|[^'\\\n])*')/, group: 'span' }, + { cls: 'tk-comment', pattern: /#.*/, group: 'span' }, // Parameter list · def foo(self, n: int = 0) { cls: 'tk-scope', @@ -440,8 +540,13 @@ { cls: 'tk-punct', pattern: /[{}[\]:,]/ }, ]; G.jsonc = [ - { cls: 'tk-comment', pattern: /\/\*[\s\S]*?\*\/|\/\/.*/ }, - ...G.json, + // Comments and strings compete by position (see `group` in tokenize). The + // old order put comments first, so a value such as "https://jsray.org" — + // everywhere in editor settings and tsconfig files — was cut at its `//`. + { cls: 'tk-comment', pattern: /\/\*[\s\S]*?\*\/|\/\/.*/, group: 'span' }, + { cls: 'tk-type', pattern: /"(?:\\.|[^"\\])*"(?=\s*:)/, group: 'span' }, + { cls: 'tk-string', pattern: /"(?:\\.|[^"\\])*"/, group: 'span' }, + ...G.json.filter((rule) => rule.cls !== 'tk-type' && rule.cls !== 'tk-string'), ]; // ============================================================ @@ -465,17 +570,20 @@ { cls: 'tk-string', pattern: /(<<(-?)[ \t]*(['"]?)([A-Za-z_]\w*)\3[^\n]*\n)/, lookbehind: true, - close: heredocEnd(4, 2) }, + close: heredocEnd(4, 2), + group: 'span' }, - // Strings before comments, else # inside "..." is eaten as a comment. + // Strings, heredocs and comments compete by position (see `group` in + // tokenize): `#` inside "..." stays text, a quote in a comment stays + // comment, and `# cat </i }, - { cls: 'tk-string', pattern: /"(?:\\.|\$[A-Za-z_]\w*|[^"\\$\n])*"/, inside: [ + close: heredocEnd(3, true, '\\b'), + group: 'span' }, + + // Comments and strings compete by position (see `group` in tokenize), so + // "https://..." and "#anchor" stay strings, "/* x */" stays a string, and + // a quote inside a comment stays comment. + { cls: 'tk-comment', pattern: /\/\*[\s\S]*?\*\//, group: 'span' }, + { cls: 'tk-string', pattern: /"(?:\\.|\$[A-Za-z_]\w*|[^"\\$\n])*"/, group: 'span', inside: [ { cls: 'tk-var', pattern: /\$[A-Za-z_]\w*/ }, ]}, - { cls: 'tk-string', pattern: /'(?:\\.|[^'\\\n])*'/ }, - { cls: 'tk-comment', pattern: /\/\/.*|#.*/ }, + { cls: 'tk-string', pattern: /'(?:\\.|[^'\\\n])*'/, group: 'span' }, + { cls: 'tk-comment', pattern: /\/\/.*|#.*/, group: 'span' }, + // Open and close tags after the spans, so "?>" inside a string is text. + { cls: 'tk-decorator', pattern: /<\?(?:php|=)?|\?>/i }, { cls: 'tk-var', pattern: /\$[A-Za-z_]\w*/ }, { cls: 'tk-var-const', pattern: /\b[A-Z][A-Z0-9_]{2,}\b/ }, { cls: 'tk-type', pattern: /(\b(?:class|interface|trait|enum|extends|implements|new)\s+)[A-Za-z_]\w*/, lookbehind: true }, @@ -556,16 +667,23 @@ function cLikeGrammar(keywords, builtins, options) { const opts = options || {}; const rules = [ - // Block comments stay before strings (license headers quote freely); - // LINE comments come after strings so "https://..." never becomes a + // Comments, strings and preprocessor lines compete by position (see + // `group` in tokenize): a license header can quote freely, a string can + // hold `/* … */` or `https://`, and a comment can hold `don't` twice. + // The option blocks below splice their literals in among these, in the + // same group. + // + // A preprocessor line is a span because it runs to the end of its line: + // `#include "a.h"` keeps its path, and a commented-out `#define` stays a // comment. - { cls: 'tk-doc', pattern: /\/\*\*[\s\S]*?\*\// }, - { cls: 'tk-comment', pattern: /\/\*[\s\S]*?\*\// }, - { cls: 'tk-decorator', pattern: /^\s*#\s*[A-Za-z_]\w*.*/m }, + { cls: 'tk-doc', pattern: /\/\*\*[\s\S]*?\*\//, group: 'span' }, + { cls: 'tk-comment', pattern: /\/\*[\s\S]*?\*\//, group: 'span' }, + { cls: 'tk-decorator', pattern: /^\s*#\s*[A-Za-z_]\w*.*/m, group: 'span' }, + { cls: 'tk-string', pattern: /"(?:\\.|[^"\\\n])*"/, group: 'span' }, + { cls: 'tk-string', pattern: /'(?:\\.|[^'\\\n])*'/, group: 'span' }, + { cls: 'tk-comment', pattern: /\/\/.*/, group: 'span' }, + // Annotations come after the spans, so `// see @Override` stays a comment. { cls: 'tk-decorator', pattern: /@[A-Za-z_]\w*/ }, - { cls: 'tk-string', pattern: /"(?:\\.|[^"\\\n])*"/ }, - { cls: 'tk-string', pattern: /'(?:\\.|[^'\\\n])*'/ }, - { cls: 'tk-comment', pattern: /\/\/.*/ }, { cls: 'tk-var-const', pattern: /\b[A-Z][A-Z0-9_]{2,}\b/ }, { cls: 'tk-type', pattern: /(\b(?:class|struct|interface|enum|trait|extends|implements|namespace|using|new|object|protocol|extension|mixin|record|actor)\s+)[A-Za-z_]\w*/, lookbehind: true }, { cls: 'tk-fn-decl', pattern: new RegExp('\\b(?!(?:' + CLIKE_DECL_SKIP + ')\\b)[A-Za-z_]\\w*(?=\\s*\\([^;{}]*\\)\\s*(?:const\\s*)?(?:->\\s*[A-Za-z_:][\\w:<>]*)?\\{)') }, @@ -604,6 +722,7 @@ rules.splice(rules.findIndex((r) => r.cls === 'tk-string'), 0, { cls: 'tk-string', pattern: /`[^`]*`/, + group: 'span', }); } @@ -623,6 +742,7 @@ rules.splice(rules.findIndex((r) => r.cls === 'tk-string'), 0, { cls: 'tk-string', pattern: /\b(?:[uU]8?|[LU])?R"([^\s()\\]{0,16})\([\s\S]*?\)\1"/, + group: 'span', }); } @@ -630,6 +750,7 @@ rules.splice(rules.findIndex((r) => r.cls === 'tk-string'), 0, { cls: 'tk-string', pattern: /\b(?:b?r)(#{0,16})"[\s\S]*?"\1/, + group: 'span', }); } @@ -642,6 +763,7 @@ rules.splice(rules.findIndex((r) => r.cls === 'tk-string'), 0, { cls: 'tk-string', pattern: /"""[\s\S]*?"""|'''[\s\S]*?'''/, + group: 'span', }); } @@ -654,8 +776,8 @@ // preceded: leaving it in place would keep one pattern around that can // still span from any apostrophe to any later one. if (opts.lifetimes) { - // The character literal stays where the string rule was, because it is a - // string and line comments deliberately come after strings. + // The character literal takes the string rule's place in the span group, + // because it is a string. const quoted = rules.findIndex( (r) => r.cls === 'tk-string' && r.pattern.source[0] === "'" ); @@ -663,6 +785,7 @@ // Exactly one character or one escape, then the closing quote. cls: 'tk-string', pattern: /'(?:\\(?:u\{[\da-fA-F]{1,6}\}|.)|[^'\\\n])'/, + group: 'span', }); // The lifetime goes AFTER the line-comment rule, and the distance between @@ -830,14 +953,14 @@ { cls: 'tk-string', pattern: /(<<([-~]?)(['"]?)([A-Z_]\w*)\3[^\n]*\n)/, lookbehind: true, - close: heredocEnd(4, 2) }, + close: heredocEnd(4, 2), + group: 'span' }, - // `=begin` / `=end` blocks come before the strings that come before - // everything else. The markers are only special at column zero, so the - // anchors here are load-bearing rather than decorative. Without this rule - // a documentation block was read as ordinary code — the body's words came - // out coloured as function calls and keywords. - { cls: 'tk-comment', pattern: /^=begin\b[\s\S]*?^=end.*$/m }, + // `=begin` / `=end` blocks. The markers are only special at column zero, + // so the anchors here are load-bearing rather than decorative. Without + // this rule a documentation block was read as ordinary code — the body's + // words came out coloured as function calls and keywords. + { cls: 'tk-comment', pattern: /^=begin\b[\s\S]*?^=end.*$/m, group: 'span' }, // %w[…] %i(…) %q{…} %Q<…>: the delimiter is picked at the call site, so // the closer is only knowable once the opener has been read, and bracket @@ -845,19 +968,24 @@ // out on purpose — it cannot be told from the modulo operator without // parsing, and it is rare enough not to be worth mistaking `a %(b)` for a // literal. - { cls: 'tk-regex', pattern: /%r([([{<|!\/])/, close: pairedEnd(1) }, - { cls: 'tk-string', pattern: /%[wWiIqQsx]([([{<|!\/])/, close: pairedEnd(1) }, - - // Strings must come before comments, else # inside "..." (incl. #{} interpolation) - // is eaten as a comment. String bodies stay single-line so an unpaired quote - // in a comment can't swallow following lines. + // + // These sat ahead of the comment rule in beta.4, which is how + // `# prefer %w[a b]` lost the rest of its comment. In the span group they + // are claimed only where they open before a `#` does. + { cls: 'tk-regex', pattern: /%r([([{<|!\/])/, close: pairedEnd(1), group: 'span' }, + { cls: 'tk-string', pattern: /%[wWiIqQsx]([([{<|!\/])/, close: pairedEnd(1), group: 'span' }, + + // Strings and comments compete by position (see `group` in tokenize), so + // `#` inside "..." (incl. #{} interpolation) stays text and a quote inside + // a comment stays comment. String bodies stay single-line so an unpaired + // quote can't swallow following lines. // Fallback excludes `#`; a bare `#` is admitted only when no `{` follows, // so `#{...}` has exactly one parse (see the JS template-string note). - { cls: 'tk-string', pattern: /"(?:\\.|#\{[^}\n]*\}|#(?!\{)|[^"\\\n#])*"/, inside: [ + { cls: 'tk-string', pattern: /"(?:\\.|#\{[^}\n]*\}|#(?!\{)|[^"\\\n#])*"/, group: 'span', inside: [ { cls: 'tk-operator', pattern: /#\{[^}\n]*\}/ }, ]}, - { cls: 'tk-string', pattern: /'(?:\\.|[^'\\\n])*'/ }, - { cls: 'tk-comment', pattern: /#.*/ }, + { cls: 'tk-string', pattern: /'(?:\\.|[^'\\\n])*'/, group: 'span' }, + { cls: 'tk-comment', pattern: /#.*/, group: 'span' }, { cls: 'tk-var-builtin', pattern: /[@$]{1,2}[A-Za-z_]\w*|\bself\b/ }, { cls: 'tk-var-const', pattern: /\b[A-Z][A-Z0-9_]{2,}\b/ }, { cls: 'tk-type', pattern: /(\b(?:class|module)\s+)[A-Z]\w*/, lookbehind: true }, @@ -884,13 +1012,14 @@ ).split(' '); G.lua = [ - // Block comments before strings; line comments after strings so - // "not -- a comment" stays a string. - { cls: 'tk-comment', pattern: /--\[\[[\s\S]*?\]\]/ }, - { cls: 'tk-string', pattern: /\[\[[\s\S]*?\]\]/ }, - { cls: 'tk-string', pattern: /"(?:\\.|[^"\\\n])*"/ }, - { cls: 'tk-string', pattern: /'(?:\\.|[^'\\\n])*'/ }, - { cls: 'tk-comment', pattern: /--.*/ }, + // Comments and strings compete by position (see `group` in tokenize), so + // "not -- a comment" stays a string and `-- don't` stays a comment. The + // long comment is listed ahead of the line comment: both open at `--`. + { cls: 'tk-comment', pattern: /--\[\[[\s\S]*?\]\]/, group: 'span' }, + { cls: 'tk-string', pattern: /\[\[[\s\S]*?\]\]/, group: 'span' }, + { cls: 'tk-string', pattern: /"(?:\\.|[^"\\\n])*"/, group: 'span' }, + { cls: 'tk-string', pattern: /'(?:\\.|[^'\\\n])*'/, group: 'span' }, + { cls: 'tk-comment', pattern: /--.*/, group: 'span' }, { cls: 'tk-var-builtin', pattern: /\b(?:self|_G|_VERSION)\b/ }, { cls: 'tk-fn-decl', pattern: /(\bfunction\s+)[A-Za-z_]\w*(?:[.:][A-Za-z_]\w*)?/, @@ -917,11 +1046,13 @@ const SQL_BUILTINS = 'count sum avg min max coalesce nullif lower upper substr substring now date'.split(' '); G.sql = [ - // Block comments before strings; line comments after strings so - // 'not -- a comment' stays a string. - { cls: 'tk-comment', pattern: /\/\*[\s\S]*?\*\// }, - { cls: 'tk-string', pattern: /'(?:''|[^'])*'|"(?:\\"|[^"])*"/ }, - { cls: 'tk-comment', pattern: /--.*/ }, + // Comments and strings compete by position (see `group` in tokenize), so + // 'not -- a comment' stays a string. This grammar's strings may span lines, + // which made the old order worse than elsewhere: `-- don't` opened a string + // that ran on until the next apostrophe anywhere below it. + { cls: 'tk-comment', pattern: /\/\*[\s\S]*?\*\//, group: 'span' }, + { cls: 'tk-string', pattern: /'(?:''|[^'])*'|"(?:\\"|[^"])*"/, group: 'span' }, + { cls: 'tk-comment', pattern: /--.*/, group: 'span' }, { cls: 'tk-keyword', pattern: new RegExp('\\b(?:' + SQL_KEYWORDS.join('|') + ')\\b', 'i') }, { cls: 'tk-fn-builtin', pattern: new RegExp('\\b(?:' + SQL_BUILTINS.join('|') + ')\\b', 'i') }, { cls: 'tk-number', pattern: /-?\b\d+(?:\.\d+)?\b/ }, @@ -940,11 +1071,14 @@ // and stopping early beats swallowing the rest of the document. { cls: 'tk-string', pattern: /(:[ \t]*)[|>][-+]?\d*[ \t]*(?:\n[ \t]+.*)*/, - lookbehind: true }, + lookbehind: true, + group: 'span' }, - // Strings before comments, else # inside "..." is eaten as a comment - { cls: 'tk-string', pattern: /"(?:\\.|[^"\\\n])*"|'(?:''|[^'\n])*'/ }, - { cls: 'tk-comment', pattern: /#.*/ }, + // Strings, block scalars and comments compete by position (see `group` in + // tokenize): `#` inside "..." stays text, and a `key: |` written inside a + // comment opens nothing. + { cls: 'tk-string', pattern: /"(?:\\.|[^"\\\n])*"|'(?:''|[^'\n])*'/, group: 'span' }, + { cls: 'tk-comment', pattern: /#.*/, group: 'span' }, { cls: 'tk-type', pattern: /^(\s*)[A-Za-z_][\w.-]*(?=\s*:)/m, lookbehind: true }, { cls: 'tk-decorator', pattern: /[&*][A-Za-z_][\w-]*/ }, { cls: 'tk-keyword', pattern: /\b(?:true|false|null|yes|no|on|off)\b/i }, @@ -997,9 +1131,9 @@ ).split(' '); G.r = [ - // Strings before comments, else # inside "..." is eaten as a comment - { cls: 'tk-string', pattern: /"(?:\\.|[^"\\\n])*"|'(?:\\.|[^'\\\n])*'/ }, - { cls: 'tk-comment', pattern: /#.*/ }, + // Strings and comments compete by position (see `group` in tokenize) + { cls: 'tk-string', pattern: /"(?:\\.|[^"\\\n])*"|'(?:\\.|[^'\\\n])*'/, group: 'span' }, + { cls: 'tk-comment', pattern: /#.*/, group: 'span' }, { cls: 'tk-fn-decl', pattern: /\b[A-Za-z._][\w.]*(?=\s*(?:<-|=)\s*function\b)/ }, { cls: 'tk-keyword', pattern: wordPattern(R_KEYWORDS) }, { cls: 'tk-fn-builtin', @@ -1025,21 +1159,22 @@ ).split(' '); G.perl = [ - { cls: 'tk-doc', pattern: /^=\w+[\s\S]*?^=cut\s*$/m }, - // Strings before comments, else # inside "..." is eaten as a comment - { cls: 'tk-string', pattern: /"(?:\\.|[^"\\\n])*"/, inside: [ + // POD, strings, quoting operators, comments and bound regexes compete by + // position (see `group` in tokenize). `#` is ordinary text inside a string, + // a `q{…}` or a `/#/`; a quote or a `q{` inside a comment is comment. + { cls: 'tk-doc', pattern: /^=\w+[\s\S]*?^=cut\s*$/m, group: 'span' }, + { cls: 'tk-string', pattern: /"(?:\\.|[^"\\\n])*"/, group: 'span', inside: [ { cls: 'tk-var', pattern: /[$@][A-Za-z_]\w*/ }, ]}, - { cls: 'tk-string', pattern: /'(?:\\.|[^'\\\n])*'/ }, + { cls: 'tk-string', pattern: /'(?:\\.|[^'\\\n])*'/, group: 'span' }, - // q{…} qq{…} qw{…} qr{…} — ahead of the comment rule, because `#` is - // ordinary text inside one. `/` is excluded as a delimiter for the - // quoting forms: after a bare word it is far more often division. - { cls: 'tk-regex', pattern: /\bqr[ \t]*([([{<|!\/])/, close: pairedEnd(1) }, - { cls: 'tk-string', pattern: /\b(?:qq|qw|q)[ \t]*([([{<|!])/, close: pairedEnd(1) }, + // q{…} qq{…} qw{…} qr{…}. `/` is excluded as a delimiter for the quoting + // forms: after a bare word it is far more often division. + { cls: 'tk-regex', pattern: /\bqr[ \t]*([([{<|!\/])/, close: pairedEnd(1), group: 'span' }, + { cls: 'tk-string', pattern: /\b(?:qq|qw|q)[ \t]*([([{<|!])/, close: pairedEnd(1), group: 'span' }, - { cls: 'tk-comment', pattern: /#.*/ }, - { cls: 'tk-regex', pattern: /((?:=~|!~)\s*)(?:m|s|tr|y)?\/(?:\\.|[^/\n])*\/[a-z]*/, lookbehind: true }, + { cls: 'tk-comment', pattern: /#.*/, group: 'span' }, + { cls: 'tk-regex', pattern: /((?:=~|!~)\s*)(?:m|s|tr|y)?\/(?:\\.|[^/\n])*\/[a-z]*/, lookbehind: true, group: 'span' }, { cls: 'tk-var-builtin', pattern: /\$[_0-9&`'+^!]|\$\^\w|\@ARGV\b|\%ENV\b|\$0\b/ }, { cls: 'tk-var', pattern: /\$#?[A-Za-z_]\w*|[@%][A-Za-z_]\w*|\$\{[^}]+\}/ }, { cls: 'tk-fn-decl', pattern: /(\bsub\s+)[A-Za-z_]\w*/, lookbehind: true }, @@ -1061,14 +1196,15 @@ ).split(' '); G.powershell = [ - // Block comments before strings; line comments after strings so - // "not # a comment" stays a string. - { cls: 'tk-comment', pattern: /<#[\s\S]*?#>/ }, - { cls: 'tk-string', pattern: /"(?:`.|\$\w+|\$\{[^}]*\}|[^"`$\n])*"/, inside: [ + // Comments and strings compete by position (see `group` in tokenize), so + // "not # a comment" stays a string. The block comment is listed ahead of + // the line comment so `<#` wins the tie over the `#` inside it. + { cls: 'tk-comment', pattern: /<#[\s\S]*?#>/, group: 'span' }, + { cls: 'tk-string', pattern: /"(?:`.|\$\w+|\$\{[^}]*\}|[^"`$\n])*"/, group: 'span', inside: [ { cls: 'tk-var-builtin', pattern: /\$\{[^}]+\}|\$\w+/ }, ]}, - { cls: 'tk-string', pattern: /'[^'\n]*'/ }, - { cls: 'tk-comment', pattern: /#.*/ }, + { cls: 'tk-string', pattern: /'[^'\n]*'/, group: 'span' }, + { cls: 'tk-comment', pattern: /#.*/, group: 'span' }, { cls: 'tk-decorator', pattern: /\[[A-Za-z][\w.]*(?:\(\)|\[\])?\]/ }, { cls: 'tk-var-builtin', pattern: /\$(?:_|PSItem|PSScriptRoot|PSCommandPath|args|input|this|null|true|false|error|home|host|profile|pid|pwd)\b|\$env:\w+/i }, { cls: 'tk-var', pattern: /\$\{[^}]+\}|\$\w+/ }, @@ -1093,22 +1229,23 @@ ).split(' '); G.elixir = [ - { cls: 'tk-doc', pattern: /@(?:moduledoc|doc)\s+"""[\s\S]*?"""/ }, - // Strings before comments, else # inside "..." (incl. #{} interpolation) is eaten + // Doc attributes, strings, sigils and comments compete by position (see + // `group` in tokenize): `#` inside "..." (incl. #{} interpolation) stays + // text, and a quote or a `~r/…/` inside a comment stays comment. + { cls: 'tk-doc', pattern: /@(?:moduledoc|doc)\s+"""[\s\S]*?"""/, group: 'span' }, // Fallback and interpolation branch kept disjoint (see the JS template-string note). - { cls: 'tk-string', pattern: /"""[\s\S]*?"""|"(?:\\.|#\{[^}\n]*\}|#(?!\{)|[^"\\\n#])*"/, inside: [ + { cls: 'tk-string', pattern: /"""[\s\S]*?"""|"(?:\\.|#\{[^}\n]*\}|#(?!\{)|[^"\\\n#])*"/, group: 'span', inside: [ { cls: 'tk-operator', pattern: /#\{[^}\n]*\}/ }, ]}, - { cls: 'tk-string', pattern: /'(?:\\.|[^'\\\n])*'/ }, + { cls: 'tk-string', pattern: /'(?:\\.|[^'\\\n])*'/, group: 'span' }, - // Sigils ~s{…} ~w[…] ~r/…/, ahead of the comment rule for the same reason - // the strings are. A `"` delimiter is not accepted here: it would end - // `~s"""…"""` at the second quote, and the triple-quote rule above - // already renders that form correctly. - { cls: 'tk-regex', pattern: /~[rR]([([{<|\/'])/, close: pairedEnd(1) }, - { cls: 'tk-string', pattern: /~[a-zA-Z]([([{<|\/'])/, close: pairedEnd(1) }, + // Sigils ~s{…} ~w[…] ~r/…/. A `"` delimiter is not accepted here: it + // would end `~s"""…"""` at the second quote, and the triple-quote rule + // above already renders that form correctly. + { cls: 'tk-regex', pattern: /~[rR]([([{<|\/'])/, close: pairedEnd(1), group: 'span' }, + { cls: 'tk-string', pattern: /~[a-zA-Z]([([{<|\/'])/, close: pairedEnd(1), group: 'span' }, - { cls: 'tk-comment', pattern: /#.*/ }, + { cls: 'tk-comment', pattern: /#.*/, group: 'span' }, { cls: 'tk-decorator', pattern: /@[a-z_]\w*/ }, { cls: 'tk-var-const', pattern: /:[a-z_]\w*[?!]?/ }, { cls: 'tk-fn-decl', pattern: /(\b(?:defp?|defmacrop?|defguard|defdelegate)\s+)[a-z_]\w*[?!]?/, lookbehind: true }, @@ -1134,11 +1271,11 @@ ).split(' '); G.haskell = [ - // Block comments before strings; line comments after strings so - // "not -- a comment" stays a string. - { cls: 'tk-comment', pattern: /\{-[\s\S]*?-\}/ }, - { cls: 'tk-string', pattern: /"(?:\\.|[^"\\\n])*"/ }, - { cls: 'tk-comment', pattern: /--.*/ }, + // Comments and strings compete by position (see `group` in tokenize), so + // "not -- a comment" stays a string and a quote in a comment stays comment. + { cls: 'tk-comment', pattern: /\{-[\s\S]*?-\}/, group: 'span' }, + { cls: 'tk-string', pattern: /"(?:\\.|[^"\\\n])*"/, group: 'span' }, + { cls: 'tk-comment', pattern: /--.*/, group: 'span' }, { cls: 'tk-fn-decl', pattern: /^[a-z_][\w']*(?=\s*::)/m }, { cls: 'tk-keyword', pattern: wordPattern(HS_KEYWORDS) }, { cls: 'tk-type', pattern: /\b[A-Z][\w']*/ }, @@ -1152,9 +1289,9 @@ // GraphQL // ============================================================ G.graphql = [ - // Strings before comments so "not # a comment" stays a string. - { cls: 'tk-string', pattern: /"""[\s\S]*?"""|"(?:\\.|[^"\\\n])*"/ }, - { cls: 'tk-comment', pattern: /#.*/ }, + // Strings and comments compete by position (see `group` in tokenize). + { cls: 'tk-string', pattern: /"""[\s\S]*?"""|"(?:\\.|[^"\\\n])*"/, group: 'span' }, + { cls: 'tk-comment', pattern: /#.*/, group: 'span' }, { cls: 'tk-decorator', pattern: /@[A-Za-z_]\w*/ }, { cls: 'tk-var-param', pattern: /\$[A-Za-z_]\w*/ }, { cls: 'tk-keyword', pattern: /\b(?:query|mutation|subscription|fragment|on|type|interface|union|enum|input|scalar|schema|directive|extend|implements|repeatable|true|false|null)\b/ }, @@ -1168,11 +1305,11 @@ // TOML / INI // ============================================================ G.toml = [ - // Table headers first (they may contain quoted keys), then strings before - // comments, else # inside "..." is eaten as a comment + // Table headers first (they may contain quoted keys), then strings and + // comments competing by position (see `group` in tokenize). { cls: 'tk-tag', pattern: /^[ \t]*\[\[?[^\]\n]+\]\]?/m }, - { cls: 'tk-string', pattern: /"""[\s\S]*?"""|'''[\s\S]*?'''|"(?:\\.|[^"\\\n])*"|'[^'\n]*'/ }, - { cls: 'tk-comment', pattern: /#.*/ }, + { cls: 'tk-string', pattern: /"""[\s\S]*?"""|'''[\s\S]*?'''|"(?:\\.|[^"\\\n])*"|'[^'\n]*'/, group: 'span' }, + { cls: 'tk-comment', pattern: /#.*/, group: 'span' }, { cls: 'tk-type', pattern: /^(\s*)[A-Za-z0-9_.-]+(?=\s*=)/m, lookbehind: true }, { cls: 'tk-keyword', pattern: /\b(?:true|false)\b/ }, { cls: 'tk-number', pattern: /\d{4}-\d{2}-\d{2}(?:[T ][\d:.]+(?:Z|[+-]\d{2}:\d{2})?)?|[+-]?\b(?:0[xX][\da-fA-F_]+|0[oO][0-7_]+|0[bB][01_]+|\d[\d_]*(?:\.\d[\d_]*)?(?:[eE][+-]?\d+)?|inf|nan)\b/ }, @@ -1194,9 +1331,9 @@ // Dockerfile // ============================================================ G.dockerfile = [ - // Strings before comments so "not # a comment" stays a string. - { cls: 'tk-string', pattern: /"(?:\\.|[^"\\\n])*"|'[^'\n]*'/ }, - { cls: 'tk-comment', pattern: /#.*/ }, + // Strings and comments compete by position (see `group` in tokenize). + { cls: 'tk-string', pattern: /"(?:\\.|[^"\\\n])*"|'[^'\n]*'/, group: 'span' }, + { cls: 'tk-comment', pattern: /#.*/, group: 'span' }, { cls: 'tk-keyword', pattern: /^\s*(?:FROM|RUN|CMD|LABEL|MAINTAINER|EXPOSE|ENV|ADD|COPY|ENTRYPOINT|VOLUME|USER|WORKDIR|ARG|ONBUILD|STOPSIGNAL|HEALTHCHECK|SHELL)\b|\bAS\b/m }, { cls: 'tk-var-builtin', pattern: /\$\{[^}]+\}|\$\w+/ }, { cls: 'tk-decorator', pattern: /(^|\s)--[\w-]+(?==|\s|$)/, lookbehind: true }, diff --git a/tests/constructs.test.mjs b/tests/constructs.test.mjs index 32d83c8..b8a71f1 100644 --- a/tests/constructs.test.mjs +++ b/tests/constructs.test.mjs @@ -343,6 +343,12 @@ test('every new string form stays linear on pathological input', () => { ['php', '<< { token(code, 'elixir', 'string', '~s{hi # not comment}'); notSwallowed(code, 'elixir', 'comment', 'not comment'); }); + +// ── Spans that compete by position ───────────────────────────────────────── +// A string may hold a comment marker and a comment may hold a quote, in the +// same grammar. No fixed rule order renders both, so these rules share a group +// and whichever opens first wins. Before beta.5 every grammar had picked one of +// the two failures: most read `// don't … won't` as a comment holding a string, +// and the C family, JavaScript and PHP read "/* x */" as a string holding a +// comment. The only test that touched this — Rust's `// don't do this` — held +// a single apostrophe, which cannot open a string that needs a closing one. + +/** Every normalised language with a discoverable line comment, and its marker. */ +function lineCommentLanguages() { + const markers = ['//', '#', '--']; + const names = [...new Set(Object.keys(JSRay.languages).map((k) => JSRay.normalizeLanguage(k)))]; + const found = []; + for (const lang of names) { + const marker = markers.find((m) => + leaves(JSRay.tokenize(`${m} plain words`, lang)).some( + (t) => t.type === 'tk-comment' && t.text === `${m} plain words` + ) + ); + if (marker) found.push([lang, marker]); + } + return found; +} + +test('A line comment holding two quotes is one comment, in every language', () => { + const langs = lineCommentLanguages(); + + // Discovery decides which languages are checked, so a grammar whose comment + // rule broke outright would silently drop out of this test. These must stay. + for (const must of ['javascript', 'python', 'php', 'shell', 'ruby', 'c', 'java', 'go', + 'rust', 'sql', 'yaml', 'lua', 'perl', 'elixir', 'haskell', 'toml', 'jsonc']) { + assert.ok(langs.some(([l]) => l === must), `${must} has no discoverable line comment`); + } + + for (const [lang, m] of langs) { + for (const body of ["don't stop, won't stop", 'say "hi" then "bye"']) { + const line = `${m} ${body}`; + token(`${line}\nx = 1`, lang, 'comment', line); + } + } +}); + +test('A string holding a comment marker is one string', () => { + for (const [lang, literal] of [ + ['js', '"https://jsray.org"'], ['js', "'/* not a comment */'"], ['js', '"/** nor this */"'], + ['c', '"/* x */"'], ['cpp', '"// x"'], ['java', '"a // b"'], ['go', '"/* x */"'], + ['php', '"/* x */"'], ['php', "'# anchor'"], ['python', '"# not a comment"'], + ['shell', '"a # b"'], ['ruby', "'# x'"], ['sql', "'-- not a comment'"], ['lua', '"-- x"'], + ['yaml', '"a # b"'], ['r', '"# x"'], ['perl', '"# x"'], ['powershell', '"# x"'], + ['elixir', '"# x"'], ['toml', '"# x"'], ['dockerfile', '"# x"'], + ['jsonc', '"https://jsray.org"'], + ]) { + token(`x = ${literal};\n`, lang, 'string', literal); + } +}); + +test('JavaScript: a regex may hold a quote without opening a string', () => { + const code = 's.split(/"/); t = "u";'; + + token(code, 'js', 'regex', '/"/'); + token(code, 'js', 'string', '"u"'); +}); + +test('A literal named inside a comment opens nothing', () => { + // beta.4 put these rules ahead of the comment rule, so each comment lost + // everything from the literal onward. + token('# prefer %w[a b] over arrays\nx = 1', 'ruby', 'comment', '# prefer %w[a b] over arrays'); + token('# wrap it in q{like this} here\nx = 1', 'perl', 'comment', '# wrap it in q{like this} here'); + token('# match with ~r/abc/ first\nx = 1', 'elixir', 'comment', '# match with ~r/abc/ first'); + + // A heredoc opening written in a comment must not turn the lines below it + // into a string, even when a matching terminator happens to follow. + for (const [lang, code] of [ + ['shell', '# pipe it: cat < t.type === 'tk-string'); + assert.equal(strings.length, 0, `[${lang}] a commented heredoc opened: ${JSON.stringify(strings)}`); + } +}); + +test('shell: a quoted redirect does not cost a heredoc its opening', () => { + // Measured from where its body starts, this heredoc would lose to "out.txt", + // which sits earlier on the opening line. Positions are compared where the + // whole match begins, lookbehind prefix included. + token('cat < "out.txt"\nbody\nEOF\n', 'shell', 'string', 'body\nEOF'); +}); + +test('JavaScript: a parameter list competes with the spans around it', () => { + // Ahead of the comments it would read a signature inside a doc comment as + // code; behind the strings it could not match a list holding a string default. + token('/** call function f(a, b) here */\nx = 1', 'js', 'doc', '/** call function f(a, b) here */'); + token('function f(a, b = "x") {}', 'js', 'var-param', 'b'); + token('function f(a /* why */, b) {}', 'js', 'comment', '/* why */'); + token('function f(a /* why */, b) {}', 'js', 'var-param', 'b'); + token('const g = (a = "x,y") => a;', 'js', 'string', '"x,y"'); +}); + +test('C family: preprocessor lines and annotations respect the spans around them', () => { + token('/*\n#define X 1\n*/\nint y;', 'c', 'comment', '/*\n#define X 1\n*/'); + token('#include "stdio.h"\nint y;', 'c', 'decorator', '#include "stdio.h"'); + token('// see @Override\nint y;', 'java', 'comment', '// see @Override'); + token('@Override\nvoid f() {}', 'java', 'decorator', '@Override'); +}); + +test('SQL: an apostrophe in a comment does not open a string across lines', () => { + // SQL strings may span lines, so the old order let `-- don't` open a literal + // that ran on to the next apostrophe anywhere below it. + const code = "-- don't\nSELECT 'x' FROM t"; + + token(code, 'sql', 'comment', "-- don't"); + token(code, 'sql', 'string', "'x'"); +}); diff --git a/types/jsray.d.ts b/types/jsray.d.ts index d5bb3fa..9655b33 100644 --- a/types/jsray.d.ts +++ b/types/jsray.d.ts @@ -19,6 +19,13 @@ declare namespace JSRay { * terminator is present, which leaves the opening to the rules behind it. */ close?: (match: RegExpExecArray, text: string, from: number) => number; + /** + * Adjacent rules carrying the same group compete by position instead of + * by order: at every step the match that begins earliest wins, and listed + * order only breaks a tie. This is how a comment may hold a quote and a + * string may hold a comment marker in the same grammar. + */ + group?: string; } type Grammar = GrammarRule[]; From 6a3b07c58f6737c6175e6bb79d6dec149a679ddc Mon Sep 17 00:00:00 2001 From: Jie Date: Sun, 13 Sep 2026 18:07:20 +0800 Subject: [PATCH 2/3] Let a template placeholder hold a template of its own MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A placeholder holding a template — ${ok ? `a ${b}` : 'c'} — ended the outer template at the inner one's closing backtick, because the placeholder pattern stopped at the first }. Under ordered rules that was wrong and mostly harmless. Once spans compete by position, the outer template's real closing backtick opens a template of its own, running to the next backtick in the file, and everything in between renders inverted. jsray-terminal's tests showed it: a template carrying a Python script, with a ternary between a template and a string in one placeholder. Rendering every source file in the four repositories with beta.4 and with this build, that file was the one place the previous commit made worse. A placeholder now admits one nested template and one level of braces. Every alternative inside it begins with its own character — a brace, a backtick, or neither — so it still has one parse, and the storm cases include nested openings that never close. --- CHANGELOG.md | 16 ++++++++++++++++ dist/jsray.js | 24 +++++++++++++++++++----- docs/languages.md | 2 +- docs/languages.zh-CN.md | 2 +- integrity.json | 2 +- src/jsray.js | 24 +++++++++++++++++++----- tests/constructs.test.mjs | 19 +++++++++++++++++++ 7 files changed, 76 insertions(+), 13 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 540df8c..6ff8ba5 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -39,6 +39,22 @@ versioning follows [SemVer](https://semver.org/). two; and JSONC read the `//` in `"https://…"`, which editor settings and tsconfig files are full of, as the start of a comment. + Checked beyond the tests by rendering every JavaScript, PHP, shell, YAML and + CSS file in the four JSRay repositories, and each code block in their docs, + with beta.4 and with this build, and reading the differences by class. None + was a regression; beta.4 had three block comments swallowing code where this + build has none. + +- **A template placeholder may hold a template of its own.** + `` `${ok ? `a ${b}` : 'c'}` `` and `` `${items.map((x) => `
  • ${x}
  • `)}` `` + ended the outer template at the inner one's closing backtick. That was wrong + before and survivable; once spans compete, the outer template's real closing + backtick opened a new template running to the next backtick in the file, and + everything between rendered inverted. jsray-terminal's own tests, which carry + a Python script in a template, were where it showed. A placeholder now admits + one nested template and one level of braces, and each alternative inside it + still begins with its own character, so the pattern has one parse. + ## [0.0.2-beta.4] — 2026-09-09 Five languages gain the literals they never had, by way of the one thing the diff --git a/dist/jsray.js b/dist/jsray.js index 9ceecb2..373e294 100644 --- a/dist/jsray.js +++ b/dist/jsray.js @@ -312,11 +312,25 @@ // engine try every combination: 26 placeholders took 8.7s to fail. Every // interpolating grammar below keeps its fallback and its interpolation // branch disjoint for the same reason. - { cls: 'tk-string', pattern: /`(?:\\.|\$\{[^}]*\}|\$(?!\{)|[^`\\$])*`/, group: 'span', inside: [ - { cls: 'tk-operator', pattern: /\$\{[^}]*\}/, inside: [ - { cls: 'tk-punct', pattern: /^\$\{|\}$/ }, - // No JS recursion here; minimal coloring avoids rule cross-talk - { cls: 'tk-var', pattern: /[A-Za-z_$][\w$]*/ }, + // + // A placeholder may hold a template of its own — `${ok ? `a ${b}` : 'c'}` + // and `${items.map((x) => `
  • ${x}
  • `)}` are ordinary JavaScript — and + // one level of braces, as in `${fn({ a })}`. The old `\$\{[^}]*\}` stopped + // at the inner template's first `}`, so the outer template ended at the + // inner one's closing backtick and its own closing backtick was left over. + // Once spans compete by position, that leftover opens a template running + // to the next backtick in the file, and everything between reads inverted. + // Inside a placeholder each alternative still begins with its own + // character — a brace, a backtick, or neither — so there is one parse. + // The operator pattern below is the placeholder branch of this one. + { cls: 'tk-string', pattern: /`(?:\\.|\$\{(?:[^{}`]|\{[^{}]*\}|`(?:\\.|\$\{[^{}]*\}|\$(?!\{)|[^`\\$])*`)*\}|\$(?!\{)|[^`\\$])*`/, group: 'span', inside: [ + { cls: 'tk-operator', pattern: /\$\{(?:[^{}`]|\{[^{}]*\}|`(?:\\.|\$\{[^{}]*\}|\$(?!\{)|[^`\\$])*`)*\}/, inside: [ + { cls: 'tk-punct', pattern: /^\$\{|\}$/ }, + // A nested template or a quoted string inside a placeholder is a + // string, not a run of variables. No JS recursion beyond that; + // minimal coloring avoids rule cross-talk. + { cls: 'tk-string', pattern: /`(?:\\.|\$\{[^{}]*\}|\$(?!\{)|[^`\\$])*`|"(?:\\.|[^"\\\n])*"|'(?:\\.|[^'\\\n])*'/ }, + { cls: 'tk-var', pattern: /[A-Za-z_$][\w$]*/ }, ]}, ]}, { cls: 'tk-string', pattern: RX.string1, group: 'span' }, diff --git a/docs/languages.md b/docs/languages.md index e4b62f7..0b942a1 100644 --- a/docs/languages.md +++ b/docs/languages.md @@ -16,7 +16,7 @@ Recognizes: - Parameter lists → `tk-var-param` (both `function name(a, b: T = 0)` and `(a, b) => ...`) - Builtin variables: `console`, `window`, `document`, `globalThis`, `Math`, `JSON`, ... - Builtin functions (as `.fn(` or `fn(`): `log`, `fetch`, `parseInt`, ... -- Template strings `` `...${id}...` `` with inline `${}` interpolation highlighted +- Template strings `` `...${id}...` `` with inline `${}` interpolation highlighted, including a template nested inside a placeholder (`` `${ok ? `a ${b}` : 'c'}` ``) and one level of braces (`${fn({ a })}`) - Regex literals `/pattern/flags`, context-sensitive (only after `=` `(` `,` `return`, etc.) - Numeric literals including separators (`1_000_000`), binary/octal/hex, and the BigInt suffix (`10n`) - `ALL_CAPS` constants, `.property` access, `@decorator` diff --git a/docs/languages.zh-CN.md b/docs/languages.zh-CN.md index 0b3372c..e1be894 100644 --- a/docs/languages.zh-CN.md +++ b/docs/languages.zh-CN.md @@ -16,7 +16,7 @@ - 形参列表 → `tk-var-param`(含 `function name(a, b: T = 0)` 和 `(a, b) => ...`) - 内置变量:`console`, `window`, `document`, `globalThis`, `Math`, `JSON`, ... - 内置函数(出现在 `.fn(` 或 `fn(` 位置):`log`, `fetch`, `parseInt`, ... -- 模板字符串 `` `...${id}...` ``,含 `${}` 内联高亮 +- 模板字符串 `` `...${id}...` ``,含 `${}` 内联高亮,包括占位符里再套一层模板串(`` `${ok ? `a ${b}` : 'c'}` ``)和一层花括号(`${fn({ a })}`) - 数字字面量含分隔符(`1_000_000`)、二/八/十六进制,以及 BigInt 后缀(`10n`) - 正则字面量 `/pattern/flags`,上下文敏感(前面是 `=` `(` `,` `return` 等才识别) - `ALL_CAPS` 常量、`.property` 访问、`@decorator` diff --git a/integrity.json b/integrity.json index 00e6074..f0a1e68 100644 --- a/integrity.json +++ b/integrity.json @@ -4,7 +4,7 @@ "algorithm": "sha256", "note": "Base64 SHA-256 digests of the released Core assets. Integrations copy the relevant digest at sync time and verify their bundled snapshot against it.", "files": { - "dist/jsray.js": "sha256-/MfXR8QVmR8xPViGqX6OPy2R3T5JHmwhx0OY11LL/uk=", + "dist/jsray.js": "sha256-zlzwEsxRVkPvuUNeWpdrO64/FqnKd9DZjkEgimdb2es=", "dist/jsray.css": "sha256-6Fuva+aZwCcmltNi2oN+1yWIJoKE+1rLpUxZvD9iutc=", "dist/themes/aurora.css": "sha256-S8X+R8XZNC8WRdHOen839zMkPxVFol6PPTrpZibbeJA=", "dist/themes/default.css": "sha256-vENOigRjwn1hW6wPjdM4tbZ58Ja44FMsa+akavPiMSI=", diff --git a/src/jsray.js b/src/jsray.js index 9ceecb2..373e294 100644 --- a/src/jsray.js +++ b/src/jsray.js @@ -312,11 +312,25 @@ // engine try every combination: 26 placeholders took 8.7s to fail. Every // interpolating grammar below keeps its fallback and its interpolation // branch disjoint for the same reason. - { cls: 'tk-string', pattern: /`(?:\\.|\$\{[^}]*\}|\$(?!\{)|[^`\\$])*`/, group: 'span', inside: [ - { cls: 'tk-operator', pattern: /\$\{[^}]*\}/, inside: [ - { cls: 'tk-punct', pattern: /^\$\{|\}$/ }, - // No JS recursion here; minimal coloring avoids rule cross-talk - { cls: 'tk-var', pattern: /[A-Za-z_$][\w$]*/ }, + // + // A placeholder may hold a template of its own — `${ok ? `a ${b}` : 'c'}` + // and `${items.map((x) => `
  • ${x}
  • `)}` are ordinary JavaScript — and + // one level of braces, as in `${fn({ a })}`. The old `\$\{[^}]*\}` stopped + // at the inner template's first `}`, so the outer template ended at the + // inner one's closing backtick and its own closing backtick was left over. + // Once spans compete by position, that leftover opens a template running + // to the next backtick in the file, and everything between reads inverted. + // Inside a placeholder each alternative still begins with its own + // character — a brace, a backtick, or neither — so there is one parse. + // The operator pattern below is the placeholder branch of this one. + { cls: 'tk-string', pattern: /`(?:\\.|\$\{(?:[^{}`]|\{[^{}]*\}|`(?:\\.|\$\{[^{}]*\}|\$(?!\{)|[^`\\$])*`)*\}|\$(?!\{)|[^`\\$])*`/, group: 'span', inside: [ + { cls: 'tk-operator', pattern: /\$\{(?:[^{}`]|\{[^{}]*\}|`(?:\\.|\$\{[^{}]*\}|\$(?!\{)|[^`\\$])*`)*\}/, inside: [ + { cls: 'tk-punct', pattern: /^\$\{|\}$/ }, + // A nested template or a quoted string inside a placeholder is a + // string, not a run of variables. No JS recursion beyond that; + // minimal coloring avoids rule cross-talk. + { cls: 'tk-string', pattern: /`(?:\\.|\$\{[^{}]*\}|\$(?!\{)|[^`\\$])*`|"(?:\\.|[^"\\\n])*"|'(?:\\.|[^'\\\n])*'/ }, + { cls: 'tk-var', pattern: /[A-Za-z_$][\w$]*/ }, ]}, ]}, { cls: 'tk-string', pattern: RX.string1, group: 'span' }, diff --git a/tests/constructs.test.mjs b/tests/constructs.test.mjs index b8a71f1..0a259cd 100644 --- a/tests/constructs.test.mjs +++ b/tests/constructs.test.mjs @@ -349,6 +349,10 @@ test('every new string form stays linear on pathological input', () => { ['js', '"/* \'// `'.repeat(3000)], ['php', '\'/* "# '.repeat(3000)], ['ruby', '%w[ "# \''.repeat(3000)], + // A placeholder may now hold a template, which holds placeholders of its + // own. Openings that nest and never close are the shape to watch. + ['js', '`${ `${'.repeat(3000)], + ['js', '`a ${ {'.repeat(3000)], ]; for (const [lang, code] of shapes) { @@ -573,3 +577,18 @@ test('SQL: an apostrophe in a comment does not open a string across lines', () = token(code, 'sql', 'comment', "-- don't"); token(code, 'sql', 'string', "'x'"); }); + +test('JavaScript: a placeholder may hold a template of its own', () => { + // Found in jsray-terminal's own tests, where a template carries a Python + // script and one of its placeholders a ternary between a template and a + // string. The outer template ended at the inner one's backtick; its real + // closing backtick then opened a template that inverted the rest of the file. + const code = 'const s = `a ${ok ? `b ${c}` : "d"} e`;\nconst after = 1;'; + + token(code, 'js', 'string', '`a ${ok ? `b ${c}` : "d"} e`'); + notSwallowed(code, 'js', 'string', 'const after'); + + token('const u = `x ${fn({ a: 1 })} y`;', 'js', 'string', '`x ${fn({ a: 1 })} y`'); + token('const h = `${items.map((x) => `
  • ${x}
  • `).join("")}`;', 'js', 'string', + '`${items.map((x) => `
  • ${x}
  • `).join("")}`'); +}); From 68858604ff29c5fa0f9460641464de2ab5556263 Mon Sep 17 00:00:00 2001 From: Jie Date: Sun, 13 Sep 2026 18:07:22 +0800 Subject: [PATCH 3/3] Colour private class members as properties #count, this.#count and #count in obj were plain text, and the type rule took #Foo apart at the word boundary, colouring Foo and leaving the # bare. The rule sits ahead of the type and constant rules for that reason. A shebang's #! is not a name and is left alone. --- CHANGELOG.md | 6 ++++++ dist/jsray.js | 6 ++++++ docs/languages.md | 2 +- docs/languages.zh-CN.md | 2 +- integrity.json | 2 +- src/jsray.js | 6 ++++++ tests/constructs.test.mjs | 9 +++++++++ 7 files changed, 30 insertions(+), 3 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 6ff8ba5..a76e481 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -55,6 +55,12 @@ versioning follows [SemVer](https://semver.org/). one nested template and one level of braces, and each alternative inside it still begins with its own character, so the pattern has one parse. +### Added + +- **Private class members** — `#count`, `this.#count`, `#count in obj` — are + coloured as properties. They were plain text, and the type rule split `#Foo` + at the word boundary, colouring `Foo` and leaving the `#` bare. + ## [0.0.2-beta.4] — 2026-09-09 Five languages gain the literals they never had, by way of the one thing the diff --git a/dist/jsray.js b/dist/jsray.js index 373e294..4be4f87 100644 --- a/dist/jsray.js +++ b/dist/jsray.js @@ -345,6 +345,12 @@ pattern: /(^|[=(,!&|?:;{}\[\]]\s*|\breturn\s*)\/(?![*\/])(?:\\.|\[(?:\\.|[^\]\\\n])*\]|[^\/\\\n])+\/[gimsuy]*/, lookbehind: true, group: 'span' }, + // Private class members — `#count`, `this.#count`, `#count in obj`. The + // `#` belongs to the name, and the rule precedes the type and constant + // rules, which would otherwise take `#Foo` apart at the word boundary. + // A shebang's `#!` is not a name and is left alone. + { cls: 'tk-property', pattern: /#[A-Za-z_$][\w$]*/ }, + // Decorators { cls: 'tk-decorator', pattern: /@[A-Za-z_$][\w$]*/ }, diff --git a/docs/languages.md b/docs/languages.md index 0b942a1..774e0b5 100644 --- a/docs/languages.md +++ b/docs/languages.md @@ -19,7 +19,7 @@ Recognizes: - Template strings `` `...${id}...` `` with inline `${}` interpolation highlighted, including a template nested inside a placeholder (`` `${ok ? `a ${b}` : 'c'}` ``) and one level of braces (`${fn({ a })}`) - Regex literals `/pattern/flags`, context-sensitive (only after `=` `(` `,` `return`, etc.) - Numeric literals including separators (`1_000_000`), binary/octal/hex, and the BigInt suffix (`10n`) -- `ALL_CAPS` constants, `.property` access, `@decorator` +- `ALL_CAPS` constants, `.property` access and private members (`#count`), `@decorator` ```ts async function fetchUser(id: number): Promise { diff --git a/docs/languages.zh-CN.md b/docs/languages.zh-CN.md index e1be894..611dc5f 100644 --- a/docs/languages.zh-CN.md +++ b/docs/languages.zh-CN.md @@ -19,7 +19,7 @@ - 模板字符串 `` `...${id}...` ``,含 `${}` 内联高亮,包括占位符里再套一层模板串(`` `${ok ? `a ${b}` : 'c'}` ``)和一层花括号(`${fn({ a })}`) - 数字字面量含分隔符(`1_000_000`)、二/八/十六进制,以及 BigInt 后缀(`10n`) - 正则字面量 `/pattern/flags`,上下文敏感(前面是 `=` `(` `,` `return` 等才识别) -- `ALL_CAPS` 常量、`.property` 访问、`@decorator` +- `ALL_CAPS` 常量、`.property` 访问与私有成员(`#count`)、`@decorator` ```ts async function fetchUser(id: number): Promise { diff --git a/integrity.json b/integrity.json index f0a1e68..49d28c4 100644 --- a/integrity.json +++ b/integrity.json @@ -4,7 +4,7 @@ "algorithm": "sha256", "note": "Base64 SHA-256 digests of the released Core assets. Integrations copy the relevant digest at sync time and verify their bundled snapshot against it.", "files": { - "dist/jsray.js": "sha256-zlzwEsxRVkPvuUNeWpdrO64/FqnKd9DZjkEgimdb2es=", + "dist/jsray.js": "sha256-KOSefuYNXQrhcQMaV4c25TpTRhgUMtNgX5WvtT3NGGc=", "dist/jsray.css": "sha256-6Fuva+aZwCcmltNi2oN+1yWIJoKE+1rLpUxZvD9iutc=", "dist/themes/aurora.css": "sha256-S8X+R8XZNC8WRdHOen839zMkPxVFol6PPTrpZibbeJA=", "dist/themes/default.css": "sha256-vENOigRjwn1hW6wPjdM4tbZ58Ja44FMsa+akavPiMSI=", diff --git a/src/jsray.js b/src/jsray.js index 373e294..4be4f87 100644 --- a/src/jsray.js +++ b/src/jsray.js @@ -345,6 +345,12 @@ pattern: /(^|[=(,!&|?:;{}\[\]]\s*|\breturn\s*)\/(?![*\/])(?:\\.|\[(?:\\.|[^\]\\\n])*\]|[^\/\\\n])+\/[gimsuy]*/, lookbehind: true, group: 'span' }, + // Private class members — `#count`, `this.#count`, `#count in obj`. The + // `#` belongs to the name, and the rule precedes the type and constant + // rules, which would otherwise take `#Foo` apart at the word boundary. + // A shebang's `#!` is not a name and is left alone. + { cls: 'tk-property', pattern: /#[A-Za-z_$][\w$]*/ }, + // Decorators { cls: 'tk-decorator', pattern: /@[A-Za-z_$][\w$]*/ }, diff --git a/tests/constructs.test.mjs b/tests/constructs.test.mjs index 0a259cd..ca32220 100644 --- a/tests/constructs.test.mjs +++ b/tests/constructs.test.mjs @@ -569,6 +569,15 @@ test('C family: preprocessor lines and annotations respect the spans around them token('@Override\nvoid f() {}', 'java', 'decorator', '@Override'); }); +test('JavaScript: a private member keeps its #', () => { + token('class A { #count = 0; }', 'js', 'property', '#count'); + token('this.#count++;', 'js', 'property', '#count'); + + const shebang = leaves(JSRay.tokenize('#!/usr/bin/env node\nx = 1', 'js')) + .filter((t) => t.type === 'tk-property'); + assert.equal(shebang.length, 0, `a shebang was read as a member: ${JSON.stringify(shebang)}`); +}); + test('SQL: an apostrophe in a comment does not open a string across lines', () => { // SQL strings may span lines, so the old order let `-- don't` open a literal // that ran on to the next apostrophe anywhere below it.