diff --git a/.changeset/tokenizer-text-fast-path.md b/.changeset/tokenizer-text-fast-path.md new file mode 100644 index 000000000..45a3e8dcd --- /dev/null +++ b/.changeset/tokenizer-text-fast-path.md @@ -0,0 +1,7 @@ +--- +'@shopify/liquid-html-parser': patch +--- + +Tokenize runs of plain text in one step instead of one character at a time. Most characters can't start a token, so the tokenizer now jumps to the next one that can. + +Tokens and ASTs are unchanged. On Dawn, Horizon and the base theme, `tokenize` is about 6× faster and `toLiquidHtmlAST`/`toLiquidAST` about 2× faster. diff --git a/packages/liquid-html-parser/src/document/tokenizer.test.ts b/packages/liquid-html-parser/src/document/tokenizer.test.ts index eda773731..a5c4f2c0c 100644 --- a/packages/liquid-html-parser/src/document/tokenizer.test.ts +++ b/packages/liquid-html-parser/src/document/tokenizer.test.ts @@ -1,6 +1,6 @@ import { describe, expect, it } from 'vitest'; -import { tokenize, TokenType } from './tokenizer'; -import type { Token } from './tokenizer'; +import { tokenize, tokenizeWithoutFastPath, TokenType } from './tokenizer'; +import type { Token, TokenizeOptions } from './tokenizer'; /** Strip the trailing EndOfInput token for cleaner assertions. */ function tokens(source: string): Token[] { @@ -497,6 +497,159 @@ describe('Unit: document-tokenizer', () => { }); }); + describe('text runs end at the next token', () => { + it('ends a Liquid tag body at a -%} preceded by another -', () => { + const source = '{% if a--%}'; + expect(tokens(source)).toMatchObject([ + { type: TokenType.LiquidTagOpen, start: 0, end: 2 }, + { type: TokenType.Text, start: 2, end: 8 }, + { type: TokenType.LiquidTagClose, start: 8, end: 11 }, + ]); + assertTokenInvariants(source); + }); + + it('keeps %} as text inside a Liquid drop and ends it at -}}', () => { + const source = '{{ "%}" -}}'; + expect(tokens(source)).toMatchObject([ + { type: TokenType.LiquidVariableOutputOpen, start: 0, end: 2 }, + { type: TokenType.Text, start: 2, end: 8 }, + { type: TokenType.LiquidVariableOutputClose, start: 8, end: 11 }, + ]); + assertTokenInvariants(source); + }); + + it('ends a curly-quoted value on its partner, not on a straight quote', () => { + const source = ''; + expect(tokens(source)).toMatchObject([ + { type: TokenType.HtmlTagOpen, start: 0, end: 1 }, + { type: TokenType.Text, start: 1, end: 4 }, + { type: TokenType.HtmlEquals, start: 4, end: 5 }, + { type: TokenType.HtmlQuoteOpen, start: 5, end: 6 }, + { type: TokenType.Text, start: 6, end: 9 }, + { type: TokenType.HtmlQuoteClose, start: 9, end: 10 }, + { type: TokenType.Text, start: 10, end: 12 }, + { type: TokenType.HtmlTagClose, start: 12, end: 13 }, + ]); + assertTokenInvariants(source); + }); + + it('finds --> and c', '', + '>', + '=', + '"', + "'", + '\u201c', + '\u201d', + '\u2018', + '\u2019', + ], + ], + ...quotePairs.map(([open, close]): [string, string, TokenizeOptions, string[]] => [ + `QuotedValue ${open}`, + '', + { insideQuotedAttribute: open }, + [...new Set(['{{', '{%', open, close])], + ]), + ['LiquidTag', '{% ', {}, ['%}', '-%}']], + ['LiquidVariableOutput', '{{ ', {}, ['}}', '-}}']], + ]; + + for (const [mode, prefix, options, starts] of cases) { + it.each(starts)(`${mode}: %j`, (start) => { + const offset = prefix.length + pad.length; + const result = tokenize(prefix + pad + start + pad, options); + expect(result.some((t) => t.type !== TokenType.Text && t.start === offset)).toBe(true); + }); + } + }); + + describe('matches tokenizeWithoutFastPath', () => { + // Every string of up to 4 of these characters, in every entry state. The + // set covers each character a mode's token can start with, plus near + // misses (`!`, `\u201a`) and plain text. + const alphabet = [...'{}%-<>/=!"\'\u201c\u201d\u2018\u2019\u201a a\n']; + let sources = ['']; + for (let length = 1, previous = ['']; length <= 4; length++) { + previous = previous.flatMap((s) => alphabet.map((c) => s + c)); + sources = sources.concat(previous); + } + + const entryStates: Array<[string, string, TokenizeOptions]> = [ + ['document start', '', {}], + ['skipFrontmatter', '', { skipFrontmatter: true }], + ['insideHtmlTag', '', { insideHtmlTag: true }], + ...['"', "'", '\u201c', '\u201d', '\u2018', '\u2019'].map( + (quote): [string, string, TokenizeOptions] => [ + `insideQuotedAttribute ${quote}`, + '', + { insideQuotedAttribute: quote }, + ], + ), + ['inside a Liquid tag', '{% ', {}], + ['inside a Liquid output', '{{ ', {}], + ]; + + function sameTokens(a: Token[], b: Token[]): boolean { + return ( + a.length === b.length && + a.every((t, i) => t.type === b[i].type && t.start === b[i].start && t.end === b[i].end) + ); + } + + it.each(entryStates)('%s', (_name, prefix, options) => { + const mismatches: string[] = []; + for (const source of sources) { + const input = prefix + source; + if (!sameTokens(tokenize(input, options), tokenizeWithoutFastPath(input, options))) { + mismatches.push(input); + if (mismatches.length === 10) break; + } + } + expect(mismatches).toEqual([]); + }); + }); + describe('structural invariants', () => { const cases = [ '{{ x }}', diff --git a/packages/liquid-html-parser/src/document/tokenizer.ts b/packages/liquid-html-parser/src/document/tokenizer.ts index dcef56ec4..5b88e84b6 100644 --- a/packages/liquid-html-parser/src/document/tokenizer.ts +++ b/packages/liquid-html-parser/src/document/tokenizer.ts @@ -62,6 +62,25 @@ export interface TokenizeOptions { } export function tokenize(source: string, options: TokenizeOptions = {}): Token[] { + return tokenizeWith(source, options, nextTextCandidate); +} + +/** + * `tokenize` without the text fast path: plain text advances one character at + * a time. Test-only reference for checking that the fast path never skips over + * a token start. Not exported from the package. + */ +export function tokenizeWithoutFastPath(source: string, options: TokenizeOptions = {}): Token[] { + return tokenizeWith(source, options, (_source, from) => from); +} + +type NextTextCandidate = (source: string, from: number, mode: Mode, quoteChar: string) => number; + +function tokenizeWith( + source: string, + options: TokenizeOptions, + nextCandidate: NextTextCandidate, +): Token[] { const tokens: Token[] = []; const modeStack: Mode[] = []; let mode = Mode.Default as Mode; @@ -169,7 +188,7 @@ export function tokenize(source: string, options: TokenizeOptions = {}): Token[] popMode(); } else { startText(); - pos++; + pos = nextCandidate(source, pos + 1, mode, quoteChar); } break; } @@ -183,7 +202,7 @@ export function tokenize(source: string, options: TokenizeOptions = {}): Token[] popMode(); } else { startText(); - pos++; + pos = nextCandidate(source, pos + 1, mode, quoteChar); } break; } @@ -208,8 +227,7 @@ export function tokenize(source: string, options: TokenizeOptions = {}): Token[] } if (match('= `from` where the + * mode's `match()` checks could succeed, so the run of plain text before it is + * consumed in one step. Returning a superset of real token starts is safe: the + * main loop re-checks that position and treats a non-match as text. Missing a + * token start is not, so a new token type needs its first character added to + * its mode's helper (tokenizer.test.ts compares against + * `tokenizeWithoutFastPath` to catch this). + */ + +function nextTextCandidate(source: string, from: number, mode: Mode, quoteChar: string): number { + switch (mode) { + case Mode.Default: + return nextDefaultCandidate(source, from); + case Mode.HtmlTag: + return nextHtmlTagCandidate(source, from); + case Mode.QuotedValue: + return nextQuotedValueCandidate(source, from, quoteChar); + case Mode.LiquidTag: + return nextLiquidCloseCandidate(source, from, '%}'); + case Mode.LiquidVariableOutput: + return nextLiquidCloseCandidate(source, from, '}}'); + default: + return assertNever(mode); + } +} + +const CHAR_DOUBLE_QUOTE = 0x22; // " +const CHAR_SINGLE_QUOTE = 0x27; // ' +const CHAR_DASH = 0x2d; // - +const CHAR_SLASH = 0x2f; // / +const CHAR_LESS_THAN = 0x3c; // < +const CHAR_EQUALS = 0x3d; // = +const CHAR_GREATER_THAN = 0x3e; // > +const CHAR_OPEN_BRACE = 0x7b; // { +const CHAR_LEFT_SINGLE_CURLY_QUOTE = 0x2018; // ‘ +const CHAR_RIGHT_DOUBLE_CURLY_QUOTE = 0x201d; // ” + +/** Default mode tokens start with `{` (Liquid), `<` (HTML), or `-` (`-->`). */ +function nextDefaultCandidate(source: string, from: number): number { + for (let i = from; i < source.length; i++) { + const c = source.charCodeAt(i); + if (c === CHAR_OPEN_BRACE || c === CHAR_LESS_THAN || c === CHAR_DASH) return i; + } + return source.length; +} + +/** HtmlTag mode tokens start with `{`, `/`, `>`, `=`, or a straight/curly quote. */ +function nextHtmlTagCandidate(source: string, from: number): number { + for (let i = from; i < source.length; i++) { + const c = source.charCodeAt(i); + if ( + c === CHAR_OPEN_BRACE || + c === CHAR_SLASH || + c === CHAR_GREATER_THAN || + c === CHAR_EQUALS || + c === CHAR_DOUBLE_QUOTE || + c === CHAR_SINGLE_QUOTE || + (c >= CHAR_LEFT_SINGLE_CURLY_QUOTE && c <= CHAR_RIGHT_DOUBLE_CURLY_QUOTE) + ) { + return i; + } + } + return source.length; +} + +/** QuotedValue mode tokens start with `{` or either quote of the open pair. */ +function nextQuotedValueCandidate(source: string, from: number, quote: string): number { + const open = quote.charCodeAt(0); + const close = closingQuoteFor(quote).charCodeAt(0); + for (let i = from; i < source.length; i++) { + const c = source.charCodeAt(i); + if (c === CHAR_OPEN_BRACE || c === open || c === close) return i; + } + return source.length; +} + +/** + * Liquid tag/output bodies only end at `%}`/`}}`, optionally preceded by `-`. + * Any `-%}` match contains a `%}` one character later, so the earliest close + * is at the first `%}` or the `-` immediately before it. + */ +function nextLiquidCloseCandidate(source: string, from: number, close: '%}' | '}}'): number { + const i = source.indexOf(close, from); + if (i === -1) return source.length; + return i > from && source.charCodeAt(i - 1) === CHAR_DASH ? i - 1 : i; +} + +/** `[a-zA-Z{]`: what may follow `<` or `= 0x61 && c <= 0x7a) || (c >= 0x41 && c <= 0x5a) || c === CHAR_OPEN_BRACE; +} + enum Mode { Default = 'Default', HtmlTag = 'HtmlTag',