Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 7 additions & 0 deletions .changeset/tokenizer-text-fast-path.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,7 @@
---
'@shopify/liquid-html-parser': patch
---

Tokenize runs of plain text in one step instead of one character at a time. Most characters can't start a token, so the tokenizer now jumps to the next one that can.

Tokens and ASTs are unchanged. On Dawn, Horizon and the base theme, `tokenize` is about 6× faster and `toLiquidHtmlAST`/`toLiquidAST` about 2× faster.
157 changes: 155 additions & 2 deletions packages/liquid-html-parser/src/document/tokenizer.test.ts
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
import { describe, expect, it } from 'vitest';
import { tokenize, TokenType } from './tokenizer';
import type { Token } from './tokenizer';
import { tokenize, tokenizeWithoutFastPath, TokenType } from './tokenizer';
import type { Token, TokenizeOptions } from './tokenizer';

/** Strip the trailing EndOfInput token for cleaner assertions. */
function tokens(source: string): Token[] {
Expand Down Expand Up @@ -497,6 +497,159 @@ describe('Unit: document-tokenizer', () => {
});
});

describe('text runs end at the next token', () => {
it('ends a Liquid tag body at a -%} preceded by another -', () => {
const source = '{% if a--%}';
expect(tokens(source)).toMatchObject([
{ type: TokenType.LiquidTagOpen, start: 0, end: 2 },
{ type: TokenType.Text, start: 2, end: 8 },
{ type: TokenType.LiquidTagClose, start: 8, end: 11 },
]);
assertTokenInvariants(source);
});

it('keeps %} as text inside a Liquid drop and ends it at -}}', () => {
const source = '{{ "%}" -}}';
expect(tokens(source)).toMatchObject([
{ type: TokenType.LiquidVariableOutputOpen, start: 0, end: 2 },
{ type: TokenType.Text, start: 2, end: 8 },
{ type: TokenType.LiquidVariableOutputClose, start: 8, end: 11 },
]);
assertTokenInvariants(source);
});

it('ends a curly-quoted value on its partner, not on a straight quote', () => {
const source = '<a b=\u201cx"y\u201d c>';
expect(tokens(source)).toMatchObject([
{ type: TokenType.HtmlTagOpen, start: 0, end: 1 },
{ type: TokenType.Text, start: 1, end: 4 },
{ type: TokenType.HtmlEquals, start: 4, end: 5 },
{ type: TokenType.HtmlQuoteOpen, start: 5, end: 6 },
{ type: TokenType.Text, start: 6, end: 9 },
{ type: TokenType.HtmlQuoteClose, start: 9, end: 10 },
{ type: TokenType.Text, start: 10, end: 12 },
{ type: TokenType.HtmlTagClose, start: 12, end: 13 },
]);
assertTokenInvariants(source);
});

it('finds --> and <!-- in the middle of text', () => {
const source = 'a-b-->c<!--d';
expect(tokens(source)).toMatchObject([
{ type: TokenType.Text, start: 0, end: 3 },
{ type: TokenType.HtmlCommentClose, start: 3, end: 6 },
{ type: TokenType.Text, start: 6, end: 7 },
{ type: TokenType.HtmlCommentOpen, start: 7, end: 11 },
{ type: TokenType.Text, start: 11, end: 12 },
]);
assertTokenInvariants(source);
});
});

describe('text fast path stops at every token start mid-text', () => {
// No character in `pad` can start a token in any mode, so the fast path
// skips the whole pad and must stop exactly where the token begins.
const pad = 'abc 1 ';
const quotePairs = [
['"', '"'],
["'", "'"],
['\u201c', '\u201d'],
['\u201d', '\u201c'],
['\u2018', '\u2019'],
['\u2019', '\u2018'],
];
const cases: Array<[string, string, TokenizeOptions, string[]]> = [
[
'Default',
'',
{},
['{{', '{{-', '{%', '{%-', '<!--', '-->', '<!', '</a', '</Z', '</{', '<a', '<Z', '<{'],
],
[
'HtmlTag',
'',
{ insideHtmlTag: true },
[
'{{',
'{{-',
'{%',
'{%-',
'/>',
'>',
'=',
'"',
"'",
'\u201c',
'\u201d',
'\u2018',
'\u2019',
],
],
...quotePairs.map(([open, close]): [string, string, TokenizeOptions, string[]] => [
`QuotedValue ${open}`,
'',
{ insideQuotedAttribute: open },
[...new Set(['{{', '{%', open, close])],
]),
['LiquidTag', '{% ', {}, ['%}', '-%}']],
['LiquidVariableOutput', '{{ ', {}, ['}}', '-}}']],
];

for (const [mode, prefix, options, starts] of cases) {
it.each(starts)(`${mode}: %j`, (start) => {
const offset = prefix.length + pad.length;
const result = tokenize(prefix + pad + start + pad, options);
expect(result.some((t) => t.type !== TokenType.Text && t.start === offset)).toBe(true);
});
}
});

describe('matches tokenizeWithoutFastPath', () => {
// Every string of up to 4 of these characters, in every entry state. The
// set covers each character a mode's token can start with, plus near
// misses (`!`, `\u201a`) and plain text.
const alphabet = [...'{}%-<>/=!"\'\u201c\u201d\u2018\u2019\u201a a\n'];
let sources = [''];
for (let length = 1, previous = ['']; length <= 4; length++) {
previous = previous.flatMap((s) => alphabet.map((c) => s + c));
sources = sources.concat(previous);
}

const entryStates: Array<[string, string, TokenizeOptions]> = [
['document start', '', {}],
['skipFrontmatter', '', { skipFrontmatter: true }],
['insideHtmlTag', '', { insideHtmlTag: true }],
...['"', "'", '\u201c', '\u201d', '\u2018', '\u2019'].map(
(quote): [string, string, TokenizeOptions] => [
`insideQuotedAttribute ${quote}`,
'',
{ insideQuotedAttribute: quote },
],
),
['inside a Liquid tag', '{% ', {}],
['inside a Liquid output', '{{ ', {}],
];

function sameTokens(a: Token[], b: Token[]): boolean {
return (
a.length === b.length &&
a.every((t, i) => t.type === b[i].type && t.start === b[i].start && t.end === b[i].end)
);
}

it.each(entryStates)('%s', (_name, prefix, options) => {
const mismatches: string[] = [];
for (const source of sources) {
const input = prefix + source;
if (!sameTokens(tokenize(input, options), tokenizeWithoutFastPath(input, options))) {
mismatches.push(input);
if (mismatches.length === 10) break;
}
}
expect(mismatches).toEqual([]);
});
});

describe('structural invariants', () => {
const cases = [
'{{ x }}',
Expand Down
129 changes: 120 additions & 9 deletions packages/liquid-html-parser/src/document/tokenizer.ts
Original file line number Diff line number Diff line change
Expand Up @@ -62,6 +62,25 @@ export interface TokenizeOptions {
}

export function tokenize(source: string, options: TokenizeOptions = {}): Token[] {
return tokenizeWith(source, options, nextTextCandidate);
}

/**
* `tokenize` without the text fast path: plain text advances one character at
* a time. Test-only reference for checking that the fast path never skips over
* a token start. Not exported from the package.
*/
export function tokenizeWithoutFastPath(source: string, options: TokenizeOptions = {}): Token[] {
return tokenizeWith(source, options, (_source, from) => from);
}

type NextTextCandidate = (source: string, from: number, mode: Mode, quoteChar: string) => number;

function tokenizeWith(
source: string,
options: TokenizeOptions,
nextCandidate: NextTextCandidate,
): Token[] {
const tokens: Token[] = [];
const modeStack: Mode[] = [];
let mode = Mode.Default as Mode;
Expand Down Expand Up @@ -169,7 +188,7 @@ export function tokenize(source: string, options: TokenizeOptions = {}): Token[]
popMode();
} else {
startText();
pos++;
pos = nextCandidate(source, pos + 1, mode, quoteChar);
}
break;
}
Expand All @@ -183,7 +202,7 @@ export function tokenize(source: string, options: TokenizeOptions = {}): Token[]
popMode();
} else {
startText();
pos++;
pos = nextCandidate(source, pos + 1, mode, quoteChar);
}
break;
}
Expand All @@ -208,25 +227,23 @@ export function tokenize(source: string, options: TokenizeOptions = {}): Token[]
}

if (match('</')) {
const after = ch(2);
if (/[a-zA-Z]/.test(after) || after === '{') {
if (isTagNameStart(source.charCodeAt(pos + 2))) {
emit(TokenType.HtmlCloseTagOpen, 2);
pushMode(Mode.HtmlTag);
continue;
}
}

if (ch(0) === '<') {
const after = ch(1);
if (/[a-zA-Z]/.test(after) || after === '{') {
if (isTagNameStart(source.charCodeAt(pos + 1))) {
emit(TokenType.HtmlTagOpen, 1);
pushMode(Mode.HtmlTag);
continue;
}
}

startText();
pos++;
pos = nextCandidate(source, pos + 1, mode, quoteChar);
break;
}

Expand Down Expand Up @@ -269,7 +286,7 @@ export function tokenize(source: string, options: TokenizeOptions = {}): Token[]
}

startText();
pos++;
pos = nextCandidate(source, pos + 1, mode, quoteChar);
break;
}

Expand All @@ -286,7 +303,7 @@ export function tokenize(source: string, options: TokenizeOptions = {}): Token[]
}

startText();
pos++;
pos = nextCandidate(source, pos + 1, mode, quoteChar);
break;
}

Expand All @@ -300,6 +317,100 @@ export function tokenize(source: string, options: TokenizeOptions = {}): Token[]
return tokens;
}

/*
* Text fast path: most characters cannot start (or close) a token in the
* current mode. Each helper returns the first index >= `from` where the
* mode's `match()` checks could succeed, so the run of plain text before it is
* consumed in one step. Returning a superset of real token starts is safe: the
* main loop re-checks that position and treats a non-match as text. Missing a
* token start is not, so a new token type needs its first character added to
* its mode's helper (tokenizer.test.ts compares against
* `tokenizeWithoutFastPath` to catch this).
*/

function nextTextCandidate(source: string, from: number, mode: Mode, quoteChar: string): number {
switch (mode) {
case Mode.Default:
return nextDefaultCandidate(source, from);
case Mode.HtmlTag:
return nextHtmlTagCandidate(source, from);
case Mode.QuotedValue:
return nextQuotedValueCandidate(source, from, quoteChar);
case Mode.LiquidTag:
return nextLiquidCloseCandidate(source, from, '%}');
case Mode.LiquidVariableOutput:
return nextLiquidCloseCandidate(source, from, '}}');
default:
return assertNever(mode);
}
}

const CHAR_DOUBLE_QUOTE = 0x22; // "
const CHAR_SINGLE_QUOTE = 0x27; // '
const CHAR_DASH = 0x2d; // -
const CHAR_SLASH = 0x2f; // /
const CHAR_LESS_THAN = 0x3c; // <
const CHAR_EQUALS = 0x3d; // =
const CHAR_GREATER_THAN = 0x3e; // >
const CHAR_OPEN_BRACE = 0x7b; // {
const CHAR_LEFT_SINGLE_CURLY_QUOTE = 0x2018; // ‘
const CHAR_RIGHT_DOUBLE_CURLY_QUOTE = 0x201d; // ”

/** Default mode tokens start with `{` (Liquid), `<` (HTML), or `-` (`-->`). */
function nextDefaultCandidate(source: string, from: number): number {
for (let i = from; i < source.length; i++) {
const c = source.charCodeAt(i);
if (c === CHAR_OPEN_BRACE || c === CHAR_LESS_THAN || c === CHAR_DASH) return i;
}
return source.length;
}

/** HtmlTag mode tokens start with `{`, `/`, `>`, `=`, or a straight/curly quote. */
function nextHtmlTagCandidate(source: string, from: number): number {
for (let i = from; i < source.length; i++) {
const c = source.charCodeAt(i);
if (
c === CHAR_OPEN_BRACE ||
c === CHAR_SLASH ||
c === CHAR_GREATER_THAN ||
c === CHAR_EQUALS ||
c === CHAR_DOUBLE_QUOTE ||
c === CHAR_SINGLE_QUOTE ||
(c >= CHAR_LEFT_SINGLE_CURLY_QUOTE && c <= CHAR_RIGHT_DOUBLE_CURLY_QUOTE)
) {
return i;
}
}
return source.length;
}

/** QuotedValue mode tokens start with `{` or either quote of the open pair. */
function nextQuotedValueCandidate(source: string, from: number, quote: string): number {
const open = quote.charCodeAt(0);
const close = closingQuoteFor(quote).charCodeAt(0);
for (let i = from; i < source.length; i++) {
const c = source.charCodeAt(i);
if (c === CHAR_OPEN_BRACE || c === open || c === close) return i;
}
return source.length;
}

/**
* Liquid tag/output bodies only end at `%}`/`}}`, optionally preceded by `-`.
* Any `-%}` match contains a `%}` one character later, so the earliest close
* is at the first `%}` or the `-` immediately before it.
*/
function nextLiquidCloseCandidate(source: string, from: number, close: '%}' | '}}'): number {
const i = source.indexOf(close, from);
if (i === -1) return source.length;
return i > from && source.charCodeAt(i - 1) === CHAR_DASH ? i - 1 : i;
}

/** `[a-zA-Z{]`: what may follow `<` or `</` to open an HTML tag. */
function isTagNameStart(c: number): boolean {
return (c >= 0x61 && c <= 0x7a) || (c >= 0x41 && c <= 0x5a) || c === CHAR_OPEN_BRACE;
}

enum Mode {
Default = 'Default',
HtmlTag = 'HtmlTag',
Expand Down
Loading