mirror of
https://github.com/harttle/liquidjs.git
synced 2026-09-16 21:00:40 -07:00
perf: introduce AST to avoid reparse
This commit is contained in:
+195
-193
@@ -1,158 +1,158 @@
|
||||
import { whiteSpaceCtrl } from './whitespace-ctrl'
|
||||
import { Substr } from './substr'
|
||||
import { NumberToken } from '../tokens/number-token'
|
||||
import { WordToken } from '../tokens/word-token'
|
||||
import { literalValues } from '../util/literal'
|
||||
import { LiteralToken } from '../tokens/literal-token'
|
||||
import { OperatorToken } from '../tokens/operator-token'
|
||||
import { PropertyAccessToken } from '../tokens/property-access-token'
|
||||
import { assert } from '../util/assert'
|
||||
import { TopLevelToken } from '../tokens/toplevel-token'
|
||||
import { FilterArg } from './filter-arg'
|
||||
import { FilterToken } from './filter-token'
|
||||
import { FilterToken } from '../tokens/filter-token'
|
||||
import { HashToken } from '../tokens/hash-token'
|
||||
import { QuotedToken } from '../tokens/quoted-token'
|
||||
import { ellipsis } from '../util/underscore'
|
||||
import { HTMLToken } from './html-token'
|
||||
import { TagToken } from './tag-token'
|
||||
import { Token } from './token'
|
||||
import { OutputToken } from './output-token'
|
||||
import { HTMLToken } from '../tokens/html-token'
|
||||
import { TagToken } from '../tokens/tag-token'
|
||||
import { Token } from '../tokens/token'
|
||||
import { RangeToken } from '../tokens/range-token'
|
||||
import { ValueToken } from '../tokens/value-token'
|
||||
import { OutputToken } from '../tokens/output-token'
|
||||
import { TokenizationError } from '../util/error'
|
||||
import { NormalizedFullOptions, defaultOptions } from '../liquid-options'
|
||||
|
||||
// bitmask character types to boost performance
|
||||
// generated by bin/char-types.js
|
||||
const TYPES = '00000000044004000000000000000000428000080000010011111111110022210111111111111111111111111110000101111111111111111111111111100000'
|
||||
const VARIABLE = 1
|
||||
const OPERATOR = 2
|
||||
const BLANK = 4
|
||||
const QUOTE = 8
|
||||
import { TYPES, QUOTE, BLANK, VARIABLE } from '../util/character'
|
||||
import { matchOperator } from './match-operator'
|
||||
|
||||
export class Tokenizer {
|
||||
private p = 0
|
||||
private N: number
|
||||
private line = 1
|
||||
private col = 1
|
||||
p = 0
|
||||
N: number
|
||||
constructor (
|
||||
private input: string,
|
||||
private file: string = '',
|
||||
private options: NormalizedFullOptions = defaultOptions
|
||||
private file: string = ''
|
||||
) {
|
||||
this.N = input.length
|
||||
}
|
||||
|
||||
* readExpression (): IterableIterator<string> {
|
||||
* readExpression (): IterableIterator<Token> {
|
||||
const operand = this.readValue()
|
||||
if (!operand) return
|
||||
|
||||
yield operand
|
||||
|
||||
while (this.p < this.N) {
|
||||
const operator = this.readOperator()
|
||||
if (!operator) return
|
||||
|
||||
const operand = this.readValue()
|
||||
if (operand.size()) {
|
||||
yield operand.toString()
|
||||
continue
|
||||
}
|
||||
this.readBlank()
|
||||
const operator = new Substr(this.input, this.p)
|
||||
while (OPERATOR & this.peekType()) operator.end = this.read()
|
||||
if (operator.size()) {
|
||||
yield operator.toString()
|
||||
continue
|
||||
}
|
||||
this.read()
|
||||
if (!operand) return
|
||||
|
||||
yield operator
|
||||
yield operand
|
||||
}
|
||||
}
|
||||
|
||||
readFilterTokens (): FilterToken[] {
|
||||
readOperator (): OperatorToken | undefined {
|
||||
this.skipBlank()
|
||||
const end = matchOperator(this.input, this.p, this.p + 8)
|
||||
if (end === -1) return
|
||||
return new OperatorToken(this.input, this.p, (this.p = end), this.file)
|
||||
}
|
||||
readFilters (): FilterToken[] {
|
||||
const filters = []
|
||||
while (true) {
|
||||
const filter = this.readFilterToken()
|
||||
const filter = this.readFilter()
|
||||
if (!filter) return filters
|
||||
filters.push(filter)
|
||||
}
|
||||
}
|
||||
|
||||
// | foo
|
||||
// | foo: a
|
||||
// | foo: a, b
|
||||
// | foo: a, b: 1
|
||||
readFilterToken (): FilterToken | null {
|
||||
readFilter (): FilterToken | null {
|
||||
this.readTo('|')
|
||||
const begin = this.p
|
||||
const name = this.readVariable().toString()
|
||||
if (!name) return null
|
||||
const name = this.readWord()
|
||||
if (!name.size()) return null
|
||||
const args = []
|
||||
this.readBlank()
|
||||
this.skipBlank()
|
||||
if (this.peek() === ':') {
|
||||
do {
|
||||
this.read()
|
||||
++this.p
|
||||
const arg = this.readFilterArg()
|
||||
arg && args.push(arg)
|
||||
while (this.p < this.N && this.peek() !== ',' && this.peek() !== '|') this.read()
|
||||
while (this.p < this.N && this.peek() !== ',' && this.peek() !== '|') ++this.p
|
||||
} while (this.peek() === ',')
|
||||
}
|
||||
const raw = this.input.slice(begin, this.p)
|
||||
return new FilterToken(name, args, raw, this.input, this.line, this.col, this.file)
|
||||
return new FilterToken(name.getText(), args, this.input, begin, this.p, this.file)
|
||||
}
|
||||
|
||||
readFilterArg (): FilterArg | null {
|
||||
readFilterArg (): FilterArg | undefined {
|
||||
const key = this.readValue()
|
||||
if (!key.size()) return null
|
||||
this.readBlank()
|
||||
if (this.peek() === ':') {
|
||||
this.read()
|
||||
return [key.toString(), this.readValue().toString()]
|
||||
}
|
||||
return key.toString()
|
||||
if (!key) return
|
||||
this.skipBlank()
|
||||
if (this.peek() !== ':') return key
|
||||
++this.p
|
||||
const value = this.readValue()
|
||||
return [key.getText(), value]
|
||||
}
|
||||
|
||||
readTokens (): Token[] {
|
||||
const tokens: Token[] = []
|
||||
readTopLevelTokens (options: NormalizedFullOptions = defaultOptions): TopLevelToken[] {
|
||||
const tokens: TopLevelToken[] = []
|
||||
while (this.p < this.N) {
|
||||
const token = this.readToken()
|
||||
const token = this.readTopLevelToken(options)
|
||||
tokens.push(token)
|
||||
}
|
||||
whiteSpaceCtrl(tokens, this.options)
|
||||
whiteSpaceCtrl(tokens, options)
|
||||
return tokens
|
||||
}
|
||||
|
||||
readToken (): Token {
|
||||
const { tagDelimiterLeft, outputDelimiterLeft } = this.options
|
||||
if (this.matchWord(tagDelimiterLeft)) return this.readTagToken()
|
||||
if (this.matchWord(outputDelimiterLeft)) return this.readOutputToken()
|
||||
return this.readHTMLToken()
|
||||
readTopLevelToken (options: NormalizedFullOptions): TopLevelToken {
|
||||
const { tagDelimiterLeft, outputDelimiterLeft } = options
|
||||
if (this.matchWord(tagDelimiterLeft)) return this.readTagToken(options)
|
||||
if (this.matchWord(outputDelimiterLeft)) return this.readOutputToken(options)
|
||||
return this.readHTMLToken(options)
|
||||
}
|
||||
|
||||
readHTMLToken (): HTMLToken {
|
||||
const html = new Substr(this.input, this.p)
|
||||
readHTMLToken (options: NormalizedFullOptions): HTMLToken {
|
||||
const begin = this.p
|
||||
while (this.p < this.N) {
|
||||
const { tagDelimiterLeft, outputDelimiterLeft } = this.options
|
||||
const { tagDelimiterLeft, outputDelimiterLeft } = options
|
||||
if (this.matchWord(tagDelimiterLeft)) break
|
||||
if (this.matchWord(outputDelimiterLeft)) break
|
||||
html.end = this.read()
|
||||
++this.p
|
||||
}
|
||||
return new HTMLToken(html.toString(), this.input, this.line, this.col, this.file)
|
||||
return new HTMLToken(this.input, begin, this.p, this.file)
|
||||
}
|
||||
|
||||
readTagToken (): TagToken {
|
||||
const { line, col, file, input, options } = this
|
||||
const { tagDelimiterLeft, tagDelimiterRight } = options
|
||||
const buffer = this.readTo(tagDelimiterRight).toString()
|
||||
if (!this.reverseMatchWord(tagDelimiterRight, buffer)) {
|
||||
throw new TokenizationError(
|
||||
`tag "${ellipsis(buffer, 16)}" not closed`,
|
||||
new Token(buffer, input, line, col, file)
|
||||
)
|
||||
readTagToken (options: NormalizedFullOptions): TagToken {
|
||||
const { file, input } = this
|
||||
const { tagDelimiterRight } = options
|
||||
const begin = this.p
|
||||
if (this.readTo(tagDelimiterRight) === -1) {
|
||||
this.mkError(`tag "${this.ellipsis(begin)}" not closed`, begin)
|
||||
}
|
||||
const value = buffer.slice(tagDelimiterLeft.length, -tagDelimiterRight.length)
|
||||
return new TagToken(buffer, value, input, line, col, options, file)
|
||||
return new TagToken(input, begin, this.p, options, file)
|
||||
}
|
||||
|
||||
readOutputToken (): OutputToken {
|
||||
const { line, col, file, input, options } = this
|
||||
const { outputDelimiterLeft, outputDelimiterRight } = options
|
||||
const buffer = this.readTo(outputDelimiterRight).toString()
|
||||
if (!this.reverseMatchWord(outputDelimiterRight, buffer)) {
|
||||
throw new TokenizationError(
|
||||
`output "${ellipsis(buffer, 16)}" not closed`,
|
||||
new Token(buffer, input, line, col, file)
|
||||
)
|
||||
readOutputToken (options: NormalizedFullOptions): OutputToken {
|
||||
const { file, input } = this
|
||||
const { outputDelimiterRight } = options
|
||||
const begin = this.p
|
||||
if (this.readTo(outputDelimiterRight) === -1) {
|
||||
this.mkError(`output "${this.ellipsis(begin)}" not closed`, begin)
|
||||
}
|
||||
const value = buffer.slice(outputDelimiterLeft.length, -outputDelimiterRight.length)
|
||||
return new OutputToken(buffer, value, input, line, col, options, file)
|
||||
return new OutputToken(input, begin, this.p, options, file)
|
||||
}
|
||||
|
||||
readVariable (): Substr {
|
||||
this.readBlank()
|
||||
const ans = new Substr(this.input, this.p)
|
||||
while (this.peekType() & VARIABLE) ans.end = this.read()
|
||||
return ans
|
||||
mkError (msg: string, begin: number) {
|
||||
throw new TokenizationError(msg, new WordToken(this.input, begin, this.N, this.file))
|
||||
}
|
||||
|
||||
ellipsis (begin: number = this.p) {
|
||||
return ellipsis(this.input.slice(begin), 16)
|
||||
}
|
||||
|
||||
readWord (): WordToken { // rename to identifier
|
||||
this.skipBlank()
|
||||
const begin = this.p
|
||||
while (this.peekType() & VARIABLE) ++this.p
|
||||
return new WordToken(this.input, begin, this.p, this.file)
|
||||
}
|
||||
|
||||
readHashes () {
|
||||
@@ -164,133 +164,135 @@ export class Tokenizer {
|
||||
}
|
||||
}
|
||||
|
||||
readHash () {
|
||||
this.readBlank()
|
||||
if (this.peek() === ',') this.read()
|
||||
const name = this.readVariable().toString()
|
||||
if (!name) return null
|
||||
readHash (): HashToken | undefined {
|
||||
this.skipBlank()
|
||||
if (this.peek() === ',') ++this.p
|
||||
const begin = this.p
|
||||
const name = this.readWord()
|
||||
if (!name.size()) return
|
||||
let value
|
||||
|
||||
this.readBlank()
|
||||
let value = ''
|
||||
this.skipBlank()
|
||||
if (this.peek() === ':') {
|
||||
this.read()
|
||||
value = this.readValue().toString()
|
||||
++this.p
|
||||
value = this.readValue()
|
||||
}
|
||||
return [name, value]
|
||||
return new HashToken(this.input, begin, this.p, name, value, this.file)
|
||||
}
|
||||
|
||||
readPropertyAccess (): Substr {
|
||||
this.readBlank()
|
||||
const ans = new Substr(this.input, this.p)
|
||||
let nested = 0
|
||||
remaining () {
|
||||
return this.input.slice(this.p)
|
||||
}
|
||||
|
||||
advance (i = 1) {
|
||||
this.p += i
|
||||
}
|
||||
|
||||
end () {
|
||||
return this.p >= this.N
|
||||
}
|
||||
|
||||
readTo (end: string): number {
|
||||
while (this.p < this.N) {
|
||||
const c = this.peek()
|
||||
const code = this.peekType()
|
||||
if (c === '[') {
|
||||
this.read()
|
||||
ans.end = this.readValue().end
|
||||
nested++
|
||||
} else if (c === ']') {
|
||||
if (!nested) break
|
||||
ans.end = this.read()
|
||||
nested--
|
||||
} else if (c === '.') {
|
||||
if (this.peekType(1) & VARIABLE) {
|
||||
this.read()
|
||||
ans.end = this.readVariable().end
|
||||
} else break
|
||||
} else if (code & VARIABLE) {
|
||||
ans.end = this.read()
|
||||
} else {
|
||||
if (nested) this.read()
|
||||
else break
|
||||
}
|
||||
++this.p
|
||||
if (this.reverseMatchWord(end)) return this.p
|
||||
}
|
||||
return ans
|
||||
return -1
|
||||
}
|
||||
readTo (end: string): Substr {
|
||||
const ans = new Substr(this.input, this.p)
|
||||
while (this.p < this.N) {
|
||||
ans.end = this.read()
|
||||
if (this.reverseMatchWord(end)) break
|
||||
|
||||
readValue (): ValueToken | undefined {
|
||||
const value = this.readQuoted() || this.readRange()
|
||||
if (value) return value
|
||||
|
||||
const variable = this.readWord()
|
||||
if (!variable.size()) return
|
||||
|
||||
let isNumber = variable.isNumber(true)
|
||||
const props: (QuotedToken | WordToken)[] = []
|
||||
while (true) {
|
||||
if (this.peek() === '[') {
|
||||
isNumber = false
|
||||
this.p++
|
||||
const prop = this.readValue() || new WordToken(this.input, this.p, this.p, this.file)
|
||||
this.readTo(']')
|
||||
props.push(prop)
|
||||
} else if (this.peek() === '.' && this.peek(1) !== '.') { // skip range syntax
|
||||
this.p++
|
||||
const prop = this.readWord()
|
||||
if (!prop.size()) break
|
||||
if (!prop.isNumber()) isNumber = false
|
||||
props.push(prop)
|
||||
} else break
|
||||
}
|
||||
return ans
|
||||
if (!props.length && literalValues.hasOwnProperty(variable.content)) {
|
||||
return new LiteralToken(this.input, variable.begin, variable.end, this.file)
|
||||
}
|
||||
if (isNumber) return new NumberToken(variable, props[0] as WordToken)
|
||||
return new PropertyAccessToken(variable, props, this.p)
|
||||
}
|
||||
readValue (): Substr {
|
||||
let val = this.readQuoted()
|
||||
if (val.size()) return val
|
||||
val = this.readBoolean()
|
||||
if (val.size()) return val
|
||||
val = this.readPropertyAccess()
|
||||
if (val.size()) return val
|
||||
return this.readRange()
|
||||
|
||||
readRange (): RangeToken | undefined {
|
||||
this.skipBlank()
|
||||
const begin = this.p
|
||||
if (this.peek() !== '(') return
|
||||
++this.p
|
||||
const lhs = this.readValueOrThrow()
|
||||
this.p += 2
|
||||
const rhs = this.readValueOrThrow()
|
||||
++this.p
|
||||
return new RangeToken(this.input, begin, this.p, lhs, rhs, this.file)
|
||||
}
|
||||
readRange (): Substr {
|
||||
this.readBlank()
|
||||
const ans = new Substr(this.input, this.p)
|
||||
if (this.peek() !== '(') return ans
|
||||
this.read()
|
||||
this.readValue()
|
||||
this.read(2)
|
||||
this.readValue()
|
||||
ans.end = this.read()
|
||||
return ans
|
||||
|
||||
readValueOrThrow (): ValueToken {
|
||||
const value = this.readValue()
|
||||
assert(value, () => `unexpected token "${this.ellipsis()}", value expected`)
|
||||
return value!
|
||||
}
|
||||
readBoolean (): Substr {
|
||||
this.readBlank()
|
||||
const ans = new Substr(this.input, this.p)
|
||||
if (this.matchWord('true') && !(this.peekType(4) & VARIABLE)) ans.end = this.read(4)
|
||||
else if (this.matchWord('false') && !(this.peekType(5) & VARIABLE)) ans.end = this.read(5)
|
||||
return ans
|
||||
}
|
||||
readQuoted (): Substr {
|
||||
this.readBlank()
|
||||
const ans = new Substr(this.input, this.p)
|
||||
if (!(this.peekType() & QUOTE)) return ans
|
||||
ans.end = this.read()
|
||||
|
||||
readQuoted (): QuotedToken | undefined {
|
||||
this.skipBlank()
|
||||
const begin = this.p
|
||||
if (!(this.peekType() & QUOTE)) return
|
||||
++this.p
|
||||
let escaped = false
|
||||
while (this.p < this.N) {
|
||||
ans.end = this.read()
|
||||
if (ans.last() === ans.first() && !escaped) break
|
||||
++this.p
|
||||
if (this.input[this.p - 1] === this.input[begin] && !escaped) break
|
||||
if (escaped) escaped = false
|
||||
else if (ans.last() === '\\') escaped = true
|
||||
else if (this.input[this.p - 1] === '\\') escaped = true
|
||||
}
|
||||
return ans
|
||||
return new QuotedToken(this.input, begin, this.p, this.file)
|
||||
}
|
||||
read (n = 1): number {
|
||||
if (n > 1) this.read(n - 1)
|
||||
const c = this.input[this.p++]
|
||||
if (c === '\n') {
|
||||
this.line++
|
||||
this.col = 1
|
||||
} else {
|
||||
this.col++
|
||||
}
|
||||
return this.p
|
||||
|
||||
readFileName (): WordToken {
|
||||
const begin = this.p
|
||||
while (!(this.peekType() & BLANK) && this.peek() !== ',' && this.p < this.N) this.p++
|
||||
return new WordToken(this.input, begin, this.p, this.file)
|
||||
}
|
||||
|
||||
matchWord (word: string) {
|
||||
for (let i = 0; i < word.length; i++) {
|
||||
if (word[i] !== this.input[this.p + i]) return false
|
||||
}
|
||||
return true
|
||||
}
|
||||
reverseMatchWord (word: string, buffer?: string) {
|
||||
const str = buffer || this.input
|
||||
const end = buffer === undefined ? this.p : buffer.length
|
||||
|
||||
reverseMatchWord (word: string) {
|
||||
for (let i = 0; i < word.length; i++) {
|
||||
if (word[word.length - 1 - i] !== str[end - 1 - i]) return false
|
||||
if (word[word.length - 1 - i] !== this.input[this.p - 1 - i]) return false
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
peekType (n = 0) {
|
||||
return +TYPES[this.input.charCodeAt(this.p + n)]
|
||||
return TYPES[this.input.charCodeAt(this.p + n)]
|
||||
}
|
||||
|
||||
peek (n = 0) {
|
||||
return this.input[this.p + n]
|
||||
}
|
||||
readBlank () {
|
||||
let ans = ''
|
||||
while (this.peekType() & BLANK) ans += this.read()
|
||||
return ans
|
||||
|
||||
skipBlank () {
|
||||
while (this.peekType() & BLANK) ++this.p
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user