mirror of
https://github.com/Shopify/liquid.git
synced 2026-10-04 01:25:14 -07:00
BlockBody: byte-walk tag tokens instead of FullToken regex
Add try_parse_tag_token that parses {%...%} tag tokens using
getbyte/byteslice + ByteTables lookup arrays instead of the
FullToken regex with 4 capture groups. Allocates only the 2
strings needed (tag_name, markup) vs 4+ from regex captures.
Uses ByteTables::WORD (no hyphen) for tag name scanning,
matching TagName = /#|\w+/ exactly. Falls back to FullToken
regex when the fast path returns nil.
This commit is contained in:
@@ -130,16 +130,21 @@ module Liquid
|
|||||||
case
|
case
|
||||||
when token.start_with?(TAGSTART)
|
when token.start_with?(TAGSTART)
|
||||||
whitespace_handler(token, parse_context)
|
whitespace_handler(token, parse_context)
|
||||||
unless token =~ FullToken
|
# rubocop:disable Metrics/BlockNesting
|
||||||
return handle_invalid_tag_token(token, parse_context, &block)
|
fast = try_parse_tag_token(token)
|
||||||
end
|
if fast
|
||||||
|
tag_name, markup, newlines = fast
|
||||||
|
elsif token =~ FullToken
|
||||||
tag_name = Regexp.last_match(2)
|
tag_name = Regexp.last_match(2)
|
||||||
markup = Regexp.last_match(4)
|
markup = Regexp.last_match(4)
|
||||||
|
newlines = parse_context.line_number ? Regexp.last_match(1).count("\n") + Regexp.last_match(3).count("\n") : 0
|
||||||
|
else
|
||||||
|
return handle_invalid_tag_token(token, parse_context, &block)
|
||||||
|
end
|
||||||
|
# rubocop:enable Metrics/BlockNesting
|
||||||
|
|
||||||
if parse_context.line_number
|
if parse_context.line_number && newlines > 0
|
||||||
# newlines inside the tag should increase the line number,
|
parse_context.line_number += newlines
|
||||||
# particularly important for multiline {% liquid %} tags
|
|
||||||
parse_context.line_number += Regexp.last_match(1).count("\n") + Regexp.last_match(3).count("\n")
|
|
||||||
end
|
end
|
||||||
|
|
||||||
if tag_name == 'liquid'
|
if tag_name == 'liquid'
|
||||||
@@ -260,6 +265,77 @@ module Liquid
|
|||||||
BlockBody.raise_missing_variable_terminator(token, parse_context)
|
BlockBody.raise_missing_variable_terminator(token, parse_context)
|
||||||
end
|
end
|
||||||
|
|
||||||
|
# Fast path for parsing "{%[-] tag_name markup [-]%}" tag tokens.
|
||||||
|
# Returns [tag_name, markup, newline_count] or nil.
|
||||||
|
#
|
||||||
|
# Accepts tokens where:
|
||||||
|
# - Tag name is '#' or starts with [a-zA-Z_] followed by \w chars
|
||||||
|
# (matching TagName = /#|\w+/ exactly — no hyphens, no '?' suffix)
|
||||||
|
# - Whitespace is spaces, tabs, newlines, \r, \f, \v
|
||||||
|
# - Whitespace control dashes are at positions 2 and len-3
|
||||||
|
# Rejects (returns nil → caller falls back to FullToken regex):
|
||||||
|
# - Tokens shorter than "{%x%}" (4 bytes)
|
||||||
|
# - Tag names starting with a digit (valid in FullToken but rare)
|
||||||
|
# - Any structure the byte-walk can't confidently parse
|
||||||
|
# Fallback: nil return triggers the original `token =~ FullToken` regex
|
||||||
|
# match in parse_for_document, preserving identical behavior for any
|
||||||
|
# input the fast path doesn't handle.
|
||||||
|
def try_parse_tag_token(token)
|
||||||
|
len = token.bytesize
|
||||||
|
pos = 2 # skip "{%"
|
||||||
|
return if pos >= len
|
||||||
|
|
||||||
|
pos += 1 if token.getbyte(pos) == ByteTables::DASH
|
||||||
|
newline_count = 0
|
||||||
|
|
||||||
|
# Skip whitespace before tag name, count newlines
|
||||||
|
while pos < len
|
||||||
|
b = token.getbyte(pos)
|
||||||
|
if b == ByteTables::NEWLINE
|
||||||
|
pos += 1
|
||||||
|
newline_count += 1
|
||||||
|
elsif ByteTables::WHITESPACE[b]
|
||||||
|
pos += 1
|
||||||
|
else
|
||||||
|
break
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return if pos >= len
|
||||||
|
|
||||||
|
# Scan tag name: '#' or \w+ (matching TagName = /#|\w+/)
|
||||||
|
name_start = pos
|
||||||
|
b = token.getbyte(pos)
|
||||||
|
if b == ByteTables::HASH
|
||||||
|
pos += 1
|
||||||
|
elsif ByteTables::IDENT_START[b]
|
||||||
|
pos += 1
|
||||||
|
pos += 1 while pos < len && ByteTables::WORD[token.getbyte(pos)]
|
||||||
|
else
|
||||||
|
return
|
||||||
|
end
|
||||||
|
tag_name = token.byteslice(name_start, pos - name_start)
|
||||||
|
|
||||||
|
# Skip whitespace after tag name, count newlines
|
||||||
|
while pos < len
|
||||||
|
b = token.getbyte(pos)
|
||||||
|
if b == ByteTables::NEWLINE
|
||||||
|
pos += 1
|
||||||
|
newline_count += 1
|
||||||
|
elsif ByteTables::WHITESPACE[b]
|
||||||
|
pos += 1
|
||||||
|
else
|
||||||
|
break
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
# Markup: everything up to optional '-' before '%}'
|
||||||
|
markup_end = len - 2 # skip '%}'
|
||||||
|
markup_end -= 1 if markup_end > pos && token.getbyte(markup_end - 1) == ByteTables::DASH
|
||||||
|
markup = pos >= markup_end ? "" : token.byteslice(pos, markup_end - pos)
|
||||||
|
|
||||||
|
[tag_name, markup, newline_count]
|
||||||
|
end
|
||||||
|
|
||||||
# @deprecated Use {.raise_missing_tag_terminator} instead
|
# @deprecated Use {.raise_missing_tag_terminator} instead
|
||||||
def raise_missing_tag_terminator(token, parse_context)
|
def raise_missing_tag_terminator(token, parse_context)
|
||||||
BlockBody.raise_missing_tag_terminator(token, parse_context)
|
BlockBody.raise_missing_tag_terminator(token, parse_context)
|
||||||
|
|||||||
Reference in New Issue
Block a user