mirror of
https://github.com/Shopify/liquid.git
synced 2026-09-12 23:40:45 -07:00
replace FullToken regex with manual byte parsing in parse_for_document
This commit is contained in:
@@ -12,6 +12,57 @@ module Liquid
|
|||||||
TAGSTART = "{%"
|
TAGSTART = "{%"
|
||||||
VARSTART = "{{"
|
VARSTART = "{{"
|
||||||
|
|
||||||
|
# Fast manual tag token parser - avoids regex MatchData allocation
|
||||||
|
# Parses "{%[-] tag_name markup [-]%}" and returns [pre_ws, tag_name, post_ws, markup] or nil
|
||||||
|
def self.parse_tag_token(token)
|
||||||
|
# token starts with "{%"
|
||||||
|
pos = 2
|
||||||
|
len = token.length
|
||||||
|
|
||||||
|
# skip optional whitespace control '-'
|
||||||
|
pos += 1 if pos < len && token.getbyte(pos) == 45 # '-'
|
||||||
|
|
||||||
|
# capture pre-whitespace (for line number counting)
|
||||||
|
ws_start = pos
|
||||||
|
while pos < len
|
||||||
|
b = token.getbyte(pos)
|
||||||
|
break unless b == 32 || b == 9 || b == 10 || b == 13 # space, tab, \n, \r
|
||||||
|
pos += 1
|
||||||
|
end
|
||||||
|
pre_ws = token.byteslice(ws_start, pos - ws_start)
|
||||||
|
|
||||||
|
# parse tag name: # or \w+
|
||||||
|
name_start = pos
|
||||||
|
if pos < len && token.getbyte(pos) == 35 # '#'
|
||||||
|
pos += 1
|
||||||
|
else
|
||||||
|
while pos < len
|
||||||
|
b = token.getbyte(pos)
|
||||||
|
break unless (b >= 97 && b <= 122) || (b >= 65 && b <= 90) || (b >= 48 && b <= 57) || b == 95 # a-z, A-Z, 0-9, _
|
||||||
|
pos += 1
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return nil if pos == name_start # no tag name found
|
||||||
|
tag_name = token.byteslice(name_start, pos - name_start)
|
||||||
|
|
||||||
|
# capture post-whitespace
|
||||||
|
post_ws_start = pos
|
||||||
|
while pos < len
|
||||||
|
b = token.getbyte(pos)
|
||||||
|
break unless b == 32 || b == 9 || b == 10 || b == 13
|
||||||
|
pos += 1
|
||||||
|
end
|
||||||
|
post_ws = token.byteslice(post_ws_start, pos - post_ws_start)
|
||||||
|
|
||||||
|
# the rest is markup, up to optional '-' and '%}'
|
||||||
|
# token ends with '%}' (guaranteed by tokenizer)
|
||||||
|
markup_end = len - 2
|
||||||
|
markup_end -= 1 if markup_end > pos && token.getbyte(markup_end - 1) == 45 # trailing '-'
|
||||||
|
markup = pos >= markup_end ? "" : token.byteslice(pos, markup_end - pos)
|
||||||
|
|
||||||
|
[pre_ws, tag_name, post_ws, markup]
|
||||||
|
end
|
||||||
|
|
||||||
attr_reader :nodelist
|
attr_reader :nodelist
|
||||||
|
|
||||||
def initialize
|
def initialize
|
||||||
@@ -130,16 +181,16 @@ module Liquid
|
|||||||
case
|
case
|
||||||
when token.start_with?(TAGSTART)
|
when token.start_with?(TAGSTART)
|
||||||
whitespace_handler(token, parse_context)
|
whitespace_handler(token, parse_context)
|
||||||
unless token =~ FullToken
|
parsed = BlockBody.parse_tag_token(token)
|
||||||
|
unless parsed
|
||||||
return handle_invalid_tag_token(token, parse_context, &block)
|
return handle_invalid_tag_token(token, parse_context, &block)
|
||||||
end
|
end
|
||||||
tag_name = Regexp.last_match(2)
|
pre_ws, tag_name, post_ws, markup = parsed
|
||||||
markup = Regexp.last_match(4)
|
|
||||||
|
|
||||||
if parse_context.line_number
|
if parse_context.line_number
|
||||||
# newlines inside the tag should increase the line number,
|
# newlines inside the tag should increase the line number,
|
||||||
# particularly important for multiline {% liquid %} tags
|
# particularly important for multiline {% liquid %} tags
|
||||||
parse_context.line_number += Regexp.last_match(1).count("\n") + Regexp.last_match(3).count("\n")
|
parse_context.line_number += pre_ws.count("\n") + post_ws.count("\n")
|
||||||
end
|
end
|
||||||
|
|
||||||
if tag_name == 'liquid'
|
if tag_name == 'liquid'
|
||||||
|
|||||||
Reference in New Issue
Block a user