mirror of
https://github.com/Shopify/liquid.git
synced 2026-09-12 23:40:45 -07:00
Improves performance on hot parse and render paths
Splits the render loop in block_body.rb on the loop-invariant check_write condition. The common case (no resource limits) now pays zero branch cost per node. Rewrites truncatewords in standardfilters.rb to scan word positions into a flat int array and builds the result string only when truncation is confirmed. No string allocation in the common no-truncation case beyond the array itself. Simplifies rest_blank? in cursor.rb: replaces manual save/skip/restore of StringScanner position with [email protected]?(/\S/). exist? does not advance position; returns nil when no non-whitespace remains; handles EOS correctly. Removes the nl newline counter from skip_ws in cursor.rb -- all callers discarded the return value. NL now handled in the same when-branch as the other whitespace bytes.
This commit is contained in:
+22
-16
@@ -1,6 +1,5 @@
|
|||||||
# frozen_string_literal: true
|
# frozen_string_literal: true
|
||||||
|
|
||||||
|
|
||||||
module Liquid
|
module Liquid
|
||||||
class BlockBody
|
class BlockBody
|
||||||
LiquidTagToken = /\A\s*(#{TagName})\s*(.*?)\z/o
|
LiquidTagToken = /\A\s*(#{TagName})\s*(.*?)\z/o
|
||||||
@@ -122,8 +121,6 @@ module Liquid
|
|||||||
end
|
end
|
||||||
end
|
end
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
def self.blank_string?(str)
|
def self.blank_string?(str)
|
||||||
str.match?(WhitespaceOrNothing)
|
str.match?(WhitespaceOrNothing)
|
||||||
end
|
end
|
||||||
@@ -257,20 +254,30 @@ module Liquid
|
|||||||
resource_limits = context.resource_limits
|
resource_limits = context.resource_limits
|
||||||
resource_limits.increment_render_score(@nodelist.length)
|
resource_limits.increment_render_score(@nodelist.length)
|
||||||
|
|
||||||
# Check if we need per-node write score tracking
|
# Hot render loop — split on check_write so the common case (no resource
|
||||||
check_write = resource_limits.render_length_limit || resource_limits.last_capture_length
|
# limits) pays zero branch cost per node.
|
||||||
|
|
||||||
idx = 0
|
idx = 0
|
||||||
while (node = @nodelist[idx])
|
if resource_limits.render_length_limit || resource_limits.last_capture_length
|
||||||
if node.instance_of?(String)
|
while (node = @nodelist[idx])
|
||||||
output << node
|
if node.instance_of?(String)
|
||||||
else
|
output << node
|
||||||
render_node(context, output, node)
|
else
|
||||||
break if context.interrupt?
|
render_node(context, output, node)
|
||||||
|
break if context.interrupt?
|
||||||
|
end
|
||||||
|
idx += 1
|
||||||
|
resource_limits.increment_write_score(output)
|
||||||
|
end
|
||||||
|
else
|
||||||
|
while (node = @nodelist[idx])
|
||||||
|
if node.instance_of?(String)
|
||||||
|
output << node
|
||||||
|
else
|
||||||
|
render_node(context, output, node)
|
||||||
|
break if context.interrupt?
|
||||||
|
end
|
||||||
|
idx += 1
|
||||||
end
|
end
|
||||||
idx += 1
|
|
||||||
|
|
||||||
resource_limits.increment_write_score(output) if check_write
|
|
||||||
end
|
end
|
||||||
|
|
||||||
output
|
output
|
||||||
@@ -283,7 +290,6 @@ module Liquid
|
|||||||
BlockBody.render_node(context, output, node)
|
BlockBody.render_node(context, output, node)
|
||||||
end
|
end
|
||||||
|
|
||||||
|
|
||||||
def create_variable(token, parse_context)
|
def create_variable(token, parse_context)
|
||||||
len = token.bytesize
|
len = token.bytesize
|
||||||
if len >= 4 && token.getbyte(len - 1) == Cursor::RCURLY && token.getbyte(len - 2) == Cursor::RCURLY
|
if len >= 4 && token.getbyte(len - 1) == Cursor::RCURLY && token.getbyte(len - 2) == Cursor::RCURLY
|
||||||
|
|||||||
+11
-11
@@ -12,29 +12,29 @@ module Liquid
|
|||||||
# ByteTables::IDENT_START[b]
|
# ByteTables::IDENT_START[b]
|
||||||
module ByteTables
|
module ByteTables
|
||||||
# [a-zA-Z_] — valid first byte of an identifier
|
# [a-zA-Z_] — valid first byte of an identifier
|
||||||
IDENT_START = Array.new(256, false).tap { |t|
|
IDENT_START = Array.new(256, false).tap do |t|
|
||||||
(97..122).each { |b| t[b] = true } # a-z
|
(97..122).each { |b| t[b] = true } # a-z
|
||||||
(65..90).each { |b| t[b] = true } # A-Z
|
(65..90).each { |b| t[b] = true } # A-Z
|
||||||
t[95] = true # _
|
t[95] = true # _
|
||||||
}.freeze
|
end.freeze
|
||||||
|
|
||||||
# [a-zA-Z0-9_-] — valid continuation byte of an identifier
|
# [a-zA-Z0-9_-] — valid continuation byte of an identifier
|
||||||
IDENT_CONT = Array.new(256, false).tap { |t|
|
IDENT_CONT = Array.new(256, false).tap do |t|
|
||||||
(97..122).each { |b| t[b] = true } # a-z
|
(97..122).each { |b| t[b] = true } # a-z
|
||||||
(65..90).each { |b| t[b] = true } # A-Z
|
(65..90).each { |b| t[b] = true } # A-Z
|
||||||
(48..57).each { |b| t[b] = true } # 0-9
|
(48..57).each { |b| t[b] = true } # 0-9
|
||||||
t[95] = true # _
|
t[95] = true # _
|
||||||
t[45] = true # -
|
t[45] = true # -
|
||||||
}.freeze
|
end.freeze
|
||||||
|
|
||||||
# [0-9] — ASCII digit
|
# [0-9] — ASCII digit
|
||||||
DIGIT = Array.new(256, false).tap { |t|
|
DIGIT = Array.new(256, false).tap do |t|
|
||||||
(48..57).each { |b| t[b] = true }
|
(48..57).each { |b| t[b] = true }
|
||||||
}.freeze
|
end.freeze
|
||||||
|
|
||||||
# [ \t\n\r\f] — ASCII whitespace
|
# [ \t\n\v\f\r] — ASCII whitespace (mirrors Ruby's \s)
|
||||||
WHITESPACE = Array.new(256, false).tap { |t|
|
WHITESPACE = Array.new(256, false).tap do |t|
|
||||||
[32, 9, 10, 13, 12].each { |b| t[b] = true } # space, tab, \n, \r, \f
|
[32, 9, 10, 11, 12, 13].each { |b| t[b] = true } # space, tab, \n, \v, \f, \r
|
||||||
}.freeze
|
end.freeze
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
|
|||||||
+10
-15
@@ -63,27 +63,20 @@ module Liquid
|
|||||||
end
|
end
|
||||||
|
|
||||||
# ── Whitespace ──────────────────────────────────────────────────
|
# ── Whitespace ──────────────────────────────────────────────────
|
||||||
# Skip spaces/tabs/newlines/cr, return count of newlines skipped
|
# Skip spaces/tabs/newlines/cr
|
||||||
def skip_ws
|
def skip_ws
|
||||||
nl = 0
|
|
||||||
while (b = @ss.peek_byte)
|
while (b = @ss.peek_byte)
|
||||||
case b
|
case b
|
||||||
when SPACE, TAB, CR, FF then @ss.scan_byte
|
when SPACE, TAB, CR, FF, NL then @ss.scan_byte
|
||||||
when NL then @ss.scan_byte
|
|
||||||
nl += 1
|
|
||||||
else break
|
else break
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
nl
|
|
||||||
end
|
end
|
||||||
|
|
||||||
# Check if remaining bytes are all whitespace
|
# Check if remaining bytes are all whitespace (or EOS).
|
||||||
|
# exist?(/\S/) returns nil when no non-whitespace remains, without advancing position.
|
||||||
def rest_blank?
|
def rest_blank?
|
||||||
saved = @ss.pos
|
!@ss.exist?(/\S/)
|
||||||
@ss.skip(/\s*/)
|
|
||||||
result = @ss.eos?
|
|
||||||
@ss.pos = saved
|
|
||||||
result
|
|
||||||
end
|
end
|
||||||
|
|
||||||
# Regex for identifier: [a-zA-Z_][\w-]*\??
|
# Regex for identifier: [a-zA-Z_][\w-]*\??
|
||||||
@@ -232,7 +225,8 @@ module Liquid
|
|||||||
b = token.getbyte(pos)
|
b = token.getbyte(pos)
|
||||||
case b
|
case b
|
||||||
when SPACE, TAB, CR, FF then pos += 1
|
when SPACE, TAB, CR, FF then pos += 1
|
||||||
when NL then pos += 1; nl += 1
|
when NL then pos += 1
|
||||||
|
nl += 1
|
||||||
else break
|
else break
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
@@ -247,6 +241,7 @@ module Liquid
|
|||||||
while pos < len
|
while pos < len
|
||||||
b = token.getbyte(pos)
|
b = token.getbyte(pos)
|
||||||
break unless ByteTables::IDENT_CONT[b]
|
break unless ByteTables::IDENT_CONT[b]
|
||||||
|
|
||||||
pos += 1
|
pos += 1
|
||||||
end
|
end
|
||||||
pos += 1 if pos < len && token.getbyte(pos) == QMARK
|
pos += 1 if pos < len && token.getbyte(pos) == QMARK
|
||||||
@@ -260,7 +255,8 @@ module Liquid
|
|||||||
b = token.getbyte(pos)
|
b = token.getbyte(pos)
|
||||||
case b
|
case b
|
||||||
when SPACE, TAB, CR, FF then pos += 1
|
when SPACE, TAB, CR, FF then pos += 1
|
||||||
when NL then pos += 1; nl += 1
|
when NL then pos += 1
|
||||||
|
nl += 1
|
||||||
else break
|
else break
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
@@ -317,6 +313,5 @@ module Liquid
|
|||||||
|
|
||||||
true
|
true
|
||||||
end
|
end
|
||||||
|
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
|
|||||||
@@ -277,16 +277,18 @@ module Liquid
|
|||||||
|
|
||||||
return input if words + 1 > MAX_I32
|
return input if words + 1 > MAX_I32
|
||||||
|
|
||||||
# Build result incrementally — avoids split() array + string allocations
|
# Scan words tracking byte positions; build the normalized (single-space)
|
||||||
|
# result string only when truncation is actually needed.
|
||||||
len = input.bytesize
|
len = input.bytesize
|
||||||
pos = 0
|
pos = 0
|
||||||
word_count = 0
|
word_count = 0
|
||||||
result = nil
|
# Flat array of [start, end, start, end, ...] for up to `words` words.
|
||||||
|
# Avoids allocating a result string in the common no-truncation case.
|
||||||
|
positions = []
|
||||||
|
|
||||||
# Skip leading whitespace
|
# Skip leading whitespace
|
||||||
while pos < len
|
while pos < len
|
||||||
b = input.getbyte(pos)
|
break unless ByteTables::WHITESPACE[input.getbyte(pos)]
|
||||||
break unless ByteTables::WHITESPACE[b]
|
|
||||||
pos += 1
|
pos += 1
|
||||||
end
|
end
|
||||||
|
|
||||||
@@ -294,34 +296,49 @@ module Liquid
|
|||||||
word_start = pos
|
word_start = pos
|
||||||
word_count += 1
|
word_count += 1
|
||||||
|
|
||||||
# Skip non-whitespace chars (word body)
|
# Scan to end of word
|
||||||
while pos < len
|
while pos < len
|
||||||
b = input.getbyte(pos)
|
break if ByteTables::WHITESPACE[input.getbyte(pos)]
|
||||||
break if ByteTables::WHITESPACE[b]
|
|
||||||
pos += 1
|
pos += 1
|
||||||
end
|
end
|
||||||
|
|
||||||
if word_count > words
|
if word_count <= words
|
||||||
# Truncate — result already has the first N words
|
positions.push(word_start, pos) # [start, end, start, end, ...]
|
||||||
truncate_string = Utils.to_s(truncate_string)
|
|
||||||
return result.concat(truncate_string)
|
|
||||||
end
|
|
||||||
|
|
||||||
# Append word to result (only allocate result when we know truncation is possible)
|
|
||||||
if result
|
|
||||||
result << " " << input.byteslice(word_start, pos - word_start)
|
|
||||||
else
|
else
|
||||||
result = +input.byteslice(word_start, pos - word_start)
|
# Truncation confirmed — build normalized result from stored positions
|
||||||
|
result = +input.byteslice(positions[0], positions[1] - positions[0])
|
||||||
|
i = 2
|
||||||
|
while i < positions.length
|
||||||
|
result << " " << input.byteslice(positions[i], positions[i + 1] - positions[i])
|
||||||
|
i += 2
|
||||||
|
end
|
||||||
|
return result << Utils.to_s(truncate_string)
|
||||||
end
|
end
|
||||||
|
|
||||||
# Skip whitespace between words
|
# Skip whitespace between words
|
||||||
while pos < len
|
while pos < len
|
||||||
b = input.getbyte(pos)
|
break unless ByteTables::WHITESPACE[input.getbyte(pos)]
|
||||||
break unless ByteTables::WHITESPACE[b]
|
|
||||||
pos += 1
|
pos += 1
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
|
|
||||||
|
# Fewer words than requested — no truncation needed, return original unchanged.
|
||||||
|
return input if word_count < words
|
||||||
|
|
||||||
|
# Exactly `words` words. Ruby's split(" ", words+1) would produce a words+1-th
|
||||||
|
# empty element when input has trailing whitespace, triggering the truncation path.
|
||||||
|
# Match that behaviour: if the input ends with whitespace, normalize and append
|
||||||
|
# truncate_string even though no word was cut.
|
||||||
|
if len > 0 && ByteTables::WHITESPACE[input.getbyte(len - 1)]
|
||||||
|
result = +input.byteslice(positions[0], positions[1] - positions[0])
|
||||||
|
i = 2
|
||||||
|
while i < positions.length
|
||||||
|
result << " " << input.byteslice(positions[i], positions[i + 1] - positions[i])
|
||||||
|
i += 2
|
||||||
|
end
|
||||||
|
return result << Utils.to_s(truncate_string)
|
||||||
|
end
|
||||||
|
|
||||||
input
|
input
|
||||||
end
|
end
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user