refactor Lexer to be static class function

This commit is contained in:
Michael Go
2025-01-07 14:32:07 -04:00
parent 3c16c27ee1
commit 002e4caea7
4 changed files with 88 additions and 92 deletions
+30 -34
View File
@@ -25,38 +25,37 @@ module Liquid
COMPARISON_OPERATOR = /==|!=|<>|<=?|>=?|contains(?=\s)/ COMPARISON_OPERATOR = /==|!=|<>|<=?|>=?|contains(?=\s)/
WHITESPACE_OR_NOTHING = /\s*/ WHITESPACE_OR_NOTHING = /\s*/
def initialize(input) class << self
@ss = StringScanner.new(input) def tokenize(input)
end ss = StringScanner.new(input)
output = []
def tokenize until ss.eos?
@output = [] ss.skip(WHITESPACE_OR_NOTHING)
break if ss.eos?
until @ss.eos? tok = if (t = ss.scan(COMPARISON_OPERATOR))
@ss.skip(WHITESPACE_OR_NOTHING)
break if @ss.eos?
tok = if (t = @ss.scan(COMPARISON_OPERATOR))
[:comparison, t] [:comparison, t]
elsif (t = @ss.scan(STRING_LITERAL)) elsif (t = ss.scan(STRING_LITERAL))
[:string, t] [:string, t]
elsif (t = @ss.scan(NUMBER_LITERAL)) elsif (t = ss.scan(NUMBER_LITERAL))
[:number, t] [:number, t]
elsif (t = @ss.scan(IDENTIFIER)) elsif (t = ss.scan(IDENTIFIER))
[:id, t] [:id, t]
elsif (t = @ss.scan(DOTDOT)) elsif (t = ss.scan(DOTDOT))
[:dotdot, t] [:dotdot, t]
else else
c = @ss.getch c = ss.getch
if (s = SPECIALS[c]) if (s = SPECIALS[c])
[s, c] [s, c]
else else
raise SyntaxError, "Unexpected character #{c}" raise SyntaxError, "Unexpected character #{c}"
end end
end end
@output << tok output << tok
end end
@output << [:end_of_string] output << [:end_of_string]
end
end end
end end
@@ -157,14 +156,11 @@ module Liquid
table.freeze table.freeze
end end
def initialize(input)
@input = input
end
# rubocop:disable Metrics/BlockNesting # rubocop:disable Metrics/BlockNesting
def tokenize class << self
ss = StringScannerPool.pop(@input) def tokenize(input)
@output = [] ss = StringScannerPool.pop(input)
output = []
until ss.eos? until ss.eos?
ss.skip(WHITESPACE_OR_NOTHING) ss.skip(WHITESPACE_OR_NOTHING)
@@ -179,22 +175,22 @@ module Liquid
# Special case for ".." # Special case for ".."
if special == DOT && ss.peek_byte == DOT_ORD if special == DOT && ss.peek_byte == DOT_ORD
ss.scan_byte ss.scan_byte
@output << DOTDOT output << DOTDOT
elsif special == DASH elsif special == DASH
# Special case for negative numbers # Special case for negative numbers
if (peeked_byte = ss.peek_byte) && NUMBER_TABLE[peeked_byte] if (peeked_byte = ss.peek_byte) && NUMBER_TABLE[peeked_byte]
ss.pos -= 1 ss.pos -= 1
@output << [:number, ss.scan(NUMBER_LITERAL)] output << [:number, ss.scan(NUMBER_LITERAL)]
else else
@output << special output << special
end end
else else
@output << special output << special
end end
elsif (sub_table = TWO_CHARS_COMPARISON_JUMP_TABLE[peeked]) elsif (sub_table = TWO_CHARS_COMPARISON_JUMP_TABLE[peeked])
ss.scan_byte ss.scan_byte
if (peeked_byte = ss.peek_byte) && (found = sub_table[peeked_byte]) if (peeked_byte = ss.peek_byte) && (found = sub_table[peeked_byte])
@output << found output << found
ss.scan_byte ss.scan_byte
else else
raise_syntax_error(start_pos, ss) raise_syntax_error(start_pos, ss)
@@ -202,17 +198,17 @@ module Liquid
elsif (sub_table = COMPARISON_JUMP_TABLE[peeked]) elsif (sub_table = COMPARISON_JUMP_TABLE[peeked])
ss.scan_byte ss.scan_byte
if (peeked_byte = ss.peek_byte) && (found = sub_table[peeked_byte]) if (peeked_byte = ss.peek_byte) && (found = sub_table[peeked_byte])
@output << found output << found
ss.scan_byte ss.scan_byte
else else
@output << SINGLE_COMPARISON_TOKENS[peeked] output << SINGLE_COMPARISON_TOKENS[peeked]
end end
else else
type, pattern = NEXT_MATCHER_JUMP_TABLE[peeked] type, pattern = NEXT_MATCHER_JUMP_TABLE[peeked]
if type && (t = ss.scan(pattern)) if type && (t = ss.scan(pattern))
# Special case for "contains" # Special case for "contains"
@output << if type == :id && t == "contains" && @output.last&.first != :dot output << if type == :id && t == "contains" && output.last&.first != :dot
COMPARISON_CONTAINS COMPARISON_CONTAINS
else else
[type, t] [type, t]
@@ -223,8 +219,7 @@ module Liquid
end end
end end
# rubocop:enable Metrics/BlockNesting # rubocop:enable Metrics/BlockNesting
output << EOS
@output << EOS
ensure ensure
StringScannerPool.release(ss) StringScannerPool.release(ss)
end end
@@ -235,6 +230,7 @@ module Liquid
raise SyntaxError, "Unexpected character #{ss.getch}" raise SyntaxError, "Unexpected character #{ss.getch}"
end end
end end
end
# Remove this once we can depend on strscan >= 3.1.1 # Remove this once we can depend on strscan >= 3.1.1
Lexer = StringScanner.instance_methods.include?(:scan_byte) ? Lexer2 : Lexer1 Lexer = StringScanner.instance_methods.include?(:scan_byte) ? Lexer2 : Lexer1
+1 -2
View File
@@ -3,8 +3,7 @@
module Liquid module Liquid
class Parser class Parser
def initialize(input) def initialize(input)
l = Lexer.new(input) @tokens = Lexer.tokenize(input)
@tokens = l.tokenize
@p = 0 # pointer to current location @p = 0 # pointer to current location
end end
+3 -2
View File
@@ -1,8 +1,10 @@
# frozen_string_literal: true
module Liquid module Liquid
class StringScannerPool class StringScannerPool
class << self class << self
def pop(input) def pop(input)
@ss_pool ||= [StringScanner.new("")] * 5 @ss_pool ||= 5.times.each_with_object([]) { |_i, arr| arr << StringScanner.new("") }
if @ss_pool.empty? if @ss_pool.empty?
StringScanner.new(input) StringScanner.new(input)
@@ -14,7 +16,6 @@ module Liquid
end end
def release(ss) def release(ss)
binding.irb if ss.nil?
@ss_pool ||= [] @ss_pool ||= []
@ss_pool << ss @ss_pool << ss
end end
+1 -1
View File
@@ -134,6 +134,6 @@ class LexerUnitTest < Minitest::Test
private private
def tokenize(input) def tokenize(input)
Lexer.new(input).tokenize Lexer.tokenize(input)
end end
end end