Class: Tokenzr::Tokenizer
- Inherits:
-
Object
- Object
- Tokenzr::Tokenizer
- Defined in:
- lib/tokenzr/tokenizer.rb
Overview
main tokenizer class
Instance Method Summary collapse
- #digit_chars ⇒ Object
-
#initialize(charset: nil, **overrides) ⇒ Tokenizer
constructor
A new instance of Tokenizer.
- #lone_chars ⇒ Object
- #parse(content) ⇒ Object
- #space_chars ⇒ Object
- #string_quotes ⇒ Object
- #text_chars ⇒ Object
Constructor Details
#initialize(charset: nil, **overrides) ⇒ Tokenizer
Returns a new instance of Tokenizer.
12 13 14 15 16 17 18 19 20 |
# File 'lib/tokenzr/tokenizer.rb', line 12 def initialize(charset: nil, **overrides) base = charset || Charset.default @charset = merge_overrides(base, overrides) conflicts = @charset.conflicts raise ConfigurationError, "Charset conflict: #{format_conflicts(conflicts)}" unless conflicts.empty? validate_operators! build_operator_index end |
Instance Method Details
#digit_chars ⇒ Object
26 27 28 |
# File 'lib/tokenzr/tokenizer.rb', line 26 def digit_chars @digit_chars ||= @charset.to_sets[:digits] end |
#lone_chars ⇒ Object
30 31 32 |
# File 'lib/tokenzr/tokenizer.rb', line 30 def lone_chars @lone_chars ||= @charset.to_sets[:lone] end |
#parse(content) ⇒ Object
42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 |
# File 'lib/tokenzr/tokenizer.rb', line 42 def parse(content) results = [] current_token = nil enum = content.each_char @line = 1 @column = 1 @cur_line = 1 @cur_col = 1 @pushback = [] while (chr = next_char(enum)) start_line = @cur_line start_col = @cur_col if string_quotes.include?(chr) results << current_token unless current_token.nil? current_token = nil results << read_string(enum, chr, start_line, start_col) next end if space_chars.include?(chr) results << current_token unless current_token.nil? current_token = nil next end if digit_chars.include?(chr) if !current_token.nil? && current_token.type == :text # digits continue an identifier started by text/underscore current_token = Token.new(current_token.content + chr, :text, current_token.line, current_token.column) next end results << current_token unless current_token.nil? current_token = nil result = read_number(enum, chr, start_line, start_col) if result.is_a?(Array) results.concat(result) else results << result end next end if text_chars.include?(chr) if !current_token.nil? && current_token.type == :text current_token = Token.new(current_token.content + chr, :text, current_token.line, current_token.column) else results << current_token unless current_token.nil? current_token = Token.new(chr, :text, start_line, start_col) end next end if lone_chars.include?(chr) results << current_token unless current_token.nil? current_token = nil op = match_operator(enum, chr, start_line, start_col) results << (op || Token.new(chr, :lone, start_line, start_col)) next end raise UnknownCharError, "Unknown character: #{chr.inspect}" end results << current_token unless current_token.nil? results end |
#space_chars ⇒ Object
34 35 36 |
# File 'lib/tokenzr/tokenizer.rb', line 34 def space_chars @space_chars ||= @charset.to_sets[:space] end |
#string_quotes ⇒ Object
38 39 40 |
# File 'lib/tokenzr/tokenizer.rb', line 38 def string_quotes @string_quotes ||= @charset.to_sets[:quotes] end |
#text_chars ⇒ Object
22 23 24 |
# File 'lib/tokenzr/tokenizer.rb', line 22 def text_chars @text_chars ||= @charset.to_sets[:text] end |