From d6e47d2adf6d9b04f7d8c72eaf3ef40da180e551 Mon Sep 17 00:00:00 2001 From: Kevin Newton Date: Mon, 28 Sep 2026 14:11:55 -0400 Subject: [PATCH 1/2] Switch ripper compat to forward look A bunch of things are required to match ripper state, including looking backward sometimes to pull a state forward because of the nature of parse.y. This changes the structure from a backward look to a forward look by keeping a couple of stacks of state. --- lib/prism/lex_compat.rb | 102 +++++++++++++++++++++++----------------- 1 file changed, 58 insertions(+), 44 deletions(-) diff --git a/lib/prism/lex_compat.rb b/lib/prism/lex_compat.rb index b7921b8d9c..3d5a10336d 100644 --- a/lib/prism/lex_compat.rb +++ b/lib/prism/lex_compat.rb @@ -616,6 +616,19 @@ def result last_heredoc_end = nil #: Integer? eof_token = nil #: Token? + # The two tokens Ripper scanned most recently, oldest last. Tokens that + # this loop does not emit are not part of Ripper's view, so they never + # become either one. + previous_token = nil #: Token? + previous_previous_token = nil #: Token? + + # The lex state in effect at each embedded expression that is still open, + # innermost last, and the state at the opener of the one that closed most + # recently. Ending an embedded expression restores the state its opener + # had, which on_regexp_end below has to reproduce. + embexpr_states = [] #: Array[Integer] + last_embexpr_state = nil #: Integer? + bom = source.slice(0, 3) == "\xEF\xBB\xBF" last_comment_token = nil #: lex_compat_token? @@ -636,6 +649,12 @@ def result last_comment_token[2] += value last_comment_token = nil last_comment_end = nil + + # The newline's characters are part of the comment Ripper emits, so + # it is still the token Ripper scanned most recently, and the trailing + # whitespace check at on_eof measures from its end. + previous_previous_token = previous_token + previous_token = prism_token next end @@ -682,7 +701,11 @@ def result # want to bother comparing the state on them. last_heredoc_end = prism_token.location.end_offset [[lineno, column], event, value, lex_state] + when :on_embexpr_beg + embexpr_states.push(prism_token._ripper_state) + [[lineno, column], event, value, lex_state] when :on_embexpr_end + last_embexpr_state = embexpr_states.pop [[lineno, column], event, value, lex_state] when :on_words_sep # Ripper emits one token each per line. @@ -701,24 +724,11 @@ def result # Ripper's lexed state. So here, if it's a regexp end token, we # output the state as the previous state, solely for the sake of # comparison. - previous_token = result_value[index - 1] lex_state = - if RIPPER.fetch(previous_token.type) == :on_embexpr_end - # If the previous token is embexpr_end, then we have to do even - # more processing. The end of an embedded expression sets the - # state to the state that it had at the beginning of the - # embedded expression. So we have to go and find that state and - # set it here. - counter = 1 - current_index = index - 1 - - until counter == 0 - current_index -= 1 - current_event = RIPPER.fetch(result_value[current_index].type) - counter += { on_embexpr_beg: -1, on_embexpr_end: 1 }[current_event] || 0 - end - - Translation::Ripper::Lexer::State[result_value[current_index]._ripper_state] + if previous_token&.type == :EMBEXPR_END && last_embexpr_state + # An embedded expression that just closed restored the state its + # opener had, so that is the state Ripper reports here. + Translation::Ripper::Lexer::State[last_embexpr_state] else previous_state end @@ -726,34 +736,36 @@ def result [[lineno, column], event, value, lex_state] when :on_eof eof_token = prism_token - previous_token = result_value[index - 1] - - # A newline that was folded back into a comment still marks the - # comment boundary for the check below. - comment_boundary = previous_token.type == :COMMENT || - (index >= 2 && %i[NEWLINE NEWLINE_TERMINATOR IGNORED_NEWLINE].include?(previous_token.type) && result_value[index - 2].type == :COMMENT && result_value[index - 2].location.end_offset == previous_token.location.start_offset) - - # If we're at the end of the file and the previous token was a - # comment and there is still whitespace after the comment, then - # Ripper will append a on_nl token (even though there isn't - # necessarily a newline). We mirror that here. - if comment_boundary - # If the comment is at the start of a heredoc: < Date: Mon, 28 Sep 2026 13:16:05 -0400 Subject: [PATCH 2/2] Add an implicit words sep token The parse.y grammar has the concept of a words separator that must delimit words in a word list literal. Prism previously didn't emit these in the token stream because they don't actually correspond to any source code. However, in order to better compare against the vendored grammars, it's much easier if we do emit them. This commit adds them into the token stream without actually touching the parser, so it barely effects anything except conformance. Since we have split up all those other tokens now and we have this in place, we now don't need any special handling: our tokens stream is pretty much exactly a superset of parse.y's token stream, meaning it's very easy to drive a parse.y grammar using prism's token stream provided you have a mapping of token names. --- config.yml | 2 ++ include/prism/internal/parser.h | 16 ++++++++++++++ lib/prism/lex_compat.rb | 4 ++++ lib/prism/translation/parser/lexer.rb | 15 +++++-------- src/prism.c | 32 +++++++++++++++++++++++++++ templates/src/tokens.c.erb | 1 + 6 files changed, 60 insertions(+), 10 deletions(-) diff --git a/config.yml b/config.yml index c226470021..14ea82f405 100644 --- a/config.yml +++ b/config.yml @@ -666,6 +666,8 @@ tokens: comment: "unary **" - name: WORDS_SEP comment: "a separator between words in a list" + - name: WORDS_SEP_IMPLICIT + comment: "a separator between words in a list that has no source characters" - name: XSTRING_BEGIN comment: "the beginning of an execution string" - name: __END__ diff --git a/include/prism/internal/parser.h b/include/prism/internal/parser.h index 0c071b1375..ad2e028909 100644 --- a/include/prism/internal/parser.h +++ b/include/prism/internal/parser.h @@ -156,6 +156,22 @@ typedef struct pm_lex_mode { /* Whether or not interpolation is allowed in this list. */ bool interpolation; + /* + * Whether any token has been emitted from this list. A word + * separator delimits the opener from the first word, so one + * without source characters is emitted when the list does not + * start with whitespace. + */ + bool started; + + /* + * Whether the previously emitted token was a word separator. A + * word separator delimits the last word from the terminator, so + * one without source characters is emitted when the list does + * not end with whitespace. + */ + bool separated; + /* * When lexing a list, it takes into account balancing the * terminator if the terminator is one of (), [], {}, or <>. diff --git a/lib/prism/lex_compat.rb b/lib/prism/lex_compat.rb index 3d5a10336d..8d4838651a 100644 --- a/lib/prism/lex_compat.rb +++ b/lib/prism/lex_compat.rb @@ -638,6 +638,10 @@ def result lineno = prism_token.location.start_line column = prism_token.location.start_column + # Ripper is a scanner, so every event it emits corresponds to source + # characters. Implicit word separators have none. + next if prism_token.type == :WORDS_SEP_IMPLICIT + event = RIPPER.fetch(prism_token.type) value = prism_token.value lex_state = Translation::Ripper::Lexer::State[prism_token._ripper_state] diff --git a/lib/prism/translation/parser/lexer.rb b/lib/prism/translation/parser/lexer.rb index a2f517298a..226128cdae 100644 --- a/lib/prism/translation/parser/lexer.rb +++ b/lib/prism/translation/parser/lexer.rb @@ -185,6 +185,7 @@ class Lexer # :nodoc: USTAR: :tSTAR, USTAR_STAR: :tDSTAR, WORDS_SEP: :tSPACE, + WORDS_SEP_IMPLICIT: :tSPACE, XSTRING_BEGIN: :tXSTRING_BEG } @@ -443,15 +444,7 @@ def to_a location = range(token.location.start_offset, token.location.start_offset + 1) end - if percent_array?(quote_stack.pop) - prev_token = lexed[index - 2] if index - 2 >= 0 - empty = %i[PERCENT_LOWER_I PERCENT_LOWER_W PERCENT_UPPER_I PERCENT_UPPER_W].include?(prev_token&.type) - ends_with_whitespace = prev_token&.type == :WORDS_SEP - # parser always emits a space token after content in a percent array, even if no actual whitespace is present. - if !empty && !ends_with_whitespace - tokens << [:tSPACE, [nil, range(token.location.start_offset, token.location.start_offset)]] - end - end + quote_stack.pop when :tSYMBEG if (next_token = lexed[index]) && next_token.type != :STRING_CONTENT && next_token.type != :EMBEXPR_BEGIN && next_token.type != :EMBVAR && next_token.type != :STRING_END next_location = token.location.join(next_token.location) @@ -470,7 +463,9 @@ def to_a when :tXSTRING_BEG quote_stack.push(value) when :tSYMBOLS_BEG, :tQSYMBOLS_BEG, :tWORDS_BEG, :tQWORDS_BEG - if (next_token = lexed[index]) && next_token.type == :WORDS_SEP + # The separator that delimits the opener from the first word is + # part of the opener for parser, so it is not emitted separately. + if (next_token = lexed[index]) && %i[WORDS_SEP WORDS_SEP_IMPLICIT].include?(next_token.type) index += 1 end diff --git a/src/prism.c b/src/prism.c index c3189f607c..0ee464f507 100644 --- a/src/prism.c +++ b/src/prism.c @@ -239,6 +239,8 @@ lex_mode_push_list(pm_parser_t *parser, bool interpolation, uint8_t delimiter) { .as.list = { .nesting = 0, .interpolation = interpolation, + .started = false, + .separated = false, .incrementor = incrementor, .terminator = terminator } @@ -11552,6 +11554,9 @@ parser_lex(pm_parser_t *parser) { // mutates next_start parser_flush_heredoc_end(parser); } + + lex_mode->as.list.started = true; + lex_mode->as.list.separated = true; LEX(PM_TOKEN_WORDS_SEP); } @@ -11561,6 +11566,18 @@ parser_lex(pm_parser_t *parser) { LEX(PM_TOKEN_EOF); } + /* A word separator delimits the opener from the first word (or + * from the terminator when the list is empty). The parser accepts + * the words without it, so when the list does not start with + * whitespace the implicit separator goes to the lex callback alone + * and lexing continues on to the token the parser receives. */ + if (!lex_mode->as.list.started) { + lex_mode->as.list.started = true; + lex_mode->as.list.separated = true; + parser->current.type = PM_TOKEN_WORDS_SEP_IMPLICIT; + parser_lex_callback(parser); + } + // Here we'll get a list of the places where strpbrk should break, // and then find the first one. const uint8_t *breakpoints = lex_mode->as.list.breakpoints; @@ -11578,6 +11595,7 @@ parser_lex(pm_parser_t *parser) { if (pm_char_is_whitespace(*breakpoint) && *breakpoint != lex_mode->as.list.terminator) { parser->current.end = breakpoint; pm_token_buffer_flush(parser, &token_buffer); + lex_mode->as.list.separated = false; LEX(PM_TOKEN_STRING_CONTENT); } @@ -11598,9 +11616,22 @@ parser_lex(pm_parser_t *parser) { if (breakpoint > parser->current.start) { parser->current.end = breakpoint; pm_token_buffer_flush(parser, &token_buffer); + lex_mode->as.list.separated = false; LEX(PM_TOKEN_STRING_CONTENT); } + /* A word separator delimits the last word from the + * terminator. The parser accepts the terminator without + * it, so when the list does not end with whitespace the + * implicit separator goes to the lex callback alone and + * lexing continues on to the terminator. */ + if (!lex_mode->as.list.separated) { + lex_mode->as.list.separated = true; + parser->current.end = breakpoint; + parser->current.type = PM_TOKEN_WORDS_SEP_IMPLICIT; + parser_lex_callback(parser); + } + // Otherwise, switch back to the default state and return // the end of the list. parser->current.end = breakpoint + 1; @@ -11708,6 +11739,7 @@ parser_lex(pm_parser_t *parser) { pm_token_buffer_flush(parser, &token_buffer); } + lex_mode->as.list.separated = false; LEX(type); } diff --git a/templates/src/tokens.c.erb b/templates/src/tokens.c.erb index 6e88d423c2..1d772ec01d 100644 --- a/templates/src/tokens.c.erb +++ b/templates/src/tokens.c.erb @@ -362,6 +362,7 @@ pm_token_str(pm_token_type_t token_type) { case PM_TOKEN_USTAR_STAR: return "**"; case PM_TOKEN_WORDS_SEP: + case PM_TOKEN_WORDS_SEP_IMPLICIT: return "string separator"; case PM_TOKEN_XSTRING_BEGIN: return "backtick string literal";