diff --git a/config.yml b/config.yml index c226470021..14ea82f405 100644 --- a/config.yml +++ b/config.yml @@ -666,6 +666,8 @@ tokens: comment: "unary **" - name: WORDS_SEP comment: "a separator between words in a list" + - name: WORDS_SEP_IMPLICIT + comment: "a separator between words in a list that has no source characters" - name: XSTRING_BEGIN comment: "the beginning of an execution string" - name: __END__ diff --git a/include/prism/internal/parser.h b/include/prism/internal/parser.h index 0c071b1375..ad2e028909 100644 --- a/include/prism/internal/parser.h +++ b/include/prism/internal/parser.h @@ -156,6 +156,22 @@ typedef struct pm_lex_mode { /* Whether or not interpolation is allowed in this list. */ bool interpolation; + /* + * Whether any token has been emitted from this list. A word + * separator delimits the opener from the first word, so one + * without source characters is emitted when the list does not + * start with whitespace. + */ + bool started; + + /* + * Whether the previously emitted token was a word separator. A + * word separator delimits the last word from the terminator, so + * one without source characters is emitted when the list does + * not end with whitespace. + */ + bool separated; + /* * When lexing a list, it takes into account balancing the * terminator if the terminator is one of (), [], {}, or <>. diff --git a/lib/prism/lex_compat.rb b/lib/prism/lex_compat.rb index b7921b8d9c..8d4838651a 100644 --- a/lib/prism/lex_compat.rb +++ b/lib/prism/lex_compat.rb @@ -616,6 +616,19 @@ def result last_heredoc_end = nil #: Integer? eof_token = nil #: Token? + # The two tokens Ripper scanned most recently, oldest last. Tokens that + # this loop does not emit are not part of Ripper's view, so they never + # become either one. + previous_token = nil #: Token? + previous_previous_token = nil #: Token? + + # The lex state in effect at each embedded expression that is still open, + # innermost last, and the state at the opener of the one that closed most + # recently. Ending an embedded expression restores the state its opener + # had, which on_regexp_end below has to reproduce. + embexpr_states = [] #: Array[Integer] + last_embexpr_state = nil #: Integer? + bom = source.slice(0, 3) == "\xEF\xBB\xBF" last_comment_token = nil #: lex_compat_token? @@ -625,6 +638,10 @@ def result lineno = prism_token.location.start_line column = prism_token.location.start_column + # Ripper is a scanner, so every event it emits corresponds to source + # characters. Implicit word separators have none. + next if prism_token.type == :WORDS_SEP_IMPLICIT + event = RIPPER.fetch(prism_token.type) value = prism_token.value lex_state = Translation::Ripper::Lexer::State[prism_token._ripper_state] @@ -636,6 +653,12 @@ def result last_comment_token[2] += value last_comment_token = nil last_comment_end = nil + + # The newline's characters are part of the comment Ripper emits, so + # it is still the token Ripper scanned most recently, and the trailing + # whitespace check at on_eof measures from its end. + previous_previous_token = previous_token + previous_token = prism_token next end @@ -682,7 +705,11 @@ def result # want to bother comparing the state on them. last_heredoc_end = prism_token.location.end_offset [[lineno, column], event, value, lex_state] + when :on_embexpr_beg + embexpr_states.push(prism_token._ripper_state) + [[lineno, column], event, value, lex_state] when :on_embexpr_end + last_embexpr_state = embexpr_states.pop [[lineno, column], event, value, lex_state] when :on_words_sep # Ripper emits one token each per line. @@ -701,24 +728,11 @@ def result # Ripper's lexed state. So here, if it's a regexp end token, we # output the state as the previous state, solely for the sake of # comparison. - previous_token = result_value[index - 1] lex_state = - if RIPPER.fetch(previous_token.type) == :on_embexpr_end - # If the previous token is embexpr_end, then we have to do even - # more processing. The end of an embedded expression sets the - # state to the state that it had at the beginning of the - # embedded expression. So we have to go and find that state and - # set it here. - counter = 1 - current_index = index - 1 - - until counter == 0 - current_index -= 1 - current_event = RIPPER.fetch(result_value[current_index].type) - counter += { on_embexpr_beg: -1, on_embexpr_end: 1 }[current_event] || 0 - end - - Translation::Ripper::Lexer::State[result_value[current_index]._ripper_state] + if previous_token&.type == :EMBEXPR_END && last_embexpr_state + # An embedded expression that just closed restored the state its + # opener had, so that is the state Ripper reports here. + Translation::Ripper::Lexer::State[last_embexpr_state] else previous_state end @@ -726,34 +740,36 @@ def result [[lineno, column], event, value, lex_state] when :on_eof eof_token = prism_token - previous_token = result_value[index - 1] - - # A newline that was folded back into a comment still marks the - # comment boundary for the check below. - comment_boundary = previous_token.type == :COMMENT || - (index >= 2 && %i[NEWLINE NEWLINE_TERMINATOR IGNORED_NEWLINE].include?(previous_token.type) && result_value[index - 2].type == :COMMENT && result_value[index - 2].location.end_offset == previous_token.location.start_offset) - - # If we're at the end of the file and the previous token was a - # comment and there is still whitespace after the comment, then - # Ripper will append a on_nl token (even though there isn't - # necessarily a newline). We mirror that here. - if comment_boundary - # If the comment is at the start of a heredoc: <= 0 - empty = %i[PERCENT_LOWER_I PERCENT_LOWER_W PERCENT_UPPER_I PERCENT_UPPER_W].include?(prev_token&.type) - ends_with_whitespace = prev_token&.type == :WORDS_SEP - # parser always emits a space token after content in a percent array, even if no actual whitespace is present. - if !empty && !ends_with_whitespace - tokens << [:tSPACE, [nil, range(token.location.start_offset, token.location.start_offset)]] - end - end + quote_stack.pop when :tSYMBEG if (next_token = lexed[index]) && next_token.type != :STRING_CONTENT && next_token.type != :EMBEXPR_BEGIN && next_token.type != :EMBVAR && next_token.type != :STRING_END next_location = token.location.join(next_token.location) @@ -470,7 +463,9 @@ def to_a when :tXSTRING_BEG quote_stack.push(value) when :tSYMBOLS_BEG, :tQSYMBOLS_BEG, :tWORDS_BEG, :tQWORDS_BEG - if (next_token = lexed[index]) && next_token.type == :WORDS_SEP + # The separator that delimits the opener from the first word is + # part of the opener for parser, so it is not emitted separately. + if (next_token = lexed[index]) && %i[WORDS_SEP WORDS_SEP_IMPLICIT].include?(next_token.type) index += 1 end diff --git a/src/prism.c b/src/prism.c index c3189f607c..0ee464f507 100644 --- a/src/prism.c +++ b/src/prism.c @@ -239,6 +239,8 @@ lex_mode_push_list(pm_parser_t *parser, bool interpolation, uint8_t delimiter) { .as.list = { .nesting = 0, .interpolation = interpolation, + .started = false, + .separated = false, .incrementor = incrementor, .terminator = terminator } @@ -11552,6 +11554,9 @@ parser_lex(pm_parser_t *parser) { // mutates next_start parser_flush_heredoc_end(parser); } + + lex_mode->as.list.started = true; + lex_mode->as.list.separated = true; LEX(PM_TOKEN_WORDS_SEP); } @@ -11561,6 +11566,18 @@ parser_lex(pm_parser_t *parser) { LEX(PM_TOKEN_EOF); } + /* A word separator delimits the opener from the first word (or + * from the terminator when the list is empty). The parser accepts + * the words without it, so when the list does not start with + * whitespace the implicit separator goes to the lex callback alone + * and lexing continues on to the token the parser receives. */ + if (!lex_mode->as.list.started) { + lex_mode->as.list.started = true; + lex_mode->as.list.separated = true; + parser->current.type = PM_TOKEN_WORDS_SEP_IMPLICIT; + parser_lex_callback(parser); + } + // Here we'll get a list of the places where strpbrk should break, // and then find the first one. const uint8_t *breakpoints = lex_mode->as.list.breakpoints; @@ -11578,6 +11595,7 @@ parser_lex(pm_parser_t *parser) { if (pm_char_is_whitespace(*breakpoint) && *breakpoint != lex_mode->as.list.terminator) { parser->current.end = breakpoint; pm_token_buffer_flush(parser, &token_buffer); + lex_mode->as.list.separated = false; LEX(PM_TOKEN_STRING_CONTENT); } @@ -11598,9 +11616,22 @@ parser_lex(pm_parser_t *parser) { if (breakpoint > parser->current.start) { parser->current.end = breakpoint; pm_token_buffer_flush(parser, &token_buffer); + lex_mode->as.list.separated = false; LEX(PM_TOKEN_STRING_CONTENT); } + /* A word separator delimits the last word from the + * terminator. The parser accepts the terminator without + * it, so when the list does not end with whitespace the + * implicit separator goes to the lex callback alone and + * lexing continues on to the terminator. */ + if (!lex_mode->as.list.separated) { + lex_mode->as.list.separated = true; + parser->current.end = breakpoint; + parser->current.type = PM_TOKEN_WORDS_SEP_IMPLICIT; + parser_lex_callback(parser); + } + // Otherwise, switch back to the default state and return // the end of the list. parser->current.end = breakpoint + 1; @@ -11708,6 +11739,7 @@ parser_lex(pm_parser_t *parser) { pm_token_buffer_flush(parser, &token_buffer); } + lex_mode->as.list.separated = false; LEX(type); } diff --git a/templates/src/tokens.c.erb b/templates/src/tokens.c.erb index 6e88d423c2..1d772ec01d 100644 --- a/templates/src/tokens.c.erb +++ b/templates/src/tokens.c.erb @@ -362,6 +362,7 @@ pm_token_str(pm_token_type_t token_type) { case PM_TOKEN_USTAR_STAR: return "**"; case PM_TOKEN_WORDS_SEP: + case PM_TOKEN_WORDS_SEP_IMPLICIT: return "string separator"; case PM_TOKEN_XSTRING_BEGIN: return "backtick string literal";