Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions config.yml
Original file line number Diff line number Diff line change
Expand Up @@ -666,6 +666,8 @@ tokens:
comment: "unary **"
- name: WORDS_SEP
comment: "a separator between words in a list"
- name: WORDS_SEP_IMPLICIT
comment: "a separator between words in a list that has no source characters"
- name: XSTRING_BEGIN
comment: "the beginning of an execution string"
- name: __END__
Expand Down
16 changes: 16 additions & 0 deletions include/prism/internal/parser.h
Original file line number Diff line number Diff line change
Expand Up @@ -156,6 +156,22 @@ typedef struct pm_lex_mode {
/* Whether or not interpolation is allowed in this list. */
bool interpolation;

/*
* Whether any token has been emitted from this list. A word
* separator delimits the opener from the first word, so one
* without source characters is emitted when the list does not
* start with whitespace.
*/
bool started;

/*
* Whether the previously emitted token was a word separator. A
* word separator delimits the last word from the terminator, so
* one without source characters is emitted when the list does
* not end with whitespace.
*/
bool separated;

/*
* When lexing a list, it takes into account balancing the
* terminator if the terminator is one of (), [], {}, or <>.
Expand Down
100 changes: 56 additions & 44 deletions lib/prism/lex_compat.rb
Original file line number Diff line number Diff line change
Expand Up @@ -616,6 +616,19 @@ def result
last_heredoc_end = nil #: Integer?
eof_token = nil #: Token?

# The two tokens Ripper scanned most recently, oldest last. Tokens that
# this loop does not emit are not part of Ripper's view, so they never
# become either one.
previous_token = nil #: Token?
previous_previous_token = nil #: Token?

# The lex state in effect at each embedded expression that is still open,
# innermost last, and the state at the opener of the one that closed most
# recently. Ending an embedded expression restores the state its opener
# had, which on_regexp_end below has to reproduce.
embexpr_states = [] #: Array[Integer]
last_embexpr_state = nil #: Integer?

bom = source.slice(0, 3) == "\xEF\xBB\xBF"

last_comment_token = nil #: lex_compat_token?
Expand All @@ -625,6 +638,10 @@ def result
lineno = prism_token.location.start_line
column = prism_token.location.start_column

# Ripper is a scanner, so every event it emits corresponds to source
# characters. Implicit word separators have none.
next if prism_token.type == :WORDS_SEP_IMPLICIT

event = RIPPER.fetch(prism_token.type)
value = prism_token.value
lex_state = Translation::Ripper::Lexer::State[prism_token._ripper_state]
Expand Down Expand Up @@ -682,7 +699,11 @@ def result
# want to bother comparing the state on them.
last_heredoc_end = prism_token.location.end_offset
[[lineno, column], event, value, lex_state]
when :on_embexpr_beg
embexpr_states.push(prism_token._ripper_state)
[[lineno, column], event, value, lex_state]
when :on_embexpr_end
last_embexpr_state = embexpr_states.pop
[[lineno, column], event, value, lex_state]
when :on_words_sep
# Ripper emits one token each per line.
Expand All @@ -701,59 +722,48 @@ def result
# Ripper's lexed state. So here, if it's a regexp end token, we
# output the state as the previous state, solely for the sake of
# comparison.
previous_token = result_value[index - 1]
lex_state =
if RIPPER.fetch(previous_token.type) == :on_embexpr_end
# If the previous token is embexpr_end, then we have to do even
# more processing. The end of an embedded expression sets the
# state to the state that it had at the beginning of the
# embedded expression. So we have to go and find that state and
# set it here.
counter = 1
current_index = index - 1

until counter == 0
current_index -= 1
current_event = RIPPER.fetch(result_value[current_index].type)
counter += { on_embexpr_beg: -1, on_embexpr_end: 1 }[current_event] || 0
end

Translation::Ripper::Lexer::State[result_value[current_index]._ripper_state]
if previous_token&.type == :EMBEXPR_END && last_embexpr_state
# An embedded expression that just closed restored the state its
# opener had, so that is the state Ripper reports here.
Translation::Ripper::Lexer::State[last_embexpr_state]
else
previous_state
end

[[lineno, column], event, value, lex_state]
when :on_eof
eof_token = prism_token
previous_token = result_value[index - 1]

# A newline that was folded back into a comment still marks the
# comment boundary for the check below.
comment_boundary = previous_token.type == :COMMENT ||
(index >= 2 && %i[NEWLINE NEWLINE_TERMINATOR IGNORED_NEWLINE].include?(previous_token.type) && result_value[index - 2].type == :COMMENT && result_value[index - 2].location.end_offset == previous_token.location.start_offset)

# If we're at the end of the file and the previous token was a
# comment and there is still whitespace after the comment, then
# Ripper will append a on_nl token (even though there isn't
# necessarily a newline). We mirror that here.
if comment_boundary
# If the comment is at the start of a heredoc: <<HEREDOC # comment
# then the comment's end_offset is up near the heredoc_beg.
# This is not the correct offset to use for figuring out if
# there is trailing whitespace after the last token.
# Use the greater offset of the two to determine the start of
# the trailing whitespace.
start_offset = [previous_token.location.end_offset, last_heredoc_end].compact.max
end_offset = prism_token.location.start_offset

if start_offset < end_offset
if bom
start_offset += 3
end_offset += 3
end

tokens << [[lineno, 0], :on_nl, source.slice(start_offset, end_offset - start_offset), lex_state]
if (previous = previous_token)
# A newline that was folded back into a comment still marks the
# comment boundary for the check below.
before = previous_previous_token
comment_boundary = previous.type == :COMMENT ||
(before && %i[NEWLINE NEWLINE_TERMINATOR IGNORED_NEWLINE].include?(previous.type) && before.type == :COMMENT && before.location.end_offset == previous.location.start_offset)

# If we're at the end of the file and the previous token was a
# comment and there is still whitespace after the comment, then
# Ripper will append a on_nl token (even though there isn't
# necessarily a newline). We mirror that here.
if comment_boundary
# If the comment is at the start of a heredoc: <<HEREDOC # comment
# then the comment's end_offset is up near the heredoc_beg.
# This is not the correct offset to use for figuring out if
# there is trailing whitespace after the last token.
# Use the greater offset of the two to determine the start of
# the trailing whitespace.
start_offset = [previous.location.end_offset, last_heredoc_end].compact.max
end_offset = prism_token.location.start_offset

if start_offset < end_offset
if bom
start_offset += 3
end_offset += 3
end

tokens << [[lineno, 0], :on_nl, source.slice(start_offset, end_offset - start_offset), lex_state]
end
end
end

Expand All @@ -763,6 +773,8 @@ def result
end #: lex_compat_token

previous_state = lex_state
previous_previous_token = previous_token
previous_token = prism_token

if event == :on_comment
last_comment_token = lex_compat_token
Expand Down
15 changes: 5 additions & 10 deletions lib/prism/translation/parser/lexer.rb
Original file line number Diff line number Diff line change
Expand Up @@ -185,6 +185,7 @@ class Lexer # :nodoc:
USTAR: :tSTAR,
USTAR_STAR: :tDSTAR,
WORDS_SEP: :tSPACE,
WORDS_SEP_IMPLICIT: :tSPACE,
XSTRING_BEGIN: :tXSTRING_BEG
}

Expand Down Expand Up @@ -443,15 +444,7 @@ def to_a
location = range(token.location.start_offset, token.location.start_offset + 1)
end

if percent_array?(quote_stack.pop)
prev_token = lexed[index - 2] if index - 2 >= 0
empty = %i[PERCENT_LOWER_I PERCENT_LOWER_W PERCENT_UPPER_I PERCENT_UPPER_W].include?(prev_token&.type)
ends_with_whitespace = prev_token&.type == :WORDS_SEP
# parser always emits a space token after content in a percent array, even if no actual whitespace is present.
if !empty && !ends_with_whitespace
tokens << [:tSPACE, [nil, range(token.location.start_offset, token.location.start_offset)]]
end
end
quote_stack.pop
when :tSYMBEG
if (next_token = lexed[index]) && next_token.type != :STRING_CONTENT && next_token.type != :EMBEXPR_BEGIN && next_token.type != :EMBVAR && next_token.type != :STRING_END
next_location = token.location.join(next_token.location)
Expand All @@ -470,7 +463,9 @@ def to_a
when :tXSTRING_BEG
quote_stack.push(value)
when :tSYMBOLS_BEG, :tQSYMBOLS_BEG, :tWORDS_BEG, :tQWORDS_BEG
if (next_token = lexed[index]) && next_token.type == :WORDS_SEP
# The separator that delimits the opener from the first word is
# part of the opener for parser, so it is not emitted separately.
if (next_token = lexed[index]) && %i[WORDS_SEP WORDS_SEP_IMPLICIT].include?(next_token.type)
index += 1
end

Expand Down
32 changes: 32 additions & 0 deletions src/prism.c
Original file line number Diff line number Diff line change
Expand Up @@ -239,6 +239,8 @@ lex_mode_push_list(pm_parser_t *parser, bool interpolation, uint8_t delimiter) {
.as.list = {
.nesting = 0,
.interpolation = interpolation,
.started = false,
.separated = false,
.incrementor = incrementor,
.terminator = terminator
}
Expand Down Expand Up @@ -11552,6 +11554,9 @@ parser_lex(pm_parser_t *parser) {
// mutates next_start
parser_flush_heredoc_end(parser);
}

lex_mode->as.list.started = true;
lex_mode->as.list.separated = true;
LEX(PM_TOKEN_WORDS_SEP);
}

Expand All @@ -11561,6 +11566,18 @@ parser_lex(pm_parser_t *parser) {
LEX(PM_TOKEN_EOF);
}

/* A word separator delimits the opener from the first word (or
* from the terminator when the list is empty). The parser accepts
* the words without it, so when the list does not start with
* whitespace the implicit separator goes to the lex callback alone
* and lexing continues on to the token the parser receives. */
if (!lex_mode->as.list.started) {
lex_mode->as.list.started = true;
lex_mode->as.list.separated = true;
parser->current.type = PM_TOKEN_WORDS_SEP_IMPLICIT;
parser_lex_callback(parser);
}

// Here we'll get a list of the places where strpbrk should break,
// and then find the first one.
const uint8_t *breakpoints = lex_mode->as.list.breakpoints;
Expand All @@ -11578,6 +11595,7 @@ parser_lex(pm_parser_t *parser) {
if (pm_char_is_whitespace(*breakpoint) && *breakpoint != lex_mode->as.list.terminator) {
parser->current.end = breakpoint;
pm_token_buffer_flush(parser, &token_buffer);
lex_mode->as.list.separated = false;
LEX(PM_TOKEN_STRING_CONTENT);
}

Expand All @@ -11598,9 +11616,22 @@ parser_lex(pm_parser_t *parser) {
if (breakpoint > parser->current.start) {
parser->current.end = breakpoint;
pm_token_buffer_flush(parser, &token_buffer);
lex_mode->as.list.separated = false;
LEX(PM_TOKEN_STRING_CONTENT);
}

/* A word separator delimits the last word from the
* terminator. The parser accepts the terminator without
* it, so when the list does not end with whitespace the
* implicit separator goes to the lex callback alone and
* lexing continues on to the terminator. */
if (!lex_mode->as.list.separated) {
lex_mode->as.list.separated = true;
parser->current.end = breakpoint;
parser->current.type = PM_TOKEN_WORDS_SEP_IMPLICIT;
parser_lex_callback(parser);
}

// Otherwise, switch back to the default state and return
// the end of the list.
parser->current.end = breakpoint + 1;
Expand Down Expand Up @@ -11708,6 +11739,7 @@ parser_lex(pm_parser_t *parser) {
pm_token_buffer_flush(parser, &token_buffer);
}

lex_mode->as.list.separated = false;
LEX(type);
}

Expand Down
1 change: 1 addition & 0 deletions templates/src/tokens.c.erb
Original file line number Diff line number Diff line change
Expand Up @@ -362,6 +362,7 @@ pm_token_str(pm_token_type_t token_type) {
case PM_TOKEN_USTAR_STAR:
return "**";
case PM_TOKEN_WORDS_SEP:
case PM_TOKEN_WORDS_SEP_IMPLICIT:
return "string separator";
case PM_TOKEN_XSTRING_BEGIN:
return "backtick string literal";
Expand Down
Loading