Ruby 4.1.0dev (2026-09-30 revision 5f5d65517e31dc76858b4b5924392ce2ced86f58)
prism.c
4
5#include "prism/internal/allocator.h"
6#include "prism/internal/arena.h"
7#include "prism/internal/bit.h"
8#include "prism/internal/buffer.h"
9#include "prism/internal/char.h"
10#include "prism/internal/comments.h"
11#include "prism/internal/constant_pool.h"
12#include "prism/internal/diagnostic.h"
13#include "prism/internal/encoding.h"
14#include "prism/internal/integer.h"
15#include "prism/internal/isinf.h"
16#include "prism/internal/line_offset_list.h"
17#include "prism/internal/list.h"
18#include "prism/internal/magic_comments.h"
19#include "prism/internal/memchr.h"
20#include "prism/internal/node.h"
21#include "prism/internal/options.h"
22#include "prism/internal/parser.h"
23#include "prism/internal/regexp.h"
24#include "prism/internal/serialize.h"
25#include "prism/internal/source.h"
26#include "prism/internal/static_literals.h"
27#include "prism/internal/stringy.h"
28#include "prism/internal/strncasecmp.h"
29#include "prism/internal/strpbrk.h"
30#include "prism/internal/tokens.h"
31
32#include "prism/excludes.h"
33#include "prism/serialize.h"
34#include "prism/stream.h"
35#include "prism/version.h"
36
37#include <assert.h>
38#include <errno.h>
39#include <limits.h>
40#include <locale.h>
41#include <math.h>
42#include <stdio.h>
43#include <stdlib.h>
44
50#ifndef PRISM_DEPTH_MAXIMUM
51 #define PRISM_DEPTH_MAXIMUM 10000
52#endif
53
58#define PM_CONCATENATE(left, right) left ## right
59
65#if defined(_Static_assert)
66# define PM_STATIC_ASSERT(line, condition, message) _Static_assert(condition, message)
67#else
68# define PM_STATIC_ASSERT(line, condition, message) typedef char PM_CONCATENATE(static_assert_, line)[(condition) ? 1 : -1]
69#endif
70
75#if defined(__GNUC__) || defined(__clang__)
77 #define PRISM_LIKELY(x) __builtin_expect(!!(x), 1)
78
80 #define PRISM_UNLIKELY(x) __builtin_expect(!!(x), 0)
81#else
83 #define PRISM_LIKELY(x) (x)
84
86 #define PRISM_UNLIKELY(x) (x)
87#endif
88
92const char *
93pm_version(void) {
94 return PRISM_VERSION;
95}
96
101#define PM_TAB_WHITESPACE_SIZE 8
102
103// Macros for min/max.
104#define MIN(a,b) (((a)<(b))?(a):(b))
105#define MAX(a,b) (((a)>(b))?(a):(b))
106
107/******************************************************************************/
108/* Helpful AST-related macros */
109/******************************************************************************/
110
111#define U32(value_) ((uint32_t) (value_))
112
113#define FL PM_NODE_FLAGS
114#define UP PM_NODE_UPCAST
115
116#define PM_LOCATION_START(location_) ((location_)->start)
117#define PM_LOCATION_END(location_) ((location_)->start + (location_)->length)
118
119#define PM_TOKEN_START(parser_, token_) U32((token_)->start - (parser_)->start)
120#define PM_TOKEN_END(parser_, token_) U32((token_)->end - (parser_)->start)
121#define PM_TOKEN_LENGTH(token_) U32((token_)->end - (token_)->start)
122#define PM_TOKENS_LENGTH(left_, right_) U32((right_)->end - (left_)->start)
123
124#define PM_NODE_START(node_) (UP(node_)->location.start)
125#define PM_NODE_LENGTH(node_) (UP(node_)->location.length)
126#define PM_NODE_END(node_) (UP(node_)->location.start + UP(node_)->location.length)
127#define PM_NODES_LENGTH(left_, right_) (PM_NODE_END(right_) - PM_NODE_START(left_))
128
129#define PM_TOKEN_NODE_LENGTH(parser_, token_, node_) (PM_NODE_END(node_) - PM_TOKEN_START(parser_, token_))
130#define PM_NODE_TOKEN_LENGTH(parser_, node_, token_) (PM_TOKEN_END(parser_, token_) - PM_NODE_START(node_))
131
132#define PM_NODE_START_SET_NODE(left_, right_) (PM_NODE_START(left_) = PM_NODE_START(right_))
133#define PM_NODE_START_SET_TOKEN(parser_, node_, token_) (PM_NODE_START(node_) = PM_TOKEN_START(parser_, token_))
134#define PM_NODE_LENGTH_SET_NODE(left_, right_) (PM_NODE_LENGTH(left_) = PM_NODE_END(right_) - PM_NODE_START(left_))
135#define PM_NODE_LENGTH_SET_TOKEN(parser_, node_, token_) (PM_NODE_LENGTH(node_) = PM_TOKEN_END(parser_, token_) - PM_NODE_START(node_))
136#define PM_NODE_LENGTH_SET_LOCATION(node_, location_) (PM_NODE_LENGTH(node_) = PM_LOCATION_END(location_) - PM_NODE_START(node_))
137
146pm_location_init(uint32_t start, uint32_t length) {
147 pm_location_t location = { .start = start, .length = length };
148 return location;
149}
150
151#define PM_LOCATION_INIT(start_, length_) pm_location_init((start_), (length_))
152#define PM_LOCATION_INIT_UNSET PM_LOCATION_INIT(0, 0)
153#define PM_LOCATION_INIT_TOKEN(parser_, token_) PM_LOCATION_INIT(PM_TOKEN_START(parser_, token_), PM_TOKEN_LENGTH(token_))
154#define PM_LOCATION_INIT_NODE(node_) UP(node_)->location
155
156#define PM_LOCATION_INIT_TOKENS(parser_, left_, right_) PM_LOCATION_INIT(PM_TOKEN_START(parser_, left_), PM_TOKENS_LENGTH(left_, right_))
157#define PM_LOCATION_INIT_NODES(left_, right_) PM_LOCATION_INIT(PM_NODE_START(left_), PM_NODES_LENGTH(left_, right_))
158#define PM_LOCATION_INIT_TOKEN_NODE(parser_, token_, node_) PM_LOCATION_INIT(PM_TOKEN_START(parser_, token_), PM_TOKEN_NODE_LENGTH(parser_, token_, node_))
159#define PM_LOCATION_INIT_NODE_TOKEN(parser_, node_, token_) PM_LOCATION_INIT(PM_NODE_START(node_), PM_NODE_TOKEN_LENGTH(parser_, node_, token_))
160
161#define TOK2LOC(parser_, token_) PM_LOCATION_INIT_TOKEN(parser_, token_)
162#define NTOK2LOC(parser_, token_) ((token_) == NULL ? PM_LOCATION_INIT_UNSET : TOK2LOC(parser_, token_))
163#define NTOK2PTR(token_) ((token_).start == NULL ? NULL : &(token_))
164
165/******************************************************************************/
166/* Lex mode manipulations */
167/******************************************************************************/
168
173static PRISM_INLINE uint8_t
174lex_mode_incrementor(const uint8_t start) {
175 switch (start) {
176 case '(':
177 case '[':
178 case '{':
179 case '<':
180 return start;
181 default:
182 return '\0';
183 }
184}
185
190static PRISM_INLINE uint8_t
191lex_mode_terminator(const uint8_t start) {
192 switch (start) {
193 case '(':
194 return ')';
195 case '[':
196 return ']';
197 case '{':
198 return '}';
199 case '<':
200 return '>';
201 default:
202 return start;
203 }
204}
205
211static bool
212lex_mode_push(pm_parser_t *parser, pm_lex_mode_t lex_mode) {
213 lex_mode.prev = parser->lex_modes.current;
214 parser->lex_modes.index++;
215
216 if (parser->lex_modes.index > PM_LEX_STACK_SIZE - 1) {
217 parser->lex_modes.current = (pm_lex_mode_t *) xmalloc(sizeof(pm_lex_mode_t));
218 if (parser->lex_modes.current == NULL) return false;
219
220 *parser->lex_modes.current = lex_mode;
221 } else {
222 parser->lex_modes.stack[parser->lex_modes.index] = lex_mode;
223 parser->lex_modes.current = &parser->lex_modes.stack[parser->lex_modes.index];
224 }
225
226 return true;
227}
228
232static PRISM_INLINE bool
233lex_mode_push_list(pm_parser_t *parser, bool interpolation, uint8_t delimiter) {
234 uint8_t incrementor = lex_mode_incrementor(delimiter);
235 uint8_t terminator = lex_mode_terminator(delimiter);
236
237 pm_lex_mode_t lex_mode = {
238 .mode = PM_LEX_LIST,
239 .as.list = {
240 .nesting = 0,
241 .interpolation = interpolation,
242 .started = false,
243 .separated = false,
244 .incrementor = incrementor,
245 .terminator = terminator
246 }
247 };
248
249 // These are the places where we need to split up the content of the list.
250 // We'll use strpbrk to find the first of these characters.
251 uint8_t *breakpoints = lex_mode.as.list.breakpoints;
252 memset(breakpoints, 0, PM_STRPBRK_CACHE_SIZE);
253 memcpy(breakpoints, "\\ \t\f\r\v\n", sizeof("\\ \t\f\r\v\n") - 1);
254 size_t index = 7;
255
256 // Now we'll add the terminator to the list of breakpoints. If the
257 // terminator is not already a NULL byte, add it to the list.
258 if (terminator != '\0') {
259 breakpoints[index++] = terminator;
260 }
261
262 // If interpolation is allowed, then we're going to check for the #
263 // character. Otherwise we'll only look for escapes and the terminator.
264 if (interpolation) {
265 breakpoints[index++] = '#';
266 }
267
268 // If there is an incrementor, then we'll check for that as well.
269 if (incrementor != '\0') {
270 breakpoints[index++] = incrementor;
271 }
272
273 parser->explicit_encoding = NULL;
274 return lex_mode_push(parser, lex_mode);
275}
276
282static PRISM_INLINE bool
283lex_mode_push_list_eof(pm_parser_t *parser) {
284 return lex_mode_push_list(parser, false, '\0');
285}
286
290static PRISM_INLINE bool
291lex_mode_push_regexp(pm_parser_t *parser, uint8_t incrementor, uint8_t terminator) {
292 pm_lex_mode_t lex_mode = {
293 .mode = PM_LEX_REGEXP,
294 .as.regexp = {
295 .nesting = 0,
296 .incrementor = incrementor,
297 .terminator = terminator
298 }
299 };
300
301 // These are the places where we need to split up the content of the
302 // regular expression. We'll use strpbrk to find the first of these
303 // characters.
304 uint8_t *breakpoints = lex_mode.as.regexp.breakpoints;
305 memset(breakpoints, 0, PM_STRPBRK_CACHE_SIZE);
306 memcpy(breakpoints, "\r\n\\#", sizeof("\r\n\\#") - 1);
307 size_t index = 4;
308
309 // First we'll add the terminator.
310 if (terminator != '\0') {
311 breakpoints[index++] = terminator;
312 }
313
314 // Next, if there is an incrementor, then we'll check for that as well.
315 if (incrementor != '\0') {
316 breakpoints[index++] = incrementor;
317 }
318
319 parser->explicit_encoding = NULL;
320 return lex_mode_push(parser, lex_mode);
321}
322
326static PRISM_INLINE bool
327lex_mode_push_string(pm_parser_t *parser, bool interpolation, bool label_allowed, uint8_t incrementor, uint8_t terminator) {
328 pm_lex_mode_t lex_mode = {
329 .mode = PM_LEX_STRING,
330 .as.string = {
331 .nesting = 0,
332 .interpolation = interpolation,
333 .label_allowed = label_allowed,
334 .incrementor = incrementor,
335 .terminator = terminator
336 }
337 };
338
339 // These are the places where we need to split up the content of the
340 // string. We'll use strpbrk to find the first of these characters.
341 uint8_t *breakpoints = lex_mode.as.string.breakpoints;
342 memset(breakpoints, 0, PM_STRPBRK_CACHE_SIZE);
343 memcpy(breakpoints, "\r\n\\", sizeof("\r\n\\") - 1);
344 size_t index = 3;
345
346 // Now add in the terminator. If the terminator is not already a NULL byte,
347 // then we'll add it.
348 if (terminator != '\0') {
349 breakpoints[index++] = terminator;
350 }
351
352 // If interpolation is allowed, then we're going to check for the #
353 // character. Otherwise we'll only look for escapes and the terminator.
354 if (interpolation) {
355 breakpoints[index++] = '#';
356 }
357
358 // If we have an incrementor, then we'll add that in as a breakpoint as
359 // well.
360 if (incrementor != '\0') {
361 breakpoints[index++] = incrementor;
362 }
363
364 parser->explicit_encoding = NULL;
365 return lex_mode_push(parser, lex_mode);
366}
367
373static PRISM_INLINE bool
374lex_mode_push_string_eof(pm_parser_t *parser) {
375 return lex_mode_push_string(parser, false, false, '\0', '\0');
376}
377
383static void
384lex_mode_pop(pm_parser_t *parser) {
385 if (parser->lex_modes.index == 0) {
386 parser->lex_modes.current->mode = PM_LEX_DEFAULT;
387 } else if (parser->lex_modes.index < PM_LEX_STACK_SIZE) {
388 parser->lex_modes.index--;
389 parser->lex_modes.current = &parser->lex_modes.stack[parser->lex_modes.index];
390 } else {
391 parser->lex_modes.index--;
392 pm_lex_mode_t *prev = parser->lex_modes.current->prev;
393 xfree_sized(parser->lex_modes.current, sizeof(pm_lex_mode_t));
394 parser->lex_modes.current = prev;
395 }
396}
397
401static PRISM_INLINE bool
402lex_state_p(const pm_parser_t *parser, pm_lex_state_t state) {
403 return parser->lex_state & state;
404}
405
406typedef enum {
407 PM_IGNORED_NEWLINE_NONE = 0,
408 PM_IGNORED_NEWLINE_ALL,
409 PM_IGNORED_NEWLINE_PATTERN
410} pm_ignored_newline_type_t;
411
412static PRISM_INLINE pm_ignored_newline_type_t
413lex_state_ignored_p(pm_parser_t *parser) {
414 bool ignored = lex_state_p(parser, PM_LEX_STATE_BEG | PM_LEX_STATE_CLASS | PM_LEX_STATE_FNAME | PM_LEX_STATE_DOT) && !lex_state_p(parser, PM_LEX_STATE_LABELED);
415
416 if (ignored) {
417 return PM_IGNORED_NEWLINE_ALL;
418 } else if ((parser->lex_state & ~((unsigned int) PM_LEX_STATE_LABEL)) == (PM_LEX_STATE_ARG | PM_LEX_STATE_LABELED)) {
419 return PM_IGNORED_NEWLINE_PATTERN;
420 } else {
421 return PM_IGNORED_NEWLINE_NONE;
422 }
423}
424
425static PRISM_INLINE bool
426lex_state_beg_p(pm_parser_t *parser) {
427 return lex_state_p(parser, PM_LEX_STATE_BEG_ANY) || ((parser->lex_state & (PM_LEX_STATE_ARG | PM_LEX_STATE_LABELED)) == (PM_LEX_STATE_ARG | PM_LEX_STATE_LABELED));
428}
429
430static PRISM_INLINE bool
431lex_state_arg_p(pm_parser_t *parser) {
432 return lex_state_p(parser, PM_LEX_STATE_ARG_ANY);
433}
434
435static PRISM_INLINE bool
436lex_state_spcarg_p(pm_parser_t *parser, bool space_seen) {
437 if (parser->current.end >= parser->end) {
438 return false;
439 }
440 return lex_state_arg_p(parser) && space_seen && !pm_char_is_whitespace(*parser->current.end);
441}
442
443static PRISM_INLINE bool
444lex_state_end_p(pm_parser_t *parser) {
445 return lex_state_p(parser, PM_LEX_STATE_END_ANY);
446}
447
451static PRISM_INLINE bool
452lex_state_operator_p(pm_parser_t *parser) {
453 return lex_state_p(parser, PM_LEX_STATE_FNAME | PM_LEX_STATE_DOT);
454}
455
460static PRISM_INLINE void
461lex_state_set(pm_parser_t *parser, pm_lex_state_t state) {
462 parser->lex_state = state;
463}
464
465#ifndef PM_DEBUG_LOGGING
470#define PM_DEBUG_LOGGING 0
471#endif
472
473#if PM_DEBUG_LOGGING
474PRISM_UNUSED static void
475debug_state(pm_parser_t *parser) {
476 fprintf(stderr, "STATE: ");
477 bool first = true;
478
479 if (parser->lex_state == PM_LEX_STATE_NONE) {
480 fprintf(stderr, "NONE\n");
481 return;
482 }
483
484#define CHECK_STATE(state) \
485 if (parser->lex_state & state) { \
486 if (!first) fprintf(stderr, "|"); \
487 fprintf(stderr, "%s", #state); \
488 first = false; \
489 }
490
491 CHECK_STATE(PM_LEX_STATE_BEG)
492 CHECK_STATE(PM_LEX_STATE_END)
493 CHECK_STATE(PM_LEX_STATE_ENDARG)
494 CHECK_STATE(PM_LEX_STATE_ENDFN)
495 CHECK_STATE(PM_LEX_STATE_ARG)
496 CHECK_STATE(PM_LEX_STATE_CMDARG)
497 CHECK_STATE(PM_LEX_STATE_MID)
498 CHECK_STATE(PM_LEX_STATE_FNAME)
499 CHECK_STATE(PM_LEX_STATE_DOT)
500 CHECK_STATE(PM_LEX_STATE_CLASS)
501 CHECK_STATE(PM_LEX_STATE_LABEL)
502 CHECK_STATE(PM_LEX_STATE_LABELED)
503 CHECK_STATE(PM_LEX_STATE_FITEM)
504
505#undef CHECK_STATE
506
507 fprintf(stderr, "\n");
508}
509
510static void
511debug_lex_state_set(pm_parser_t *parser, pm_lex_state_t state, char const * caller_name, int line_number) {
512 fprintf(stderr, "Caller: %s:%d\nPrevious: ", caller_name, line_number);
513 debug_state(parser);
514 lex_state_set(parser, state);
515 fprintf(stderr, "Now: ");
516 debug_state(parser);
517 fprintf(stderr, "\n");
518}
519
520#define lex_state_set(parser, state) debug_lex_state_set(parser, state, __func__, __LINE__)
521#endif
522
523/******************************************************************************/
524/* Command-line macro helpers */
525/******************************************************************************/
526
528#define PM_PARSER_COMMAND_LINE_OPTION(parser, option) ((parser)->command_line & (option))
529
531#define PM_PARSER_COMMAND_LINE_OPTION_A(parser) PM_PARSER_COMMAND_LINE_OPTION(parser, PM_OPTIONS_COMMAND_LINE_A)
532
534#define PM_PARSER_COMMAND_LINE_OPTION_E(parser) PM_PARSER_COMMAND_LINE_OPTION(parser, PM_OPTIONS_COMMAND_LINE_E)
535
537#define PM_PARSER_COMMAND_LINE_OPTION_L(parser) PM_PARSER_COMMAND_LINE_OPTION(parser, PM_OPTIONS_COMMAND_LINE_L)
538
540#define PM_PARSER_COMMAND_LINE_OPTION_N(parser) PM_PARSER_COMMAND_LINE_OPTION(parser, PM_OPTIONS_COMMAND_LINE_N)
541
543#define PM_PARSER_COMMAND_LINE_OPTION_P(parser) PM_PARSER_COMMAND_LINE_OPTION(parser, PM_OPTIONS_COMMAND_LINE_P)
544
546#define PM_PARSER_COMMAND_LINE_OPTION_X(parser) PM_PARSER_COMMAND_LINE_OPTION(parser, PM_OPTIONS_COMMAND_LINE_X)
547
548/******************************************************************************/
549/* Diagnostic-related functions */
550/******************************************************************************/
551
555static PRISM_INLINE void
556pm_parser_err(pm_parser_t *parser, uint32_t start, uint32_t length, pm_diagnostic_id_t diag_id) {
557 pm_diagnostic_list_append(&parser->metadata_arena, &parser->error_list, start, length, diag_id);
558}
559
564static PRISM_INLINE void
565pm_parser_err_token(pm_parser_t *parser, const pm_token_t *token, pm_diagnostic_id_t diag_id) {
566 pm_parser_err(parser, PM_TOKEN_START(parser, token), PM_TOKEN_LENGTH(token), diag_id);
567}
568
573static PRISM_INLINE void
574pm_parser_err_current(pm_parser_t *parser, pm_diagnostic_id_t diag_id) {
575 pm_parser_err_token(parser, &parser->current, diag_id);
576}
577
582static PRISM_INLINE void
583pm_parser_err_previous(pm_parser_t *parser, pm_diagnostic_id_t diag_id) {
584 pm_parser_err_token(parser, &parser->previous, diag_id);
585}
586
591static PRISM_INLINE void
592pm_parser_err_node(pm_parser_t *parser, const pm_node_t *node, pm_diagnostic_id_t diag_id) {
593 pm_parser_err(parser, PM_NODE_START(node), PM_NODE_LENGTH(node), diag_id);
594}
595
599#define PM_PARSER_ERR_FORMAT(parser_, start_, length_, diag_id_, ...) \
600 pm_diagnostic_list_append_format(&(parser_)->metadata_arena, &(parser_)->error_list, start_, length_, diag_id_, __VA_ARGS__)
601
606#define PM_PARSER_ERR_NODE_FORMAT(parser_, node_, diag_id_, ...) \
607 PM_PARSER_ERR_FORMAT(parser_, PM_NODE_START(node_), PM_NODE_LENGTH(node_), diag_id_, __VA_ARGS__)
608
613#define PM_PARSER_ERR_NODE_FORMAT_CONTENT(parser_, node_, diag_id_) \
614 PM_PARSER_ERR_NODE_FORMAT(parser_, node_, diag_id_, (int) PM_NODE_LENGTH(node_), (const char *) (parser_->start + PM_NODE_START(node_)))
615
620#define PM_PARSER_ERR_TOKEN_FORMAT(parser_, token_, diag_id, ...) \
621 PM_PARSER_ERR_FORMAT(parser_, PM_TOKEN_START(parser_, token_), PM_TOKEN_LENGTH(token_), diag_id, __VA_ARGS__)
622
627#define PM_PARSER_ERR_TOKEN_FORMAT_CONTENT(parser_, token_, diag_id_) \
628 PM_PARSER_ERR_TOKEN_FORMAT(parser_, token_, diag_id_, (int) PM_TOKEN_LENGTH(token_), (const char *) (token_)->start)
629
633static PRISM_INLINE void
634pm_parser_warn(pm_parser_t *parser, uint32_t start, uint32_t length, pm_diagnostic_id_t diag_id) {
635 pm_diagnostic_list_append(&parser->metadata_arena, &parser->warning_list, start, length, diag_id);
636}
637
642static PRISM_INLINE void
643pm_parser_warn_token(pm_parser_t *parser, const pm_token_t *token, pm_diagnostic_id_t diag_id) {
644 pm_parser_warn(parser, PM_TOKEN_START(parser, token), PM_TOKEN_LENGTH(token), diag_id);
645}
646
651static PRISM_INLINE void
652pm_parser_warn_node(pm_parser_t *parser, const pm_node_t *node, pm_diagnostic_id_t diag_id) {
653 pm_parser_warn(parser, PM_NODE_START(node), PM_NODE_LENGTH(node), diag_id);
654}
655
660#define PM_PARSER_WARN_FORMAT(parser_, start_, length_, diag_id_, ...) \
661 pm_diagnostic_list_append_format(&(parser_)->metadata_arena, &(parser_)->warning_list, start_, length_, diag_id_, __VA_ARGS__)
662
667#define PM_PARSER_WARN_TOKEN_FORMAT(parser_, token_, diag_id_, ...) \
668 PM_PARSER_WARN_FORMAT(parser_, PM_TOKEN_START(parser_, token_), PM_TOKEN_LENGTH(token_), diag_id_, __VA_ARGS__)
669
674#define PM_PARSER_WARN_TOKEN_FORMAT_CONTENT(parser_, token_, diag_id_) \
675 PM_PARSER_WARN_TOKEN_FORMAT(parser_, token_, diag_id_, (int) PM_TOKEN_LENGTH(token_), (const char *) (token_)->start)
676
681#define PM_PARSER_WARN_NODE_FORMAT(parser_, node_, diag_id_, ...) \
682 PM_PARSER_WARN_FORMAT(parser_, PM_NODE_START(node_), PM_NODE_LENGTH(node_), diag_id_, __VA_ARGS__)
683
689static void
690pm_parser_err_heredoc_term(pm_parser_t *parser, const uint8_t *ident_start, size_t ident_length) {
691 PM_PARSER_ERR_FORMAT(
692 parser,
693 U32(ident_start - parser->start),
694 U32(ident_length),
695 PM_ERR_HEREDOC_TERM,
696 (int) ident_length,
697 (const char *) ident_start
698 );
699}
700
701/******************************************************************************/
702/* Scope-related functions */
703/******************************************************************************/
704
708static bool
709pm_parser_scope_push(pm_parser_t *parser, bool closed) {
710 pm_scope_t *scope = (pm_scope_t *) xmalloc(sizeof(pm_scope_t));
711 if (scope == NULL) return false;
712
713 *scope = (pm_scope_t) {
714 .previous = parser->current_scope,
715 .locals = { 0 },
716 .parameters = PM_SCOPE_PARAMETERS_NONE,
717 .implicit_parameters = { 0 },
718 .shareable_constant = parser->current_scope == NULL ? PM_SCOPE_SHAREABLE_CONSTANT_NONE : parser->current_scope->shareable_constant,
719 .closed = closed
720 };
721
722 parser->current_scope = scope;
723 return true;
724}
725
730static bool
731pm_parser_scope_toplevel_p(pm_parser_t *parser) {
732 pm_scope_t *scope = parser->current_scope;
733
734 do {
735 if (scope->previous == NULL) return true;
736 if (scope->closed) return false;
737 } while ((scope = scope->previous) != NULL);
738
739 assert(false && "unreachable");
740 return true;
741}
742
746static pm_scope_t *
747pm_parser_scope_find(pm_parser_t *parser, uint32_t depth) {
748 pm_scope_t *scope = parser->current_scope;
749
750 while (depth-- > 0) {
751 assert(scope != NULL);
752 scope = scope->previous;
753 }
754
755 return scope;
756}
757
758typedef enum {
759 PM_SCOPE_FORWARDING_PARAM_CHECK_RESULT_PASS,
760 PM_SCOPE_FORWARDING_PARAM_CHECK_RESULT_CONFLICT,
761 PM_SCOPE_FORWARDING_PARAM_CHECK_RESULT_FAIL
762} pm_scope_forwarding_param_check_result_t;
763
764static pm_scope_forwarding_param_check_result_t
765pm_parser_scope_forwarding_param_check(pm_parser_t *parser, const uint8_t mask) {
766 pm_scope_t *scope = parser->current_scope;
767 bool conflict = false;
768
769 while (scope != NULL) {
770 if (scope->parameters & mask) {
771 if (scope->closed) {
772 if (conflict) {
773 return PM_SCOPE_FORWARDING_PARAM_CHECK_RESULT_CONFLICT;
774 } else {
775 return PM_SCOPE_FORWARDING_PARAM_CHECK_RESULT_PASS;
776 }
777 }
778
779 conflict = true;
780 }
781
782 if (scope->closed) break;
783 scope = scope->previous;
784 }
785
786 return PM_SCOPE_FORWARDING_PARAM_CHECK_RESULT_FAIL;
787}
788
789static void
790pm_parser_scope_forwarding_block_check(pm_parser_t *parser, const pm_token_t * token) {
791 switch (pm_parser_scope_forwarding_param_check(parser, PM_SCOPE_PARAMETERS_FORWARDING_BLOCK)) {
792 case PM_SCOPE_FORWARDING_PARAM_CHECK_RESULT_PASS:
793 // Pass.
794 break;
795 case PM_SCOPE_FORWARDING_PARAM_CHECK_RESULT_CONFLICT:
796 pm_parser_err_token(parser, token, PM_ERR_ARGUMENT_CONFLICT_AMPERSAND);
797 break;
798 case PM_SCOPE_FORWARDING_PARAM_CHECK_RESULT_FAIL:
799 pm_parser_err_token(parser, token, PM_ERR_ARGUMENT_NO_FORWARDING_AMPERSAND);
800 break;
801 }
802}
803
804static void
805pm_parser_scope_forwarding_positionals_check(pm_parser_t *parser, const pm_token_t * token) {
806 switch (pm_parser_scope_forwarding_param_check(parser, PM_SCOPE_PARAMETERS_FORWARDING_POSITIONALS)) {
807 case PM_SCOPE_FORWARDING_PARAM_CHECK_RESULT_PASS:
808 // Pass.
809 break;
810 case PM_SCOPE_FORWARDING_PARAM_CHECK_RESULT_CONFLICT:
811 pm_parser_err_token(parser, token, PM_ERR_ARGUMENT_CONFLICT_STAR);
812 break;
813 case PM_SCOPE_FORWARDING_PARAM_CHECK_RESULT_FAIL:
814 pm_parser_err_token(parser, token, PM_ERR_ARGUMENT_NO_FORWARDING_STAR);
815 break;
816 }
817}
818
819static void
820pm_parser_scope_forwarding_all_check(pm_parser_t *parser, const pm_token_t *token) {
821 switch (pm_parser_scope_forwarding_param_check(parser, PM_SCOPE_PARAMETERS_FORWARDING_ALL)) {
822 case PM_SCOPE_FORWARDING_PARAM_CHECK_RESULT_PASS:
823 // Pass.
824 break;
825 case PM_SCOPE_FORWARDING_PARAM_CHECK_RESULT_CONFLICT:
826 // This shouldn't happen, because ... is not allowed in the
827 // declaration of blocks. If we get here, we assume we already have
828 // an error for this.
829 break;
830 case PM_SCOPE_FORWARDING_PARAM_CHECK_RESULT_FAIL:
831 pm_parser_err_token(parser, token, PM_ERR_ARGUMENT_NO_FORWARDING_ELLIPSES);
832 break;
833 }
834}
835
836static void
837pm_parser_scope_forwarding_keywords_check(pm_parser_t *parser, const pm_token_t * token) {
838 switch (pm_parser_scope_forwarding_param_check(parser, PM_SCOPE_PARAMETERS_FORWARDING_KEYWORDS)) {
839 case PM_SCOPE_FORWARDING_PARAM_CHECK_RESULT_PASS:
840 // Pass.
841 break;
842 case PM_SCOPE_FORWARDING_PARAM_CHECK_RESULT_CONFLICT:
843 pm_parser_err_token(parser, token, PM_ERR_ARGUMENT_CONFLICT_STAR_STAR);
844 break;
845 case PM_SCOPE_FORWARDING_PARAM_CHECK_RESULT_FAIL:
846 pm_parser_err_token(parser, token, PM_ERR_ARGUMENT_NO_FORWARDING_STAR_STAR);
847 break;
848 }
849}
850
854static PRISM_INLINE pm_shareable_constant_value_t
855pm_parser_scope_shareable_constant_get(pm_parser_t *parser) {
856 return parser->current_scope->shareable_constant;
857}
858
863static void
864pm_parser_scope_shareable_constant_set(pm_parser_t *parser, pm_shareable_constant_value_t shareable_constant) {
865 pm_scope_t *scope = parser->current_scope;
866
867 do {
868 scope->shareable_constant = shareable_constant;
869 } while (!scope->closed && (scope = scope->previous) != NULL);
870}
871
872/******************************************************************************/
873/* Local variable-related functions */
874/******************************************************************************/
875
879#define PM_LOCALS_HASH_THRESHOLD 5
880
881static void
882pm_locals_free(pm_locals_t *locals) {
883 if (locals->capacity > 0) {
884 xfree_sized(locals->locals, locals->capacity * sizeof(pm_local_t));
885 }
886}
887
892static void
893pm_locals_resize(pm_locals_t *locals) {
894 uint32_t next_capacity = locals->capacity == 0 ? 4 : (locals->capacity * 2);
895 assert(next_capacity > locals->capacity);
896
897 pm_local_t *next_locals = xcalloc(next_capacity, sizeof(pm_local_t));
898 if (next_locals == NULL) abort();
899
900 if (next_capacity < PM_LOCALS_HASH_THRESHOLD) {
901 if (locals->size > 0) {
902 memcpy(next_locals, locals->locals, locals->size * sizeof(pm_local_t));
903 }
904 } else {
905 // If we just switched from a list to a hash, then we need to fill in
906 // the hash values of all of the locals.
907 bool hash_needed = (locals->capacity <= PM_LOCALS_HASH_THRESHOLD);
908 uint32_t mask = next_capacity - 1;
909
910 for (uint32_t index = 0; index < locals->capacity; index++) {
911 pm_local_t *local = &locals->locals[index];
912
913 if (local->name != PM_CONSTANT_ID_UNSET) {
914 if (hash_needed) local->hash = pm_constant_id_hash(local->name);
915
916 uint32_t hash = local->hash;
917 while (next_locals[hash & mask].name != PM_CONSTANT_ID_UNSET) hash++;
918 next_locals[hash & mask] = *local;
919 }
920 }
921 }
922
923 pm_locals_free(locals);
924 locals->locals = next_locals;
925 locals->capacity = next_capacity;
926}
927
943static bool
944pm_locals_write(pm_locals_t *locals, pm_constant_id_t name, uint32_t start, uint32_t length, uint32_t reads) {
945 if (locals->size >= (locals->capacity / 4 * 3)) {
946 pm_locals_resize(locals);
947 }
948
949 locals->bloom |= (1u << (name & 31));
950
951 if (locals->capacity < PM_LOCALS_HASH_THRESHOLD) {
952 for (uint32_t index = 0; index < locals->capacity; index++) {
953 pm_local_t *local = &locals->locals[index];
954
955 if (local->name == PM_CONSTANT_ID_UNSET) {
956 *local = (pm_local_t) {
957 .name = name,
958 .location = { .start = start, .length = length },
959 .index = locals->size++,
960 .reads = reads,
961 .hash = 0
962 };
963 return true;
964 } else if (local->name == name) {
965 return false;
966 }
967 }
968 } else {
969 uint32_t mask = locals->capacity - 1;
970 uint32_t hash = pm_constant_id_hash(name);
971 uint32_t initial_hash = hash;
972
973 do {
974 pm_local_t *local = &locals->locals[hash & mask];
975
976 if (local->name == PM_CONSTANT_ID_UNSET) {
977 *local = (pm_local_t) {
978 .name = name,
979 .location = { .start = start, .length = length },
980 .index = locals->size++,
981 .reads = reads,
982 .hash = initial_hash
983 };
984 return true;
985 } else if (local->name == name) {
986 return false;
987 } else {
988 hash++;
989 }
990 } while ((hash & mask) != initial_hash);
991 }
992
993 assert(false && "unreachable");
994 return true;
995}
996
1001static uint32_t
1002pm_locals_find(pm_locals_t *locals, pm_constant_id_t name) {
1003 if (!(locals->bloom & (1u << (name & 31)))) return UINT32_MAX;
1004
1005 if (locals->capacity < PM_LOCALS_HASH_THRESHOLD) {
1006 for (uint32_t index = 0; index < locals->size; index++) {
1007 pm_local_t *local = &locals->locals[index];
1008 if (local->name == name) return index;
1009 }
1010 } else {
1011 uint32_t mask = locals->capacity - 1;
1012 uint32_t hash = pm_constant_id_hash(name);
1013 uint32_t initial_hash = hash & mask;
1014
1015 do {
1016 pm_local_t *local = &locals->locals[hash & mask];
1017
1018 if (local->name == PM_CONSTANT_ID_UNSET) {
1019 return UINT32_MAX;
1020 } else if (local->name == name) {
1021 return hash & mask;
1022 } else {
1023 hash++;
1024 }
1025 } while ((hash & mask) != initial_hash);
1026 }
1027
1028 return UINT32_MAX;
1029}
1030
1035static void
1036pm_locals_read(pm_locals_t *locals, pm_constant_id_t name) {
1037 uint32_t index = pm_locals_find(locals, name);
1038 assert(index != UINT32_MAX);
1039
1040 pm_local_t *local = &locals->locals[index];
1041 assert(local->reads < UINT32_MAX);
1042
1043 local->reads++;
1044}
1045
1050static void
1051pm_locals_unread(pm_locals_t *locals, pm_constant_id_t name) {
1052 uint32_t index = pm_locals_find(locals, name);
1053 assert(index != UINT32_MAX);
1054
1055 pm_local_t *local = &locals->locals[index];
1056 assert(local->reads > 0);
1057
1058 local->reads--;
1059}
1060
1064static uint32_t
1065pm_locals_reads(pm_locals_t *locals, pm_constant_id_t name) {
1066 uint32_t index = pm_locals_find(locals, name);
1067 assert(index != UINT32_MAX);
1068
1069 return locals->locals[index].reads;
1070}
1071
1080static void
1081pm_locals_order(pm_parser_t *parser, pm_locals_t *locals, pm_constant_id_list_t *list, bool toplevel) {
1082 pm_constant_id_list_init_capacity(parser->arena, list, locals->size);
1083
1084 // If we're still below the threshold for switching to a hash, then we only
1085 // need to loop over the locals until we hit the size because the locals are
1086 // stored in a list.
1087 uint32_t capacity = locals->capacity < PM_LOCALS_HASH_THRESHOLD ? locals->size : locals->capacity;
1088
1089 // We will only warn for unused variables if we're not at the top level, or
1090 // if we're parsing a file outside of eval or -e.
1091 bool warn_unused = !toplevel || (!parser->parsing_eval && !PM_PARSER_COMMAND_LINE_OPTION_E(parser));
1092
1093 for (uint32_t index = 0; index < capacity; index++) {
1094 pm_local_t *local = &locals->locals[index];
1095
1096 if (local->name != PM_CONSTANT_ID_UNSET) {
1097 pm_constant_id_list_insert(list, (size_t) local->index, local->name);
1098
1099 if (warn_unused && local->reads == 0 && ((parser->start_line >= 0) || (pm_line_offset_list_line(&parser->line_offsets, local->location.start, parser->start_line) >= 0))) {
1100 pm_constant_t *constant = pm_constant_pool_id_to_constant(&parser->constant_pool, local->name);
1101
1102 if (constant->length >= 1 && *constant->start != '_') {
1103 PM_PARSER_WARN_FORMAT(
1104 parser,
1105 local->location.start,
1106 local->location.length,
1107 PM_WARN_UNUSED_LOCAL_VARIABLE,
1108 (int) constant->length,
1109 (const char *) constant->start
1110 );
1111 }
1112 }
1113 }
1114 }
1115}
1116
1117/******************************************************************************/
1118/* Node-related functions */
1119/******************************************************************************/
1120
1125pm_parser_constant_id_raw(pm_parser_t *parser, const uint8_t *start, const uint8_t *end) {
1126 /* Fast path: if this is the same token as the last lookup (same pointer
1127 * range), return the cached result. */
1128 if (start == parser->constant_cache.start && end == parser->constant_cache.end) {
1129 return parser->constant_cache.id;
1130 }
1131
1132 pm_constant_id_t id = pm_constant_pool_insert_shared(&parser->metadata_arena, &parser->constant_pool, start, (size_t) (end - start));
1133
1134 parser->constant_cache.start = start;
1135 parser->constant_cache.end = end;
1136 parser->constant_cache.id = id;
1137
1138 return id;
1139}
1140
1145pm_parser_constant_id_owned(pm_parser_t *parser, uint8_t *start, size_t length) {
1146 return pm_constant_pool_insert_owned(&parser->metadata_arena, &parser->constant_pool, start, length);
1147}
1148
1153pm_parser_constant_id_constant(pm_parser_t *parser, const char *start, size_t length) {
1154 return pm_constant_pool_insert_constant(&parser->metadata_arena, &parser->constant_pool, (const uint8_t *) start, length);
1155}
1156
1161pm_parser_constant_id_token(pm_parser_t *parser, const pm_token_t *token) {
1162 return pm_parser_constant_id_raw(parser, token->start, token->end);
1163}
1164
1169#define PM_CASE_VOID_VALUE PM_RETURN_NODE: case PM_BREAK_NODE: case PM_NEXT_NODE: \
1170 case PM_REDO_NODE: case PM_RETRY_NODE: case PM_MATCH_REQUIRED_NODE
1171
1177static pm_node_t *
1178pm_check_value_expression(pm_parser_t *parser, pm_node_t *node) {
1179 pm_node_t *void_node = NULL;
1180
1181 while (node != NULL) {
1182 switch (PM_NODE_TYPE(node)) {
1183 case PM_CASE_VOID_VALUE:
1184 return void_node != NULL ? void_node : node;
1185 case PM_MATCH_PREDICATE_NODE:
1186 return NULL;
1187 case PM_BEGIN_NODE: {
1188 pm_begin_node_t *cast = (pm_begin_node_t *) node;
1189
1190 if (cast->ensure_clause != NULL) {
1191 if (cast->rescue_clause != NULL) {
1192 pm_node_t *vn = pm_check_value_expression(parser, UP(cast->rescue_clause));
1193 if (vn != NULL) return vn;
1194 }
1195
1196 if (cast->statements != NULL) {
1197 pm_node_t *vn = pm_check_value_expression(parser, UP(cast->statements));
1198 if (vn != NULL) return vn;
1199 }
1200
1201 node = UP(cast->ensure_clause);
1202 } else if (cast->rescue_clause != NULL) {
1203 // https://bugs.ruby-lang.org/issues/21669
1204 if (cast->else_clause == NULL || parser->version < PM_OPTIONS_VERSION_CRUBY_4_1) {
1205 if (cast->statements == NULL) return NULL;
1206
1207 pm_node_t *vn = pm_check_value_expression(parser, UP(cast->statements));
1208 if (vn == NULL) return NULL;
1209 if (void_node == NULL) void_node = vn;
1210 }
1211
1212 for (pm_rescue_node_t *rescue_clause = cast->rescue_clause; rescue_clause != NULL; rescue_clause = rescue_clause->subsequent) {
1213 pm_node_t *vn = pm_check_value_expression(parser, UP(rescue_clause->statements));
1214
1215 if (vn == NULL) {
1216 // https://bugs.ruby-lang.org/issues/21669
1217 if (parser->version >= PM_OPTIONS_VERSION_CRUBY_4_1) {
1218 return NULL;
1219 }
1220 void_node = NULL;
1221 break;
1222 }
1223 }
1224
1225 if (cast->else_clause != NULL) {
1226 node = UP(cast->else_clause);
1227
1228 // https://bugs.ruby-lang.org/issues/21669
1229 if (parser->version >= PM_OPTIONS_VERSION_CRUBY_4_1) {
1230 pm_node_t *vn = pm_check_value_expression(parser, node);
1231 if (vn != NULL) return vn;
1232 }
1233 } else {
1234 return void_node;
1235 }
1236 } else {
1237 node = UP(cast->statements);
1238 }
1239
1240 break;
1241 }
1242 case PM_CASE_NODE: {
1243 // https://bugs.ruby-lang.org/issues/21669
1244 if (parser->version < PM_OPTIONS_VERSION_CRUBY_4_1) {
1245 return NULL;
1246 }
1247
1248 pm_case_node_t *cast = (pm_case_node_t *) node;
1249 if (cast->else_clause == NULL) return NULL;
1250
1251 pm_node_t *condition;
1252 PM_NODE_LIST_FOREACH(&cast->conditions, index, condition) {
1253 assert(PM_NODE_TYPE_P(condition, PM_WHEN_NODE));
1254
1255 pm_when_node_t *cast = (pm_when_node_t *) condition;
1256 pm_node_t *vn = pm_check_value_expression(parser, UP(cast->statements));
1257 if (vn == NULL) return NULL;
1258 if (void_node == NULL) void_node = vn;
1259 }
1260
1261 node = UP(cast->else_clause);
1262 break;
1263 }
1264 case PM_CASE_MATCH_NODE: {
1265 // https://bugs.ruby-lang.org/issues/21669
1266 if (parser->version < PM_OPTIONS_VERSION_CRUBY_4_1) {
1267 return NULL;
1268 }
1269
1271 if (cast->else_clause == NULL) return NULL;
1272
1273 pm_node_t *condition;
1274 PM_NODE_LIST_FOREACH(&cast->conditions, index, condition) {
1275 assert(PM_NODE_TYPE_P(condition, PM_IN_NODE));
1276
1277 pm_in_node_t *cast = (pm_in_node_t *) condition;
1278 pm_node_t *vn = pm_check_value_expression(parser, UP(cast->statements));
1279 if (vn == NULL) return NULL;
1280 if (void_node == NULL) void_node = vn;
1281 }
1282
1283 node = UP(cast->else_clause);
1284 break;
1285 }
1286 case PM_ENSURE_NODE: {
1287 pm_ensure_node_t *cast = (pm_ensure_node_t *) node;
1288 node = UP(cast->statements);
1289 break;
1290 }
1291 case PM_PARENTHESES_NODE: {
1293 node = UP(cast->body);
1294 break;
1295 }
1296 case PM_STATEMENTS_NODE: {
1298
1299 // https://bugs.ruby-lang.org/issues/21669
1300 if (parser->version >= PM_OPTIONS_VERSION_CRUBY_4_1) {
1301 pm_node_t *body_part;
1302 PM_NODE_LIST_FOREACH(&cast->body, index, body_part) {
1303 switch (PM_NODE_TYPE(body_part)) {
1304 case PM_CASE_VOID_VALUE:
1305 if (void_node == NULL) {
1306 void_node = body_part;
1307 }
1308 return void_node;
1309 default: break;
1310 }
1311 }
1312 }
1313
1314 node = cast->body.nodes[cast->body.size - 1];
1315 break;
1316 }
1317 case PM_IF_NODE: {
1318 pm_if_node_t *cast = (pm_if_node_t *) node;
1319 if (cast->statements == NULL || cast->subsequent == NULL) {
1320 return NULL;
1321 }
1322 pm_node_t *vn = pm_check_value_expression(parser, UP(cast->statements));
1323 if (vn == NULL) {
1324 return NULL;
1325 }
1326 if (void_node == NULL) {
1327 void_node = vn;
1328 }
1329 node = cast->subsequent;
1330 break;
1331 }
1332 case PM_UNLESS_NODE: {
1333 pm_unless_node_t *cast = (pm_unless_node_t *) node;
1334 if (cast->statements == NULL || cast->else_clause == NULL) {
1335 return NULL;
1336 }
1337 pm_node_t *vn = pm_check_value_expression(parser, UP(cast->statements));
1338 if (vn == NULL) {
1339 return NULL;
1340 }
1341 if (void_node == NULL) {
1342 void_node = vn;
1343 }
1344 node = UP(cast->else_clause);
1345 break;
1346 }
1347 case PM_ELSE_NODE: {
1348 pm_else_node_t *cast = (pm_else_node_t *) node;
1349 node = UP(cast->statements);
1350 break;
1351 }
1352 case PM_AND_NODE:
1353 case PM_OR_NODE:
1354 // The left operand of an and/or node was already checked for a
1355 // value when the node was created, so descending into it again
1356 // would re-report the same void value and, in a chain such as
1357 // `a && a && ...`, walk the whole left branch on every operator,
1358 // which is quadratic in the length of the chain.
1359 return NULL;
1360 case PM_LOCAL_VARIABLE_WRITE_NODE: {
1362
1363 pm_scope_t *scope = parser->current_scope;
1364 for (uint32_t depth = 0; depth < cast->depth; depth++) scope = scope->previous;
1365
1366 pm_locals_read(&scope->locals, cast->name);
1367 return NULL;
1368 }
1369 default:
1370 return NULL;
1371 }
1372 }
1373
1374 return NULL;
1375}
1376
1377static PRISM_INLINE void
1378pm_assert_value_expression(pm_parser_t *parser, pm_node_t *node) {
1379 pm_node_t *void_node = pm_check_value_expression(parser, node);
1380 if (void_node != NULL) {
1381 pm_parser_err_node(parser, void_node, PM_ERR_VOID_EXPRESSION);
1382 }
1383}
1384
1388static void
1389pm_void_statement_check(pm_parser_t *parser, const pm_node_t *node) {
1390 const char *type = NULL;
1391 int length = 0;
1392
1393 switch (PM_NODE_TYPE(node)) {
1394 case PM_BACK_REFERENCE_READ_NODE:
1395 case PM_CLASS_VARIABLE_READ_NODE:
1396 case PM_GLOBAL_VARIABLE_READ_NODE:
1397 case PM_INSTANCE_VARIABLE_READ_NODE:
1398 case PM_LOCAL_VARIABLE_READ_NODE:
1399 case PM_NUMBERED_REFERENCE_READ_NODE:
1400 type = "a variable";
1401 length = 10;
1402 break;
1403 case PM_CALL_NODE: {
1404 const pm_call_node_t *cast = (const pm_call_node_t *) node;
1405 if (cast->call_operator_loc.length > 0 || cast->message_loc.length == 0) break;
1406
1407 const pm_constant_t *message = pm_constant_pool_id_to_constant(&parser->constant_pool, cast->name);
1408 switch (message->length) {
1409 case 1:
1410 switch (message->start[0]) {
1411 case '+':
1412 case '-':
1413 case '*':
1414 case '/':
1415 case '%':
1416 case '|':
1417 case '^':
1418 case '&':
1419 case '>':
1420 case '<':
1421 type = (const char *) message->start;
1422 length = 1;
1423 break;
1424 }
1425 break;
1426 case 2:
1427 switch (message->start[1]) {
1428 case '=':
1429 if (message->start[0] == '<' || message->start[0] == '>' || message->start[0] == '!' || message->start[0] == '=') {
1430 type = (const char *) message->start;
1431 length = 2;
1432 }
1433 break;
1434 case '@':
1435 if (message->start[0] == '+' || message->start[0] == '-') {
1436 type = (const char *) message->start;
1437 length = 2;
1438 }
1439 break;
1440 case '*':
1441 if (message->start[0] == '*') {
1442 type = (const char *) message->start;
1443 length = 2;
1444 }
1445 break;
1446 }
1447 break;
1448 case 3:
1449 if (memcmp(message->start, "<=>", 3) == 0) {
1450 type = "<=>";
1451 length = 3;
1452 }
1453 break;
1454 }
1455
1456 break;
1457 }
1458 case PM_CONSTANT_PATH_NODE:
1459 type = "::";
1460 length = 2;
1461 break;
1462 case PM_CONSTANT_READ_NODE:
1463 type = "a constant";
1464 length = 10;
1465 break;
1466 case PM_DEFINED_NODE:
1467 type = "defined?";
1468 length = 8;
1469 break;
1470 case PM_FALSE_NODE:
1471 type = "false";
1472 length = 5;
1473 break;
1474 case PM_FLOAT_NODE:
1475 case PM_IMAGINARY_NODE:
1476 case PM_INTEGER_NODE:
1477 case PM_INTERPOLATED_REGULAR_EXPRESSION_NODE:
1478 case PM_INTERPOLATED_STRING_NODE:
1479 case PM_RATIONAL_NODE:
1480 case PM_REGULAR_EXPRESSION_NODE:
1481 case PM_SOURCE_ENCODING_NODE:
1482 case PM_SOURCE_FILE_NODE:
1483 case PM_SOURCE_LINE_NODE:
1484 case PM_STRING_NODE:
1485 case PM_SYMBOL_NODE:
1486 type = "a literal";
1487 length = 9;
1488 break;
1489 case PM_NIL_NODE:
1490 type = "nil";
1491 length = 3;
1492 break;
1493 case PM_RANGE_NODE: {
1494 const pm_range_node_t *cast = (const pm_range_node_t *) node;
1495
1496 if (PM_NODE_FLAG_P(cast, PM_RANGE_FLAGS_EXCLUDE_END)) {
1497 type = "...";
1498 length = 3;
1499 } else {
1500 type = "..";
1501 length = 2;
1502 }
1503
1504 break;
1505 }
1506 case PM_SELF_NODE:
1507 type = "self";
1508 length = 4;
1509 break;
1510 case PM_TRUE_NODE:
1511 type = "true";
1512 length = 4;
1513 break;
1514 default:
1515 break;
1516 }
1517
1518 if (type != NULL) {
1519 PM_PARSER_WARN_NODE_FORMAT(parser, node, PM_WARN_VOID_STATEMENT, length, type);
1520 }
1521}
1522
1527static void
1528pm_void_statements_check(pm_parser_t *parser, const pm_statements_node_t *node, bool last_value) {
1529 assert(node->body.size > 0);
1530 const size_t size = node->body.size - (last_value ? 1 : 0);
1531 for (size_t index = 0; index < size; index++) {
1532 pm_void_statement_check(parser, node->body.nodes[index]);
1533 }
1534}
1535
1541typedef enum {
1542 PM_CONDITIONAL_PREDICATE_TYPE_CONDITIONAL,
1543 PM_CONDITIONAL_PREDICATE_TYPE_FLIP_FLOP,
1544 PM_CONDITIONAL_PREDICATE_TYPE_NOT
1545} pm_conditional_predicate_type_t;
1546
1550static void
1551pm_parser_warn_conditional_predicate_literal(pm_parser_t *parser, pm_node_t *node, pm_conditional_predicate_type_t type, pm_diagnostic_id_t diag_id, const char *prefix) {
1552 switch (type) {
1553 case PM_CONDITIONAL_PREDICATE_TYPE_CONDITIONAL:
1554 PM_PARSER_WARN_NODE_FORMAT(parser, node, diag_id, prefix, "condition");
1555 break;
1556 case PM_CONDITIONAL_PREDICATE_TYPE_FLIP_FLOP:
1557 PM_PARSER_WARN_NODE_FORMAT(parser, node, diag_id, prefix, "flip-flop");
1558 break;
1559 case PM_CONDITIONAL_PREDICATE_TYPE_NOT:
1560 break;
1561 }
1562}
1563
1568static bool
1569pm_conditional_predicate_warn_write_literal_p(const pm_node_t *node) {
1570 switch (PM_NODE_TYPE(node)) {
1571 case PM_ARRAY_NODE: {
1572 if (PM_NODE_FLAG_P(node, PM_NODE_FLAG_STATIC_LITERAL)) return true;
1573
1574 const pm_array_node_t *cast = (const pm_array_node_t *) node;
1575 for (size_t index = 0; index < cast->elements.size; index++) {
1576 if (!pm_conditional_predicate_warn_write_literal_p(cast->elements.nodes[index])) return false;
1577 }
1578
1579 return true;
1580 }
1581 case PM_HASH_NODE: {
1582 if (PM_NODE_FLAG_P(node, PM_NODE_FLAG_STATIC_LITERAL)) return true;
1583
1584 const pm_hash_node_t *cast = (const pm_hash_node_t *) node;
1585 for (size_t index = 0; index < cast->elements.size; index++) {
1586 const pm_node_t *element = cast->elements.nodes[index];
1587 if (!PM_NODE_TYPE_P(element, PM_ASSOC_NODE)) return false;
1588
1589 const pm_assoc_node_t *assoc = (const pm_assoc_node_t *) element;
1590 if (!pm_conditional_predicate_warn_write_literal_p(assoc->key) || !pm_conditional_predicate_warn_write_literal_p(assoc->value)) return false;
1591 }
1592
1593 return true;
1594 }
1595 case PM_FALSE_NODE:
1596 case PM_FLOAT_NODE:
1597 case PM_IMAGINARY_NODE:
1598 case PM_INTEGER_NODE:
1599 case PM_NIL_NODE:
1600 case PM_RATIONAL_NODE:
1601 case PM_REGULAR_EXPRESSION_NODE:
1602 case PM_SOURCE_ENCODING_NODE:
1603 case PM_SOURCE_FILE_NODE:
1604 case PM_SOURCE_LINE_NODE:
1605 case PM_STRING_NODE:
1606 case PM_SYMBOL_NODE:
1607 case PM_TRUE_NODE:
1608 return true;
1609 default:
1610 return false;
1611 }
1612}
1613
1618static PRISM_INLINE void
1619pm_conditional_predicate_warn_write_literal(pm_parser_t *parser, const pm_node_t *node) {
1620 if (pm_conditional_predicate_warn_write_literal_p(node)) {
1621 pm_parser_warn_node(parser, node, parser->version <= PM_OPTIONS_VERSION_CRUBY_3_3 ? PM_WARN_EQUAL_IN_CONDITIONAL_3_3 : PM_WARN_EQUAL_IN_CONDITIONAL);
1622 }
1623}
1624
1637static void
1638pm_conditional_predicate(pm_parser_t *parser, pm_node_t *node, pm_conditional_predicate_type_t type) {
1639 switch (PM_NODE_TYPE(node)) {
1640 case PM_AND_NODE: {
1641 pm_and_node_t *cast = (pm_and_node_t *) node;
1642 pm_conditional_predicate(parser, cast->left, PM_CONDITIONAL_PREDICATE_TYPE_CONDITIONAL);
1643 pm_conditional_predicate(parser, cast->right, PM_CONDITIONAL_PREDICATE_TYPE_CONDITIONAL);
1644 break;
1645 }
1646 case PM_OR_NODE: {
1647 pm_or_node_t *cast = (pm_or_node_t *) node;
1648 pm_conditional_predicate(parser, cast->left, PM_CONDITIONAL_PREDICATE_TYPE_CONDITIONAL);
1649 pm_conditional_predicate(parser, cast->right, PM_CONDITIONAL_PREDICATE_TYPE_CONDITIONAL);
1650 break;
1651 }
1652 case PM_PARENTHESES_NODE: {
1654
1655 if ((cast->body != NULL) && PM_NODE_TYPE_P(cast->body, PM_STATEMENTS_NODE)) {
1656 pm_statements_node_t *statements = (pm_statements_node_t *) cast->body;
1657 if (statements->body.size == 1) pm_conditional_predicate(parser, statements->body.nodes[0], type);
1658 }
1659
1660 break;
1661 }
1662 case PM_BEGIN_NODE: {
1663 pm_begin_node_t *cast = (pm_begin_node_t *) node;
1664 if (cast->statements != NULL) {
1665 pm_statements_node_t *statements = cast->statements;
1666 if (statements->body.size == 1) pm_conditional_predicate(parser, statements->body.nodes[0], type);
1667 }
1668 break;
1669 }
1670 case PM_RANGE_NODE: {
1671 pm_range_node_t *cast = (pm_range_node_t *) node;
1672
1673 if (cast->left != NULL) pm_conditional_predicate(parser, cast->left, PM_CONDITIONAL_PREDICATE_TYPE_FLIP_FLOP);
1674 if (cast->right != NULL) pm_conditional_predicate(parser, cast->right, PM_CONDITIONAL_PREDICATE_TYPE_FLIP_FLOP);
1675
1676 // Here we change the range node into a flip flop node. We can do
1677 // this since the nodes are exactly the same except for the type.
1678 // We're only asserting against the size when we should probably
1679 // assert against the entire layout, but we'll assume tests will
1680 // catch this.
1681 assert(sizeof(pm_range_node_t) == sizeof(pm_flip_flop_node_t));
1682 node->type = PM_FLIP_FLOP_NODE;
1683
1684 break;
1685 }
1686 case PM_REGULAR_EXPRESSION_NODE:
1687 // Here we change the regular expression node into a match last line
1688 // node. We can do this since the nodes are exactly the same except
1689 // for the type.
1691 node->type = PM_MATCH_LAST_LINE_NODE;
1692
1693 if (!PM_PARSER_COMMAND_LINE_OPTION_E(parser)) {
1694 pm_parser_warn_conditional_predicate_literal(parser, node, type, PM_WARN_LITERAL_IN_CONDITION_DEFAULT, "regex ");
1695 }
1696
1697 break;
1698 case PM_INTERPOLATED_REGULAR_EXPRESSION_NODE:
1699 // Here we change the interpolated regular expression node into an
1700 // interpolated match last line node. We can do this since the nodes
1701 // are exactly the same except for the type.
1703 node->type = PM_INTERPOLATED_MATCH_LAST_LINE_NODE;
1704
1705 if (!PM_PARSER_COMMAND_LINE_OPTION_E(parser)) {
1706 pm_parser_warn_conditional_predicate_literal(parser, node, type, PM_WARN_LITERAL_IN_CONDITION_VERBOSE, "regex ");
1707 }
1708
1709 break;
1710 case PM_INTEGER_NODE:
1711 if (type == PM_CONDITIONAL_PREDICATE_TYPE_FLIP_FLOP) {
1712 if (!PM_PARSER_COMMAND_LINE_OPTION_E(parser)) {
1713 pm_parser_warn_node(parser, node, PM_WARN_INTEGER_IN_FLIP_FLOP);
1714 }
1715 } else {
1716 pm_parser_warn_conditional_predicate_literal(parser, node, type, PM_WARN_LITERAL_IN_CONDITION_VERBOSE, "");
1717 }
1718 break;
1719 case PM_STRING_NODE:
1720 case PM_SOURCE_FILE_NODE:
1721 case PM_INTERPOLATED_STRING_NODE:
1722 pm_parser_warn_conditional_predicate_literal(parser, node, type, PM_WARN_LITERAL_IN_CONDITION_DEFAULT, "string ");
1723 break;
1724 case PM_SYMBOL_NODE:
1725 case PM_INTERPOLATED_SYMBOL_NODE:
1726 pm_parser_warn_conditional_predicate_literal(parser, node, type, PM_WARN_LITERAL_IN_CONDITION_VERBOSE, "symbol ");
1727 break;
1728 case PM_SOURCE_LINE_NODE:
1729 case PM_SOURCE_ENCODING_NODE:
1730 case PM_FLOAT_NODE:
1731 case PM_RATIONAL_NODE:
1732 case PM_IMAGINARY_NODE:
1733 pm_parser_warn_conditional_predicate_literal(parser, node, type, PM_WARN_LITERAL_IN_CONDITION_VERBOSE, "");
1734 break;
1735 case PM_CLASS_VARIABLE_WRITE_NODE:
1736 pm_conditional_predicate_warn_write_literal(parser, ((pm_class_variable_write_node_t *) node)->value);
1737 break;
1738 case PM_CONSTANT_WRITE_NODE:
1739 pm_conditional_predicate_warn_write_literal(parser, ((pm_constant_write_node_t *) node)->value);
1740 break;
1741 case PM_GLOBAL_VARIABLE_WRITE_NODE:
1742 pm_conditional_predicate_warn_write_literal(parser, ((pm_global_variable_write_node_t *) node)->value);
1743 break;
1744 case PM_INSTANCE_VARIABLE_WRITE_NODE:
1745 pm_conditional_predicate_warn_write_literal(parser, ((pm_instance_variable_write_node_t *) node)->value);
1746 break;
1747 case PM_LOCAL_VARIABLE_WRITE_NODE:
1748 pm_conditional_predicate_warn_write_literal(parser, ((pm_local_variable_write_node_t *) node)->value);
1749 break;
1750 case PM_MULTI_WRITE_NODE:
1751 pm_conditional_predicate_warn_write_literal(parser, ((pm_multi_write_node_t *) node)->value);
1752 break;
1753 default:
1754 break;
1755 }
1756}
1757
1780
1784static PRISM_INLINE const pm_location_t *
1785pm_arguments_end(pm_arguments_t *arguments) {
1786 if (arguments->block != NULL) {
1787 uint32_t end = PM_NODE_END(arguments->block);
1788
1789 if (arguments->closing_loc.length > 0) {
1790 uint32_t arguments_end = PM_LOCATION_END(&arguments->closing_loc);
1791 if (arguments_end > end) {
1792 return &arguments->closing_loc;
1793 }
1794 }
1795 return &arguments->block->location;
1796 }
1797 if (arguments->closing_loc.length > 0) {
1798 return &arguments->closing_loc;
1799 }
1800 if (arguments->arguments != NULL) {
1801 return &arguments->arguments->base.location;
1802 }
1803 if (arguments->opening_loc.length > 0) {
1804 return &arguments->opening_loc;
1805 }
1806 return NULL;
1807}
1808
1813static void
1814pm_arguments_validate_block(pm_parser_t *parser, pm_arguments_t *arguments, pm_block_node_t *block) {
1815 // First, check that we have arguments and that we don't have a closing
1816 // location for them.
1817 if (arguments->arguments == NULL || arguments->closing_loc.length > 0) {
1818 return;
1819 }
1820
1821 // Next, check that we don't have a single parentheses argument. This would
1822 // look like:
1823 //
1824 // foo (1) {}
1825 //
1826 // In this case, it's actually okay for the block to be attached to the
1827 // call, even though it looks like it's attached to the argument.
1828 if (arguments->arguments->arguments.size == 1 && PM_NODE_TYPE_P(arguments->arguments->arguments.nodes[0], PM_PARENTHESES_NODE)) {
1829 return;
1830 }
1831
1832 // If we didn't hit a case before this check, then at this point we need to
1833 // add a syntax error.
1834 pm_parser_err_node(parser, UP(block), PM_ERR_ARGUMENT_UNEXPECTED_BLOCK);
1835}
1836
1837/******************************************************************************/
1838/* Basic character checks */
1839/******************************************************************************/
1840
1847static PRISM_INLINE size_t
1848char_is_identifier_start(const pm_parser_t *parser, const uint8_t *b, ptrdiff_t n) {
1849 if (n <= 0) return 0;
1850
1851 if (parser->encoding_changed) {
1852 size_t width;
1853
1854 if ((width = parser->encoding->alpha_char(b, n)) != 0) {
1855 return width;
1856 } else if (*b == '_') {
1857 return 1;
1858 } else if (*b >= 0x80) {
1859 return parser->encoding->char_width(b, n);
1860 } else {
1861 return 0;
1862 }
1863 } else if (*b < 0x80) {
1864 return (pm_encoding_unicode_table[*b] & PRISM_ENCODING_ALPHABETIC_BIT ? 1 : 0) || (*b == '_');
1865 } else {
1866 return pm_encoding_utf_8_char_width(b, n);
1867 }
1868}
1869
1874static PRISM_INLINE size_t
1875char_is_identifier_utf8(const uint8_t *b, ptrdiff_t n) {
1876 if (n <= 0) {
1877 return 0;
1878 } else if (*b < 0x80) {
1879 return (*b == '_') || (pm_encoding_unicode_table[*b] & PRISM_ENCODING_ALPHANUMERIC_BIT ? 1 : 0);
1880 } else {
1881 return pm_encoding_utf_8_char_width(b, n);
1882 }
1883}
1884
1898#if defined(PRISM_HAS_NEON)
1899#include <arm_neon.h>
1900
1901static PRISM_INLINE size_t
1902scan_identifier_ascii(const uint8_t *start, const uint8_t *end) {
1903 const uint8_t *cursor = start;
1904
1905 // Nibble-based lookup tables for classifying [a-zA-Z0-9_].
1906 // Each high nibble is assigned a unique bit; the low nibble table
1907 // contains the OR of bits for all high nibbles that have an
1908 // identifier character at that low nibble position. A byte is an
1909 // identifier character iff (low_lut[lo] & high_lut[hi]) != 0.
1910 static const uint8_t low_lut_data[16] = {
1911 0x15, 0x1F, 0x1F, 0x1F, 0x1F, 0x1F, 0x1F, 0x1F,
1912 0x1F, 0x1F, 0x1E, 0x0A, 0x0A, 0x0A, 0x0A, 0x0E
1913 };
1914 static const uint8_t high_lut_data[16] = {
1915 0x00, 0x00, 0x00, 0x01, 0x02, 0x04, 0x08, 0x10,
1916 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00
1917 };
1918 const uint8x16_t low_lut = vld1q_u8(low_lut_data);
1919 const uint8x16_t high_lut = vld1q_u8(high_lut_data);
1920 const uint8x16_t mask_0f = vdupq_n_u8(0x0F);
1921
1922 while (cursor + 16 <= end) {
1923 uint8x16_t v = vld1q_u8(cursor);
1924
1925 uint8x16_t lo_class = vqtbl1q_u8(low_lut, vandq_u8(v, mask_0f));
1926 uint8x16_t hi_class = vqtbl1q_u8(high_lut, vshrq_n_u8(v, 4));
1927 uint8x16_t ident = vandq_u8(lo_class, hi_class);
1928
1929 // Fast check: if the per-byte minimum is nonzero, every byte matched.
1930 if (vminvq_u8(ident) != 0) {
1931 cursor += 16;
1932 continue;
1933 }
1934
1935 // Find the first non-identifier byte (zero in ident).
1936 uint8x16_t is_zero = vceqq_u8(ident, vdupq_n_u8(0));
1937 uint64_t lo = vgetq_lane_u64(vreinterpretq_u64_u8(is_zero), 0);
1938
1939 if (lo != 0) {
1940 cursor += pm_ctzll(lo) / 8;
1941 } else {
1942 uint64_t hi = vgetq_lane_u64(vreinterpretq_u64_u8(is_zero), 1);
1943 cursor += 8 + pm_ctzll(hi) / 8;
1944 }
1945
1946 return (size_t) (cursor - start);
1947 }
1948
1949 return (size_t) (cursor - start);
1950}
1951
1952#elif defined(PRISM_HAS_SSSE3)
1953#include <tmmintrin.h>
1954
1955static PRISM_INLINE size_t
1956scan_identifier_ascii(const uint8_t *start, const uint8_t *end) {
1957 const uint8_t *cursor = start;
1958
1959 while (cursor + 16 <= end) {
1960 __m128i v = _mm_loadu_si128((const __m128i *) cursor);
1961 __m128i zero = _mm_setzero_si128();
1962
1963 // Unsigned range check via saturating subtraction:
1964 // byte >= lo ⟺ saturate(lo - byte) == 0
1965 // byte <= hi ⟺ saturate(byte - hi) == 0
1966
1967 // Fold case: OR with 0x20 maps A-Z to a-z.
1968 __m128i lowered = _mm_or_si128(v, _mm_set1_epi8(0x20));
1969 __m128i letter = _mm_and_si128(
1970 _mm_cmpeq_epi8(_mm_subs_epu8(_mm_set1_epi8(0x61), lowered), zero),
1971 _mm_cmpeq_epi8(_mm_subs_epu8(lowered, _mm_set1_epi8(0x7A)), zero));
1972
1973 __m128i digit = _mm_and_si128(
1974 _mm_cmpeq_epi8(_mm_subs_epu8(_mm_set1_epi8(0x30), v), zero),
1975 _mm_cmpeq_epi8(_mm_subs_epu8(v, _mm_set1_epi8(0x39)), zero));
1976
1977 __m128i underscore = _mm_cmpeq_epi8(v, _mm_set1_epi8(0x5F));
1978
1979 __m128i ident = _mm_or_si128(_mm_or_si128(letter, digit), underscore);
1980 int mask = _mm_movemask_epi8(ident);
1981
1982 if (mask == 0xFFFF) {
1983 cursor += 16;
1984 continue;
1985 }
1986
1987 cursor += pm_ctzll((uint64_t) (~mask & 0xFFFF));
1988 return (size_t) (cursor - start);
1989 }
1990
1991 return (size_t) (cursor - start);
1992}
1993
1994// The SWAR path uses pm_ctzll to find the first non-matching byte within a
1995// word, which only yields the correct byte index on little-endian targets.
1996// We gate on a positive little-endian check so that unknown-endianness
1997// platforms safely fall through to the no-op fallback.
1998#elif defined(PRISM_HAS_SWAR)
1999
2009static PRISM_INLINE size_t
2010scan_identifier_ascii(const uint8_t *start, const uint8_t *end) {
2011 static const uint64_t ones = 0x0101010101010101ULL;
2012 static const uint64_t highs = 0x8080808080808080ULL;
2013 const uint8_t *cursor = start;
2014
2015 while (cursor + 8 <= end) {
2016 uint64_t word;
2017 memcpy(&word, cursor, 8);
2018
2019 // Bail on any non-ASCII byte.
2020 if (word & highs) break;
2021
2022 uint64_t digit = ((word | highs) - ones * 0x30) & ((ones * 0x39 | highs) - word) & highs;
2023
2024 // Fold upper- and lowercase together by forcing bit 5 (OR 0x20),
2025 // then check the lowercase range once. A-Z maps to a-z; the
2026 // only non-letter byte that could alias into [0x61,0x7A] is one
2027 // whose original value was in [0x41,0x5A] — which is exactly
2028 // the uppercase letters we want to match.
2029 uint64_t lowered = word | (ones * 0x20);
2030 uint64_t letter = ((lowered | highs) - ones * 0x61) & ((ones * 0x7A | highs) - lowered) & highs;
2031
2032 // Standard SWAR "has zero byte" idiom on (word XOR 0x5F) to find
2033 // bytes equal to underscore. Safe from cross-byte borrows because
2034 // the ASCII guard above ensures all bytes are < 0x80.
2035 uint64_t xor_us = word ^ (ones * 0x5F);
2036 uint64_t underscore = (xor_us - ones) & ~xor_us & highs;
2037
2038 uint64_t ident = digit | letter | underscore;
2039
2040 if (ident == highs) {
2041 cursor += 8;
2042 continue;
2043 }
2044
2045 // Find the first non-identifier byte. On little-endian the first
2046 // byte sits in the least-significant position.
2047 uint64_t not_ident = ~ident & highs;
2048 cursor += pm_ctzll(not_ident) / 8;
2049 return (size_t) (cursor - start);
2050 }
2051
2052 return (size_t) (cursor - start);
2053}
2054
2055#else
2056
2057// No-op fallback for big-endian or other unsupported platforms.
2058// The caller's byte-at-a-time loop handles everything.
2059#define scan_identifier_ascii(start, end) ((size_t) 0)
2060
2061#endif
2062
2068static PRISM_INLINE size_t
2069char_is_identifier(const pm_parser_t *parser, const uint8_t *b, ptrdiff_t n) {
2070 if (n <= 0) {
2071 return 0;
2072 } else if (parser->encoding_changed) {
2073 size_t width;
2074
2075 if ((width = parser->encoding->alnum_char(b, n)) != 0) {
2076 return width;
2077 } else if (*b == '_') {
2078 return 1;
2079 } else if (*b >= 0x80) {
2080 return parser->encoding->char_width(b, n);
2081 } else {
2082 return 0;
2083 }
2084 } else {
2085 return char_is_identifier_utf8(b, n);
2086 }
2087}
2088
2089// Here we're defining a perfect hash for the characters that are allowed in
2090// global names. This is used to quickly check the next character after a $ to
2091// see if it's a valid character for a global name.
2092#define BIT(c, idx) (((c) / 32 - 1 == idx) ? (1U << ((c) % 32)) : 0)
2093#define PUNCT(idx) ( \
2094 BIT('~', idx) | BIT('*', idx) | BIT('$', idx) | BIT('?', idx) | \
2095 BIT('!', idx) | BIT('@', idx) | BIT('/', idx) | BIT('\\', idx) | \
2096 BIT(';', idx) | BIT(',', idx) | BIT('.', idx) | BIT('=', idx) | \
2097 BIT(':', idx) | BIT('<', idx) | BIT('>', idx) | BIT('\"', idx) | \
2098 BIT('&', idx) | BIT('`', idx) | BIT('\'', idx) | BIT('+', idx) | \
2099 BIT('0', idx))
2100
2101const unsigned int pm_global_name_punctuation_hash[(0x7e - 0x20 + 31) / 32] = { PUNCT(0), PUNCT(1), PUNCT(2) };
2102
2103#undef BIT
2104#undef PUNCT
2105
2106static PRISM_INLINE bool
2107char_is_global_name_punctuation(const uint8_t b) {
2108 const unsigned int i = (const unsigned int) b;
2109 if (i <= 0x20 || 0x7e < i) return false;
2110
2111 return (pm_global_name_punctuation_hash[(i - 0x20) / 32] >> (i % 32)) & 1;
2112}
2113
2114static PRISM_INLINE bool
2115token_is_setter_name(pm_token_t *token) {
2116 return (
2117 (token->type == PM_TOKEN_BRACKET_LEFT_RIGHT_EQUAL) ||
2118 ((token->type == PM_TOKEN_IDENTIFIER) &&
2119 (token->end - token->start >= 2) &&
2120 (token->end[-1] == '='))
2121 );
2122}
2123
2127static bool
2128pm_local_is_keyword(const char *source, size_t length) {
2129#define KEYWORD(name) if (memcmp(source, name, length) == 0) return true
2130
2131 switch (length) {
2132 case 2:
2133 switch (source[0]) {
2134 case 'd': KEYWORD("do"); return false;
2135 case 'i': KEYWORD("if"); KEYWORD("in"); return false;
2136 case 'o': KEYWORD("or"); return false;
2137 default: return false;
2138 }
2139 case 3:
2140 switch (source[0]) {
2141 case 'a': KEYWORD("and"); return false;
2142 case 'd': KEYWORD("def"); return false;
2143 case 'e': KEYWORD("end"); return false;
2144 case 'f': KEYWORD("for"); return false;
2145 case 'n': KEYWORD("nil"); KEYWORD("not"); return false;
2146 default: return false;
2147 }
2148 case 4:
2149 switch (source[0]) {
2150 case 'c': KEYWORD("case"); return false;
2151 case 'e': KEYWORD("else"); return false;
2152 case 'n': KEYWORD("next"); return false;
2153 case 'r': KEYWORD("redo"); return false;
2154 case 's': KEYWORD("self"); return false;
2155 case 't': KEYWORD("then"); KEYWORD("true"); return false;
2156 case 'w': KEYWORD("when"); return false;
2157 default: return false;
2158 }
2159 case 5:
2160 switch (source[0]) {
2161 case 'a': KEYWORD("alias"); return false;
2162 case 'b': KEYWORD("begin"); KEYWORD("break"); return false;
2163 case 'c': KEYWORD("class"); return false;
2164 case 'e': KEYWORD("elsif"); return false;
2165 case 'f': KEYWORD("false"); return false;
2166 case 'r': KEYWORD("retry"); return false;
2167 case 's': KEYWORD("super"); return false;
2168 case 'u': KEYWORD("undef"); KEYWORD("until"); return false;
2169 case 'w': KEYWORD("while"); return false;
2170 case 'y': KEYWORD("yield"); return false;
2171 default: return false;
2172 }
2173 case 6:
2174 switch (source[0]) {
2175 case 'e': KEYWORD("ensure"); return false;
2176 case 'm': KEYWORD("module"); return false;
2177 case 'r': KEYWORD("rescue"); KEYWORD("return"); return false;
2178 case 'u': KEYWORD("unless"); return false;
2179 default: return false;
2180 }
2181 case 8:
2182 KEYWORD("__LINE__");
2183 KEYWORD("__FILE__");
2184 return false;
2185 case 12:
2186 KEYWORD("__ENCODING__");
2187 return false;
2188 default:
2189 return false;
2190 }
2191
2192#undef KEYWORD
2193}
2194
2195/******************************************************************************/
2196/* Node flag handling functions */
2197/******************************************************************************/
2198
2202static PRISM_INLINE void
2203pm_node_flag_set(pm_node_t *node, pm_node_flags_t flag) {
2204 node->flags |= flag;
2205}
2206
2210static PRISM_INLINE void
2211pm_node_flag_unset(pm_node_t *node, pm_node_flags_t flag) {
2212 node->flags &= (pm_node_flags_t) ~flag;
2213}
2214
2218static PRISM_INLINE void
2219pm_node_flag_set_repeated_parameter(pm_node_t *node) {
2220 assert(PM_NODE_TYPE(node) == PM_BLOCK_LOCAL_VARIABLE_NODE ||
2221 PM_NODE_TYPE(node) == PM_BLOCK_PARAMETER_NODE ||
2222 PM_NODE_TYPE(node) == PM_KEYWORD_REST_PARAMETER_NODE ||
2223 PM_NODE_TYPE(node) == PM_OPTIONAL_KEYWORD_PARAMETER_NODE ||
2224 PM_NODE_TYPE(node) == PM_OPTIONAL_PARAMETER_NODE ||
2225 PM_NODE_TYPE(node) == PM_REQUIRED_KEYWORD_PARAMETER_NODE ||
2226 PM_NODE_TYPE(node) == PM_REQUIRED_PARAMETER_NODE ||
2227 PM_NODE_TYPE(node) == PM_REST_PARAMETER_NODE);
2228
2229 pm_node_flag_set(node, PM_PARAMETER_FLAGS_REPEATED_PARAMETER);
2230}
2231
2232/******************************************************************************/
2233/* Node creation functions */
2234/******************************************************************************/
2235
2241#define PM_REGULAR_EXPRESSION_ENCODING_MASK ~(PM_REGULAR_EXPRESSION_FLAGS_EUC_JP | PM_REGULAR_EXPRESSION_FLAGS_ASCII_8BIT | PM_REGULAR_EXPRESSION_FLAGS_WINDOWS_31J | PM_REGULAR_EXPRESSION_FLAGS_UTF_8)
2242
2246static PRISM_INLINE pm_node_flags_t
2247pm_regular_expression_flags_create(pm_parser_t *parser, const pm_token_t *closing) {
2248 pm_node_flags_t flags = 0;
2249
2250 if (closing->type == PM_TOKEN_REGEXP_END) {
2251 pm_buffer_t unknown_flags = { 0 };
2252
2253 // The closing delimiter is normally a single byte, so the options
2254 // follow it. A `\r\n` newline delimiter is two bytes, however, so we
2255 // skip past it to avoid misreading the trailing `\n` as an option.
2256 const uint8_t *flag = closing->start + 1;
2257 if ((closing->end - closing->start) >= 2 && closing->start[0] == '\r' && closing->start[1] == '\n') {
2258 flag++;
2259 }
2260
2261 for (; flag < closing->end; flag++) {
2262 switch (*flag) {
2263 case 'i': flags |= PM_REGULAR_EXPRESSION_FLAGS_IGNORE_CASE; break;
2264 case 'm': flags |= PM_REGULAR_EXPRESSION_FLAGS_MULTI_LINE; break;
2265 case 'x': flags |= PM_REGULAR_EXPRESSION_FLAGS_EXTENDED; break;
2266 case 'o': flags |= PM_REGULAR_EXPRESSION_FLAGS_ONCE; break;
2267
2268 case 'e': flags = (pm_node_flags_t) (((pm_node_flags_t) (flags & PM_REGULAR_EXPRESSION_ENCODING_MASK)) | PM_REGULAR_EXPRESSION_FLAGS_EUC_JP); break;
2269 case 'n': flags = (pm_node_flags_t) (((pm_node_flags_t) (flags & PM_REGULAR_EXPRESSION_ENCODING_MASK)) | PM_REGULAR_EXPRESSION_FLAGS_ASCII_8BIT); break;
2270 case 's': flags = (pm_node_flags_t) (((pm_node_flags_t) (flags & PM_REGULAR_EXPRESSION_ENCODING_MASK)) | PM_REGULAR_EXPRESSION_FLAGS_WINDOWS_31J); break;
2271 case 'u': flags = (pm_node_flags_t) (((pm_node_flags_t) (flags & PM_REGULAR_EXPRESSION_ENCODING_MASK)) | PM_REGULAR_EXPRESSION_FLAGS_UTF_8); break;
2272
2273 default: pm_buffer_append_byte(&unknown_flags, *flag);
2274 }
2275 }
2276
2277 size_t unknown_flags_length = pm_buffer_length(&unknown_flags);
2278 if (unknown_flags_length != 0) {
2279 const char *word = unknown_flags_length >= 2 ? "options" : "option";
2280 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->previous, PM_ERR_REGEXP_UNKNOWN_OPTIONS, word, unknown_flags_length, pm_buffer_value(&unknown_flags));
2281 }
2282 pm_buffer_cleanup(&unknown_flags);
2283 }
2284
2285 return flags;
2286}
2287
2288#undef PM_REGULAR_EXPRESSION_ENCODING_MASK
2289
2290static pm_statements_node_t *
2291pm_statements_node_create(pm_parser_t *parser);
2292
2293static void
2294pm_statements_node_body_append(pm_parser_t *parser, pm_statements_node_t *node, pm_node_t *statement, bool newline);
2295
2296static size_t
2297pm_statements_node_body_length(pm_statements_node_t *node);
2298
2303static PRISM_INLINE void
2304pm_integer_arena_move(pm_arena_t *arena, pm_integer_t *integer) {
2305 if (integer->values != NULL) {
2306 size_t byte_size = integer->length * sizeof(uint32_t);
2307 uint32_t *old_values = integer->values;
2308 integer->values = (uint32_t *) pm_arena_memdup(arena, old_values, byte_size, PRISM_ALIGNOF(uint32_t));
2309 xfree(old_values);
2310 }
2311}
2312
2316static pm_error_recovery_node_t *
2317pm_error_recovery_node_create(pm_parser_t *parser, uint32_t start, uint32_t length) {
2318 return pm_error_recovery_node_new(
2319 parser->arena,
2320 ++parser->node_id,
2321 0,
2322 ((pm_location_t) { .start = start, .length = length }),
2323 NULL
2324 );
2325}
2326
2330static pm_error_recovery_node_t *
2331pm_error_recovery_node_create_unexpected(pm_parser_t *parser, pm_node_t *unexpected) {
2332 return pm_error_recovery_node_new(
2333 parser->arena,
2334 ++parser->node_id,
2335 0,
2336 unexpected->location,
2337 unexpected
2338 );
2339}
2340
2344static pm_alias_global_variable_node_t *
2345pm_alias_global_variable_node_create(pm_parser_t *parser, const pm_token_t *keyword, pm_node_t *new_name, pm_node_t *old_name) {
2346 assert(keyword->type == PM_TOKEN_KEYWORD_ALIAS);
2347
2348 return pm_alias_global_variable_node_new(
2349 parser->arena,
2350 ++parser->node_id,
2351 0,
2352 PM_LOCATION_INIT_TOKEN_NODE(parser, keyword, old_name),
2353 new_name,
2354 old_name,
2355 TOK2LOC(parser, keyword)
2356 );
2357}
2358
2362static pm_alias_method_node_t *
2363pm_alias_method_node_create(pm_parser_t *parser, const pm_token_t *keyword, pm_node_t *new_name, pm_node_t *old_name) {
2364 assert(keyword->type == PM_TOKEN_KEYWORD_ALIAS);
2365
2366 return pm_alias_method_node_new(
2367 parser->arena,
2368 ++parser->node_id,
2369 0,
2370 PM_LOCATION_INIT_TOKEN_NODE(parser, keyword, old_name),
2371 new_name,
2372 old_name,
2373 TOK2LOC(parser, keyword)
2374 );
2375}
2376
2380static pm_alternation_pattern_node_t *
2381pm_alternation_pattern_node_create(pm_parser_t *parser, pm_node_t *left, pm_node_t *right, const pm_token_t *operator) {
2382 return pm_alternation_pattern_node_new(
2383 parser->arena,
2384 ++parser->node_id,
2385 0,
2386 PM_LOCATION_INIT_NODES(left, right),
2387 left,
2388 right,
2389 TOK2LOC(parser, operator)
2390 );
2391}
2392
2396static pm_and_node_t *
2397pm_and_node_create(pm_parser_t *parser, pm_node_t *left, const pm_token_t *operator, pm_node_t *right) {
2398 pm_assert_value_expression(parser, left);
2399
2400 return pm_and_node_new(
2401 parser->arena,
2402 ++parser->node_id,
2403 0,
2404 PM_LOCATION_INIT_NODES(left, right),
2405 left,
2406 right,
2407 TOK2LOC(parser, operator)
2408 );
2409}
2410
2414static pm_arguments_node_t *
2415pm_arguments_node_create(pm_parser_t *parser) {
2416 return pm_arguments_node_new(
2417 parser->arena,
2418 ++parser->node_id,
2419 0,
2420 PM_LOCATION_INIT_UNSET,
2421 ((pm_node_list_t) { 0 })
2422 );
2423}
2424
2428static size_t
2429pm_arguments_node_size(pm_arguments_node_t *node) {
2430 return node->arguments.size;
2431}
2432
2436static void
2437pm_arguments_node_arguments_append(pm_arena_t *arena, pm_arguments_node_t *node, pm_node_t *argument) {
2438 if (pm_arguments_node_size(node) == 0) {
2439 PM_NODE_START_SET_NODE(node, argument);
2440 }
2441
2442 if (PM_NODE_END(node) < PM_NODE_END(argument)) {
2443 PM_NODE_LENGTH_SET_NODE(node, argument);
2444 }
2445
2446 pm_node_list_append(arena, &node->arguments, argument);
2447
2448 if (PM_NODE_TYPE_P(argument, PM_SPLAT_NODE)) {
2449 if (PM_NODE_FLAG_P(node, PM_ARGUMENTS_NODE_FLAGS_CONTAINS_SPLAT)) {
2450 pm_node_flag_set(UP(node), PM_ARGUMENTS_NODE_FLAGS_CONTAINS_MULTIPLE_SPLATS);
2451 } else {
2452 pm_node_flag_set(UP(node), PM_ARGUMENTS_NODE_FLAGS_CONTAINS_SPLAT);
2453 }
2454 }
2455}
2456
2460static pm_array_node_t *
2461pm_array_node_create(pm_parser_t *parser, const pm_token_t *opening) {
2462 if (opening == NULL) {
2463 return pm_array_node_new(
2464 parser->arena,
2465 ++parser->node_id,
2466 PM_NODE_FLAG_STATIC_LITERAL,
2467 PM_LOCATION_INIT_UNSET,
2468 ((pm_node_list_t) { 0 }),
2469 ((pm_location_t) { 0 }),
2470 ((pm_location_t) { 0 })
2471 );
2472 } else {
2473 return pm_array_node_new(
2474 parser->arena,
2475 ++parser->node_id,
2476 PM_NODE_FLAG_STATIC_LITERAL,
2477 PM_LOCATION_INIT_TOKEN(parser, opening),
2478 ((pm_node_list_t) { 0 }),
2479 TOK2LOC(parser, opening),
2480 TOK2LOC(parser, opening)
2481 );
2482 }
2483}
2484
2488static PRISM_INLINE void
2489pm_array_node_elements_append(pm_arena_t *arena, pm_array_node_t *node, pm_node_t *element) {
2490 if (!node->elements.size && !node->opening_loc.length) {
2491 PM_NODE_START_SET_NODE(node, element);
2492 }
2493
2494 pm_node_list_append(arena, &node->elements, element);
2495 PM_NODE_LENGTH_SET_NODE(node, element);
2496
2497 // If the element is not a static literal, then the array is not a static
2498 // literal. Turn that flag off.
2499 if (PM_NODE_TYPE_P(element, PM_ARRAY_NODE) || PM_NODE_TYPE_P(element, PM_HASH_NODE) || PM_NODE_TYPE_P(element, PM_RANGE_NODE) || !PM_NODE_FLAG_P(element, PM_NODE_FLAG_STATIC_LITERAL)) {
2500 pm_node_flag_unset(UP(node), PM_NODE_FLAG_STATIC_LITERAL);
2501 }
2502
2503 if (PM_NODE_TYPE_P(element, PM_SPLAT_NODE)) {
2504 pm_node_flag_set(UP(node), PM_ARRAY_NODE_FLAGS_CONTAINS_SPLAT);
2505 }
2506}
2507
2511static void
2512pm_array_node_close_set(const pm_parser_t *parser, pm_array_node_t *node, const pm_token_t *closing) {
2513 assert(closing->type == PM_TOKEN_BRACKET_RIGHT || closing->type == PM_TOKEN_STRING_END || closing->type == 0);
2514 PM_NODE_LENGTH_SET_TOKEN(parser, node, closing);
2515 node->closing_loc = TOK2LOC(parser, closing);
2516}
2517
2522static pm_array_pattern_node_t *
2523pm_array_pattern_node_node_list_create(pm_parser_t *parser, pm_node_list_t *nodes) {
2524 pm_array_pattern_node_t *node = pm_array_pattern_node_new(
2525 parser->arena,
2526 ++parser->node_id,
2527 0,
2528 PM_LOCATION_INIT_NODES(nodes->nodes[0], nodes->nodes[nodes->size - 1]),
2529 NULL,
2530 ((pm_node_list_t) { 0 }),
2531 NULL,
2532 ((pm_node_list_t) { 0 }),
2533 ((pm_location_t) { 0 }),
2534 ((pm_location_t) { 0 })
2535 );
2536
2537 // For now we're going to just copy over each pointer manually. This could be
2538 // much more efficient, as we could instead resize the node list.
2539 bool found_rest = false;
2540 pm_node_t *child;
2541
2542 PM_NODE_LIST_FOREACH(nodes, index, child) {
2543 if (!found_rest && (PM_NODE_TYPE_P(child, PM_SPLAT_NODE) || PM_NODE_TYPE_P(child, PM_IMPLICIT_REST_NODE))) {
2544 node->rest = child;
2545 found_rest = true;
2546 } else if (found_rest) {
2547 pm_node_list_append(parser->arena, &node->posts, child);
2548 } else {
2549 pm_node_list_append(parser->arena, &node->requireds, child);
2550 }
2551 }
2552
2553 return node;
2554}
2555
2559static pm_array_pattern_node_t *
2560pm_array_pattern_node_rest_create(pm_parser_t *parser, pm_node_t *rest) {
2561 return pm_array_pattern_node_new(
2562 parser->arena,
2563 ++parser->node_id,
2564 0,
2565 PM_LOCATION_INIT_NODE(rest),
2566 NULL,
2567 ((pm_node_list_t) { 0 }),
2568 rest,
2569 ((pm_node_list_t) { 0 }),
2570 ((pm_location_t) { 0 }),
2571 ((pm_location_t) { 0 })
2572 );
2573}
2574
2579static pm_array_pattern_node_t *
2580pm_array_pattern_node_constant_create(pm_parser_t *parser, pm_node_t *constant, const pm_token_t *opening, const pm_token_t *closing) {
2581 return pm_array_pattern_node_new(
2582 parser->arena,
2583 ++parser->node_id,
2584 0,
2585 PM_LOCATION_INIT_NODE_TOKEN(parser, constant, closing),
2586 constant,
2587 ((pm_node_list_t) { 0 }),
2588 NULL,
2589 ((pm_node_list_t) { 0 }),
2590 TOK2LOC(parser, opening),
2591 TOK2LOC(parser, closing)
2592 );
2593}
2594
2599static pm_array_pattern_node_t *
2600pm_array_pattern_node_empty_create(pm_parser_t *parser, const pm_token_t *opening, const pm_token_t *closing) {
2601 return pm_array_pattern_node_new(
2602 parser->arena,
2603 ++parser->node_id,
2604 0,
2605 PM_LOCATION_INIT_TOKENS(parser, opening, closing),
2606 NULL,
2607 ((pm_node_list_t) { 0 }),
2608 NULL,
2609 ((pm_node_list_t) { 0 }),
2610 TOK2LOC(parser, opening),
2611 TOK2LOC(parser, closing)
2612 );
2613}
2614
2615static PRISM_INLINE void
2616pm_array_pattern_node_requireds_append(pm_arena_t *arena, pm_array_pattern_node_t *node, pm_node_t *inner) {
2617 pm_node_list_append(arena, &node->requireds, inner);
2618}
2619
2623static pm_assoc_node_t *
2624pm_assoc_node_create(pm_parser_t *parser, pm_node_t *key, const pm_token_t *operator, pm_node_t *value) {
2625 uint32_t end;
2626
2627 if (value != NULL && PM_NODE_END(value) > PM_NODE_END(key)) {
2628 end = PM_NODE_END(value);
2629 } else if (operator != NULL) {
2630 end = PM_TOKEN_END(parser, operator);
2631 } else {
2632 end = PM_NODE_END(key);
2633 }
2634
2635 // Hash string keys will be frozen, so we can mark them as frozen here so
2636 // that the compiler picks them up and also when we check for static literal
2637 // on the keys it gets factored in.
2638 if (PM_NODE_TYPE_P(key, PM_STRING_NODE)) {
2639 key->flags |= PM_STRING_FLAGS_FROZEN | PM_NODE_FLAG_STATIC_LITERAL;
2640 }
2641
2642 // If the key and value of this assoc node are both static literals, then
2643 // we can mark this node as a static literal.
2644 pm_node_flags_t flags = 0;
2645 if (
2646 !PM_NODE_TYPE_P(key, PM_ARRAY_NODE) && !PM_NODE_TYPE_P(key, PM_HASH_NODE) && !PM_NODE_TYPE_P(key, PM_RANGE_NODE) &&
2647 value && !PM_NODE_TYPE_P(value, PM_ARRAY_NODE) && !PM_NODE_TYPE_P(value, PM_HASH_NODE) && !PM_NODE_TYPE_P(value, PM_RANGE_NODE)
2648 ) {
2649 flags = key->flags & value->flags & PM_NODE_FLAG_STATIC_LITERAL;
2650 }
2651
2652 return pm_assoc_node_new(
2653 parser->arena,
2654 ++parser->node_id,
2655 flags,
2656 ((pm_location_t) { .start = PM_NODE_START(key), .length = U32(end - PM_NODE_START(key)) }),
2657 key,
2658 value,
2659 NTOK2LOC(parser, operator)
2660 );
2661}
2662
2666static pm_assoc_splat_node_t *
2667pm_assoc_splat_node_create(pm_parser_t *parser, pm_node_t *value, const pm_token_t *operator) {
2668 assert(operator->type == PM_TOKEN_USTAR_STAR);
2669
2670 return pm_assoc_splat_node_new(
2671 parser->arena,
2672 ++parser->node_id,
2673 0,
2674 (value == NULL) ? PM_LOCATION_INIT_TOKEN(parser, operator) : PM_LOCATION_INIT_TOKEN_NODE(parser, operator, value),
2675 value,
2676 TOK2LOC(parser, operator)
2677 );
2678}
2679
2683static pm_back_reference_read_node_t *
2684pm_back_reference_read_node_create(pm_parser_t *parser, const pm_token_t *name) {
2685 assert(name->type == PM_TOKEN_BACK_REFERENCE);
2686
2687 return pm_back_reference_read_node_new(
2688 parser->arena,
2689 ++parser->node_id,
2690 0,
2691 PM_LOCATION_INIT_TOKEN(parser, name),
2692 pm_parser_constant_id_token(parser, name)
2693 );
2694}
2695
2699static pm_begin_node_t *
2700pm_begin_node_create(pm_parser_t *parser, const pm_token_t *begin_keyword, pm_statements_node_t *statements) {
2701 uint32_t start = begin_keyword == NULL ? 0 : PM_TOKEN_START(parser, begin_keyword);
2702 uint32_t end = statements == NULL ? (begin_keyword == NULL ? 0 : PM_TOKEN_END(parser, begin_keyword)) : PM_NODE_END(statements);
2703
2704 return pm_begin_node_new(
2705 parser->arena,
2706 ++parser->node_id,
2707 0,
2708 ((pm_location_t) { .start = start, .length = U32(end - start) }),
2709 NTOK2LOC(parser, begin_keyword),
2710 statements,
2711 NULL,
2712 NULL,
2713 NULL,
2714 ((pm_location_t) { 0 })
2715 );
2716}
2717
2721static void
2722pm_begin_node_rescue_clause_set(pm_begin_node_t *node, pm_rescue_node_t *rescue_clause) {
2723 if (node->begin_keyword_loc.length == 0) {
2724 PM_NODE_START_SET_NODE(node, rescue_clause);
2725 }
2726 PM_NODE_LENGTH_SET_NODE(node, rescue_clause);
2727 node->rescue_clause = rescue_clause;
2728}
2729
2733static void
2734pm_begin_node_else_clause_set(pm_begin_node_t *node, pm_else_node_t *else_clause) {
2735 if ((node->begin_keyword_loc.length == 0) && PM_NODE_START(node) == 0) {
2736 PM_NODE_START_SET_NODE(node, else_clause);
2737 }
2738 PM_NODE_LENGTH_SET_NODE(node, else_clause);
2739 node->else_clause = else_clause;
2740}
2741
2745static void
2746pm_begin_node_ensure_clause_set(pm_begin_node_t *node, pm_ensure_node_t *ensure_clause) {
2747 if ((node->begin_keyword_loc.length == 0) && PM_NODE_START(node) == 0) {
2748 PM_NODE_START_SET_NODE(node, ensure_clause);
2749 }
2750 PM_NODE_LENGTH_SET_NODE(node, ensure_clause);
2751 node->ensure_clause = ensure_clause;
2752}
2753
2757static void
2758pm_begin_node_end_keyword_set(const pm_parser_t *parser, pm_begin_node_t *node, const pm_token_t *end_keyword) {
2759 assert(end_keyword->type == PM_TOKEN_KEYWORD_END || end_keyword->type == 0);
2760 PM_NODE_LENGTH_SET_TOKEN(parser, node, end_keyword);
2761 node->end_keyword_loc = TOK2LOC(parser, end_keyword);
2762}
2763
2767static pm_block_argument_node_t *
2768pm_block_argument_node_create(pm_parser_t *parser, const pm_token_t *operator, pm_node_t *expression) {
2769 assert(operator->type == PM_TOKEN_UAMPERSAND);
2770
2771 return pm_block_argument_node_new(
2772 parser->arena,
2773 ++parser->node_id,
2774 0,
2775 (expression == NULL) ? PM_LOCATION_INIT_TOKEN(parser, operator) : PM_LOCATION_INIT_TOKEN_NODE(parser, operator, expression),
2776 expression,
2777 TOK2LOC(parser, operator)
2778 );
2779}
2780
2784static pm_block_node_t *
2785pm_block_node_create(pm_parser_t *parser, pm_constant_id_list_t *locals, const pm_token_t *opening, pm_node_t *parameters, pm_node_t *body, const pm_token_t *closing) {
2786 return pm_block_node_new(
2787 parser->arena,
2788 ++parser->node_id,
2789 0,
2790 PM_LOCATION_INIT_TOKENS(parser, opening, closing),
2791 *locals,
2792 parameters,
2793 body,
2794 TOK2LOC(parser, opening),
2795 TOK2LOC(parser, closing)
2796 );
2797}
2798
2802static pm_block_parameter_node_t *
2803pm_block_parameter_node_create(pm_parser_t *parser, const pm_token_t *name, const pm_token_t *operator) {
2804 assert(operator->type == PM_TOKEN_UAMPERSAND || operator->type == PM_TOKEN_AMPERSAND);
2805
2806 return pm_block_parameter_node_new(
2807 parser->arena,
2808 ++parser->node_id,
2809 0,
2810 (name == NULL) ? PM_LOCATION_INIT_TOKEN(parser, operator) : PM_LOCATION_INIT_TOKENS(parser, operator, name),
2811 name == NULL ? 0 : pm_parser_constant_id_token(parser, name),
2812 NTOK2LOC(parser, name),
2813 TOK2LOC(parser, operator)
2814 );
2815}
2816
2820static pm_block_parameters_node_t *
2821pm_block_parameters_node_create(pm_parser_t *parser, pm_parameters_node_t *parameters, const pm_token_t *opening) {
2822 uint32_t start;
2823 if (opening != NULL) {
2824 start = PM_TOKEN_START(parser, opening);
2825 } else if (parameters != NULL) {
2826 start = PM_NODE_START(parameters);
2827 } else {
2828 start = 0;
2829 }
2830
2831 uint32_t end;
2832 if (parameters != NULL) {
2833 end = PM_NODE_END(parameters);
2834 } else if (opening != NULL) {
2835 end = PM_TOKEN_END(parser, opening);
2836 } else {
2837 end = 0;
2838 }
2839
2840 return pm_block_parameters_node_new(
2841 parser->arena,
2842 ++parser->node_id,
2843 0,
2844 ((pm_location_t) { .start = start, .length = U32(end - start) }),
2845 parameters,
2846 ((pm_node_list_t) { 0 }),
2847 NTOK2LOC(parser, opening),
2848 ((pm_location_t) { 0 })
2849 );
2850}
2851
2855static void
2856pm_block_parameters_node_closing_set(const pm_parser_t *parser, pm_block_parameters_node_t *node, const pm_token_t *closing) {
2857 assert(closing->type == PM_TOKEN_PIPE || closing->type == PM_TOKEN_PARENTHESIS_RIGHT || closing->type == 0);
2858 PM_NODE_LENGTH_SET_TOKEN(parser, node, closing);
2859 node->closing_loc = TOK2LOC(parser, closing);
2860}
2861
2865static pm_block_local_variable_node_t *
2866pm_block_local_variable_node_create(pm_parser_t *parser, const pm_token_t *name) {
2867 return pm_block_local_variable_node_new(
2868 parser->arena,
2869 ++parser->node_id,
2870 0,
2871 PM_LOCATION_INIT_TOKEN(parser, name),
2872 pm_parser_constant_id_token(parser, name)
2873 );
2874}
2875
2879static void
2880pm_block_parameters_node_append_local(pm_arena_t *arena, pm_block_parameters_node_t *node, const pm_block_local_variable_node_t *local) {
2881 pm_node_list_append(arena, &node->locals, UP(local));
2882
2883 if (PM_NODE_LENGTH(node) == 0) {
2884 PM_NODE_START_SET_NODE(node, local);
2885 }
2886
2887 PM_NODE_LENGTH_SET_NODE(node, local);
2888}
2889
2893static pm_break_node_t *
2894pm_break_node_create(pm_parser_t *parser, const pm_token_t *keyword, pm_arguments_node_t *arguments) {
2895 assert(keyword->type == PM_TOKEN_KEYWORD_BREAK);
2896
2897 return pm_break_node_new(
2898 parser->arena,
2899 ++parser->node_id,
2900 0,
2901 (arguments == NULL) ? PM_LOCATION_INIT_TOKEN(parser, keyword) : PM_LOCATION_INIT_TOKEN_NODE(parser, keyword, arguments),
2902 arguments,
2903 TOK2LOC(parser, keyword)
2904 );
2905}
2906
2907// There are certain flags that we want to use internally but don't want to
2908// expose because they are not relevant beyond parsing. Therefore we'll define
2909// them here and not define them in config.yml/a header file.
2910static const pm_node_flags_t PM_WRITE_NODE_FLAGS_IMPLICIT_ARRAY = (1 << 2);
2911
2912static const pm_node_flags_t PM_CALL_NODE_FLAGS_IMPLICIT_ARRAY = ((PM_CALL_NODE_FLAGS_LAST - 1) << 1);
2913static const pm_node_flags_t PM_CALL_NODE_FLAGS_COMPARISON = ((PM_CALL_NODE_FLAGS_LAST - 1) << 2);
2914static const pm_node_flags_t PM_CALL_NODE_FLAGS_INDEX = ((PM_CALL_NODE_FLAGS_LAST - 1) << 3);
2915
2921static pm_call_node_t *
2922pm_call_node_create(pm_parser_t *parser, pm_node_flags_t flags) {
2923 return pm_call_node_new(
2924 parser->arena,
2925 ++parser->node_id,
2926 flags,
2927 PM_LOCATION_INIT_UNSET,
2928 NULL,
2929 ((pm_location_t) { 0 }),
2930 0,
2931 ((pm_location_t) { 0 }),
2932 ((pm_location_t) { 0 }),
2933 NULL,
2934 ((pm_location_t) { 0 }),
2935 ((pm_location_t) { 0 }),
2936 NULL
2937 );
2938}
2939
2944static PRISM_INLINE pm_node_flags_t
2945pm_call_node_ignore_visibility_flag(const pm_node_t *receiver) {
2946 return PM_NODE_TYPE_P(receiver, PM_SELF_NODE) ? PM_CALL_NODE_FLAGS_IGNORE_VISIBILITY : 0;
2947}
2948
2953static pm_call_node_t *
2954pm_call_node_aref_create(pm_parser_t *parser, pm_node_t *receiver, pm_arguments_t *arguments) {
2955 pm_assert_value_expression(parser, receiver);
2956
2957 pm_node_flags_t flags = pm_call_node_ignore_visibility_flag(receiver);
2958 if (arguments->block == NULL || PM_NODE_TYPE_P(arguments->block, PM_BLOCK_ARGUMENT_NODE)) {
2959 flags |= PM_CALL_NODE_FLAGS_INDEX;
2960 }
2961
2962 pm_call_node_t *node = pm_call_node_create(parser, flags);
2963
2964 PM_NODE_START_SET_NODE(node, receiver);
2965
2966 const pm_location_t *end = pm_arguments_end(arguments);
2967 assert(end != NULL && "unreachable");
2968 PM_NODE_LENGTH_SET_LOCATION(node, end);
2969
2970 node->receiver = receiver;
2971 node->message_loc.start = arguments->opening_loc.start;
2972 node->message_loc.length = (arguments->closing_loc.start + arguments->closing_loc.length) - arguments->opening_loc.start;
2973
2974 node->opening_loc = arguments->opening_loc;
2975 node->arguments = arguments->arguments;
2976 node->closing_loc = arguments->closing_loc;
2977 node->block = arguments->block;
2978
2979 node->name = pm_parser_constant_id_constant(parser, "[]", 2);
2980 return node;
2981}
2982
2986static pm_call_node_t *
2987pm_call_node_binary_create(pm_parser_t *parser, pm_node_t *receiver, pm_token_t *operator, pm_node_t *argument, pm_node_flags_t flags) {
2988 pm_assert_value_expression(parser, receiver);
2989 pm_assert_value_expression(parser, argument);
2990
2991 pm_call_node_t *node = pm_call_node_create(parser, pm_call_node_ignore_visibility_flag(receiver) | flags);
2992
2993 PM_NODE_START_SET_NODE(node, PM_NODE_START(receiver) < PM_NODE_START(argument) ? receiver : argument);
2994 PM_NODE_LENGTH_SET_NODE(node, PM_NODE_END(receiver) > PM_NODE_END(argument) ? receiver : argument);
2995
2996 node->receiver = receiver;
2997 node->message_loc = TOK2LOC(parser, operator);
2998
2999 pm_arguments_node_t *arguments = pm_arguments_node_create(parser);
3000 pm_arguments_node_arguments_append(parser->arena, arguments, argument);
3001 node->arguments = arguments;
3002
3003 node->name = pm_parser_constant_id_token(parser, operator);
3004 return node;
3005}
3006
3007static const uint8_t * parse_operator_symbol_name(const pm_token_t *);
3008
3012static pm_call_node_t *
3013pm_call_node_call_create(pm_parser_t *parser, pm_node_t *receiver, pm_token_t *operator, pm_token_t *message, pm_arguments_t *arguments) {
3014 pm_assert_value_expression(parser, receiver);
3015
3016 pm_call_node_t *node = pm_call_node_create(parser, pm_call_node_ignore_visibility_flag(receiver));
3017
3018 PM_NODE_START_SET_NODE(node, receiver);
3019 const pm_location_t *end = pm_arguments_end(arguments);
3020 if (end == NULL) {
3021 PM_NODE_LENGTH_SET_TOKEN(parser, node, message);
3022 } else {
3023 PM_NODE_LENGTH_SET_LOCATION(node, end);
3024 }
3025
3026 node->receiver = receiver;
3027 node->call_operator_loc = TOK2LOC(parser, operator);
3028 node->message_loc = TOK2LOC(parser, message);
3029 node->opening_loc = arguments->opening_loc;
3030 node->arguments = arguments->arguments;
3031 node->closing_loc = arguments->closing_loc;
3032 node->block = arguments->block;
3033
3034 if (operator->type == PM_TOKEN_AMPERSAND_DOT) {
3035 pm_node_flag_set(UP(node), PM_CALL_NODE_FLAGS_SAFE_NAVIGATION);
3036 }
3037
3042 node->name = pm_parser_constant_id_raw(parser, message->start, parse_operator_symbol_name(message));
3043 return node;
3044}
3045
3049static pm_call_node_t *
3050pm_call_node_call_synthesized_create(pm_parser_t *parser, pm_node_t *receiver, const char *message, pm_arguments_node_t *arguments) {
3051 pm_call_node_t *node = pm_call_node_create(parser, 0);
3052 node->base.location = (pm_location_t) { .start = 0, .length = U32(parser->end - parser->start) };
3053
3054 node->receiver = receiver;
3055 node->arguments = arguments;
3056
3057 node->name = pm_parser_constant_id_constant(parser, message, strlen(message));
3058 return node;
3059}
3060
3065static pm_call_node_t *
3066pm_call_node_fcall_create(pm_parser_t *parser, pm_token_t *message, pm_arguments_t *arguments) {
3067 pm_call_node_t *node = pm_call_node_create(parser, PM_CALL_NODE_FLAGS_IGNORE_VISIBILITY);
3068
3069 PM_NODE_START_SET_TOKEN(parser, node, message);
3070 const pm_location_t *end = pm_arguments_end(arguments);
3071 assert(end != NULL && "unreachable");
3072 PM_NODE_LENGTH_SET_LOCATION(node, end);
3073
3074 node->message_loc = TOK2LOC(parser, message);
3075 node->opening_loc = arguments->opening_loc;
3076 node->arguments = arguments->arguments;
3077 node->closing_loc = arguments->closing_loc;
3078 node->block = arguments->block;
3079
3080 node->name = pm_parser_constant_id_token(parser, message);
3081 return node;
3082}
3083
3088static pm_call_node_t *
3089pm_call_node_fcall_synthesized_create(pm_parser_t *parser, pm_arguments_node_t *arguments, pm_constant_id_t name) {
3090 pm_call_node_t *node = pm_call_node_create(parser, PM_CALL_NODE_FLAGS_IGNORE_VISIBILITY);
3091
3092 node->base.location = (pm_location_t) { 0 };
3093 node->arguments = arguments;
3094
3095 node->name = name;
3096 return node;
3097}
3098
3102static pm_call_node_t *
3103pm_call_node_not_create(pm_parser_t *parser, pm_node_t *receiver, pm_token_t *message, pm_arguments_t *arguments) {
3104 pm_assert_value_expression(parser, receiver);
3105 if (receiver != NULL) pm_conditional_predicate(parser, receiver, PM_CONDITIONAL_PREDICATE_TYPE_NOT);
3106
3107 pm_call_node_t *node = pm_call_node_create(parser, receiver == NULL ? 0 : pm_call_node_ignore_visibility_flag(receiver));
3108
3109 PM_NODE_START_SET_TOKEN(parser, node, message);
3110 if (arguments->closing_loc.length > 0) {
3111 PM_NODE_LENGTH_SET_LOCATION(node, &arguments->closing_loc);
3112 } else {
3113 assert(receiver != NULL);
3114 PM_NODE_LENGTH_SET_NODE(node, receiver);
3115 }
3116
3117 node->receiver = receiver;
3118 node->message_loc = TOK2LOC(parser, message);
3119 node->opening_loc = arguments->opening_loc;
3120 node->arguments = arguments->arguments;
3121 node->closing_loc = arguments->closing_loc;
3122
3123 node->name = pm_parser_constant_id_constant(parser, "!", 1);
3124 return node;
3125}
3126
3130static pm_call_node_t *
3131pm_call_node_shorthand_create(pm_parser_t *parser, pm_node_t *receiver, pm_token_t *operator, pm_arguments_t *arguments) {
3132 pm_assert_value_expression(parser, receiver);
3133
3134 pm_call_node_t *node = pm_call_node_create(parser, pm_call_node_ignore_visibility_flag(receiver));
3135
3136 PM_NODE_START_SET_NODE(node, receiver);
3137 const pm_location_t *end = pm_arguments_end(arguments);
3138 assert(end != NULL && "unreachable");
3139 PM_NODE_LENGTH_SET_LOCATION(node, end);
3140
3141 node->receiver = receiver;
3142 node->call_operator_loc = TOK2LOC(parser, operator);
3143 node->opening_loc = arguments->opening_loc;
3144 node->arguments = arguments->arguments;
3145 node->closing_loc = arguments->closing_loc;
3146 node->block = arguments->block;
3147
3148 if (operator->type == PM_TOKEN_AMPERSAND_DOT) {
3149 pm_node_flag_set(UP(node), PM_CALL_NODE_FLAGS_SAFE_NAVIGATION);
3150 }
3151
3152 node->name = pm_parser_constant_id_constant(parser, "call", 4);
3153 return node;
3154}
3155
3159static pm_call_node_t *
3160pm_call_node_unary_create(pm_parser_t *parser, pm_token_t *operator, pm_node_t *receiver, const char *name) {
3161 pm_assert_value_expression(parser, receiver);
3162
3163 pm_call_node_t *node = pm_call_node_create(parser, pm_call_node_ignore_visibility_flag(receiver));
3164
3165 PM_NODE_START_SET_TOKEN(parser, node, operator);
3166 PM_NODE_LENGTH_SET_NODE(node, receiver);
3167
3168 node->receiver = receiver;
3169 node->message_loc = TOK2LOC(parser, operator);
3170
3171 node->name = pm_parser_constant_id_constant(parser, name, strlen(name));
3172 return node;
3173}
3174
3179static pm_call_node_t *
3180pm_call_node_variable_call_create(pm_parser_t *parser, pm_token_t *message) {
3181 pm_call_node_t *node = pm_call_node_create(parser, PM_CALL_NODE_FLAGS_IGNORE_VISIBILITY);
3182
3183 node->base.location = TOK2LOC(parser, message);
3184 node->message_loc = TOK2LOC(parser, message);
3185
3186 node->name = pm_parser_constant_id_token(parser, message);
3187 return node;
3188}
3189
3194static PRISM_INLINE bool
3195pm_call_node_writable_p(const pm_parser_t *parser, const pm_call_node_t *node) {
3196 return (
3197 (node->message_loc.length > 0) &&
3198 (parser->start[node->message_loc.start + node->message_loc.length - 1] != '!') &&
3199 (parser->start[node->message_loc.start + node->message_loc.length - 1] != '?') &&
3200 char_is_identifier_start(parser, parser->start + node->message_loc.start, (ptrdiff_t) node->message_loc.length) &&
3201 (node->opening_loc.length == 0) &&
3202 (node->arguments == NULL) &&
3203 (node->block == NULL)
3204 );
3205}
3206
3210static void
3211pm_call_write_read_name_init(pm_parser_t *parser, pm_constant_id_t *read_name, pm_constant_id_t *write_name) {
3212 pm_constant_t *write_constant = pm_constant_pool_id_to_constant(&parser->constant_pool, *write_name);
3213
3214 if (write_constant->length > 0) {
3215 size_t length = write_constant->length - 1;
3216
3217 uint8_t *memory = (uint8_t *) pm_arena_alloc(parser->arena, length, 1);
3218 memcpy(memory, write_constant->start, length);
3219
3220 *read_name = pm_constant_pool_insert_owned(&parser->metadata_arena, &parser->constant_pool, memory, length);
3221 } else {
3222 // We can get here if the message was missing because of a syntax error.
3223 *read_name = pm_parser_constant_id_constant(parser, "", 0);
3224 }
3225}
3226
3230static pm_call_and_write_node_t *
3231pm_call_and_write_node_create(pm_parser_t *parser, pm_call_node_t *target, const pm_token_t *operator, pm_node_t *value) {
3232 assert(target->block == NULL);
3233 assert(operator->type == PM_TOKEN_AMPERSAND_AMPERSAND_EQUAL);
3234
3235 pm_call_and_write_node_t *node = pm_call_and_write_node_new(
3236 parser->arena,
3237 ++parser->node_id,
3238 FL(target),
3239 PM_LOCATION_INIT_NODES(target, value),
3240 target->receiver,
3241 target->call_operator_loc,
3242 target->message_loc,
3243 0,
3244 target->name,
3245 TOK2LOC(parser, operator),
3246 value
3247 );
3248
3249 pm_call_write_read_name_init(parser, &node->read_name, &node->write_name);
3250
3251 // The target is no longer necessary because we've reused its children.
3252 // It is arena-allocated so no explicit free is needed.
3253
3254 return node;
3255}
3256
3261static void
3262pm_index_arguments_check(pm_parser_t *parser, const pm_arguments_node_t *arguments, const pm_node_t *block) {
3263 if (parser->version >= PM_OPTIONS_VERSION_CRUBY_3_4) {
3264 if (arguments != NULL && PM_NODE_FLAG_P(arguments, PM_ARGUMENTS_NODE_FLAGS_CONTAINS_KEYWORDS)) {
3265 pm_node_t *node;
3266 PM_NODE_LIST_FOREACH(&arguments->arguments, index, node) {
3267 if (PM_NODE_TYPE_P(node, PM_KEYWORD_HASH_NODE)) {
3268 pm_parser_err_node(parser, node, PM_ERR_UNEXPECTED_INDEX_KEYWORDS);
3269 break;
3270 }
3271 }
3272 }
3273
3274 if (block != NULL) {
3275 pm_parser_err_node(parser, block, PM_ERR_UNEXPECTED_INDEX_BLOCK);
3276 }
3277 }
3278}
3279
3283static pm_index_and_write_node_t *
3284pm_index_and_write_node_create(pm_parser_t *parser, pm_call_node_t *target, const pm_token_t *operator, pm_node_t *value) {
3285 assert(operator->type == PM_TOKEN_AMPERSAND_AMPERSAND_EQUAL);
3286
3287 pm_index_arguments_check(parser, target->arguments, target->block);
3288
3289 assert(!target->block || PM_NODE_TYPE_P(target->block, PM_BLOCK_ARGUMENT_NODE));
3290
3291 pm_index_and_write_node_t *node = pm_index_and_write_node_new(
3292 parser->arena,
3293 ++parser->node_id,
3294 FL(target),
3295 PM_LOCATION_INIT_NODES(target, value),
3296 target->receiver,
3297 target->call_operator_loc,
3298 target->opening_loc,
3299 target->arguments,
3300 target->closing_loc,
3301 (pm_block_argument_node_t *) target->block,
3302 TOK2LOC(parser, operator),
3303 value
3304 );
3305
3306 // The target is no longer necessary because we've reused its children.
3307 // It is arena-allocated so no explicit free is needed.
3308
3309 return node;
3310}
3311
3315static pm_call_operator_write_node_t *
3316pm_call_operator_write_node_create(pm_parser_t *parser, pm_call_node_t *target, const pm_token_t *operator, pm_node_t *value) {
3317 assert(target->block == NULL);
3318
3319 pm_call_operator_write_node_t *node = pm_call_operator_write_node_new(
3320 parser->arena,
3321 ++parser->node_id,
3322 FL(target),
3323 PM_LOCATION_INIT_NODES(target, value),
3324 target->receiver,
3325 target->call_operator_loc,
3326 target->message_loc,
3327 0,
3328 target->name,
3329 pm_parser_constant_id_raw(parser, operator->start, operator->end - 1),
3330 TOK2LOC(parser, operator),
3331 value
3332 );
3333
3334 pm_call_write_read_name_init(parser, &node->read_name, &node->write_name);
3335
3336 // The target is no longer necessary because we've reused its children.
3337 // It is arena-allocated so no explicit free is needed.
3338
3339 return node;
3340}
3341
3345static pm_index_operator_write_node_t *
3346pm_index_operator_write_node_create(pm_parser_t *parser, pm_call_node_t *target, const pm_token_t *operator, pm_node_t *value) {
3347 pm_index_arguments_check(parser, target->arguments, target->block);
3348
3349 assert(!target->block || PM_NODE_TYPE_P(target->block, PM_BLOCK_ARGUMENT_NODE));
3350
3351 pm_index_operator_write_node_t *node = pm_index_operator_write_node_new(
3352 parser->arena,
3353 ++parser->node_id,
3354 FL(target),
3355 PM_LOCATION_INIT_NODES(target, value),
3356 target->receiver,
3357 target->call_operator_loc,
3358 target->opening_loc,
3359 target->arguments,
3360 target->closing_loc,
3361 (pm_block_argument_node_t *) target->block,
3362 pm_parser_constant_id_raw(parser, operator->start, operator->end - 1),
3363 TOK2LOC(parser, operator),
3364 value
3365 );
3366
3367 // The target is no longer necessary because we've reused its children.
3368 // It is arena-allocated so no explicit free is needed.
3369
3370 return node;
3371}
3372
3376static pm_call_or_write_node_t *
3377pm_call_or_write_node_create(pm_parser_t *parser, pm_call_node_t *target, const pm_token_t *operator, pm_node_t *value) {
3378 assert(target->block == NULL);
3379 assert(operator->type == PM_TOKEN_PIPE_PIPE_EQUAL);
3380
3381 pm_call_or_write_node_t *node = pm_call_or_write_node_new(
3382 parser->arena,
3383 ++parser->node_id,
3384 FL(target),
3385 PM_LOCATION_INIT_NODES(target, value),
3386 target->receiver,
3387 target->call_operator_loc,
3388 target->message_loc,
3389 0,
3390 target->name,
3391 TOK2LOC(parser, operator),
3392 value
3393 );
3394
3395 pm_call_write_read_name_init(parser, &node->read_name, &node->write_name);
3396
3397 // The target is no longer necessary because we've reused its children.
3398 // It is arena-allocated so no explicit free is needed.
3399
3400 return node;
3401}
3402
3406static pm_index_or_write_node_t *
3407pm_index_or_write_node_create(pm_parser_t *parser, pm_call_node_t *target, const pm_token_t *operator, pm_node_t *value) {
3408 assert(operator->type == PM_TOKEN_PIPE_PIPE_EQUAL);
3409
3410 pm_index_arguments_check(parser, target->arguments, target->block);
3411
3412 assert(!target->block || PM_NODE_TYPE_P(target->block, PM_BLOCK_ARGUMENT_NODE));
3413
3414 pm_index_or_write_node_t *node = pm_index_or_write_node_new(
3415 parser->arena,
3416 ++parser->node_id,
3417 FL(target),
3418 PM_LOCATION_INIT_NODES(target, value),
3419 target->receiver,
3420 target->call_operator_loc,
3421 target->opening_loc,
3422 target->arguments,
3423 target->closing_loc,
3424 (pm_block_argument_node_t *) target->block,
3425 TOK2LOC(parser, operator),
3426 value
3427 );
3428
3429 // The target is no longer necessary because we've reused its children.
3430 // It is arena-allocated so no explicit free is needed.
3431
3432 return node;
3433}
3434
3439static pm_call_target_node_t *
3440pm_call_target_node_create(pm_parser_t *parser, pm_call_node_t *target) {
3441 pm_call_target_node_t *node = pm_call_target_node_new(
3442 parser->arena,
3443 ++parser->node_id,
3444 FL(target),
3445 PM_LOCATION_INIT_NODE(target),
3446 target->receiver,
3447 target->call_operator_loc,
3448 target->name,
3449 target->message_loc
3450 );
3451
3452 /* It is possible to get here where we have parsed an invalid syntax tree
3453 * where the call operator was not present. In that case we will have a
3454 * problem because it is a required location. In this case we need to fill
3455 * it in with a fake location so that the syntax tree remains valid. */
3456 if (node->call_operator_loc.length == 0) {
3457 node->call_operator_loc = target->base.location;
3458 }
3459
3460 // The target is no longer necessary because we've reused its children.
3461 // It is arena-allocated so no explicit free is needed.
3462
3463 return node;
3464}
3465
3470static pm_index_target_node_t *
3471pm_index_target_node_create(pm_parser_t *parser, pm_call_node_t *target) {
3472 pm_index_arguments_check(parser, target->arguments, target->block);
3473 assert(!target->block || PM_NODE_TYPE_P(target->block, PM_BLOCK_ARGUMENT_NODE));
3474
3475 pm_index_target_node_t *node = pm_index_target_node_new(
3476 parser->arena,
3477 ++parser->node_id,
3478 FL(target) | PM_CALL_NODE_FLAGS_ATTRIBUTE_WRITE,
3479 PM_LOCATION_INIT_NODE(target),
3480 target->receiver,
3481 target->opening_loc,
3482 target->arguments,
3483 target->closing_loc,
3484 (pm_block_argument_node_t *) target->block
3485 );
3486
3487 // The target is no longer necessary because we've reused its children.
3488 // It is arena-allocated so no explicit free is needed.
3489
3490 return node;
3491}
3492
3496static pm_capture_pattern_node_t *
3497pm_capture_pattern_node_create(pm_parser_t *parser, pm_node_t *value, pm_local_variable_target_node_t *target, const pm_token_t *operator) {
3498 return pm_capture_pattern_node_new(
3499 parser->arena,
3500 ++parser->node_id,
3501 0,
3502 PM_LOCATION_INIT_NODES(value, target),
3503 value,
3504 target,
3505 TOK2LOC(parser, operator)
3506 );
3507}
3508
3512static pm_case_node_t *
3513pm_case_node_create(pm_parser_t *parser, const pm_token_t *case_keyword, pm_node_t *predicate, const pm_token_t *end_keyword) {
3514 return pm_case_node_new(
3515 parser->arena,
3516 ++parser->node_id,
3517 0,
3518 PM_LOCATION_INIT_TOKENS(parser, case_keyword, end_keyword == NULL ? case_keyword : end_keyword),
3519 predicate,
3520 ((pm_node_list_t) { 0 }),
3521 NULL,
3522 TOK2LOC(parser, case_keyword),
3523 NTOK2LOC(parser, end_keyword)
3524 );
3525}
3526
3530static void
3531pm_case_node_condition_append(pm_arena_t *arena, pm_case_node_t *node, pm_node_t *condition) {
3532 assert(PM_NODE_TYPE_P(condition, PM_WHEN_NODE));
3533
3534 pm_node_list_append(arena, &node->conditions, condition);
3535 PM_NODE_LENGTH_SET_NODE(node, condition);
3536}
3537
3541static void
3542pm_case_node_else_clause_set(pm_case_node_t *node, pm_else_node_t *else_clause) {
3543 node->else_clause = else_clause;
3544 PM_NODE_LENGTH_SET_NODE(node, else_clause);
3545}
3546
3550static void
3551pm_case_node_end_keyword_loc_set(const pm_parser_t *parser, pm_case_node_t *node, const pm_token_t *end_keyword) {
3552 PM_NODE_LENGTH_SET_TOKEN(parser, node, end_keyword);
3553 node->end_keyword_loc = TOK2LOC(parser, end_keyword);
3554}
3555
3559static pm_case_match_node_t *
3560pm_case_match_node_create(pm_parser_t *parser, const pm_token_t *case_keyword, pm_node_t *predicate) {
3561 return pm_case_match_node_new(
3562 parser->arena,
3563 ++parser->node_id,
3564 0,
3565 PM_LOCATION_INIT_TOKEN(parser, case_keyword),
3566 predicate,
3567 ((pm_node_list_t) { 0 }),
3568 NULL,
3569 TOK2LOC(parser, case_keyword),
3570 ((pm_location_t) { 0 })
3571 );
3572}
3573
3577static void
3578pm_case_match_node_condition_append(pm_arena_t *arena, pm_case_match_node_t *node, pm_node_t *condition) {
3579 assert(PM_NODE_TYPE_P(condition, PM_IN_NODE));
3580
3581 pm_node_list_append(arena, &node->conditions, condition);
3582 PM_NODE_LENGTH_SET_NODE(node, condition);
3583}
3584
3588static void
3589pm_case_match_node_else_clause_set(pm_case_match_node_t *node, pm_else_node_t *else_clause) {
3590 node->else_clause = else_clause;
3591 PM_NODE_LENGTH_SET_NODE(node, else_clause);
3592}
3593
3597static void
3598pm_case_match_node_end_keyword_loc_set(const pm_parser_t *parser, pm_case_match_node_t *node, const pm_token_t *end_keyword) {
3599 PM_NODE_LENGTH_SET_TOKEN(parser, node, end_keyword);
3600 node->end_keyword_loc = TOK2LOC(parser, end_keyword);
3601}
3602
3606static pm_class_node_t *
3607pm_class_node_create(pm_parser_t *parser, pm_constant_id_list_t *locals, const pm_token_t *class_keyword, pm_node_t *constant_path, const pm_token_t *name, const pm_token_t *inheritance_operator, pm_node_t *superclass, pm_node_t *body, const pm_token_t *end_keyword) {
3608 return pm_class_node_new(
3609 parser->arena,
3610 ++parser->node_id,
3611 0,
3612 PM_LOCATION_INIT_TOKENS(parser, class_keyword, end_keyword),
3613 *locals,
3614 TOK2LOC(parser, class_keyword),
3615 constant_path,
3616 NTOK2LOC(parser, inheritance_operator),
3617 superclass,
3618 body,
3619 TOK2LOC(parser, end_keyword),
3620 pm_parser_constant_id_token(parser, name)
3621 );
3622}
3623
3627static pm_class_variable_and_write_node_t *
3628pm_class_variable_and_write_node_create(pm_parser_t *parser, pm_class_variable_read_node_t *target, const pm_token_t *operator, pm_node_t *value) {
3629 assert(operator->type == PM_TOKEN_AMPERSAND_AMPERSAND_EQUAL);
3630
3631 return pm_class_variable_and_write_node_new(
3632 parser->arena,
3633 ++parser->node_id,
3634 0,
3635 PM_LOCATION_INIT_NODES(target, value),
3636 target->name,
3637 target->base.location,
3638 TOK2LOC(parser, operator),
3639 value
3640 );
3641}
3642
3646static pm_class_variable_operator_write_node_t *
3647pm_class_variable_operator_write_node_create(pm_parser_t *parser, pm_class_variable_read_node_t *target, const pm_token_t *operator, pm_node_t *value) {
3648 return pm_class_variable_operator_write_node_new(
3649 parser->arena,
3650 ++parser->node_id,
3651 0,
3652 PM_LOCATION_INIT_NODES(target, value),
3653 target->name,
3654 target->base.location,
3655 TOK2LOC(parser, operator),
3656 value,
3657 pm_parser_constant_id_raw(parser, operator->start, operator->end - 1)
3658 );
3659}
3660
3664static pm_class_variable_or_write_node_t *
3665pm_class_variable_or_write_node_create(pm_parser_t *parser, pm_class_variable_read_node_t *target, const pm_token_t *operator, pm_node_t *value) {
3666 assert(operator->type == PM_TOKEN_PIPE_PIPE_EQUAL);
3667
3668 return pm_class_variable_or_write_node_new(
3669 parser->arena,
3670 ++parser->node_id,
3671 0,
3672 PM_LOCATION_INIT_NODES(target, value),
3673 target->name,
3674 target->base.location,
3675 TOK2LOC(parser, operator),
3676 value
3677 );
3678}
3679
3683static pm_class_variable_read_node_t *
3684pm_class_variable_read_node_create(pm_parser_t *parser, const pm_token_t *token) {
3685 assert(token->type == PM_TOKEN_CLASS_VARIABLE);
3686
3687 return pm_class_variable_read_node_new(
3688 parser->arena,
3689 ++parser->node_id,
3690 0,
3691 PM_LOCATION_INIT_TOKEN(parser, token),
3692 pm_parser_constant_id_token(parser, token)
3693 );
3694}
3695
3702static PRISM_INLINE pm_node_flags_t
3703pm_implicit_array_write_flags(const pm_node_t *node, pm_node_flags_t flags) {
3704 if (PM_NODE_TYPE_P(node, PM_ARRAY_NODE) && ((const pm_array_node_t *) node)->opening_loc.length == 0) {
3705 return flags;
3706 }
3707 return 0;
3708}
3709
3713static pm_class_variable_write_node_t *
3714pm_class_variable_write_node_create(pm_parser_t *parser, pm_class_variable_read_node_t *read_node, pm_token_t *operator, pm_node_t *value) {
3715 return pm_class_variable_write_node_new(
3716 parser->arena,
3717 ++parser->node_id,
3718 pm_implicit_array_write_flags(value, PM_WRITE_NODE_FLAGS_IMPLICIT_ARRAY),
3719 PM_LOCATION_INIT_NODES(read_node, value),
3720 read_node->name,
3721 read_node->base.location,
3722 value,
3723 TOK2LOC(parser, operator)
3724 );
3725}
3726
3730static pm_constant_path_and_write_node_t *
3731pm_constant_path_and_write_node_create(pm_parser_t *parser, pm_constant_path_node_t *target, const pm_token_t *operator, pm_node_t *value) {
3732 assert(operator->type == PM_TOKEN_AMPERSAND_AMPERSAND_EQUAL);
3733
3734 return pm_constant_path_and_write_node_new(
3735 parser->arena,
3736 ++parser->node_id,
3737 0,
3738 PM_LOCATION_INIT_NODES(target, value),
3739 target,
3740 TOK2LOC(parser, operator),
3741 value
3742 );
3743}
3744
3748static pm_constant_path_operator_write_node_t *
3749pm_constant_path_operator_write_node_create(pm_parser_t *parser, pm_constant_path_node_t *target, const pm_token_t *operator, pm_node_t *value) {
3750 return pm_constant_path_operator_write_node_new(
3751 parser->arena,
3752 ++parser->node_id,
3753 0,
3754 PM_LOCATION_INIT_NODES(target, value),
3755 target,
3756 TOK2LOC(parser, operator),
3757 value,
3758 pm_parser_constant_id_raw(parser, operator->start, operator->end - 1)
3759 );
3760}
3761
3765static pm_constant_path_or_write_node_t *
3766pm_constant_path_or_write_node_create(pm_parser_t *parser, pm_constant_path_node_t *target, const pm_token_t *operator, pm_node_t *value) {
3767 assert(operator->type == PM_TOKEN_PIPE_PIPE_EQUAL);
3768
3769 return pm_constant_path_or_write_node_new(
3770 parser->arena,
3771 ++parser->node_id,
3772 0,
3773 PM_LOCATION_INIT_NODES(target, value),
3774 target,
3775 TOK2LOC(parser, operator),
3776 value
3777 );
3778}
3779
3783static pm_constant_path_node_t *
3784pm_constant_path_node_create(pm_parser_t *parser, pm_node_t *parent, const pm_token_t *delimiter, const pm_token_t *name_token) {
3785 pm_assert_value_expression(parser, parent);
3786
3787 pm_constant_id_t name = PM_CONSTANT_ID_UNSET;
3788 if (name_token->type == PM_TOKEN_CONSTANT) {
3789 name = pm_parser_constant_id_token(parser, name_token);
3790 }
3791
3792 return pm_constant_path_node_new(
3793 parser->arena,
3794 ++parser->node_id,
3795 0,
3796 (parent == NULL) ? PM_LOCATION_INIT_TOKENS(parser, delimiter, name_token) : PM_LOCATION_INIT_NODE_TOKEN(parser, parent, name_token),
3797 parent,
3798 name,
3799 TOK2LOC(parser, delimiter),
3800 TOK2LOC(parser, name_token)
3801 );
3802}
3803
3807static pm_constant_path_write_node_t *
3808pm_constant_path_write_node_create(pm_parser_t *parser, pm_constant_path_node_t *target, const pm_token_t *operator, pm_node_t *value) {
3809 return pm_constant_path_write_node_new(
3810 parser->arena,
3811 ++parser->node_id,
3812 pm_implicit_array_write_flags(value, PM_WRITE_NODE_FLAGS_IMPLICIT_ARRAY),
3813 PM_LOCATION_INIT_NODES(target, value),
3814 target,
3815 TOK2LOC(parser, operator),
3816 value
3817 );
3818}
3819
3823static pm_constant_and_write_node_t *
3824pm_constant_and_write_node_create(pm_parser_t *parser, pm_constant_read_node_t *target, const pm_token_t *operator, pm_node_t *value) {
3825 assert(operator->type == PM_TOKEN_AMPERSAND_AMPERSAND_EQUAL);
3826
3827 return pm_constant_and_write_node_new(
3828 parser->arena,
3829 ++parser->node_id,
3830 0,
3831 PM_LOCATION_INIT_NODES(target, value),
3832 target->name,
3833 target->base.location,
3834 TOK2LOC(parser, operator),
3835 value
3836 );
3837}
3838
3842static pm_constant_operator_write_node_t *
3843pm_constant_operator_write_node_create(pm_parser_t *parser, pm_constant_read_node_t *target, const pm_token_t *operator, pm_node_t *value) {
3844 return pm_constant_operator_write_node_new(
3845 parser->arena,
3846 ++parser->node_id,
3847 0,
3848 PM_LOCATION_INIT_NODES(target, value),
3849 target->name,
3850 target->base.location,
3851 TOK2LOC(parser, operator),
3852 value,
3853 pm_parser_constant_id_raw(parser, operator->start, operator->end - 1)
3854 );
3855}
3856
3860static pm_constant_or_write_node_t *
3861pm_constant_or_write_node_create(pm_parser_t *parser, pm_constant_read_node_t *target, const pm_token_t *operator, pm_node_t *value) {
3862 assert(operator->type == PM_TOKEN_PIPE_PIPE_EQUAL);
3863
3864 return pm_constant_or_write_node_new(
3865 parser->arena,
3866 ++parser->node_id,
3867 0,
3868 PM_LOCATION_INIT_NODES(target, value),
3869 target->name,
3870 target->base.location,
3871 TOK2LOC(parser, operator),
3872 value
3873 );
3874}
3875
3879static pm_constant_read_node_t *
3880pm_constant_read_node_create(pm_parser_t *parser, const pm_token_t *name) {
3881 assert(name->type == PM_TOKEN_CONSTANT || name->type == 0);
3882
3883 return pm_constant_read_node_new(
3884 parser->arena,
3885 ++parser->node_id,
3886 0,
3887 PM_LOCATION_INIT_TOKEN(parser, name),
3888 pm_parser_constant_id_token(parser, name)
3889 );
3890}
3891
3895static pm_constant_write_node_t *
3896pm_constant_write_node_create(pm_parser_t *parser, pm_constant_read_node_t *target, const pm_token_t *operator, pm_node_t *value) {
3897 return pm_constant_write_node_new(
3898 parser->arena,
3899 ++parser->node_id,
3900 pm_implicit_array_write_flags(value, PM_WRITE_NODE_FLAGS_IMPLICIT_ARRAY),
3901 PM_LOCATION_INIT_NODES(target, value),
3902 target->name,
3903 target->base.location,
3904 value,
3905 TOK2LOC(parser, operator)
3906 );
3907}
3908
3912static void
3913pm_def_node_receiver_check(pm_parser_t *parser, const pm_node_t *node) {
3914 switch (PM_NODE_TYPE(node)) {
3915 case PM_BEGIN_NODE: {
3916 const pm_begin_node_t *cast = (pm_begin_node_t *) node;
3917 if (cast->statements != NULL) pm_def_node_receiver_check(parser, UP(cast->statements));
3918 break;
3919 }
3920 case PM_PARENTHESES_NODE: {
3921 const pm_parentheses_node_t *cast = (const pm_parentheses_node_t *) node;
3922 if (cast->body != NULL) pm_def_node_receiver_check(parser, cast->body);
3923 break;
3924 }
3925 case PM_STATEMENTS_NODE: {
3926 const pm_statements_node_t *cast = (const pm_statements_node_t *) node;
3927 pm_def_node_receiver_check(parser, cast->body.nodes[cast->body.size - 1]);
3928 break;
3929 }
3930 case PM_ARRAY_NODE:
3931 case PM_FLOAT_NODE:
3932 case PM_IMAGINARY_NODE:
3933 case PM_INTEGER_NODE:
3934 case PM_INTERPOLATED_REGULAR_EXPRESSION_NODE:
3935 case PM_INTERPOLATED_STRING_NODE:
3936 case PM_INTERPOLATED_SYMBOL_NODE:
3937 case PM_INTERPOLATED_X_STRING_NODE:
3938 case PM_RATIONAL_NODE:
3939 case PM_REGULAR_EXPRESSION_NODE:
3940 case PM_SOURCE_ENCODING_NODE:
3941 case PM_SOURCE_FILE_NODE:
3942 case PM_SOURCE_LINE_NODE:
3943 case PM_STRING_NODE:
3944 case PM_SYMBOL_NODE:
3945 case PM_X_STRING_NODE:
3946 pm_parser_err_node(parser, node, PM_ERR_SINGLETON_FOR_LITERALS);
3947 break;
3948 default:
3949 break;
3950 }
3951}
3952
3956static pm_def_node_t *
3957pm_def_node_create(
3958 pm_parser_t *parser,
3959 pm_constant_id_t name,
3960 const pm_token_t *name_loc,
3961 pm_node_t *receiver,
3962 pm_parameters_node_t *parameters,
3963 pm_node_t *body,
3964 pm_constant_id_list_t *locals,
3965 const pm_token_t *def_keyword,
3966 const pm_token_t *operator,
3967 const pm_token_t *lparen,
3968 const pm_token_t *rparen,
3969 const pm_token_t *equal,
3970 const pm_token_t *end_keyword
3971) {
3972 if (receiver != NULL) {
3973 pm_def_node_receiver_check(parser, receiver);
3974 }
3975
3976 return pm_def_node_new(
3977 parser->arena,
3978 ++parser->node_id,
3979 0,
3980 (end_keyword == NULL) ? PM_LOCATION_INIT_TOKEN_NODE(parser, def_keyword, body) : PM_LOCATION_INIT_TOKENS(parser, def_keyword, end_keyword),
3981 name,
3982 TOK2LOC(parser, name_loc),
3983 receiver,
3984 parameters,
3985 body,
3986 *locals,
3987 TOK2LOC(parser, def_keyword),
3988 NTOK2LOC(parser, operator),
3989 NTOK2LOC(parser, lparen),
3990 NTOK2LOC(parser, rparen),
3991 NTOK2LOC(parser, equal),
3992 NTOK2LOC(parser, end_keyword)
3993 );
3994}
3995
3999static pm_defined_node_t *
4000pm_defined_node_create(pm_parser_t *parser, const pm_token_t *lparen, pm_node_t *value, const pm_token_t *rparen, const pm_token_t *keyword) {
4001 return pm_defined_node_new(
4002 parser->arena,
4003 ++parser->node_id,
4004 0,
4005 (rparen == NULL) ? PM_LOCATION_INIT_TOKEN_NODE(parser, keyword, value) : PM_LOCATION_INIT_TOKENS(parser, keyword, rparen),
4006 NTOK2LOC(parser, lparen),
4007 value,
4008 NTOK2LOC(parser, rparen),
4009 TOK2LOC(parser, keyword)
4010 );
4011}
4012
4016static pm_else_node_t *
4017pm_else_node_create(pm_parser_t *parser, const pm_token_t *else_keyword, pm_statements_node_t *statements, const pm_token_t *end_keyword) {
4018 return pm_else_node_new(
4019 parser->arena,
4020 ++parser->node_id,
4021 0,
4022 ((end_keyword == NULL) && (statements != NULL)) ? PM_LOCATION_INIT_TOKEN_NODE(parser, else_keyword, statements) : PM_LOCATION_INIT_TOKENS(parser, else_keyword, end_keyword),
4023 TOK2LOC(parser, else_keyword),
4024 statements,
4025 NTOK2LOC(parser, end_keyword)
4026 );
4027}
4028
4032static pm_embedded_statements_node_t *
4033pm_embedded_statements_node_create(pm_parser_t *parser, const pm_token_t *opening, pm_statements_node_t *statements, const pm_token_t *closing) {
4034 return pm_embedded_statements_node_new(
4035 parser->arena,
4036 ++parser->node_id,
4037 0,
4038 PM_LOCATION_INIT_TOKENS(parser, opening, closing),
4039 TOK2LOC(parser, opening),
4040 statements,
4041 TOK2LOC(parser, closing)
4042 );
4043}
4044
4048static pm_embedded_variable_node_t *
4049pm_embedded_variable_node_create(pm_parser_t *parser, const pm_token_t *operator, pm_node_t *variable) {
4050 return pm_embedded_variable_node_new(
4051 parser->arena,
4052 ++parser->node_id,
4053 0,
4054 PM_LOCATION_INIT_TOKEN_NODE(parser, operator, variable),
4055 TOK2LOC(parser, operator),
4056 variable
4057 );
4058}
4059
4063static pm_ensure_node_t *
4064pm_ensure_node_create(pm_parser_t *parser, const pm_token_t *ensure_keyword, pm_statements_node_t *statements, const pm_token_t *end_keyword) {
4065 return pm_ensure_node_new(
4066 parser->arena,
4067 ++parser->node_id,
4068 0,
4069 PM_LOCATION_INIT_TOKENS(parser, ensure_keyword, end_keyword),
4070 TOK2LOC(parser, ensure_keyword),
4071 statements,
4072 TOK2LOC(parser, end_keyword)
4073 );
4074}
4075
4079static pm_false_node_t *
4080pm_false_node_create(pm_parser_t *parser, const pm_token_t *token) {
4081 assert(token->type == PM_TOKEN_KEYWORD_FALSE);
4082
4083 return pm_false_node_new(
4084 parser->arena,
4085 ++parser->node_id,
4086 PM_NODE_FLAG_STATIC_LITERAL,
4087 PM_LOCATION_INIT_TOKEN(parser, token)
4088 );
4089}
4090
4095static pm_find_pattern_node_t *
4096pm_find_pattern_node_create(pm_parser_t *parser, pm_node_list_t *nodes) {
4097 assert(nodes->size >= 2);
4098 pm_node_t *left = nodes->nodes[0];
4099 pm_node_t *right = nodes->nodes[nodes->size - 1];
4100
4101 assert(PM_NODE_TYPE_P(left, PM_SPLAT_NODE));
4102 assert(PM_NODE_TYPE_P(right, PM_SPLAT_NODE));
4103
4104 pm_find_pattern_node_t *node = pm_find_pattern_node_new(
4105 parser->arena,
4106 ++parser->node_id,
4107 0,
4108 PM_LOCATION_INIT_NODES(left, right),
4109 NULL,
4110 (pm_splat_node_t *) left,
4111 ((pm_node_list_t) { 0 }),
4112 (pm_splat_node_t *) right,
4113 ((pm_location_t) { 0 }),
4114 ((pm_location_t) { 0 })
4115 );
4116
4117 // For now we're going to just copy over each pointer manually. This could be
4118 // much more efficient, as we could instead resize the node list to only point
4119 // to 1...-1.
4120 for (size_t index = 1; index < nodes->size - 1; index++) {
4121 pm_node_list_append(parser->arena, &node->requireds, nodes->nodes[index]);
4122 }
4123
4124 return node;
4125}
4126
4131static double
4132pm_double_parse(pm_parser_t *parser, const pm_token_t *token) {
4133 ptrdiff_t diff = token->end - token->start;
4134 if (diff <= 0) return 0.0;
4135
4136 // First, get a buffer of the content.
4137 size_t length = (size_t) diff;
4138 const size_t buffer_size = sizeof(char) * (length + 1);
4139 char *buffer = xmalloc(buffer_size);
4140 memcpy((void *) buffer, token->start, length);
4141
4142 // Next, determine if we need to replace the decimal point because of
4143 // locale-specific options, and then normalize them if we have to.
4144 char decimal_point = *localeconv()->decimal_point;
4145 if (decimal_point != '.') {
4146 for (size_t index = 0; index < length; index++) {
4147 if (buffer[index] == '.') buffer[index] = decimal_point;
4148 }
4149 }
4150
4151 // Next, handle underscores by removing them from the buffer.
4152 for (size_t index = 0; index < length; index++) {
4153 if (buffer[index] == '_') {
4154 memmove((void *) (buffer + index), (void *) (buffer + index + 1), length - index);
4155 length--;
4156 }
4157 }
4158
4159 // Null-terminate the buffer so that strtod cannot read off the end.
4160 buffer[length] = '\0';
4161
4162 // Now, call strtod to parse the value. Note that CRuby has their own
4163 // version of strtod which avoids locales. We're okay using the locale-aware
4164 // version because we've already validated through the parser that the token
4165 // is in a valid format.
4166 errno = 0;
4167 char *eptr;
4168 double value = strtod(buffer, &eptr);
4169
4170 // This should never happen, because we've already checked that the token
4171 // is in a valid format. However it's good to be safe.
4172 if ((eptr != buffer + length) || (errno != 0 && errno != ERANGE)) {
4173 PM_PARSER_ERR_TOKEN_FORMAT_CONTENT(parser, token, PM_ERR_FLOAT_PARSE);
4174 xfree_sized(buffer, buffer_size);
4175 return 0.0;
4176 }
4177
4178 // If errno is set, then it should only be ERANGE. At this point we need to
4179 // check if it's infinity (it should be).
4180 if (errno == ERANGE && PRISM_ISINF(value)) {
4181 int warn_width;
4182 const char *ellipsis;
4183
4184 if (length > 20) {
4185 warn_width = 20;
4186 ellipsis = "...";
4187 } else {
4188 warn_width = (int) length;
4189 ellipsis = "";
4190 }
4191
4192 pm_diagnostic_list_append_format(&parser->metadata_arena, &parser->warning_list, PM_TOKEN_START(parser, token), PM_TOKEN_LENGTH(token), PM_WARN_FLOAT_OUT_OF_RANGE, warn_width, (const char *) token->start, ellipsis);
4193 value = (value < 0.0) ? -HUGE_VAL : HUGE_VAL;
4194 }
4195
4196 // Finally we can free the buffer and return the value.
4197 xfree_sized(buffer, buffer_size);
4198 return value;
4199}
4200
4204static pm_float_node_t *
4205pm_float_node_create(pm_parser_t *parser, const pm_token_t *token) {
4206 assert(token->type == PM_TOKEN_FLOAT);
4207
4208 return pm_float_node_new(
4209 parser->arena,
4210 ++parser->node_id,
4211 PM_NODE_FLAG_STATIC_LITERAL,
4212 PM_LOCATION_INIT_TOKEN(parser, token),
4213 pm_double_parse(parser, token)
4214 );
4215}
4216
4220static pm_imaginary_node_t *
4221pm_float_node_imaginary_create(pm_parser_t *parser, const pm_token_t *token) {
4222 assert(token->type == PM_TOKEN_FLOAT_IMAGINARY);
4223
4224 return pm_imaginary_node_new(
4225 parser->arena,
4226 ++parser->node_id,
4227 PM_NODE_FLAG_STATIC_LITERAL,
4228 PM_LOCATION_INIT_TOKEN(parser, token),
4229 UP(pm_float_node_create(parser, &((pm_token_t) {
4230 .type = PM_TOKEN_FLOAT,
4231 .start = token->start,
4232 .end = token->end - 1
4233 })))
4234 );
4235}
4236
4240static pm_rational_node_t *
4241pm_float_node_rational_create(pm_parser_t *parser, const pm_token_t *token) {
4242 assert(token->type == PM_TOKEN_FLOAT_RATIONAL);
4243
4244 pm_rational_node_t *node = pm_rational_node_new(
4245 parser->arena,
4246 ++parser->node_id,
4247 PM_INTEGER_BASE_FLAGS_DECIMAL | PM_NODE_FLAG_STATIC_LITERAL,
4248 PM_LOCATION_INIT_TOKEN(parser, token),
4249 ((pm_integer_t) { 0 }),
4250 ((pm_integer_t) { 0 })
4251 );
4252
4253 const uint8_t *start = token->start;
4254 const uint8_t *end = token->end - 1; // r
4255
4256 while (start < end && *start == '0') start++; // 0.1 -> .1
4257 while (end > start && end[-1] == '0') end--; // 1.0 -> 1.
4258
4259 size_t length = (size_t) (end - start);
4260 if (length == 1) {
4261 node->denominator.value = 1;
4262 return node;
4263 }
4264
4265 const uint8_t *point = memchr(start, '.', length);
4266 assert(point && "should have a decimal point");
4267
4268 uint8_t *digits = xmalloc(length);
4269 if (digits == NULL) {
4270 fputs("[pm_float_node_rational_create] Failed to allocate memory", stderr);
4271 abort();
4272 }
4273
4274 memcpy(digits, start, (unsigned long) (point - start));
4275 memcpy(digits + (point - start), point + 1, (unsigned long) (end - point - 1));
4276 pm_integer_parse(&node->numerator, PM_INTEGER_BASE_DEFAULT, digits, digits + length - 1);
4277
4278 size_t fract_length = 0;
4279 for (const uint8_t *fract = point; fract < end; ++fract) {
4280 if (*fract != '_') ++fract_length;
4281 }
4282 digits[0] = '1';
4283 if (fract_length > 1) memset(digits + 1, '0', fract_length - 1);
4284 pm_integer_parse(&node->denominator, PM_INTEGER_BASE_DEFAULT, digits, digits + fract_length);
4285 xfree_sized(digits, length);
4286
4287 pm_integers_reduce(&node->numerator, &node->denominator);
4288 pm_integer_arena_move(parser->arena, &node->numerator);
4289 pm_integer_arena_move(parser->arena, &node->denominator);
4290 return node;
4291}
4292
4297static pm_imaginary_node_t *
4298pm_float_node_rational_imaginary_create(pm_parser_t *parser, const pm_token_t *token) {
4299 assert(token->type == PM_TOKEN_FLOAT_RATIONAL_IMAGINARY);
4300
4301 return pm_imaginary_node_new(
4302 parser->arena,
4303 ++parser->node_id,
4304 PM_NODE_FLAG_STATIC_LITERAL,
4305 PM_LOCATION_INIT_TOKEN(parser, token),
4306 UP(pm_float_node_rational_create(parser, &((pm_token_t) {
4307 .type = PM_TOKEN_FLOAT_RATIONAL,
4308 .start = token->start,
4309 .end = token->end - 1
4310 })))
4311 );
4312}
4313
4317static pm_for_node_t *
4318pm_for_node_create(
4319 pm_parser_t *parser,
4320 pm_node_t *index,
4321 pm_node_t *collection,
4322 pm_statements_node_t *statements,
4323 const pm_token_t *for_keyword,
4324 const pm_token_t *in_keyword,
4325 const pm_token_t *do_keyword,
4326 const pm_token_t *end_keyword
4327) {
4328 return pm_for_node_new(
4329 parser->arena,
4330 ++parser->node_id,
4331 0,
4332 PM_LOCATION_INIT_TOKENS(parser, for_keyword, end_keyword),
4333 index,
4334 collection,
4335 statements,
4336 TOK2LOC(parser, for_keyword),
4337 TOK2LOC(parser, in_keyword),
4338 NTOK2LOC(parser, do_keyword),
4339 TOK2LOC(parser, end_keyword)
4340 );
4341}
4342
4346static pm_forwarding_arguments_node_t *
4347pm_forwarding_arguments_node_create(pm_parser_t *parser, const pm_token_t *token) {
4348 assert(token->type == PM_TOKEN_UDOT_DOT_DOT);
4349
4350 return pm_forwarding_arguments_node_new(
4351 parser->arena,
4352 ++parser->node_id,
4353 0,
4354 PM_LOCATION_INIT_TOKEN(parser, token)
4355 );
4356}
4357
4361static pm_forwarding_parameter_node_t *
4362pm_forwarding_parameter_node_create(pm_parser_t *parser, const pm_token_t *token) {
4363 assert(token->type == PM_TOKEN_UDOT_DOT_DOT);
4364
4365 return pm_forwarding_parameter_node_new(
4366 parser->arena,
4367 ++parser->node_id,
4368 0,
4369 PM_LOCATION_INIT_TOKEN(parser, token)
4370 );
4371}
4372
4376static pm_forwarding_super_node_t *
4377pm_forwarding_super_node_create(pm_parser_t *parser, const pm_token_t *token, pm_arguments_t *arguments) {
4378 assert(arguments->block == NULL || PM_NODE_TYPE_P(arguments->block, PM_BLOCK_NODE));
4379 assert(token->type == PM_TOKEN_KEYWORD_SUPER);
4380
4381 pm_block_node_t *block = NULL;
4382 if (arguments->block != NULL) {
4383 block = (pm_block_node_t *) arguments->block;
4384 }
4385
4386 return pm_forwarding_super_node_new(
4387 parser->arena,
4388 ++parser->node_id,
4389 0,
4390 (block == NULL) ? PM_LOCATION_INIT_TOKEN(parser, token) : PM_LOCATION_INIT_TOKEN_NODE(parser, token, block),
4391 PM_LOCATION_INIT_TOKEN(parser, token),
4392 block
4393 );
4394}
4395
4400static pm_hash_pattern_node_t *
4401pm_hash_pattern_node_empty_create(pm_parser_t *parser, const pm_token_t *opening, const pm_token_t *closing) {
4402 return pm_hash_pattern_node_new(
4403 parser->arena,
4404 ++parser->node_id,
4405 0,
4406 PM_LOCATION_INIT_TOKENS(parser, opening, closing),
4407 NULL,
4408 ((pm_node_list_t) { 0 }),
4409 NULL,
4410 TOK2LOC(parser, opening),
4411 TOK2LOC(parser, closing)
4412 );
4413}
4414
4418static pm_hash_pattern_node_t *
4419pm_hash_pattern_node_node_list_create(pm_parser_t *parser, pm_node_list_t *elements, pm_node_t *rest) {
4420 uint32_t start;
4421 uint32_t end;
4422
4423 if (elements->size > 0) {
4424 if (rest) {
4425 start = MIN(PM_NODE_START(rest), PM_NODE_START(elements->nodes[0]));
4426 end = MAX(PM_NODE_END(rest), PM_NODE_END(elements->nodes[elements->size - 1]));
4427 } else {
4428 start = PM_NODE_START(elements->nodes[0]);
4429 end = PM_NODE_END(elements->nodes[elements->size - 1]);
4430 }
4431 } else {
4432 assert(rest != NULL);
4433 start = PM_NODE_START(rest);
4434 end = PM_NODE_END(rest);
4435 }
4436
4437 pm_hash_pattern_node_t *node = pm_hash_pattern_node_new(
4438 parser->arena,
4439 ++parser->node_id,
4440 0,
4441 ((pm_location_t) { .start = start, .length = U32(end - start) }),
4442 NULL,
4443 ((pm_node_list_t) { 0 }),
4444 rest,
4445 ((pm_location_t) { 0 }),
4446 ((pm_location_t) { 0 })
4447 );
4448
4449 pm_node_list_concat(parser->arena, &node->elements, elements);
4450 return node;
4451}
4452
4456static pm_constant_id_t
4457pm_global_variable_write_name(pm_parser_t *parser, const pm_node_t *target) {
4458 switch (PM_NODE_TYPE(target)) {
4459 case PM_GLOBAL_VARIABLE_READ_NODE:
4460 return ((pm_global_variable_read_node_t *) target)->name;
4461 case PM_BACK_REFERENCE_READ_NODE:
4462 return ((pm_back_reference_read_node_t *) target)->name;
4463 case PM_NUMBERED_REFERENCE_READ_NODE:
4464 // This will only ever happen in the event of a syntax error, but we
4465 // still need to provide something for the node.
4466 return pm_parser_constant_id_raw(parser, parser->start + PM_NODE_START(target), parser->start + PM_NODE_END(target));
4467 default:
4468 assert(false && "unreachable");
4469 return (pm_constant_id_t) -1;
4470 }
4471}
4472
4476static pm_global_variable_and_write_node_t *
4477pm_global_variable_and_write_node_create(pm_parser_t *parser, pm_node_t *target, const pm_token_t *operator, pm_node_t *value) {
4478 assert(operator->type == PM_TOKEN_AMPERSAND_AMPERSAND_EQUAL);
4479
4480 return pm_global_variable_and_write_node_new(
4481 parser->arena,
4482 ++parser->node_id,
4483 0,
4484 PM_LOCATION_INIT_NODES(target, value),
4485 pm_global_variable_write_name(parser, target),
4486 target->location,
4487 TOK2LOC(parser, operator),
4488 value
4489 );
4490}
4491
4495static pm_global_variable_operator_write_node_t *
4496pm_global_variable_operator_write_node_create(pm_parser_t *parser, pm_node_t *target, const pm_token_t *operator, pm_node_t *value) {
4497 return pm_global_variable_operator_write_node_new(
4498 parser->arena,
4499 ++parser->node_id,
4500 0,
4501 PM_LOCATION_INIT_NODES(target, value),
4502 pm_global_variable_write_name(parser, target),
4503 target->location,
4504 TOK2LOC(parser, operator),
4505 value,
4506 pm_parser_constant_id_raw(parser, operator->start, operator->end - 1)
4507 );
4508}
4509
4513static pm_global_variable_or_write_node_t *
4514pm_global_variable_or_write_node_create(pm_parser_t *parser, pm_node_t *target, const pm_token_t *operator, pm_node_t *value) {
4515 assert(operator->type == PM_TOKEN_PIPE_PIPE_EQUAL);
4516
4517 return pm_global_variable_or_write_node_new(
4518 parser->arena,
4519 ++parser->node_id,
4520 0,
4521 PM_LOCATION_INIT_NODES(target, value),
4522 pm_global_variable_write_name(parser, target),
4523 target->location,
4524 TOK2LOC(parser, operator),
4525 value
4526 );
4527}
4528
4532static pm_global_variable_read_node_t *
4533pm_global_variable_read_node_create(pm_parser_t *parser, const pm_token_t *name) {
4534 return pm_global_variable_read_node_new(
4535 parser->arena,
4536 ++parser->node_id,
4537 0,
4538 PM_LOCATION_INIT_TOKEN(parser, name),
4539 pm_parser_constant_id_token(parser, name)
4540 );
4541}
4542
4546static pm_global_variable_read_node_t *
4547pm_global_variable_read_node_synthesized_create(pm_parser_t *parser, pm_constant_id_t name) {
4548 return pm_global_variable_read_node_new(
4549 parser->arena,
4550 ++parser->node_id,
4551 0,
4552 PM_LOCATION_INIT_UNSET,
4553 name
4554 );
4555}
4556
4560static pm_global_variable_write_node_t *
4561pm_global_variable_write_node_create(pm_parser_t *parser, pm_node_t *target, const pm_token_t *operator, pm_node_t *value) {
4562 return pm_global_variable_write_node_new(
4563 parser->arena,
4564 ++parser->node_id,
4565 pm_implicit_array_write_flags(value, PM_WRITE_NODE_FLAGS_IMPLICIT_ARRAY),
4566 PM_LOCATION_INIT_NODES(target, value),
4567 pm_global_variable_write_name(parser, target),
4568 target->location,
4569 value,
4570 TOK2LOC(parser, operator)
4571 );
4572}
4573
4577static pm_global_variable_write_node_t *
4578pm_global_variable_write_node_synthesized_create(pm_parser_t *parser, pm_constant_id_t name, pm_node_t *value) {
4579 return pm_global_variable_write_node_new(
4580 parser->arena,
4581 ++parser->node_id,
4582 0,
4583 PM_LOCATION_INIT_UNSET,
4584 name,
4585 ((pm_location_t) { 0 }),
4586 value,
4587 ((pm_location_t) { 0 })
4588 );
4589}
4590
4594static pm_hash_node_t *
4595pm_hash_node_create(pm_parser_t *parser, const pm_token_t *opening) {
4596 assert(opening != NULL);
4597
4598 return pm_hash_node_new(
4599 parser->arena,
4600 ++parser->node_id,
4601 PM_NODE_FLAG_STATIC_LITERAL,
4602 PM_LOCATION_INIT_TOKEN(parser, opening),
4603 TOK2LOC(parser, opening),
4604 ((pm_node_list_t) { 0 }),
4605 ((pm_location_t) { 0 })
4606 );
4607}
4608
4612static PRISM_INLINE void
4613pm_hash_node_elements_append(pm_arena_t *arena, pm_hash_node_t *hash, pm_node_t *element) {
4614 pm_node_list_append(arena, &hash->elements, element);
4615
4616 bool static_literal = PM_NODE_TYPE_P(element, PM_ASSOC_NODE);
4617 if (static_literal) {
4618 pm_assoc_node_t *assoc = (pm_assoc_node_t *) element;
4619 static_literal = !PM_NODE_TYPE_P(assoc->key, PM_ARRAY_NODE) && !PM_NODE_TYPE_P(assoc->key, PM_HASH_NODE) && !PM_NODE_TYPE_P(assoc->key, PM_RANGE_NODE);
4620 static_literal = static_literal && PM_NODE_FLAG_P(assoc->key, PM_NODE_FLAG_STATIC_LITERAL);
4621 static_literal = static_literal && PM_NODE_FLAG_P(assoc, PM_NODE_FLAG_STATIC_LITERAL);
4622 }
4623
4624 if (!static_literal) {
4625 pm_node_flag_unset(UP(hash), PM_NODE_FLAG_STATIC_LITERAL);
4626 }
4627}
4628
4629static PRISM_INLINE void
4630pm_hash_node_closing_loc_set(const pm_parser_t *parser, pm_hash_node_t *hash, pm_token_t *token) {
4631 PM_NODE_LENGTH_SET_TOKEN(parser, hash, token);
4632 hash->closing_loc = TOK2LOC(parser, token);
4633}
4634
4638static pm_if_node_t *
4639pm_if_node_create(pm_parser_t *parser,
4640 const pm_token_t *if_keyword,
4641 pm_node_t *predicate,
4642 const pm_token_t *then_keyword,
4643 pm_statements_node_t *statements,
4644 pm_node_t *subsequent,
4645 const pm_token_t *end_keyword
4646) {
4647 pm_conditional_predicate(parser, predicate, PM_CONDITIONAL_PREDICATE_TYPE_CONDITIONAL);
4648
4649 uint32_t start = PM_TOKEN_START(parser, if_keyword);
4650 uint32_t end;
4651
4652 if (end_keyword != NULL) {
4653 end = PM_TOKEN_END(parser, end_keyword);
4654 } else if (subsequent != NULL) {
4655 end = PM_NODE_END(subsequent);
4656 } else if (pm_statements_node_body_length(statements) != 0) {
4657 end = PM_NODE_END(statements);
4658 } else {
4659 end = PM_NODE_END(predicate);
4660 }
4661
4662 return pm_if_node_new(
4663 parser->arena,
4664 ++parser->node_id,
4665 PM_NODE_FLAG_NEWLINE,
4666 ((pm_location_t) { .start = start, .length = U32(end - start) }),
4667 TOK2LOC(parser, if_keyword),
4668 predicate,
4669 NTOK2LOC(parser, then_keyword),
4670 statements,
4671 subsequent,
4672 NTOK2LOC(parser, end_keyword)
4673 );
4674}
4675
4679static pm_if_node_t *
4680pm_if_node_modifier_create(pm_parser_t *parser, pm_node_t *statement, const pm_token_t *if_keyword, pm_node_t *predicate) {
4681 pm_conditional_predicate(parser, predicate, PM_CONDITIONAL_PREDICATE_TYPE_CONDITIONAL);
4682
4683 pm_statements_node_t *statements = pm_statements_node_create(parser);
4684 pm_statements_node_body_append(parser, statements, statement, true);
4685
4686 return pm_if_node_new(
4687 parser->arena,
4688 ++parser->node_id,
4689 PM_NODE_FLAG_NEWLINE,
4690 PM_LOCATION_INIT_NODES(statement, predicate),
4691 TOK2LOC(parser, if_keyword),
4692 predicate,
4693 ((pm_location_t) { 0 }),
4694 statements,
4695 NULL,
4696 ((pm_location_t) { 0 })
4697 );
4698}
4699
4703static pm_if_node_t *
4704pm_if_node_ternary_create(pm_parser_t *parser, pm_node_t *predicate, const pm_token_t *qmark, pm_node_t *true_expression, const pm_token_t *colon, pm_node_t *false_expression) {
4705 pm_assert_value_expression(parser, predicate);
4706 pm_conditional_predicate(parser, predicate, PM_CONDITIONAL_PREDICATE_TYPE_CONDITIONAL);
4707
4708 pm_statements_node_t *if_statements = pm_statements_node_create(parser);
4709 pm_statements_node_body_append(parser, if_statements, true_expression, true);
4710
4711 pm_statements_node_t *else_statements = pm_statements_node_create(parser);
4712 pm_statements_node_body_append(parser, else_statements, false_expression, true);
4713
4714 pm_else_node_t *else_node = pm_else_node_create(parser, colon, else_statements, NULL);
4715 return pm_if_node_new(
4716 parser->arena,
4717 ++parser->node_id,
4718 PM_NODE_FLAG_NEWLINE,
4719 PM_LOCATION_INIT_NODES(predicate, false_expression),
4720 ((pm_location_t) { 0 }),
4721 predicate,
4722 TOK2LOC(parser, qmark),
4723 if_statements,
4724 UP(else_node),
4725 ((pm_location_t) { 0 })
4726 );
4727}
4728
4729static PRISM_INLINE void
4730pm_if_node_end_keyword_loc_set(const pm_parser_t *parser, pm_if_node_t *node, const pm_token_t *keyword) {
4731 PM_NODE_LENGTH_SET_TOKEN(parser, node, keyword);
4732 node->end_keyword_loc = TOK2LOC(parser, keyword);
4733}
4734
4735static PRISM_INLINE void
4736pm_else_node_end_keyword_loc_set(const pm_parser_t *parser, pm_else_node_t *node, const pm_token_t *keyword) {
4737 PM_NODE_LENGTH_SET_TOKEN(parser, node, keyword);
4738 node->end_keyword_loc = TOK2LOC(parser, keyword);
4739}
4740
4744static pm_implicit_node_t *
4745pm_implicit_node_create(pm_parser_t *parser, pm_node_t *value) {
4746 return pm_implicit_node_new(
4747 parser->arena,
4748 ++parser->node_id,
4749 0,
4750 PM_LOCATION_INIT_NODE(value),
4751 value
4752 );
4753}
4754
4758static pm_implicit_rest_node_t *
4759pm_implicit_rest_node_create(pm_parser_t *parser, const pm_token_t *token) {
4760 assert(token->type == PM_TOKEN_COMMA);
4761
4762 return pm_implicit_rest_node_new(
4763 parser->arena,
4764 ++parser->node_id,
4765 0,
4766 PM_LOCATION_INIT_TOKEN(parser, token)
4767 );
4768}
4769
4773static pm_integer_node_t *
4774pm_integer_node_create(pm_parser_t *parser, pm_node_flags_t base, const pm_token_t *token) {
4775 assert(token->type == PM_TOKEN_INTEGER);
4776
4777 pm_integer_node_t *node = pm_integer_node_new(
4778 parser->arena,
4779 ++parser->node_id,
4780 base | PM_NODE_FLAG_STATIC_LITERAL,
4781 PM_LOCATION_INIT_TOKEN(parser, token),
4782 ((pm_integer_t) { 0 })
4783 );
4784
4785 if (parser->integer.lexed) {
4786 // The value was already computed during lexing.
4787 node->value.value = parser->integer.value;
4788 parser->integer.lexed = false;
4789 } else {
4790 pm_integer_base_t integer_base = PM_INTEGER_BASE_DECIMAL;
4791 switch (base) {
4792 case PM_INTEGER_BASE_FLAGS_BINARY: integer_base = PM_INTEGER_BASE_BINARY; break;
4793 case PM_INTEGER_BASE_FLAGS_OCTAL: integer_base = PM_INTEGER_BASE_OCTAL; break;
4794 case PM_INTEGER_BASE_FLAGS_DECIMAL: break;
4795 case PM_INTEGER_BASE_FLAGS_HEXADECIMAL: integer_base = PM_INTEGER_BASE_HEXADECIMAL; break;
4796 default: assert(false && "unreachable"); break;
4797 }
4798
4799 pm_integer_parse(&node->value, integer_base, token->start, token->end);
4800 pm_integer_arena_move(parser->arena, &node->value);
4801 }
4802
4803 return node;
4804}
4805
4810static pm_imaginary_node_t *
4811pm_integer_node_imaginary_create(pm_parser_t *parser, pm_node_flags_t base, const pm_token_t *token) {
4812 assert(token->type == PM_TOKEN_INTEGER_IMAGINARY);
4813
4814 return pm_imaginary_node_new(
4815 parser->arena,
4816 ++parser->node_id,
4817 PM_NODE_FLAG_STATIC_LITERAL,
4818 PM_LOCATION_INIT_TOKEN(parser, token),
4819 UP(pm_integer_node_create(parser, base, &((pm_token_t) {
4820 .type = PM_TOKEN_INTEGER,
4821 .start = token->start,
4822 .end = token->end - 1
4823 })))
4824 );
4825}
4826
4831static pm_rational_node_t *
4832pm_integer_node_rational_create(pm_parser_t *parser, pm_node_flags_t base, const pm_token_t *token) {
4833 assert(token->type == PM_TOKEN_INTEGER_RATIONAL);
4834
4835 pm_rational_node_t *node = pm_rational_node_new(
4836 parser->arena,
4837 ++parser->node_id,
4838 base | PM_NODE_FLAG_STATIC_LITERAL,
4839 PM_LOCATION_INIT_TOKEN(parser, token),
4840 ((pm_integer_t) { 0 }),
4841 ((pm_integer_t) { .value = 1 })
4842 );
4843
4844 pm_integer_base_t integer_base = PM_INTEGER_BASE_DECIMAL;
4845 switch (base) {
4846 case PM_INTEGER_BASE_FLAGS_BINARY: integer_base = PM_INTEGER_BASE_BINARY; break;
4847 case PM_INTEGER_BASE_FLAGS_OCTAL: integer_base = PM_INTEGER_BASE_OCTAL; break;
4848 case PM_INTEGER_BASE_FLAGS_DECIMAL: break;
4849 case PM_INTEGER_BASE_FLAGS_HEXADECIMAL: integer_base = PM_INTEGER_BASE_HEXADECIMAL; break;
4850 default: assert(false && "unreachable"); break;
4851 }
4852
4853 pm_integer_parse(&node->numerator, integer_base, token->start, token->end - 1);
4854 pm_integer_arena_move(parser->arena, &node->numerator);
4855
4856 return node;
4857}
4858
4863static pm_imaginary_node_t *
4864pm_integer_node_rational_imaginary_create(pm_parser_t *parser, pm_node_flags_t base, const pm_token_t *token) {
4865 assert(token->type == PM_TOKEN_INTEGER_RATIONAL_IMAGINARY);
4866
4867 return pm_imaginary_node_new(
4868 parser->arena,
4869 ++parser->node_id,
4870 PM_NODE_FLAG_STATIC_LITERAL,
4871 PM_LOCATION_INIT_TOKEN(parser, token),
4872 UP(pm_integer_node_rational_create(parser, base, &((pm_token_t) {
4873 .type = PM_TOKEN_INTEGER_RATIONAL,
4874 .start = token->start,
4875 .end = token->end - 1
4876 })))
4877 );
4878}
4879
4883static pm_in_node_t *
4884pm_in_node_create(pm_parser_t *parser, pm_node_t *pattern, pm_statements_node_t *statements, const pm_token_t *in_keyword, const pm_token_t *then_keyword) {
4885 uint32_t start = PM_TOKEN_START(parser, in_keyword);
4886 uint32_t end;
4887
4888 if (statements != NULL) {
4889 end = PM_NODE_END(statements);
4890 } else if (then_keyword != NULL) {
4891 end = PM_TOKEN_END(parser, then_keyword);
4892 } else {
4893 end = PM_NODE_END(pattern);
4894 }
4895
4896 return pm_in_node_new(
4897 parser->arena,
4898 ++parser->node_id,
4899 0,
4900 ((pm_location_t) { .start = start, .length = U32(end - start) }),
4901 pattern,
4902 statements,
4903 TOK2LOC(parser, in_keyword),
4904 NTOK2LOC(parser, then_keyword)
4905 );
4906}
4907
4911static pm_instance_variable_and_write_node_t *
4912pm_instance_variable_and_write_node_create(pm_parser_t *parser, pm_instance_variable_read_node_t *target, const pm_token_t *operator, pm_node_t *value) {
4913 assert(operator->type == PM_TOKEN_AMPERSAND_AMPERSAND_EQUAL);
4914
4915 return pm_instance_variable_and_write_node_new(
4916 parser->arena,
4917 ++parser->node_id,
4918 0,
4919 PM_LOCATION_INIT_NODES(target, value),
4920 target->name,
4921 target->base.location,
4922 TOK2LOC(parser, operator),
4923 value
4924 );
4925}
4926
4930static pm_instance_variable_operator_write_node_t *
4931pm_instance_variable_operator_write_node_create(pm_parser_t *parser, pm_instance_variable_read_node_t *target, const pm_token_t *operator, pm_node_t *value) {
4932 return pm_instance_variable_operator_write_node_new(
4933 parser->arena,
4934 ++parser->node_id,
4935 0,
4936 PM_LOCATION_INIT_NODES(target, value),
4937 target->name,
4938 target->base.location,
4939 TOK2LOC(parser, operator),
4940 value,
4941 pm_parser_constant_id_raw(parser, operator->start, operator->end - 1)
4942 );
4943}
4944
4948static pm_instance_variable_or_write_node_t *
4949pm_instance_variable_or_write_node_create(pm_parser_t *parser, pm_instance_variable_read_node_t *target, const pm_token_t *operator, pm_node_t *value) {
4950 assert(operator->type == PM_TOKEN_PIPE_PIPE_EQUAL);
4951
4952 return pm_instance_variable_or_write_node_new(
4953 parser->arena,
4954 ++parser->node_id,
4955 0,
4956 PM_LOCATION_INIT_NODES(target, value),
4957 target->name,
4958 target->base.location,
4959 TOK2LOC(parser, operator),
4960 value
4961 );
4962}
4963
4967static pm_instance_variable_read_node_t *
4968pm_instance_variable_read_node_create(pm_parser_t *parser, const pm_token_t *token) {
4969 assert(token->type == PM_TOKEN_INSTANCE_VARIABLE);
4970
4971 return pm_instance_variable_read_node_new(
4972 parser->arena,
4973 ++parser->node_id,
4974 0,
4975 PM_LOCATION_INIT_TOKEN(parser, token),
4976 pm_parser_constant_id_token(parser, token)
4977 );
4978}
4979
4984static pm_instance_variable_write_node_t *
4985pm_instance_variable_write_node_create(pm_parser_t *parser, pm_instance_variable_read_node_t *read_node, pm_token_t *operator, pm_node_t *value) {
4986 return pm_instance_variable_write_node_new(
4987 parser->arena,
4988 ++parser->node_id,
4989 pm_implicit_array_write_flags(value, PM_WRITE_NODE_FLAGS_IMPLICIT_ARRAY),
4990 PM_LOCATION_INIT_NODES(read_node, value),
4991 read_node->name,
4992 read_node->base.location,
4993 value,
4994 TOK2LOC(parser, operator)
4995 );
4996}
4997
5003static void
5004pm_interpolated_node_append(pm_arena_t *arena, pm_node_t *node, pm_node_list_t *parts, pm_node_t *part) {
5005 switch (PM_NODE_TYPE(part)) {
5006 case PM_STRING_NODE:
5007 pm_node_flag_set(part, PM_NODE_FLAG_STATIC_LITERAL | PM_STRING_FLAGS_FROZEN);
5008 break;
5009 case PM_EMBEDDED_STATEMENTS_NODE: {
5010 pm_embedded_statements_node_t *cast = (pm_embedded_statements_node_t *) part;
5011 pm_node_t *embedded = (cast->statements != NULL && cast->statements->body.size == 1) ? cast->statements->body.nodes[0] : NULL;
5012
5013 if (embedded == NULL) {
5014 // If there are no statements or more than one statement, then
5015 // we lose the static literal flag.
5016 pm_node_flag_unset(node, PM_NODE_FLAG_STATIC_LITERAL);
5017 } else if (PM_NODE_TYPE_P(embedded, PM_STRING_NODE)) {
5018 // If the embedded statement is a string, then we can keep the
5019 // static literal flag and mark the string as frozen.
5020 pm_node_flag_set(embedded, PM_NODE_FLAG_STATIC_LITERAL | PM_STRING_FLAGS_FROZEN);
5021 } else if (PM_NODE_TYPE_P(embedded, PM_INTERPOLATED_STRING_NODE) && PM_NODE_FLAG_P(embedded, PM_NODE_FLAG_STATIC_LITERAL)) {
5022 // If the embedded statement is an interpolated string and it's
5023 // a static literal, then we can keep the static literal flag.
5024 } else {
5025 // Otherwise we lose the static literal flag.
5026 pm_node_flag_unset(node, PM_NODE_FLAG_STATIC_LITERAL);
5027 }
5028
5029 break;
5030 }
5031 case PM_EMBEDDED_VARIABLE_NODE:
5032 pm_node_flag_unset(UP(node), PM_NODE_FLAG_STATIC_LITERAL);
5033 break;
5034 default:
5035 assert(false && "unexpected node type");
5036 break;
5037 }
5038
5039 pm_node_list_append(arena, parts, part);
5040}
5041
5045static pm_interpolated_regular_expression_node_t *
5046pm_interpolated_regular_expression_node_create(pm_parser_t *parser, const pm_token_t *opening) {
5047 return pm_interpolated_regular_expression_node_new(
5048 parser->arena,
5049 ++parser->node_id,
5050 PM_NODE_FLAG_STATIC_LITERAL,
5051 PM_LOCATION_INIT_TOKEN(parser, opening),
5052 TOK2LOC(parser, opening),
5053 ((pm_node_list_t) { 0 }),
5054 TOK2LOC(parser, opening)
5055 );
5056}
5057
5058static PRISM_INLINE void
5059pm_interpolated_regular_expression_node_append(pm_arena_t *arena, pm_interpolated_regular_expression_node_t *node, pm_node_t *part) {
5060 if (PM_NODE_START(node) > PM_NODE_START(part)) {
5061 PM_NODE_START_SET_NODE(node, part);
5062 }
5063 if (PM_NODE_END(node) < PM_NODE_END(part)) {
5064 PM_NODE_LENGTH_SET_NODE(node, part);
5065 }
5066
5067 pm_interpolated_node_append(arena, UP(node), &node->parts, part);
5068}
5069
5070static PRISM_INLINE void
5071pm_interpolated_regular_expression_node_closing_set(pm_parser_t *parser, pm_interpolated_regular_expression_node_t *node, const pm_token_t *closing) {
5072 node->closing_loc = TOK2LOC(parser, closing);
5073 PM_NODE_LENGTH_SET_TOKEN(parser, node, closing);
5074 pm_node_flag_set(UP(node), pm_regular_expression_flags_create(parser, closing));
5075}
5076
5100static PRISM_INLINE void
5101pm_interpolated_string_node_append(pm_parser_t *parser, pm_interpolated_string_node_t *node, pm_node_t *part) {
5102 pm_arena_t *arena = parser->arena;
5103#define CLEAR_FLAGS(node) \
5104 node->base.flags = (pm_node_flags_t) (FL(node) & ~(PM_NODE_FLAG_STATIC_LITERAL | PM_INTERPOLATED_STRING_NODE_FLAGS_FROZEN | PM_INTERPOLATED_STRING_NODE_FLAGS_MUTABLE))
5105
5106#define MUTABLE_FLAGS(node) \
5107 node->base.flags = (pm_node_flags_t) ((FL(node) | PM_INTERPOLATED_STRING_NODE_FLAGS_MUTABLE) & ~PM_INTERPOLATED_STRING_NODE_FLAGS_FROZEN);
5108
5109 if (node->parts.size == 0 && node->opening_loc.length == 0) {
5110 PM_NODE_START_SET_NODE(node, part);
5111 }
5112
5113 if (PM_NODE_END(part) > PM_NODE_END(node)) {
5114 PM_NODE_LENGTH_SET_NODE(node, part);
5115 }
5116
5117 switch (PM_NODE_TYPE(part)) {
5118 case PM_STRING_NODE:
5119 // If inner string is not frozen, it stops being a static literal. We should *not* clear other flags,
5120 // because concatenating two frozen strings (`'foo' 'bar'`) is still frozen. This holds true for
5121 // as long as this interpolation only consists of other string literals.
5122 if (!PM_NODE_FLAG_P(part, PM_STRING_FLAGS_FROZEN)) {
5123 pm_node_flag_unset(UP(node), PM_NODE_FLAG_STATIC_LITERAL);
5124 }
5125 part->flags = (pm_node_flags_t) ((part->flags | PM_NODE_FLAG_STATIC_LITERAL | PM_STRING_FLAGS_FROZEN) & ~PM_STRING_FLAGS_MUTABLE);
5126 break;
5127 case PM_INTERPOLATED_STRING_NODE:
5128 if (PM_NODE_FLAG_P(part, PM_NODE_FLAG_STATIC_LITERAL)) {
5129 // If the string that we're concatenating is a static literal,
5130 // then we can keep the static literal flag for this string.
5131 } else {
5132 // Otherwise, we lose the static literal flag here and we should
5133 // also clear the mutability flags.
5134 CLEAR_FLAGS(node);
5135 }
5136 break;
5137 case PM_EMBEDDED_STATEMENTS_NODE: {
5138 pm_embedded_statements_node_t *cast = (pm_embedded_statements_node_t *) part;
5139 pm_node_t *embedded = (cast->statements != NULL && cast->statements->body.size == 1) ? cast->statements->body.nodes[0] : NULL;
5140
5141 if (embedded == NULL) {
5142 // If we're embedding multiple statements or no statements, then
5143 // the string is not longer a static literal.
5144 CLEAR_FLAGS(node);
5145 } else if (PM_NODE_TYPE_P(embedded, PM_STRING_NODE)) {
5146 // If the embedded statement is a string, then we can make that
5147 // string as frozen and static literal, and not touch the static
5148 // literal status of this string.
5149 embedded->flags = (pm_node_flags_t) ((embedded->flags | PM_NODE_FLAG_STATIC_LITERAL | PM_STRING_FLAGS_FROZEN) & ~PM_STRING_FLAGS_MUTABLE);
5150
5151 if (PM_NODE_FLAG_P(node, PM_NODE_FLAG_STATIC_LITERAL)) {
5152 MUTABLE_FLAGS(node);
5153 }
5154 } else if (PM_NODE_TYPE_P(embedded, PM_INTERPOLATED_STRING_NODE) && PM_NODE_FLAG_P(embedded, PM_NODE_FLAG_STATIC_LITERAL)) {
5155 // If the embedded statement is an interpolated string, but that
5156 // string is marked as static literal, then we can keep our
5157 // static literal status for this string.
5158 if (PM_NODE_FLAG_P(node, PM_NODE_FLAG_STATIC_LITERAL)) {
5159 MUTABLE_FLAGS(node);
5160 }
5161 } else {
5162 // In all other cases, we lose the static literal flag here and
5163 // become mutable.
5164 CLEAR_FLAGS(node);
5165 }
5166
5167 break;
5168 }
5169 case PM_EMBEDDED_VARIABLE_NODE:
5170 // Embedded variables clear static literal, which means we also
5171 // should clear the mutability flags.
5172 CLEAR_FLAGS(node);
5173 break;
5174 case PM_X_STRING_NODE:
5175 case PM_INTERPOLATED_X_STRING_NODE:
5176 case PM_SYMBOL_NODE:
5177 case PM_INTERPOLATED_SYMBOL_NODE:
5178 // These will only happen in error cases. But we want to handle it
5179 // here so that we don't fail the assertion.
5180 CLEAR_FLAGS(node);
5181 pm_node_list_append(arena, &node->parts, UP(pm_error_recovery_node_create_unexpected(parser, part)));
5182 return;
5183 case PM_ERROR_RECOVERY_NODE:
5184 CLEAR_FLAGS(node);
5185 break;
5186 default:
5187 assert(false && "unexpected node type");
5188 break;
5189 }
5190
5191 pm_node_list_append(arena, &node->parts, part);
5192
5193#undef CLEAR_FLAGS
5194#undef MUTABLE_FLAGS
5195}
5196
5200static pm_interpolated_string_node_t *
5201pm_interpolated_string_node_create(pm_parser_t *parser, const pm_token_t *opening, const pm_node_list_t *parts, const pm_token_t *closing) {
5202 pm_node_flags_t flags = PM_NODE_FLAG_STATIC_LITERAL;
5203
5204 switch (parser->frozen_string_literal) {
5205 case PM_OPTIONS_FROZEN_STRING_LITERAL_DISABLED:
5206 flags |= PM_INTERPOLATED_STRING_NODE_FLAGS_MUTABLE;
5207 break;
5208 case PM_OPTIONS_FROZEN_STRING_LITERAL_ENABLED:
5209 flags |= PM_INTERPOLATED_STRING_NODE_FLAGS_FROZEN;
5210 break;
5211 }
5212
5213 uint32_t start = opening == NULL ? 0 : PM_TOKEN_START(parser, opening);
5214 uint32_t end = closing == NULL ? 0 : PM_TOKEN_END(parser, closing);
5215
5216 pm_interpolated_string_node_t *node = pm_interpolated_string_node_new(
5217 parser->arena,
5218 ++parser->node_id,
5219 flags,
5220 ((pm_location_t) { .start = start, .length = U32(end - start) }),
5221 NTOK2LOC(parser, opening),
5222 ((pm_node_list_t) { 0 }),
5223 NTOK2LOC(parser, closing)
5224 );
5225
5226 if (parts != NULL) {
5227 pm_node_t *part;
5228 PM_NODE_LIST_FOREACH(parts, index, part) {
5229 pm_interpolated_string_node_append(parser, node, part);
5230 }
5231 }
5232
5233 return node;
5234}
5235
5239static void
5240pm_interpolated_string_node_closing_set(const pm_parser_t *parser, pm_interpolated_string_node_t *node, const pm_token_t *closing) {
5241 node->closing_loc = TOK2LOC(parser, closing);
5242 PM_NODE_LENGTH_SET_TOKEN(parser, node, closing);
5243}
5244
5245static void
5246pm_interpolated_symbol_node_append(pm_arena_t *arena, pm_interpolated_symbol_node_t *node, pm_node_t *part) {
5247 if (node->parts.size == 0 && node->opening_loc.length == 0) {
5248 PM_NODE_START_SET_NODE(node, part);
5249 }
5250
5251 pm_interpolated_node_append(arena, UP(node), &node->parts, part);
5252
5253 if (PM_NODE_END(part) > PM_NODE_END(node)) {
5254 PM_NODE_LENGTH_SET_NODE(node, part);
5255 }
5256}
5257
5258static void
5259pm_interpolated_symbol_node_closing_loc_set(const pm_parser_t *parser, pm_interpolated_symbol_node_t *node, const pm_token_t *closing) {
5260 node->closing_loc = TOK2LOC(parser, closing);
5261 PM_NODE_LENGTH_SET_TOKEN(parser, node, closing);
5262}
5263
5267static pm_interpolated_symbol_node_t *
5268pm_interpolated_symbol_node_create(pm_parser_t *parser, const pm_token_t *opening, const pm_node_list_t *parts, const pm_token_t *closing) {
5269 uint32_t start = opening == NULL ? 0 : PM_TOKEN_START(parser, opening);
5270 uint32_t end = closing == NULL ? 0 : PM_TOKEN_END(parser, closing);
5271
5272 pm_interpolated_symbol_node_t *node = pm_interpolated_symbol_node_new(
5273 parser->arena,
5274 ++parser->node_id,
5275 PM_NODE_FLAG_STATIC_LITERAL,
5276 ((pm_location_t) { .start = start, .length = U32(end - start) }),
5277 NTOK2LOC(parser, opening),
5278 ((pm_node_list_t) { 0 }),
5279 NTOK2LOC(parser, closing)
5280 );
5281
5282 if (parts != NULL) {
5283 pm_node_t *part;
5284 PM_NODE_LIST_FOREACH(parts, index, part) {
5285 pm_interpolated_symbol_node_append(parser->arena, node, part);
5286 }
5287 }
5288
5289 return node;
5290}
5291
5295static pm_interpolated_x_string_node_t *
5296pm_interpolated_xstring_node_create(pm_parser_t *parser, const pm_token_t *opening, const pm_token_t *closing) {
5297 return pm_interpolated_x_string_node_new(
5298 parser->arena,
5299 ++parser->node_id,
5300 0,
5301 PM_LOCATION_INIT_TOKENS(parser, opening, closing),
5302 TOK2LOC(parser, opening),
5303 ((pm_node_list_t) { 0 }),
5304 TOK2LOC(parser, closing)
5305 );
5306}
5307
5308static PRISM_INLINE void
5309pm_interpolated_xstring_node_append(pm_arena_t *arena, pm_interpolated_x_string_node_t *node, pm_node_t *part) {
5310 pm_interpolated_node_append(arena, UP(node), &node->parts, part);
5311 PM_NODE_LENGTH_SET_NODE(node, part);
5312}
5313
5314static PRISM_INLINE void
5315pm_interpolated_xstring_node_closing_set(const pm_parser_t *parser, pm_interpolated_x_string_node_t *node, const pm_token_t *closing) {
5316 node->closing_loc = TOK2LOC(parser, closing);
5317 PM_NODE_LENGTH_SET_TOKEN(parser, node, closing);
5318}
5319
5323static pm_it_local_variable_read_node_t *
5324pm_it_local_variable_read_node_create(pm_parser_t *parser, const pm_token_t *name) {
5325 return pm_it_local_variable_read_node_new(
5326 parser->arena,
5327 ++parser->node_id,
5328 0,
5329 PM_LOCATION_INIT_TOKEN(parser, name)
5330 );
5331}
5332
5336static pm_it_parameters_node_t *
5337pm_it_parameters_node_create(pm_parser_t *parser, const pm_token_t *opening, const pm_token_t *closing) {
5338 return pm_it_parameters_node_new(
5339 parser->arena,
5340 ++parser->node_id,
5341 0,
5342 PM_LOCATION_INIT_TOKENS(parser, opening, closing)
5343 );
5344}
5345
5349static pm_keyword_hash_node_t *
5350pm_keyword_hash_node_create(pm_parser_t *parser) {
5351 return pm_keyword_hash_node_new(
5352 parser->arena,
5353 ++parser->node_id,
5354 PM_KEYWORD_HASH_NODE_FLAGS_SYMBOL_KEYS,
5355 PM_LOCATION_INIT_UNSET,
5356 ((pm_node_list_t) { 0 })
5357 );
5358}
5359
5363static void
5364pm_keyword_hash_node_elements_append(pm_arena_t *arena, pm_keyword_hash_node_t *hash, pm_node_t *element) {
5365 // If the element being added is not an AssocNode or does not have a symbol
5366 // key, then we want to turn the SYMBOL_KEYS flag off.
5367 if (!PM_NODE_TYPE_P(element, PM_ASSOC_NODE) || !PM_NODE_TYPE_P(((pm_assoc_node_t *) element)->key, PM_SYMBOL_NODE)) {
5368 pm_node_flag_unset(UP(hash), PM_KEYWORD_HASH_NODE_FLAGS_SYMBOL_KEYS);
5369 }
5370
5371 pm_node_list_append(arena, &hash->elements, element);
5372 if (PM_NODE_LENGTH(hash) == 0) {
5373 PM_NODE_START_SET_NODE(hash, element);
5374 }
5375 PM_NODE_LENGTH_SET_NODE(hash, element);
5376}
5377
5381static pm_required_keyword_parameter_node_t *
5382pm_required_keyword_parameter_node_create(pm_parser_t *parser, const pm_token_t *name) {
5383 return pm_required_keyword_parameter_node_new(
5384 parser->arena,
5385 ++parser->node_id,
5386 0,
5387 PM_LOCATION_INIT_TOKEN(parser, name),
5388 pm_parser_constant_id_raw(parser, name->start, name->end - 1),
5389 TOK2LOC(parser, name)
5390 );
5391}
5392
5396static pm_optional_keyword_parameter_node_t *
5397pm_optional_keyword_parameter_node_create(pm_parser_t *parser, const pm_token_t *name, pm_node_t *value) {
5398 return pm_optional_keyword_parameter_node_new(
5399 parser->arena,
5400 ++parser->node_id,
5401 0,
5402 PM_LOCATION_INIT_TOKEN_NODE(parser, name, value),
5403 pm_parser_constant_id_raw(parser, name->start, name->end - 1),
5404 TOK2LOC(parser, name),
5405 value
5406 );
5407}
5408
5412static pm_keyword_rest_parameter_node_t *
5413pm_keyword_rest_parameter_node_create(pm_parser_t *parser, const pm_token_t *operator, const pm_token_t *name) {
5414 return pm_keyword_rest_parameter_node_new(
5415 parser->arena,
5416 ++parser->node_id,
5417 0,
5418 (name == NULL) ? PM_LOCATION_INIT_TOKEN(parser, operator) : PM_LOCATION_INIT_TOKENS(parser, operator, name),
5419 name == NULL ? 0 : pm_parser_constant_id_token(parser, name),
5420 NTOK2LOC(parser, name),
5421 TOK2LOC(parser, operator)
5422 );
5423}
5424
5428static pm_lambda_node_t *
5429pm_lambda_node_create(
5430 pm_parser_t *parser,
5431 pm_constant_id_list_t *locals,
5432 const pm_token_t *operator,
5433 const pm_token_t *opening,
5434 const pm_token_t *closing,
5435 pm_node_t *parameters,
5436 pm_node_t *body
5437) {
5438 return pm_lambda_node_new(
5439 parser->arena,
5440 ++parser->node_id,
5441 0,
5442 PM_LOCATION_INIT_TOKENS(parser, operator, closing),
5443 *locals,
5444 TOK2LOC(parser, operator),
5445 TOK2LOC(parser, opening),
5446 TOK2LOC(parser, closing),
5447 parameters,
5448 body
5449 );
5450}
5451
5455static pm_local_variable_and_write_node_t *
5456pm_local_variable_and_write_node_create(pm_parser_t *parser, pm_node_t *target, const pm_token_t *operator, pm_node_t *value, pm_constant_id_t name, uint32_t depth) {
5457 assert(PM_NODE_TYPE_P(target, PM_LOCAL_VARIABLE_READ_NODE) || PM_NODE_TYPE_P(target, PM_IT_LOCAL_VARIABLE_READ_NODE) || PM_NODE_TYPE_P(target, PM_CALL_NODE));
5458 assert(operator->type == PM_TOKEN_AMPERSAND_AMPERSAND_EQUAL);
5459
5460 return pm_local_variable_and_write_node_new(
5461 parser->arena,
5462 ++parser->node_id,
5463 0,
5464 PM_LOCATION_INIT_NODES(target, value),
5465 target->location,
5466 TOK2LOC(parser, operator),
5467 value,
5468 name,
5469 depth
5470 );
5471}
5472
5476static pm_local_variable_operator_write_node_t *
5477pm_local_variable_operator_write_node_create(pm_parser_t *parser, pm_node_t *target, const pm_token_t *operator, pm_node_t *value, pm_constant_id_t name, uint32_t depth) {
5478 return pm_local_variable_operator_write_node_new(
5479 parser->arena,
5480 ++parser->node_id,
5481 0,
5482 PM_LOCATION_INIT_NODES(target, value),
5483 target->location,
5484 TOK2LOC(parser, operator),
5485 value,
5486 name,
5487 pm_parser_constant_id_raw(parser, operator->start, operator->end - 1),
5488 depth
5489 );
5490}
5491
5495static pm_local_variable_or_write_node_t *
5496pm_local_variable_or_write_node_create(pm_parser_t *parser, pm_node_t *target, const pm_token_t *operator, pm_node_t *value, pm_constant_id_t name, uint32_t depth) {
5497 assert(PM_NODE_TYPE_P(target, PM_LOCAL_VARIABLE_READ_NODE) || PM_NODE_TYPE_P(target, PM_IT_LOCAL_VARIABLE_READ_NODE) || PM_NODE_TYPE_P(target, PM_CALL_NODE));
5498 assert(operator->type == PM_TOKEN_PIPE_PIPE_EQUAL);
5499
5500 return pm_local_variable_or_write_node_new(
5501 parser->arena,
5502 ++parser->node_id,
5503 0,
5504 PM_LOCATION_INIT_NODES(target, value),
5505 target->location,
5506 TOK2LOC(parser, operator),
5507 value,
5508 name,
5509 depth
5510 );
5511}
5512
5516static pm_local_variable_read_node_t *
5517pm_local_variable_read_node_create_constant_id(pm_parser_t *parser, const pm_token_t *name, pm_constant_id_t name_id, uint32_t depth, bool missing) {
5518 if (!missing) pm_locals_read(&pm_parser_scope_find(parser, depth)->locals, name_id);
5519
5520 return pm_local_variable_read_node_new(
5521 parser->arena,
5522 ++parser->node_id,
5523 0,
5524 PM_LOCATION_INIT_TOKEN(parser, name),
5525 name_id,
5526 depth
5527 );
5528}
5529
5533static pm_local_variable_read_node_t *
5534pm_local_variable_read_node_create(pm_parser_t *parser, const pm_token_t *name, uint32_t depth) {
5535 pm_constant_id_t name_id = pm_parser_constant_id_token(parser, name);
5536 return pm_local_variable_read_node_create_constant_id(parser, name, name_id, depth, false);
5537}
5538
5543static pm_local_variable_read_node_t *
5544pm_local_variable_read_node_missing_create(pm_parser_t *parser, const pm_token_t *name, uint32_t depth) {
5545 pm_constant_id_t name_id = pm_parser_constant_id_token(parser, name);
5546 return pm_local_variable_read_node_create_constant_id(parser, name, name_id, depth, true);
5547}
5548
5552static pm_local_variable_write_node_t *
5553pm_local_variable_write_node_create(pm_parser_t *parser, pm_constant_id_t name, uint32_t depth, pm_node_t *value, const pm_location_t *name_loc, const pm_token_t *operator) {
5554 return pm_local_variable_write_node_new(
5555 parser->arena,
5556 ++parser->node_id,
5557 pm_implicit_array_write_flags(value, PM_WRITE_NODE_FLAGS_IMPLICIT_ARRAY),
5558 ((pm_location_t) { .start = name_loc->start, .length = PM_NODE_END(value) - name_loc->start }),
5559 name,
5560 depth,
5561 *name_loc,
5562 value,
5563 TOK2LOC(parser, operator)
5564 );
5565}
5566
5570static PRISM_INLINE bool
5571pm_token_is_it(const uint8_t *start, const uint8_t *end) {
5572 return (end - start == 2) && (start[0] == 'i') && (start[1] == 't');
5573}
5574
5579static PRISM_INLINE bool
5580pm_token_is_numbered_parameter(const pm_parser_t *parser, uint32_t start, uint32_t length) {
5581 return (
5582 (length == 2) &&
5583 (parser->start[start] == '_') &&
5584 (parser->start[start + 1] != '0') &&
5585 pm_char_is_decimal_digit(parser->start[start + 1])
5586 );
5587}
5588
5593static PRISM_INLINE void
5594pm_refute_numbered_parameter(pm_parser_t *parser, uint32_t start, uint32_t length) {
5595 if (pm_token_is_numbered_parameter(parser, start, length)) {
5596 PM_PARSER_ERR_FORMAT(parser, start, length, PM_ERR_PARAMETER_NUMBERED_RESERVED, parser->start + start);
5597 }
5598}
5599
5604static pm_local_variable_target_node_t *
5605pm_local_variable_target_node_create(pm_parser_t *parser, const pm_location_t *location, pm_constant_id_t name, uint32_t depth) {
5606 pm_refute_numbered_parameter(parser, location->start, location->length);
5607
5608 return pm_local_variable_target_node_new(
5609 parser->arena,
5610 ++parser->node_id,
5611 0,
5612 ((pm_location_t) { .start = location->start, .length = location->length }),
5613 name,
5614 depth
5615 );
5616}
5617
5621static pm_match_predicate_node_t *
5622pm_match_predicate_node_create(pm_parser_t *parser, pm_node_t *value, pm_node_t *pattern, const pm_token_t *operator) {
5623 pm_assert_value_expression(parser, value);
5624
5625 return pm_match_predicate_node_new(
5626 parser->arena,
5627 ++parser->node_id,
5628 0,
5629 PM_LOCATION_INIT_NODES(value, pattern),
5630 value,
5631 pattern,
5632 TOK2LOC(parser, operator)
5633 );
5634}
5635
5639static pm_match_required_node_t *
5640pm_match_required_node_create(pm_parser_t *parser, pm_node_t *value, pm_node_t *pattern, const pm_token_t *operator) {
5641 pm_assert_value_expression(parser, value);
5642
5643 return pm_match_required_node_new(
5644 parser->arena,
5645 ++parser->node_id,
5646 0,
5647 PM_LOCATION_INIT_NODES(value, pattern),
5648 value,
5649 pattern,
5650 TOK2LOC(parser, operator)
5651 );
5652}
5653
5657static pm_match_write_node_t *
5658pm_match_write_node_create(pm_parser_t *parser, pm_call_node_t *call) {
5659 return pm_match_write_node_new(
5660 parser->arena,
5661 ++parser->node_id,
5662 0,
5663 PM_LOCATION_INIT_NODE(call),
5664 call,
5665 ((pm_node_list_t) { 0 })
5666 );
5667}
5668
5672static pm_module_node_t *
5673pm_module_node_create(pm_parser_t *parser, pm_constant_id_list_t *locals, const pm_token_t *module_keyword, pm_node_t *constant_path, const pm_token_t *name, pm_node_t *body, const pm_token_t *end_keyword) {
5674 pm_constant_id_list_t module_locals = { .ids = NULL, .size = 0, .capacity = 0 };
5675 if (locals != NULL) module_locals = *locals;
5676
5677 return pm_module_node_new(
5678 parser->arena,
5679 ++parser->node_id,
5680 0,
5681 PM_LOCATION_INIT_TOKENS(parser, module_keyword, end_keyword),
5682 module_locals,
5683 TOK2LOC(parser, module_keyword),
5684 constant_path,
5685 body,
5686 TOK2LOC(parser, end_keyword),
5687 pm_parser_constant_id_token(parser, name)
5688 );
5689}
5690
5694static pm_multi_target_node_t *
5695pm_multi_target_node_create(pm_parser_t *parser) {
5696 return pm_multi_target_node_new(
5697 parser->arena,
5698 ++parser->node_id,
5699 0,
5700 PM_LOCATION_INIT_UNSET,
5701 ((pm_node_list_t) { 0 }),
5702 NULL,
5703 ((pm_node_list_t) { 0 }),
5704 ((pm_location_t) { 0 }),
5705 ((pm_location_t) { 0 })
5706 );
5707}
5708
5712static void
5713pm_multi_target_node_targets_append(pm_parser_t *parser, pm_multi_target_node_t *node, pm_node_t *target) {
5714 if (PM_NODE_TYPE_P(target, PM_SPLAT_NODE)) {
5715 if (node->rest == NULL) {
5716 node->rest = target;
5717 } else {
5718 pm_parser_err_node(parser, target, PM_ERR_MULTI_ASSIGN_MULTI_SPLATS);
5719 pm_node_list_append(parser->arena, &node->rights, target);
5720 }
5721 } else if (PM_NODE_TYPE_P(target, PM_IMPLICIT_REST_NODE)) {
5722 if (node->rest == NULL) {
5723 node->rest = target;
5724 } else {
5725 PM_PARSER_ERR_TOKEN_FORMAT_CONTENT(parser, &parser->current, PM_ERR_MULTI_ASSIGN_UNEXPECTED_REST);
5726 pm_node_list_append(parser->arena, &node->rights, target);
5727 }
5728 } else if (node->rest == NULL) {
5729 pm_node_list_append(parser->arena, &node->lefts, target);
5730 } else {
5731 pm_node_list_append(parser->arena, &node->rights, target);
5732 }
5733
5734 if (PM_NODE_LENGTH(node) == 0 || (PM_NODE_START(node) > PM_NODE_START(target))) {
5735 PM_NODE_START_SET_NODE(node, target);
5736 }
5737
5738 if (PM_NODE_LENGTH(node) == 0 || (PM_NODE_END(node) < PM_NODE_END(target))) {
5739 PM_NODE_LENGTH_SET_NODE(node, target);
5740 }
5741}
5742
5746static void
5747pm_multi_target_node_opening_set(const pm_parser_t *parser, pm_multi_target_node_t *node, const pm_token_t *lparen) {
5748 PM_NODE_START_SET_TOKEN(parser, node, lparen);
5749 PM_NODE_LENGTH_SET_TOKEN(parser, node, lparen);
5750 node->lparen_loc = TOK2LOC(parser, lparen);
5751}
5752
5756static void
5757pm_multi_target_node_closing_set(const pm_parser_t *parser, pm_multi_target_node_t *node, const pm_token_t *rparen) {
5758 PM_NODE_LENGTH_SET_TOKEN(parser, node, rparen);
5759 node->rparen_loc = TOK2LOC(parser, rparen);
5760}
5761
5765static pm_multi_write_node_t *
5766pm_multi_write_node_create(pm_parser_t *parser, pm_multi_target_node_t *target, const pm_token_t *operator, pm_node_t *value) {
5767 /* The target is no longer necessary because we have reused its children. It
5768 * is arena-allocated so no explicit free is needed. */
5769 return pm_multi_write_node_new(
5770 parser->arena,
5771 ++parser->node_id,
5772 pm_implicit_array_write_flags(value, PM_WRITE_NODE_FLAGS_IMPLICIT_ARRAY),
5773 PM_LOCATION_INIT_NODES(target, value),
5774 target->lefts,
5775 target->rest,
5776 target->rights,
5777 target->lparen_loc,
5778 target->rparen_loc,
5779 TOK2LOC(parser, operator),
5780 value
5781 );
5782}
5783
5787static pm_next_node_t *
5788pm_next_node_create(pm_parser_t *parser, const pm_token_t *keyword, pm_arguments_node_t *arguments) {
5789 assert(keyword->type == PM_TOKEN_KEYWORD_NEXT);
5790
5791 return pm_next_node_new(
5792 parser->arena,
5793 ++parser->node_id,
5794 0,
5795 (arguments == NULL) ? PM_LOCATION_INIT_TOKEN(parser, keyword) : PM_LOCATION_INIT_TOKEN_NODE(parser, keyword, arguments),
5796 arguments,
5797 TOK2LOC(parser, keyword)
5798 );
5799}
5800
5804static pm_nil_node_t *
5805pm_nil_node_create(pm_parser_t *parser, const pm_token_t *token) {
5806 assert(token->type == PM_TOKEN_KEYWORD_NIL);
5807
5808 return pm_nil_node_new(
5809 parser->arena,
5810 ++parser->node_id,
5811 PM_NODE_FLAG_STATIC_LITERAL,
5812 PM_LOCATION_INIT_TOKEN(parser, token)
5813 );
5814}
5815
5819static pm_no_block_parameter_node_t *
5820pm_no_block_parameter_node_create(pm_parser_t *parser, const pm_token_t *operator, const pm_token_t *keyword) {
5821 assert(operator->type == PM_TOKEN_AMPERSAND || operator->type == PM_TOKEN_UAMPERSAND);
5822 assert(keyword->type == PM_TOKEN_KEYWORD_NIL);
5823
5824 return pm_no_block_parameter_node_new(
5825 parser->arena,
5826 ++parser->node_id,
5827 0,
5828 PM_LOCATION_INIT_TOKENS(parser, operator, keyword),
5829 TOK2LOC(parser, operator),
5830 TOK2LOC(parser, keyword)
5831 );
5832}
5833
5837static pm_no_keywords_parameter_node_t *
5838pm_no_keywords_parameter_node_create(pm_parser_t *parser, const pm_token_t *operator, const pm_token_t *keyword) {
5839 assert(operator->type == PM_TOKEN_USTAR_STAR || operator->type == PM_TOKEN_STAR_STAR);
5840 assert(keyword->type == PM_TOKEN_KEYWORD_NIL);
5841
5842 return pm_no_keywords_parameter_node_new(
5843 parser->arena,
5844 ++parser->node_id,
5845 0,
5846 PM_LOCATION_INIT_TOKENS(parser, operator, keyword),
5847 TOK2LOC(parser, operator),
5848 TOK2LOC(parser, keyword)
5849 );
5850}
5851
5855static pm_numbered_parameters_node_t *
5856pm_numbered_parameters_node_create(pm_parser_t *parser, const pm_token_t *opening, const pm_token_t *closing, uint8_t maximum) {
5857 return pm_numbered_parameters_node_new(
5858 parser->arena,
5859 ++parser->node_id,
5860 0,
5861 PM_LOCATION_INIT_TOKENS(parser, opening, closing),
5862 maximum
5863 );
5864}
5865
5870#define NTH_REF_MAX ((uint32_t) (INT_MAX >> 1))
5871
5878static uint32_t
5879pm_numbered_reference_read_node_number(pm_parser_t *parser, const pm_token_t *token) {
5880 const uint8_t *start = token->start + 1;
5881 const uint8_t *end = token->end;
5882
5883 ptrdiff_t diff = end - start;
5884 assert(diff > 0);
5885#if PTRDIFF_MAX > SIZE_MAX
5886 assert(diff < (ptrdiff_t) SIZE_MAX);
5887#endif
5888 size_t length = (size_t) diff;
5889
5890 char *digits = xcalloc(length + 1, sizeof(char));
5891 memcpy(digits, start, length);
5892 digits[length] = '\0';
5893
5894 char *endptr;
5895 errno = 0;
5896 unsigned long value = strtoul(digits, &endptr, 10);
5897
5898 if ((digits == endptr) || (*endptr != '\0')) {
5899 pm_parser_err(parser, U32(start - parser->start), U32(length), PM_ERR_INVALID_NUMBER_DECIMAL);
5900 value = 0;
5901 }
5902
5903 xfree_sized(digits, sizeof(char) * (length + 1));
5904
5905 if ((errno == ERANGE) || (value > NTH_REF_MAX)) {
5906 PM_PARSER_WARN_FORMAT(parser, U32(start - parser->start), U32(length), PM_WARN_INVALID_NUMBERED_REFERENCE, (int) (length + 1), (const char *) token->start);
5907 value = 0;
5908 }
5909
5910 return (uint32_t) value;
5911}
5912
5913#undef NTH_REF_MAX
5914
5918static pm_numbered_reference_read_node_t *
5919pm_numbered_reference_read_node_create(pm_parser_t *parser, const pm_token_t *name) {
5920 assert(name->type == PM_TOKEN_NUMBERED_REFERENCE);
5921
5922 return pm_numbered_reference_read_node_new(
5923 parser->arena,
5924 ++parser->node_id,
5925 0,
5926 PM_LOCATION_INIT_TOKEN(parser, name),
5927 pm_numbered_reference_read_node_number(parser, name)
5928 );
5929}
5930
5934static pm_optional_parameter_node_t *
5935pm_optional_parameter_node_create(pm_parser_t *parser, const pm_token_t *name, const pm_token_t *operator, pm_node_t *value) {
5936 return pm_optional_parameter_node_new(
5937 parser->arena,
5938 ++parser->node_id,
5939 0,
5940 PM_LOCATION_INIT_TOKEN_NODE(parser, name, value),
5941 pm_parser_constant_id_token(parser, name),
5942 TOK2LOC(parser, name),
5943 TOK2LOC(parser, operator),
5944 value
5945 );
5946}
5947
5951static pm_or_node_t *
5952pm_or_node_create(pm_parser_t *parser, pm_node_t *left, const pm_token_t *operator, pm_node_t *right) {
5953 pm_assert_value_expression(parser, left);
5954
5955 return pm_or_node_new(
5956 parser->arena,
5957 ++parser->node_id,
5958 0,
5959 PM_LOCATION_INIT_NODES(left, right),
5960 left,
5961 right,
5962 TOK2LOC(parser, operator)
5963 );
5964}
5965
5969static pm_parameters_node_t *
5970pm_parameters_node_create(pm_parser_t *parser) {
5971 return pm_parameters_node_new(
5972 parser->arena,
5973 ++parser->node_id,
5974 0,
5975 PM_LOCATION_INIT_UNSET,
5976 ((pm_node_list_t) { 0 }),
5977 ((pm_node_list_t) { 0 }),
5978 NULL,
5979 ((pm_node_list_t) { 0 }),
5980 ((pm_node_list_t) { 0 }),
5981 NULL,
5982 NULL
5983 );
5984}
5985
5989static void
5990pm_parameters_node_location_set(pm_parameters_node_t *params, pm_node_t *param) {
5991 if ((params->base.location.length == 0) || PM_NODE_START(params) > PM_NODE_START(param)) {
5992 PM_NODE_START_SET_NODE(params, param);
5993 }
5994
5995 if ((params->base.location.length == 0) || (PM_NODE_END(params) < PM_NODE_END(param))) {
5996 PM_NODE_LENGTH_SET_NODE(params, param);
5997 }
5998}
5999
6003static void
6004pm_parameters_node_requireds_append(pm_arena_t *arena, pm_parameters_node_t *params, pm_node_t *param) {
6005 pm_parameters_node_location_set(params, param);
6006 pm_node_list_append(arena, &params->requireds, param);
6007}
6008
6012static void
6013pm_parameters_node_optionals_append(pm_arena_t *arena, pm_parameters_node_t *params, pm_optional_parameter_node_t *param) {
6014 pm_parameters_node_location_set(params, UP(param));
6015 pm_node_list_append(arena, &params->optionals, UP(param));
6016}
6017
6021static void
6022pm_parameters_node_posts_append(pm_arena_t *arena, pm_parameters_node_t *params, pm_node_t *param) {
6023 pm_parameters_node_location_set(params, param);
6024 pm_node_list_append(arena, &params->posts, param);
6025}
6026
6030static void
6031pm_parameters_node_rest_set(pm_parameters_node_t *params, pm_node_t *param) {
6032 pm_parameters_node_location_set(params, param);
6033 params->rest = param;
6034}
6035
6039static void
6040pm_parameters_node_keywords_append(pm_arena_t *arena, pm_parameters_node_t *params, pm_node_t *param) {
6041 pm_parameters_node_location_set(params, param);
6042 pm_node_list_append(arena, &params->keywords, param);
6043}
6044
6048static void
6049pm_parameters_node_keyword_rest_set(pm_parameters_node_t *params, pm_node_t *param) {
6050 assert(params->keyword_rest == NULL);
6051 pm_parameters_node_location_set(params, param);
6052 params->keyword_rest = param;
6053}
6054
6058static void
6059pm_parameters_node_block_set(pm_parameters_node_t *params, pm_node_t *param) {
6060 assert(params->block == NULL);
6061 pm_parameters_node_location_set(params, param);
6062 params->block = param;
6063}
6064
6068static pm_program_node_t *
6069pm_program_node_create(pm_parser_t *parser, pm_constant_id_list_t *locals, pm_statements_node_t *statements) {
6070 return pm_program_node_new(
6071 parser->arena,
6072 ++parser->node_id,
6073 0,
6074 PM_LOCATION_INIT_NODE(statements),
6075 *locals,
6076 statements
6077 );
6078}
6079
6083static pm_parentheses_node_t *
6084pm_parentheses_node_create(pm_parser_t *parser, const pm_token_t *opening, pm_node_t *body, const pm_token_t *closing, pm_node_flags_t flags) {
6085 return pm_parentheses_node_new(
6086 parser->arena,
6087 ++parser->node_id,
6088 flags,
6089 PM_LOCATION_INIT_TOKENS(parser, opening, closing),
6090 body,
6091 TOK2LOC(parser, opening),
6092 TOK2LOC(parser, closing)
6093 );
6094}
6095
6099static pm_pinned_expression_node_t *
6100pm_pinned_expression_node_create(pm_parser_t *parser, pm_node_t *expression, const pm_token_t *operator, const pm_token_t *lparen, const pm_token_t *rparen) {
6101 return pm_pinned_expression_node_new(
6102 parser->arena,
6103 ++parser->node_id,
6104 0,
6105 PM_LOCATION_INIT_TOKENS(parser, operator, rparen),
6106 expression,
6107 TOK2LOC(parser, operator),
6108 TOK2LOC(parser, lparen),
6109 TOK2LOC(parser, rparen)
6110 );
6111}
6112
6116static pm_pinned_variable_node_t *
6117pm_pinned_variable_node_create(pm_parser_t *parser, const pm_token_t *operator, pm_node_t *variable) {
6118 return pm_pinned_variable_node_new(
6119 parser->arena,
6120 ++parser->node_id,
6121 0,
6122 PM_LOCATION_INIT_TOKEN_NODE(parser, operator, variable),
6123 variable,
6124 TOK2LOC(parser, operator)
6125 );
6126}
6127
6131static pm_post_execution_node_t *
6132pm_post_execution_node_create(pm_parser_t *parser, const pm_token_t *keyword, const pm_token_t *opening, pm_statements_node_t *statements, const pm_token_t *closing) {
6133 return pm_post_execution_node_new(
6134 parser->arena,
6135 ++parser->node_id,
6136 0,
6137 PM_LOCATION_INIT_TOKENS(parser, keyword, closing),
6138 statements,
6139 TOK2LOC(parser, keyword),
6140 TOK2LOC(parser, opening),
6141 TOK2LOC(parser, closing)
6142 );
6143}
6144
6148static pm_pre_execution_node_t *
6149pm_pre_execution_node_create(pm_parser_t *parser, const pm_token_t *keyword, const pm_token_t *opening, pm_statements_node_t *statements, const pm_token_t *closing) {
6150 return pm_pre_execution_node_new(
6151 parser->arena,
6152 ++parser->node_id,
6153 0,
6154 PM_LOCATION_INIT_TOKENS(parser, keyword, closing),
6155 statements,
6156 TOK2LOC(parser, keyword),
6157 TOK2LOC(parser, opening),
6158 TOK2LOC(parser, closing)
6159 );
6160}
6161
6165static pm_range_node_t *
6166pm_range_node_create(pm_parser_t *parser, pm_node_t *left, const pm_token_t *operator, pm_node_t *right) {
6167 pm_assert_value_expression(parser, left);
6168 pm_assert_value_expression(parser, right);
6169 pm_node_flags_t flags = 0;
6170
6171 // Indicate that this node is an exclusive range if the operator is `...`.
6172 if (operator->type == PM_TOKEN_DOT_DOT_DOT || operator->type == PM_TOKEN_UDOT_DOT_DOT) {
6173 flags |= PM_RANGE_FLAGS_EXCLUDE_END;
6174 }
6175
6176 // Indicate that this node is a static literal (i.e., can be compiled with
6177 // a putobject in CRuby) if the left and right are implicit nil, explicit
6178 // nil, or integers.
6179 if (
6180 (left == NULL || PM_NODE_TYPE_P(left, PM_NIL_NODE) || PM_NODE_TYPE_P(left, PM_INTEGER_NODE)) &&
6181 (right == NULL || PM_NODE_TYPE_P(right, PM_NIL_NODE) || PM_NODE_TYPE_P(right, PM_INTEGER_NODE))
6182 ) {
6183 flags |= PM_NODE_FLAG_STATIC_LITERAL;
6184 }
6185
6186 uint32_t start = left == NULL ? PM_TOKEN_START(parser, operator) : PM_NODE_START(left);
6187 uint32_t end = right == NULL ? PM_TOKEN_END(parser, operator) : PM_NODE_END(right);
6188
6189 return pm_range_node_new(
6190 parser->arena,
6191 ++parser->node_id,
6192 flags,
6193 ((pm_location_t) { .start = start, .length = U32(end - start) }),
6194 left,
6195 right,
6196 TOK2LOC(parser, operator)
6197 );
6198}
6199
6203static pm_redo_node_t *
6204pm_redo_node_create(pm_parser_t *parser, const pm_token_t *token) {
6205 assert(token->type == PM_TOKEN_KEYWORD_REDO);
6206
6207 return pm_redo_node_new(
6208 parser->arena,
6209 ++parser->node_id,
6210 0,
6211 PM_LOCATION_INIT_TOKEN(parser, token)
6212 );
6213}
6214
6219static pm_regular_expression_node_t *
6220pm_regular_expression_node_create_unescaped(pm_parser_t *parser, const pm_token_t *opening, const pm_token_t *content, const pm_token_t *closing, const pm_string_t *unescaped) {
6221 return pm_regular_expression_node_new(
6222 parser->arena,
6223 ++parser->node_id,
6224 pm_regular_expression_flags_create(parser, closing) | PM_NODE_FLAG_STATIC_LITERAL,
6225 PM_LOCATION_INIT_TOKENS(parser, opening, closing),
6226 TOK2LOC(parser, opening),
6227 TOK2LOC(parser, content),
6228 TOK2LOC(parser, closing),
6229 *unescaped
6230 );
6231}
6232
6236static PRISM_INLINE pm_regular_expression_node_t *
6237pm_regular_expression_node_create(pm_parser_t *parser, const pm_token_t *opening, const pm_token_t *content, const pm_token_t *closing) {
6238 return pm_regular_expression_node_create_unescaped(parser, opening, content, closing, &PM_STRING_EMPTY);
6239}
6240
6244static pm_required_parameter_node_t *
6245pm_required_parameter_node_create(pm_parser_t *parser, const pm_token_t *token) {
6246 return pm_required_parameter_node_new(
6247 parser->arena,
6248 ++parser->node_id,
6249 0,
6250 PM_LOCATION_INIT_TOKEN(parser, token),
6251 pm_parser_constant_id_token(parser, token)
6252 );
6253}
6254
6258static pm_rescue_modifier_node_t *
6259pm_rescue_modifier_node_create(pm_parser_t *parser, pm_node_t *expression, const pm_token_t *keyword, pm_node_t *rescue_expression) {
6260 return pm_rescue_modifier_node_new(
6261 parser->arena,
6262 ++parser->node_id,
6263 0,
6264 PM_LOCATION_INIT_NODES(expression, rescue_expression),
6265 expression,
6266 TOK2LOC(parser, keyword),
6267 rescue_expression
6268 );
6269}
6270
6274static pm_rescue_node_t *
6275pm_rescue_node_create(pm_parser_t *parser, const pm_token_t *keyword) {
6276 return pm_rescue_node_new(
6277 parser->arena,
6278 ++parser->node_id,
6279 0,
6280 PM_LOCATION_INIT_TOKEN(parser, keyword),
6281 TOK2LOC(parser, keyword),
6282 ((pm_node_list_t) { 0 }),
6283 ((pm_location_t) { 0 }),
6284 NULL,
6285 ((pm_location_t) { 0 }),
6286 NULL,
6287 NULL
6288 );
6289}
6290
6291static PRISM_INLINE void
6292pm_rescue_node_operator_set(const pm_parser_t *parser, pm_rescue_node_t *node, const pm_token_t *operator) {
6293 node->operator_loc = TOK2LOC(parser, operator);
6294}
6295
6299static void
6300pm_rescue_node_reference_set(pm_rescue_node_t *node, pm_node_t *reference) {
6301 node->reference = reference;
6302 PM_NODE_LENGTH_SET_NODE(node, reference);
6303}
6304
6308static void
6309pm_rescue_node_statements_set(pm_rescue_node_t *node, pm_statements_node_t *statements) {
6310 node->statements = statements;
6311 if (pm_statements_node_body_length(statements) > 0) {
6312 PM_NODE_LENGTH_SET_NODE(node, statements);
6313 }
6314}
6315
6319static void
6320pm_rescue_node_subsequent_set(pm_rescue_node_t *node, pm_rescue_node_t *subsequent) {
6321 node->subsequent = subsequent;
6322 PM_NODE_LENGTH_SET_NODE(node, subsequent);
6323}
6324
6328static void
6329pm_rescue_node_exceptions_append(pm_arena_t *arena, pm_rescue_node_t *node, pm_node_t *exception) {
6330 pm_node_list_append(arena, &node->exceptions, exception);
6331 PM_NODE_LENGTH_SET_NODE(node, exception);
6332}
6333
6337static pm_rest_parameter_node_t *
6338pm_rest_parameter_node_create(pm_parser_t *parser, const pm_token_t *operator, const pm_token_t *name) {
6339 return pm_rest_parameter_node_new(
6340 parser->arena,
6341 ++parser->node_id,
6342 0,
6343 (name == NULL) ? PM_LOCATION_INIT_TOKEN(parser, operator) : PM_LOCATION_INIT_TOKENS(parser, operator, name),
6344 name == NULL ? 0 : pm_parser_constant_id_token(parser, name),
6345 NTOK2LOC(parser, name),
6346 TOK2LOC(parser, operator)
6347 );
6348}
6349
6353static pm_retry_node_t *
6354pm_retry_node_create(pm_parser_t *parser, const pm_token_t *token) {
6355 assert(token->type == PM_TOKEN_KEYWORD_RETRY);
6356
6357 return pm_retry_node_new(
6358 parser->arena,
6359 ++parser->node_id,
6360 0,
6361 PM_LOCATION_INIT_TOKEN(parser, token)
6362 );
6363}
6364
6368static pm_return_node_t *
6369pm_return_node_create(pm_parser_t *parser, const pm_token_t *keyword, pm_arguments_node_t *arguments) {
6370 return pm_return_node_new(
6371 parser->arena,
6372 ++parser->node_id,
6373 0,
6374 (arguments == NULL) ? PM_LOCATION_INIT_TOKEN(parser, keyword) : PM_LOCATION_INIT_TOKEN_NODE(parser, keyword, arguments),
6375 TOK2LOC(parser, keyword),
6376 arguments
6377 );
6378}
6379
6383static pm_self_node_t *
6384pm_self_node_create(pm_parser_t *parser, const pm_token_t *token) {
6385 assert(token->type == PM_TOKEN_KEYWORD_SELF);
6386
6387 return pm_self_node_new(
6388 parser->arena,
6389 ++parser->node_id,
6390 0,
6391 PM_LOCATION_INIT_TOKEN(parser, token)
6392 );
6393}
6394
6398static pm_shareable_constant_node_t *
6399pm_shareable_constant_node_create(pm_parser_t *parser, pm_node_t *write, pm_shareable_constant_value_t value) {
6400 return pm_shareable_constant_node_new(
6401 parser->arena,
6402 ++parser->node_id,
6403 (pm_node_flags_t) value,
6404 PM_LOCATION_INIT_NODE(write),
6405 write
6406 );
6407}
6408
6412static pm_singleton_class_node_t *
6413pm_singleton_class_node_create(pm_parser_t *parser, pm_constant_id_list_t *locals, const pm_token_t *class_keyword, const pm_token_t *operator, pm_node_t *expression, pm_node_t *body, const pm_token_t *end_keyword) {
6414 return pm_singleton_class_node_new(
6415 parser->arena,
6416 ++parser->node_id,
6417 0,
6418 PM_LOCATION_INIT_TOKENS(parser, class_keyword, end_keyword),
6419 *locals,
6420 TOK2LOC(parser, class_keyword),
6421 TOK2LOC(parser, operator),
6422 expression,
6423 body,
6424 TOK2LOC(parser, end_keyword)
6425 );
6426}
6427
6431static pm_source_encoding_node_t *
6432pm_source_encoding_node_create(pm_parser_t *parser, const pm_token_t *token) {
6433 assert(token->type == PM_TOKEN_KEYWORD___ENCODING__);
6434
6435 return pm_source_encoding_node_new(
6436 parser->arena,
6437 ++parser->node_id,
6438 PM_NODE_FLAG_STATIC_LITERAL,
6439 PM_LOCATION_INIT_TOKEN(parser, token)
6440 );
6441}
6442
6446static pm_source_file_node_t*
6447pm_source_file_node_create(pm_parser_t *parser, const pm_token_t *file_keyword) {
6448 assert(file_keyword->type == PM_TOKEN_KEYWORD___FILE__);
6449
6450 pm_node_flags_t flags = 0;
6451
6452 switch (parser->frozen_string_literal) {
6453 case PM_OPTIONS_FROZEN_STRING_LITERAL_DISABLED:
6454 flags |= PM_STRING_FLAGS_MUTABLE;
6455 break;
6456 case PM_OPTIONS_FROZEN_STRING_LITERAL_ENABLED:
6457 flags |= PM_STRING_FLAGS_FROZEN;
6458 if (parser->version < PM_OPTIONS_VERSION_CRUBY_4_1) {
6459 flags |= PM_NODE_FLAG_STATIC_LITERAL;
6460 }
6461 break;
6462 }
6463
6464 return pm_source_file_node_new(
6465 parser->arena,
6466 ++parser->node_id,
6467 flags,
6468 PM_LOCATION_INIT_TOKEN(parser, file_keyword),
6469 parser->filepath
6470 );
6471}
6472
6476static pm_source_line_node_t *
6477pm_source_line_node_create(pm_parser_t *parser, const pm_token_t *token) {
6478 assert(token->type == PM_TOKEN_KEYWORD___LINE__);
6479
6480 return pm_source_line_node_new(
6481 parser->arena,
6482 ++parser->node_id,
6483 PM_NODE_FLAG_STATIC_LITERAL,
6484 PM_LOCATION_INIT_TOKEN(parser, token)
6485 );
6486}
6487
6491static pm_splat_node_t *
6492pm_splat_node_create(pm_parser_t *parser, const pm_token_t *operator, pm_node_t *expression) {
6493 return pm_splat_node_new(
6494 parser->arena,
6495 ++parser->node_id,
6496 0,
6497 (expression == NULL) ? PM_LOCATION_INIT_TOKEN(parser, operator) : PM_LOCATION_INIT_TOKEN_NODE(parser, operator, expression),
6498 TOK2LOC(parser, operator),
6499 expression
6500 );
6501}
6502
6506static pm_statements_node_t *
6507pm_statements_node_create(pm_parser_t *parser) {
6508 return pm_statements_node_new(
6509 parser->arena,
6510 ++parser->node_id,
6511 0,
6512 PM_LOCATION_INIT_UNSET,
6513 ((pm_node_list_t) { 0 })
6514 );
6515}
6516
6520static size_t
6521pm_statements_node_body_length(pm_statements_node_t *node) {
6522 return node && node->body.size;
6523}
6524
6529static PRISM_INLINE void
6530pm_statements_node_body_update(pm_statements_node_t *node, pm_node_t *statement) {
6531 if (pm_statements_node_body_length(node) == 0 || PM_NODE_START(statement) < PM_NODE_START(node)) {
6532 PM_NODE_START_SET_NODE(node, statement);
6533 }
6534
6535 if (PM_NODE_END(statement) > PM_NODE_END(node)) {
6536 PM_NODE_LENGTH_SET_NODE(node, statement);
6537 }
6538}
6539
6543static void
6544pm_statements_node_body_append(pm_parser_t *parser, pm_statements_node_t *node, pm_node_t *statement, bool newline) {
6545 pm_statements_node_body_update(node, statement);
6546
6547 if (node->body.size > 0) {
6548 const pm_node_t *previous = node->body.nodes[node->body.size - 1];
6549
6550 switch (PM_NODE_TYPE(previous)) {
6551 case PM_BREAK_NODE:
6552 case PM_NEXT_NODE:
6553 case PM_REDO_NODE:
6554 case PM_RETRY_NODE:
6555 case PM_RETURN_NODE:
6556 pm_parser_warn_node(parser, statement, PM_WARN_UNREACHABLE_STATEMENT);
6557 break;
6558 default:
6559 break;
6560 }
6561 }
6562
6563 pm_node_list_append(parser->arena, &node->body, statement);
6564 if (newline) pm_node_flag_set(statement, PM_NODE_FLAG_NEWLINE);
6565}
6566
6570static void
6571pm_statements_node_body_prepend(pm_arena_t *arena, pm_statements_node_t *node, pm_node_t *statement) {
6572 pm_statements_node_body_update(node, statement);
6573 pm_node_list_prepend(arena, &node->body, statement);
6574 pm_node_flag_set(statement, PM_NODE_FLAG_NEWLINE);
6575}
6576
6580static PRISM_INLINE pm_string_node_t *
6581pm_string_node_create_unescaped(pm_parser_t *parser, const pm_token_t *opening, const pm_token_t *content, const pm_token_t *closing, const pm_string_t *string) {
6582 pm_node_flags_t flags = 0;
6583
6584 switch (parser->frozen_string_literal) {
6585 case PM_OPTIONS_FROZEN_STRING_LITERAL_DISABLED:
6586 flags = PM_STRING_FLAGS_MUTABLE;
6587 break;
6588 case PM_OPTIONS_FROZEN_STRING_LITERAL_ENABLED:
6589 flags = PM_NODE_FLAG_STATIC_LITERAL | PM_STRING_FLAGS_FROZEN;
6590 break;
6591 }
6592
6593 uint32_t start = PM_TOKEN_START(parser, opening == NULL ? content : opening);
6594 uint32_t end = PM_TOKEN_END(parser, closing == NULL ? content : closing);
6595
6596 return pm_string_node_new(
6597 parser->arena,
6598 ++parser->node_id,
6599 flags,
6600 ((pm_location_t) { .start = start, .length = U32(end - start) }),
6601 NTOK2LOC(parser, opening),
6602 TOK2LOC(parser, content),
6603 NTOK2LOC(parser, closing),
6604 *string
6605 );
6606}
6607
6611static pm_string_node_t *
6612pm_string_node_create(pm_parser_t *parser, const pm_token_t *opening, const pm_token_t *content, const pm_token_t *closing) {
6613 return pm_string_node_create_unescaped(parser, opening, content, closing, &PM_STRING_EMPTY);
6614}
6615
6620static pm_string_node_t *
6621pm_string_node_create_current_string(pm_parser_t *parser, const pm_token_t *opening, const pm_token_t *content, const pm_token_t *closing) {
6622 pm_string_node_t *node = pm_string_node_create_unescaped(parser, opening, content, closing, &parser->current_string);
6623 parser->current_string = PM_STRING_EMPTY;
6624 return node;
6625}
6626
6630static pm_super_node_t *
6631pm_super_node_create(pm_parser_t *parser, const pm_token_t *keyword, pm_arguments_t *arguments) {
6632 assert(keyword->type == PM_TOKEN_KEYWORD_SUPER);
6633
6634 const pm_location_t *end = pm_arguments_end(arguments);
6635 assert(end != NULL && "unreachable");
6636
6637 return pm_super_node_new(
6638 parser->arena,
6639 ++parser->node_id,
6640 0,
6641 ((pm_location_t) { .start = PM_TOKEN_START(parser, keyword), .length = PM_LOCATION_END(end) - PM_TOKEN_START(parser, keyword) }),
6642 TOK2LOC(parser, keyword),
6643 arguments->opening_loc,
6644 arguments->arguments,
6645 arguments->closing_loc,
6646 arguments->block
6647 );
6648}
6649
6654static bool
6655pm_ascii_only_p(const pm_string_t *contents) {
6656 const size_t length = pm_string_length(contents);
6657 const uint8_t *source = pm_string_source(contents);
6658
6659 for (size_t index = 0; index < length; index++) {
6660 if (source[index] & 0x80) return false;
6661 }
6662
6663 return true;
6664}
6665
6669static void
6670parse_symbol_encoding_validate_utf8(pm_parser_t *parser, const pm_token_t *location, const pm_string_t *contents) {
6671 for (const uint8_t *cursor = pm_string_source(contents), *end = cursor + pm_string_length(contents); cursor < end;) {
6672 size_t width = pm_encoding_utf_8_char_width(cursor, end - cursor);
6673
6674 if (width == 0) {
6675 pm_parser_err(parser, PM_TOKEN_START(parser, location), PM_TOKEN_LENGTH(location), PM_ERR_INVALID_SYMBOL);
6676 break;
6677 }
6678
6679 cursor += width;
6680 }
6681}
6682
6687static void
6688parse_symbol_encoding_validate_other(pm_parser_t *parser, const pm_token_t *location, const pm_string_t *contents) {
6689 const pm_encoding_t *encoding = parser->encoding;
6690
6691 for (const uint8_t *cursor = pm_string_source(contents), *end = cursor + pm_string_length(contents); cursor < end;) {
6692 size_t width = encoding->char_width(cursor, end - cursor);
6693
6694 if (width == 0) {
6695 pm_parser_err(parser, PM_TOKEN_START(parser, location), PM_TOKEN_LENGTH(location), PM_ERR_INVALID_SYMBOL);
6696 break;
6697 }
6698
6699 cursor += width;
6700 }
6701}
6702
6712static PRISM_INLINE pm_node_flags_t
6713parse_symbol_encoding(pm_parser_t *parser, const pm_encoding_t *explicit_encoding, const pm_token_t *location, const pm_string_t *contents, bool validate) {
6714 if (explicit_encoding != NULL) {
6715 // A Symbol may optionally have its encoding explicitly set. This will
6716 // happen if an escape sequence results in a non-ASCII code point.
6717 if (explicit_encoding == PM_ENCODING_UTF_8_ENTRY) {
6718 if (validate) parse_symbol_encoding_validate_utf8(parser, location, contents);
6719 return PM_SYMBOL_FLAGS_FORCED_UTF8_ENCODING;
6720 } else if (parser->encoding == PM_ENCODING_US_ASCII_ENTRY) {
6721 return PM_SYMBOL_FLAGS_FORCED_BINARY_ENCODING;
6722 } else if (validate) {
6723 parse_symbol_encoding_validate_other(parser, location, contents);
6724 }
6725 } else if (pm_ascii_only_p(contents)) {
6726 // Ruby stipulates that all source files must use an ASCII-compatible
6727 // encoding. Thus, all symbols appearing in source are eligible for
6728 // "downgrading" to US-ASCII.
6729 return PM_SYMBOL_FLAGS_FORCED_US_ASCII_ENCODING;
6730 } else if (validate) {
6731 parse_symbol_encoding_validate_other(parser, location, contents);
6732 }
6733
6734 return 0;
6735}
6736
6741static pm_symbol_node_t *
6742pm_symbol_node_create_unescaped(pm_parser_t *parser, const pm_token_t *opening, const pm_token_t *value, const pm_token_t *closing, const pm_string_t *unescaped, pm_node_flags_t flags) {
6743 uint32_t start = opening == NULL ? PM_TOKEN_START(parser, value) : PM_TOKEN_START(parser, opening);
6744 uint32_t end = closing == NULL ? PM_TOKEN_END(parser, value) : PM_TOKEN_END(parser, closing);
6745
6746 return pm_symbol_node_new(
6747 parser->arena,
6748 ++parser->node_id,
6749 PM_NODE_FLAG_STATIC_LITERAL | flags,
6750 ((pm_location_t) { .start = start, .length = U32(end - start) }),
6751 NTOK2LOC(parser, opening),
6752 NTOK2LOC(parser, value),
6753 NTOK2LOC(parser, closing),
6754 *unescaped
6755 );
6756}
6757
6761static PRISM_INLINE pm_symbol_node_t *
6762pm_symbol_node_create(pm_parser_t *parser, const pm_token_t *opening, const pm_token_t *value, const pm_token_t *closing) {
6763 return pm_symbol_node_create_unescaped(parser, opening, value, closing, &PM_STRING_EMPTY, 0);
6764}
6765
6769static pm_symbol_node_t *
6770pm_symbol_node_create_current_string(pm_parser_t *parser, const pm_token_t *opening, const pm_token_t *value, const pm_token_t *closing) {
6771 pm_symbol_node_t *node = pm_symbol_node_create_unescaped(parser, opening, value, closing, &parser->current_string, parse_symbol_encoding(parser, parser->explicit_encoding, value, &parser->current_string, false));
6772 parser->current_string = PM_STRING_EMPTY;
6773 return node;
6774}
6775
6779static pm_symbol_node_t *
6780pm_symbol_node_label_create(pm_parser_t *parser, const pm_token_t *token) {
6781 assert(token->type == PM_TOKEN_LABEL);
6782
6783 pm_token_t closing = { .type = PM_TOKEN_LABEL_END, .start = token->end - 1, .end = token->end };
6784 pm_token_t label = { .type = PM_TOKEN_LABEL, .start = token->start, .end = token->end - 1 };
6785 pm_symbol_node_t *node = pm_symbol_node_create(parser, NULL, &label, &closing);
6786
6787 assert((label.end - label.start) >= 0);
6788 pm_string_shared_init(&node->unescaped, label.start, label.end);
6789 pm_node_flag_set(UP(node), parse_symbol_encoding(parser, parser->explicit_encoding, &label, &node->unescaped, false));
6790
6791 return node;
6792}
6793
6797static pm_symbol_node_t *
6798pm_symbol_node_synthesized_create(pm_parser_t *parser, const char *content) {
6799 pm_symbol_node_t *node = pm_symbol_node_new(
6800 parser->arena,
6801 ++parser->node_id,
6802 PM_NODE_FLAG_STATIC_LITERAL | PM_SYMBOL_FLAGS_FORCED_US_ASCII_ENCODING,
6803 PM_LOCATION_INIT_UNSET,
6804 ((pm_location_t) { 0 }),
6805 ((pm_location_t) { 0 }),
6806 ((pm_location_t) { 0 }),
6807 ((pm_string_t) { 0 })
6808 );
6809
6810 pm_string_constant_init(&node->unescaped, content, strlen(content));
6811 return node;
6812}
6813
6817static bool
6818pm_symbol_node_label_p(const pm_parser_t *parser, const pm_node_t *node) {
6819 const pm_location_t *location = NULL;
6820
6821 switch (PM_NODE_TYPE(node)) {
6822 case PM_SYMBOL_NODE: {
6823 const pm_symbol_node_t *cast = (pm_symbol_node_t *) node;
6824 if (cast->closing_loc.length > 0) {
6825 location = &cast->closing_loc;
6826 }
6827 break;
6828 }
6829 case PM_INTERPOLATED_SYMBOL_NODE: {
6830 const pm_interpolated_symbol_node_t *cast = (pm_interpolated_symbol_node_t *) node;
6831 if (cast->closing_loc.length > 0) {
6832 location = &cast->closing_loc;
6833 }
6834 break;
6835 }
6836 default:
6837 return false;
6838 }
6839
6840 return (location != NULL) && (parser->start[PM_LOCATION_END(location) - 1] == ':');
6841}
6842
6846static pm_symbol_node_t *
6847pm_string_node_to_symbol_node(pm_parser_t *parser, pm_string_node_t *node, const pm_token_t *opening, const pm_token_t *closing) {
6848 pm_symbol_node_t *new_node = pm_symbol_node_new(
6849 parser->arena,
6850 ++parser->node_id,
6851 PM_NODE_FLAG_STATIC_LITERAL,
6852 PM_LOCATION_INIT_TOKENS(parser, opening, closing),
6853 TOK2LOC(parser, opening),
6854 node->content_loc,
6855 TOK2LOC(parser, closing),
6856 node->unescaped
6857 );
6858
6859 pm_token_t content = {
6860 .type = PM_TOKEN_IDENTIFIER,
6861 .start = parser->start + node->content_loc.start,
6862 .end = parser->start + node->content_loc.start + node->content_loc.length
6863 };
6864
6865 pm_node_flag_set(UP(new_node), parse_symbol_encoding(parser, parser->explicit_encoding, &content, &node->unescaped, true));
6866
6867 /* The old node is arena-allocated so no explicit free is needed. */
6868 return new_node;
6869}
6870
6874static pm_string_node_t *
6875pm_symbol_node_to_string_node(pm_parser_t *parser, pm_symbol_node_t *node) {
6876 pm_node_flags_t flags = 0;
6877
6878 switch (parser->frozen_string_literal) {
6879 case PM_OPTIONS_FROZEN_STRING_LITERAL_DISABLED:
6880 flags = PM_STRING_FLAGS_MUTABLE;
6881 break;
6882 case PM_OPTIONS_FROZEN_STRING_LITERAL_ENABLED:
6883 flags = PM_NODE_FLAG_STATIC_LITERAL | PM_STRING_FLAGS_FROZEN;
6884 break;
6885 }
6886
6887 pm_string_node_t *new_node = pm_string_node_new(
6888 parser->arena,
6889 ++parser->node_id,
6890 flags,
6891 PM_LOCATION_INIT_NODE(node),
6892 node->opening_loc,
6893 node->content_loc,
6894 node->closing_loc,
6895 node->unescaped
6896 );
6897
6898 /* The old node is arena-allocated so no explicit free is needed. */
6899 return new_node;
6900}
6901
6905static pm_true_node_t *
6906pm_true_node_create(pm_parser_t *parser, const pm_token_t *token) {
6907 assert(token->type == PM_TOKEN_KEYWORD_TRUE);
6908
6909 return pm_true_node_new(
6910 parser->arena,
6911 ++parser->node_id,
6912 PM_NODE_FLAG_STATIC_LITERAL,
6913 PM_LOCATION_INIT_TOKEN(parser, token)
6914 );
6915}
6916
6920static pm_true_node_t *
6921pm_true_node_synthesized_create(pm_parser_t *parser) {
6922 return pm_true_node_new(
6923 parser->arena,
6924 ++parser->node_id,
6925 PM_NODE_FLAG_STATIC_LITERAL,
6926 PM_LOCATION_INIT_UNSET
6927 );
6928}
6929
6933static pm_undef_node_t *
6934pm_undef_node_create(pm_parser_t *parser, const pm_token_t *token) {
6935 assert(token->type == PM_TOKEN_KEYWORD_UNDEF);
6936
6937 return pm_undef_node_new(
6938 parser->arena,
6939 ++parser->node_id,
6940 0,
6941 PM_LOCATION_INIT_TOKEN(parser, token),
6942 ((pm_node_list_t) { 0 }),
6943 TOK2LOC(parser, token)
6944 );
6945}
6946
6950static void
6951pm_undef_node_append(pm_arena_t *arena, pm_undef_node_t *node, pm_node_t *name) {
6952 PM_NODE_LENGTH_SET_NODE(node, name);
6953 pm_node_list_append(arena, &node->names, name);
6954}
6955
6959static pm_unless_node_t *
6960pm_unless_node_create(pm_parser_t *parser, const pm_token_t *keyword, pm_node_t *predicate, const pm_token_t *then_keyword, pm_statements_node_t *statements) {
6961 pm_conditional_predicate(parser, predicate, PM_CONDITIONAL_PREDICATE_TYPE_CONDITIONAL);
6962 pm_node_t *end = statements == NULL ? predicate : UP(statements);
6963
6964 return pm_unless_node_new(
6965 parser->arena,
6966 ++parser->node_id,
6967 PM_NODE_FLAG_NEWLINE,
6968 PM_LOCATION_INIT_TOKEN_NODE(parser, keyword, end),
6969 TOK2LOC(parser, keyword),
6970 predicate,
6971 NTOK2LOC(parser, then_keyword),
6972 statements,
6973 NULL,
6974 ((pm_location_t) { 0 })
6975 );
6976}
6977
6981static pm_unless_node_t *
6982pm_unless_node_modifier_create(pm_parser_t *parser, pm_node_t *statement, const pm_token_t *unless_keyword, pm_node_t *predicate) {
6983 pm_conditional_predicate(parser, predicate, PM_CONDITIONAL_PREDICATE_TYPE_CONDITIONAL);
6984
6985 pm_statements_node_t *statements = pm_statements_node_create(parser);
6986 pm_statements_node_body_append(parser, statements, statement, true);
6987
6988 return pm_unless_node_new(
6989 parser->arena,
6990 ++parser->node_id,
6991 PM_NODE_FLAG_NEWLINE,
6992 PM_LOCATION_INIT_NODES(statement, predicate),
6993 TOK2LOC(parser, unless_keyword),
6994 predicate,
6995 ((pm_location_t) { 0 }),
6996 statements,
6997 NULL,
6998 ((pm_location_t) { 0 })
6999 );
7000}
7001
7002static PRISM_INLINE void
7003pm_unless_node_end_keyword_loc_set(const pm_parser_t *parser, pm_unless_node_t *node, const pm_token_t *end_keyword) {
7004 node->end_keyword_loc = TOK2LOC(parser, end_keyword);
7005 PM_NODE_LENGTH_SET_TOKEN(parser, node, end_keyword);
7006}
7007
7013static void
7014pm_loop_modifier_block_exits(pm_parser_t *parser, pm_statements_node_t *statements) {
7015 assert(parser->current_block_exits != NULL);
7016
7017 // All of the block exits that we want to remove should be within the
7018 // statements, and since we are modifying the statements, we shouldn't have
7019 // to check the end location.
7020 uint32_t start = statements->base.location.start;
7021
7022 for (size_t index = parser->current_block_exits->size; index > 0; index--) {
7023 pm_node_t *block_exit = parser->current_block_exits->nodes[index - 1];
7024 if (block_exit->location.start < start) break;
7025
7026 // Implicitly remove from the list by lowering the size.
7027 parser->current_block_exits->size--;
7028 }
7029}
7030
7034static pm_until_node_t *
7035pm_until_node_create(pm_parser_t *parser, const pm_token_t *keyword, const pm_token_t *do_keyword, const pm_token_t *closing, pm_node_t *predicate, pm_statements_node_t *statements, pm_node_flags_t flags) {
7036 pm_conditional_predicate(parser, predicate, PM_CONDITIONAL_PREDICATE_TYPE_CONDITIONAL);
7037
7038 return pm_until_node_new(
7039 parser->arena,
7040 ++parser->node_id,
7041 flags,
7042 PM_LOCATION_INIT_TOKENS(parser, keyword, closing),
7043 TOK2LOC(parser, keyword),
7044 NTOK2LOC(parser, do_keyword),
7045 TOK2LOC(parser, closing),
7046 predicate,
7047 statements
7048 );
7049}
7050
7054static pm_until_node_t *
7055pm_until_node_modifier_create(pm_parser_t *parser, const pm_token_t *keyword, pm_node_t *predicate, pm_statements_node_t *statements, pm_node_flags_t flags) {
7056 pm_conditional_predicate(parser, predicate, PM_CONDITIONAL_PREDICATE_TYPE_CONDITIONAL);
7057 pm_loop_modifier_block_exits(parser, statements);
7058
7059 return pm_until_node_new(
7060 parser->arena,
7061 ++parser->node_id,
7062 flags,
7063 PM_LOCATION_INIT_NODES(statements, predicate),
7064 TOK2LOC(parser, keyword),
7065 ((pm_location_t) { 0 }),
7066 ((pm_location_t) { 0 }),
7067 predicate,
7068 statements
7069 );
7070}
7071
7075static pm_when_node_t *
7076pm_when_node_create(pm_parser_t *parser, const pm_token_t *keyword) {
7077 return pm_when_node_new(
7078 parser->arena,
7079 ++parser->node_id,
7080 0,
7081 PM_LOCATION_INIT_TOKEN(parser, keyword),
7082 TOK2LOC(parser, keyword),
7083 ((pm_node_list_t) { 0 }),
7084 ((pm_location_t) { 0 }),
7085 NULL
7086 );
7087}
7088
7092static void
7093pm_when_node_conditions_append(pm_arena_t *arena, pm_when_node_t *node, pm_node_t *condition) {
7094 PM_NODE_LENGTH_SET_NODE(node, condition);
7095 pm_node_list_append(arena, &node->conditions, condition);
7096}
7097
7101static PRISM_INLINE void
7102pm_when_node_then_keyword_loc_set(const pm_parser_t *parser, pm_when_node_t *node, const pm_token_t *then_keyword) {
7103 PM_NODE_LENGTH_SET_TOKEN(parser, node, then_keyword);
7104 node->then_keyword_loc = TOK2LOC(parser, then_keyword);
7105}
7106
7110static void
7111pm_when_node_statements_set(pm_when_node_t *node, pm_statements_node_t *statements) {
7112 if (PM_NODE_END(statements) > PM_NODE_END(node)) {
7113 PM_NODE_LENGTH_SET_NODE(node, statements);
7114 }
7115
7116 node->statements = statements;
7117}
7118
7122static pm_while_node_t *
7123pm_while_node_create(pm_parser_t *parser, const pm_token_t *keyword, const pm_token_t *do_keyword, const pm_token_t *closing, pm_node_t *predicate, pm_statements_node_t *statements, pm_node_flags_t flags) {
7124 pm_conditional_predicate(parser, predicate, PM_CONDITIONAL_PREDICATE_TYPE_CONDITIONAL);
7125
7126 return pm_while_node_new(
7127 parser->arena,
7128 ++parser->node_id,
7129 flags,
7130 PM_LOCATION_INIT_TOKENS(parser, keyword, closing),
7131 TOK2LOC(parser, keyword),
7132 NTOK2LOC(parser, do_keyword),
7133 TOK2LOC(parser, closing),
7134 predicate,
7135 statements
7136 );
7137}
7138
7142static pm_while_node_t *
7143pm_while_node_modifier_create(pm_parser_t *parser, const pm_token_t *keyword, pm_node_t *predicate, pm_statements_node_t *statements, pm_node_flags_t flags) {
7144 pm_conditional_predicate(parser, predicate, PM_CONDITIONAL_PREDICATE_TYPE_CONDITIONAL);
7145 pm_loop_modifier_block_exits(parser, statements);
7146
7147 return pm_while_node_new(
7148 parser->arena,
7149 ++parser->node_id,
7150 flags,
7151 PM_LOCATION_INIT_NODES(statements, predicate),
7152 TOK2LOC(parser, keyword),
7153 ((pm_location_t) { 0 }),
7154 ((pm_location_t) { 0 }),
7155 predicate,
7156 statements
7157 );
7158}
7159
7163static pm_while_node_t *
7164pm_while_node_synthesized_create(pm_parser_t *parser, pm_node_t *predicate, pm_statements_node_t *statements) {
7165 return pm_while_node_new(
7166 parser->arena,
7167 ++parser->node_id,
7168 0,
7169 PM_LOCATION_INIT_UNSET,
7170 ((pm_location_t) { 0 }),
7171 ((pm_location_t) { 0 }),
7172 ((pm_location_t) { 0 }),
7173 predicate,
7174 statements
7175 );
7176}
7177
7182static pm_x_string_node_t *
7183pm_xstring_node_create_unescaped(pm_parser_t *parser, const pm_token_t *opening, const pm_token_t *content, const pm_token_t *closing, const pm_string_t *unescaped) {
7184 return pm_x_string_node_new(
7185 parser->arena,
7186 ++parser->node_id,
7187 PM_STRING_FLAGS_FROZEN,
7188 PM_LOCATION_INIT_TOKENS(parser, opening, closing),
7189 TOK2LOC(parser, opening),
7190 TOK2LOC(parser, content),
7191 TOK2LOC(parser, closing),
7192 *unescaped
7193 );
7194}
7195
7199static PRISM_INLINE pm_x_string_node_t *
7200pm_xstring_node_create(pm_parser_t *parser, const pm_token_t *opening, const pm_token_t *content, const pm_token_t *closing) {
7201 return pm_xstring_node_create_unescaped(parser, opening, content, closing, &PM_STRING_EMPTY);
7202}
7203
7207static pm_yield_node_t *
7208pm_yield_node_create(pm_parser_t *parser, const pm_token_t *keyword, const pm_location_t *lparen_loc, pm_arguments_node_t *arguments, const pm_location_t *rparen_loc) {
7209 uint32_t start = PM_TOKEN_START(parser, keyword);
7210 uint32_t end;
7211
7212 if (rparen_loc->length > 0) {
7213 end = PM_LOCATION_END(rparen_loc);
7214 } else if (arguments != NULL) {
7215 end = PM_NODE_END(arguments);
7216 } else if (lparen_loc->length > 0) {
7217 end = PM_LOCATION_END(lparen_loc);
7218 } else {
7219 end = PM_TOKEN_END(parser, keyword);
7220 }
7221
7222 return pm_yield_node_new(
7223 parser->arena,
7224 ++parser->node_id,
7225 0,
7226 ((pm_location_t) { .start = start, .length = U32(end - start) }),
7227 TOK2LOC(parser, keyword),
7228 *lparen_loc,
7229 arguments,
7230 *rparen_loc
7231 );
7232}
7233
7238static int
7239pm_parser_local_depth_constant_id(pm_parser_t *parser, pm_constant_id_t constant_id) {
7240 pm_scope_t *scope = parser->current_scope;
7241 int depth = 0;
7242
7243 while (scope != NULL) {
7244 if (pm_locals_find(&scope->locals, constant_id) != UINT32_MAX) return depth;
7245 if (scope->closed) break;
7246
7247 scope = scope->previous;
7248 depth++;
7249 }
7250
7251 return -1;
7252}
7253
7259static PRISM_INLINE int
7260pm_parser_local_depth(pm_parser_t *parser, pm_token_t *token) {
7261 return pm_parser_local_depth_constant_id(parser, pm_parser_constant_id_token(parser, token));
7262}
7263
7267static PRISM_INLINE void
7268pm_parser_local_add(pm_parser_t *parser, pm_constant_id_t constant_id, const uint8_t *start, const uint8_t *end, uint32_t reads) {
7269 pm_locals_write(&parser->current_scope->locals, constant_id, U32(start - parser->start), U32(end - start), reads);
7270}
7271
7275static pm_constant_id_t
7276pm_parser_local_add_raw(pm_parser_t *parser, const uint8_t *start, const uint8_t *end, uint32_t reads) {
7277 pm_constant_id_t constant_id = pm_parser_constant_id_raw(parser, start, end);
7278 if (constant_id != 0) pm_parser_local_add(parser, constant_id, start, end, reads);
7279 return constant_id;
7280}
7281
7285static PRISM_INLINE pm_constant_id_t
7286pm_parser_local_add_location(pm_parser_t *parser, pm_location_t *location, uint32_t reads) {
7287 return pm_parser_local_add_raw(parser, parser->start + location->start, parser->start + location->start + location->length, reads);
7288}
7289
7293static PRISM_INLINE pm_constant_id_t
7294pm_parser_local_add_token(pm_parser_t *parser, pm_token_t *token, uint32_t reads) {
7295 return pm_parser_local_add_raw(parser, token->start, token->end, reads);
7296}
7297
7301static pm_constant_id_t
7302pm_parser_local_add_owned(pm_parser_t *parser, uint8_t *start, size_t length) {
7303 pm_constant_id_t constant_id = pm_parser_constant_id_owned(parser, start, length);
7304 if (constant_id != 0) pm_parser_local_add(parser, constant_id, parser->start, parser->start, 1);
7305 return constant_id;
7306}
7307
7311static pm_constant_id_t
7312pm_parser_local_add_constant(pm_parser_t *parser, const char *start, size_t length) {
7313 pm_constant_id_t constant_id = pm_parser_constant_id_constant(parser, start, length);
7314 if (constant_id != 0) pm_parser_local_add(parser, constant_id, parser->start, parser->start, 1);
7315 return constant_id;
7316}
7317
7325static bool
7326pm_parser_parameter_name_check(pm_parser_t *parser, const pm_token_t *name) {
7327 // We want to check whether the parameter name is a numbered parameter or
7328 // not.
7329 pm_refute_numbered_parameter(parser, PM_TOKEN_START(parser, name), PM_TOKEN_LENGTH(name));
7330
7331 // Otherwise we'll fetch the constant id for the parameter name and check
7332 // whether it's already in the current scope.
7333 pm_constant_id_t constant_id = pm_parser_constant_id_token(parser, name);
7334
7335 if (pm_locals_find(&parser->current_scope->locals, constant_id) != UINT32_MAX) {
7336 // Add an error if the parameter doesn't start with _ and has been seen before
7337 if ((name->start < name->end) && (*name->start != '_')) {
7338 pm_parser_err_token(parser, name, PM_ERR_PARAMETER_NAME_DUPLICATED);
7339 }
7340 return true;
7341 }
7342 return false;
7343}
7344
7348static void
7349pm_parser_scope_pop(pm_parser_t *parser) {
7350 pm_scope_t *scope = parser->current_scope;
7351 parser->current_scope = scope->previous;
7352 pm_locals_free(&scope->locals);
7353 xfree_sized(scope, sizeof(pm_scope_t));
7354}
7355
7356/******************************************************************************/
7357/* Stack helpers */
7358/******************************************************************************/
7359
7363static PRISM_INLINE void
7364pm_state_stack_push(pm_state_stack_t *stack, bool value) {
7365 *stack = (*stack << 1) | (value & 1);
7366}
7367
7371static PRISM_INLINE void
7372pm_state_stack_pop(pm_state_stack_t *stack) {
7373 *stack >>= 1;
7374}
7375
7379static PRISM_INLINE bool
7380pm_state_stack_p(const pm_state_stack_t *stack) {
7381 return *stack & 1;
7382}
7383
7384static PRISM_INLINE void
7385pm_accepts_block_stack_push(pm_parser_t *parser, bool value) {
7386 // Use the negation of the value to prevent stack overflow.
7387 pm_state_stack_push(&parser->accepts_block_stack, !value);
7388}
7389
7390static PRISM_INLINE void
7391pm_accepts_block_stack_pop(pm_parser_t *parser) {
7392 pm_state_stack_pop(&parser->accepts_block_stack);
7393}
7394
7395static PRISM_INLINE bool
7396pm_accepts_block_stack_p(pm_parser_t *parser) {
7397 return !pm_state_stack_p(&parser->accepts_block_stack);
7398}
7399
7400static PRISM_INLINE void
7401pm_do_loop_stack_push(pm_parser_t *parser, bool value) {
7402 pm_state_stack_push(&parser->do_loop_stack, value);
7403}
7404
7405static PRISM_INLINE void
7406pm_do_loop_stack_pop(pm_parser_t *parser) {
7407 pm_state_stack_pop(&parser->do_loop_stack);
7408}
7409
7410static PRISM_INLINE bool
7411pm_do_loop_stack_p(pm_parser_t *parser) {
7412 return pm_state_stack_p(&parser->do_loop_stack);
7413}
7414
7439static PRISM_INLINE void
7440pm_enclosure_frame_push(pm_parser_t *parser) {
7441 pm_do_loop_stack_push(parser, false);
7442 pm_accepts_block_stack_push(parser, true);
7443}
7444
7445static PRISM_INLINE void
7446pm_enclosure_frame_pop(pm_parser_t *parser) {
7447 pm_do_loop_stack_pop(parser);
7448 pm_accepts_block_stack_pop(parser);
7449}
7450
7451/******************************************************************************/
7452/* Lexer check helpers */
7453/******************************************************************************/
7454
7459static PRISM_INLINE uint8_t
7460peek_at(const pm_parser_t *parser, const uint8_t *cursor) {
7461 if (cursor < parser->end) {
7462 return *cursor;
7463 } else {
7464 return '\0';
7465 }
7466}
7467
7473static PRISM_INLINE uint8_t
7474peek_offset(pm_parser_t *parser, ptrdiff_t offset) {
7475 return peek_at(parser, parser->current.end + offset);
7476}
7477
7482static PRISM_INLINE uint8_t
7483peek(const pm_parser_t *parser) {
7484 return peek_at(parser, parser->current.end);
7485}
7486
7491static PRISM_INLINE bool
7492match(pm_parser_t *parser, uint8_t value) {
7493 if (peek(parser) == value) {
7494 parser->current.end++;
7495 return true;
7496 }
7497 return false;
7498}
7499
7504static PRISM_INLINE size_t
7505match_eol_at(pm_parser_t *parser, const uint8_t *cursor) {
7506 if (peek_at(parser, cursor) == '\n') {
7507 return 1;
7508 }
7509 if (peek_at(parser, cursor) == '\r' && peek_at(parser, cursor + 1) == '\n') {
7510 return 2;
7511 }
7512 return 0;
7513}
7514
7520static PRISM_INLINE size_t
7521match_eol_offset(pm_parser_t *parser, ptrdiff_t offset) {
7522 return match_eol_at(parser, parser->current.end + offset);
7523}
7524
7530static PRISM_INLINE size_t
7531match_eol(pm_parser_t *parser) {
7532 return match_eol_at(parser, parser->current.end);
7533}
7534
7538static PRISM_INLINE const uint8_t *
7539next_newline(const uint8_t *cursor, ptrdiff_t length) {
7540 assert(length >= 0);
7541
7542 // Note that it's okay for us to use memchr here to look for \n because none
7543 // of the encodings that we support have \n as a component of a multi-byte
7544 // character.
7545 return memchr(cursor, '\n', (size_t) length);
7546}
7547
7551static PRISM_INLINE bool
7552ambiguous_operator_p(const pm_parser_t *parser, bool space_seen) {
7553 return !lex_state_p(parser, PM_LEX_STATE_CLASS | PM_LEX_STATE_DOT | PM_LEX_STATE_FNAME | PM_LEX_STATE_ENDFN) && space_seen && !pm_char_is_whitespace(peek(parser));
7554}
7555
7560static bool
7561parser_lex_magic_comment_encoding_value(pm_parser_t *parser, const uint8_t *start, const uint8_t *end) {
7562 const pm_encoding_t *encoding = pm_encoding_find(start, end);
7563
7564 if (encoding != NULL) {
7565 if (parser->encoding != encoding) {
7566 parser->encoding = encoding;
7567 if (parser->encoding_changed_callback != NULL) parser->encoding_changed_callback(parser);
7568 }
7569
7570 parser->encoding_changed = (encoding != PM_ENCODING_UTF_8_ENTRY);
7571 return true;
7572 }
7573
7574 return false;
7575}
7576
7581static void
7582parser_lex_magic_comment_encoding(pm_parser_t *parser) {
7583 const uint8_t *cursor = parser->current.start + 1;
7584 const uint8_t *end = parser->current.end;
7585
7586 bool separator = false;
7587 while (true) {
7588 if (end - cursor <= 6) return;
7589 switch (cursor[6]) {
7590 case 'C': case 'c': cursor += 6; continue;
7591 case 'O': case 'o': cursor += 5; continue;
7592 case 'D': case 'd': cursor += 4; continue;
7593 case 'I': case 'i': cursor += 3; continue;
7594 case 'N': case 'n': cursor += 2; continue;
7595 case 'G': case 'g': cursor += 1; continue;
7596 case '=': case ':':
7597 separator = true;
7598 cursor += 6;
7599 break;
7600 default:
7601 cursor += 6;
7602 if (pm_char_is_whitespace(*cursor)) break;
7603 continue;
7604 }
7605 if (pm_strncasecmp(cursor - 6, (const uint8_t *) "coding", 6) == 0) break;
7606 separator = false;
7607 }
7608
7609 while (true) {
7610 do {
7611 if (++cursor >= end) return;
7612 } while (pm_char_is_whitespace(*cursor));
7613
7614 if (separator) break;
7615 if (*cursor != '=' && *cursor != ':') return;
7616
7617 separator = true;
7618 cursor++;
7619 }
7620
7621 const uint8_t *value_start = cursor;
7622 while ((*cursor == '-' || *cursor == '_' || parser->encoding->alnum_char(cursor, 1)) && ++cursor < end);
7623
7624 if (!parser_lex_magic_comment_encoding_value(parser, value_start, cursor)) {
7625 // If we were unable to parse the encoding value, then we've got an
7626 // issue because we didn't understand the encoding that the user was
7627 // trying to use. In this case we'll keep using the default encoding but
7628 // add an error to the parser to indicate an unsuccessful parse.
7629 pm_parser_err(parser, U32(value_start - parser->start), U32(cursor - value_start), PM_ERR_INVALID_ENCODING_MAGIC_COMMENT);
7630 }
7631}
7632
7633typedef enum {
7634 PM_MAGIC_COMMENT_BOOLEAN_VALUE_TRUE,
7635 PM_MAGIC_COMMENT_BOOLEAN_VALUE_FALSE,
7636 PM_MAGIC_COMMENT_BOOLEAN_VALUE_INVALID
7637} pm_magic_comment_boolean_value_t;
7638
7643static pm_magic_comment_boolean_value_t
7644parser_lex_magic_comment_boolean_value(const uint8_t *value_start, uint32_t value_length) {
7645 if (value_length == 4 && pm_strncasecmp(value_start, (const uint8_t *) "true", 4) == 0) {
7646 return PM_MAGIC_COMMENT_BOOLEAN_VALUE_TRUE;
7647 } else if (value_length == 5 && pm_strncasecmp(value_start, (const uint8_t *) "false", 5) == 0) {
7648 return PM_MAGIC_COMMENT_BOOLEAN_VALUE_FALSE;
7649 } else {
7650 return PM_MAGIC_COMMENT_BOOLEAN_VALUE_INVALID;
7651 }
7652}
7653
7654static PRISM_INLINE bool
7655pm_char_is_magic_comment_key_delimiter(const uint8_t b) {
7656 return b == '\'' || b == '"' || b == ':' || b == ';';
7657}
7658
7664static PRISM_INLINE const uint8_t *
7665parser_lex_magic_comment_emacs_marker(pm_parser_t *parser, const uint8_t *cursor, const uint8_t *end) {
7666 // Scan for '*' as the middle character, since it is rarer than '-' in
7667 // typical comments and avoids repeated memchr calls for '-' that hit
7668 // dashes in words like "foo-bar".
7669 while ((cursor + 3 <= end) && (cursor = pm_memchr(cursor + 1, '*', (size_t) (end - cursor - 1), parser->encoding_changed, parser->encoding)) != NULL) {
7670 if (cursor[-1] == '-' && cursor + 1 < end && cursor[1] == '-') {
7671 return cursor - 1;
7672 }
7673 }
7674 return NULL;
7675}
7676
7687static PRISM_INLINE bool
7688parser_lex_magic_comment(pm_parser_t *parser, bool semantic_token_seen) {
7689 bool result = true;
7690
7691 const uint8_t *start = parser->current.start + 1;
7692 const uint8_t *end = parser->current.end;
7693 if (end - start <= 7) return false;
7694
7695 const uint8_t *cursor;
7696 bool indicator = false;
7697
7698 if ((cursor = parser_lex_magic_comment_emacs_marker(parser, start, end)) != NULL) {
7699 start = cursor + 3;
7700
7701 if ((cursor = parser_lex_magic_comment_emacs_marker(parser, start, end)) != NULL) {
7702 end = cursor;
7703 indicator = true;
7704 } else {
7705 // If we have a start marker but not an end marker, then we cannot
7706 // have a magic comment.
7707 return false;
7708 }
7709 } else {
7710 // Non-emacs magic comments must contain a colon for `key: value`.
7711 // Reject early if there is no colon to avoid scanning the entire
7712 // comment character-by-character.
7713 if (pm_memchr(start, ':', (size_t) (end - start), parser->encoding_changed, parser->encoding) == NULL) {
7714 return false;
7715 }
7716
7717 // Advance start past leading whitespace so the main loop begins
7718 // directly at the key, avoiding a redundant whitespace scan.
7719 start += pm_strspn_whitespace(start, end - start);
7720 }
7721
7722 cursor = start;
7723 while (cursor < end) {
7724 if (indicator) {
7725 while (cursor < end && (pm_char_is_magic_comment_key_delimiter(*cursor) || pm_char_is_whitespace(*cursor))) cursor++;
7726 }
7727
7728 const uint8_t *key_start = cursor;
7729 while (cursor < end && (!pm_char_is_magic_comment_key_delimiter(*cursor) && !pm_char_is_whitespace(*cursor))) cursor++;
7730
7731 const uint8_t *key_end = cursor;
7732 while (cursor < end && pm_char_is_whitespace(*cursor)) cursor++;
7733 if (cursor == end) break;
7734
7735 if (*cursor == ':') {
7736 cursor++;
7737 } else {
7738 if (!indicator) return false;
7739 continue;
7740 }
7741
7742 while (cursor < end && pm_char_is_whitespace(*cursor)) cursor++;
7743 if (cursor == end) break;
7744
7745 const uint8_t *value_start;
7746 const uint8_t *value_end;
7747
7748 if (*cursor == '"') {
7749 value_start = ++cursor;
7750 for (; cursor < end && *cursor != '"'; cursor++) {
7751 if (*cursor == '\\' && (cursor + 1 < end)) cursor++;
7752 }
7753 value_end = cursor;
7754 if (cursor < end && *cursor == '"') cursor++;
7755 } else {
7756 value_start = cursor;
7757 while (cursor < end && *cursor != '"' && *cursor != ';' && !pm_char_is_whitespace(*cursor)) cursor++;
7758 value_end = cursor;
7759 }
7760
7761 if (indicator) {
7762 while (cursor < end && (*cursor == ';' || pm_char_is_whitespace(*cursor))) cursor++;
7763 } else {
7764 while (cursor < end && pm_char_is_whitespace(*cursor)) cursor++;
7765 if (cursor != end) return false;
7766 }
7767
7768 // Here, we need to do some processing on the key to swap out dashes for
7769 // underscores. We only need to do this if there _is_ a dash in the key.
7770 pm_string_t key;
7771 const size_t key_length = (size_t) (key_end - key_start);
7772 const uint8_t *dash = pm_memchr(key_start, '-', key_length, parser->encoding_changed, parser->encoding);
7773
7774 if (dash == NULL) {
7775 pm_string_shared_init(&key, key_start, key_end);
7776 } else {
7777 uint8_t *buffer = xmalloc(key_length);
7778 if (buffer == NULL) break;
7779
7780 memcpy(buffer, key_start, key_length);
7781 buffer[dash - key_start] = '_';
7782
7783 while ((dash = pm_memchr(dash + 1, '-', (size_t) (key_end - dash - 1), parser->encoding_changed, parser->encoding)) != NULL) {
7784 buffer[dash - key_start] = '_';
7785 }
7786
7787 pm_string_owned_init(&key, buffer, key_length);
7788 }
7789
7790 // Finally, we can start checking the key against the list of known
7791 // magic comment keys, and potentially change state based on that.
7792 const uint8_t *key_source = pm_string_source(&key);
7793 uint32_t value_length = (uint32_t) (value_end - value_start);
7794
7795 // We only want to attempt to compare against encoding comments if it's
7796 // the first line in the file (or the second in the case of a shebang).
7797 if (parser->current.start == parser->encoding_comment_start && !parser->encoding_locked) {
7798 if (
7799 (key_length == 8 && pm_strncasecmp(key_source, (const uint8_t *) "encoding", 8) == 0) ||
7800 (key_length == 6 && pm_strncasecmp(key_source, (const uint8_t *) "coding", 6) == 0)
7801 ) {
7802 result = parser_lex_magic_comment_encoding_value(parser, value_start, value_end);
7803 }
7804 }
7805
7806 if (key_length == 11) {
7807 if (pm_strncasecmp(key_source, (const uint8_t *) "warn_indent", 11) == 0) {
7808 switch (parser_lex_magic_comment_boolean_value(value_start, value_length)) {
7809 case PM_MAGIC_COMMENT_BOOLEAN_VALUE_INVALID:
7810 PM_PARSER_WARN_TOKEN_FORMAT(
7811 parser,
7812 &parser->current,
7813 PM_WARN_INVALID_MAGIC_COMMENT_VALUE,
7814 (int) key_length,
7815 (const char *) key_source,
7816 (int) value_length,
7817 (const char *) value_start
7818 );
7819 break;
7820 case PM_MAGIC_COMMENT_BOOLEAN_VALUE_FALSE:
7821 parser->warn_mismatched_indentation = false;
7822 break;
7823 case PM_MAGIC_COMMENT_BOOLEAN_VALUE_TRUE:
7824 parser->warn_mismatched_indentation = true;
7825 break;
7826 }
7827 }
7828 } else if (key_length == 21) {
7829 if (pm_strncasecmp(key_source, (const uint8_t *) "frozen_string_literal", 21) == 0) {
7830 // We only want to handle frozen string literal comments if it's
7831 // before any semantic tokens have been seen.
7832 if (semantic_token_seen) {
7833 pm_parser_warn_token(parser, &parser->current, PM_WARN_IGNORED_FROZEN_STRING_LITERAL);
7834 } else {
7835 switch (parser_lex_magic_comment_boolean_value(value_start, value_length)) {
7836 case PM_MAGIC_COMMENT_BOOLEAN_VALUE_INVALID:
7837 PM_PARSER_WARN_TOKEN_FORMAT(
7838 parser,
7839 &parser->current,
7840 PM_WARN_INVALID_MAGIC_COMMENT_VALUE,
7841 (int) key_length,
7842 (const char *) key_source,
7843 (int) value_length,
7844 (const char *) value_start
7845 );
7846 break;
7847 case PM_MAGIC_COMMENT_BOOLEAN_VALUE_FALSE:
7848 parser->frozen_string_literal = PM_OPTIONS_FROZEN_STRING_LITERAL_DISABLED;
7849 break;
7850 case PM_MAGIC_COMMENT_BOOLEAN_VALUE_TRUE:
7851 parser->frozen_string_literal = PM_OPTIONS_FROZEN_STRING_LITERAL_ENABLED;
7852 break;
7853 }
7854 }
7855 }
7856 } else if (key_length == 24) {
7857 if (pm_strncasecmp(key_source, (const uint8_t *) "shareable_constant_value", 24) == 0) {
7858 const uint8_t *cursor = parser->current.start;
7859 while ((cursor > parser->start) && ((cursor[-1] == ' ') || (cursor[-1] == '\t'))) cursor--;
7860
7861 if (!((cursor == parser->start) || (cursor[-1] == '\n'))) {
7862 pm_parser_warn_token(parser, &parser->current, PM_WARN_SHAREABLE_CONSTANT_VALUE_LINE);
7863 } else if (value_length == 4 && pm_strncasecmp(value_start, (const uint8_t *) "none", 4) == 0) {
7864 pm_parser_scope_shareable_constant_set(parser, PM_SCOPE_SHAREABLE_CONSTANT_NONE);
7865 } else if (value_length == 7 && pm_strncasecmp(value_start, (const uint8_t *) "literal", 7) == 0) {
7866 pm_parser_scope_shareable_constant_set(parser, PM_SCOPE_SHAREABLE_CONSTANT_LITERAL);
7867 } else if (value_length == 23 && pm_strncasecmp(value_start, (const uint8_t *) "experimental_everything", 23) == 0) {
7868 pm_parser_scope_shareable_constant_set(parser, PM_SCOPE_SHAREABLE_CONSTANT_EXPERIMENTAL_EVERYTHING);
7869 } else if (value_length == 17 && pm_strncasecmp(value_start, (const uint8_t *) "experimental_copy", 17) == 0) {
7870 pm_parser_scope_shareable_constant_set(parser, PM_SCOPE_SHAREABLE_CONSTANT_EXPERIMENTAL_COPY);
7871 } else {
7872 PM_PARSER_WARN_TOKEN_FORMAT(
7873 parser,
7874 &parser->current,
7875 PM_WARN_INVALID_MAGIC_COMMENT_VALUE,
7876 (int) key_length,
7877 (const char *) key_source,
7878 (int) value_length,
7879 (const char *) value_start
7880 );
7881 }
7882 }
7883 }
7884
7885 // When we're done, we want to free the string in case we had to
7886 // allocate memory for it.
7887 pm_string_cleanup(&key);
7888
7889 // Allocate a new magic comment node to append to the parser's list.
7890 pm_magic_comment_t *magic_comment = (pm_magic_comment_t *) pm_arena_alloc(&parser->metadata_arena, sizeof(pm_magic_comment_t), PRISM_ALIGNOF(pm_magic_comment_t));
7891 magic_comment->node.next = NULL;
7892 magic_comment->key = (pm_location_t) { .start = U32(key_start - parser->start), .length = U32(key_length) };
7893 magic_comment->value = (pm_location_t) { .start = U32(value_start - parser->start), .length = value_length };
7894 pm_list_append(&parser->magic_comment_list, (pm_list_node_t *) magic_comment);
7895 }
7896
7897 return result;
7898}
7899
7900/******************************************************************************/
7901/* Context manipulations */
7902/******************************************************************************/
7903
7904static const uint32_t context_terminators[] = {
7905 [PM_CONTEXT_NONE] = 0,
7906 [PM_CONTEXT_BEGIN] = (1U << PM_TOKEN_KEYWORD_ENSURE) | (1U << PM_TOKEN_KEYWORD_RESCUE) | (1U << PM_TOKEN_KEYWORD_ELSE) | (1U << PM_TOKEN_KEYWORD_END),
7907 [PM_CONTEXT_BEGIN_ENSURE] = (1U << PM_TOKEN_KEYWORD_END),
7908 [PM_CONTEXT_BEGIN_ELSE] = (1U << PM_TOKEN_KEYWORD_ENSURE) | (1U << PM_TOKEN_KEYWORD_END),
7909 [PM_CONTEXT_BEGIN_RESCUE] = (1U << PM_TOKEN_KEYWORD_ENSURE) | (1U << PM_TOKEN_KEYWORD_RESCUE) | (1U << PM_TOKEN_KEYWORD_ELSE) | (1U << PM_TOKEN_KEYWORD_END),
7910 [PM_CONTEXT_BLOCK_BRACES] = (1U << PM_TOKEN_BRACE_RIGHT),
7911 [PM_CONTEXT_BLOCK_KEYWORDS] = (1U << PM_TOKEN_KEYWORD_END) | (1U << PM_TOKEN_KEYWORD_RESCUE) | (1U << PM_TOKEN_KEYWORD_ENSURE),
7912 [PM_CONTEXT_BLOCK_ENSURE] = (1U << PM_TOKEN_KEYWORD_END),
7913 [PM_CONTEXT_BLOCK_ELSE] = (1U << PM_TOKEN_KEYWORD_ENSURE) | (1U << PM_TOKEN_KEYWORD_END),
7914 [PM_CONTEXT_BLOCK_PARAMETERS] = (1U << PM_TOKEN_PIPE),
7915 [PM_CONTEXT_BLOCK_RESCUE] = (1U << PM_TOKEN_KEYWORD_ENSURE) | (1U << PM_TOKEN_KEYWORD_RESCUE) | (1U << PM_TOKEN_KEYWORD_ELSE) | (1U << PM_TOKEN_KEYWORD_END),
7916 [PM_CONTEXT_CASE_WHEN] = (1U << PM_TOKEN_KEYWORD_WHEN) | (1U << PM_TOKEN_KEYWORD_END) | (1U << PM_TOKEN_KEYWORD_ELSE),
7917 [PM_CONTEXT_CASE_IN] = (1U << PM_TOKEN_KEYWORD_IN) | (1U << PM_TOKEN_KEYWORD_END) | (1U << PM_TOKEN_KEYWORD_ELSE),
7918 [PM_CONTEXT_CLASS] = (1U << PM_TOKEN_KEYWORD_END) | (1U << PM_TOKEN_KEYWORD_RESCUE) | (1U << PM_TOKEN_KEYWORD_ENSURE),
7919 [PM_CONTEXT_CLASS_ENSURE] = (1U << PM_TOKEN_KEYWORD_END),
7920 [PM_CONTEXT_CLASS_ELSE] = (1U << PM_TOKEN_KEYWORD_ENSURE) | (1U << PM_TOKEN_KEYWORD_END),
7921 [PM_CONTEXT_CLASS_RESCUE] = (1U << PM_TOKEN_KEYWORD_ENSURE) | (1U << PM_TOKEN_KEYWORD_RESCUE) | (1U << PM_TOKEN_KEYWORD_ELSE) | (1U << PM_TOKEN_KEYWORD_END),
7922 [PM_CONTEXT_DEF] = (1U << PM_TOKEN_KEYWORD_END) | (1U << PM_TOKEN_KEYWORD_RESCUE) | (1U << PM_TOKEN_KEYWORD_ENSURE),
7923 [PM_CONTEXT_DEF_ENSURE] = (1U << PM_TOKEN_KEYWORD_END),
7924 [PM_CONTEXT_DEF_ELSE] = (1U << PM_TOKEN_KEYWORD_ENSURE) | (1U << PM_TOKEN_KEYWORD_END),
7925 [PM_CONTEXT_DEF_RESCUE] = (1U << PM_TOKEN_KEYWORD_ENSURE) | (1U << PM_TOKEN_KEYWORD_RESCUE) | (1U << PM_TOKEN_KEYWORD_ELSE) | (1U << PM_TOKEN_KEYWORD_END),
7926 [PM_CONTEXT_DEF_PARAMS] = (1U << PM_TOKEN_EOF),
7927 [PM_CONTEXT_DEFINED] = (1U << PM_TOKEN_EOF),
7928 [PM_CONTEXT_DEFAULT_PARAMS] = (1U << PM_TOKEN_COMMA) | (1U << PM_TOKEN_PARENTHESIS_RIGHT),
7929 [PM_CONTEXT_ELSE] = (1U << PM_TOKEN_KEYWORD_END),
7930 [PM_CONTEXT_ELSIF] = (1U << PM_TOKEN_KEYWORD_ELSE) | (1U << PM_TOKEN_KEYWORD_ELSIF) | (1U << PM_TOKEN_KEYWORD_END),
7931 [PM_CONTEXT_EMBEXPR] = (1U << PM_TOKEN_EMBEXPR_END),
7932 [PM_CONTEXT_FOR] = (1U << PM_TOKEN_KEYWORD_END),
7933 [PM_CONTEXT_FOR_INDEX] = (1U << PM_TOKEN_KEYWORD_IN),
7934 [PM_CONTEXT_IF] = (1U << PM_TOKEN_KEYWORD_ELSE) | (1U << PM_TOKEN_KEYWORD_ELSIF) | (1U << PM_TOKEN_KEYWORD_END),
7935 [PM_CONTEXT_LAMBDA_BRACES] = (1U << PM_TOKEN_BRACE_RIGHT),
7936 [PM_CONTEXT_LAMBDA_DO_END] = (1U << PM_TOKEN_KEYWORD_END) | (1U << PM_TOKEN_KEYWORD_RESCUE) | (1U << PM_TOKEN_KEYWORD_ENSURE),
7937 [PM_CONTEXT_LAMBDA_ENSURE] = (1U << PM_TOKEN_KEYWORD_END),
7938 [PM_CONTEXT_LAMBDA_ELSE] = (1U << PM_TOKEN_KEYWORD_ENSURE) | (1U << PM_TOKEN_KEYWORD_END),
7939 [PM_CONTEXT_LAMBDA_RESCUE] = (1U << PM_TOKEN_KEYWORD_ENSURE) | (1U << PM_TOKEN_KEYWORD_RESCUE) | (1U << PM_TOKEN_KEYWORD_ELSE) | (1U << PM_TOKEN_KEYWORD_END),
7940 [PM_CONTEXT_LOOP_PREDICATE] = (1U << PM_TOKEN_KEYWORD_DO) | (1U << PM_TOKEN_KEYWORD_THEN),
7941 [PM_CONTEXT_MAIN] = (1U << PM_TOKEN_EOF),
7942 [PM_CONTEXT_MODULE] = (1U << PM_TOKEN_KEYWORD_END) | (1U << PM_TOKEN_KEYWORD_RESCUE) | (1U << PM_TOKEN_KEYWORD_ENSURE),
7943 [PM_CONTEXT_MODULE_ENSURE] = (1U << PM_TOKEN_KEYWORD_END),
7944 [PM_CONTEXT_MODULE_ELSE] = (1U << PM_TOKEN_KEYWORD_ENSURE) | (1U << PM_TOKEN_KEYWORD_END),
7945 [PM_CONTEXT_MODULE_RESCUE] = (1U << PM_TOKEN_KEYWORD_ENSURE) | (1U << PM_TOKEN_KEYWORD_RESCUE) | (1U << PM_TOKEN_KEYWORD_ELSE) | (1U << PM_TOKEN_KEYWORD_END),
7946 [PM_CONTEXT_MULTI_TARGET] = (1U << PM_TOKEN_EOF),
7947 [PM_CONTEXT_PARENS] = (1U << PM_TOKEN_PARENTHESIS_RIGHT),
7948 [PM_CONTEXT_POSTEXE] = (1U << PM_TOKEN_BRACE_RIGHT),
7949 [PM_CONTEXT_PREDICATE] = (1U << PM_TOKEN_KEYWORD_THEN) | (1U << PM_TOKEN_NEWLINE) | (1U << PM_TOKEN_SEMICOLON),
7950 [PM_CONTEXT_PREEXE] = (1U << PM_TOKEN_BRACE_RIGHT),
7951 [PM_CONTEXT_RESCUE_MODIFIER] = (1U << PM_TOKEN_EOF),
7952 [PM_CONTEXT_SCLASS] = (1U << PM_TOKEN_KEYWORD_END) | (1U << PM_TOKEN_KEYWORD_RESCUE) | (1U << PM_TOKEN_KEYWORD_ENSURE),
7953 [PM_CONTEXT_SCLASS_ENSURE] = (1U << PM_TOKEN_KEYWORD_END),
7954 [PM_CONTEXT_SCLASS_ELSE] = (1U << PM_TOKEN_KEYWORD_ENSURE) | (1U << PM_TOKEN_KEYWORD_END),
7955 [PM_CONTEXT_SCLASS_RESCUE] = (1U << PM_TOKEN_KEYWORD_ENSURE) | (1U << PM_TOKEN_KEYWORD_RESCUE) | (1U << PM_TOKEN_KEYWORD_ELSE) | (1U << PM_TOKEN_KEYWORD_END),
7956 [PM_CONTEXT_TERNARY] = (1U << PM_TOKEN_EOF),
7957 [PM_CONTEXT_UNLESS] = (1U << PM_TOKEN_KEYWORD_ELSE) | (1U << PM_TOKEN_KEYWORD_END),
7958 [PM_CONTEXT_UNTIL] = (1U << PM_TOKEN_KEYWORD_END),
7959 [PM_CONTEXT_WHILE] = (1U << PM_TOKEN_KEYWORD_END),
7960};
7961
7962static PRISM_INLINE bool
7963context_terminator(pm_context_t context, pm_token_t *token) {
7964 return token->type < 32 && (context_terminators[context] & (1U << token->type));
7965}
7966
7971static pm_context_t
7972context_recoverable(const pm_parser_t *parser, pm_token_t *token) {
7973 pm_context_node_t *context_node = parser->current_context;
7974
7975 while (context_node != NULL) {
7976 if (context_terminator(context_node->context, token)) return context_node->context;
7977 context_node = context_node->prev;
7978 }
7979
7980 return PM_CONTEXT_NONE;
7981}
7982
7983PM_STATIC_ASSERT(__LINE__, PM_CONTEXT_MAXIMUM <= 64, "Expected every context to fit in the context mask.");
7984
7985static bool
7986context_push(pm_parser_t *parser, pm_context_t context) {
7987 pm_context_node_t *context_node = (pm_context_node_t *) xmalloc(sizeof(pm_context_node_t));
7988 if (context_node == NULL) return false;
7989
7990 pm_context_node_t *previous = parser->current_context;
7991 *context_node = (pm_context_node_t) {
7992 .context = context,
7993 .prev = previous,
7994 .mask = (previous == NULL ? 0 : previous->mask) | ((uint64_t) 1 << context)
7995 };
7996
7997 parser->current_context = context_node;
7998 return true;
7999}
8000
8001static void
8002context_pop(pm_parser_t *parser) {
8003 pm_context_node_t *prev = parser->current_context->prev;
8004 xfree_sized(parser->current_context, sizeof(pm_context_node_t));
8005 parser->current_context = prev;
8006}
8007
8008static bool
8009context_p(const pm_parser_t *parser, pm_context_t context) {
8010 return parser->current_context != NULL && (parser->current_context->mask & ((uint64_t) 1 << context)) != 0;
8011}
8012
8013static bool
8014context_def_p(const pm_parser_t *parser) {
8015 pm_context_node_t *context_node = parser->current_context;
8016
8017 while (context_node != NULL) {
8018 switch (context_node->context) {
8019 case PM_CONTEXT_DEF:
8020 case PM_CONTEXT_DEF_PARAMS:
8021 case PM_CONTEXT_DEF_ENSURE:
8022 case PM_CONTEXT_DEF_RESCUE:
8023 case PM_CONTEXT_DEF_ELSE:
8024 return true;
8025 case PM_CONTEXT_CLASS:
8026 case PM_CONTEXT_CLASS_ENSURE:
8027 case PM_CONTEXT_CLASS_RESCUE:
8028 case PM_CONTEXT_CLASS_ELSE:
8029 case PM_CONTEXT_MODULE:
8030 case PM_CONTEXT_MODULE_ENSURE:
8031 case PM_CONTEXT_MODULE_RESCUE:
8032 case PM_CONTEXT_MODULE_ELSE:
8033 case PM_CONTEXT_SCLASS:
8034 case PM_CONTEXT_SCLASS_ENSURE:
8035 case PM_CONTEXT_SCLASS_RESCUE:
8036 case PM_CONTEXT_SCLASS_ELSE:
8037 return false;
8038 default:
8039 context_node = context_node->prev;
8040 }
8041 }
8042
8043 return false;
8044}
8045
8050static const char *
8051context_human(pm_context_t context) {
8052 switch (context) {
8053 case PM_CONTEXT_NONE:
8054 case PM_CONTEXT_MAXIMUM:
8055 assert(false && "unreachable");
8056 return "";
8057 case PM_CONTEXT_BEGIN: return "begin statement";
8058 case PM_CONTEXT_BLOCK_BRACES: return "'{'..'}' block";
8059 case PM_CONTEXT_BLOCK_KEYWORDS: return "'do'..'end' block";
8060 case PM_CONTEXT_BLOCK_PARAMETERS: return "'|'..'|' block parameter";
8061 case PM_CONTEXT_CASE_WHEN: return "'when' clause";
8062 case PM_CONTEXT_CASE_IN: return "'in' clause";
8063 case PM_CONTEXT_CLASS: return "class definition";
8064 case PM_CONTEXT_DEF: return "method definition";
8065 case PM_CONTEXT_DEF_PARAMS: return "method parameters";
8066 case PM_CONTEXT_DEFAULT_PARAMS: return "parameter default value";
8067 case PM_CONTEXT_DEFINED: return "'defined?' expression";
8068 case PM_CONTEXT_ELSE:
8069 case PM_CONTEXT_BEGIN_ELSE:
8070 case PM_CONTEXT_BLOCK_ELSE:
8071 case PM_CONTEXT_CLASS_ELSE:
8072 case PM_CONTEXT_DEF_ELSE:
8073 case PM_CONTEXT_LAMBDA_ELSE:
8074 case PM_CONTEXT_MODULE_ELSE:
8075 case PM_CONTEXT_SCLASS_ELSE: return "'else' clause";
8076 case PM_CONTEXT_ELSIF: return "'elsif' clause";
8077 case PM_CONTEXT_EMBEXPR: return "embedded expression";
8078 case PM_CONTEXT_BEGIN_ENSURE:
8079 case PM_CONTEXT_BLOCK_ENSURE:
8080 case PM_CONTEXT_CLASS_ENSURE:
8081 case PM_CONTEXT_DEF_ENSURE:
8082 case PM_CONTEXT_LAMBDA_ENSURE:
8083 case PM_CONTEXT_MODULE_ENSURE:
8084 case PM_CONTEXT_SCLASS_ENSURE: return "'ensure' clause";
8085 case PM_CONTEXT_FOR: return "for loop";
8086 case PM_CONTEXT_FOR_INDEX: return "for loop index";
8087 case PM_CONTEXT_IF: return "if statement";
8088 case PM_CONTEXT_LAMBDA_BRACES: return "'{'..'}' lambda block";
8089 case PM_CONTEXT_LAMBDA_DO_END: return "'do'..'end' lambda block";
8090 case PM_CONTEXT_LOOP_PREDICATE: return "loop predicate";
8091 case PM_CONTEXT_MAIN: return "top level context";
8092 case PM_CONTEXT_MODULE: return "module definition";
8093 case PM_CONTEXT_MULTI_TARGET: return "multiple targets";
8094 case PM_CONTEXT_PARENS: return "parentheses";
8095 case PM_CONTEXT_POSTEXE: return "'END' block";
8096 case PM_CONTEXT_PREDICATE: return "predicate";
8097 case PM_CONTEXT_PREEXE: return "'BEGIN' block";
8098 case PM_CONTEXT_BEGIN_RESCUE:
8099 case PM_CONTEXT_BLOCK_RESCUE:
8100 case PM_CONTEXT_CLASS_RESCUE:
8101 case PM_CONTEXT_DEF_RESCUE:
8102 case PM_CONTEXT_LAMBDA_RESCUE:
8103 case PM_CONTEXT_MODULE_RESCUE:
8104 case PM_CONTEXT_RESCUE_MODIFIER:
8105 case PM_CONTEXT_SCLASS_RESCUE: return "'rescue' clause";
8106 case PM_CONTEXT_SCLASS: return "singleton class definition";
8107 case PM_CONTEXT_TERNARY: return "ternary expression";
8108 case PM_CONTEXT_UNLESS: return "unless statement";
8109 case PM_CONTEXT_UNTIL: return "until statement";
8110 case PM_CONTEXT_WHILE: return "while statement";
8111 }
8112
8113 assert(false && "unreachable");
8114 return "";
8115}
8116
8117/******************************************************************************/
8118/* Specific token lexers */
8119/******************************************************************************/
8120
8121static PRISM_INLINE void
8122pm_strspn_number_validate(pm_parser_t *parser, const uint8_t *string, size_t length, const uint8_t *invalid) {
8123 if (invalid != NULL) {
8124 pm_diagnostic_id_t diag_id = (invalid == (string + length - 1)) ? PM_ERR_INVALID_NUMBER_UNDERSCORE_TRAILING : PM_ERR_INVALID_NUMBER_UNDERSCORE_INNER;
8125 pm_parser_err(parser, U32(invalid - parser->start), 1, diag_id);
8126 }
8127}
8128
8129static size_t
8130pm_strspn_binary_number_validate(pm_parser_t *parser, const uint8_t *string) {
8131 const uint8_t *invalid = NULL;
8132 size_t length = pm_strspn_binary_number(string, parser->end - string, &invalid);
8133 pm_strspn_number_validate(parser, string, length, invalid);
8134 return length;
8135}
8136
8137static size_t
8138pm_strspn_octal_number_validate(pm_parser_t *parser, const uint8_t *string) {
8139 const uint8_t *invalid = NULL;
8140 size_t length = pm_strspn_octal_number(string, parser->end - string, &invalid);
8141 pm_strspn_number_validate(parser, string, length, invalid);
8142 return length;
8143}
8144
8145static size_t
8146pm_strspn_decimal_number_validate(pm_parser_t *parser, const uint8_t *string) {
8147 const uint8_t *invalid = NULL;
8148 size_t length = pm_strspn_decimal_number(string, parser->end - string, &invalid);
8149 pm_strspn_number_validate(parser, string, length, invalid);
8150 return length;
8151}
8152
8153static size_t
8154pm_strspn_hexadecimal_number_validate(pm_parser_t *parser, const uint8_t *string) {
8155 const uint8_t *invalid = NULL;
8156 size_t length = pm_strspn_hexadecimal_number(string, parser->end - string, &invalid);
8157 pm_strspn_number_validate(parser, string, length, invalid);
8158 return length;
8159}
8160
8161static pm_token_type_t
8162lex_optional_float_suffix(pm_parser_t *parser, bool* seen_e) {
8163 pm_token_type_t type = PM_TOKEN_INTEGER;
8164
8165 // Here we're going to attempt to parse the optional decimal portion of a
8166 // float. If it's not there, then it's okay and we'll just continue on.
8167 if (peek(parser) == '.') {
8168 if (pm_char_is_decimal_digit(peek_offset(parser, 1))) {
8169 parser->current.end += 2;
8170 parser->current.end += pm_strspn_decimal_number_validate(parser, parser->current.end);
8171 type = PM_TOKEN_FLOAT;
8172 } else {
8173 // If we had a . and then something else, then it's not a float
8174 // suffix on a number it's a method call or something else.
8175 return type;
8176 }
8177 }
8178
8179 // Here we're going to attempt to parse the optional exponent portion of a
8180 // float. If it's not there, it's okay and we'll just continue on.
8181 if ((peek(parser) == 'e') || (peek(parser) == 'E')) {
8182 if ((peek_offset(parser, 1) == '+') || (peek_offset(parser, 1) == '-')) {
8183 parser->current.end += 2;
8184
8185 if (pm_char_is_decimal_digit(peek(parser))) {
8186 parser->current.end++;
8187 parser->current.end += pm_strspn_decimal_number_validate(parser, parser->current.end);
8188 } else {
8189 pm_parser_err_current(parser, PM_ERR_INVALID_FLOAT_EXPONENT);
8190 }
8191 } else if (pm_char_is_decimal_digit(peek_offset(parser, 1))) {
8192 parser->current.end++;
8193 parser->current.end += pm_strspn_decimal_number_validate(parser, parser->current.end);
8194 } else {
8195 return type;
8196 }
8197
8198 *seen_e = true;
8199 type = PM_TOKEN_FLOAT;
8200 }
8201
8202 return type;
8203}
8204
8205static pm_token_type_t
8206lex_numeric_prefix(pm_parser_t *parser, bool* seen_e) {
8207 pm_token_type_t type = PM_TOKEN_INTEGER;
8208 *seen_e = false;
8209
8210 if (peek_offset(parser, -1) == '0') {
8211 switch (*parser->current.end) {
8212 // 0d1111 is a decimal number
8213 case 'd':
8214 case 'D':
8215 parser->current.end++;
8216 if (pm_char_is_decimal_digit(peek(parser))) {
8217 parser->current.end += pm_strspn_decimal_number_validate(parser, parser->current.end);
8218 } else {
8219 match(parser, '_');
8220 pm_parser_err_current(parser, PM_ERR_INVALID_NUMBER_DECIMAL);
8221 }
8222
8223 break;
8224
8225 // 0b1111 is a binary number
8226 case 'b':
8227 case 'B':
8228 parser->current.end++;
8229 if (pm_char_is_binary_digit(peek(parser))) {
8230 parser->current.end += pm_strspn_binary_number_validate(parser, parser->current.end);
8231 } else {
8232 match(parser, '_');
8233 pm_parser_err_current(parser, PM_ERR_INVALID_NUMBER_BINARY);
8234 }
8235
8236 parser->integer.base = PM_INTEGER_BASE_FLAGS_BINARY;
8237 break;
8238
8239 // 0o1111 is an octal number
8240 case 'o':
8241 case 'O':
8242 parser->current.end++;
8243 if (pm_char_is_octal_digit(peek(parser))) {
8244 parser->current.end += pm_strspn_octal_number_validate(parser, parser->current.end);
8245 } else {
8246 match(parser, '_');
8247 pm_parser_err_current(parser, PM_ERR_INVALID_NUMBER_OCTAL);
8248 }
8249
8250 parser->integer.base = PM_INTEGER_BASE_FLAGS_OCTAL;
8251 break;
8252
8253 // 01111 is an octal number
8254 case '_':
8255 case '0':
8256 case '1':
8257 case '2':
8258 case '3':
8259 case '4':
8260 case '5':
8261 case '6':
8262 case '7':
8263 parser->current.end += pm_strspn_octal_number_validate(parser, parser->current.end);
8264 parser->integer.base = PM_INTEGER_BASE_FLAGS_OCTAL;
8265 break;
8266
8267 // 0x1111 is a hexadecimal number
8268 case 'x':
8269 case 'X':
8270 parser->current.end++;
8271 if (pm_char_is_hexadecimal_digit(peek(parser))) {
8272 parser->current.end += pm_strspn_hexadecimal_number_validate(parser, parser->current.end);
8273 } else {
8274 match(parser, '_');
8275 pm_parser_err_current(parser, PM_ERR_INVALID_NUMBER_HEXADECIMAL);
8276 }
8277
8278 parser->integer.base = PM_INTEGER_BASE_FLAGS_HEXADECIMAL;
8279 break;
8280
8281 // 0.xxx is a float
8282 case '.': {
8283 type = lex_optional_float_suffix(parser, seen_e);
8284 break;
8285 }
8286
8287 // 0exxx is a float
8288 case 'e':
8289 case 'E': {
8290 type = lex_optional_float_suffix(parser, seen_e);
8291 break;
8292 }
8293 }
8294 } else {
8295 // If it didn't start with a 0, then we'll lex as far as we can into a
8296 // decimal number. We compute the integer value inline to avoid
8297 // re-scanning the digits later in pm_integer_parse.
8298 {
8299 const uint8_t *cursor = parser->current.end;
8300 const uint8_t *end = parser->end;
8301 uint64_t value = (uint64_t) (cursor[-1] - '0');
8302
8303 bool has_underscore = false;
8304 bool prev_underscore = false;
8305 const uint8_t *invalid = NULL;
8306
8307 while (cursor < end) {
8308 uint8_t c = *cursor;
8309 if (c >= '0' && c <= '9') {
8310 if (value <= UINT32_MAX) value = value * 10 + (uint64_t) (c - '0');
8311 prev_underscore = false;
8312 cursor++;
8313 } else if (c == '_') {
8314 has_underscore = true;
8315 if (prev_underscore && invalid == NULL) invalid = cursor;
8316 prev_underscore = true;
8317 cursor++;
8318 } else {
8319 break;
8320 }
8321 }
8322
8323 if (has_underscore) {
8324 if (prev_underscore && invalid == NULL) invalid = cursor - 1;
8325 pm_strspn_number_validate(parser, parser->current.end, (size_t) (cursor - parser->current.end), invalid);
8326 }
8327
8328 if (value <= UINT32_MAX) {
8329 parser->integer.value = (uint32_t) value;
8330 parser->integer.lexed = true;
8331 }
8332
8333 parser->current.end = cursor;
8334 }
8335
8336 // Afterward, we'll lex as far as we can into an optional float suffix.
8337 // Guard the function call: the vast majority of decimal numbers are
8338 // plain integers, so avoid the call when the next byte cannot start a
8339 // float suffix.
8340 {
8341 uint8_t next = peek(parser);
8342 if (next == '.' || next == 'e' || next == 'E') {
8343 type = lex_optional_float_suffix(parser, seen_e);
8344
8345 // If it turned out to be a float, the cached integer value is
8346 // invalid.
8347 if (type != PM_TOKEN_INTEGER) {
8348 parser->integer.lexed = false;
8349 }
8350 }
8351 }
8352 }
8353
8354 // At this point we have a completed number, but we want to provide the user
8355 // with a good experience if they put an additional .xxx fractional
8356 // component on the end, so we'll check for that here.
8357 if (peek_offset(parser, 0) == '.' && pm_char_is_decimal_digit(peek_offset(parser, 1))) {
8358 const uint8_t *fraction_start = parser->current.end;
8359 const uint8_t *fraction_end = parser->current.end + 2;
8360 fraction_end += pm_strspn_decimal_digit(fraction_end, parser->end - fraction_end);
8361 pm_parser_err(parser, U32(fraction_start - parser->start), U32(fraction_end - fraction_start), PM_ERR_INVALID_NUMBER_FRACTION);
8362 }
8363
8364 return type;
8365}
8366
8367static pm_token_type_t
8368lex_numeric(pm_parser_t *parser) {
8369 pm_token_type_t type = PM_TOKEN_INTEGER;
8370 parser->integer.base = PM_INTEGER_BASE_FLAGS_DECIMAL;
8371 parser->integer.lexed = false;
8372
8373 if (parser->current.end < parser->end) {
8374 bool seen_e = false;
8375 type = lex_numeric_prefix(parser, &seen_e);
8376
8377 const uint8_t *end = parser->current.end;
8378 pm_token_type_t suffix_type = type;
8379
8380 if (type == PM_TOKEN_INTEGER) {
8381 if (match(parser, 'r')) {
8382 suffix_type = PM_TOKEN_INTEGER_RATIONAL;
8383
8384 if (match(parser, 'i')) {
8385 suffix_type = PM_TOKEN_INTEGER_RATIONAL_IMAGINARY;
8386 }
8387 } else if (match(parser, 'i')) {
8388 suffix_type = PM_TOKEN_INTEGER_IMAGINARY;
8389 }
8390 } else {
8391 if (!seen_e && match(parser, 'r')) {
8392 suffix_type = PM_TOKEN_FLOAT_RATIONAL;
8393
8394 if (match(parser, 'i')) {
8395 suffix_type = PM_TOKEN_FLOAT_RATIONAL_IMAGINARY;
8396 }
8397 } else if (match(parser, 'i')) {
8398 suffix_type = PM_TOKEN_FLOAT_IMAGINARY;
8399 }
8400 }
8401
8402 const uint8_t b = peek(parser);
8403 if (b != '\0' && (b >= 0x80 || ((b >= 'a' && b <= 'z') || (b >= 'A' && b <= 'Z')) || b == '_')) {
8404 parser->current.end = end;
8405 } else {
8406 type = suffix_type;
8407 }
8408 }
8409
8410 return type;
8411}
8412
8413static pm_token_type_t
8414lex_global_variable(pm_parser_t *parser) {
8415 if (parser->current.end >= parser->end) {
8416 pm_parser_err_token(parser, &parser->current, PM_ERR_GLOBAL_VARIABLE_BARE);
8417 return PM_TOKEN_GLOBAL_VARIABLE;
8418 }
8419
8420 // True if multiple characters are allowed after the declaration of the
8421 // global variable. Not true when it starts with "$-".
8422 bool allow_multiple = true;
8423
8424 switch (*parser->current.end) {
8425 case '~': // $~: match-data
8426 case '*': // $*: argv
8427 case '$': // $$: pid
8428 case '?': // $?: last status
8429 case '!': // $!: error string
8430 case '@': // $@: error position
8431 case '/': // $/: input record separator
8432 case '\\': // $\: output record separator
8433 case ';': // $;: field separator
8434 case ',': // $,: output field separator
8435 case '.': // $.: last read line number
8436 case '=': // $=: ignorecase
8437 case ':': // $:: load path
8438 case '<': // $<: reading filename
8439 case '>': // $>: default output handle
8440 case '\"': // $": already loaded files
8441 parser->current.end++;
8442 return PM_TOKEN_GLOBAL_VARIABLE;
8443
8444 case '&': // $&: last match
8445 case '`': // $`: string before last match
8446 case '\'': // $': string after last match
8447 case '+': // $+: string matches last paren.
8448 parser->current.end++;
8449 return lex_state_p(parser, PM_LEX_STATE_FNAME) ? PM_TOKEN_GLOBAL_VARIABLE : PM_TOKEN_BACK_REFERENCE;
8450
8451 case '0': {
8452 parser->current.end++;
8453 size_t width;
8454
8455 if ((width = char_is_identifier(parser, parser->current.end, parser->end - parser->current.end)) > 0) {
8456 do {
8457 parser->current.end += width;
8458 } while ((width = char_is_identifier(parser, parser->current.end, parser->end - parser->current.end)) > 0);
8459
8460 // $0 isn't allowed to be followed by anything.
8461 pm_diagnostic_id_t diag_id = parser->version <= PM_OPTIONS_VERSION_CRUBY_3_3 ? PM_ERR_INVALID_VARIABLE_GLOBAL_3_3 : PM_ERR_INVALID_VARIABLE_GLOBAL;
8462 PM_PARSER_ERR_TOKEN_FORMAT_CONTENT(parser, &parser->current, diag_id);
8463 }
8464
8465 return PM_TOKEN_GLOBAL_VARIABLE;
8466 }
8467
8468 case '1':
8469 case '2':
8470 case '3':
8471 case '4':
8472 case '5':
8473 case '6':
8474 case '7':
8475 case '8':
8476 case '9':
8477 parser->current.end += pm_strspn_decimal_digit(parser->current.end, parser->end - parser->current.end);
8478 return lex_state_p(parser, PM_LEX_STATE_FNAME) ? PM_TOKEN_GLOBAL_VARIABLE : PM_TOKEN_NUMBERED_REFERENCE;
8479
8480 case '-':
8481 parser->current.end++;
8482 allow_multiple = false;
8484 default: {
8485 size_t width;
8486
8487 if ((width = char_is_identifier(parser, parser->current.end, parser->end - parser->current.end)) > 0) {
8488 do {
8489 parser->current.end += width;
8490 } while (allow_multiple && (width = char_is_identifier(parser, parser->current.end, parser->end - parser->current.end)) > 0);
8491 } else if (pm_char_is_whitespace(peek(parser))) {
8492 // If we get here, then we have a $ followed by whitespace,
8493 // which is not allowed.
8494 pm_parser_err_token(parser, &parser->current, PM_ERR_GLOBAL_VARIABLE_BARE);
8495 } else {
8496 // If we get here, then we have a $ followed by something that
8497 // isn't recognized as a global variable.
8498 pm_diagnostic_id_t diag_id = parser->version <= PM_OPTIONS_VERSION_CRUBY_3_3 ? PM_ERR_INVALID_VARIABLE_GLOBAL_3_3 : PM_ERR_INVALID_VARIABLE_GLOBAL;
8499 size_t width = parser->encoding->char_width(parser->current.end, parser->end - parser->current.end);
8500 PM_PARSER_ERR_FORMAT(parser, PM_TOKEN_START(parser, &parser->current), PM_TOKEN_LENGTH(&parser->current) + U32(width), diag_id, (int) (PM_TOKEN_LENGTH(&parser->current) + U32(width)), (const char *) parser->current.start);
8501 }
8502
8503 return PM_TOKEN_GLOBAL_VARIABLE;
8504 }
8505 }
8506}
8507
8520static PRISM_INLINE pm_token_type_t
8521lex_keyword(pm_parser_t *parser, const uint8_t *current_start, const char *value, size_t vlen, pm_lex_state_t state, pm_token_type_t type, pm_token_type_t modifier_type) {
8522 if (memcmp(current_start, value, vlen) == 0) {
8523 pm_lex_state_t last_state = parser->lex_state;
8524
8525 if (parser->lex_state & PM_LEX_STATE_FNAME) {
8526 lex_state_set(parser, PM_LEX_STATE_ENDFN);
8527 } else {
8528 lex_state_set(parser, state);
8529 if (state == PM_LEX_STATE_BEG) {
8530 parser->command_start = true;
8531 }
8532
8533 if ((modifier_type != PM_TOKEN_EOF) && !(last_state & (PM_LEX_STATE_BEG | PM_LEX_STATE_LABELED | PM_LEX_STATE_CLASS))) {
8534 lex_state_set(parser, PM_LEX_STATE_BEG | PM_LEX_STATE_LABEL);
8535 return modifier_type;
8536 }
8537 }
8538
8539 return type;
8540 }
8541
8542 return PM_TOKEN_EOF;
8543}
8544
8545static pm_token_type_t
8546lex_identifier(pm_parser_t *parser, bool previous_command_start) {
8547 // Lex as far as we can into the current identifier.
8548 size_t width;
8549 const uint8_t *end = parser->end;
8550 const uint8_t *current_start = parser->current.start;
8551 const uint8_t *current_end = parser->current.end;
8552 bool encoding_changed = parser->encoding_changed;
8553
8554 if (encoding_changed) {
8555 while ((width = char_is_identifier(parser, current_end, end - current_end)) > 0) {
8556 current_end += width;
8557 }
8558 } else {
8559 // Fast path: scan ASCII identifier bytes using wide operations.
8560 current_end += scan_identifier_ascii(current_end, end);
8561
8562 // Byte-at-a-time fallback for the tail and any UTF-8 sequences.
8563 while ((width = char_is_identifier_utf8(current_end, end - current_end)) > 0) {
8564 current_end += width;
8565 }
8566 }
8567 parser->current.end = current_end;
8568
8569 // Now cache the length of the identifier so that we can quickly compare it
8570 // against known keywords.
8571 width = (size_t) (current_end - current_start);
8572
8573 if (current_end < end) {
8574 if (((current_end + 1 >= end) || (current_end[1] != '=')) && (match(parser, '!') || match(parser, '?'))) {
8575 // First we'll attempt to extend the identifier by a ! or ?. Then we'll
8576 // check if we're returning the defined? keyword or just an identifier.
8577 width++;
8578
8579 if (
8580 ((lex_state_p(parser, PM_LEX_STATE_LABEL | PM_LEX_STATE_ENDFN) && !previous_command_start) || lex_state_arg_p(parser)) &&
8581 (peek(parser) == ':') && (peek_offset(parser, 1) != ':')
8582 ) {
8583 // If we're in a position where we can accept a : at the end of an
8584 // identifier, then we'll optionally accept it.
8585 lex_state_set(parser, PM_LEX_STATE_ARG | PM_LEX_STATE_LABELED);
8586 (void) match(parser, ':');
8587
8588 /* A label is a symbol lexed inline rather than through a lex
8589 * mode, so it clears the encoding here. */
8590 parser->explicit_encoding = NULL;
8591 return PM_TOKEN_LABEL;
8592 }
8593
8594 if (parser->lex_state != PM_LEX_STATE_DOT) {
8595 if (width == 8 && (lex_keyword(parser, current_start, "defined?", width, PM_LEX_STATE_ARG, PM_TOKEN_KEYWORD_DEFINED, PM_TOKEN_EOF) != PM_TOKEN_EOF)) {
8596 return PM_TOKEN_KEYWORD_DEFINED;
8597 }
8598 }
8599
8600 return PM_TOKEN_METHOD_NAME;
8601 }
8602
8603 if (lex_state_p(parser, PM_LEX_STATE_FNAME) && peek_offset(parser, 1) != '~' && peek_offset(parser, 1) != '>' && (peek_offset(parser, 1) != '=' || peek_offset(parser, 2) == '>') && match(parser, '=')) {
8604 // If we're in a position where we can accept a = at the end of an
8605 // identifier, then we'll optionally accept it.
8606 return PM_TOKEN_IDENTIFIER;
8607 }
8608
8609 if (
8610 ((lex_state_p(parser, PM_LEX_STATE_LABEL | PM_LEX_STATE_ENDFN) && !previous_command_start) || lex_state_arg_p(parser)) &&
8611 peek(parser) == ':' && peek_offset(parser, 1) != ':'
8612 ) {
8613 // If we're in a position where we can accept a : at the end of an
8614 // identifier, then we'll optionally accept it.
8615 lex_state_set(parser, PM_LEX_STATE_ARG | PM_LEX_STATE_LABELED);
8616 (void) match(parser, ':');
8617
8618 /* A label is a symbol lexed inline rather than through a lex
8619 * mode, so it clears the encoding here. */
8620 parser->explicit_encoding = NULL;
8621 return PM_TOKEN_LABEL;
8622 }
8623 }
8624
8625 if (parser->lex_state != PM_LEX_STATE_DOT) {
8626 pm_token_type_t type;
8627
8628 /* The lex state from before lex_keyword transitions it, mirroring the
8629 * `state = p->lex.state` capture in parse.y's keyword handling. */
8630 pm_lex_state_t previous_lex_state = parser->lex_state;
8631
8632 switch (width) {
8633 case 2:
8634 if (lex_keyword(parser, current_start, "do", width, PM_LEX_STATE_BEG, PM_TOKEN_KEYWORD_DO, PM_TOKEN_EOF) != PM_TOKEN_EOF) {
8635 /* In FNAME position (a symbol like `:do` or a method name
8636 * like `def do`), `do` is a plain name rather than a
8637 * block, loop, or lambda opener, so none of the
8638 * discrimination below applies. This mirrors parse.y,
8639 * whose EXPR_FNAME early-return precedes all of the
8640 * keyword_do special-casing (and never touches
8641 * lpar_beg). */
8642 if (previous_lex_state & PM_LEX_STATE_FNAME) {
8643 return PM_TOKEN_KEYWORD_DO;
8644 }
8645 if (parser->enclosure_nesting == parser->lambda_enclosure_nesting) {
8646 // At the bare nesting level of a lambda literal (no
8647 // delimiter opened since `->`), a `do` opens the lambda
8648 // body. This is a distinct token so that a command in a
8649 // parameter default cannot consume it as its own block
8650 // (`-> a = foo do end` is `->(a = foo) do end`). It
8651 // mirrors CRuby's keyword_do_LAMBDA.
8652 //
8653 // Clear the nesting so that no token within the
8654 // `do`/`end` body is considered to be at the beginning
8655 // of a lambda; the parser restores the enclosing value
8656 // once the lambda has been fully parsed. This mirrors
8657 // parse.y setting `p->lex.lpar_beg = -1` when lexing
8658 // keyword_do_LAMBDA.
8659 parser->lambda_enclosure_nesting = -1;
8660 return PM_TOKEN_KEYWORD_DO_LAMBDA;
8661 }
8662 if (pm_do_loop_stack_p(parser)) {
8663 return PM_TOKEN_KEYWORD_DO_LOOP;
8664 }
8665 if (!pm_accepts_block_stack_p(parser)) {
8666 return PM_TOKEN_KEYWORD_DO_BLOCK;
8667 }
8668 return PM_TOKEN_KEYWORD_DO;
8669 }
8670
8671 if ((type = lex_keyword(parser, current_start, "if", width, PM_LEX_STATE_BEG, PM_TOKEN_KEYWORD_IF, PM_TOKEN_KEYWORD_IF_MODIFIER)) != PM_TOKEN_EOF) return type;
8672 if ((type = lex_keyword(parser, current_start, "in", width, PM_LEX_STATE_BEG, PM_TOKEN_KEYWORD_IN, PM_TOKEN_EOF)) != PM_TOKEN_EOF) return type;
8673 if ((type = lex_keyword(parser, current_start, "or", width, PM_LEX_STATE_BEG, PM_TOKEN_KEYWORD_OR, PM_TOKEN_EOF)) != PM_TOKEN_EOF) return type;
8674 break;
8675 case 3:
8676 if ((type = lex_keyword(parser, current_start, "and", width, PM_LEX_STATE_BEG, PM_TOKEN_KEYWORD_AND, PM_TOKEN_EOF)) != PM_TOKEN_EOF) return type;
8677 if ((type = lex_keyword(parser, current_start, "def", width, PM_LEX_STATE_FNAME, PM_TOKEN_KEYWORD_DEF, PM_TOKEN_EOF)) != PM_TOKEN_EOF) return type;
8678 if ((type = lex_keyword(parser, current_start, "end", width, PM_LEX_STATE_END, PM_TOKEN_KEYWORD_END, PM_TOKEN_EOF)) != PM_TOKEN_EOF) return type;
8679 if ((type = lex_keyword(parser, current_start, "END", width, PM_LEX_STATE_END, PM_TOKEN_KEYWORD_END_UPCASE, PM_TOKEN_EOF)) != PM_TOKEN_EOF) return type;
8680 if ((type = lex_keyword(parser, current_start, "for", width, PM_LEX_STATE_BEG, PM_TOKEN_KEYWORD_FOR, PM_TOKEN_EOF)) != PM_TOKEN_EOF) return type;
8681 if ((type = lex_keyword(parser, current_start, "nil", width, PM_LEX_STATE_END, PM_TOKEN_KEYWORD_NIL, PM_TOKEN_EOF)) != PM_TOKEN_EOF) return type;
8682 if ((type = lex_keyword(parser, current_start, "not", width, PM_LEX_STATE_ARG, PM_TOKEN_KEYWORD_NOT, PM_TOKEN_EOF)) != PM_TOKEN_EOF) return type;
8683 break;
8684 case 4:
8685 if ((type = lex_keyword(parser, current_start, "case", width, PM_LEX_STATE_BEG, PM_TOKEN_KEYWORD_CASE, PM_TOKEN_EOF)) != PM_TOKEN_EOF) return type;
8686 if ((type = lex_keyword(parser, current_start, "else", width, PM_LEX_STATE_BEG, PM_TOKEN_KEYWORD_ELSE, PM_TOKEN_EOF)) != PM_TOKEN_EOF) return type;
8687 if ((type = lex_keyword(parser, current_start, "next", width, PM_LEX_STATE_MID, PM_TOKEN_KEYWORD_NEXT, PM_TOKEN_EOF)) != PM_TOKEN_EOF) return type;
8688 if ((type = lex_keyword(parser, current_start, "redo", width, PM_LEX_STATE_END, PM_TOKEN_KEYWORD_REDO, PM_TOKEN_EOF)) != PM_TOKEN_EOF) return type;
8689 if ((type = lex_keyword(parser, current_start, "self", width, PM_LEX_STATE_END, PM_TOKEN_KEYWORD_SELF, PM_TOKEN_EOF)) != PM_TOKEN_EOF) return type;
8690 if ((type = lex_keyword(parser, current_start, "then", width, PM_LEX_STATE_BEG, PM_TOKEN_KEYWORD_THEN, PM_TOKEN_EOF)) != PM_TOKEN_EOF) return type;
8691 if ((type = lex_keyword(parser, current_start, "true", width, PM_LEX_STATE_END, PM_TOKEN_KEYWORD_TRUE, PM_TOKEN_EOF)) != PM_TOKEN_EOF) return type;
8692 if ((type = lex_keyword(parser, current_start, "when", width, PM_LEX_STATE_BEG, PM_TOKEN_KEYWORD_WHEN, PM_TOKEN_EOF)) != PM_TOKEN_EOF) return type;
8693 break;
8694 case 5:
8695 if ((type = lex_keyword(parser, current_start, "alias", width, PM_LEX_STATE_FNAME | PM_LEX_STATE_FITEM, PM_TOKEN_KEYWORD_ALIAS, PM_TOKEN_EOF)) != PM_TOKEN_EOF) return type;
8696 if ((type = lex_keyword(parser, current_start, "begin", width, PM_LEX_STATE_BEG, PM_TOKEN_KEYWORD_BEGIN, PM_TOKEN_EOF)) != PM_TOKEN_EOF) return type;
8697 if ((type = lex_keyword(parser, current_start, "BEGIN", width, PM_LEX_STATE_END, PM_TOKEN_KEYWORD_BEGIN_UPCASE, PM_TOKEN_EOF)) != PM_TOKEN_EOF) return type;
8698 if ((type = lex_keyword(parser, current_start, "break", width, PM_LEX_STATE_MID, PM_TOKEN_KEYWORD_BREAK, PM_TOKEN_EOF)) != PM_TOKEN_EOF) return type;
8699 if ((type = lex_keyword(parser, current_start, "class", width, PM_LEX_STATE_CLASS, PM_TOKEN_KEYWORD_CLASS, PM_TOKEN_EOF)) != PM_TOKEN_EOF) return type;
8700 if ((type = lex_keyword(parser, current_start, "elsif", width, PM_LEX_STATE_BEG, PM_TOKEN_KEYWORD_ELSIF, PM_TOKEN_EOF)) != PM_TOKEN_EOF) return type;
8701 if ((type = lex_keyword(parser, current_start, "false", width, PM_LEX_STATE_END, PM_TOKEN_KEYWORD_FALSE, PM_TOKEN_EOF)) != PM_TOKEN_EOF) return type;
8702 if ((type = lex_keyword(parser, current_start, "retry", width, PM_LEX_STATE_END, PM_TOKEN_KEYWORD_RETRY, PM_TOKEN_EOF)) != PM_TOKEN_EOF) return type;
8703 if ((type = lex_keyword(parser, current_start, "super", width, PM_LEX_STATE_ARG, PM_TOKEN_KEYWORD_SUPER, PM_TOKEN_EOF)) != PM_TOKEN_EOF) return type;
8704 if ((type = lex_keyword(parser, current_start, "undef", width, PM_LEX_STATE_FNAME | PM_LEX_STATE_FITEM, PM_TOKEN_KEYWORD_UNDEF, PM_TOKEN_EOF)) != PM_TOKEN_EOF) return type;
8705 if ((type = lex_keyword(parser, current_start, "until", width, PM_LEX_STATE_BEG, PM_TOKEN_KEYWORD_UNTIL, PM_TOKEN_KEYWORD_UNTIL_MODIFIER)) != PM_TOKEN_EOF) return type;
8706 if ((type = lex_keyword(parser, current_start, "while", width, PM_LEX_STATE_BEG, PM_TOKEN_KEYWORD_WHILE, PM_TOKEN_KEYWORD_WHILE_MODIFIER)) != PM_TOKEN_EOF) return type;
8707 if ((type = lex_keyword(parser, current_start, "yield", width, PM_LEX_STATE_ARG, PM_TOKEN_KEYWORD_YIELD, PM_TOKEN_EOF)) != PM_TOKEN_EOF) return type;
8708 break;
8709 case 6:
8710 if ((type = lex_keyword(parser, current_start, "ensure", width, PM_LEX_STATE_BEG, PM_TOKEN_KEYWORD_ENSURE, PM_TOKEN_EOF)) != PM_TOKEN_EOF) return type;
8711 if ((type = lex_keyword(parser, current_start, "module", width, PM_LEX_STATE_BEG, PM_TOKEN_KEYWORD_MODULE, PM_TOKEN_EOF)) != PM_TOKEN_EOF) return type;
8712 if ((type = lex_keyword(parser, current_start, "rescue", width, PM_LEX_STATE_MID, PM_TOKEN_KEYWORD_RESCUE, PM_TOKEN_KEYWORD_RESCUE_MODIFIER)) != PM_TOKEN_EOF) return type;
8713 if ((type = lex_keyword(parser, current_start, "return", width, PM_LEX_STATE_MID, PM_TOKEN_KEYWORD_RETURN, PM_TOKEN_EOF)) != PM_TOKEN_EOF) return type;
8714 if ((type = lex_keyword(parser, current_start, "unless", width, PM_LEX_STATE_BEG, PM_TOKEN_KEYWORD_UNLESS, PM_TOKEN_KEYWORD_UNLESS_MODIFIER)) != PM_TOKEN_EOF) return type;
8715 break;
8716 case 8:
8717 if ((type = lex_keyword(parser, current_start, "__LINE__", width, PM_LEX_STATE_END, PM_TOKEN_KEYWORD___LINE__, PM_TOKEN_EOF)) != PM_TOKEN_EOF) return type;
8718 if ((type = lex_keyword(parser, current_start, "__FILE__", width, PM_LEX_STATE_END, PM_TOKEN_KEYWORD___FILE__, PM_TOKEN_EOF)) != PM_TOKEN_EOF) return type;
8719 break;
8720 case 12:
8721 if ((type = lex_keyword(parser, current_start, "__ENCODING__", width, PM_LEX_STATE_END, PM_TOKEN_KEYWORD___ENCODING__, PM_TOKEN_EOF)) != PM_TOKEN_EOF) return type;
8722 break;
8723 }
8724 }
8725
8726 if (encoding_changed) {
8727 return parser->encoding->isupper_char(current_start, end - current_start) ? PM_TOKEN_CONSTANT : PM_TOKEN_IDENTIFIER;
8728 }
8729
8730 /* Identifiers usually start with an ASCII byte, for which the uppercase
8731 * check is a simple range comparison. This avoids the call into the
8732 * encoding module for every identifier. */
8733 if (*current_start < 0x80) {
8734 return (*current_start >= 'A' && *current_start <= 'Z') ? PM_TOKEN_CONSTANT : PM_TOKEN_IDENTIFIER;
8735 }
8736 return pm_encoding_utf_8_isupper_char(current_start, end - current_start) ? PM_TOKEN_CONSTANT : PM_TOKEN_IDENTIFIER;
8737}
8738
8743static bool
8744current_token_starts_line(pm_parser_t *parser) {
8745 return (parser->current.start == parser->start) || (parser->current.start[-1] == '\n');
8746}
8747
8762static pm_token_type_t
8763lex_interpolation(pm_parser_t *parser, const uint8_t *pound) {
8764 // If there is no content following this #, then we're at the end of
8765 // the string and we can safely return string content.
8766 if (pound + 1 >= parser->end) {
8767 parser->current.end = pound + 1;
8768 return PM_TOKEN_STRING_CONTENT;
8769 }
8770
8771 // Now we'll check against the character that follows the #. If it
8772 // constitutes valid interplation, we'll handle that, otherwise we'll return
8773 // 0.
8774 switch (pound[1]) {
8775 case '@': {
8776 // In this case we may have hit an embedded instance or class variable.
8777 if (pound + 2 >= parser->end) {
8778 parser->current.end = pound + 1;
8779 return PM_TOKEN_STRING_CONTENT;
8780 }
8781
8782 // If we're looking at a @ and there's another @, then we'll skip past the
8783 // second @.
8784 const uint8_t *variable = pound + 2;
8785 if (*variable == '@' && pound + 3 < parser->end) variable++;
8786
8787 if (char_is_identifier_start(parser, variable, parser->end - variable)) {
8788 // At this point we're sure that we've either hit an embedded instance
8789 // or class variable. In this case we'll first need to check if we've
8790 // already consumed content.
8791 if (pound > parser->current.start) {
8792 parser->current.end = pound;
8793 return PM_TOKEN_STRING_CONTENT;
8794 }
8795
8796 // Otherwise we need to return the embedded variable token
8797 // and then switch to the embedded variable lex mode.
8798 lex_mode_push(parser, (pm_lex_mode_t) { .mode = PM_LEX_EMBVAR });
8799 parser->current.end = pound + 1;
8800 return PM_TOKEN_EMBVAR;
8801 }
8802
8803 // If we didn't get a valid interpolation, then this is just regular
8804 // string content. This is like if we get "#@-". In this case the caller
8805 // should keep lexing.
8806 parser->current.end = pound + 1;
8807 return 0;
8808 }
8809 case '$':
8810 // In this case we may have hit an embedded global variable. If there's
8811 // not enough room, then we'll just return string content.
8812 if (pound + 2 >= parser->end) {
8813 parser->current.end = pound + 1;
8814 return PM_TOKEN_STRING_CONTENT;
8815 }
8816
8817 // This is the character that we're going to check to see if it is the
8818 // start of an identifier that would indicate that this is a global
8819 // variable.
8820 const uint8_t *check = pound + 2;
8821
8822 if (pound[2] == '-') {
8823 if (pound + 3 >= parser->end) {
8824 parser->current.end = pound + 2;
8825 return PM_TOKEN_STRING_CONTENT;
8826 }
8827
8828 check++;
8829 }
8830
8831 // If the character that we're going to check is the start of an
8832 // identifier, or we don't have a - and the character is a decimal number
8833 // or a global name punctuation character, then we've hit an embedded
8834 // global variable.
8835 if (
8836 char_is_identifier_start(parser, check, parser->end - check) ||
8837 (pound[2] != '-' && (pm_char_is_decimal_digit(pound[2]) || char_is_global_name_punctuation(pound[2])))
8838 ) {
8839 // In this case we've hit an embedded global variable. First check to
8840 // see if we've already consumed content. If we have, then we need to
8841 // return that content as string content first.
8842 if (pound > parser->current.start) {
8843 parser->current.end = pound;
8844 return PM_TOKEN_STRING_CONTENT;
8845 }
8846
8847 // Otherwise, we need to return the embedded variable token and switch
8848 // to the embedded variable lex mode.
8849 lex_mode_push(parser, (pm_lex_mode_t) { .mode = PM_LEX_EMBVAR });
8850 parser->current.end = pound + 1;
8851 return PM_TOKEN_EMBVAR;
8852 }
8853
8854 // In this case we've hit a #$ that does not indicate a global variable.
8855 // In this case we'll continue lexing past it.
8856 parser->current.end = pound + 1;
8857 return 0;
8858 case '{':
8859 // In this case it's the start of an embedded expression. If we have
8860 // already consumed content, then we need to return that content as string
8861 // content first.
8862 if (pound > parser->current.start) {
8863 parser->current.end = pound;
8864 return PM_TOKEN_STRING_CONTENT;
8865 }
8866
8867 parser->enclosure_nesting++;
8868
8869 // Otherwise we'll skip past the #{ and begin lexing the embedded
8870 // expression.
8871 lex_mode_push(parser, (pm_lex_mode_t) { .mode = PM_LEX_EMBEXPR });
8872 parser->current.end = pound + 2;
8873 parser->command_start = true;
8874 pm_enclosure_frame_push(parser);
8875 return PM_TOKEN_EMBEXPR_BEGIN;
8876 default:
8877 // In this case we've hit a # that doesn't constitute interpolation. We'll
8878 // mark that by returning the not provided token type. This tells the
8879 // consumer to keep lexing forward.
8880 parser->current.end = pound + 1;
8881 return 0;
8882 }
8883}
8884
8885static const uint8_t PM_ESCAPE_FLAG_NONE = 0x0;
8886static const uint8_t PM_ESCAPE_FLAG_CONTROL = 0x1;
8887static const uint8_t PM_ESCAPE_FLAG_META = 0x2;
8888static const uint8_t PM_ESCAPE_FLAG_SINGLE = 0x4;
8889static const uint8_t PM_ESCAPE_FLAG_REGEXP = 0x8;
8890
8894static const bool ascii_printable_chars[] = {
8895 0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 0, 0,
8896 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
8897 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1,
8898 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1,
8899 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1,
8900 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 1, 1, 1,
8901 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1,
8902 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0
8903};
8904
8905static PRISM_INLINE bool
8906char_is_ascii_printable(const uint8_t b) {
8907 return (b < 0x80) && ascii_printable_chars[b];
8908}
8909
8914static PRISM_INLINE uint8_t
8915escape_hexadecimal_digit(const uint8_t value) {
8916 return (uint8_t) ((value <= '9') ? (value - '0') : (value & 0x7) + 9);
8917}
8918
8924static PRISM_INLINE uint32_t
8925escape_unicode(pm_parser_t *parser, const uint8_t *string, size_t length, const pm_location_t *error_location, const uint8_t flags) {
8926 uint32_t value = 0;
8927 for (size_t index = 0; index < length; index++) {
8928 if (index != 0) value <<= 4;
8929 value |= escape_hexadecimal_digit(string[index]);
8930 }
8931
8932 // Here we're going to verify that the value is actually a valid Unicode
8933 // codepoint and not a surrogate pair.
8934 if (value >= 0xD800 && value <= 0xDFFF) {
8935 if (flags & PM_ESCAPE_FLAG_REGEXP) {
8936 // In regexp context, defer the error to regexp encoding
8937 // validation where we can produce a regexp-specific message.
8938 } else if (error_location != NULL) {
8939 pm_parser_err(parser, error_location->start, error_location->length, PM_ERR_ESCAPE_INVALID_UNICODE);
8940 } else {
8941 pm_parser_err(parser, U32(string - parser->start), U32(length), PM_ERR_ESCAPE_INVALID_UNICODE);
8942 }
8943 return 0xFFFD;
8944 }
8945
8946 return value;
8947}
8948
8952static PRISM_INLINE uint8_t
8953escape_byte(uint8_t value, const uint8_t flags) {
8954 if (flags & PM_ESCAPE_FLAG_CONTROL) value &= 0x9f;
8955 if (flags & PM_ESCAPE_FLAG_META) value |= 0x80;
8956 return value;
8957}
8958
8962static PRISM_INLINE void
8963escape_write_unicode(pm_parser_t *parser, pm_buffer_t *buffer, const uint8_t flags, const uint8_t *start, const uint8_t *end, uint32_t value) {
8964 // \u escape sequences in string-like structures implicitly change the
8965 // encoding to UTF-8 if they are >= 0x80 or if they are used in a character
8966 // literal.
8967 if (value >= 0x80 || flags & PM_ESCAPE_FLAG_SINGLE) {
8968 if (parser->explicit_encoding != NULL && parser->explicit_encoding != PM_ENCODING_UTF_8_ENTRY) {
8969 if (flags & PM_ESCAPE_FLAG_REGEXP) {
8970 // In regexp context, suppress this error — the regexp encoding
8971 // validation will produce a more specific error message.
8972 } else {
8973 PM_PARSER_ERR_FORMAT(parser, U32(start - parser->start), U32(end - start), PM_ERR_MIXED_ENCODING, parser->explicit_encoding->name);
8974 }
8975 }
8976
8977 parser->explicit_encoding = PM_ENCODING_UTF_8_ENTRY;
8978 }
8979
8980 if (!pm_buffer_append_unicode_codepoint(buffer, value)) {
8981 if (flags & PM_ESCAPE_FLAG_REGEXP) {
8982 // In regexp context, defer the error to the regexp encoding
8983 // validation which produces a regexp-specific message.
8984 } else {
8985 pm_parser_err(parser, U32(start - parser->start), U32(end - start), PM_ERR_ESCAPE_INVALID_UNICODE);
8986 }
8987
8988 pm_buffer_append_byte(buffer, 0xEF);
8989 pm_buffer_append_byte(buffer, 0xBF);
8990 pm_buffer_append_byte(buffer, 0xBD);
8991 }
8992}
8993
8998static PRISM_INLINE void
8999escape_write_byte_encoded(pm_parser_t *parser, pm_buffer_t *buffer, const uint8_t flags, uint8_t byte) {
9000 if (byte >= 0x80) {
9001 if (parser->explicit_encoding != NULL && parser->explicit_encoding == PM_ENCODING_UTF_8_ENTRY && parser->encoding != PM_ENCODING_UTF_8_ENTRY) {
9002 if (flags & PM_ESCAPE_FLAG_REGEXP) {
9003 // In regexp context, suppress this error — the regexp encoding
9004 // validation will produce a more specific error message.
9005 } else {
9006 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->current, PM_ERR_MIXED_ENCODING, parser->encoding->name);
9007 }
9008 }
9009
9010 parser->explicit_encoding = parser->encoding;
9011 }
9012
9013 pm_buffer_append_byte(buffer, byte);
9014}
9015
9031static PRISM_INLINE void
9032escape_write_byte(pm_parser_t *parser, pm_buffer_t *buffer, pm_buffer_t *regular_expression_buffer, uint8_t flags, uint8_t byte) {
9033 if (flags & PM_ESCAPE_FLAG_REGEXP) {
9034 pm_buffer_append_format(regular_expression_buffer, "\\x%02X", byte);
9035 }
9036
9037 escape_write_byte_encoded(parser, buffer, flags, byte);
9038}
9039
9043static PRISM_INLINE void
9044escape_write_escape_encoded(pm_parser_t *parser, pm_buffer_t *buffer, pm_buffer_t *regular_expression_buffer, uint8_t flags) {
9045 size_t width;
9046 if (parser->encoding_changed) {
9047 width = parser->encoding->char_width(parser->current.end, parser->end - parser->current.end);
9048 } else {
9049 width = pm_encoding_utf_8_char_width(parser->current.end, parser->end - parser->current.end);
9050 }
9051
9052 if (width == 1) {
9053 if (parser->heredoc_end == NULL && *parser->current.end == '\n') pm_line_offset_list_append(&parser->metadata_arena, &parser->line_offsets, PM_TOKEN_END(parser, &parser->current) + 1);
9054 escape_write_byte(parser, buffer, regular_expression_buffer, flags, escape_byte(*parser->current.end++, flags));
9055 } else if (width > 1) {
9056 // Valid multibyte character. Just ignore escape.
9057 pm_buffer_t *b = (flags & PM_ESCAPE_FLAG_REGEXP) ? regular_expression_buffer : buffer;
9058 pm_buffer_append_bytes(b, parser->current.end, width);
9059 parser->current.end += width;
9060 } else {
9061 // Assume the next character wasn't meant to be part of this escape
9062 // sequence since it is invalid. Add an error and move on.
9063 parser->current.end++;
9064 pm_parser_err_current(parser, PM_ERR_ESCAPE_INVALID_CONTROL);
9065 }
9066}
9067
9073static void
9074escape_read_warn(pm_parser_t *parser, uint8_t flags, uint8_t flag, const char *type) {
9075#define FLAG(value) ((value & PM_ESCAPE_FLAG_CONTROL) ? "\\C-" : (value & PM_ESCAPE_FLAG_META) ? "\\M-" : "")
9076
9077 PM_PARSER_WARN_TOKEN_FORMAT(
9078 parser,
9079 &parser->current,
9080 PM_WARN_INVALID_CHARACTER,
9081 FLAG(flags),
9082 FLAG(flag),
9083 type
9084 );
9085
9086#undef FLAG
9087}
9088
9092static void
9093escape_read(pm_parser_t *parser, pm_buffer_t *buffer, pm_buffer_t *regular_expression_buffer, uint8_t flags) {
9094 uint8_t peeked = peek(parser);
9095 switch (peeked) {
9096 case '\\': {
9097 parser->current.end++;
9098 escape_write_byte(parser, buffer, regular_expression_buffer, flags, escape_byte('\\', flags));
9099 return;
9100 }
9101 case '\'': {
9102 parser->current.end++;
9103 escape_write_byte(parser, buffer, regular_expression_buffer, flags, escape_byte('\'', flags));
9104 return;
9105 }
9106 case 'a': {
9107 parser->current.end++;
9108 escape_write_byte(parser, buffer, regular_expression_buffer, flags, escape_byte('\a', flags));
9109 return;
9110 }
9111 case 'b': {
9112 parser->current.end++;
9113 escape_write_byte(parser, buffer, regular_expression_buffer, flags, escape_byte('\b', flags));
9114 return;
9115 }
9116 case 'e': {
9117 parser->current.end++;
9118 escape_write_byte(parser, buffer, regular_expression_buffer, flags, escape_byte('\033', flags));
9119 return;
9120 }
9121 case 'f': {
9122 parser->current.end++;
9123 escape_write_byte(parser, buffer, regular_expression_buffer, flags, escape_byte('\f', flags));
9124 return;
9125 }
9126 case 'n': {
9127 parser->current.end++;
9128 escape_write_byte(parser, buffer, regular_expression_buffer, flags, escape_byte('\n', flags));
9129 return;
9130 }
9131 case 'r': {
9132 parser->current.end++;
9133 escape_write_byte(parser, buffer, regular_expression_buffer, flags, escape_byte('\r', flags));
9134 return;
9135 }
9136 case 's': {
9137 parser->current.end++;
9138 escape_write_byte(parser, buffer, regular_expression_buffer, flags, escape_byte(' ', flags));
9139 return;
9140 }
9141 case 't': {
9142 parser->current.end++;
9143 escape_write_byte(parser, buffer, regular_expression_buffer, flags, escape_byte('\t', flags));
9144 return;
9145 }
9146 case 'v': {
9147 parser->current.end++;
9148 escape_write_byte(parser, buffer, regular_expression_buffer, flags, escape_byte('\v', flags));
9149 return;
9150 }
9151 case '0': case '1': case '2': case '3': case '4': case '5': case '6': case '7': {
9152 uint8_t value = (uint8_t) (*parser->current.end - '0');
9153 parser->current.end++;
9154
9155 if (pm_char_is_octal_digit(peek(parser))) {
9156 value = ((uint8_t) (value << 3)) | ((uint8_t) (*parser->current.end - '0'));
9157 parser->current.end++;
9158
9159 if (pm_char_is_octal_digit(peek(parser))) {
9160 value = ((uint8_t) (value << 3)) | ((uint8_t) (*parser->current.end - '0'));
9161 parser->current.end++;
9162 }
9163 }
9164
9165 value = escape_byte(value, flags);
9166 escape_write_byte(parser, buffer, regular_expression_buffer, flags, value);
9167 return;
9168 }
9169 case 'x': {
9170 const uint8_t *start = parser->current.end - 1;
9171
9172 parser->current.end++;
9173 uint8_t byte = peek(parser);
9174
9175 if (pm_char_is_hexadecimal_digit(byte)) {
9176 uint8_t value = escape_hexadecimal_digit(byte);
9177 parser->current.end++;
9178
9179 byte = peek(parser);
9180 if (pm_char_is_hexadecimal_digit(byte)) {
9181 value = (uint8_t) ((value << 4) | escape_hexadecimal_digit(byte));
9182 parser->current.end++;
9183 }
9184
9185 value = escape_byte(value, flags);
9186 if (flags & PM_ESCAPE_FLAG_REGEXP) {
9187 if (flags & (PM_ESCAPE_FLAG_CONTROL | PM_ESCAPE_FLAG_META)) {
9188 pm_buffer_append_format(regular_expression_buffer, "\\x%02X", value);
9189 } else {
9190 pm_buffer_append_bytes(regular_expression_buffer, start, (size_t) (parser->current.end - start));
9191 }
9192 }
9193
9194 escape_write_byte_encoded(parser, buffer, flags, value);
9195 } else {
9196 pm_parser_err_current(parser, PM_ERR_ESCAPE_INVALID_HEXADECIMAL);
9197 }
9198
9199 return;
9200 }
9201 case 'u': {
9202 const uint8_t *start = parser->current.end - 1;
9203 parser->current.end++;
9204
9205 if (parser->current.end == parser->end) {
9206 const uint8_t *start = parser->current.end - 2;
9207 PM_PARSER_ERR_FORMAT(parser, U32(start - parser->start), U32(parser->current.end - start), PM_ERR_ESCAPE_INVALID_UNICODE_SHORT, 2, start);
9208 } else if (peek(parser) == '{') {
9209 const uint8_t *unicode_codepoints_start = parser->current.end - 2;
9210 parser->current.end++;
9211
9212 size_t whitespace;
9213 while (true) {
9214 if ((whitespace = pm_strspn_inline_whitespace(parser->current.end, parser->end - parser->current.end)) > 0) {
9215 parser->current.end += whitespace;
9216 } else if (peek(parser) == '\\' && peek_offset(parser, 1) == 'n') {
9217 // This is super hacky, but it gets us nicer error
9218 // messages because we can still pass it off to the
9219 // regular expression engine even if we hit an
9220 // unterminated regular expression.
9221 parser->current.end += 2;
9222 } else {
9223 break;
9224 }
9225 }
9226
9227 const uint8_t *extra_codepoints_start = NULL;
9228 int codepoints_count = 0;
9229
9230 while ((parser->current.end < parser->end) && (*parser->current.end != '}')) {
9231 const uint8_t *unicode_start = parser->current.end;
9232 size_t hexadecimal_length = pm_strspn_hexadecimal_digit(parser->current.end, parser->end - parser->current.end);
9233
9234 if (hexadecimal_length > 6) {
9235 // \u{nnnn} character literal allows only 1-6 hexadecimal digits
9236 pm_parser_err(parser, U32(unicode_start - parser->start), U32(hexadecimal_length), PM_ERR_ESCAPE_INVALID_UNICODE_LONG);
9237 } else if (hexadecimal_length == 0) {
9238 // there are not hexadecimal characters
9239
9240 if (flags & PM_ESCAPE_FLAG_REGEXP) {
9241 // If this is a regular expression, we are going to
9242 // let the regular expression engine handle this
9243 // error instead of us because we don't know at this
9244 // point if we're inside a comment in /x mode.
9245 pm_buffer_append_bytes(regular_expression_buffer, start, (size_t) (parser->current.end - start));
9246 } else {
9247 pm_parser_err(parser, PM_TOKEN_END(parser, &parser->current), 0, PM_ERR_ESCAPE_INVALID_UNICODE);
9248 pm_parser_err(parser, PM_TOKEN_END(parser, &parser->current), 0, PM_ERR_ESCAPE_INVALID_UNICODE_TERM);
9249 }
9250
9251 return;
9252 }
9253
9254 parser->current.end += hexadecimal_length;
9255 codepoints_count++;
9256 if (flags & PM_ESCAPE_FLAG_SINGLE && codepoints_count == 2) {
9257 extra_codepoints_start = unicode_start;
9258 }
9259
9260 uint32_t value = escape_unicode(parser, unicode_start, hexadecimal_length, NULL, flags);
9261 escape_write_unicode(parser, buffer, flags, unicode_start, parser->current.end, value);
9262
9263 parser->current.end += pm_strspn_inline_whitespace(parser->current.end, parser->end - parser->current.end);
9264 }
9265
9266 // ?\u{nnnn} character literal should contain only one codepoint
9267 // and cannot be like ?\u{nnnn mmmm}.
9268 if (flags & PM_ESCAPE_FLAG_SINGLE && codepoints_count > 1) {
9269 pm_parser_err(parser, U32(extra_codepoints_start - parser->start), U32(parser->current.end - 1 - extra_codepoints_start), PM_ERR_ESCAPE_INVALID_UNICODE_LITERAL);
9270 }
9271
9272 if (parser->current.end == parser->end) {
9273 PM_PARSER_ERR_FORMAT(parser, U32(start - parser->start), U32(parser->current.end - start), PM_ERR_ESCAPE_INVALID_UNICODE_LIST, (int) (parser->current.end - start), start);
9274 } else if (peek(parser) == '}') {
9275 parser->current.end++;
9276 } else {
9277 if (flags & PM_ESCAPE_FLAG_REGEXP) {
9278 // If this is a regular expression, we are going to let
9279 // the regular expression engine handle this error
9280 // instead of us because we don't know at this point if
9281 // we're inside a comment in /x mode.
9282 pm_buffer_append_bytes(regular_expression_buffer, start, (size_t) (parser->current.end - start));
9283 } else {
9284 pm_parser_err(parser, U32(unicode_codepoints_start - parser->start), U32(parser->current.end - unicode_codepoints_start), PM_ERR_ESCAPE_INVALID_UNICODE_TERM);
9285 }
9286 }
9287
9288 if (flags & PM_ESCAPE_FLAG_REGEXP) {
9289 pm_buffer_append_bytes(regular_expression_buffer, unicode_codepoints_start, (size_t) (parser->current.end - unicode_codepoints_start));
9290 }
9291 } else {
9292 size_t length = pm_strspn_hexadecimal_digit(parser->current.end, MIN(parser->end - parser->current.end, 4));
9293
9294 if (length == 0) {
9295 if (flags & PM_ESCAPE_FLAG_REGEXP) {
9296 pm_buffer_append_bytes(regular_expression_buffer, start, (size_t) (parser->current.end - start));
9297 } else {
9298 const uint8_t *start = parser->current.end - 2;
9299 PM_PARSER_ERR_FORMAT(parser, U32(start - parser->start), U32(parser->current.end - start), PM_ERR_ESCAPE_INVALID_UNICODE_SHORT, 2, start);
9300 }
9301 } else if (length == 4) {
9302 uint32_t value = escape_unicode(parser, parser->current.end, 4, NULL, flags);
9303
9304 if (flags & PM_ESCAPE_FLAG_REGEXP) {
9305 pm_buffer_append_bytes(regular_expression_buffer, start, (size_t) (parser->current.end + 4 - start));
9306 }
9307
9308 escape_write_unicode(parser, buffer, flags, start, parser->current.end + 4, value);
9309 parser->current.end += 4;
9310 } else {
9311 parser->current.end += length;
9312
9313 if (flags & PM_ESCAPE_FLAG_REGEXP) {
9314 // If this is a regular expression, we are going to let
9315 // the regular expression engine handle this error
9316 // instead of us.
9317 pm_buffer_append_bytes(regular_expression_buffer, start, (size_t) (parser->current.end - start));
9318 } else {
9319 pm_parser_err_current(parser, PM_ERR_ESCAPE_INVALID_UNICODE);
9320 }
9321 }
9322 }
9323
9324 return;
9325 }
9326 case 'c': {
9327 parser->current.end++;
9328 if (flags & PM_ESCAPE_FLAG_CONTROL) {
9329 pm_parser_err_current(parser, PM_ERR_ESCAPE_INVALID_CONTROL_REPEAT);
9330 }
9331
9332 if (parser->current.end == parser->end) {
9333 pm_parser_err_current(parser, PM_ERR_ESCAPE_INVALID_CONTROL);
9334 return;
9335 }
9336
9337 uint8_t peeked = peek(parser);
9338 switch (peeked) {
9339 case '?': {
9340 parser->current.end++;
9341 escape_write_byte(parser, buffer, regular_expression_buffer, flags, escape_byte(0x7f, flags));
9342 return;
9343 }
9344 case '\\':
9345 parser->current.end++;
9346
9347 if (match(parser, 'u') || match(parser, 'U')) {
9348 pm_parser_err(parser, PM_TOKEN_START(parser, &parser->current), PM_TOKEN_LENGTH(&parser->current), PM_ERR_INVALID_ESCAPE_CHARACTER);
9349 return;
9350 }
9351
9352 escape_read(parser, buffer, regular_expression_buffer, flags | PM_ESCAPE_FLAG_CONTROL);
9353 return;
9354 case ' ':
9355 parser->current.end++;
9356 escape_read_warn(parser, flags, PM_ESCAPE_FLAG_CONTROL, "\\s");
9357 escape_write_byte(parser, buffer, regular_expression_buffer, flags, escape_byte(peeked, flags | PM_ESCAPE_FLAG_CONTROL));
9358 return;
9359 case '\t':
9360 parser->current.end++;
9361 escape_read_warn(parser, flags, 0, "\\t");
9362 escape_write_byte(parser, buffer, regular_expression_buffer, flags, escape_byte(peeked, flags | PM_ESCAPE_FLAG_CONTROL));
9363 return;
9364 default: {
9365 if (!char_is_ascii_printable(peeked)) {
9366 pm_parser_err_current(parser, PM_ERR_ESCAPE_INVALID_CONTROL);
9367 return;
9368 }
9369
9370 if (parser->heredoc_end == NULL && peeked == '\n') pm_line_offset_list_append(&parser->metadata_arena, &parser->line_offsets, PM_TOKEN_END(parser, &parser->current) + 1);
9371 parser->current.end++;
9372 escape_write_byte(parser, buffer, regular_expression_buffer, flags, escape_byte(peeked, flags | PM_ESCAPE_FLAG_CONTROL));
9373 return;
9374 }
9375 }
9376 }
9377 case 'C': {
9378 parser->current.end++;
9379 if (flags & PM_ESCAPE_FLAG_CONTROL) {
9380 pm_parser_err_current(parser, PM_ERR_ESCAPE_INVALID_CONTROL_REPEAT);
9381 }
9382
9383 if (peek(parser) != '-') {
9384 size_t width = parser->encoding->char_width(parser->current.end, parser->end - parser->current.end);
9385 pm_parser_err(parser, PM_TOKEN_START(parser, &parser->current), PM_TOKEN_LENGTH(&parser->current) + U32(width), PM_ERR_ESCAPE_INVALID_CONTROL);
9386 return;
9387 }
9388
9389 parser->current.end++;
9390 if (parser->current.end == parser->end) {
9391 pm_parser_err_current(parser, PM_ERR_ESCAPE_INVALID_CONTROL);
9392 return;
9393 }
9394
9395 uint8_t peeked = peek(parser);
9396 switch (peeked) {
9397 case '?': {
9398 parser->current.end++;
9399 escape_write_byte(parser, buffer, regular_expression_buffer, flags, escape_byte(0x7f, flags));
9400 return;
9401 }
9402 case '\\':
9403 parser->current.end++;
9404
9405 if (match(parser, 'u') || match(parser, 'U')) {
9406 pm_parser_err(parser, PM_TOKEN_START(parser, &parser->current), PM_TOKEN_LENGTH(&parser->current), PM_ERR_INVALID_ESCAPE_CHARACTER);
9407 return;
9408 }
9409
9410 escape_read(parser, buffer, regular_expression_buffer, flags | PM_ESCAPE_FLAG_CONTROL);
9411 return;
9412 case ' ':
9413 parser->current.end++;
9414 escape_read_warn(parser, flags, PM_ESCAPE_FLAG_CONTROL, "\\s");
9415 escape_write_byte(parser, buffer, regular_expression_buffer, flags, escape_byte(peeked, flags | PM_ESCAPE_FLAG_CONTROL));
9416 return;
9417 case '\t':
9418 parser->current.end++;
9419 escape_read_warn(parser, flags, 0, "\\t");
9420 escape_write_byte(parser, buffer, regular_expression_buffer, flags, escape_byte(peeked, flags | PM_ESCAPE_FLAG_CONTROL));
9421 return;
9422 default: {
9423 if (!char_is_ascii_printable(peeked)) {
9424 size_t width = parser->encoding->char_width(parser->current.end, parser->end - parser->current.end);
9425 pm_parser_err(parser, PM_TOKEN_START(parser, &parser->current), PM_TOKEN_LENGTH(&parser->current) + U32(width), PM_ERR_ESCAPE_INVALID_CONTROL);
9426 return;
9427 }
9428
9429 if (parser->heredoc_end == NULL && peeked == '\n') pm_line_offset_list_append(&parser->metadata_arena, &parser->line_offsets, PM_TOKEN_END(parser, &parser->current) + 1);
9430 parser->current.end++;
9431 escape_write_byte(parser, buffer, regular_expression_buffer, flags, escape_byte(peeked, flags | PM_ESCAPE_FLAG_CONTROL));
9432 return;
9433 }
9434 }
9435 }
9436 case 'M': {
9437 parser->current.end++;
9438 if (flags & PM_ESCAPE_FLAG_META) {
9439 pm_parser_err_current(parser, PM_ERR_ESCAPE_INVALID_META_REPEAT);
9440 }
9441
9442 if (peek(parser) != '-') {
9443 size_t width = parser->encoding->char_width(parser->current.end, parser->end - parser->current.end);
9444 pm_parser_err(parser, PM_TOKEN_START(parser, &parser->current), PM_TOKEN_LENGTH(&parser->current) + U32(width), PM_ERR_ESCAPE_INVALID_META);
9445 return;
9446 }
9447
9448 parser->current.end++;
9449 if (parser->current.end == parser->end) {
9450 pm_parser_err_current(parser, PM_ERR_ESCAPE_INVALID_META);
9451 return;
9452 }
9453
9454 uint8_t peeked = peek(parser);
9455 switch (peeked) {
9456 case '\\':
9457 parser->current.end++;
9458
9459 if (match(parser, 'u') || match(parser, 'U')) {
9460 pm_parser_err(parser, PM_TOKEN_START(parser, &parser->current), PM_TOKEN_LENGTH(&parser->current), PM_ERR_INVALID_ESCAPE_CHARACTER);
9461 return;
9462 }
9463
9464 escape_read(parser, buffer, regular_expression_buffer, flags | PM_ESCAPE_FLAG_META);
9465 return;
9466 case ' ':
9467 parser->current.end++;
9468 escape_read_warn(parser, flags, PM_ESCAPE_FLAG_META, "\\s");
9469 escape_write_byte(parser, buffer, regular_expression_buffer, flags, escape_byte(peeked, flags | PM_ESCAPE_FLAG_META));
9470 return;
9471 case '\t':
9472 parser->current.end++;
9473 escape_read_warn(parser, flags & ((uint8_t) ~PM_ESCAPE_FLAG_CONTROL), PM_ESCAPE_FLAG_META, "\\t");
9474 escape_write_byte(parser, buffer, regular_expression_buffer, flags, escape_byte(peeked, flags | PM_ESCAPE_FLAG_META));
9475 return;
9476 default:
9477 if (!char_is_ascii_printable(peeked)) {
9478 size_t width = parser->encoding->char_width(parser->current.end, parser->end - parser->current.end);
9479 pm_parser_err(parser, PM_TOKEN_START(parser, &parser->current), PM_TOKEN_LENGTH(&parser->current) + U32(width), PM_ERR_ESCAPE_INVALID_META);
9480 return;
9481 }
9482
9483 if (parser->heredoc_end == NULL && peeked == '\n') pm_line_offset_list_append(&parser->metadata_arena, &parser->line_offsets, PM_TOKEN_END(parser, &parser->current) + 1);
9484 parser->current.end++;
9485 escape_write_byte(parser, buffer, regular_expression_buffer, flags, escape_byte(peeked, flags | PM_ESCAPE_FLAG_META));
9486 return;
9487 }
9488 }
9489 case '\r': {
9490 if (peek_offset(parser, 1) == '\n') {
9491 if (parser->heredoc_end == NULL) pm_line_offset_list_append(&parser->metadata_arena, &parser->line_offsets, PM_TOKEN_END(parser, &parser->current) + 2);
9492 parser->current.end += 2;
9493 escape_write_byte_encoded(parser, buffer, flags, escape_byte('\n', flags));
9494 return;
9495 }
9497 }
9498 default: {
9499 if ((flags & (PM_ESCAPE_FLAG_CONTROL | PM_ESCAPE_FLAG_META)) && !char_is_ascii_printable(peeked)) {
9500 size_t width = parser->encoding->char_width(parser->current.end, parser->end - parser->current.end);
9501 pm_parser_err(parser, PM_TOKEN_START(parser, &parser->current), PM_TOKEN_LENGTH(&parser->current) + U32(width), PM_ERR_ESCAPE_INVALID_META);
9502 return;
9503 }
9504 if (parser->current.end < parser->end) {
9505 escape_write_escape_encoded(parser, buffer, regular_expression_buffer, flags);
9506 } else {
9507 pm_parser_err_current(parser, PM_ERR_INVALID_ESCAPE_CHARACTER);
9508 }
9509 return;
9510 }
9511 }
9512}
9513
9539static pm_token_type_t
9540lex_question_mark(pm_parser_t *parser) {
9541 if (lex_state_end_p(parser)) {
9542 lex_state_set(parser, PM_LEX_STATE_BEG);
9543 return PM_TOKEN_QUESTION_MARK;
9544 }
9545
9546 /*
9547 * A literal takes its encoding from its own contents. Literals that push a
9548 * lex mode clear this in lex_mode_push_*; a character literal is lexed
9549 * inline, so it clears the encoding here.
9550 */
9551 parser->explicit_encoding = NULL;
9552
9553 if (parser->current.end >= parser->end) {
9554 pm_parser_err_current(parser, PM_ERR_INCOMPLETE_QUESTION_MARK);
9555 pm_string_shared_init(&parser->current_string, parser->current.start + 1, parser->current.end);
9556 return PM_TOKEN_CHARACTER_LITERAL;
9557 }
9558
9559 if (pm_char_is_whitespace(*parser->current.end)) {
9560 lex_state_set(parser, PM_LEX_STATE_BEG);
9561 return PM_TOKEN_QUESTION_MARK;
9562 }
9563
9564 lex_state_set(parser, PM_LEX_STATE_BEG);
9565
9566 if (match(parser, '\\')) {
9567 lex_state_set(parser, PM_LEX_STATE_END);
9568
9569 pm_buffer_t buffer;
9570 pm_buffer_init(&buffer, 3);
9571
9572 escape_read(parser, &buffer, NULL, PM_ESCAPE_FLAG_SINGLE);
9573
9574 // Copy buffer data into the arena and free the heap buffer.
9575 void *arena_data = pm_arena_memdup(parser->arena, buffer.value, buffer.length, PRISM_ALIGNOF(uint8_t));
9576 pm_string_constant_init(&parser->current_string, (const char *) arena_data, buffer.length);
9577 pm_buffer_cleanup(&buffer);
9578
9579 return PM_TOKEN_CHARACTER_LITERAL;
9580 } else {
9581 size_t encoding_width = parser->encoding->char_width(parser->current.end, parser->end - parser->current.end);
9582
9583 // Ternary operators can have a ? immediately followed by an identifier
9584 // which starts with an underscore. We check for this case here.
9585 if (
9586 !(parser->encoding->alnum_char(parser->current.end, parser->end - parser->current.end) || peek(parser) == '_') ||
9587 (
9588 (parser->current.end + encoding_width >= parser->end) ||
9589 !char_is_identifier(parser, parser->current.end + encoding_width, parser->end - (parser->current.end + encoding_width))
9590 )
9591 ) {
9592 lex_state_set(parser, PM_LEX_STATE_END);
9593 parser->current.end += encoding_width;
9594 pm_string_shared_init(&parser->current_string, parser->current.start + 1, parser->current.end);
9595 return PM_TOKEN_CHARACTER_LITERAL;
9596 }
9597 }
9598
9599 return PM_TOKEN_QUESTION_MARK;
9600}
9601
9606static pm_token_type_t
9607lex_at_variable(pm_parser_t *parser) {
9608 pm_token_type_t type = match(parser, '@') ? PM_TOKEN_CLASS_VARIABLE : PM_TOKEN_INSTANCE_VARIABLE;
9609 const uint8_t *end = parser->end;
9610
9611 size_t width;
9612 if ((width = char_is_identifier_start(parser, parser->current.end, end - parser->current.end)) > 0) {
9613 parser->current.end += width;
9614
9615 while ((width = char_is_identifier(parser, parser->current.end, end - parser->current.end)) > 0) {
9616 parser->current.end += width;
9617 }
9618 } else if (parser->current.end < end && pm_char_is_decimal_digit(*parser->current.end)) {
9619 pm_diagnostic_id_t diag_id = (type == PM_TOKEN_CLASS_VARIABLE) ? PM_ERR_INCOMPLETE_VARIABLE_CLASS : PM_ERR_INCOMPLETE_VARIABLE_INSTANCE;
9620 if (parser->version <= PM_OPTIONS_VERSION_CRUBY_3_3) {
9621 diag_id = (type == PM_TOKEN_CLASS_VARIABLE) ? PM_ERR_INCOMPLETE_VARIABLE_CLASS_3_3 : PM_ERR_INCOMPLETE_VARIABLE_INSTANCE_3_3;
9622 }
9623
9624 size_t width = parser->encoding->char_width(parser->current.end, end - parser->current.end);
9625 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->current, diag_id, (int) ((parser->current.end + width) - parser->current.start), (const char *) parser->current.start);
9626 } else {
9627 pm_diagnostic_id_t diag_id = (type == PM_TOKEN_CLASS_VARIABLE) ? PM_ERR_CLASS_VARIABLE_BARE : PM_ERR_INSTANCE_VARIABLE_BARE;
9628 pm_parser_err_token(parser, &parser->current, diag_id);
9629 }
9630
9631 // If we're lexing an embedded variable, then we need to pop back into the
9632 // parent lex context.
9633 if (parser->lex_modes.current->mode == PM_LEX_EMBVAR) {
9634 lex_mode_pop(parser);
9635 }
9636
9637 return type;
9638}
9639
9643static PRISM_INLINE void
9644parser_lex_callback(pm_parser_t *parser) {
9645 if (parser->lex_callback.callback) {
9646 parser->lex_callback.callback(parser, &parser->current, parser->lex_callback.data);
9647 }
9648}
9649
9654parser_comment(pm_parser_t *parser, pm_comment_type_t type) {
9655 pm_comment_t *comment = (pm_comment_t *) pm_arena_alloc(&parser->metadata_arena, sizeof(pm_comment_t), PRISM_ALIGNOF(pm_comment_t));
9656
9657 *comment = (pm_comment_t) {
9658 .type = type,
9659 .location = TOK2LOC(parser, &parser->current)
9660 };
9661
9662 return comment;
9663}
9664
9670static pm_token_type_t
9671lex_embdoc(pm_parser_t *parser) {
9672 // First, lex out the EMBDOC_BEGIN token.
9673 const uint8_t *newline = next_newline(parser->current.end, parser->end - parser->current.end);
9674
9675 if (newline == NULL) {
9676 parser->current.end = parser->end;
9677 } else {
9678 pm_line_offset_list_append(&parser->metadata_arena, &parser->line_offsets, U32(newline - parser->start + 1));
9679 parser->current.end = newline + 1;
9680 }
9681
9682 parser->current.type = PM_TOKEN_EMBDOC_BEGIN;
9683 parser_lex_callback(parser);
9684
9685 // Now, create a comment that is going to be attached to the parser.
9686 const uint8_t *comment_start = parser->current.start;
9687 pm_comment_t *comment = parser_comment(parser, PM_COMMENT_EMBDOC);
9688
9689 // Now, loop until we find the end of the embedded documentation or the end
9690 // of the file.
9691 while (parser->current.end + 4 <= parser->end) {
9692 parser->current.start = parser->current.end;
9693
9694 // If we've hit the end of the embedded documentation then we'll return
9695 // that token here.
9696 if (
9697 (memcmp(parser->current.end, "=end", 4) == 0) &&
9698 (
9699 (parser->current.end + 4 == parser->end) || // end of file
9700 pm_char_is_whitespace(parser->current.end[4]) || // whitespace
9701 (parser->current.end[4] == '\0') || // NUL or end of script
9702 (parser->current.end[4] == '\004') || // ^D
9703 (parser->current.end[4] == '\032') // ^Z
9704 )
9705 ) {
9706 const uint8_t *newline = next_newline(parser->current.end, parser->end - parser->current.end);
9707
9708 if (newline == NULL) {
9709 parser->current.end = parser->end;
9710 } else {
9711 pm_line_offset_list_append(&parser->metadata_arena, &parser->line_offsets, U32(newline - parser->start + 1));
9712 parser->current.end = newline + 1;
9713 }
9714
9715 parser->current.type = PM_TOKEN_EMBDOC_END;
9716 parser_lex_callback(parser);
9717
9718 comment->location.length = (uint32_t) (parser->current.end - comment_start);
9719 pm_list_append(&parser->comment_list, (pm_list_node_t *) comment);
9720
9721 return PM_TOKEN_EMBDOC_END;
9722 }
9723
9724 // Otherwise, we'll parse until the end of the line and return a line of
9725 // embedded documentation.
9726 const uint8_t *newline = next_newline(parser->current.end, parser->end - parser->current.end);
9727
9728 if (newline == NULL) {
9729 parser->current.end = parser->end;
9730 } else {
9731 pm_line_offset_list_append(&parser->metadata_arena, &parser->line_offsets, U32(newline - parser->start + 1));
9732 parser->current.end = newline + 1;
9733 }
9734
9735 parser->current.type = PM_TOKEN_EMBDOC_LINE;
9736 parser_lex_callback(parser);
9737 }
9738
9739 pm_parser_err_current(parser, PM_ERR_EMBDOC_TERM);
9740
9741 comment->location.length = (uint32_t) (parser->current.end - comment_start);
9742 pm_list_append(&parser->comment_list, (pm_list_node_t *) comment);
9743
9744 return PM_TOKEN_EOF;
9745}
9746
9752static PRISM_INLINE void
9753parser_lex_ignored_newline(pm_parser_t *parser) {
9754 parser->current.type = PM_TOKEN_IGNORED_NEWLINE;
9755 parser_lex_callback(parser);
9756}
9757
9767static PRISM_INLINE void
9768parser_flush_heredoc_end(pm_parser_t *parser) {
9769 assert(parser->heredoc_end <= parser->end);
9770 parser->next_start = parser->heredoc_end;
9771 parser->heredoc_end = NULL;
9772}
9773
9777static bool
9778parser_end_of_line_p(const pm_parser_t *parser) {
9779 const uint8_t *cursor = parser->current.end;
9780
9781 while (cursor < parser->end && *cursor != '\n' && *cursor != '#') {
9782 if (!pm_char_is_inline_whitespace(*cursor++)) return false;
9783 }
9784
9785 return true;
9786}
9787
9806typedef struct {
9812
9817 const uint8_t *cursor;
9819
9839
9843static PRISM_INLINE void
9844pm_token_buffer_push_byte(pm_token_buffer_t *token_buffer, uint8_t byte) {
9845 pm_buffer_append_byte(&token_buffer->buffer, byte);
9846}
9847
9848static PRISM_INLINE void
9849pm_regexp_token_buffer_push_byte(pm_regexp_token_buffer_t *token_buffer, uint8_t byte) {
9850 pm_buffer_append_byte(&token_buffer->regexp_buffer, byte);
9851}
9852
9856static PRISM_INLINE size_t
9857parser_char_width(const pm_parser_t *parser) {
9858 size_t width;
9859 if (parser->encoding_changed) {
9860 width = parser->encoding->char_width(parser->current.end, parser->end - parser->current.end);
9861 } else {
9862 width = pm_encoding_utf_8_char_width(parser->current.end, parser->end - parser->current.end);
9863 }
9864
9865 // TODO: If the character is invalid in the given encoding, then we'll just
9866 // push one byte into the buffer. This should actually be an error.
9867 return (width == 0 ? 1 : width);
9868}
9869
9873static void
9874pm_token_buffer_push_escaped(pm_token_buffer_t *token_buffer, pm_parser_t *parser) {
9875 size_t width = parser_char_width(parser);
9876 pm_buffer_append_bytes(&token_buffer->buffer, parser->current.end, width);
9877 parser->current.end += width;
9878}
9879
9880static void
9881pm_regexp_token_buffer_push_escaped(pm_regexp_token_buffer_t *token_buffer, pm_parser_t *parser) {
9882 size_t width = parser_char_width(parser);
9883 const uint8_t *start = parser->current.end;
9884 pm_buffer_append_bytes(&token_buffer->base.buffer, start, width);
9885 pm_buffer_append_bytes(&token_buffer->regexp_buffer, start, width);
9886 parser->current.end += width;
9887}
9888
9895static PRISM_INLINE void
9896pm_token_buffer_copy(pm_parser_t *parser, pm_token_buffer_t *token_buffer) {
9897 // Copy buffer data into the arena and free the heap buffer.
9898 size_t len = pm_buffer_length(&token_buffer->buffer);
9899 void *arena_data = pm_arena_memdup(parser->arena, pm_buffer_value(&token_buffer->buffer), len, PRISM_ALIGNOF(uint8_t));
9900 pm_string_constant_init(&parser->current_string, (const char *) arena_data, len);
9901 pm_buffer_cleanup(&token_buffer->buffer);
9902}
9903
9904static PRISM_INLINE void
9905pm_regexp_token_buffer_copy(pm_parser_t *parser, pm_regexp_token_buffer_t *token_buffer) {
9906 pm_token_buffer_copy(parser, &token_buffer->base);
9907 pm_buffer_cleanup(&token_buffer->regexp_buffer);
9908}
9909
9919static void
9920pm_token_buffer_flush(pm_parser_t *parser, pm_token_buffer_t *token_buffer) {
9921 if (token_buffer->cursor == NULL) {
9922 pm_string_shared_init(&parser->current_string, parser->current.start, parser->current.end);
9923 } else {
9924 pm_buffer_append_bytes(&token_buffer->buffer, token_buffer->cursor, (size_t) (parser->current.end - token_buffer->cursor));
9925 pm_token_buffer_copy(parser, token_buffer);
9926 }
9927}
9928
9929static void
9930pm_regexp_token_buffer_flush(pm_parser_t *parser, pm_regexp_token_buffer_t *token_buffer) {
9931 if (token_buffer->base.cursor == NULL) {
9932 pm_string_shared_init(&parser->current_string, parser->current.start, parser->current.end);
9933 } else {
9934 const uint8_t *cursor = token_buffer->base.cursor;
9935 size_t length = (size_t) (parser->current.end - cursor);
9936 pm_buffer_append_bytes(&token_buffer->base.buffer, cursor, length);
9937 pm_buffer_append_bytes(&token_buffer->regexp_buffer, cursor, length);
9938 pm_regexp_token_buffer_copy(parser, token_buffer);
9939 }
9940}
9941
9942#define PM_TOKEN_BUFFER_DEFAULT_SIZE 16
9943
9952static void
9953pm_token_buffer_escape(pm_parser_t *parser, pm_token_buffer_t *token_buffer) {
9954 const uint8_t *start;
9955 if (token_buffer->cursor == NULL) {
9956 pm_buffer_init(&token_buffer->buffer, PM_TOKEN_BUFFER_DEFAULT_SIZE);
9957 start = parser->current.start;
9958 } else {
9959 start = token_buffer->cursor;
9960 }
9961
9962 const uint8_t *end = parser->current.end - 1;
9963 assert(end >= start);
9964 pm_buffer_append_bytes(&token_buffer->buffer, start, (size_t) (end - start));
9965
9966 token_buffer->cursor = end;
9967}
9968
9969static void
9970pm_regexp_token_buffer_escape(pm_parser_t *parser, pm_regexp_token_buffer_t *token_buffer) {
9971 const uint8_t *start;
9972 if (token_buffer->base.cursor == NULL) {
9973 pm_buffer_init(&token_buffer->base.buffer, PM_TOKEN_BUFFER_DEFAULT_SIZE);
9974 pm_buffer_init(&token_buffer->regexp_buffer, PM_TOKEN_BUFFER_DEFAULT_SIZE);
9975 start = parser->current.start;
9976 } else {
9977 start = token_buffer->base.cursor;
9978 }
9979
9980 const uint8_t *end = parser->current.end - 1;
9981 pm_buffer_append_bytes(&token_buffer->base.buffer, start, (size_t) (end - start));
9982 pm_buffer_append_bytes(&token_buffer->regexp_buffer, start, (size_t) (end - start));
9983
9984 token_buffer->base.cursor = end;
9985}
9986
9987#undef PM_TOKEN_BUFFER_DEFAULT_SIZE
9988
9993static PRISM_INLINE size_t
9994pm_heredoc_strspn_inline_whitespace(pm_parser_t *parser, const uint8_t **cursor, pm_heredoc_indent_t indent) {
9995 size_t whitespace = 0;
9996
9997 switch (indent) {
9998 case PM_HEREDOC_INDENT_NONE:
9999 // Do nothing, we can't match a terminator with
10000 // indentation and there's no need to calculate common
10001 // whitespace.
10002 break;
10003 case PM_HEREDOC_INDENT_DASH:
10004 // Skip past inline whitespace.
10005 *cursor += pm_strspn_inline_whitespace(*cursor, parser->end - *cursor);
10006 break;
10007 case PM_HEREDOC_INDENT_TILDE:
10008 // Skip past inline whitespace and calculate common
10009 // whitespace.
10010 while (*cursor < parser->end && pm_char_is_inline_whitespace(**cursor)) {
10011 if (**cursor == '\t') {
10012 whitespace = (whitespace / PM_TAB_WHITESPACE_SIZE + 1) * PM_TAB_WHITESPACE_SIZE;
10013 } else {
10014 whitespace++;
10015 }
10016 (*cursor)++;
10017 }
10018
10019 break;
10020 }
10021
10022 return whitespace;
10023}
10024
10029static uint8_t
10030pm_lex_percent_delimiter(pm_parser_t *parser) {
10031 size_t eol_length = match_eol(parser);
10032
10033 if (eol_length) {
10034 if (parser->heredoc_end) {
10035 // If we have already lexed a heredoc, then the newline has already
10036 // been added to the list. In this case we want to just flush the
10037 // heredoc end.
10038 parser_flush_heredoc_end(parser);
10039 } else {
10040 // Otherwise, we'll add the newline to the list of newlines.
10041 pm_line_offset_list_append(&parser->metadata_arena, &parser->line_offsets, PM_TOKEN_END(parser, &parser->current) + U32(eol_length));
10042 }
10043
10044 uint8_t delimiter = *parser->current.end;
10045
10046 // If our delimiter is \r\n, we want to treat it as if it's \n.
10047 // For example, %\r\nfoo\r\n should be "foo"
10048 if (eol_length == 2) {
10049 delimiter = *(parser->current.end + 1);
10050 }
10051
10052 parser->current.end += eol_length;
10053 return delimiter;
10054 }
10055
10056 return *parser->current.end++;
10057}
10058
10063#define LEX(token_type) parser->current.type = token_type; parser_lex_callback(parser); return
10064
10071static void
10072parser_lex(pm_parser_t *parser) {
10073 assert(parser->current.end <= parser->end);
10074 parser->previous = parser->current;
10075
10076 // This value mirrors cmd_state from CRuby.
10077 bool previous_command_start = parser->command_start;
10078 parser->command_start = false;
10079
10080 // This is used to communicate to the newline lexing function that we've
10081 // already seen a comment.
10082 bool lexed_comment = false;
10083
10084 // Here we cache the current value of the semantic token seen flag. This is
10085 // used to reset it in case we find a token that shouldn't flip this flag.
10086 unsigned int semantic_token_seen = parser->semantic_token_seen;
10087 parser->semantic_token_seen = true;
10088
10089 // We'll jump to this label when we are about to encounter an EOF.
10090 // If we still have lex_modes on the stack, we pop them so that cleanup
10091 // can happen. For example, we should still continue parsing after a heredoc
10092 // identifier, even if the heredoc body was syntax invalid.
10093 switch_lex_modes:
10094
10095 switch (parser->lex_modes.current->mode) {
10096 case PM_LEX_DEFAULT:
10097 case PM_LEX_EMBEXPR:
10098 case PM_LEX_EMBVAR:
10099
10100 // We have a specific named label here because we are going to jump back to
10101 // this location in the event that we have lexed a token that should not be
10102 // returned to the parser. This includes comments, ignored newlines, and
10103 // invalid tokens of some form.
10104 lex_next_token: {
10105 // If we have the special next_start pointer set, then we're going to jump
10106 // to that location and start lexing from there.
10107 if (parser->next_start != NULL) {
10108 parser->current.end = parser->next_start;
10109 parser->next_start = NULL;
10110 }
10111
10112 // This value mirrors space_seen from CRuby. It tracks whether or not
10113 // space has been eaten before the start of the next token.
10114 bool space_seen = false;
10115
10116 // First, we're going to skip past any whitespace at the front of the next
10117 // token. Skip runs of inline whitespace in bulk to avoid per-character
10118 // stores back to parser->current.end.
10119 bool chomping = true;
10120 while (chomping) {
10121 /* Skip the run of inline whitespace in bulk, then decide what
10122 * to do based on the first byte after it. Handling both in a
10123 * single pass avoids re-entering the scan when the run was
10124 * non-empty, which is the common case. */
10125 {
10126 static const uint8_t inline_whitespace[256] = {
10127 [' '] = 1, ['\t'] = 1, ['\f'] = 1, ['\v'] = 1
10128 };
10129 const uint8_t *scan = parser->current.end;
10130 while (scan < parser->end && inline_whitespace[*scan]) scan++;
10131 if (scan > parser->current.end) {
10132 parser->current.end = scan;
10133 space_seen = true;
10134 }
10135 if (scan >= parser->end) break;
10136 }
10137
10138 switch (*parser->current.end) {
10139 case '\r':
10140 if (match_eol_offset(parser, 1)) {
10141 chomping = false;
10142 } else {
10143 pm_parser_warn(parser, PM_TOKEN_END(parser, &parser->current), 1, PM_WARN_UNEXPECTED_CARRIAGE_RETURN);
10144 parser->current.end++;
10145 space_seen = true;
10146 }
10147 break;
10148 case '\\': {
10149 size_t eol_length = match_eol_offset(parser, 1);
10150 if (eol_length) {
10151 if (parser->heredoc_end) {
10152 parser->current.end = parser->heredoc_end;
10153 parser->heredoc_end = NULL;
10154 } else {
10155 parser->current.end += eol_length + 1;
10156 pm_line_offset_list_append(&parser->metadata_arena, &parser->line_offsets, PM_TOKEN_END(parser, &parser->current));
10157 space_seen = true;
10158 }
10159 } else if (pm_char_is_inline_whitespace(*parser->current.end)) {
10160 parser->current.end += 2;
10161 } else {
10162 chomping = false;
10163 }
10164
10165 break;
10166 }
10167 default:
10168 chomping = false;
10169 break;
10170 }
10171 }
10172
10173 // Next, we'll set to start of this token to be the current end.
10174 parser->current.start = parser->current.end;
10175
10176 // We'll check if we're at the end of the file. If we are, then we
10177 // need to return the EOF token.
10178 if (parser->current.end >= parser->end) {
10179 // We may be missing closing tokens. We should pop modes one by one
10180 // to do the appropriate cleanup like moving next_start for heredocs.
10181 // Only when no mode is remaining will we actually emit the EOF token.
10182 if (parser->lex_modes.current->mode != PM_LEX_DEFAULT) {
10183 lex_mode_pop(parser);
10184 goto switch_lex_modes;
10185 }
10186
10187 // If we hit EOF, but the EOF came immediately after a newline,
10188 // set the start of the token to the newline. This way any EOF
10189 // errors will be reported as happening on that line rather than
10190 // a line after. For example "foo(\n" should report an error
10191 // on line 1 even though EOF technically occurs on line 2.
10192 if (parser->current.start > parser->start && (*(parser->current.start - 1) == '\n')) {
10193 parser->current.start -= 1;
10194 }
10195 LEX(PM_TOKEN_EOF);
10196 }
10197
10198 // Finally, we'll check the current character to determine the next
10199 // token.
10200 switch (*parser->current.end++) {
10201 case '\0': // NUL or end of script
10202 case '\004': // ^D
10203 case '\032': // ^Z
10204 parser->current.end--;
10205 LEX(PM_TOKEN_EOF);
10206
10207 case '#': { // comments
10208 const uint8_t *ending = next_newline(parser->current.end, parser->end - parser->current.end);
10209 parser->current.end = ending == NULL ? parser->end : ending;
10210
10211 // If we found a comment while lexing, then we're going to
10212 // add it to the list of comments in the file and keep
10213 // lexing.
10214 pm_comment_t *comment = parser_comment(parser, PM_COMMENT_INLINE);
10215 pm_list_append(&parser->comment_list, (pm_list_node_t *) comment);
10216
10217 parser->current.type = PM_TOKEN_COMMENT;
10218 parser_lex_callback(parser);
10219
10220 // Here, parse the comment to see if it's a magic comment
10221 // and potentially change state on the parser.
10222 if (!parser_lex_magic_comment(parser, semantic_token_seen) && (parser->current.start == parser->encoding_comment_start)) {
10223 ptrdiff_t length = parser->current.end - parser->current.start;
10224
10225 // If we didn't find a magic comment within the first
10226 // pass and we're at the start of the file, then we need
10227 // to do another pass to potentially find other patterns
10228 // for encoding comments.
10229 if (length >= 10 && !parser->encoding_locked) {
10230 parser_lex_magic_comment_encoding(parser);
10231 }
10232 }
10233
10234 /* The comment does not include its terminating newline,
10235 * which lexes through the newline handling below as its
10236 * own token. A comment that ends the file has no newline,
10237 * so the newline handling runs without one to emit. */
10238 if (ending == NULL) {
10239 lexed_comment = true;
10240 } else {
10241 parser->current.start = ending;
10242 parser->current.end = ending + 1;
10243 }
10244 }
10246 case '\r':
10247 case '\n': {
10248 parser->semantic_token_seen = semantic_token_seen & 0x1;
10249 size_t eol_length = match_eol_at(parser, parser->current.end - 1);
10250
10251 if (eol_length) {
10252 // The only way you can have carriage returns in this
10253 // particular loop is if you have a carriage return
10254 // followed by a newline. In that case we'll just skip
10255 // over the carriage return and continue lexing, in
10256 // order to make it so that the newline token
10257 // encapsulates both the carriage return and the
10258 // newline. Note that we need to check that we haven't
10259 // already lexed a comment here because that falls
10260 // through into here as well.
10261 if (!lexed_comment) {
10262 parser->current.end += eol_length - 1; // skip CR
10263 }
10264
10265 if (parser->heredoc_end == NULL) {
10266 pm_line_offset_list_append(&parser->metadata_arena, &parser->line_offsets, PM_TOKEN_END(parser, &parser->current));
10267 }
10268 }
10269
10270 if (parser->heredoc_end) {
10271 parser_flush_heredoc_end(parser);
10272 }
10273
10274 // If this is an ignored newline, then we can continue lexing after
10275 // calling the callback with the ignored newline token.
10276 switch (lex_state_ignored_p(parser)) {
10277 case PM_IGNORED_NEWLINE_NONE:
10278 break;
10279 case PM_IGNORED_NEWLINE_PATTERN:
10280 if (parser->pattern_matching_newlines || parser->in_keyword_arg) {
10281 if (!lexed_comment) {
10282 parser->current.type = PM_TOKEN_NEWLINE_TERMINATOR;
10283 parser_lex_callback(parser);
10284 }
10285
10286 lex_state_set(parser, PM_LEX_STATE_BEG);
10287 parser->command_start = true;
10288 parser->current.type = PM_TOKEN_NEWLINE;
10289 return;
10290 }
10292 case PM_IGNORED_NEWLINE_ALL:
10293 if (!lexed_comment) parser_lex_ignored_newline(parser);
10294 lexed_comment = false;
10295 goto lex_next_token;
10296 }
10297
10298 // Here we need to look ahead and see if there is a call operator
10299 // (either . or &.) that starts the next line. If there is, then this
10300 // is going to become an ignored newline and we're going to instead
10301 // return the call operator.
10302 const uint8_t *next_content = parser->next_start == NULL ? parser->current.end : parser->next_start;
10303 next_content += pm_strspn_inline_whitespace(next_content, parser->end - next_content);
10304
10305 if (next_content < parser->end) {
10306 // If we hit a comment after a newline, then we're going to check
10307 // if it's ignored or if it's followed by a method call ('.').
10308 // If it is, then we're going to call the
10309 // callback with an ignored newline and then continue lexing.
10310 // Otherwise we'll return a regular newline.
10311 if (next_content[0] == '#') {
10312 // Here we look for a "." or "&." following a "\n".
10313 const uint8_t *following = next_newline(next_content, parser->end - next_content);
10314
10315 while (following && (following + 1 < parser->end)) {
10316 following++;
10317 following += pm_strspn_inline_whitespace(following, parser->end - following);
10318
10319 // If this is not followed by a comment, then we can break out
10320 // of this loop.
10321 if (peek_at(parser, following) != '#') break;
10322
10323 // If there is a comment, then we need to find the end of the
10324 // comment and continue searching from there.
10325 following = next_newline(following, parser->end - following);
10326 }
10327
10328 // If the lex state was ignored, we will lex the
10329 // ignored newline.
10330 if (lex_state_ignored_p(parser)) {
10331 if (!lexed_comment) parser_lex_ignored_newline(parser);
10332 lexed_comment = false;
10333 goto lex_next_token;
10334 }
10335
10336 // If we hit a '.' or a '&.' we will lex the ignored
10337 // newline.
10338 if (following && (
10339 (peek_at(parser, following) == '.') ||
10340 (peek_at(parser, following) == '&' && peek_at(parser, following + 1) == '.')
10341 )) {
10342 if (!lexed_comment) parser_lex_ignored_newline(parser);
10343 lexed_comment = false;
10344 goto lex_next_token;
10345 }
10346
10347
10348 // If we are parsing as CRuby 4.0 or later and we
10349 // hit a '&&' or a '||' then we will lex the ignored
10350 // newline.
10351 if (
10352 (parser->version >= PM_OPTIONS_VERSION_CRUBY_4_0) &&
10353 following && (
10354 (peek_at(parser, following) == '&' && peek_at(parser, following + 1) == '&') ||
10355 (peek_at(parser, following) == '|' && peek_at(parser, following + 1) == '|') ||
10356 (
10357 peek_at(parser, following) == 'a' &&
10358 peek_at(parser, following + 1) == 'n' &&
10359 peek_at(parser, following + 2) == 'd' &&
10360 peek_at(parser, next_content + 3) != '!' &&
10361 peek_at(parser, next_content + 3) != '?' &&
10362 !char_is_identifier(parser, following + 3, parser->end - (following + 3))
10363 ) ||
10364 (
10365 peek_at(parser, following) == 'o' &&
10366 peek_at(parser, following + 1) == 'r' &&
10367 peek_at(parser, next_content + 2) != '!' &&
10368 peek_at(parser, next_content + 2) != '?' &&
10369 !char_is_identifier(parser, following + 2, parser->end - (following + 2))
10370 )
10371 )
10372 ) {
10373 if (!lexed_comment) parser_lex_ignored_newline(parser);
10374 lexed_comment = false;
10375 goto lex_next_token;
10376 }
10377 }
10378
10379 // If we hit a . after a newline, then we're in a call chain and
10380 // we need to return the call operator.
10381 if (next_content[0] == '.') {
10382 /* A beginless range on the next line means this
10383 * newline terminates the statement rather than
10384 * continuing a method chain. */
10385 if (peek_at(parser, next_content + 1) == '.') {
10386 if (!lexed_comment) {
10387 parser->current.type = PM_TOKEN_NEWLINE_TERMINATOR;
10388 parser_lex_callback(parser);
10389 }
10390
10391 lex_state_set(parser, PM_LEX_STATE_BEG);
10392 parser->command_start = true;
10393 parser->current.type = PM_TOKEN_NEWLINE;
10394 return;
10395 }
10396
10397 if (!lexed_comment) parser_lex_ignored_newline(parser);
10398 lex_state_set(parser, PM_LEX_STATE_DOT);
10399 parser->current.start = next_content;
10400 parser->current.end = next_content + 1;
10401 parser->next_start = NULL;
10402 LEX(PM_TOKEN_DOT);
10403 }
10404
10405 // If we hit a &. after a newline, then we're in a call chain and
10406 // we need to return the call operator.
10407 if (peek_at(parser, next_content) == '&' && peek_at(parser, next_content + 1) == '.') {
10408 if (!lexed_comment) parser_lex_ignored_newline(parser);
10409 lex_state_set(parser, PM_LEX_STATE_DOT);
10410 parser->current.start = next_content;
10411 parser->current.end = next_content + 2;
10412 parser->next_start = NULL;
10413 LEX(PM_TOKEN_AMPERSAND_DOT);
10414 }
10415
10416 if (parser->version >= PM_OPTIONS_VERSION_CRUBY_4_0) {
10417 // If we hit an && then we are in a logical chain
10418 // and we need to return the logical operator.
10419 if (peek_at(parser, next_content) == '&' && peek_at(parser, next_content + 1) == '&') {
10420 if (!lexed_comment) parser_lex_ignored_newline(parser);
10421 lex_state_set(parser, PM_LEX_STATE_BEG);
10422 parser->current.start = next_content;
10423 parser->current.end = next_content + 2;
10424 parser->next_start = NULL;
10425 LEX(PM_TOKEN_AMPERSAND_AMPERSAND);
10426 }
10427
10428 // If we hit a || then we are in a logical chain and
10429 // we need to return the logical operator.
10430 if (peek_at(parser, next_content) == '|' && peek_at(parser, next_content + 1) == '|') {
10431 if (!lexed_comment) parser_lex_ignored_newline(parser);
10432 lex_state_set(parser, PM_LEX_STATE_BEG);
10433 parser->current.start = next_content;
10434 parser->current.end = next_content + 2;
10435 parser->next_start = NULL;
10436 LEX(PM_TOKEN_PIPE_PIPE);
10437 }
10438
10439 // If we hit an 'and' then we are in a logical chain
10440 // and we need to return the logical operator.
10441 if (
10442 peek_at(parser, next_content) == 'a' &&
10443 peek_at(parser, next_content + 1) == 'n' &&
10444 peek_at(parser, next_content + 2) == 'd' &&
10445 peek_at(parser, next_content + 3) != '!' &&
10446 peek_at(parser, next_content + 3) != '?' &&
10447 !char_is_identifier(parser, next_content + 3, parser->end - (next_content + 3))
10448 ) {
10449 if (!lexed_comment) parser_lex_ignored_newline(parser);
10450 lex_state_set(parser, PM_LEX_STATE_BEG);
10451 parser->current.start = next_content;
10452 parser->current.end = next_content + 3;
10453 parser->next_start = NULL;
10454 parser->command_start = true;
10455 LEX(PM_TOKEN_KEYWORD_AND);
10456 }
10457
10458 // If we hit a 'or' then we are in a logical chain
10459 // and we need to return the logical operator.
10460 if (
10461 peek_at(parser, next_content) == 'o' &&
10462 peek_at(parser, next_content + 1) == 'r' &&
10463 peek_at(parser, next_content + 2) != '!' &&
10464 peek_at(parser, next_content + 2) != '?' &&
10465 !char_is_identifier(parser, next_content + 2, parser->end - (next_content + 2))
10466 ) {
10467 if (!lexed_comment) parser_lex_ignored_newline(parser);
10468 lex_state_set(parser, PM_LEX_STATE_BEG);
10469 parser->current.start = next_content;
10470 parser->current.end = next_content + 2;
10471 parser->next_start = NULL;
10472 parser->command_start = true;
10473 LEX(PM_TOKEN_KEYWORD_OR);
10474 }
10475 }
10476 }
10477
10478 // At this point we know this is a regular newline, and we can set the
10479 // necessary state and return the token.
10480 lex_state_set(parser, PM_LEX_STATE_BEG);
10481 parser->command_start = true;
10482 parser->current.type = PM_TOKEN_NEWLINE;
10483 if (!lexed_comment) parser_lex_callback(parser);
10484 return;
10485 }
10486
10487 // ,
10488 case ',':
10489 if ((parser->previous.type == PM_TOKEN_COMMA) && (parser->enclosure_nesting > 0)) {
10490 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->current, PM_ERR_ARRAY_TERM, pm_token_str(parser->current.type));
10491 }
10492
10493 lex_state_set(parser, PM_LEX_STATE_BEG | PM_LEX_STATE_LABEL);
10494 LEX(PM_TOKEN_COMMA);
10495
10496 // (
10497 case '(': {
10498 /* A parenthesis scanned at the beginning of an expression
10499 * groups the expression it wraps, while one scanned in
10500 * argument position with a preceding space wraps a command
10501 * argument. Everything else opens an argument list. */
10502 pm_token_type_t type = PM_TOKEN_PARENTHESIS_LEFT;
10503
10504 if (lex_state_beg_p(parser)) {
10505 type = PM_TOKEN_PARENTHESIS_LEFT_GROUPING;
10506 } else if (space_seen && (lex_state_arg_p(parser) || parser->lex_state == (PM_LEX_STATE_END | PM_LEX_STATE_LABEL))) {
10507 type = PM_TOKEN_PARENTHESIS_LEFT_PARENTHESES;
10508 }
10509
10510 parser->enclosure_nesting++;
10511 lex_state_set(parser, PM_LEX_STATE_BEG | PM_LEX_STATE_LABEL);
10512 pm_enclosure_frame_push(parser);
10513 LEX(type);
10514 }
10515
10516 // )
10517 case ')':
10518 parser->enclosure_nesting--;
10519 lex_state_set(parser, PM_LEX_STATE_ENDFN);
10520 pm_enclosure_frame_pop(parser);
10521 LEX(PM_TOKEN_PARENTHESIS_RIGHT);
10522
10523 // ;
10524 case ';':
10525 lex_state_set(parser, PM_LEX_STATE_BEG);
10526 parser->command_start = true;
10527 LEX(PM_TOKEN_SEMICOLON);
10528
10529 // [ [] []=
10530 case '[':
10531 parser->enclosure_nesting++;
10532 pm_token_type_t type = PM_TOKEN_BRACKET_LEFT;
10533
10534 if (lex_state_operator_p(parser)) {
10535 if (match(parser, ']')) {
10536 parser->enclosure_nesting--;
10537 lex_state_set(parser, PM_LEX_STATE_ARG);
10538 LEX(match(parser, '=') ? PM_TOKEN_BRACKET_LEFT_RIGHT_EQUAL : PM_TOKEN_BRACKET_LEFT_RIGHT);
10539 }
10540
10541 lex_state_set(parser, PM_LEX_STATE_ARG | PM_LEX_STATE_LABEL);
10542 LEX(type);
10543 }
10544
10545 if (lex_state_beg_p(parser) || (lex_state_arg_p(parser) && (space_seen || lex_state_p(parser, PM_LEX_STATE_LABELED)))) {
10546 type = PM_TOKEN_BRACKET_LEFT_ARRAY;
10547 }
10548
10549 lex_state_set(parser, PM_LEX_STATE_BEG | PM_LEX_STATE_LABEL);
10550 pm_enclosure_frame_push(parser);
10551 LEX(type);
10552
10553 // ]
10554 case ']':
10555 parser->enclosure_nesting--;
10556 lex_state_set(parser, PM_LEX_STATE_END);
10557 pm_enclosure_frame_pop(parser);
10558 LEX(PM_TOKEN_BRACKET_RIGHT);
10559
10560 // {
10561 case '{': {
10562 pm_token_type_t type = PM_TOKEN_BRACE_LEFT;
10563
10564 if (parser->enclosure_nesting == parser->lambda_enclosure_nesting) {
10565 /* This { begins a lambda */
10566 parser->command_start = true;
10567 lex_state_set(parser, PM_LEX_STATE_BEG);
10568 type = PM_TOKEN_LAMBDA_BEGIN;
10569 } else if (lex_state_p(parser, PM_LEX_STATE_LABELED)) {
10570 /* This { begins a hash literal */
10571 lex_state_set(parser, PM_LEX_STATE_BEG | PM_LEX_STATE_LABEL);
10572 type = PM_TOKEN_BRACE_LEFT_HASH;
10573 } else if (lex_state_p(parser, PM_LEX_STATE_ARG_ANY | PM_LEX_STATE_END | PM_LEX_STATE_ENDFN)) {
10574 /* This { begins a block */
10575 parser->command_start = true;
10576 lex_state_set(parser, PM_LEX_STATE_BEG);
10577 } else if (lex_state_p(parser, PM_LEX_STATE_ENDARG)) {
10578 /* This { begins a block following a parenthesized
10579 * command argument */
10580 parser->command_start = true;
10581 lex_state_set(parser, PM_LEX_STATE_BEG);
10582 type = PM_TOKEN_BRACE_LEFT_ARGUMENT;
10583 } else {
10584 /* This { begins a hash literal */
10585 lex_state_set(parser, PM_LEX_STATE_BEG | PM_LEX_STATE_LABEL);
10586 type = PM_TOKEN_BRACE_LEFT_HASH;
10587 }
10588
10589 parser->enclosure_nesting++;
10590 parser->brace_nesting++;
10591 pm_enclosure_frame_push(parser);
10592
10593 LEX(type);
10594 }
10595
10596 // }
10597 case '}':
10598 parser->enclosure_nesting--;
10599 pm_enclosure_frame_pop(parser);
10600
10601 if ((parser->lex_modes.current->mode == PM_LEX_EMBEXPR) && (parser->brace_nesting == 0)) {
10602 lex_mode_pop(parser);
10603 LEX(PM_TOKEN_EMBEXPR_END);
10604 }
10605
10606 parser->brace_nesting--;
10607 lex_state_set(parser, PM_LEX_STATE_END);
10608 LEX(PM_TOKEN_BRACE_RIGHT);
10609
10610 // * ** **= *=
10611 case '*': {
10612 if (match(parser, '*')) {
10613 if (match(parser, '=')) {
10614 lex_state_set(parser, PM_LEX_STATE_BEG);
10615 LEX(PM_TOKEN_STAR_STAR_EQUAL);
10616 }
10617
10618 pm_token_type_t type = PM_TOKEN_STAR_STAR;
10619
10620 if (lex_state_spcarg_p(parser, space_seen)) {
10621 pm_parser_warn_token(parser, &parser->current, PM_WARN_AMBIGUOUS_PREFIX_STAR_STAR);
10622 type = PM_TOKEN_USTAR_STAR;
10623 } else if (lex_state_beg_p(parser)) {
10624 type = PM_TOKEN_USTAR_STAR;
10625 } else if (ambiguous_operator_p(parser, space_seen)) {
10626 PM_PARSER_WARN_TOKEN_FORMAT(parser, &parser->current, PM_WARN_AMBIGUOUS_BINARY_OPERATOR, "**", "argument prefix");
10627 }
10628
10629 if (lex_state_operator_p(parser)) {
10630 lex_state_set(parser, PM_LEX_STATE_ARG);
10631 } else {
10632 lex_state_set(parser, PM_LEX_STATE_BEG);
10633 }
10634
10635 LEX(type);
10636 }
10637
10638 if (match(parser, '=')) {
10639 lex_state_set(parser, PM_LEX_STATE_BEG);
10640 LEX(PM_TOKEN_STAR_EQUAL);
10641 }
10642
10643 pm_token_type_t type = PM_TOKEN_STAR;
10644
10645 if (lex_state_spcarg_p(parser, space_seen)) {
10646 pm_parser_warn_token(parser, &parser->current, PM_WARN_AMBIGUOUS_PREFIX_STAR);
10647 type = PM_TOKEN_USTAR;
10648 } else if (lex_state_beg_p(parser)) {
10649 type = PM_TOKEN_USTAR;
10650 } else if (ambiguous_operator_p(parser, space_seen)) {
10651 PM_PARSER_WARN_TOKEN_FORMAT(parser, &parser->current, PM_WARN_AMBIGUOUS_BINARY_OPERATOR, "*", "argument prefix");
10652 }
10653
10654 if (lex_state_operator_p(parser)) {
10655 lex_state_set(parser, PM_LEX_STATE_ARG);
10656 } else {
10657 lex_state_set(parser, PM_LEX_STATE_BEG);
10658 }
10659
10660 LEX(type);
10661 }
10662
10663 // ! != !~ !@
10664 case '!':
10665 if (lex_state_operator_p(parser)) {
10666 lex_state_set(parser, PM_LEX_STATE_ARG);
10667 if (match(parser, '@')) {
10668 LEX(PM_TOKEN_BANG);
10669 }
10670 } else {
10671 lex_state_set(parser, PM_LEX_STATE_BEG);
10672 }
10673
10674 if (match(parser, '=')) {
10675 LEX(PM_TOKEN_BANG_EQUAL);
10676 }
10677
10678 if (match(parser, '~')) {
10679 LEX(PM_TOKEN_BANG_TILDE);
10680 }
10681
10682 LEX(PM_TOKEN_BANG);
10683
10684 // = => =~ == === =begin
10685 case '=':
10686 if (
10687 current_token_starts_line(parser) &&
10688 (parser->current.end + 5 <= parser->end) &&
10689 memcmp(parser->current.end, "begin", 5) == 0 &&
10690 (pm_char_is_whitespace(peek_offset(parser, 5)) || (peek_offset(parser, 5) == '\0'))
10691 ) {
10692 pm_token_type_t type = lex_embdoc(parser);
10693 if (type == PM_TOKEN_EOF) {
10694 LEX(type);
10695 }
10696
10697 goto lex_next_token;
10698 }
10699
10700 if (lex_state_operator_p(parser)) {
10701 lex_state_set(parser, PM_LEX_STATE_ARG);
10702 } else {
10703 lex_state_set(parser, PM_LEX_STATE_BEG);
10704 }
10705
10706 if (match(parser, '>')) {
10707 LEX(PM_TOKEN_EQUAL_GREATER);
10708 }
10709
10710 if (match(parser, '~')) {
10711 LEX(PM_TOKEN_EQUAL_TILDE);
10712 }
10713
10714 if (match(parser, '=')) {
10715 LEX(match(parser, '=') ? PM_TOKEN_EQUAL_EQUAL_EQUAL : PM_TOKEN_EQUAL_EQUAL);
10716 }
10717
10718 LEX(PM_TOKEN_EQUAL);
10719
10720 // < << <<= <= <=>
10721 case '<':
10722 if (match(parser, '<')) {
10723 if (
10724 !lex_state_p(parser, PM_LEX_STATE_DOT | PM_LEX_STATE_CLASS) &&
10725 !lex_state_end_p(parser) &&
10726 (!lex_state_p(parser, PM_LEX_STATE_ARG_ANY) || lex_state_p(parser, PM_LEX_STATE_LABELED) || space_seen)
10727 ) {
10728 const uint8_t *end = parser->current.end;
10729
10730 pm_heredoc_quote_t quote = PM_HEREDOC_QUOTE_NONE;
10731 pm_heredoc_indent_t indent = PM_HEREDOC_INDENT_NONE;
10732
10733 if (match(parser, '-')) {
10734 indent = PM_HEREDOC_INDENT_DASH;
10735 }
10736 else if (match(parser, '~')) {
10737 indent = PM_HEREDOC_INDENT_TILDE;
10738 }
10739
10740 if (match(parser, '`')) {
10741 quote = PM_HEREDOC_QUOTE_BACKTICK;
10742 }
10743 else if (match(parser, '"')) {
10744 quote = PM_HEREDOC_QUOTE_DOUBLE;
10745 }
10746 else if (match(parser, '\'')) {
10747 quote = PM_HEREDOC_QUOTE_SINGLE;
10748 }
10749
10750 const uint8_t *ident_start = parser->current.end;
10751 size_t width = 0;
10752
10753 if (parser->current.end >= parser->end) {
10754 parser->current.end = end;
10755 } else if (quote == PM_HEREDOC_QUOTE_NONE && (width = char_is_identifier(parser, parser->current.end, parser->end - parser->current.end)) == 0) {
10756 parser->current.end = end;
10757 } else {
10758 if (quote == PM_HEREDOC_QUOTE_NONE) {
10759 parser->current.end += width;
10760
10761 while ((width = char_is_identifier(parser, parser->current.end, parser->end - parser->current.end))) {
10762 parser->current.end += width;
10763 }
10764 } else {
10765 // If we have quotes, then we're going to go until we find the
10766 // end quote.
10767 while ((parser->current.end < parser->end) && quote != (pm_heredoc_quote_t) (*parser->current.end)) {
10768 if (*parser->current.end == '\r' || *parser->current.end == '\n') break;
10769 parser->current.end++;
10770 }
10771 }
10772
10773 size_t ident_length = (size_t) (parser->current.end - ident_start);
10774 bool ident_error = false;
10775
10776 if (quote != PM_HEREDOC_QUOTE_NONE && !match(parser, (uint8_t) quote)) {
10777 pm_parser_err(parser, U32(ident_start - parser->start), U32(ident_length), PM_ERR_HEREDOC_IDENTIFIER);
10778 ident_error = true;
10779 }
10780
10781 parser->explicit_encoding = NULL;
10782 lex_mode_push(parser, (pm_lex_mode_t) {
10783 .mode = PM_LEX_HEREDOC,
10784 .as.heredoc = {
10785 .base = {
10786 .ident_start = ident_start,
10787 .ident_length = ident_length,
10788 .quote = quote,
10789 .indent = indent
10790 },
10791 .next_start = parser->current.end,
10792 .common_whitespace = NULL,
10793 .line_continuation = false
10794 }
10795 });
10796
10797 if (parser->heredoc_end == NULL) {
10798 const uint8_t *body_start = next_newline(parser->current.end, parser->end - parser->current.end);
10799
10800 if (body_start == NULL) {
10801 // If there is no newline after the heredoc identifier, then
10802 // this is not a valid heredoc declaration. In this case we
10803 // will add an error, but we will still return a heredoc
10804 // start.
10805 if (!ident_error) pm_parser_err_heredoc_term(parser, ident_start, ident_length);
10806 body_start = parser->end;
10807 } else {
10808 // Otherwise, we want to indicate that the body of the
10809 // heredoc starts on the character after the next newline.
10810 pm_line_offset_list_append(&parser->metadata_arena, &parser->line_offsets, U32(body_start - parser->start + 1));
10811 body_start++;
10812 }
10813
10814 parser->next_start = body_start;
10815 } else {
10816 parser->next_start = parser->heredoc_end;
10817 }
10818
10819 LEX(PM_TOKEN_HEREDOC_START);
10820 }
10821 }
10822
10823 if (match(parser, '=')) {
10824 lex_state_set(parser, PM_LEX_STATE_BEG);
10825 LEX(PM_TOKEN_LESS_LESS_EQUAL);
10826 }
10827
10828 if (ambiguous_operator_p(parser, space_seen)) {
10829 PM_PARSER_WARN_TOKEN_FORMAT(parser, &parser->current, PM_WARN_AMBIGUOUS_BINARY_OPERATOR, "<<", "here document");
10830 }
10831
10832 if (lex_state_operator_p(parser)) {
10833 lex_state_set(parser, PM_LEX_STATE_ARG);
10834 } else {
10835 if (lex_state_p(parser, PM_LEX_STATE_CLASS)) parser->command_start = true;
10836 lex_state_set(parser, PM_LEX_STATE_BEG);
10837 }
10838
10839 LEX(PM_TOKEN_LESS_LESS);
10840 }
10841
10842 if (lex_state_operator_p(parser)) {
10843 lex_state_set(parser, PM_LEX_STATE_ARG);
10844 } else {
10845 if (lex_state_p(parser, PM_LEX_STATE_CLASS)) parser->command_start = true;
10846 lex_state_set(parser, PM_LEX_STATE_BEG);
10847 }
10848
10849 if (match(parser, '=')) {
10850 if (match(parser, '>')) {
10851 LEX(PM_TOKEN_LESS_EQUAL_GREATER);
10852 }
10853
10854 LEX(PM_TOKEN_LESS_EQUAL);
10855 }
10856
10857 LEX(PM_TOKEN_LESS);
10858
10859 // > >> >>= >=
10860 case '>':
10861 if (match(parser, '>')) {
10862 if (lex_state_operator_p(parser)) {
10863 lex_state_set(parser, PM_LEX_STATE_ARG);
10864 } else {
10865 lex_state_set(parser, PM_LEX_STATE_BEG);
10866 }
10867 LEX(match(parser, '=') ? PM_TOKEN_GREATER_GREATER_EQUAL : PM_TOKEN_GREATER_GREATER);
10868 }
10869
10870 if (lex_state_operator_p(parser)) {
10871 lex_state_set(parser, PM_LEX_STATE_ARG);
10872 } else {
10873 lex_state_set(parser, PM_LEX_STATE_BEG);
10874 }
10875
10876 LEX(match(parser, '=') ? PM_TOKEN_GREATER_EQUAL : PM_TOKEN_GREATER);
10877
10878 // double-quoted string literal
10879 case '"': {
10880 bool label_allowed = (lex_state_p(parser, PM_LEX_STATE_LABEL | PM_LEX_STATE_ENDFN) && !previous_command_start) || lex_state_arg_p(parser);
10881 lex_mode_push_string(parser, true, label_allowed, '\0', '"');
10882 LEX(PM_TOKEN_STRING_BEGIN);
10883 }
10884
10885 // xstring literal
10886 case '`': {
10887 if (lex_state_p(parser, PM_LEX_STATE_FNAME)) {
10888 lex_state_set(parser, PM_LEX_STATE_ENDFN);
10889 LEX(PM_TOKEN_BACKTICK);
10890 }
10891
10892 if (lex_state_p(parser, PM_LEX_STATE_DOT)) {
10893 if (previous_command_start) {
10894 lex_state_set(parser, PM_LEX_STATE_CMDARG);
10895 } else {
10896 lex_state_set(parser, PM_LEX_STATE_ARG);
10897 }
10898
10899 LEX(PM_TOKEN_BACKTICK);
10900 }
10901
10902 lex_mode_push_string(parser, true, false, '\0', '`');
10903 LEX(PM_TOKEN_XSTRING_BEGIN);
10904 }
10905
10906 // single-quoted string literal
10907 case '\'': {
10908 bool label_allowed = (lex_state_p(parser, PM_LEX_STATE_LABEL | PM_LEX_STATE_ENDFN) && !previous_command_start) || lex_state_arg_p(parser);
10909 lex_mode_push_string(parser, false, label_allowed, '\0', '\'');
10910 LEX(PM_TOKEN_STRING_BEGIN);
10911 }
10912
10913 // ? character literal
10914 case '?':
10915 LEX(lex_question_mark(parser));
10916
10917 // & && &&= &=
10918 case '&': {
10919 if (match(parser, '&')) {
10920 lex_state_set(parser, PM_LEX_STATE_BEG);
10921
10922 if (match(parser, '=')) {
10923 LEX(PM_TOKEN_AMPERSAND_AMPERSAND_EQUAL);
10924 }
10925
10926 LEX(PM_TOKEN_AMPERSAND_AMPERSAND);
10927 }
10928
10929 if (match(parser, '=')) {
10930 lex_state_set(parser, PM_LEX_STATE_BEG);
10931 LEX(PM_TOKEN_AMPERSAND_EQUAL);
10932 }
10933
10934 if (match(parser, '.')) {
10935 lex_state_set(parser, PM_LEX_STATE_DOT);
10936 LEX(PM_TOKEN_AMPERSAND_DOT);
10937 }
10938
10939 pm_token_type_t type = PM_TOKEN_AMPERSAND;
10940 if (lex_state_spcarg_p(parser, space_seen)) {
10941 if ((peek(parser) != ':') || (peek_offset(parser, 1) == '\0')) {
10942 pm_parser_warn_token(parser, &parser->current, PM_WARN_AMBIGUOUS_PREFIX_AMPERSAND);
10943 } else {
10944 const uint8_t delim = peek_offset(parser, 1);
10945
10946 if ((delim != '\'') && (delim != '"') && !char_is_identifier(parser, parser->current.end + 1, parser->end - (parser->current.end + 1))) {
10947 pm_parser_warn_token(parser, &parser->current, PM_WARN_AMBIGUOUS_PREFIX_AMPERSAND);
10948 }
10949 }
10950
10951 type = PM_TOKEN_UAMPERSAND;
10952 } else if (lex_state_beg_p(parser)) {
10953 type = PM_TOKEN_UAMPERSAND;
10954 } else if (ambiguous_operator_p(parser, space_seen)) {
10955 PM_PARSER_WARN_TOKEN_FORMAT(parser, &parser->current, PM_WARN_AMBIGUOUS_BINARY_OPERATOR, "&", "argument prefix");
10956 }
10957
10958 if (lex_state_operator_p(parser)) {
10959 lex_state_set(parser, PM_LEX_STATE_ARG);
10960 } else {
10961 lex_state_set(parser, PM_LEX_STATE_BEG);
10962 }
10963
10964 LEX(type);
10965 }
10966
10967 // | || ||= |=
10968 case '|':
10969 if (match(parser, '|')) {
10970 if (match(parser, '=')) {
10971 lex_state_set(parser, PM_LEX_STATE_BEG);
10972 LEX(PM_TOKEN_PIPE_PIPE_EQUAL);
10973 }
10974
10975 if (lex_state_p(parser, PM_LEX_STATE_BEG)) {
10976 parser->current.end--;
10977 LEX(PM_TOKEN_PIPE);
10978 }
10979
10980 lex_state_set(parser, PM_LEX_STATE_BEG);
10981 LEX(PM_TOKEN_PIPE_PIPE);
10982 }
10983
10984 if (match(parser, '=')) {
10985 lex_state_set(parser, PM_LEX_STATE_BEG);
10986 LEX(PM_TOKEN_PIPE_EQUAL);
10987 }
10988
10989 if (lex_state_operator_p(parser)) {
10990 lex_state_set(parser, PM_LEX_STATE_ARG);
10991 } else {
10992 lex_state_set(parser, PM_LEX_STATE_BEG | PM_LEX_STATE_LABEL);
10993 }
10994
10995 LEX(PM_TOKEN_PIPE);
10996
10997 // + += +@
10998 case '+': {
10999 if (lex_state_operator_p(parser)) {
11000 lex_state_set(parser, PM_LEX_STATE_ARG);
11001
11002 if (match(parser, '@')) {
11003 LEX(PM_TOKEN_UPLUS);
11004 }
11005
11006 LEX(PM_TOKEN_PLUS);
11007 }
11008
11009 if (match(parser, '=')) {
11010 lex_state_set(parser, PM_LEX_STATE_BEG);
11011 LEX(PM_TOKEN_PLUS_EQUAL);
11012 }
11013
11014 if (
11015 lex_state_beg_p(parser) ||
11016 (lex_state_spcarg_p(parser, space_seen) ? (pm_parser_warn_token(parser, &parser->current, PM_WARN_AMBIGUOUS_FIRST_ARGUMENT_PLUS), true) : false)
11017 ) {
11018 lex_state_set(parser, PM_LEX_STATE_BEG);
11019
11020 if (pm_char_is_decimal_digit(peek(parser))) {
11021 parser->current.end++;
11022 pm_token_type_t type = lex_numeric(parser);
11023 lex_state_set(parser, PM_LEX_STATE_END);
11024 LEX(type);
11025 }
11026
11027 LEX(PM_TOKEN_UPLUS);
11028 }
11029
11030 if (ambiguous_operator_p(parser, space_seen)) {
11031 PM_PARSER_WARN_TOKEN_FORMAT(parser, &parser->current, PM_WARN_AMBIGUOUS_BINARY_OPERATOR, "+", "unary operator");
11032 }
11033
11034 lex_state_set(parser, PM_LEX_STATE_BEG);
11035 LEX(PM_TOKEN_PLUS);
11036 }
11037
11038 // - -= -@
11039 case '-': {
11040 if (lex_state_operator_p(parser)) {
11041 lex_state_set(parser, PM_LEX_STATE_ARG);
11042
11043 if (match(parser, '@')) {
11044 LEX(PM_TOKEN_UMINUS);
11045 }
11046
11047 LEX(PM_TOKEN_MINUS);
11048 }
11049
11050 if (match(parser, '=')) {
11051 lex_state_set(parser, PM_LEX_STATE_BEG);
11052 LEX(PM_TOKEN_MINUS_EQUAL);
11053 }
11054
11055 if (match(parser, '>')) {
11056 lex_state_set(parser, PM_LEX_STATE_ENDFN);
11057 LEX(PM_TOKEN_MINUS_GREATER);
11058 }
11059
11060 bool spcarg = lex_state_spcarg_p(parser, space_seen);
11061 bool is_beg = lex_state_beg_p(parser);
11062 if (!is_beg && spcarg) {
11063 pm_parser_warn_token(parser, &parser->current, PM_WARN_AMBIGUOUS_FIRST_ARGUMENT_MINUS);
11064 }
11065
11066 if (is_beg || spcarg) {
11067 lex_state_set(parser, PM_LEX_STATE_BEG);
11068 LEX(pm_char_is_decimal_digit(peek(parser)) ? PM_TOKEN_UMINUS_NUM : PM_TOKEN_UMINUS);
11069 }
11070
11071 if (ambiguous_operator_p(parser, space_seen)) {
11072 PM_PARSER_WARN_TOKEN_FORMAT(parser, &parser->current, PM_WARN_AMBIGUOUS_BINARY_OPERATOR, "-", "unary operator");
11073 }
11074
11075 lex_state_set(parser, PM_LEX_STATE_BEG);
11076 LEX(PM_TOKEN_MINUS);
11077 }
11078
11079 // . .. ...
11080 case '.': {
11081 bool beg_p = lex_state_beg_p(parser);
11082
11083 if (match(parser, '.')) {
11084 if (match(parser, '.')) {
11085 // If we're _not_ inside a range within default parameters
11086 if (!context_p(parser, PM_CONTEXT_DEFAULT_PARAMS) && context_p(parser, PM_CONTEXT_DEF_PARAMS)) {
11087 if (lex_state_p(parser, PM_LEX_STATE_END)) {
11088 lex_state_set(parser, PM_LEX_STATE_BEG);
11089 } else {
11090 lex_state_set(parser, PM_LEX_STATE_ENDARG);
11091 }
11092 LEX(PM_TOKEN_UDOT_DOT_DOT);
11093 }
11094
11095 if (parser->enclosure_nesting == 0 && parser_end_of_line_p(parser)) {
11096 pm_parser_warn_token(parser, &parser->current, PM_WARN_DOT_DOT_DOT_EOL);
11097 }
11098
11099 lex_state_set(parser, PM_LEX_STATE_BEG);
11100 LEX(beg_p ? PM_TOKEN_UDOT_DOT_DOT : PM_TOKEN_DOT_DOT_DOT);
11101 }
11102
11103 lex_state_set(parser, PM_LEX_STATE_BEG);
11104 LEX(beg_p ? PM_TOKEN_UDOT_DOT : PM_TOKEN_DOT_DOT);
11105 }
11106
11107 lex_state_set(parser, PM_LEX_STATE_DOT);
11108 LEX(PM_TOKEN_DOT);
11109 }
11110
11111 // integer
11112 case '0':
11113 case '1':
11114 case '2':
11115 case '3':
11116 case '4':
11117 case '5':
11118 case '6':
11119 case '7':
11120 case '8':
11121 case '9': {
11122 pm_token_type_t type = lex_numeric(parser);
11123 lex_state_set(parser, PM_LEX_STATE_END);
11124 LEX(type);
11125 }
11126
11127 // :: symbol
11128 case ':':
11129 if (match(parser, ':')) {
11130 if (lex_state_beg_p(parser) || lex_state_p(parser, PM_LEX_STATE_CLASS) || (lex_state_p(parser, PM_LEX_STATE_ARG_ANY) && space_seen)) {
11131 lex_state_set(parser, PM_LEX_STATE_BEG);
11132 LEX(PM_TOKEN_UCOLON_COLON);
11133 }
11134
11135 lex_state_set(parser, PM_LEX_STATE_DOT);
11136 LEX(PM_TOKEN_COLON_COLON);
11137 }
11138
11139 if (lex_state_end_p(parser) || pm_char_is_whitespace(peek(parser)) || peek(parser) == '#') {
11140 lex_state_set(parser, PM_LEX_STATE_BEG);
11141 LEX(PM_TOKEN_COLON);
11142 }
11143
11144 if (peek(parser) == '"' || peek(parser) == '\'') {
11145 lex_mode_push_string(parser, peek(parser) == '"', false, '\0', *parser->current.end);
11146 parser->current.end++;
11147 } else {
11148 /*
11149 * A quoted symbol clears its encoding by pushing a lex
11150 * mode above. A bare symbol is lexed inline, so it
11151 * clears the encoding here.
11152 */
11153 parser->explicit_encoding = NULL;
11154 }
11155
11156 lex_state_set(parser, PM_LEX_STATE_FNAME);
11157 LEX(PM_TOKEN_SYMBOL_BEGIN);
11158
11159 // / /=
11160 case '/':
11161 if (lex_state_beg_p(parser)) {
11162 lex_mode_push_regexp(parser, '\0', '/');
11163 LEX(PM_TOKEN_REGEXP_BEGIN);
11164 }
11165
11166 if (match(parser, '=')) {
11167 lex_state_set(parser, PM_LEX_STATE_BEG);
11168 LEX(PM_TOKEN_SLASH_EQUAL);
11169 }
11170
11171 if (lex_state_spcarg_p(parser, space_seen)) {
11172 // https://bugs.ruby-lang.org/issues/21994
11173 if (parser->version <= PM_OPTIONS_VERSION_CRUBY_4_0) {
11174 pm_parser_warn_token(parser, &parser->current, PM_WARN_AMBIGUOUS_SLASH);
11175 }
11176 lex_mode_push_regexp(parser, '\0', '/');
11177 LEX(PM_TOKEN_REGEXP_BEGIN);
11178 }
11179
11180 if (ambiguous_operator_p(parser, space_seen)) {
11181 PM_PARSER_WARN_TOKEN_FORMAT(parser, &parser->current, PM_WARN_AMBIGUOUS_BINARY_OPERATOR, "/", "regexp literal");
11182 }
11183
11184 if (lex_state_operator_p(parser)) {
11185 lex_state_set(parser, PM_LEX_STATE_ARG);
11186 } else {
11187 lex_state_set(parser, PM_LEX_STATE_BEG);
11188 }
11189
11190 LEX(PM_TOKEN_SLASH);
11191
11192 // ^ ^=
11193 case '^':
11194 if (lex_state_operator_p(parser)) {
11195 lex_state_set(parser, PM_LEX_STATE_ARG);
11196 } else {
11197 lex_state_set(parser, PM_LEX_STATE_BEG);
11198 }
11199 LEX(match(parser, '=') ? PM_TOKEN_CARET_EQUAL : PM_TOKEN_CARET);
11200
11201 // ~ ~@
11202 case '~':
11203 if (lex_state_operator_p(parser)) {
11204 (void) match(parser, '@');
11205 lex_state_set(parser, PM_LEX_STATE_ARG);
11206 } else {
11207 lex_state_set(parser, PM_LEX_STATE_BEG);
11208 }
11209
11210 LEX(PM_TOKEN_TILDE);
11211
11212 // % %= %i %I %q %Q %w %W
11213 case '%': {
11214 // If there is no subsequent character then we have an
11215 // invalid token. We're going to say it's the percent
11216 // operator because we don't want to move into the string
11217 // lex mode unnecessarily.
11218 if ((lex_state_beg_p(parser) || lex_state_arg_p(parser)) && (parser->current.end >= parser->end)) {
11219 pm_parser_err_current(parser, PM_ERR_INVALID_PERCENT_EOF);
11220 LEX(PM_TOKEN_PERCENT);
11221 }
11222
11223 if (!lex_state_beg_p(parser) && match(parser, '=')) {
11224 lex_state_set(parser, PM_LEX_STATE_BEG);
11225 LEX(PM_TOKEN_PERCENT_EQUAL);
11226 } else if (
11227 lex_state_beg_p(parser) ||
11228 (lex_state_p(parser, PM_LEX_STATE_FITEM) && (peek(parser) == 's')) ||
11229 lex_state_spcarg_p(parser, space_seen)
11230 ) {
11231 if (!parser->encoding->alnum_char(parser->current.end, parser->end - parser->current.end)) {
11232 if (*parser->current.end >= 0x80) {
11233 pm_parser_err_current(parser, PM_ERR_INVALID_PERCENT);
11234 goto lex_next_token;
11235 }
11236
11237 const uint8_t delimiter = pm_lex_percent_delimiter(parser);
11238 lex_mode_push_string(parser, true, false, lex_mode_incrementor(delimiter), lex_mode_terminator(delimiter));
11239 LEX(PM_TOKEN_STRING_BEGIN);
11240 }
11241
11242 // Delimiters for %-literals cannot be alphanumeric. We
11243 // validate that here.
11244 uint8_t delimiter = peek_offset(parser, 1);
11245 if (delimiter >= 0x80 || parser->encoding->alnum_char(&delimiter, 1)) {
11246 pm_parser_err_current(parser, PM_ERR_INVALID_PERCENT);
11247 goto lex_next_token;
11248 }
11249
11250 switch (peek(parser)) {
11251 case 'i': {
11252 parser->current.end++;
11253
11254 if (parser->current.end < parser->end) {
11255 lex_mode_push_list(parser, false, pm_lex_percent_delimiter(parser));
11256 } else {
11257 lex_mode_push_list_eof(parser);
11258 }
11259
11260 LEX(PM_TOKEN_PERCENT_LOWER_I);
11261 }
11262 case 'I': {
11263 parser->current.end++;
11264
11265 if (parser->current.end < parser->end) {
11266 lex_mode_push_list(parser, true, pm_lex_percent_delimiter(parser));
11267 } else {
11268 lex_mode_push_list_eof(parser);
11269 }
11270
11271 LEX(PM_TOKEN_PERCENT_UPPER_I);
11272 }
11273 case 'r': {
11274 parser->current.end++;
11275
11276 if (parser->current.end < parser->end) {
11277 const uint8_t delimiter = pm_lex_percent_delimiter(parser);
11278 lex_mode_push_regexp(parser, lex_mode_incrementor(delimiter), lex_mode_terminator(delimiter));
11279 } else {
11280 lex_mode_push_regexp(parser, '\0', '\0');
11281 }
11282
11283 LEX(PM_TOKEN_REGEXP_BEGIN);
11284 }
11285 case 'q': {
11286 parser->current.end++;
11287
11288 if (parser->current.end < parser->end) {
11289 const uint8_t delimiter = pm_lex_percent_delimiter(parser);
11290 lex_mode_push_string(parser, false, false, lex_mode_incrementor(delimiter), lex_mode_terminator(delimiter));
11291 } else {
11292 lex_mode_push_string_eof(parser);
11293 }
11294
11295 LEX(PM_TOKEN_STRING_BEGIN);
11296 }
11297 case 'Q': {
11298 parser->current.end++;
11299
11300 if (parser->current.end < parser->end) {
11301 const uint8_t delimiter = pm_lex_percent_delimiter(parser);
11302 lex_mode_push_string(parser, true, false, lex_mode_incrementor(delimiter), lex_mode_terminator(delimiter));
11303 } else {
11304 lex_mode_push_string_eof(parser);
11305 }
11306
11307 LEX(PM_TOKEN_STRING_BEGIN);
11308 }
11309 case 's': {
11310 parser->current.end++;
11311
11312 if (parser->current.end < parser->end) {
11313 const uint8_t delimiter = pm_lex_percent_delimiter(parser);
11314 lex_mode_push_string(parser, false, false, lex_mode_incrementor(delimiter), lex_mode_terminator(delimiter));
11315 lex_state_set(parser, PM_LEX_STATE_FNAME | PM_LEX_STATE_FITEM);
11316 } else {
11317 lex_mode_push_string_eof(parser);
11318 }
11319
11320 LEX(PM_TOKEN_SYMBOL_BEGIN);
11321 }
11322 case 'w': {
11323 parser->current.end++;
11324
11325 if (parser->current.end < parser->end) {
11326 lex_mode_push_list(parser, false, pm_lex_percent_delimiter(parser));
11327 } else {
11328 lex_mode_push_list_eof(parser);
11329 }
11330
11331 LEX(PM_TOKEN_PERCENT_LOWER_W);
11332 }
11333 case 'W': {
11334 parser->current.end++;
11335
11336 if (parser->current.end < parser->end) {
11337 lex_mode_push_list(parser, true, pm_lex_percent_delimiter(parser));
11338 } else {
11339 lex_mode_push_list_eof(parser);
11340 }
11341
11342 LEX(PM_TOKEN_PERCENT_UPPER_W);
11343 }
11344 case 'x': {
11345 parser->current.end++;
11346
11347 if (parser->current.end < parser->end) {
11348 const uint8_t delimiter = pm_lex_percent_delimiter(parser);
11349 lex_mode_push_string(parser, true, false, lex_mode_incrementor(delimiter), lex_mode_terminator(delimiter));
11350 } else {
11351 lex_mode_push_string_eof(parser);
11352 }
11353
11354 LEX(PM_TOKEN_PERCENT_LOWER_X);
11355 }
11356 default:
11357 // If we get to this point, then we have a % that is completely
11358 // unparsable. In this case we'll just drop it from the parser
11359 // and skip past it and hope that the next token is something
11360 // that we can parse.
11361 pm_parser_err_current(parser, PM_ERR_INVALID_PERCENT);
11362 goto lex_next_token;
11363 }
11364 }
11365
11366 if (ambiguous_operator_p(parser, space_seen)) {
11367 PM_PARSER_WARN_TOKEN_FORMAT(parser, &parser->current, PM_WARN_AMBIGUOUS_BINARY_OPERATOR, "%", "string literal");
11368 }
11369
11370 lex_state_set(parser, lex_state_operator_p(parser) ? PM_LEX_STATE_ARG : PM_LEX_STATE_BEG);
11371 LEX(PM_TOKEN_PERCENT);
11372 }
11373
11374 // global variable
11375 case '$': {
11376 pm_token_type_t type = lex_global_variable(parser);
11377
11378 // If we're lexing an embedded variable, then we need to pop back into
11379 // the parent lex context.
11380 if (parser->lex_modes.current->mode == PM_LEX_EMBVAR) {
11381 lex_mode_pop(parser);
11382 }
11383
11384 lex_state_set(parser, PM_LEX_STATE_END);
11385 LEX(type);
11386 }
11387
11388 // instance variable, class variable
11389 case '@':
11390 lex_state_set(parser, parser->lex_state & PM_LEX_STATE_FNAME ? PM_LEX_STATE_ENDFN : PM_LEX_STATE_END);
11391 LEX(lex_at_variable(parser));
11392
11393 default: {
11394 if (*parser->current.start != '_') {
11395 size_t width = char_is_identifier_start(parser, parser->current.start, parser->end - parser->current.start);
11396
11397 // If this isn't the beginning of an identifier, then
11398 // it's an invalid token as we've exhausted all of the
11399 // other options. We'll skip past it and return the next
11400 // token after adding an appropriate error message.
11401 if (!width) {
11402 if (*parser->current.start >= 0x80) {
11403 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->current, PM_ERR_INVALID_MULTIBYTE_CHARACTER, *parser->current.start);
11404 } else if (*parser->current.start == '\\') {
11405 switch (peek_at(parser, parser->current.start + 1)) {
11406 case ' ':
11407 parser->current.end++;
11408 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->current, PM_ERR_UNEXPECTED_TOKEN_IGNORE, "escaped space");
11409 break;
11410 case '\f':
11411 parser->current.end++;
11412 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->current, PM_ERR_UNEXPECTED_TOKEN_IGNORE, "escaped form feed");
11413 break;
11414 case '\t':
11415 parser->current.end++;
11416 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->current, PM_ERR_UNEXPECTED_TOKEN_IGNORE, "escaped horizontal tab");
11417 break;
11418 case '\v':
11419 parser->current.end++;
11420 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->current, PM_ERR_UNEXPECTED_TOKEN_IGNORE, "escaped vertical tab");
11421 break;
11422 case '\r':
11423 if (peek_at(parser, parser->current.start + 2) != '\n') {
11424 parser->current.end++;
11425 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->current, PM_ERR_UNEXPECTED_TOKEN_IGNORE, "escaped carriage return");
11426 break;
11427 }
11429 default:
11430 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->current, PM_ERR_UNEXPECTED_TOKEN_IGNORE, "backslash");
11431 break;
11432 }
11433 } else if (char_is_ascii_printable(*parser->current.start)) {
11434 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->current, PM_ERR_INVALID_PRINTABLE_CHARACTER, *parser->current.start);
11435 } else {
11436 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->current, PM_ERR_INVALID_CHARACTER, *parser->current.start);
11437 }
11438
11439 goto lex_next_token;
11440 }
11441
11442 parser->current.end = parser->current.start + width;
11443 }
11444
11445 pm_token_type_t type = lex_identifier(parser, previous_command_start);
11446
11447 // If we've hit a __END__ and it was at the start of the
11448 // line or the start of the file and it is followed by
11449 // either a \n or a \r\n, then this is the last token of the
11450 // file.
11451 if (
11452 ((parser->current.end - parser->current.start) == 7) &&
11453 current_token_starts_line(parser) &&
11454 (memcmp(parser->current.start, "__END__", 7) == 0) &&
11455 (parser->current.end == parser->end || match_eol(parser))
11456 ) {
11457 // Since we know we're about to add an __END__ comment,
11458 // we know we need to add all of the newlines to get the
11459 // correct column information for it.
11460 const uint8_t *cursor = parser->current.end;
11461 while ((cursor = next_newline(cursor, parser->end - cursor)) != NULL) {
11462 pm_line_offset_list_append(&parser->metadata_arena, &parser->line_offsets, U32(++cursor - parser->start));
11463 }
11464
11465 parser->current.end = parser->end;
11466 parser->current.type = PM_TOKEN___END__;
11467 parser_lex_callback(parser);
11468
11469 parser->data_loc.start = PM_TOKEN_START(parser, &parser->current);
11470 parser->data_loc.length = PM_TOKEN_LENGTH(&parser->current);
11471
11472 LEX(PM_TOKEN_EOF);
11473 }
11474
11475 pm_lex_state_t last_state = parser->lex_state;
11476
11477 if (type == PM_TOKEN_IDENTIFIER || type == PM_TOKEN_CONSTANT || type == PM_TOKEN_METHOD_NAME) {
11478 if (lex_state_p(parser, PM_LEX_STATE_BEG_ANY | PM_LEX_STATE_ARG_ANY | PM_LEX_STATE_DOT)) {
11479 if (previous_command_start) {
11480 lex_state_set(parser, PM_LEX_STATE_CMDARG);
11481 } else {
11482 lex_state_set(parser, PM_LEX_STATE_ARG);
11483 }
11484 } else if (parser->lex_state == PM_LEX_STATE_FNAME) {
11485 lex_state_set(parser, PM_LEX_STATE_ENDFN);
11486 } else {
11487 lex_state_set(parser, PM_LEX_STATE_END);
11488 }
11489 }
11490
11491 if (
11492 !(last_state & (PM_LEX_STATE_DOT | PM_LEX_STATE_FNAME)) &&
11493 (type == PM_TOKEN_IDENTIFIER) &&
11494 ((pm_parser_local_depth(parser, &parser->current) != -1) ||
11495 pm_token_is_numbered_parameter(parser, PM_TOKEN_START(parser, &parser->current), PM_TOKEN_LENGTH(&parser->current)))
11496 ) {
11497 lex_state_set(parser, PM_LEX_STATE_END | PM_LEX_STATE_LABEL);
11498 }
11499
11500 LEX(type);
11501 }
11502 }
11503 }
11504 case PM_LEX_LIST: {
11505 if (parser->next_start != NULL) {
11506 parser->current.end = parser->next_start;
11507 parser->next_start = NULL;
11508 }
11509
11510 // First we'll set the beginning of the token.
11511 parser->current.start = parser->current.end;
11512
11513 pm_lex_mode_t *lex_mode = parser->lex_modes.current;
11514
11515 // If there's any whitespace at the start of the list, then we're
11516 // going to trim it off the beginning and create a new token.
11517 size_t whitespace;
11518
11519 if (parser->heredoc_end) {
11520 whitespace = pm_strspn_inline_whitespace(parser->current.end, parser->end - parser->current.end);
11521 if (peek_offset(parser, (ptrdiff_t)whitespace) == '\n') {
11522 whitespace += 1;
11523 }
11524 } else if (lex_mode->as.list.terminator == '\n') {
11525 // When the list delimiter is a newline (e.g. `%w` followed by a
11526 // newline), the newline is the terminator rather than a word
11527 // separator. We only trim inline whitespace here so that the
11528 // terminating newline is left for the terminator handling below.
11529 whitespace = pm_strspn_inline_whitespace(parser->current.end, parser->end - parser->current.end);
11530 } else {
11531 whitespace = pm_strspn_whitespace_newlines(parser->current.end, parser->end - parser->current.end, &parser->metadata_arena, &parser->line_offsets, PM_TOKEN_END(parser, &parser->current));
11532 }
11533
11534 if (whitespace > 0) {
11535 parser->current.end += whitespace;
11536 if (peek_offset(parser, -1) == '\n') {
11537 // mutates next_start
11538 parser_flush_heredoc_end(parser);
11539 }
11540
11541 lex_mode->as.list.started = true;
11542 lex_mode->as.list.separated = true;
11543 LEX(PM_TOKEN_WORDS_SEP);
11544 }
11545
11546 // We'll check if we're at the end of the file. If we are, then we
11547 // need to return the EOF token.
11548 if (parser->current.end >= parser->end) {
11549 LEX(PM_TOKEN_EOF);
11550 }
11551
11552 /* A word separator delimits the opener from the first word (or
11553 * from the terminator when the list is empty). The parser accepts
11554 * the words without it, so when the list does not start with
11555 * whitespace the implicit separator goes to the lex callback alone
11556 * and lexing continues on to the token the parser receives. */
11557 if (!lex_mode->as.list.started) {
11558 lex_mode->as.list.started = true;
11559 lex_mode->as.list.separated = true;
11560 parser->current.type = PM_TOKEN_WORDS_SEP_IMPLICIT;
11561 parser_lex_callback(parser);
11562 }
11563
11564 // Here we'll get a list of the places where strpbrk should break,
11565 // and then find the first one.
11566 const uint8_t *breakpoints = lex_mode->as.list.breakpoints;
11567 const uint8_t *breakpoint = pm_strpbrk(parser, parser->current.end, breakpoints, parser->end - parser->current.end, true);
11568
11569 // If we haven't found an escape yet, then this buffer will be
11570 // unallocated since we can refer directly to the source string.
11571 pm_token_buffer_t token_buffer = { 0 };
11572
11573 while (breakpoint != NULL) {
11574 // If we hit whitespace, then we must have received content by
11575 // now, so we can return an element of the list. A whitespace
11576 // character that is also the terminator (e.g. a newline
11577 // delimiter) is handled by the terminator check below, not here.
11578 if (pm_char_is_whitespace(*breakpoint) && *breakpoint != lex_mode->as.list.terminator) {
11579 parser->current.end = breakpoint;
11580 pm_token_buffer_flush(parser, &token_buffer);
11581 lex_mode->as.list.separated = false;
11582 LEX(PM_TOKEN_STRING_CONTENT);
11583 }
11584
11585 // If we hit the terminator, we need to check which token to
11586 // return.
11587 if (*breakpoint == lex_mode->as.list.terminator) {
11588 // If this terminator doesn't actually close the list, then
11589 // we need to continue on past it.
11590 if (lex_mode->as.list.nesting > 0) {
11591 parser->current.end = breakpoint + 1;
11592 breakpoint = pm_strpbrk(parser, parser->current.end, breakpoints, parser->end - parser->current.end, true);
11593 lex_mode->as.list.nesting--;
11594 continue;
11595 }
11596
11597 // If we've hit the terminator and we've already skipped
11598 // past content, then we can return a list node.
11599 if (breakpoint > parser->current.start) {
11600 parser->current.end = breakpoint;
11601 pm_token_buffer_flush(parser, &token_buffer);
11602 lex_mode->as.list.separated = false;
11603 LEX(PM_TOKEN_STRING_CONTENT);
11604 }
11605
11606 /* A word separator delimits the last word from the
11607 * terminator. The parser accepts the terminator without
11608 * it, so when the list does not end with whitespace the
11609 * implicit separator goes to the lex callback alone and
11610 * lexing continues on to the terminator. */
11611 if (!lex_mode->as.list.separated) {
11612 lex_mode->as.list.separated = true;
11613 parser->current.end = breakpoint;
11614 parser->current.type = PM_TOKEN_WORDS_SEP_IMPLICIT;
11615 parser_lex_callback(parser);
11616 }
11617
11618 // Otherwise, switch back to the default state and return
11619 // the end of the list.
11620 parser->current.end = breakpoint + 1;
11621
11622 // If the terminator is a newline (i.e. the list delimiter
11623 // was a newline), then we need to record it so that line
11624 // numbers after the list remain accurate.
11625 if (*breakpoint == '\n') {
11626 pm_line_offset_list_append(&parser->metadata_arena, &parser->line_offsets, PM_TOKEN_END(parser, &parser->current));
11627 }
11628
11629 lex_mode_pop(parser);
11630 lex_state_set(parser, PM_LEX_STATE_END);
11631 LEX(PM_TOKEN_STRING_END);
11632 }
11633
11634 // If we hit a null byte, skip directly past it.
11635 if (*breakpoint == '\0') {
11636 breakpoint = pm_strpbrk(parser, breakpoint + 1, breakpoints, parser->end - (breakpoint + 1), true);
11637 continue;
11638 }
11639
11640 // If we hit escapes, then we need to treat the next token
11641 // literally. In this case we'll skip past the next character
11642 // and find the next breakpoint.
11643 if (*breakpoint == '\\') {
11644 parser->current.end = breakpoint + 1;
11645
11646 // If we've hit the end of the file, then break out of the
11647 // loop by setting the breakpoint to NULL.
11648 if (parser->current.end == parser->end) {
11649 breakpoint = NULL;
11650 continue;
11651 }
11652
11653 pm_token_buffer_escape(parser, &token_buffer);
11654 uint8_t peeked = peek(parser);
11655
11656 switch (peeked) {
11657 case ' ':
11658 case '\f':
11659 case '\t':
11660 case '\v':
11661 case '\\':
11662 pm_token_buffer_push_byte(&token_buffer, peeked);
11663 parser->current.end++;
11664 break;
11665 case '\r':
11666 parser->current.end++;
11667 if (peek(parser) != '\n') {
11668 pm_token_buffer_push_byte(&token_buffer, '\r');
11669 break;
11670 }
11672 case '\n':
11673 pm_token_buffer_push_byte(&token_buffer, '\n');
11674
11675 if (parser->heredoc_end) {
11676 // ... if we are on the same line as a heredoc,
11677 // flush the heredoc and continue parsing after
11678 // heredoc_end.
11679 parser_flush_heredoc_end(parser);
11680 pm_token_buffer_copy(parser, &token_buffer);
11681 LEX(PM_TOKEN_STRING_CONTENT);
11682 } else {
11683 // ... else track the newline.
11684 pm_line_offset_list_append(&parser->metadata_arena, &parser->line_offsets, PM_TOKEN_END(parser, &parser->current) + 1);
11685 }
11686
11687 parser->current.end++;
11688 break;
11689 default:
11690 if (peeked == lex_mode->as.list.incrementor || peeked == lex_mode->as.list.terminator) {
11691 pm_token_buffer_push_byte(&token_buffer, peeked);
11692 parser->current.end++;
11693 } else if (lex_mode->as.list.interpolation) {
11694 escape_read(parser, &token_buffer.buffer, NULL, PM_ESCAPE_FLAG_NONE);
11695 } else {
11696 pm_token_buffer_push_byte(&token_buffer, '\\');
11697 pm_token_buffer_push_escaped(&token_buffer, parser);
11698 }
11699
11700 break;
11701 }
11702
11703 token_buffer.cursor = parser->current.end;
11704 breakpoint = pm_strpbrk(parser, parser->current.end, breakpoints, parser->end - parser->current.end, true);
11705 continue;
11706 }
11707
11708 // If we hit a #, then we will attempt to lex interpolation.
11709 if (*breakpoint == '#') {
11710 pm_token_type_t type = lex_interpolation(parser, breakpoint);
11711
11712 if (!type) {
11713 // If we haven't returned at this point then we had something
11714 // that looked like an interpolated class or instance variable
11715 // like "#@" but wasn't actually. In this case we'll just skip
11716 // to the next breakpoint.
11717 breakpoint = pm_strpbrk(parser, parser->current.end, breakpoints, parser->end - parser->current.end, true);
11718 continue;
11719 }
11720
11721 if (type == PM_TOKEN_STRING_CONTENT) {
11722 pm_token_buffer_flush(parser, &token_buffer);
11723 }
11724
11725 lex_mode->as.list.separated = false;
11726 LEX(type);
11727 }
11728
11729 // If we've hit the incrementor, then we need to skip past it
11730 // and find the next breakpoint.
11731 assert(*breakpoint == lex_mode->as.list.incrementor);
11732 parser->current.end = breakpoint + 1;
11733 breakpoint = pm_strpbrk(parser, parser->current.end, breakpoints, parser->end - parser->current.end, true);
11734 lex_mode->as.list.nesting++;
11735 continue;
11736 }
11737
11738 if (parser->current.end > parser->current.start) {
11739 pm_token_buffer_flush(parser, &token_buffer);
11740 LEX(PM_TOKEN_STRING_CONTENT);
11741 }
11742
11743 // If we were unable to find a breakpoint, then this token hits the
11744 // end of the file.
11745 parser->current.end = parser->end;
11746 pm_token_buffer_flush(parser, &token_buffer);
11747 LEX(PM_TOKEN_STRING_CONTENT);
11748 }
11749 case PM_LEX_REGEXP: {
11750 // First, we'll set to start of this token to be the current end.
11751 if (parser->next_start == NULL) {
11752 parser->current.start = parser->current.end;
11753 } else {
11754 parser->current.start = parser->next_start;
11755 parser->current.end = parser->next_start;
11756 parser->next_start = NULL;
11757 }
11758
11759 // We'll check if we're at the end of the file. If we are, then we
11760 // need to return the EOF token.
11761 if (parser->current.end >= parser->end) {
11762 LEX(PM_TOKEN_EOF);
11763 }
11764
11765 // Get a reference to the current mode.
11766 pm_lex_mode_t *lex_mode = parser->lex_modes.current;
11767
11768 // These are the places where we need to split up the content of the
11769 // regular expression. We'll use strpbrk to find the first of these
11770 // characters.
11771 const uint8_t *breakpoints = lex_mode->as.regexp.breakpoints;
11772 const uint8_t *breakpoint = pm_strpbrk(parser, parser->current.end, breakpoints, parser->end - parser->current.end, false);
11773 pm_regexp_token_buffer_t token_buffer = { 0 };
11774
11775 while (breakpoint != NULL) {
11776 uint8_t term = lex_mode->as.regexp.terminator;
11777 bool is_terminator = (*breakpoint == term);
11778
11779 // If the terminator is newline, we need to consider \r\n _also_ a newline
11780 // For example: `%\nfoo\r\n`
11781 // The string should be "foo", not "foo\r"
11782 if (*breakpoint == '\r' && peek_at(parser, breakpoint + 1) == '\n') {
11783 if (term == '\n') {
11784 is_terminator = true;
11785 }
11786
11787 // If the terminator is a CR, but we see a CRLF, we need to
11788 // treat the CRLF as a newline, meaning this is _not_ the
11789 // terminator
11790 if (term == '\r') {
11791 is_terminator = false;
11792 }
11793 }
11794
11795 // If we hit the terminator, we need to determine what kind of
11796 // token to return.
11797 if (is_terminator) {
11798 if (lex_mode->as.regexp.nesting > 0) {
11799 parser->current.end = breakpoint + 1;
11800 breakpoint = pm_strpbrk(parser, parser->current.end, breakpoints, parser->end - parser->current.end, false);
11801 lex_mode->as.regexp.nesting--;
11802 continue;
11803 }
11804
11805 // Here we've hit the terminator. If we have already consumed
11806 // content then we need to return that content as string content
11807 // first.
11808 if (breakpoint > parser->current.start) {
11809 parser->current.end = breakpoint;
11810 pm_regexp_token_buffer_flush(parser, &token_buffer);
11811 LEX(PM_TOKEN_STRING_CONTENT);
11812 }
11813
11814 // Check here if we need to track the newline.
11815 size_t eol_length = match_eol_at(parser, breakpoint);
11816 if (eol_length) {
11817 parser->current.end = breakpoint + eol_length;
11818
11819 // Track the newline if we're not in a heredoc that
11820 // would have already have added the newline to the
11821 // list.
11822 if (parser->heredoc_end == NULL) {
11823 pm_line_offset_list_append(&parser->metadata_arena, &parser->line_offsets, PM_TOKEN_END(parser, &parser->current));
11824 }
11825 } else {
11826 parser->current.end = breakpoint + 1;
11827 }
11828
11829 // Since we've hit the terminator of the regular expression,
11830 // we now need to parse the options.
11831 parser->current.end += pm_strspn_regexp_option(parser->current.end, parser->end - parser->current.end);
11832
11833 lex_mode_pop(parser);
11834 lex_state_set(parser, PM_LEX_STATE_END);
11835 LEX(PM_TOKEN_REGEXP_END);
11836 }
11837
11838 // If we've hit the incrementor, then we need to skip past it
11839 // and find the next breakpoint.
11840 if (*breakpoint && *breakpoint == lex_mode->as.regexp.incrementor) {
11841 parser->current.end = breakpoint + 1;
11842 breakpoint = pm_strpbrk(parser, parser->current.end, breakpoints, parser->end - parser->current.end, false);
11843 lex_mode->as.regexp.nesting++;
11844 continue;
11845 }
11846
11847 switch (*breakpoint) {
11848 case '\0':
11849 // If we hit a null byte, skip directly past it.
11850 parser->current.end = breakpoint + 1;
11851 breakpoint = pm_strpbrk(parser, parser->current.end, breakpoints, parser->end - parser->current.end, false);
11852 break;
11853 case '\r':
11854 if (peek_at(parser, breakpoint + 1) != '\n') {
11855 parser->current.end = breakpoint + 1;
11856 breakpoint = pm_strpbrk(parser, parser->current.end, breakpoints, parser->end - parser->current.end, false);
11857 break;
11858 }
11859
11860 breakpoint++;
11861 parser->current.end = breakpoint;
11862 pm_regexp_token_buffer_escape(parser, &token_buffer);
11863 token_buffer.base.cursor = breakpoint;
11864
11866 case '\n':
11867 // If we've hit a newline, then we need to track that in
11868 // the list of newlines.
11869 if (parser->heredoc_end == NULL) {
11870 pm_line_offset_list_append(&parser->metadata_arena, &parser->line_offsets, U32(breakpoint - parser->start + 1));
11871 parser->current.end = breakpoint + 1;
11872 breakpoint = pm_strpbrk(parser, parser->current.end, breakpoints, parser->end - parser->current.end, false);
11873 break;
11874 }
11875
11876 parser->current.end = breakpoint + 1;
11877 parser_flush_heredoc_end(parser);
11878 pm_regexp_token_buffer_flush(parser, &token_buffer);
11879 LEX(PM_TOKEN_STRING_CONTENT);
11880 case '\\': {
11881 // If we hit escapes, then we need to treat the next
11882 // token literally. In this case we'll skip past the
11883 // next character and find the next breakpoint.
11884 parser->current.end = breakpoint + 1;
11885
11886 // If we've hit the end of the file, then break out of
11887 // the loop by setting the breakpoint to NULL.
11888 if (parser->current.end == parser->end) {
11889 breakpoint = NULL;
11890 break;
11891 }
11892
11893 pm_regexp_token_buffer_escape(parser, &token_buffer);
11894 uint8_t peeked = peek(parser);
11895
11896 switch (peeked) {
11897 case '\r':
11898 parser->current.end++;
11899 if (peek(parser) != '\n') {
11900 if (lex_mode->as.regexp.terminator != '\r') {
11901 pm_token_buffer_push_byte(&token_buffer.base, '\\');
11902 }
11903 pm_regexp_token_buffer_push_byte(&token_buffer, '\r');
11904 pm_token_buffer_push_byte(&token_buffer.base, '\r');
11905 break;
11906 }
11908 case '\n':
11909 if (parser->heredoc_end) {
11910 // ... if we are on the same line as a heredoc,
11911 // flush the heredoc and continue parsing after
11912 // heredoc_end.
11913 parser_flush_heredoc_end(parser);
11914 pm_regexp_token_buffer_copy(parser, &token_buffer);
11915 LEX(PM_TOKEN_STRING_CONTENT);
11916 } else {
11917 // ... else track the newline.
11918 pm_line_offset_list_append(&parser->metadata_arena, &parser->line_offsets, PM_TOKEN_END(parser, &parser->current) + 1);
11919 }
11920
11921 parser->current.end++;
11922 break;
11923 case 'c':
11924 case 'C':
11925 case 'M':
11926 case 'u':
11927 case 'x':
11928 escape_read(parser, &token_buffer.regexp_buffer, &token_buffer.base.buffer, PM_ESCAPE_FLAG_REGEXP);
11929 break;
11930 default:
11931 if (lex_mode->as.regexp.terminator == peeked) {
11932 // Some characters when they are used as the
11933 // terminator also receive an escape. They are
11934 // enumerated here.
11935 switch (peeked) {
11936 case '$': case ')': case '*': case '+':
11937 case '.': case '>': case '?': case ']':
11938 case '^': case '|': case '}':
11939 pm_token_buffer_push_byte(&token_buffer.base, '\\');
11940 break;
11941 default:
11942 break;
11943 }
11944
11945 pm_regexp_token_buffer_push_byte(&token_buffer, peeked);
11946 pm_token_buffer_push_byte(&token_buffer.base, peeked);
11947 parser->current.end++;
11948 break;
11949 }
11950
11951 if (peeked < 0x80) pm_token_buffer_push_byte(&token_buffer.base, '\\');
11952 pm_regexp_token_buffer_push_escaped(&token_buffer, parser);
11953 break;
11954 }
11955
11956 token_buffer.base.cursor = parser->current.end;
11957 breakpoint = pm_strpbrk(parser, parser->current.end, breakpoints, parser->end - parser->current.end, false);
11958 break;
11959 }
11960 case '#': {
11961 // If we hit a #, then we will attempt to lex
11962 // interpolation.
11963 pm_token_type_t type = lex_interpolation(parser, breakpoint);
11964
11965 if (!type) {
11966 // If we haven't returned at this point then we had
11967 // something that looked like an interpolated class or
11968 // instance variable like "#@" but wasn't actually. In
11969 // this case we'll just skip to the next breakpoint.
11970 breakpoint = pm_strpbrk(parser, parser->current.end, breakpoints, parser->end - parser->current.end, false);
11971 break;
11972 }
11973
11974 if (type == PM_TOKEN_STRING_CONTENT) {
11975 pm_regexp_token_buffer_flush(parser, &token_buffer);
11976 }
11977
11978 LEX(type);
11979 }
11980 default:
11981 assert(false && "unreachable");
11982 break;
11983 }
11984 }
11985
11986 if (parser->current.end > parser->current.start) {
11987 pm_regexp_token_buffer_flush(parser, &token_buffer);
11988 LEX(PM_TOKEN_STRING_CONTENT);
11989 }
11990
11991 // If we were unable to find a breakpoint, then this token hits the
11992 // end of the file.
11993 parser->current.end = parser->end;
11994 pm_regexp_token_buffer_flush(parser, &token_buffer);
11995 LEX(PM_TOKEN_STRING_CONTENT);
11996 }
11997 case PM_LEX_STRING: {
11998 // First, we'll set to start of this token to be the current end.
11999 if (parser->next_start == NULL) {
12000 parser->current.start = parser->current.end;
12001 } else {
12002 parser->current.start = parser->next_start;
12003 parser->current.end = parser->next_start;
12004 parser->next_start = NULL;
12005 }
12006
12007 // We'll check if we're at the end of the file. If we are, then we need to
12008 // return the EOF token.
12009 if (parser->current.end >= parser->end) {
12010 LEX(PM_TOKEN_EOF);
12011 }
12012
12013 // These are the places where we need to split up the content of the
12014 // string. We'll use strpbrk to find the first of these characters.
12015 pm_lex_mode_t *lex_mode = parser->lex_modes.current;
12016 const uint8_t *breakpoints = lex_mode->as.string.breakpoints;
12017 const uint8_t *breakpoint = pm_strpbrk(parser, parser->current.end, breakpoints, parser->end - parser->current.end, true);
12018
12019 // If we haven't found an escape yet, then this buffer will be
12020 // unallocated since we can refer directly to the source string.
12021 pm_token_buffer_t token_buffer = { 0 };
12022
12023 while (breakpoint != NULL) {
12024 // If we hit the incrementor, then we'll increment then nesting and
12025 // continue lexing.
12026 if (lex_mode->as.string.incrementor != '\0' && *breakpoint == lex_mode->as.string.incrementor) {
12027 lex_mode->as.string.nesting++;
12028 parser->current.end = breakpoint + 1;
12029 breakpoint = pm_strpbrk(parser, parser->current.end, breakpoints, parser->end - parser->current.end, true);
12030 continue;
12031 }
12032
12033 uint8_t term = lex_mode->as.string.terminator;
12034 bool is_terminator = (*breakpoint == term);
12035
12036 // If the terminator is newline, we need to consider \r\n _also_ a newline
12037 // For example: `%r\nfoo\r\n`
12038 // The string should be /foo/, not /foo\r/
12039 if (*breakpoint == '\r' && peek_at(parser, breakpoint + 1) == '\n') {
12040 if (term == '\n') {
12041 is_terminator = true;
12042 }
12043
12044 // If the terminator is a CR, but we see a CRLF, we need to
12045 // treat the CRLF as a newline, meaning this is _not_ the
12046 // terminator
12047 if (term == '\r') {
12048 is_terminator = false;
12049 }
12050 }
12051
12052 // Note that we have to check the terminator here first because we could
12053 // potentially be parsing a % string that has a # character as the
12054 // terminator.
12055 if (is_terminator) {
12056 // If this terminator doesn't actually close the string, then we need
12057 // to continue on past it.
12058 if (lex_mode->as.string.nesting > 0) {
12059 parser->current.end = breakpoint + 1;
12060 breakpoint = pm_strpbrk(parser, parser->current.end, breakpoints, parser->end - parser->current.end, true);
12061 lex_mode->as.string.nesting--;
12062 continue;
12063 }
12064
12065 // Here we've hit the terminator. If we have already consumed content
12066 // then we need to return that content as string content first.
12067 if (breakpoint > parser->current.start) {
12068 parser->current.end = breakpoint;
12069 pm_token_buffer_flush(parser, &token_buffer);
12070 LEX(PM_TOKEN_STRING_CONTENT);
12071 }
12072
12073 // Otherwise we need to switch back to the parent lex mode and
12074 // return the end of the string.
12075 size_t eol_length = match_eol_at(parser, breakpoint);
12076 if (eol_length) {
12077 parser->current.end = breakpoint + eol_length;
12078
12079 // Track the newline if we're not in a heredoc that
12080 // would have already have added the newline to the
12081 // list.
12082 if (parser->heredoc_end == NULL) {
12083 pm_line_offset_list_append(&parser->metadata_arena, &parser->line_offsets, PM_TOKEN_END(parser, &parser->current));
12084 }
12085 } else {
12086 parser->current.end = breakpoint + 1;
12087 }
12088
12089 if (lex_mode->as.string.label_allowed && (peek(parser) == ':') && (peek_offset(parser, 1) != ':')) {
12090 parser->current.end++;
12091 lex_state_set(parser, PM_LEX_STATE_ARG | PM_LEX_STATE_LABELED);
12092 lex_mode_pop(parser);
12093 LEX(PM_TOKEN_LABEL_END);
12094 }
12095
12096 // When the delimiter itself is a newline, we won't
12097 // get a chance to flush heredocs in the usual places since
12098 // the newline is already consumed.
12099 if (term == '\n' && parser->heredoc_end) {
12100 parser_flush_heredoc_end(parser);
12101 }
12102
12103 lex_state_set(parser, PM_LEX_STATE_END);
12104 lex_mode_pop(parser);
12105 LEX(PM_TOKEN_STRING_END);
12106 }
12107
12108 switch (*breakpoint) {
12109 case '\0':
12110 // Skip directly past the null character.
12111 parser->current.end = breakpoint + 1;
12112 breakpoint = pm_strpbrk(parser, parser->current.end, breakpoints, parser->end - parser->current.end, true);
12113 break;
12114 case '\r':
12115 if (peek_at(parser, breakpoint + 1) != '\n') {
12116 parser->current.end = breakpoint + 1;
12117 breakpoint = pm_strpbrk(parser, parser->current.end, breakpoints, parser->end - parser->current.end, true);
12118 break;
12119 }
12120
12121 // If we hit a \r\n sequence, then we need to treat it
12122 // as a newline.
12123 breakpoint++;
12124 parser->current.end = breakpoint;
12125 pm_token_buffer_escape(parser, &token_buffer);
12126 token_buffer.cursor = breakpoint;
12127
12129 case '\n':
12130 // When we hit a newline, we need to flush any potential
12131 // heredocs. Note that this has to happen after we check
12132 // for the terminator in case the terminator is a
12133 // newline character.
12134 if (parser->heredoc_end == NULL) {
12135 pm_line_offset_list_append(&parser->metadata_arena, &parser->line_offsets, U32(breakpoint - parser->start + 1));
12136 parser->current.end = breakpoint + 1;
12137 breakpoint = pm_strpbrk(parser, parser->current.end, breakpoints, parser->end - parser->current.end, true);
12138 break;
12139 }
12140
12141 parser->current.end = breakpoint + 1;
12142 parser_flush_heredoc_end(parser);
12143 pm_token_buffer_flush(parser, &token_buffer);
12144 LEX(PM_TOKEN_STRING_CONTENT);
12145 case '\\': {
12146 // Here we hit escapes.
12147 parser->current.end = breakpoint + 1;
12148
12149 // If we've hit the end of the file, then break out of
12150 // the loop by setting the breakpoint to NULL.
12151 if (parser->current.end == parser->end) {
12152 breakpoint = NULL;
12153 continue;
12154 }
12155
12156 pm_token_buffer_escape(parser, &token_buffer);
12157 uint8_t peeked = peek(parser);
12158
12159 switch (peeked) {
12160 case '\\':
12161 pm_token_buffer_push_byte(&token_buffer, '\\');
12162 parser->current.end++;
12163 break;
12164 case '\r':
12165 parser->current.end++;
12166 if (peek(parser) != '\n') {
12167 if (!lex_mode->as.string.interpolation) {
12168 pm_token_buffer_push_byte(&token_buffer, '\\');
12169 }
12170 pm_token_buffer_push_byte(&token_buffer, '\r');
12171 break;
12172 }
12174 case '\n':
12175 if (!lex_mode->as.string.interpolation) {
12176 pm_token_buffer_push_byte(&token_buffer, '\\');
12177 pm_token_buffer_push_byte(&token_buffer, '\n');
12178 }
12179
12180 if (parser->heredoc_end) {
12181 // ... if we are on the same line as a heredoc,
12182 // flush the heredoc and continue parsing after
12183 // heredoc_end.
12184 parser_flush_heredoc_end(parser);
12185 pm_token_buffer_copy(parser, &token_buffer);
12186 LEX(PM_TOKEN_STRING_CONTENT);
12187 } else {
12188 // ... else track the newline.
12189 pm_line_offset_list_append(&parser->metadata_arena, &parser->line_offsets, PM_TOKEN_END(parser, &parser->current) + 1);
12190 }
12191
12192 parser->current.end++;
12193 break;
12194 default:
12195 if (lex_mode->as.string.incrementor != '\0' && peeked == lex_mode->as.string.incrementor) {
12196 pm_token_buffer_push_byte(&token_buffer, peeked);
12197 parser->current.end++;
12198 } else if (lex_mode->as.string.terminator != '\0' && peeked == lex_mode->as.string.terminator) {
12199 pm_token_buffer_push_byte(&token_buffer, peeked);
12200 parser->current.end++;
12201 } else if (lex_mode->as.string.interpolation) {
12202 escape_read(parser, &token_buffer.buffer, NULL, PM_ESCAPE_FLAG_NONE);
12203 } else {
12204 pm_token_buffer_push_byte(&token_buffer, '\\');
12205 pm_token_buffer_push_escaped(&token_buffer, parser);
12206 }
12207
12208 break;
12209 }
12210
12211 token_buffer.cursor = parser->current.end;
12212 breakpoint = pm_strpbrk(parser, parser->current.end, breakpoints, parser->end - parser->current.end, true);
12213 break;
12214 }
12215 case '#': {
12216 pm_token_type_t type = lex_interpolation(parser, breakpoint);
12217
12218 if (!type) {
12219 // If we haven't returned at this point then we had something that
12220 // looked like an interpolated class or instance variable like "#@"
12221 // but wasn't actually. In this case we'll just skip to the next
12222 // breakpoint.
12223 breakpoint = pm_strpbrk(parser, parser->current.end, breakpoints, parser->end - parser->current.end, true);
12224 break;
12225 }
12226
12227 if (type == PM_TOKEN_STRING_CONTENT) {
12228 pm_token_buffer_flush(parser, &token_buffer);
12229 }
12230
12231 LEX(type);
12232 }
12233 default:
12234 assert(false && "unreachable");
12235 }
12236 }
12237
12238 if (parser->current.end > parser->current.start) {
12239 pm_token_buffer_flush(parser, &token_buffer);
12240 LEX(PM_TOKEN_STRING_CONTENT);
12241 }
12242
12243 // If we've hit the end of the string, then this is an unterminated
12244 // string. In that case we'll return a string content token.
12245 parser->current.end = parser->end;
12246 pm_token_buffer_flush(parser, &token_buffer);
12247 LEX(PM_TOKEN_STRING_CONTENT);
12248 }
12249 case PM_LEX_HEREDOC: {
12250 // First, we'll set to start of this token.
12251 if (parser->next_start == NULL) {
12252 parser->current.start = parser->current.end;
12253 } else {
12254 parser->current.start = parser->next_start;
12255 parser->current.end = parser->next_start;
12256 parser->heredoc_end = NULL;
12257 parser->next_start = NULL;
12258 }
12259
12260 // Now let's grab the information about the identifier off of the
12261 // current lex mode.
12262 pm_lex_mode_t *lex_mode = parser->lex_modes.current;
12263 pm_heredoc_lex_mode_t *heredoc_lex_mode = &lex_mode->as.heredoc.base;
12264
12265 bool line_continuation = lex_mode->as.heredoc.line_continuation;
12266 lex_mode->as.heredoc.line_continuation = false;
12267
12268 // We'll check if we're at the end of the file. If we are, then we
12269 // will add an error (because we weren't able to find the
12270 // terminator) but still continue parsing so that content after the
12271 // declaration of the heredoc can be parsed.
12272 if (parser->current.end >= parser->end) {
12273 pm_parser_err_heredoc_term(parser, heredoc_lex_mode->ident_start, heredoc_lex_mode->ident_length);
12274 parser->next_start = lex_mode->as.heredoc.next_start;
12275 parser->heredoc_end = parser->current.end;
12276 lex_state_set(parser, PM_LEX_STATE_END);
12277 lex_mode_pop(parser);
12278 LEX(PM_TOKEN_HEREDOC_END);
12279 }
12280
12281 const uint8_t *ident_start = heredoc_lex_mode->ident_start;
12282 size_t ident_length = heredoc_lex_mode->ident_length;
12283
12284 // If we are immediately following a newline and we have hit the
12285 // terminator, then we need to return the ending of the heredoc.
12286 if (current_token_starts_line(parser)) {
12287 const uint8_t *start = parser->current.start;
12288
12289 if (!line_continuation && (start + ident_length <= parser->end)) {
12290 const uint8_t *newline = next_newline(start, parser->end - start);
12291 const uint8_t *ident_end = newline;
12292 const uint8_t *terminator_end = newline;
12293
12294 if (newline == NULL) {
12295 terminator_end = parser->end;
12296 ident_end = parser->end;
12297 } else {
12298 terminator_end++;
12299 if (newline[-1] == '\r') {
12300 ident_end--; // Remove \r
12301 }
12302 }
12303
12304 const uint8_t *terminator_start = ident_end - ident_length;
12305 const uint8_t *cursor = start;
12306
12307 if (heredoc_lex_mode->indent == PM_HEREDOC_INDENT_DASH || heredoc_lex_mode->indent == PM_HEREDOC_INDENT_TILDE) {
12308 while (cursor < terminator_start && pm_char_is_inline_whitespace(*cursor)) {
12309 cursor++;
12310 }
12311 }
12312
12313 if (
12314 (cursor == terminator_start) &&
12315 (memcmp(terminator_start, ident_start, ident_length) == 0)
12316 ) {
12317 if (newline != NULL) {
12318 pm_line_offset_list_append(&parser->metadata_arena, &parser->line_offsets, U32(newline - parser->start + 1));
12319 }
12320
12321 parser->current.end = terminator_end;
12322 if (*lex_mode->as.heredoc.next_start == '\\') {
12323 parser->next_start = NULL;
12324 } else {
12325 parser->next_start = lex_mode->as.heredoc.next_start;
12326 parser->heredoc_end = parser->current.end;
12327 }
12328
12329 lex_state_set(parser, PM_LEX_STATE_END);
12330 lex_mode_pop(parser);
12331 LEX(PM_TOKEN_HEREDOC_END);
12332 }
12333 }
12334
12335 size_t whitespace = pm_heredoc_strspn_inline_whitespace(parser, &start, heredoc_lex_mode->indent);
12336 if (
12337 heredoc_lex_mode->indent == PM_HEREDOC_INDENT_TILDE &&
12338 lex_mode->as.heredoc.common_whitespace != NULL &&
12339 (*lex_mode->as.heredoc.common_whitespace > whitespace) &&
12340 peek_at(parser, start) != '\n'
12341 ) {
12342 *lex_mode->as.heredoc.common_whitespace = whitespace;
12343 }
12344 }
12345
12346 // Otherwise we'll be parsing string content. These are the places
12347 // where we need to split up the content of the heredoc. We'll use
12348 // strpbrk to find the first of these characters.
12349 uint8_t breakpoints[PM_STRPBRK_CACHE_SIZE] = "\r\n\\#";
12350
12351 pm_heredoc_quote_t quote = heredoc_lex_mode->quote;
12352 if (quote == PM_HEREDOC_QUOTE_SINGLE) {
12353 breakpoints[3] = '\0';
12354 }
12355
12356 const uint8_t *breakpoint = pm_strpbrk(parser, parser->current.end, breakpoints, parser->end - parser->current.end, true);
12357 pm_token_buffer_t token_buffer = { 0 };
12358 bool was_line_continuation = false;
12359
12360 while (breakpoint != NULL) {
12361 switch (*breakpoint) {
12362 case '\0':
12363 // Skip directly past the null character.
12364 parser->current.end = breakpoint + 1;
12365 breakpoint = pm_strpbrk(parser, parser->current.end, breakpoints, parser->end - parser->current.end, true);
12366 break;
12367 case '\r':
12368 parser->current.end = breakpoint + 1;
12369
12370 if (peek_at(parser, breakpoint + 1) != '\n') {
12371 breakpoint = pm_strpbrk(parser, parser->current.end, breakpoints, parser->end - parser->current.end, true);
12372 break;
12373 }
12374
12375 // If we hit a \r\n sequence, then we want to replace it
12376 // with a single \n character in the final string.
12377 breakpoint++;
12378 pm_token_buffer_escape(parser, &token_buffer);
12379 token_buffer.cursor = breakpoint;
12380
12382 case '\n': {
12383 if (parser->heredoc_end != NULL && (parser->heredoc_end > breakpoint)) {
12384 parser_flush_heredoc_end(parser);
12385 parser->current.end = breakpoint + 1;
12386 pm_token_buffer_flush(parser, &token_buffer);
12387 LEX(PM_TOKEN_STRING_CONTENT);
12388 }
12389
12390 pm_line_offset_list_append(&parser->metadata_arena, &parser->line_offsets, U32(breakpoint - parser->start + 1));
12391
12392 // If we have a - or ~ heredoc, then we can match after
12393 // some leading whitespace.
12394 const uint8_t *start = breakpoint + 1;
12395
12396 if (!was_line_continuation && (start + ident_length <= parser->end)) {
12397 // We want to match the terminator starting from the end of the line in case
12398 // there is whitespace in the ident such as <<-' DOC' or <<~' DOC'.
12399 const uint8_t *newline = next_newline(start, parser->end - start);
12400
12401 if (newline == NULL) {
12402 newline = parser->end;
12403 } else if (newline[-1] == '\r') {
12404 newline--; // Remove \r
12405 }
12406
12407 // Start of a possible terminator.
12408 const uint8_t *terminator_start = newline - ident_length;
12409
12410 // Cursor to check for the leading whitespace. We skip the
12411 // leading whitespace if we have a - or ~ heredoc.
12412 const uint8_t *cursor = start;
12413
12414 if (heredoc_lex_mode->indent == PM_HEREDOC_INDENT_DASH || heredoc_lex_mode->indent == PM_HEREDOC_INDENT_TILDE) {
12415 while (cursor < terminator_start && pm_char_is_inline_whitespace(*cursor)) {
12416 cursor++;
12417 }
12418 }
12419
12420 if (
12421 cursor == terminator_start &&
12422 (memcmp(terminator_start, ident_start, ident_length) == 0)
12423 ) {
12424 parser->current.end = breakpoint + 1;
12425 pm_token_buffer_flush(parser, &token_buffer);
12426 LEX(PM_TOKEN_STRING_CONTENT);
12427 }
12428 }
12429
12430 size_t whitespace = pm_heredoc_strspn_inline_whitespace(parser, &start, lex_mode->as.heredoc.base.indent);
12431
12432 // If we have hit a newline that is followed by a valid
12433 // terminator, then we need to return the content of the
12434 // heredoc here as string content. Then, the next time a
12435 // token is lexed, it will match again and return the
12436 // end of the heredoc.
12437 if (lex_mode->as.heredoc.base.indent == PM_HEREDOC_INDENT_TILDE) {
12438 if ((lex_mode->as.heredoc.common_whitespace != NULL) && (*lex_mode->as.heredoc.common_whitespace > whitespace) && peek_at(parser, start) != '\n') {
12439 *lex_mode->as.heredoc.common_whitespace = whitespace;
12440 }
12441
12442 parser->current.end = breakpoint + 1;
12443 pm_token_buffer_flush(parser, &token_buffer);
12444 LEX(PM_TOKEN_STRING_CONTENT);
12445 }
12446
12447 // Otherwise we hit a newline and it wasn't followed by
12448 // a terminator, so we can continue parsing.
12449 parser->current.end = breakpoint + 1;
12450 breakpoint = pm_strpbrk(parser, parser->current.end, breakpoints, parser->end - parser->current.end, true);
12451 break;
12452 }
12453 case '\\': {
12454 // If we hit an escape, then we need to skip past
12455 // however many characters the escape takes up. However
12456 // it's important that if \n or \r\n are escaped, we
12457 // stop looping before the newline and not after the
12458 // newline so that we can still potentially find the
12459 // terminator of the heredoc.
12460 parser->current.end = breakpoint + 1;
12461
12462 // If we've hit the end of the file, then break out of
12463 // the loop by setting the breakpoint to NULL.
12464 if (parser->current.end == parser->end) {
12465 breakpoint = NULL;
12466 continue;
12467 }
12468
12469 pm_token_buffer_escape(parser, &token_buffer);
12470 uint8_t peeked = peek(parser);
12471
12472 if (quote == PM_HEREDOC_QUOTE_SINGLE) {
12473 switch (peeked) {
12474 case '\r':
12475 parser->current.end++;
12476 if (peek(parser) != '\n') {
12477 pm_token_buffer_push_byte(&token_buffer, '\\');
12478 pm_token_buffer_push_byte(&token_buffer, '\r');
12479 break;
12480 }
12482 case '\n':
12483 pm_token_buffer_push_byte(&token_buffer, '\\');
12484 pm_token_buffer_push_byte(&token_buffer, '\n');
12485 token_buffer.cursor = parser->current.end + 1;
12486 breakpoint = parser->current.end;
12487 continue;
12488 default:
12489 pm_token_buffer_push_byte(&token_buffer, '\\');
12490 pm_token_buffer_push_escaped(&token_buffer, parser);
12491 break;
12492 }
12493 } else {
12494 switch (peeked) {
12495 case '\r':
12496 parser->current.end++;
12497 if (peek(parser) != '\n') {
12498 pm_token_buffer_push_byte(&token_buffer, '\r');
12499 break;
12500 }
12502 case '\n':
12503 // If we are in a tilde here, we should
12504 // break out of the loop and return the
12505 // string content.
12506 if (heredoc_lex_mode->indent == PM_HEREDOC_INDENT_TILDE) {
12507 const uint8_t *end = parser->current.end;
12508
12509 if (parser->heredoc_end == NULL) {
12510 pm_line_offset_list_append(&parser->metadata_arena, &parser->line_offsets, U32(end - parser->start + 1));
12511 }
12512
12513 // Here we want the buffer to only
12514 // include up to the backslash.
12515 parser->current.end = breakpoint;
12516 pm_token_buffer_flush(parser, &token_buffer);
12517
12518 // Now we can advance the end of the
12519 // token past the newline.
12520 parser->current.end = end + 1;
12521 lex_mode->as.heredoc.line_continuation = true;
12522 LEX(PM_TOKEN_STRING_CONTENT);
12523 }
12524
12525 was_line_continuation = true;
12526 token_buffer.cursor = parser->current.end + 1;
12527 breakpoint = parser->current.end;
12528 continue;
12529 default:
12530 escape_read(parser, &token_buffer.buffer, NULL, PM_ESCAPE_FLAG_NONE);
12531 break;
12532 }
12533 }
12534
12535 token_buffer.cursor = parser->current.end;
12536 breakpoint = pm_strpbrk(parser, parser->current.end, breakpoints, parser->end - parser->current.end, true);
12537 break;
12538 }
12539 case '#': {
12540 pm_token_type_t type = lex_interpolation(parser, breakpoint);
12541
12542 if (!type) {
12543 // If we haven't returned at this point then we had
12544 // something that looked like an interpolated class
12545 // or instance variable like "#@" but wasn't
12546 // actually. In this case we'll just skip to the
12547 // next breakpoint.
12548 breakpoint = pm_strpbrk(parser, parser->current.end, breakpoints, parser->end - parser->current.end, true);
12549 break;
12550 }
12551
12552 if (type == PM_TOKEN_STRING_CONTENT) {
12553 pm_token_buffer_flush(parser, &token_buffer);
12554 }
12555
12556 LEX(type);
12557 }
12558 default:
12559 assert(false && "unreachable");
12560 }
12561
12562 was_line_continuation = false;
12563 }
12564
12565 if (parser->current.end > parser->current.start) {
12566 parser->current.end = parser->end;
12567 pm_token_buffer_flush(parser, &token_buffer);
12568 LEX(PM_TOKEN_STRING_CONTENT);
12569 }
12570
12571 // If we've hit the end of the string, then this is an unterminated
12572 // heredoc. In that case we'll return a string content token.
12573 parser->current.end = parser->end;
12574 pm_token_buffer_flush(parser, &token_buffer);
12575 LEX(PM_TOKEN_STRING_CONTENT);
12576 }
12577 }
12578
12579 assert(false && "unreachable");
12580}
12581
12582#undef LEX
12583
12584/******************************************************************************/
12585/* Parse functions */
12586/******************************************************************************/
12587
12596typedef enum {
12597 PM_BINDING_POWER_UNSET = 0, // used to indicate this token cannot be used as an infix operator
12598 PM_BINDING_POWER_STATEMENT = 2,
12599 PM_BINDING_POWER_MODIFIER_RESCUE = 4, // rescue
12600 PM_BINDING_POWER_MODIFIER = 6, // if unless until while
12601 PM_BINDING_POWER_COMPOSITION = 8, // and or
12602 PM_BINDING_POWER_NOT = 10, // not
12603 PM_BINDING_POWER_MATCH = 12, // => in
12604 PM_BINDING_POWER_DEFINED = 14, // defined?
12605 PM_BINDING_POWER_MULTI_ASSIGNMENT = 16, // =
12606 PM_BINDING_POWER_ASSIGNMENT = 18, // = += -= *= /= %= &= |= ^= &&= ||= <<= >>= **=
12607 PM_BINDING_POWER_TERNARY = 20, // ?:
12608 PM_BINDING_POWER_RANGE = 22, // .. ...
12609 PM_BINDING_POWER_LOGICAL_OR = 24, // ||
12610 PM_BINDING_POWER_LOGICAL_AND = 26, // &&
12611 PM_BINDING_POWER_EQUALITY = 28, // <=> == === != =~ !~
12612 PM_BINDING_POWER_COMPARISON = 30, // > >= < <=
12613 PM_BINDING_POWER_BITWISE_OR = 32, // | ^
12614 PM_BINDING_POWER_BITWISE_AND = 34, // &
12615 PM_BINDING_POWER_SHIFT = 36, // << >>
12616 PM_BINDING_POWER_TERM = 38, // + -
12617 PM_BINDING_POWER_FACTOR = 40, // * / %
12618 PM_BINDING_POWER_UMINUS = 42, // -@
12619 PM_BINDING_POWER_EXPONENT = 44, // **
12620 PM_BINDING_POWER_UNARY = 46, // ! ~ +@
12621 PM_BINDING_POWER_INDEX = 48, // [] []=
12622 PM_BINDING_POWER_CALL = 50, // :: .
12623 PM_BINDING_POWER_MAX = 52
12624} pm_binding_power_t;
12625
12630typedef struct {
12632 pm_binding_power_t left;
12633
12635 pm_binding_power_t right;
12636
12639
12646
12647#define BINDING_POWER_ASSIGNMENT { PM_BINDING_POWER_UNARY, PM_BINDING_POWER_ASSIGNMENT, true, false }
12648#define LEFT_ASSOCIATIVE(precedence) { precedence, precedence + 1, true, false }
12649#define RIGHT_ASSOCIATIVE(precedence) { precedence, precedence, true, false }
12650#define NON_ASSOCIATIVE(precedence) { precedence, precedence + 1, true, true }
12651#define RIGHT_ASSOCIATIVE_UNARY(precedence) { precedence, precedence, false, false }
12652
12653pm_binding_powers_t pm_binding_powers[PM_TOKEN_MAXIMUM] = {
12654 // rescue
12655 [PM_TOKEN_KEYWORD_RESCUE_MODIFIER] = { PM_BINDING_POWER_MODIFIER_RESCUE, PM_BINDING_POWER_COMPOSITION, true, false },
12656
12657 // if unless until while
12658 [PM_TOKEN_KEYWORD_IF_MODIFIER] = LEFT_ASSOCIATIVE(PM_BINDING_POWER_MODIFIER),
12659 [PM_TOKEN_KEYWORD_UNLESS_MODIFIER] = LEFT_ASSOCIATIVE(PM_BINDING_POWER_MODIFIER),
12660 [PM_TOKEN_KEYWORD_UNTIL_MODIFIER] = LEFT_ASSOCIATIVE(PM_BINDING_POWER_MODIFIER),
12661 [PM_TOKEN_KEYWORD_WHILE_MODIFIER] = LEFT_ASSOCIATIVE(PM_BINDING_POWER_MODIFIER),
12662
12663 // and or
12664 [PM_TOKEN_KEYWORD_AND] = LEFT_ASSOCIATIVE(PM_BINDING_POWER_COMPOSITION),
12665 [PM_TOKEN_KEYWORD_OR] = LEFT_ASSOCIATIVE(PM_BINDING_POWER_COMPOSITION),
12666
12667 // => in
12668 [PM_TOKEN_EQUAL_GREATER] = NON_ASSOCIATIVE(PM_BINDING_POWER_MATCH),
12669 [PM_TOKEN_KEYWORD_IN] = NON_ASSOCIATIVE(PM_BINDING_POWER_MATCH),
12670
12671 // &&= &= ^= = >>= <<= -= %= |= ||= += /= *= **=
12672 [PM_TOKEN_AMPERSAND_AMPERSAND_EQUAL] = BINDING_POWER_ASSIGNMENT,
12673 [PM_TOKEN_AMPERSAND_EQUAL] = BINDING_POWER_ASSIGNMENT,
12674 [PM_TOKEN_CARET_EQUAL] = BINDING_POWER_ASSIGNMENT,
12675 [PM_TOKEN_EQUAL] = BINDING_POWER_ASSIGNMENT,
12676 [PM_TOKEN_GREATER_GREATER_EQUAL] = BINDING_POWER_ASSIGNMENT,
12677 [PM_TOKEN_LESS_LESS_EQUAL] = BINDING_POWER_ASSIGNMENT,
12678 [PM_TOKEN_MINUS_EQUAL] = BINDING_POWER_ASSIGNMENT,
12679 [PM_TOKEN_PERCENT_EQUAL] = BINDING_POWER_ASSIGNMENT,
12680 [PM_TOKEN_PIPE_EQUAL] = BINDING_POWER_ASSIGNMENT,
12681 [PM_TOKEN_PIPE_PIPE_EQUAL] = BINDING_POWER_ASSIGNMENT,
12682 [PM_TOKEN_PLUS_EQUAL] = BINDING_POWER_ASSIGNMENT,
12683 [PM_TOKEN_SLASH_EQUAL] = BINDING_POWER_ASSIGNMENT,
12684 [PM_TOKEN_STAR_EQUAL] = BINDING_POWER_ASSIGNMENT,
12685 [PM_TOKEN_STAR_STAR_EQUAL] = BINDING_POWER_ASSIGNMENT,
12686
12687 // ?:
12688 [PM_TOKEN_QUESTION_MARK] = RIGHT_ASSOCIATIVE(PM_BINDING_POWER_TERNARY),
12689
12690 // .. ...
12691 [PM_TOKEN_DOT_DOT] = NON_ASSOCIATIVE(PM_BINDING_POWER_RANGE),
12692 [PM_TOKEN_DOT_DOT_DOT] = NON_ASSOCIATIVE(PM_BINDING_POWER_RANGE),
12693 [PM_TOKEN_UDOT_DOT] = RIGHT_ASSOCIATIVE_UNARY(PM_BINDING_POWER_LOGICAL_OR),
12694 [PM_TOKEN_UDOT_DOT_DOT] = RIGHT_ASSOCIATIVE_UNARY(PM_BINDING_POWER_LOGICAL_OR),
12695
12696 // ||
12697 [PM_TOKEN_PIPE_PIPE] = LEFT_ASSOCIATIVE(PM_BINDING_POWER_LOGICAL_OR),
12698
12699 // &&
12700 [PM_TOKEN_AMPERSAND_AMPERSAND] = LEFT_ASSOCIATIVE(PM_BINDING_POWER_LOGICAL_AND),
12701
12702 // != !~ == === =~ <=>
12703 [PM_TOKEN_BANG_EQUAL] = NON_ASSOCIATIVE(PM_BINDING_POWER_EQUALITY),
12704 [PM_TOKEN_BANG_TILDE] = NON_ASSOCIATIVE(PM_BINDING_POWER_EQUALITY),
12705 [PM_TOKEN_EQUAL_EQUAL] = NON_ASSOCIATIVE(PM_BINDING_POWER_EQUALITY),
12706 [PM_TOKEN_EQUAL_EQUAL_EQUAL] = NON_ASSOCIATIVE(PM_BINDING_POWER_EQUALITY),
12707 [PM_TOKEN_EQUAL_TILDE] = NON_ASSOCIATIVE(PM_BINDING_POWER_EQUALITY),
12708 [PM_TOKEN_LESS_EQUAL_GREATER] = NON_ASSOCIATIVE(PM_BINDING_POWER_EQUALITY),
12709
12710 // > >= < <=
12711 [PM_TOKEN_GREATER] = LEFT_ASSOCIATIVE(PM_BINDING_POWER_COMPARISON),
12712 [PM_TOKEN_GREATER_EQUAL] = LEFT_ASSOCIATIVE(PM_BINDING_POWER_COMPARISON),
12713 [PM_TOKEN_LESS] = LEFT_ASSOCIATIVE(PM_BINDING_POWER_COMPARISON),
12714 [PM_TOKEN_LESS_EQUAL] = LEFT_ASSOCIATIVE(PM_BINDING_POWER_COMPARISON),
12715
12716 // ^ |
12717 [PM_TOKEN_CARET] = LEFT_ASSOCIATIVE(PM_BINDING_POWER_BITWISE_OR),
12718 [PM_TOKEN_PIPE] = LEFT_ASSOCIATIVE(PM_BINDING_POWER_BITWISE_OR),
12719
12720 // &
12721 [PM_TOKEN_AMPERSAND] = LEFT_ASSOCIATIVE(PM_BINDING_POWER_BITWISE_AND),
12722
12723 // >> <<
12724 [PM_TOKEN_GREATER_GREATER] = LEFT_ASSOCIATIVE(PM_BINDING_POWER_SHIFT),
12725 [PM_TOKEN_LESS_LESS] = LEFT_ASSOCIATIVE(PM_BINDING_POWER_SHIFT),
12726
12727 // - +
12728 [PM_TOKEN_MINUS] = LEFT_ASSOCIATIVE(PM_BINDING_POWER_TERM),
12729 [PM_TOKEN_PLUS] = LEFT_ASSOCIATIVE(PM_BINDING_POWER_TERM),
12730
12731 // % / *
12732 [PM_TOKEN_PERCENT] = LEFT_ASSOCIATIVE(PM_BINDING_POWER_FACTOR),
12733 [PM_TOKEN_SLASH] = LEFT_ASSOCIATIVE(PM_BINDING_POWER_FACTOR),
12734 [PM_TOKEN_STAR] = LEFT_ASSOCIATIVE(PM_BINDING_POWER_FACTOR),
12735 [PM_TOKEN_USTAR] = RIGHT_ASSOCIATIVE_UNARY(PM_BINDING_POWER_FACTOR),
12736
12737 // -@
12738 [PM_TOKEN_UMINUS] = RIGHT_ASSOCIATIVE_UNARY(PM_BINDING_POWER_UMINUS),
12739 [PM_TOKEN_UMINUS_NUM] = { PM_BINDING_POWER_UMINUS, PM_BINDING_POWER_MAX, false, false },
12740
12741 // **
12742 [PM_TOKEN_STAR_STAR] = RIGHT_ASSOCIATIVE(PM_BINDING_POWER_EXPONENT),
12743 [PM_TOKEN_USTAR_STAR] = RIGHT_ASSOCIATIVE_UNARY(PM_BINDING_POWER_UNARY),
12744
12745 // ! ~ +@
12746 [PM_TOKEN_BANG] = RIGHT_ASSOCIATIVE_UNARY(PM_BINDING_POWER_UNARY),
12747 [PM_TOKEN_TILDE] = RIGHT_ASSOCIATIVE_UNARY(PM_BINDING_POWER_UNARY),
12748 [PM_TOKEN_UPLUS] = RIGHT_ASSOCIATIVE_UNARY(PM_BINDING_POWER_UNARY),
12749
12750 // [
12751 [PM_TOKEN_BRACKET_LEFT] = LEFT_ASSOCIATIVE(PM_BINDING_POWER_INDEX),
12752
12753 // :: . &.
12754 [PM_TOKEN_COLON_COLON] = RIGHT_ASSOCIATIVE(PM_BINDING_POWER_CALL),
12755 [PM_TOKEN_DOT] = RIGHT_ASSOCIATIVE(PM_BINDING_POWER_CALL),
12756 [PM_TOKEN_AMPERSAND_DOT] = RIGHT_ASSOCIATIVE(PM_BINDING_POWER_CALL)
12757};
12758
12759#undef BINDING_POWER_ASSIGNMENT
12760#undef LEFT_ASSOCIATIVE
12761#undef RIGHT_ASSOCIATIVE
12762#undef RIGHT_ASSOCIATIVE_UNARY
12763
12767static PRISM_INLINE bool
12768match1(const pm_parser_t *parser, pm_token_type_t type) {
12769 return parser->current.type == type;
12770}
12771
12775static PRISM_INLINE bool
12776match2(const pm_parser_t *parser, pm_token_type_t type1, pm_token_type_t type2) {
12777 return match1(parser, type1) || match1(parser, type2);
12778}
12779
12783static PRISM_INLINE bool
12784match3(const pm_parser_t *parser, pm_token_type_t type1, pm_token_type_t type2, pm_token_type_t type3) {
12785 return match1(parser, type1) || match1(parser, type2) || match1(parser, type3);
12786}
12787
12791static PRISM_INLINE bool
12792match4(const pm_parser_t *parser, pm_token_type_t type1, pm_token_type_t type2, pm_token_type_t type3, pm_token_type_t type4) {
12793 return match1(parser, type1) || match1(parser, type2) || match1(parser, type3) || match1(parser, type4);
12794}
12795
12799static PRISM_INLINE bool
12800match5(const pm_parser_t *parser, pm_token_type_t type1, pm_token_type_t type2, pm_token_type_t type3, pm_token_type_t type4, pm_token_type_t type5) {
12801 return match1(parser, type1) || match1(parser, type2) || match1(parser, type3) || match1(parser, type4) || match1(parser, type5);
12802}
12803
12807static PRISM_INLINE bool
12808match6(const pm_parser_t *parser, pm_token_type_t type1, pm_token_type_t type2, pm_token_type_t type3, pm_token_type_t type4, pm_token_type_t type5, pm_token_type_t type6) {
12809 return match1(parser, type1) || match1(parser, type2) || match1(parser, type3) || match1(parser, type4) || match1(parser, type5) || match1(parser, type6);
12810}
12811
12815static PRISM_INLINE bool
12816match8(const pm_parser_t *parser, pm_token_type_t type1, pm_token_type_t type2, pm_token_type_t type3, pm_token_type_t type4, pm_token_type_t type5, pm_token_type_t type6, pm_token_type_t type7, pm_token_type_t type8) {
12817 return match1(parser, type1) || match1(parser, type2) || match1(parser, type3) || match1(parser, type4) || match1(parser, type5) || match1(parser, type6) || match1(parser, type7) || match1(parser, type8);
12818}
12819
12826static bool
12827accept1(pm_parser_t *parser, pm_token_type_t type) {
12828 if (match1(parser, type)) {
12829 parser_lex(parser);
12830 return true;
12831 }
12832 return false;
12833}
12834
12839static PRISM_INLINE bool
12840accept2(pm_parser_t *parser, pm_token_type_t type1, pm_token_type_t type2) {
12841 if (match2(parser, type1, type2)) {
12842 parser_lex(parser);
12843 return true;
12844 }
12845 return false;
12846}
12847
12859static void
12860expect1(pm_parser_t *parser, pm_token_type_t type, pm_diagnostic_id_t diag_id) {
12861 if (accept1(parser, type)) return;
12862
12863 const uint8_t *location = parser->previous.end;
12864 pm_parser_err(parser, U32(location - parser->start), 0, diag_id);
12865
12866 parser->previous.start = location;
12867 parser->previous.type = 0;
12868}
12869
12874static void
12875expect2(pm_parser_t *parser, pm_token_type_t type1, pm_token_type_t type2, pm_diagnostic_id_t diag_id) {
12876 if (accept2(parser, type1, type2)) return;
12877
12878 const uint8_t *location = parser->previous.end;
12879 pm_parser_err(parser, U32(location - parser->start), 0, diag_id);
12880
12881 parser->previous.start = location;
12882 parser->previous.type = 0;
12883}
12884
12889static void
12890expect1_heredoc_term(pm_parser_t *parser, const uint8_t *ident_start, size_t ident_length) {
12891 if (match1(parser, PM_TOKEN_HEREDOC_END)) {
12892 parser_lex(parser);
12893 } else {
12894 pm_parser_err_heredoc_term(parser, ident_start, ident_length);
12895 parser->previous.start = parser->previous.end;
12896 parser->previous.type = 0;
12897 }
12898}
12899
12906static void
12907expect1_opening(pm_parser_t *parser, pm_token_type_t type, pm_diagnostic_id_t diag_id, const pm_token_t *opening) {
12908 if (accept1(parser, type)) return;
12909
12910 const uint8_t *start = opening->start;
12911 pm_parser_err(parser, U32(start - parser->start), U32(opening->end - start), diag_id);
12912
12913 parser->previous.start = parser->previous.end;
12914 parser->previous.type = 0;
12915}
12916
12918#define PM_PARSE_ACCEPTS_COMMAND_CALL ((uint8_t) 0x1)
12919#define PM_PARSE_ACCEPTS_LABEL ((uint8_t) 0x2)
12920#define PM_PARSE_ACCEPTS_DO_BLOCK ((uint8_t) 0x4)
12921#define PM_PARSE_IN_ENDLESS_DEF ((uint8_t) 0x8)
12922
12932#define PM_PARSE_ACCEPTS_STATEMENT ((uint8_t) 0x10)
12933
12934static pm_node_t *
12935parse_expression(pm_parser_t *parser, pm_binding_power_t binding_power, uint8_t flags, pm_diagnostic_id_t diag_id, uint16_t depth);
12936
12941static pm_node_t *
12942parse_value_expression(pm_parser_t *parser, pm_binding_power_t binding_power, uint8_t flags, pm_diagnostic_id_t diag_id, uint16_t depth) {
12943 pm_node_t *node = parse_expression(parser, binding_power, flags, diag_id, depth);
12944 pm_assert_value_expression(parser, node);
12945 return node;
12946}
12947
12966static PRISM_INLINE bool
12967token_begins_expression_p(pm_token_type_t type) {
12968 switch (type) {
12969 case PM_TOKEN_EQUAL_GREATER:
12970 case PM_TOKEN_KEYWORD_IN:
12971 // We need to special case this because it is a binary operator that
12972 // should not be marked as beginning an expression.
12973 return false;
12974 case PM_TOKEN_BRACE_RIGHT:
12975 case PM_TOKEN_BRACKET_RIGHT:
12976 case PM_TOKEN_COLON:
12977 case PM_TOKEN_COMMA:
12978 case PM_TOKEN_EMBEXPR_END:
12979 case PM_TOKEN_EOF:
12980 case PM_TOKEN_LAMBDA_BEGIN:
12981 case PM_TOKEN_KEYWORD_DO:
12982 case PM_TOKEN_KEYWORD_DO_BLOCK:
12983 case PM_TOKEN_KEYWORD_DO_LAMBDA:
12984 case PM_TOKEN_KEYWORD_DO_LOOP:
12985 case PM_TOKEN_KEYWORD_END:
12986 case PM_TOKEN_KEYWORD_ELSE:
12987 case PM_TOKEN_KEYWORD_ELSIF:
12988 case PM_TOKEN_KEYWORD_ENSURE:
12989 case PM_TOKEN_KEYWORD_THEN:
12990 case PM_TOKEN_KEYWORD_RESCUE:
12991 case PM_TOKEN_KEYWORD_WHEN:
12992 case PM_TOKEN_NEWLINE:
12993 case PM_TOKEN_PARENTHESIS_RIGHT:
12994 case PM_TOKEN_SEMICOLON:
12995 // The reason we need this short-circuit is because we're using the
12996 // binding powers table to tell us if the subsequent token could
12997 // potentially be the start of an expression. If there _is_ a binding
12998 // power for one of these tokens, then we should remove it from this list
12999 // and let it be handled by the default case below.
13000 assert(pm_binding_powers[type].left == PM_BINDING_POWER_UNSET);
13001 return false;
13002 case PM_TOKEN_UAMPERSAND:
13003 // This is a special case because this unary operator cannot appear
13004 // as a general operator, it only appears in certain circumstances.
13005 return false;
13006 case PM_TOKEN_UCOLON_COLON:
13007 case PM_TOKEN_UMINUS:
13008 case PM_TOKEN_UMINUS_NUM:
13009 case PM_TOKEN_UPLUS:
13010 case PM_TOKEN_BANG:
13011 case PM_TOKEN_TILDE:
13012 case PM_TOKEN_UDOT_DOT:
13013 case PM_TOKEN_UDOT_DOT_DOT:
13014 // These unary tokens actually do have binding power associated with them
13015 // so that we can correctly place them into the precedence order. But we
13016 // want them to be marked as beginning an expression, so we need to
13017 // special case them here.
13018 return true;
13019 default:
13020 return pm_binding_powers[type].left == PM_BINDING_POWER_UNSET;
13021 }
13022}
13023
13039static PRISM_INLINE bool
13040token_begins_pattern_p(pm_token_type_t type) {
13041 return (
13042 token_begins_expression_p(type) ||
13043 type == PM_TOKEN_USTAR ||
13044 type == PM_TOKEN_USTAR_STAR ||
13045 type == PM_TOKEN_CARET
13046 );
13047}
13048
13053static pm_node_t *
13054parse_starred_expression(pm_parser_t *parser, pm_binding_power_t binding_power, uint8_t flags, pm_diagnostic_id_t diag_id, uint16_t depth) {
13055 if (accept1(parser, PM_TOKEN_USTAR)) {
13056 pm_token_t operator = parser->previous;
13057 pm_node_t *expression = parse_value_expression(parser, binding_power, (uint8_t) (flags & PM_PARSE_ACCEPTS_DO_BLOCK), PM_ERR_EXPECT_EXPRESSION_AFTER_STAR, (uint16_t) (depth + 1));
13058 return UP(pm_splat_node_create(parser, &operator, expression));
13059 }
13060
13061 return parse_value_expression(parser, binding_power, flags, diag_id, depth);
13062}
13063
13064static bool
13065pm_node_unreference_each(const pm_node_t *node, void *data) {
13066 switch (PM_NODE_TYPE(node)) {
13067 /* When we are about to destroy a set of nodes that could potentially
13068 * contain block exits for the current scope, we need to check if they
13069 * are contained in the list of block exits and remove them if they are.
13070 */
13071 case PM_BREAK_NODE:
13072 case PM_NEXT_NODE:
13073 case PM_REDO_NODE: {
13074 pm_parser_t *parser = (pm_parser_t *) data;
13075 size_t index = 0;
13076
13077 while (index < parser->current_block_exits->size) {
13078 pm_node_t *block_exit = parser->current_block_exits->nodes[index];
13079
13080 if (block_exit == node) {
13081 if (index + 1 < parser->current_block_exits->size) {
13082 memmove(
13083 &parser->current_block_exits->nodes[index],
13084 &parser->current_block_exits->nodes[index + 1],
13085 (parser->current_block_exits->size - index - 1) * sizeof(pm_node_t *)
13086 );
13087 }
13088 parser->current_block_exits->size--;
13089
13090 /* Note returning true here because these nodes could have
13091 * arguments that are themselves block exits. */
13092 return true;
13093 }
13094
13095 index++;
13096 }
13097
13098 return true;
13099 }
13100 /* When an implicit local variable is written to or targeted, it becomes
13101 * a regular, named local variable. This branch removes it from the list
13102 * of implicit parameters when that happens. */
13103 case PM_LOCAL_VARIABLE_READ_NODE:
13104 case PM_IT_LOCAL_VARIABLE_READ_NODE: {
13105 pm_parser_t *parser = (pm_parser_t *) data;
13106 pm_node_list_t *implicit_parameters = &parser->current_scope->implicit_parameters;
13107
13108 for (size_t index = 0; index < implicit_parameters->size; index++) {
13109 if (implicit_parameters->nodes[index] == node) {
13110 /* If the node is not the last one in the list, we need to
13111 * shift the remaining nodes down to fill the gap. This is
13112 * extremely unlikely to happen. */
13113 if (index != implicit_parameters->size - 1) {
13114 memmove(&implicit_parameters->nodes[index], &implicit_parameters->nodes[index + 1], (implicit_parameters->size - index - 1) * sizeof(pm_node_t *));
13115 }
13116
13117 implicit_parameters->size--;
13118 break;
13119 }
13120 }
13121
13122 return false;
13123 }
13124 default:
13125 return true;
13126 }
13127}
13128
13134static void
13135pm_node_unreference(pm_parser_t *parser, const pm_node_t *node) {
13136 pm_visit_node(node, pm_node_unreference_each, parser);
13137}
13138
13143static void
13144parse_write_name(pm_parser_t *parser, pm_constant_id_t *name_field) {
13145 // The method name needs to change. If we previously had
13146 // foo, we now need foo=. In this case we'll allocate a new
13147 // owned string, copy the previous method name in, and
13148 // append an =.
13149 pm_constant_t *constant = pm_constant_pool_id_to_constant(&parser->constant_pool, *name_field);
13150 size_t length = constant->length;
13151 uint8_t *name = (uint8_t *) pm_arena_alloc(parser->arena, length + 1, 1);
13152
13153 memcpy(name, constant->start, length);
13154 name[length] = '=';
13155
13156 *name_field = pm_constant_pool_insert_owned(&parser->metadata_arena, &parser->constant_pool, name, length + 1);
13157}
13158
13165static pm_node_t *
13166parse_unwriteable_target(pm_parser_t *parser, pm_node_t *target) {
13167 switch (PM_NODE_TYPE(target)) {
13168 case PM_SOURCE_ENCODING_NODE: pm_parser_err_node(parser, target, PM_ERR_EXPRESSION_NOT_WRITABLE_ENCODING); break;
13169 case PM_FALSE_NODE: pm_parser_err_node(parser, target, PM_ERR_EXPRESSION_NOT_WRITABLE_FALSE); break;
13170 case PM_SOURCE_FILE_NODE: pm_parser_err_node(parser, target, PM_ERR_EXPRESSION_NOT_WRITABLE_FILE); break;
13171 case PM_SOURCE_LINE_NODE: pm_parser_err_node(parser, target, PM_ERR_EXPRESSION_NOT_WRITABLE_LINE); break;
13172 case PM_NIL_NODE: pm_parser_err_node(parser, target, PM_ERR_EXPRESSION_NOT_WRITABLE_NIL); break;
13173 case PM_SELF_NODE: pm_parser_err_node(parser, target, PM_ERR_EXPRESSION_NOT_WRITABLE_SELF); break;
13174 case PM_TRUE_NODE: pm_parser_err_node(parser, target, PM_ERR_EXPRESSION_NOT_WRITABLE_TRUE); break;
13175 default: break;
13176 }
13177
13178 pm_constant_id_t name = pm_parser_constant_id_raw(parser, parser->start + PM_NODE_START(target), parser->start + PM_NODE_END(target));
13179 pm_local_variable_target_node_t *result = pm_local_variable_target_node_create(parser, &target->location, name, 0);
13180
13181 return UP(result);
13182}
13183
13192static pm_node_t *
13193parse_target(pm_parser_t *parser, pm_node_t *target, bool multiple, bool splat_parent) {
13194 switch (PM_NODE_TYPE(target)) {
13195 case PM_ERROR_RECOVERY_NODE:
13196 return target;
13197 case PM_SOURCE_ENCODING_NODE:
13198 case PM_FALSE_NODE:
13199 case PM_SOURCE_FILE_NODE:
13200 case PM_SOURCE_LINE_NODE:
13201 case PM_NIL_NODE:
13202 case PM_SELF_NODE:
13203 case PM_TRUE_NODE: {
13204 // In these special cases, we have specific error messages and we
13205 // will replace them with local variable writes.
13206 return parse_unwriteable_target(parser, target);
13207 }
13208 case PM_CLASS_VARIABLE_READ_NODE:
13210 target->type = PM_CLASS_VARIABLE_TARGET_NODE;
13211 return target;
13212 case PM_CONSTANT_PATH_NODE:
13213 if (context_def_p(parser)) {
13214 pm_parser_err_node(parser, target, PM_ERR_WRITE_TARGET_IN_METHOD);
13215 }
13216
13218 target->type = PM_CONSTANT_PATH_TARGET_NODE;
13219
13220 return target;
13221 case PM_CONSTANT_READ_NODE:
13222 if (context_def_p(parser)) {
13223 pm_parser_err_node(parser, target, PM_ERR_WRITE_TARGET_IN_METHOD);
13224 }
13225
13226 assert(sizeof(pm_constant_target_node_t) == sizeof(pm_constant_read_node_t));
13227 target->type = PM_CONSTANT_TARGET_NODE;
13228
13229 return target;
13230 case PM_BACK_REFERENCE_READ_NODE:
13231 case PM_NUMBERED_REFERENCE_READ_NODE:
13232 PM_PARSER_ERR_NODE_FORMAT_CONTENT(parser, target, PM_ERR_WRITE_TARGET_READONLY);
13233 return UP(pm_error_recovery_node_create_unexpected(parser, target));
13234 case PM_GLOBAL_VARIABLE_READ_NODE:
13236 target->type = PM_GLOBAL_VARIABLE_TARGET_NODE;
13237 return target;
13238 case PM_LOCAL_VARIABLE_READ_NODE: {
13239 if (pm_token_is_numbered_parameter(parser, PM_NODE_START(target), PM_NODE_LENGTH(target))) {
13240 PM_PARSER_ERR_FORMAT(parser, PM_NODE_START(target), PM_NODE_LENGTH(target), PM_ERR_PARAMETER_NUMBERED_RESERVED, parser->start + PM_NODE_START(target));
13241 pm_node_unreference(parser, target);
13242 }
13243
13244 const pm_local_variable_read_node_t *cast = (const pm_local_variable_read_node_t *) target;
13245 uint32_t name = cast->name;
13246 uint32_t depth = cast->depth;
13247 pm_locals_unread(&pm_parser_scope_find(parser, depth)->locals, name);
13248
13250 target->type = PM_LOCAL_VARIABLE_TARGET_NODE;
13251
13252 return target;
13253 }
13254 case PM_IT_LOCAL_VARIABLE_READ_NODE: {
13255 pm_constant_id_t name = pm_parser_local_add_constant(parser, "it", 2);
13256 pm_node_t *node = UP(pm_local_variable_target_node_create(parser, &target->location, name, 0));
13257
13258 pm_node_unreference(parser, target);
13259
13260 return node;
13261 }
13262 case PM_INSTANCE_VARIABLE_READ_NODE:
13264 target->type = PM_INSTANCE_VARIABLE_TARGET_NODE;
13265 return target;
13266 case PM_MULTI_TARGET_NODE:
13267 if (splat_parent) {
13268 // Multi target is not accepted in all positions. If this is one
13269 // of them, then we need to add an error.
13270 pm_parser_err_node(parser, target, PM_ERR_WRITE_TARGET_UNEXPECTED);
13271 }
13272
13273 return target;
13274 case PM_SPLAT_NODE: {
13275 pm_splat_node_t *splat = (pm_splat_node_t *) target;
13276
13277 if (splat->expression != NULL) {
13278 splat->expression = parse_target(parser, splat->expression, multiple, true);
13279 }
13280
13281 return UP(splat);
13282 }
13283 case PM_CALL_NODE: {
13284 pm_call_node_t *call = (pm_call_node_t *) target;
13285
13286 // If we have no arguments to the call node and we need this to be a
13287 // target then this is either a method call or a local variable
13288 // write.
13289 if (
13290 (call->message_loc.length > 0) &&
13291 (parser->start[call->message_loc.start + call->message_loc.length - 1] != '!') &&
13292 (parser->start[call->message_loc.start + call->message_loc.length - 1] != '?') &&
13293 (call->opening_loc.length == 0) &&
13294 (call->arguments == NULL) &&
13295 (call->block == NULL)
13296 ) {
13297 if (call->receiver == NULL) {
13298 // When we get here, we have a local variable write, because it
13299 // was previously marked as a method call but now we have an =.
13300 // This looks like:
13301 //
13302 // foo = 1
13303 //
13304 // When it was parsed in the prefix position, foo was seen as a
13305 // method call with no receiver and no arguments. Now we have an
13306 // =, so we know it's a local variable write.
13307 pm_location_t message_loc = call->message_loc;
13308 pm_constant_id_t name = pm_parser_local_add_location(parser, &message_loc, 0);
13309
13310 return UP(pm_local_variable_target_node_create(parser, &message_loc, name, 0));
13311 }
13312
13313 if (peek_at(parser, parser->start + call->message_loc.start) == '_' || parser->encoding->alnum_char(parser->start + call->message_loc.start, (ptrdiff_t) call->message_loc.length)) {
13314 if (multiple && PM_NODE_FLAG_P(call, PM_CALL_NODE_FLAGS_SAFE_NAVIGATION)) {
13315 pm_parser_err_node(parser, (const pm_node_t *) call, PM_ERR_UNEXPECTED_SAFE_NAVIGATION);
13316 }
13317
13318 parse_write_name(parser, &call->name);
13319 return UP(pm_call_target_node_create(parser, call));
13320 }
13321 }
13322
13323 // If there is no call operator and the message is "[]" then this is
13324 // an aref expression, and we can transform it into an aset
13325 // expression.
13326 if (PM_NODE_FLAG_P(call, PM_CALL_NODE_FLAGS_INDEX)) {
13327 return UP(pm_index_target_node_create(parser, call));
13328 }
13329 }
13331 default:
13332 // In this case we have a node that we don't know how to convert
13333 // into a target. We need to treat it as an error. For now, we'll
13334 // mark it as an error and just skip right past it.
13335 pm_parser_err_node(parser, target, PM_ERR_WRITE_TARGET_UNEXPECTED);
13336 return target;
13337 }
13338}
13339
13344static pm_node_t *
13345parse_target_validate(pm_parser_t *parser, pm_node_t *target, bool multiple) {
13346 pm_node_t *result = parse_target(parser, target, multiple, false);
13347
13348 // Ensure that we have one of an =, an 'in' in for indexes, and a ')' in
13349 // parens after the targets.
13350 if (
13351 !match1(parser, PM_TOKEN_EQUAL) &&
13352 !(context_p(parser, PM_CONTEXT_FOR_INDEX) && match1(parser, PM_TOKEN_KEYWORD_IN)) &&
13353 !(context_p(parser, PM_CONTEXT_PARENS) && match1(parser, PM_TOKEN_PARENTHESIS_RIGHT))
13354 ) {
13355 pm_parser_err_node(parser, result, PM_ERR_WRITE_TARGET_UNEXPECTED);
13356 }
13357
13358 return result;
13359}
13360
13365static pm_node_t *
13366parse_shareable_constant_write(pm_parser_t *parser, pm_node_t *write) {
13367 pm_shareable_constant_value_t shareable_constant = pm_parser_scope_shareable_constant_get(parser);
13368
13369 if (shareable_constant != PM_SCOPE_SHAREABLE_CONSTANT_NONE) {
13370 return UP(pm_shareable_constant_node_create(parser, write, shareable_constant));
13371 }
13372
13373 return write;
13374}
13375
13379static pm_node_t *
13380parse_write(pm_parser_t *parser, pm_node_t *target, pm_token_t *operator, pm_node_t *value) {
13381 switch (PM_NODE_TYPE(target)) {
13382 case PM_ERROR_RECOVERY_NODE:
13383 return target;
13384 case PM_CLASS_VARIABLE_READ_NODE: {
13385 pm_class_variable_write_node_t *node = pm_class_variable_write_node_create(parser, (pm_class_variable_read_node_t *) target, operator, value);
13386 return UP(node);
13387 }
13388 case PM_CONSTANT_PATH_NODE: {
13389 pm_node_t *node = UP(pm_constant_path_write_node_create(parser, (pm_constant_path_node_t *) target, operator, value));
13390
13391 if (context_def_p(parser)) {
13392 pm_parser_err_node(parser, node, PM_ERR_WRITE_TARGET_IN_METHOD);
13393 }
13394
13395 return parse_shareable_constant_write(parser, node);
13396 }
13397 case PM_CONSTANT_READ_NODE: {
13398 pm_node_t *node = UP(pm_constant_write_node_create(parser, (pm_constant_read_node_t *) target, operator, value));
13399
13400 if (context_def_p(parser)) {
13401 pm_parser_err_node(parser, node, PM_ERR_WRITE_TARGET_IN_METHOD);
13402 }
13403
13404 return parse_shareable_constant_write(parser, node);
13405 }
13406 case PM_BACK_REFERENCE_READ_NODE:
13407 case PM_NUMBERED_REFERENCE_READ_NODE:
13408 PM_PARSER_ERR_NODE_FORMAT_CONTENT(parser, target, PM_ERR_WRITE_TARGET_READONLY);
13410 case PM_GLOBAL_VARIABLE_READ_NODE: {
13411 pm_global_variable_write_node_t *node = pm_global_variable_write_node_create(parser, target, operator, value);
13412 return UP(node);
13413 }
13414 case PM_LOCAL_VARIABLE_READ_NODE: {
13416
13417 pm_location_t location = target->location;
13418 pm_constant_id_t name = local_read->name;
13419 uint32_t depth = local_read->depth;
13420 pm_scope_t *scope = pm_parser_scope_find(parser, depth);
13421
13422 if (pm_token_is_numbered_parameter(parser, PM_NODE_START(target), PM_NODE_LENGTH(target))) {
13423 pm_diagnostic_id_t diag_id = (scope->parameters & PM_SCOPE_PARAMETERS_NUMBERED_FOUND) ? PM_ERR_EXPRESSION_NOT_WRITABLE_NUMBERED : PM_ERR_PARAMETER_NUMBERED_RESERVED;
13424 PM_PARSER_ERR_FORMAT(parser, PM_NODE_START(target), PM_NODE_LENGTH(target), diag_id, parser->start + PM_NODE_START(target));
13425 pm_node_unreference(parser, target);
13426 }
13427
13428 pm_locals_unread(&scope->locals, name);
13429
13430 return UP(pm_local_variable_write_node_create(parser, name, depth, value, &location, operator));
13431 }
13432 case PM_IT_LOCAL_VARIABLE_READ_NODE: {
13433 pm_constant_id_t name = pm_parser_local_add_constant(parser, "it", 2);
13434 pm_node_t *node = UP(pm_local_variable_write_node_create(parser, name, 0, value, &target->location, operator));
13435
13436 pm_node_unreference(parser, target);
13437
13438 return node;
13439 }
13440 case PM_INSTANCE_VARIABLE_READ_NODE: {
13441 pm_node_t *write_node = UP(pm_instance_variable_write_node_create(parser, (pm_instance_variable_read_node_t *) target, operator, value));
13442 return write_node;
13443 }
13444 case PM_MULTI_TARGET_NODE:
13445 return UP(pm_multi_write_node_create(parser, (pm_multi_target_node_t *) target, operator, value));
13446 case PM_SPLAT_NODE: {
13447 pm_splat_node_t *splat = (pm_splat_node_t *) target;
13448
13449 if (splat->expression != NULL) {
13450 splat->expression = parse_write(parser, splat->expression, operator, value);
13451 }
13452
13453 pm_multi_target_node_t *multi_target = pm_multi_target_node_create(parser);
13454 pm_multi_target_node_targets_append(parser, multi_target, UP(splat));
13455
13456 return UP(pm_multi_write_node_create(parser, multi_target, operator, value));
13457 }
13458 case PM_CALL_NODE: {
13459 pm_call_node_t *call = (pm_call_node_t *) target;
13460
13461 // If we have no arguments to the call node and we need this to be a
13462 // target then this is either a method call or a local variable
13463 // write.
13464 if (
13465 (call->message_loc.length > 0) &&
13466 (parser->start[call->message_loc.start + call->message_loc.length - 1] != '!') &&
13467 (parser->start[call->message_loc.start + call->message_loc.length - 1] != '?') &&
13468 (call->opening_loc.length == 0) &&
13469 (call->arguments == NULL) &&
13470 (call->block == NULL)
13471 ) {
13472 if (call->receiver == NULL) {
13473 // When we get here, we have a local variable write, because it
13474 // was previously marked as a method call but now we have an =.
13475 // This looks like:
13476 //
13477 // foo = 1
13478 //
13479 // When it was parsed in the prefix position, foo was seen as a
13480 // method call with no receiver and no arguments. Now we have an
13481 // =, so we know it's a local variable write.
13482 pm_location_t message_loc = call->message_loc;
13483
13484 pm_refute_numbered_parameter(parser, message_loc.start, message_loc.length);
13485 pm_parser_local_add_location(parser, &message_loc, 0);
13486
13487 pm_constant_id_t constant_id = pm_parser_constant_id_raw(parser, parser->start + PM_LOCATION_START(&message_loc), parser->start + PM_LOCATION_END(&message_loc));
13488 target = UP(pm_local_variable_write_node_create(parser, constant_id, 0, value, &message_loc, operator));
13489
13490 return target;
13491 }
13492
13493 if (char_is_identifier_start(parser, parser->start + call->message_loc.start, (ptrdiff_t) call->message_loc.length)) {
13494 // When we get here, we have a method call, because it was
13495 // previously marked as a method call but now we have an =. This
13496 // looks like:
13497 //
13498 // foo.bar = 1
13499 //
13500 // When it was parsed in the prefix position, foo.bar was seen as a
13501 // method call with no arguments. Now we have an =, so we know it's
13502 // a method call with an argument. In this case we will create the
13503 // arguments node, parse the argument, and add it to the list.
13504 pm_arguments_node_t *arguments = pm_arguments_node_create(parser);
13505 call->arguments = arguments;
13506
13507 pm_arguments_node_arguments_append(parser->arena, arguments, value);
13508 PM_NODE_LENGTH_SET_NODE(call, arguments);
13509 call->equal_loc = TOK2LOC(parser, operator);
13510
13511 parse_write_name(parser, &call->name);
13512 pm_node_flag_set(UP(call), PM_CALL_NODE_FLAGS_ATTRIBUTE_WRITE | pm_implicit_array_write_flags(value, PM_CALL_NODE_FLAGS_IMPLICIT_ARRAY));
13513
13514 return UP(call);
13515 }
13516 }
13517
13518 // If there is no call operator and the message is "[]" then this is
13519 // an aref expression, and we can transform it into an aset
13520 // expression.
13521 if (PM_NODE_FLAG_P(call, PM_CALL_NODE_FLAGS_INDEX)) {
13522 if (call->arguments == NULL) {
13523 call->arguments = pm_arguments_node_create(parser);
13524 }
13525
13526 pm_arguments_node_arguments_append(parser->arena, call->arguments, value);
13527 PM_NODE_LENGTH_SET_NODE(target, value);
13528
13529 // Replace the name with "[]=".
13530 call->name = pm_parser_constant_id_constant(parser, "[]=", 3);
13531 call->equal_loc = TOK2LOC(parser, operator);
13532
13533 // Ensure that the arguments for []= don't contain keywords
13534 pm_index_arguments_check(parser, call->arguments, call->block);
13535 pm_node_flag_set(UP(call), PM_CALL_NODE_FLAGS_ATTRIBUTE_WRITE | pm_implicit_array_write_flags(value, PM_CALL_NODE_FLAGS_IMPLICIT_ARRAY));
13536
13537 return target;
13538 }
13539
13540 // If there are arguments on the call node, then it can't be a
13541 // method call ending with = or a local variable write, so it must
13542 // be a syntax error. In this case we'll fall through to our default
13543 // handling. We need to free the value that we parsed because there
13544 // is no way for us to attach it to the tree at this point.
13545 //
13546 // Since it is possible for the value to contain an implicit
13547 // parameter somewhere in its subtree, we need to walk it and remove
13548 // any implicit parameters from the list of implicit parameters for
13549 // the current scope.
13550 pm_node_unreference(parser, value);
13551 }
13553 default:
13554 // In this case we have a node that we don't know how to convert into a
13555 // target. We need to treat it as an error. For now, we'll mark it as an
13556 // error and just skip right past it.
13557 pm_parser_err_token(parser, operator, PM_ERR_WRITE_TARGET_UNEXPECTED);
13558 return target;
13559 }
13560}
13561
13568static pm_node_t *
13569parse_unwriteable_write(pm_parser_t *parser, pm_node_t *target, const pm_token_t *equals, pm_node_t *value) {
13570 switch (PM_NODE_TYPE(target)) {
13571 case PM_SOURCE_ENCODING_NODE: pm_parser_err_token(parser, equals, PM_ERR_EXPRESSION_NOT_WRITABLE_ENCODING); break;
13572 case PM_FALSE_NODE: pm_parser_err_token(parser, equals, PM_ERR_EXPRESSION_NOT_WRITABLE_FALSE); break;
13573 case PM_SOURCE_FILE_NODE: pm_parser_err_token(parser, equals, PM_ERR_EXPRESSION_NOT_WRITABLE_FILE); break;
13574 case PM_SOURCE_LINE_NODE: pm_parser_err_token(parser, equals, PM_ERR_EXPRESSION_NOT_WRITABLE_LINE); break;
13575 case PM_NIL_NODE: pm_parser_err_token(parser, equals, PM_ERR_EXPRESSION_NOT_WRITABLE_NIL); break;
13576 case PM_SELF_NODE: pm_parser_err_token(parser, equals, PM_ERR_EXPRESSION_NOT_WRITABLE_SELF); break;
13577 case PM_TRUE_NODE: pm_parser_err_token(parser, equals, PM_ERR_EXPRESSION_NOT_WRITABLE_TRUE); break;
13578 default: break;
13579 }
13580
13581 pm_constant_id_t name = pm_parser_local_add_location(parser, &target->location, 1);
13582 pm_local_variable_write_node_t *result = pm_local_variable_write_node_create(parser, name, 0, value, &target->location, equals);
13583
13584 return UP(result);
13585}
13586
13597static pm_node_t *
13598parse_targets(pm_parser_t *parser, pm_node_t *first_target, pm_binding_power_t binding_power, uint16_t depth) {
13599 bool has_rest = PM_NODE_TYPE_P(first_target, PM_SPLAT_NODE);
13600
13601 pm_multi_target_node_t *result = pm_multi_target_node_create(parser);
13602 pm_multi_target_node_targets_append(parser, result, parse_target(parser, first_target, true, false));
13603
13604 while (accept1(parser, PM_TOKEN_COMMA)) {
13605 if (accept1(parser, PM_TOKEN_USTAR)) {
13606 // Here we have a splat operator. It can have a name or be
13607 // anonymous. It can be the final target or be in the middle if
13608 // there haven't been any others yet.
13609 if (has_rest) {
13610 pm_parser_err_previous(parser, PM_ERR_MULTI_ASSIGN_MULTI_SPLATS);
13611 }
13612
13613 pm_token_t star_operator = parser->previous;
13614 pm_node_t *name = NULL;
13615
13616 if (token_begins_expression_p(parser->current.type)) {
13617 name = parse_expression(parser, binding_power, PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_EXPECT_EXPRESSION_AFTER_STAR, (uint16_t) (depth + 1));
13618 name = parse_target(parser, name, true, true);
13619 }
13620
13621 pm_node_t *splat = UP(pm_splat_node_create(parser, &star_operator, name));
13622 pm_multi_target_node_targets_append(parser, result, splat);
13623 has_rest = true;
13624 } else if (match1(parser, PM_TOKEN_PARENTHESIS_LEFT_GROUPING)) {
13625 context_push(parser, PM_CONTEXT_MULTI_TARGET);
13626 pm_node_t *target = parse_expression(parser, binding_power, PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_EXPECT_EXPRESSION_AFTER_COMMA, (uint16_t) (depth + 1));
13627 target = parse_target(parser, target, true, false);
13628
13629 pm_multi_target_node_targets_append(parser, result, target);
13630 context_pop(parser);
13631 } else if (token_begins_expression_p(parser->current.type)) {
13632 pm_node_t *target = parse_expression(parser, binding_power, PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_EXPECT_EXPRESSION_AFTER_COMMA, (uint16_t) (depth + 1));
13633 target = parse_target(parser, target, true, false);
13634
13635 pm_multi_target_node_targets_append(parser, result, target);
13636 } else if (!match1(parser, PM_TOKEN_EOF)) {
13637 // If we get here, then we have a trailing , in a multi target node.
13638 // We'll add an implicit rest node to represent this.
13639 pm_node_t *rest = UP(pm_implicit_rest_node_create(parser, &parser->previous));
13640 pm_multi_target_node_targets_append(parser, result, rest);
13641 break;
13642 }
13643 }
13644
13645 return UP(result);
13646}
13647
13652static pm_node_t *
13653parse_targets_validate(pm_parser_t *parser, pm_node_t *first_target, pm_binding_power_t binding_power, uint16_t depth) {
13654 pm_node_t *result = parse_targets(parser, first_target, binding_power, depth);
13655
13656 // If we're inside parentheses, then we allow a newline before the
13657 // closing parenthesis or equals sign. Outside of parentheses, a newline
13658 // is not allowed (e.g., `a, b\n= 1, 2` is not valid).
13659 if (context_p(parser, PM_CONTEXT_PARENS) || context_p(parser, PM_CONTEXT_MULTI_TARGET)) {
13660 accept1(parser, PM_TOKEN_NEWLINE);
13661 }
13662
13663 // Ensure that we have either an = or a ) after the targets.
13664 if (!match2(parser, PM_TOKEN_EQUAL, PM_TOKEN_PARENTHESIS_RIGHT)) {
13665 pm_parser_err_node(parser, result, PM_ERR_WRITE_TARGET_UNEXPECTED);
13666 }
13667
13668 return result;
13669}
13670
13674static pm_statements_node_t *
13675parse_statements(pm_parser_t *parser, pm_context_t context, uint16_t depth) {
13676 // First, skip past any optional terminators that might be at the beginning
13677 // of the statements.
13678 while (accept2(parser, PM_TOKEN_SEMICOLON, PM_TOKEN_NEWLINE));
13679
13680 // If we have a terminator, then we can just return NULL.
13681 if (context_terminator(context, &parser->current)) return NULL;
13682
13683 pm_statements_node_t *statements = pm_statements_node_create(parser);
13684
13685 // At this point we know we have at least one statement, and that it
13686 // immediately follows the current token.
13687 context_push(parser, context);
13688
13689 while (true) {
13690 pm_node_t *node = parse_expression(parser, PM_BINDING_POWER_STATEMENT, PM_PARSE_ACCEPTS_COMMAND_CALL | PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_CANNOT_PARSE_EXPRESSION, (uint16_t) (depth + 1));
13691 pm_statements_node_body_append(parser, statements, node, true);
13692
13693 // If we're recovering from a syntax error, then we need to stop parsing
13694 // the statements now.
13695 if (parser->recovering) {
13696 // If this is the level of context where the recovery has happened,
13697 // then we can mark the parser as done recovering.
13698 if (context_terminator(context, &parser->current)) parser->recovering = false;
13699 break;
13700 }
13701
13702 // If we have a terminator, then we will parse all consecutive
13703 // terminators and then continue parsing the statements list.
13704 if (accept2(parser, PM_TOKEN_NEWLINE, PM_TOKEN_SEMICOLON)) {
13705 // If we have a terminator, then we will continue parsing the
13706 // statements list.
13707 while (accept2(parser, PM_TOKEN_NEWLINE, PM_TOKEN_SEMICOLON));
13708 if (context_terminator(context, &parser->current)) break;
13709
13710 // Now we can continue parsing the list of statements.
13711 continue;
13712 }
13713
13714 // At this point we have a list of statements that are not terminated by
13715 // a newline or semicolon. At this point we need to check if we're at
13716 // the end of the statements list. If we are, then we should break out
13717 // of the loop.
13718 if (context_terminator(context, &parser->current)) break;
13719
13720 // At this point, we have a syntax error, because the statement was not
13721 // terminated by a newline or semicolon, and we're not at the end of the
13722 // statements list. Ideally we should scan forward to determine if we
13723 // should insert a missing terminator or break out of parsing the
13724 // statements list at this point.
13725 //
13726 // We don't have that yet, so instead we'll do a more naive approach. If
13727 // we were unable to parse an expression, then we will skip past this
13728 // token and continue parsing the statements list. Otherwise we'll add
13729 // an error and continue parsing the statements list.
13730 if (PM_NODE_TYPE_P(node, PM_ERROR_RECOVERY_NODE)) {
13731 parser_lex(parser);
13732
13733 // If we are at the end of the file, then we need to stop parsing
13734 // the statements entirely at this point. Mark the parser as
13735 // recovering, as we know that EOF closes the top-level context, and
13736 // then break out of the loop.
13737 if (match1(parser, PM_TOKEN_EOF)) {
13738 parser->recovering = true;
13739 break;
13740 }
13741
13742 while (accept2(parser, PM_TOKEN_NEWLINE, PM_TOKEN_SEMICOLON));
13743 if (context_terminator(context, &parser->current)) break;
13744 } else if (!accept2(parser, PM_TOKEN_NEWLINE, PM_TOKEN_EOF)) {
13745 // This is an inlined version of accept1 because the error that we
13746 // want to add has varargs. If this happens again, we should
13747 // probably extract a helper function.
13748 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->current, PM_ERR_EXPECT_EOL_AFTER_STATEMENT, pm_token_str(parser->current.type));
13749 parser->previous.start = parser->previous.end;
13750 parser->previous.type = 0;
13751 }
13752 }
13753
13754 context_pop(parser);
13755
13756 bool last_value = true;
13757 switch (context) {
13758 case PM_CONTEXT_BEGIN_ENSURE:
13759 case PM_CONTEXT_DEF_ENSURE:
13760 last_value = false;
13761 break;
13762 default:
13763 break;
13764 }
13765 pm_void_statements_check(parser, statements, last_value);
13766
13767 return statements;
13768}
13769
13773static void
13774pm_hash_key_duplicated_warn(pm_parser_t *parser, const pm_node_t *duplicated, const pm_node_t *node) {
13775 pm_buffer_t buffer = { 0 };
13776 pm_static_literal_inspect(&buffer, &parser->line_offsets, parser->start, parser->start_line, parser->encoding, duplicated);
13777
13778 pm_diagnostic_list_append_format(
13779 &parser->metadata_arena,
13780 &parser->warning_list,
13781 duplicated->location.start,
13782 duplicated->location.length,
13783 PM_WARN_DUPLICATED_HASH_KEY,
13784 (int) pm_buffer_length(&buffer),
13785 pm_buffer_value(&buffer),
13786 pm_line_offset_list_line_column(&parser->line_offsets, PM_NODE_START(node), parser->start_line).line
13787 );
13788
13789 pm_buffer_cleanup(&buffer);
13790}
13791
13796static void
13797pm_hash_key_static_literals_add(pm_parser_t *parser, pm_static_literals_t *literals, pm_node_t *node) {
13798 const pm_node_t *duplicated = pm_static_literals_add(&parser->line_offsets, parser->start, parser->start_line, parser->encoding, literals, node, true);
13799
13800 if (duplicated != NULL) {
13801 pm_hash_key_duplicated_warn(parser, duplicated, node);
13802 }
13803}
13804
13812static void
13813pm_hash_key_static_literals_merge(pm_parser_t *parser, pm_static_literals_t *literals, const pm_hash_node_t *hash, uint32_t boundary) {
13814 const pm_node_list_t *elements = &hash->elements;
13815
13816 for (size_t index = 0; index < elements->size; index++) {
13817 pm_node_t *element = elements->nodes[index];
13818
13819 switch (PM_NODE_TYPE(element)) {
13820 case PM_ASSOC_NODE: {
13821 pm_node_t *key = ((pm_assoc_node_t *) element)->key;
13822 const pm_node_t *duplicated = pm_static_literals_add(&parser->line_offsets, parser->start, parser->start_line, parser->encoding, literals, key, true);
13823
13824 if (duplicated != NULL && PM_NODE_START(duplicated) < boundary) {
13825 pm_hash_key_duplicated_warn(parser, duplicated, key);
13826 }
13827
13828 break;
13829 }
13830 case PM_ASSOC_SPLAT_NODE: {
13831 const pm_node_t *value = ((pm_assoc_splat_node_t *) element)->value;
13832
13833 if (value != NULL && PM_NODE_TYPE_P(value, PM_HASH_NODE)) {
13834 pm_hash_key_static_literals_merge(parser, literals, (const pm_hash_node_t *) value, boundary);
13835 }
13836
13837 break;
13838 }
13839 default:
13840 break;
13841 }
13842 }
13843}
13844
13849static void
13850pm_when_clause_static_literals_add(pm_parser_t *parser, pm_static_literals_t *literals, pm_node_t *node) {
13851 pm_node_t *previous;
13852
13853 if ((previous = pm_static_literals_add(&parser->line_offsets, parser->start, parser->start_line, parser->encoding, literals, node, false)) != NULL) {
13854 pm_diagnostic_list_append_format(
13855 &parser->metadata_arena,
13856 &parser->warning_list,
13857 PM_NODE_START(node),
13858 PM_NODE_LENGTH(node),
13859 PM_WARN_DUPLICATED_WHEN_CLAUSE,
13860 pm_line_offset_list_line_column(&parser->line_offsets, PM_NODE_START(node), parser->start_line).line,
13861 pm_line_offset_list_line_column(&parser->line_offsets, PM_NODE_START(previous), parser->start_line).line
13862 );
13863 }
13864}
13865
13869static bool
13870parse_assocs(pm_parser_t *parser, pm_static_literals_t *literals, pm_node_t *node, uint16_t depth) {
13871 assert(PM_NODE_TYPE_P(node, PM_HASH_NODE) || PM_NODE_TYPE_P(node, PM_KEYWORD_HASH_NODE));
13872 bool contains_keyword_splat = false;
13873
13874 while (true) {
13875 pm_node_t *element;
13876
13877 switch (parser->current.type) {
13878 case PM_TOKEN_USTAR_STAR: {
13879 parser_lex(parser);
13880 pm_token_t operator = parser->previous;
13881 pm_node_t *value = NULL;
13882
13883 if (token_begins_expression_p(parser->current.type)) {
13884 value = parse_value_expression(parser, PM_BINDING_POWER_DEFINED, PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_EXPECT_EXPRESSION_AFTER_SPLAT_HASH, (uint16_t) (depth + 1));
13885
13886 /* If the splatted value is itself a hash literal, its keys
13887 * become part of this hash for the duplicate key warning. */
13888 if (value != NULL && PM_NODE_TYPE_P(value, PM_HASH_NODE)) {
13889 pm_hash_key_static_literals_merge(parser, literals, (const pm_hash_node_t *) value, PM_NODE_START(value));
13890 }
13891 } else {
13892 pm_parser_scope_forwarding_keywords_check(parser, &operator);
13893 }
13894
13895 element = UP(pm_assoc_splat_node_create(parser, value, &operator));
13896 contains_keyword_splat = true;
13897 break;
13898 }
13899 case PM_TOKEN_LABEL: {
13900 pm_token_t label = parser->current;
13901 parser_lex(parser);
13902
13903 pm_node_t *key = UP(pm_symbol_node_label_create(parser, &label));
13904 pm_hash_key_static_literals_add(parser, literals, key);
13905
13906 pm_node_t *value = NULL;
13907
13908 if (token_begins_expression_p(parser->current.type)) {
13909 value = parse_value_expression(parser, PM_BINDING_POWER_DEFINED, PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_HASH_EXPRESSION_AFTER_LABEL, (uint16_t) (depth + 1));
13910 } else {
13911 if (parser->encoding->isupper_char(label.start, (label.end - 1) - label.start)) {
13912 pm_token_t constant = { .type = PM_TOKEN_CONSTANT, .start = label.start, .end = label.end - 1 };
13913 value = UP(pm_constant_read_node_create(parser, &constant));
13914 } else {
13915 int depth = -1;
13916 pm_token_t identifier = { .type = PM_TOKEN_IDENTIFIER, .start = label.start, .end = label.end - 1 };
13917
13918 if (identifier.end[-1] == '!' || identifier.end[-1] == '?') {
13919 PM_PARSER_ERR_TOKEN_FORMAT_CONTENT(parser, &identifier, PM_ERR_INVALID_LOCAL_VARIABLE_READ);
13920 } else {
13921 depth = pm_parser_local_depth(parser, &identifier);
13922 }
13923
13924 if (depth == -1) {
13925 value = UP(pm_call_node_variable_call_create(parser, &identifier));
13926 } else {
13927 value = UP(pm_local_variable_read_node_create(parser, &identifier, (uint32_t) depth));
13928 }
13929 }
13930
13931 value->location.length++;
13932 value = UP(pm_implicit_node_create(parser, value));
13933 }
13934
13935 element = UP(pm_assoc_node_create(parser, key, NULL, value));
13936 break;
13937 }
13938 default: {
13939 pm_node_t *key = parse_value_expression(parser, PM_BINDING_POWER_DEFINED, PM_PARSE_ACCEPTS_DO_BLOCK | PM_PARSE_ACCEPTS_LABEL, PM_ERR_HASH_KEY, (uint16_t) (depth + 1));
13940
13941 // Hash keys that are strings are automatically frozen. We will
13942 // mark that here.
13943 if (PM_NODE_TYPE_P(key, PM_STRING_NODE)) {
13944 pm_node_flag_set(key, PM_STRING_FLAGS_FROZEN | PM_NODE_FLAG_STATIC_LITERAL);
13945 }
13946
13947 pm_hash_key_static_literals_add(parser, literals, key);
13948
13949 pm_token_t operator = { 0 };
13950 if (!pm_symbol_node_label_p(parser, key)) {
13951 expect1(parser, PM_TOKEN_EQUAL_GREATER, PM_ERR_HASH_ROCKET);
13952 operator = parser->previous;
13953 }
13954
13955 pm_node_t *value = parse_value_expression(parser, PM_BINDING_POWER_DEFINED, PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_HASH_VALUE, (uint16_t) (depth + 1));
13956 element = UP(pm_assoc_node_create(parser, key, NTOK2PTR(operator), value));
13957 break;
13958 }
13959 }
13960
13961 if (PM_NODE_TYPE_P(node, PM_HASH_NODE)) {
13962 pm_hash_node_elements_append(parser->arena, (pm_hash_node_t *) node, element);
13963 } else {
13964 pm_keyword_hash_node_elements_append(parser->arena, (pm_keyword_hash_node_t *) node, element);
13965 }
13966
13967 // If there's no comma after the element, then we're done.
13968 if (!accept1(parser, PM_TOKEN_COMMA)) break;
13969
13970 // If the next element starts with a label or a **, then we know we have
13971 // another element in the hash, so we'll continue parsing.
13972 if (match2(parser, PM_TOKEN_USTAR_STAR, PM_TOKEN_LABEL)) continue;
13973
13974 // Otherwise we need to check if the subsequent token begins an expression.
13975 // If it does, then we'll continue parsing.
13976 if (token_begins_expression_p(parser->current.type)) continue;
13977
13978 // Otherwise by default we will exit out of this loop.
13979 break;
13980 }
13981
13982 return contains_keyword_splat;
13983}
13984
13985static PRISM_INLINE bool
13986argument_allowed_for_bare_hash(pm_parser_t *parser, pm_node_t *argument) {
13987 if (pm_symbol_node_label_p(parser, argument)) {
13988 return true;
13989 }
13990
13991 switch (PM_NODE_TYPE(argument)) {
13992 case PM_CALL_NODE: {
13993 pm_call_node_t *cast = (pm_call_node_t *) argument;
13994 if (cast->opening_loc.length == 0 && cast->arguments != NULL) {
13995 if (PM_NODE_FLAG_P(cast->arguments, PM_ARGUMENTS_NODE_FLAGS_CONTAINS_KEYWORDS | PM_ARGUMENTS_NODE_FLAGS_CONTAINS_SPLAT)) {
13996 return false;
13997 }
13998 if (cast->block != NULL) {
13999 return false;
14000 }
14001 }
14002 break;
14003 }
14004 default: break;
14005 }
14006 return accept1(parser, PM_TOKEN_EQUAL_GREATER);
14007}
14008
14012static PRISM_INLINE void
14013parse_arguments_append(pm_parser_t *parser, pm_arguments_t *arguments, pm_node_t *argument) {
14014 if (arguments->arguments == NULL) {
14015 arguments->arguments = pm_arguments_node_create(parser);
14016 }
14017
14018 pm_arguments_node_arguments_append(parser->arena, arguments->arguments, argument);
14019}
14020
14025static PRISM_INLINE bool
14026pm_call_node_command_p(const pm_call_node_t *node) {
14027 return (
14028 (node->opening_loc.length == 0) &&
14029 (node->block == NULL || PM_NODE_TYPE_P(node->block, PM_BLOCK_ARGUMENT_NODE)) &&
14030 (node->arguments != NULL || node->block != NULL)
14031 );
14032}
14033
14043static bool
14044pm_constant_path_command_call_p(const pm_parser_t *parser, const pm_call_node_t *call) {
14045 return (
14046 call->receiver != NULL &&
14047 call->opening_loc.length == 0 &&
14048 call->block != NULL && PM_NODE_TYPE_P(call->block, PM_BLOCK_NODE) &&
14049 call->call_operator_loc.length > 0 &&
14050 parser->start[call->call_operator_loc.start] == ':' &&
14051 call->message_loc.length > 0 &&
14052 parser->encoding->isupper_char(parser->start + call->message_loc.start, (ptrdiff_t) call->message_loc.length)
14053 );
14054}
14055
14061static bool
14062pm_command_call_value_p(const pm_parser_t *parser, const pm_node_t *node) {
14063 switch (PM_NODE_TYPE(node)) {
14064 case PM_CALL_NODE: {
14065 const pm_call_node_t *call = (const pm_call_node_t *) node;
14066
14067 /* Command-style calls (e.g., foo bar, obj.foo bar). Attribute
14068 * writes (e.g., a.b = 1) are not commands. */
14069 if (pm_call_node_command_p(call) && !PM_NODE_FLAG_P(node, PM_CALL_NODE_FLAGS_ATTRIBUTE_WRITE) && (call->receiver == NULL || call->call_operator_loc.length > 0)) {
14070 return true;
14071 }
14072
14073 /* A constant-path command with a brace block, e.g. `Foo::Bar { }`. */
14074 if (pm_constant_path_command_call_p(parser, call)) {
14075 return true;
14076 }
14077
14078 /* A `!` or `not` prefix wrapping a command call (e.g., `!foo bar`,
14079 * `not foo bar`) is also a command-call value. */
14080 if (call->receiver != NULL && call->arguments == NULL && call->opening_loc.length == 0 && call->call_operator_loc.length == 0) {
14081 return pm_command_call_value_p(parser, call->receiver);
14082 }
14083
14084 return false;
14085 }
14086 case PM_SUPER_NODE: {
14087 /* A command-style super (no parens). A super carrying a do-block is
14088 * a block call (it permits chaining), so it is excluded here and
14089 * handled by pm_block_call_p instead. */
14090 const pm_super_node_t *cast = (const pm_super_node_t *) node;
14091 return cast->lparen_loc.length == 0 &&
14092 (cast->arguments != NULL || cast->block != NULL) &&
14093 !(cast->block != NULL && PM_NODE_TYPE_P(cast->block, PM_BLOCK_NODE));
14094 }
14095 case PM_YIELD_NODE: {
14096 const pm_yield_node_t *cast = (const pm_yield_node_t *) node;
14097 return cast->lparen_loc.length == 0 && cast->arguments != NULL;
14098 }
14099 case PM_RESCUE_MODIFIER_NODE:
14100 return pm_command_call_value_p(parser, ((const pm_rescue_modifier_node_t *) node)->expression);
14101 case PM_DEF_NODE: {
14102 const pm_def_node_t *cast = (const pm_def_node_t *) node;
14103 if (cast->equal_loc.length > 0 && cast->body != NULL) {
14104 const pm_node_t *body = cast->body;
14105 if (PM_NODE_TYPE_P(body, PM_STATEMENTS_NODE)) {
14106 body = ((const pm_statements_node_t *) body)->body.nodes[((const pm_statements_node_t *) body)->body.size - 1];
14107 }
14108 return pm_command_call_value_p(parser, body);
14109 }
14110 return false;
14111 }
14112 default:
14113 return false;
14114 }
14115}
14116
14123static bool
14124pm_block_call_p(const pm_node_t *node) {
14125 while (PM_NODE_TYPE_P(node, PM_CALL_NODE)) {
14126 const pm_call_node_t *call = (const pm_call_node_t *) node;
14127
14128 /* Root: a command (no parentheses) carrying command arguments and a
14129 * block (brace or do), e.g. `foo bar do end`, `foo bar { }`. The
14130 * no-parentheses requirement is what distinguishes a command root from
14131 * a method call root like `foo.bar(1) { }`, which is a primary value
14132 * and may be used as an argument.
14133 */
14134 if (call->opening_loc.length == 0 && call->arguments != NULL && call->block != NULL && PM_NODE_TYPE_P(call->block, PM_BLOCK_NODE)) {
14135 return true;
14136 }
14137
14138 /* Walk up the receiver chain of a `.`/`::`/`&.` call (e.g.,
14139 * `foo bar do end.baz(1)`). Parentheses on the chained call are allowed
14140 * here -- in parse.y a `block_call` can be extended by
14141 * `call_op2 operation2 opt_paren_args` and remains a block call.
14142 */
14143 if (call->call_operator_loc.length > 0 && call->receiver != NULL) {
14144 node = call->receiver;
14145 continue;
14146 }
14147
14148 return false;
14149 }
14150
14151 /* A `super` with command arguments and a do-block is also a block-call root
14152 * (parse.y: `command do_block`, where the command is `keyword_super
14153 * command_args`). `super do end` with no arguments is a forwarding super
14154 * (a primary value) and is handled elsewhere.
14155 */
14156 if (PM_NODE_TYPE_P(node, PM_SUPER_NODE)) {
14157 const pm_super_node_t *super = (const pm_super_node_t *) node;
14158 return super->lparen_loc.length == 0 && super->block != NULL && PM_NODE_TYPE_P(super->block, PM_BLOCK_NODE);
14159 }
14160
14161 return false;
14162}
14163
14167static void
14168parse_arguments(pm_parser_t *parser, pm_arguments_t *arguments, bool accepts_forwarding, pm_token_type_t terminator, uint8_t flags, uint16_t depth) {
14169 pm_binding_power_t binding_power = pm_binding_powers[parser->current.type].left;
14170
14171 // First we need to check if the next token is one that could be the start
14172 // of an argument. If it's not, then we can just return.
14173 if (
14174 match2(parser, terminator, PM_TOKEN_EOF) ||
14175 (binding_power != PM_BINDING_POWER_UNSET && binding_power < PM_BINDING_POWER_RANGE) ||
14176 context_terminator(parser->current_context->context, &parser->current)
14177 ) {
14178 return;
14179 }
14180
14181 bool parsed_first_argument = false;
14182 bool parsed_bare_hash = false;
14183 bool parsed_block_argument = false;
14184 bool parsed_forwarding_arguments = false;
14185
14186 while (!match1(parser, PM_TOKEN_EOF)) {
14187 if (parsed_forwarding_arguments) {
14188 pm_parser_err_current(parser, PM_ERR_ARGUMENT_AFTER_FORWARDING_ELLIPSES);
14189 }
14190
14191 pm_node_t *argument = NULL;
14192
14193 switch (parser->current.type) {
14194 case PM_TOKEN_USTAR_STAR:
14195 case PM_TOKEN_LABEL: {
14196 if (parsed_bare_hash) {
14197 pm_parser_err_current(parser, PM_ERR_ARGUMENT_BARE_HASH);
14198 }
14199
14200 pm_keyword_hash_node_t *hash = pm_keyword_hash_node_create(parser);
14201 argument = UP(hash);
14202
14203 pm_static_literals_t hash_keys = { 0 };
14204 bool contains_keyword_splat = parse_assocs(parser, &hash_keys, UP(hash), (uint16_t) (depth + 1));
14205
14206 parse_arguments_append(parser, arguments, argument);
14207
14208 pm_node_flags_t node_flags = PM_ARGUMENTS_NODE_FLAGS_CONTAINS_KEYWORDS;
14209 if (contains_keyword_splat) node_flags |= PM_ARGUMENTS_NODE_FLAGS_CONTAINS_KEYWORD_SPLAT;
14210 pm_node_flag_set(UP(arguments->arguments), node_flags);
14211
14212 pm_static_literals_free(&hash_keys);
14213 parsed_bare_hash = true;
14214
14215 break;
14216 }
14217 case PM_TOKEN_UAMPERSAND: {
14218 parser_lex(parser);
14219 pm_token_t operator = parser->previous;
14220 pm_node_t *expression = NULL;
14221
14222 if (token_begins_expression_p(parser->current.type)) {
14223 expression = parse_value_expression(parser, PM_BINDING_POWER_DEFINED, flags & PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_EXPECT_ARGUMENT, (uint16_t) (depth + 1));
14224 } else {
14225 pm_parser_scope_forwarding_block_check(parser, &operator);
14226 }
14227
14228 argument = UP(pm_block_argument_node_create(parser, &operator, expression));
14229 if (parsed_block_argument) {
14230 parse_arguments_append(parser, arguments, argument);
14231 } else {
14232 arguments->block = argument;
14233 }
14234
14235 if (match1(parser, PM_TOKEN_COMMA)) {
14236 pm_parser_err_current(parser, PM_ERR_ARGUMENT_AFTER_BLOCK);
14237 }
14238
14239 parsed_block_argument = true;
14240 break;
14241 }
14242 case PM_TOKEN_USTAR: {
14243 parser_lex(parser);
14244 pm_token_t operator = parser->previous;
14245
14246 if (match4(parser, PM_TOKEN_PARENTHESIS_RIGHT, PM_TOKEN_COMMA, PM_TOKEN_SEMICOLON, PM_TOKEN_BRACKET_RIGHT)) {
14247 pm_parser_scope_forwarding_positionals_check(parser, &operator);
14248 argument = UP(pm_splat_node_create(parser, &operator, NULL));
14249 if (parsed_bare_hash) {
14250 pm_parser_err_previous(parser, PM_ERR_ARGUMENT_SPLAT_AFTER_ASSOC_SPLAT);
14251 }
14252 } else {
14253 pm_node_t *expression = parse_value_expression(parser, PM_BINDING_POWER_DEFINED, flags & PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_EXPECT_EXPRESSION_AFTER_SPLAT, (uint16_t) (depth + 1));
14254
14255 if (parsed_bare_hash) {
14256 pm_parser_err(parser, PM_TOKEN_START(parser, &operator), PM_NODE_END(expression) - PM_TOKEN_START(parser, &operator), PM_ERR_ARGUMENT_SPLAT_AFTER_ASSOC_SPLAT);
14257 }
14258
14259 argument = UP(pm_splat_node_create(parser, &operator, expression));
14260 }
14261
14262 parse_arguments_append(parser, arguments, argument);
14263 break;
14264 }
14265 case PM_TOKEN_UDOT_DOT_DOT: {
14266 if (accepts_forwarding) {
14267 parser_lex(parser);
14268
14269 if (token_begins_expression_p(parser->current.type)) {
14270 // If the token begins an expression then this ... was
14271 // not actually argument forwarding but was instead a
14272 // range.
14273 pm_token_t operator = parser->previous;
14274 pm_node_t *right = parse_expression(parser, PM_BINDING_POWER_RANGE, flags & PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_EXPECT_EXPRESSION_AFTER_OPERATOR, (uint16_t) (depth + 1));
14275
14276 // If we parse a range, we need to validate that we
14277 // didn't accidentally violate the nonassoc rules of the
14278 // ... operator.
14279 if (PM_NODE_TYPE_P(right, PM_RANGE_NODE)) {
14280 pm_range_node_t *range = (pm_range_node_t *) right;
14281 pm_parser_err(parser, range->operator_loc.start, range->operator_loc.length, PM_ERR_UNEXPECTED_RANGE_OPERATOR);
14282 }
14283
14284 argument = UP(pm_range_node_create(parser, NULL, &operator, right));
14285 } else {
14286 pm_parser_scope_forwarding_all_check(parser, &parser->previous);
14287 if (parsed_first_argument && terminator == PM_TOKEN_EOF) {
14288 pm_parser_err_previous(parser, PM_ERR_ARGUMENT_FORWARDING_UNBOUND);
14289 }
14290
14291 argument = UP(pm_forwarding_arguments_node_create(parser, &parser->previous));
14292 parse_arguments_append(parser, arguments, argument);
14293 pm_node_flag_set(UP(arguments->arguments), PM_ARGUMENTS_NODE_FLAGS_CONTAINS_FORWARDING);
14294 arguments->has_forwarding = true;
14295 parsed_forwarding_arguments = true;
14296 break;
14297 }
14298 }
14299 }
14301 default: {
14302 if (argument == NULL) {
14303 argument = parse_value_expression(parser, PM_BINDING_POWER_DEFINED, (uint8_t) ((flags & PM_PARSE_ACCEPTS_DO_BLOCK) | (!parsed_first_argument ? PM_PARSE_ACCEPTS_COMMAND_CALL : 0u) | PM_PARSE_ACCEPTS_LABEL), PM_ERR_EXPECT_ARGUMENT, (uint16_t) (depth + 1));
14304 }
14305
14306 bool contains_keywords = false;
14307 bool contains_keyword_splat = false;
14308
14309 if (argument_allowed_for_bare_hash(parser, argument)) {
14310 if (parsed_bare_hash) {
14311 pm_parser_err_previous(parser, PM_ERR_ARGUMENT_BARE_HASH);
14312 }
14313
14314 /* A hash key must be an argument (`arg`). A command call or
14315 * block call (e.g. `Foo::Bar { } => v`, `foo bar do end =>
14316 * v`) is not an argument, so reject it as a key. Plain
14317 * command calls never reach here as a key because they
14318 * absorb the `=>` into their own arguments first.
14319 */
14320 if (pm_command_call_value_p(parser, argument) || pm_block_call_p(argument)) {
14321 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->previous, PM_ERR_EXPECT_EOL_AFTER_STATEMENT, pm_token_str(parser->previous.type));
14322 }
14323
14324 pm_token_t operator = { 0 };
14325 if (parser->previous.type == PM_TOKEN_EQUAL_GREATER) {
14326 operator = parser->previous;
14327 }
14328
14329 pm_keyword_hash_node_t *bare_hash = pm_keyword_hash_node_create(parser);
14330 contains_keywords = true;
14331
14332 // Create the set of static literals for this hash.
14333 pm_static_literals_t hash_keys = { 0 };
14334 pm_hash_key_static_literals_add(parser, &hash_keys, argument);
14335
14336 // Finish parsing the one we are part way through.
14337 pm_node_t *value = parse_value_expression(parser, PM_BINDING_POWER_DEFINED, flags & PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_HASH_VALUE, (uint16_t) (depth + 1));
14338 argument = UP(pm_assoc_node_create(parser, argument, NTOK2PTR(operator), value));
14339
14340 pm_keyword_hash_node_elements_append(parser->arena, bare_hash, argument);
14341 argument = UP(bare_hash);
14342
14343 // Then parse more if we have a comma
14344 if (accept1(parser, PM_TOKEN_COMMA) && (
14345 token_begins_expression_p(parser->current.type) ||
14346 match2(parser, PM_TOKEN_USTAR_STAR, PM_TOKEN_LABEL)
14347 )) {
14348 contains_keyword_splat = parse_assocs(parser, &hash_keys, UP(bare_hash), (uint16_t) (depth + 1));
14349 }
14350
14351 pm_static_literals_free(&hash_keys);
14352 parsed_bare_hash = true;
14353 }
14354
14355 parse_arguments_append(parser, arguments, argument);
14356
14357 pm_node_flags_t node_flags = 0;
14358 if (contains_keywords) node_flags |= PM_ARGUMENTS_NODE_FLAGS_CONTAINS_KEYWORDS;
14359 if (contains_keyword_splat) node_flags |= PM_ARGUMENTS_NODE_FLAGS_CONTAINS_KEYWORD_SPLAT;
14360 pm_node_flag_set(UP(arguments->arguments), node_flags);
14361
14362 break;
14363 }
14364 }
14365
14366 parsed_first_argument = true;
14367
14368 // If parsing the argument failed, we need to stop parsing arguments.
14369 if (PM_NODE_TYPE_P(argument, PM_ERROR_RECOVERY_NODE) || parser->recovering) break;
14370
14371 // If the terminator of these arguments is not EOF, then we have a
14372 // specific token we're looking for. In that case we can accept a
14373 // newline here because it is not functioning as a statement terminator.
14374 bool accepted_newline = false;
14375 if (terminator != PM_TOKEN_EOF) {
14376 accepted_newline = accept1(parser, PM_TOKEN_NEWLINE);
14377 }
14378
14379 if (parser->previous.type == PM_TOKEN_COMMA && parsed_bare_hash) {
14380 // If we previously were on a comma and we just parsed a bare hash,
14381 // then we want to continue parsing arguments. This is because the
14382 // comma was grabbed up by the hash parser.
14383 } else if (accept1(parser, PM_TOKEN_COMMA)) {
14384 // If there was a comma, then we need to check if we also accepted a
14385 // newline. If we did, then this is a syntax error.
14386 if (accepted_newline) {
14387 pm_parser_err_previous(parser, PM_ERR_INVALID_COMMA);
14388 }
14389
14390 // If this is a command call and an argument takes a block,
14391 // there can be no further arguments. For example,
14392 // `foo(bar 1 do end, 2)` should be rejected.
14393 if (PM_NODE_TYPE_P(argument, PM_CALL_NODE)) {
14394 pm_call_node_t *call = (pm_call_node_t *) argument;
14395 if (call->opening_loc.length == 0 && call->arguments != NULL && call->block != NULL) {
14396 pm_parser_err_previous(parser, PM_ERR_INVALID_COMMA);
14397 break;
14398 }
14399 }
14400 } else {
14401 // If there is no comma at the end of the argument list then we're
14402 // done parsing arguments and can break out of this loop.
14403 break;
14404 }
14405
14406 // If we hit the terminator, then that means we have a trailing comma so
14407 // we can accept that output as well.
14408 if (match1(parser, terminator)) {
14409 // A forwarding `...` argument must be the last argument and cannot
14410 // be followed by a trailing comma, e.g. `foo(...,)`. A comma
14411 // followed by another argument is already rejected at the top of
14412 // this loop, so the only case left to reject here is the trailing
14413 // one.
14414 if (parsed_forwarding_arguments) {
14415 pm_parser_err_previous(parser, PM_ERR_INVALID_COMMA);
14416 }
14417
14418 break;
14419 }
14420 }
14421}
14422
14434parse_required_destructured_parameter(pm_parser_t *parser) {
14435 expect1(parser, PM_TOKEN_PARENTHESIS_LEFT_GROUPING, PM_ERR_EXPECT_LPAREN_REQ_PARAMETER);
14436
14437 pm_multi_target_node_t *node = pm_multi_target_node_create(parser);
14438 pm_multi_target_node_opening_set(parser, node, &parser->previous);
14439
14440 do {
14441 pm_node_t *param;
14442
14443 // If we get here then we have a trailing comma, which isn't allowed in
14444 // the grammar. In other places, multi targets _do_ allow trailing
14445 // commas, so here we'll assume this is a mistake of the user not
14446 // knowing it's not allowed here.
14447 if (node->lefts.size > 0 && match1(parser, PM_TOKEN_PARENTHESIS_RIGHT)) {
14448 param = UP(pm_implicit_rest_node_create(parser, &parser->previous));
14449 pm_multi_target_node_targets_append(parser, node, param);
14450 pm_parser_err_current(parser, PM_ERR_PARAMETER_WILD_LOOSE_COMMA);
14451 break;
14452 }
14453
14454 if (match1(parser, PM_TOKEN_PARENTHESIS_LEFT_GROUPING)) {
14455 param = UP(parse_required_destructured_parameter(parser));
14456 } else if (accept1(parser, PM_TOKEN_USTAR)) {
14457 pm_token_t star = parser->previous;
14458 pm_node_t *value = NULL;
14459
14460 if (accept1(parser, PM_TOKEN_IDENTIFIER)) {
14461 pm_token_t name = parser->previous;
14462 value = UP(pm_required_parameter_node_create(parser, &name));
14463 if (pm_parser_parameter_name_check(parser, &name)) {
14464 pm_node_flag_set_repeated_parameter(value);
14465 }
14466 pm_parser_local_add_token(parser, &name, 1);
14467 }
14468
14469 param = UP(pm_splat_node_create(parser, &star, value));
14470 } else {
14471 expect1(parser, PM_TOKEN_IDENTIFIER, PM_ERR_EXPECT_IDENT_REQ_PARAMETER);
14472 pm_token_t name = parser->previous;
14473
14474 param = UP(pm_required_parameter_node_create(parser, &name));
14475 if (pm_parser_parameter_name_check(parser, &name)) {
14476 pm_node_flag_set_repeated_parameter(param);
14477 }
14478 pm_parser_local_add_token(parser, &name, 1);
14479 }
14480
14481 pm_multi_target_node_targets_append(parser, node, param);
14482 } while (accept1(parser, PM_TOKEN_COMMA));
14483
14484 accept1(parser, PM_TOKEN_NEWLINE);
14485 expect1(parser, PM_TOKEN_PARENTHESIS_RIGHT, PM_ERR_EXPECT_RPAREN_REQ_PARAMETER);
14486 pm_multi_target_node_closing_set(parser, node, &parser->previous);
14487
14488 return node;
14489}
14490
14495typedef enum {
14496 PM_PARAMETERS_NO_CHANGE = 0, // Extra state for tokens that should not change the state
14497 PM_PARAMETERS_ORDER_NOTHING_AFTER = 1,
14498 PM_PARAMETERS_ORDER_KEYWORDS_REST,
14499 PM_PARAMETERS_ORDER_KEYWORDS,
14500 PM_PARAMETERS_ORDER_REST,
14501 PM_PARAMETERS_ORDER_AFTER_OPTIONAL,
14502 PM_PARAMETERS_ORDER_OPTIONAL,
14503 PM_PARAMETERS_ORDER_NAMED,
14504 PM_PARAMETERS_ORDER_NONE,
14505} pm_parameters_order_t;
14506
14510static pm_parameters_order_t parameters_ordering[PM_TOKEN_MAXIMUM] = {
14511 [0] = PM_PARAMETERS_NO_CHANGE,
14512 [PM_TOKEN_UAMPERSAND] = PM_PARAMETERS_ORDER_NOTHING_AFTER,
14513 [PM_TOKEN_AMPERSAND] = PM_PARAMETERS_ORDER_NOTHING_AFTER,
14514 [PM_TOKEN_UDOT_DOT_DOT] = PM_PARAMETERS_ORDER_NOTHING_AFTER,
14515 [PM_TOKEN_IDENTIFIER] = PM_PARAMETERS_ORDER_NAMED,
14516 [PM_TOKEN_PARENTHESIS_LEFT_GROUPING] = PM_PARAMETERS_ORDER_NAMED,
14517 [PM_TOKEN_EQUAL] = PM_PARAMETERS_ORDER_OPTIONAL,
14518 [PM_TOKEN_LABEL] = PM_PARAMETERS_ORDER_KEYWORDS,
14519 [PM_TOKEN_USTAR] = PM_PARAMETERS_ORDER_AFTER_OPTIONAL,
14520 [PM_TOKEN_STAR] = PM_PARAMETERS_ORDER_AFTER_OPTIONAL,
14521 [PM_TOKEN_USTAR_STAR] = PM_PARAMETERS_ORDER_KEYWORDS_REST,
14522 [PM_TOKEN_STAR_STAR] = PM_PARAMETERS_ORDER_KEYWORDS_REST
14523};
14524
14532static bool
14533update_parameter_state(pm_parser_t *parser, pm_token_t *token, pm_parameters_order_t *current) {
14534 pm_parameters_order_t state = parameters_ordering[token->type];
14535 if (state == PM_PARAMETERS_NO_CHANGE) return true;
14536
14537 // If we see another ordered argument after a optional argument
14538 // we only continue parsing ordered arguments until we stop seeing ordered arguments.
14539 if (*current == PM_PARAMETERS_ORDER_OPTIONAL && state == PM_PARAMETERS_ORDER_NAMED) {
14540 *current = PM_PARAMETERS_ORDER_AFTER_OPTIONAL;
14541 return true;
14542 } else if (*current == PM_PARAMETERS_ORDER_AFTER_OPTIONAL && state == PM_PARAMETERS_ORDER_NAMED) {
14543 return true;
14544 }
14545
14546 if (token->type == PM_TOKEN_USTAR && *current == PM_PARAMETERS_ORDER_AFTER_OPTIONAL) {
14547 pm_parser_err_token(parser, token, PM_ERR_PARAMETER_STAR);
14548 return false;
14549 } else if (token->type == PM_TOKEN_UDOT_DOT_DOT && (*current >= PM_PARAMETERS_ORDER_KEYWORDS_REST && *current <= PM_PARAMETERS_ORDER_AFTER_OPTIONAL)) {
14550 pm_parser_err_token(parser, token, *current == PM_PARAMETERS_ORDER_AFTER_OPTIONAL ? PM_ERR_PARAMETER_FORWARDING_AFTER_REST : PM_ERR_PARAMETER_ORDER);
14551 return false;
14552 } else if (*current == PM_PARAMETERS_ORDER_NOTHING_AFTER || state > *current) {
14553 // We know what transition we failed on, so we can provide a better error here.
14554 pm_parser_err_token(parser, token, PM_ERR_PARAMETER_ORDER);
14555 return false;
14556 }
14557
14558 if (state < *current) *current = state;
14559 return true;
14560}
14561
14562static PRISM_INLINE void
14563parse_parameters_handle_trailing_comma(
14564 pm_parser_t *parser,
14565 pm_parameters_node_t *params,
14566 pm_parameters_order_t order,
14567 bool in_block,
14568 bool allows_trailing_comma
14569) {
14570 if (!allows_trailing_comma) {
14571 pm_parser_err_previous(parser, PM_ERR_PARAMETER_WILD_LOOSE_COMMA);
14572 return;
14573 }
14574
14575 if (in_block) {
14576 if (order >= PM_PARAMETERS_ORDER_NAMED) {
14577 // foo do |bar,|; end
14578 pm_node_t *param = UP(pm_implicit_rest_node_create(parser, &parser->previous));
14579
14580 if (params->rest == NULL) {
14581 pm_parameters_node_rest_set(params, param);
14582 } else {
14583 pm_parser_err_node(parser, UP(param), PM_ERR_PARAMETER_SPLAT_MULTI);
14584 pm_parameters_node_posts_append(parser->arena, params, UP(param));
14585 }
14586 } else {
14587 // foo do |*bar,|; end
14588 pm_parser_err_previous(parser, PM_ERR_PARAMETER_WILD_LOOSE_COMMA);
14589 }
14590 } else {
14591 // https://bugs.ruby-lang.org/issues/19107
14592 // Allow `def foo(bar,); end`, `def foo(*bar,); end`, etc. but not `def foo(...,); end`
14593 if (parser->version < PM_OPTIONS_VERSION_CRUBY_4_1 || order == PM_PARAMETERS_ORDER_NOTHING_AFTER) {
14594 pm_parser_err_previous(parser, PM_ERR_PARAMETER_WILD_LOOSE_COMMA);
14595 }
14596 }
14597}
14598
14602static pm_parameters_node_t *
14603parse_parameters(
14604 pm_parser_t *parser,
14605 pm_binding_power_t binding_power,
14606 bool uses_parentheses,
14607 bool allows_trailing_comma,
14608 bool allows_forwarding_parameters,
14609 bool accepts_blocks_in_defaults,
14610 bool in_block,
14611 pm_diagnostic_id_t diag_id_forwarding,
14612 uint16_t depth
14613) {
14614 pm_do_loop_stack_push(parser, false);
14615
14616 pm_parameters_node_t *params = pm_parameters_node_create(parser);
14617 pm_parameters_order_t order = PM_PARAMETERS_ORDER_NONE;
14618
14619 while (true) {
14620 bool parsing = true;
14621
14622 switch (parser->current.type) {
14623 case PM_TOKEN_PARENTHESIS_LEFT_GROUPING: {
14624 update_parameter_state(parser, &parser->current, &order);
14625 pm_node_t *param = UP(parse_required_destructured_parameter(parser));
14626
14627 if (order > PM_PARAMETERS_ORDER_AFTER_OPTIONAL) {
14628 pm_parameters_node_requireds_append(parser->arena, params, param);
14629 } else {
14630 pm_parameters_node_posts_append(parser->arena, params, param);
14631 }
14632 break;
14633 }
14634 case PM_TOKEN_UAMPERSAND:
14635 case PM_TOKEN_AMPERSAND: {
14636 update_parameter_state(parser, &parser->current, &order);
14637 parser_lex(parser);
14638
14639 pm_token_t operator = parser->previous;
14640 pm_node_t *param;
14641
14642 if (parser->version >= PM_OPTIONS_VERSION_CRUBY_4_1 && accept1(parser, PM_TOKEN_KEYWORD_NIL)) {
14643 param = (pm_node_t *) pm_no_block_parameter_node_create(parser, &operator, &parser->previous);
14644 } else {
14645 pm_token_t name = {0};
14646
14647 bool repeated = false;
14648 if (accept1(parser, PM_TOKEN_IDENTIFIER)) {
14649 name = parser->previous;
14650 repeated = pm_parser_parameter_name_check(parser, &name);
14651 pm_parser_local_add_token(parser, &name, 1);
14652 } else {
14653 parser->current_scope->parameters |= PM_SCOPE_PARAMETERS_FORWARDING_BLOCK;
14654 }
14655
14656 param = (pm_node_t *) pm_block_parameter_node_create(parser, NTOK2PTR(name), &operator);
14657 if (repeated) {
14658 pm_node_flag_set_repeated_parameter(param);
14659 }
14660 }
14661
14662 if (params->block == NULL) {
14663 pm_parameters_node_block_set(params, param);
14664 } else {
14665 pm_parser_err_node(parser, param, PM_ERR_PARAMETER_BLOCK_MULTI);
14666 pm_parameters_node_posts_append(parser->arena, params, UP(pm_error_recovery_node_create_unexpected(parser, param)));
14667 }
14668
14669 break;
14670 }
14671 case PM_TOKEN_UDOT_DOT_DOT: {
14672 if (!allows_forwarding_parameters) {
14673 pm_parser_err_current(parser, diag_id_forwarding);
14674 }
14675
14676 bool succeeded = update_parameter_state(parser, &parser->current, &order);
14677 parser_lex(parser);
14678
14679 parser->current_scope->parameters |= PM_SCOPE_PARAMETERS_FORWARDING_ALL;
14680 pm_forwarding_parameter_node_t *param = pm_forwarding_parameter_node_create(parser, &parser->previous);
14681
14682 if (params->keyword_rest != NULL) {
14683 // If we already have a keyword rest parameter, then we replace it with the
14684 // forwarding parameter and move the keyword rest parameter to the posts list.
14685 pm_node_t *keyword_rest = params->keyword_rest;
14686 pm_parameters_node_posts_append(parser->arena, params, UP(pm_error_recovery_node_create_unexpected(parser, keyword_rest)));
14687 if (succeeded) pm_parser_err_previous(parser, PM_ERR_PARAMETER_UNEXPECTED_FWD);
14688 params->keyword_rest = NULL;
14689 }
14690
14691 pm_parameters_node_keyword_rest_set(params, UP(param));
14692 break;
14693 }
14694 case PM_TOKEN_CLASS_VARIABLE:
14695 case PM_TOKEN_IDENTIFIER:
14696 case PM_TOKEN_CONSTANT:
14697 case PM_TOKEN_INSTANCE_VARIABLE:
14698 case PM_TOKEN_GLOBAL_VARIABLE:
14699 case PM_TOKEN_METHOD_NAME: {
14700 parser_lex(parser);
14701 switch (parser->previous.type) {
14702 case PM_TOKEN_CONSTANT:
14703 pm_parser_err_previous(parser, PM_ERR_ARGUMENT_FORMAL_CONSTANT);
14704 break;
14705 case PM_TOKEN_INSTANCE_VARIABLE:
14706 pm_parser_err_previous(parser, PM_ERR_ARGUMENT_FORMAL_IVAR);
14707 break;
14708 case PM_TOKEN_GLOBAL_VARIABLE:
14709 pm_parser_err_previous(parser, PM_ERR_ARGUMENT_FORMAL_GLOBAL);
14710 break;
14711 case PM_TOKEN_CLASS_VARIABLE:
14712 pm_parser_err_previous(parser, PM_ERR_ARGUMENT_FORMAL_CLASS);
14713 break;
14714 case PM_TOKEN_METHOD_NAME:
14715 pm_parser_err_previous(parser, PM_ERR_PARAMETER_METHOD_NAME);
14716 break;
14717 default: break;
14718 }
14719
14720 if (parser->current.type == PM_TOKEN_EQUAL) {
14721 update_parameter_state(parser, &parser->current, &order);
14722 } else {
14723 update_parameter_state(parser, &parser->previous, &order);
14724 }
14725
14726 pm_token_t name = parser->previous;
14727 bool repeated = pm_parser_parameter_name_check(parser, &name);
14728 pm_parser_local_add_token(parser, &name, 1);
14729
14730 if (match1(parser, PM_TOKEN_EQUAL)) {
14731 pm_token_t operator = parser->current;
14732 context_push(parser, PM_CONTEXT_DEFAULT_PARAMS);
14733 parser_lex(parser);
14734
14735 pm_constant_id_t name_id = pm_parser_constant_id_token(parser, &name);
14736 uint32_t reads = parser->version <= PM_OPTIONS_VERSION_CRUBY_3_3 ? pm_locals_reads(&parser->current_scope->locals, name_id) : 0;
14737
14738 if (accepts_blocks_in_defaults) pm_accepts_block_stack_push(parser, true);
14739 pm_node_t *value = parse_value_expression(parser, binding_power, PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_PARAMETER_NO_DEFAULT, (uint16_t) (depth + 1));
14740 if (accepts_blocks_in_defaults) pm_accepts_block_stack_pop(parser);
14741
14742 pm_optional_parameter_node_t *param = pm_optional_parameter_node_create(parser, &name, &operator, value);
14743
14744 if (repeated) {
14745 pm_node_flag_set_repeated_parameter(UP(param));
14746 }
14747 pm_parameters_node_optionals_append(parser->arena, params, param);
14748
14749 // If the value of the parameter increased the number of
14750 // reads of that parameter, then we need to warn that we
14751 // have a circular definition.
14752 if ((parser->version <= PM_OPTIONS_VERSION_CRUBY_3_3) && (pm_locals_reads(&parser->current_scope->locals, name_id) != reads)) {
14753 PM_PARSER_ERR_TOKEN_FORMAT_CONTENT(parser, &name, PM_ERR_PARAMETER_CIRCULAR);
14754 }
14755
14756 context_pop(parser);
14757
14758 // If parsing the value of the parameter resulted in error recovery,
14759 // then we can put a missing node in its place and stop parsing the
14760 // parameters entirely now.
14761 if (parser->recovering) {
14762 parsing = false;
14763 break;
14764 }
14765 } else if (order > PM_PARAMETERS_ORDER_AFTER_OPTIONAL) {
14766 pm_required_parameter_node_t *param = pm_required_parameter_node_create(parser, &name);
14767 if (repeated) {
14768 pm_node_flag_set_repeated_parameter(UP(param));
14769 }
14770 pm_parameters_node_requireds_append(parser->arena, params, UP(param));
14771 } else {
14772 pm_required_parameter_node_t *param = pm_required_parameter_node_create(parser, &name);
14773 if (repeated) {
14774 pm_node_flag_set_repeated_parameter(UP(param));
14775 }
14776 pm_parameters_node_posts_append(parser->arena, params, UP(param));
14777 }
14778
14779 break;
14780 }
14781 case PM_TOKEN_LABEL: {
14782 if (!uses_parentheses && !in_block) parser->in_keyword_arg = true;
14783 update_parameter_state(parser, &parser->current, &order);
14784
14785 context_push(parser, PM_CONTEXT_DEFAULT_PARAMS);
14786 parser_lex(parser);
14787
14788 pm_token_t name = parser->previous;
14789 pm_token_t local = name;
14790 local.end -= 1;
14791
14792 if (parser->encoding_changed ? parser->encoding->isupper_char(local.start, local.end - local.start) : pm_encoding_utf_8_isupper_char(local.start, local.end - local.start)) {
14793 pm_parser_err(parser, PM_TOKEN_START(parser, &local), PM_TOKEN_LENGTH(&local), PM_ERR_ARGUMENT_FORMAL_CONSTANT);
14794 } else if (local.end[-1] == '!' || local.end[-1] == '?') {
14795 PM_PARSER_ERR_TOKEN_FORMAT_CONTENT(parser, &local, PM_ERR_INVALID_LOCAL_VARIABLE_WRITE);
14796 }
14797
14798 bool repeated = pm_parser_parameter_name_check(parser, &local);
14799 pm_parser_local_add_token(parser, &local, 1);
14800
14801 switch (parser->current.type) {
14802 case PM_TOKEN_COMMA:
14803 case PM_TOKEN_PARENTHESIS_RIGHT:
14804 case PM_TOKEN_PIPE: {
14805 context_pop(parser);
14806
14807 pm_node_t *param = UP(pm_required_keyword_parameter_node_create(parser, &name));
14808 if (repeated) {
14809 pm_node_flag_set_repeated_parameter(param);
14810 }
14811
14812 pm_parameters_node_keywords_append(parser->arena, params, param);
14813 break;
14814 }
14815 case PM_TOKEN_SEMICOLON:
14816 case PM_TOKEN_NEWLINE: {
14817 context_pop(parser);
14818
14819 if (uses_parentheses) {
14820 parsing = false;
14821 break;
14822 }
14823
14824 pm_node_t *param = UP(pm_required_keyword_parameter_node_create(parser, &name));
14825 if (repeated) {
14826 pm_node_flag_set_repeated_parameter(param);
14827 }
14828
14829 pm_parameters_node_keywords_append(parser->arena, params, param);
14830 break;
14831 }
14832 default: {
14833 pm_node_t *param;
14834
14835 if (token_begins_expression_p(parser->current.type)) {
14836 pm_constant_id_t name_id = pm_parser_constant_id_token(parser, &local);
14837 uint32_t reads = parser->version <= PM_OPTIONS_VERSION_CRUBY_3_3 ? pm_locals_reads(&parser->current_scope->locals, name_id) : 0;
14838
14839 if (accepts_blocks_in_defaults) pm_accepts_block_stack_push(parser, true);
14840 pm_node_t *value = parse_value_expression(parser, binding_power, PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_PARAMETER_NO_DEFAULT_KW, (uint16_t) (depth + 1));
14841 if (accepts_blocks_in_defaults) pm_accepts_block_stack_pop(parser);
14842
14843 if (parser->version <= PM_OPTIONS_VERSION_CRUBY_3_3 && (pm_locals_reads(&parser->current_scope->locals, name_id) != reads)) {
14844 PM_PARSER_ERR_TOKEN_FORMAT_CONTENT(parser, &local, PM_ERR_PARAMETER_CIRCULAR);
14845 }
14846
14847 param = UP(pm_optional_keyword_parameter_node_create(parser, &name, value));
14848 }
14849 else {
14850 param = UP(pm_required_keyword_parameter_node_create(parser, &name));
14851 }
14852
14853 if (repeated) {
14854 pm_node_flag_set_repeated_parameter(param);
14855 }
14856
14857 context_pop(parser);
14858 pm_parameters_node_keywords_append(parser->arena, params, param);
14859
14860 // If parsing the value of the parameter resulted in error recovery,
14861 // then we can put a missing node in its place and stop parsing the
14862 // parameters entirely now.
14863 if (parser->recovering) {
14864 parsing = false;
14865 break;
14866 }
14867 }
14868 }
14869
14870 parser->in_keyword_arg = false;
14871 break;
14872 }
14873 case PM_TOKEN_USTAR:
14874 case PM_TOKEN_STAR: {
14875 update_parameter_state(parser, &parser->current, &order);
14876 parser_lex(parser);
14877
14878 pm_token_t operator = parser->previous;
14879 pm_token_t name = { 0 };
14880 bool repeated = false;
14881
14882 if (accept1(parser, PM_TOKEN_IDENTIFIER)) {
14883 name = parser->previous;
14884 repeated = pm_parser_parameter_name_check(parser, &name);
14885 pm_parser_local_add_token(parser, &name, 1);
14886 } else {
14887 parser->current_scope->parameters |= PM_SCOPE_PARAMETERS_FORWARDING_POSITIONALS;
14888 }
14889
14890 pm_node_t *param = UP(pm_rest_parameter_node_create(parser, &operator, NTOK2PTR(name)));
14891 if (repeated) {
14892 pm_node_flag_set_repeated_parameter(param);
14893 }
14894
14895 if (params->rest == NULL) {
14896 pm_parameters_node_rest_set(params, param);
14897 } else {
14898 pm_parser_err_node(parser, param, PM_ERR_PARAMETER_SPLAT_MULTI);
14899 pm_parameters_node_posts_append(parser->arena, params, param);
14900 }
14901
14902 break;
14903 }
14904 case PM_TOKEN_STAR_STAR:
14905 case PM_TOKEN_USTAR_STAR: {
14906 pm_parameters_order_t previous_order = order;
14907 update_parameter_state(parser, &parser->current, &order);
14908 parser_lex(parser);
14909
14910 pm_token_t operator = parser->previous;
14911 pm_node_t *param;
14912
14913 if (accept1(parser, PM_TOKEN_KEYWORD_NIL)) {
14914 if (previous_order <= PM_PARAMETERS_ORDER_KEYWORDS) {
14915 pm_parser_err_previous(parser, PM_ERR_PARAMETER_UNEXPECTED_NO_KW);
14916 }
14917
14918 param = UP(pm_no_keywords_parameter_node_create(parser, &operator, &parser->previous));
14919 } else {
14920 pm_token_t name = { 0 };
14921
14922 bool repeated = false;
14923 if (accept1(parser, PM_TOKEN_IDENTIFIER)) {
14924 name = parser->previous;
14925 repeated = pm_parser_parameter_name_check(parser, &name);
14926 pm_parser_local_add_token(parser, &name, 1);
14927 } else {
14928 parser->current_scope->parameters |= PM_SCOPE_PARAMETERS_FORWARDING_KEYWORDS;
14929 }
14930
14931 param = UP(pm_keyword_rest_parameter_node_create(parser, &operator, NTOK2PTR(name)));
14932 if (repeated) {
14933 pm_node_flag_set_repeated_parameter(param);
14934 }
14935 }
14936
14937 if (params->keyword_rest == NULL) {
14938 pm_parameters_node_keyword_rest_set(params, param);
14939 } else {
14940 pm_parser_err_node(parser, param, PM_ERR_PARAMETER_ASSOC_SPLAT_MULTI);
14941 pm_parameters_node_posts_append(parser->arena, params, UP(pm_error_recovery_node_create_unexpected(parser, param)));
14942 }
14943
14944 break;
14945 }
14946 default:
14947 if (parser->previous.type == PM_TOKEN_COMMA) {
14948 parse_parameters_handle_trailing_comma(parser, params, order, in_block, allows_trailing_comma);
14949 }
14950
14951 parsing = false;
14952 break;
14953 }
14954
14955 // If we hit some kind of issue while parsing the parameter, this would
14956 // have been set to false. In that case, we need to break out of the
14957 // loop.
14958 if (!parsing) break;
14959
14960 bool accepted_newline = false;
14961 if (uses_parentheses) {
14962 accepted_newline = accept1(parser, PM_TOKEN_NEWLINE);
14963 }
14964
14965 if (accept1(parser, PM_TOKEN_COMMA)) {
14966 // If there was a comma, but we also accepted a newline, then this
14967 // is a syntax error.
14968 if (accepted_newline) {
14969 pm_parser_err_previous(parser, PM_ERR_INVALID_COMMA);
14970 }
14971 } else {
14972 // If there was no comma, then we're done parsing parameters.
14973 break;
14974 }
14975 }
14976
14977 pm_do_loop_stack_pop(parser);
14978
14979 // If we don't have any parameters, return `NULL` instead of an empty `ParametersNode`.
14980 if (PM_NODE_START(params) == PM_NODE_END(params)) {
14981 return NULL;
14982 }
14983
14984 return params;
14985}
14986
14991static size_t
14992token_newline_index(const pm_parser_t *parser) {
14993 if (parser->heredoc_end == NULL) {
14994 // This is the common case. In this case we can look at the previously
14995 // recorded newline in the newline list and subtract from the current
14996 // offset.
14997 return parser->line_offsets.size - 1;
14998 } else {
14999 // This is unlikely. This is the case that we have already parsed the
15000 // start of a heredoc, so we cannot rely on looking at the previous
15001 // offset of the newline list, and instead must go through the whole
15002 // process of a binary search for the line number.
15003 return (size_t) pm_line_offset_list_line(&parser->line_offsets, PM_TOKEN_START(parser, &parser->current), 0);
15004 }
15005}
15006
15011static int64_t
15012token_column(const pm_parser_t *parser, size_t newline_index, const pm_token_t *token, bool break_on_non_space) {
15013 const uint8_t *cursor = parser->start + parser->line_offsets.offsets[newline_index];
15014 const uint8_t *end = token->start;
15015
15016 // Skip over the BOM if it is present.
15017 if (
15018 newline_index == 0 &&
15019 parser->start[0] == 0xef &&
15020 parser->start[1] == 0xbb &&
15021 parser->start[2] == 0xbf
15022 ) cursor += 3;
15023
15024 int64_t column = 0;
15025 for (; cursor < end; cursor++) {
15026 switch (*cursor) {
15027 case '\t':
15028 column = ((column / PM_TAB_WHITESPACE_SIZE) + 1) * PM_TAB_WHITESPACE_SIZE;
15029 break;
15030 case ' ':
15031 column++;
15032 break;
15033 default:
15034 column++;
15035 if (break_on_non_space) return -1;
15036 break;
15037 }
15038 }
15039
15040 return column;
15041}
15042
15047static void
15048parser_warn_indentation_mismatch(pm_parser_t *parser, size_t opening_newline_index, const pm_token_t *opening_token, bool if_after_else, bool allow_indent) {
15049 // If these warnings are disabled (unlikely), then we can just return.
15050 if (!parser->warn_mismatched_indentation) return;
15051
15052 // If the tokens are on the same line, we do not warn.
15053 size_t closing_newline_index = token_newline_index(parser);
15054 if (opening_newline_index == closing_newline_index) return;
15055
15056 // If the opening token has anything other than spaces or tabs before it,
15057 // then we do not warn. This is unless we are matching up an `if`/`end` pair
15058 // and the `if` immediately follows an `else` keyword.
15059 int64_t opening_column = token_column(parser, opening_newline_index, opening_token, !if_after_else);
15060 if (!if_after_else && (opening_column == -1)) return;
15061
15062 // Get a reference to the closing token off the current parser. This assumes
15063 // that the caller has placed this in the correct position.
15064 pm_token_t *closing_token = &parser->current;
15065
15066 // If the tokens are at the same indentation, we do not warn.
15067 int64_t closing_column = token_column(parser, closing_newline_index, closing_token, true);
15068 if ((closing_column == -1) || (opening_column == closing_column)) return;
15069
15070 // If the closing column is greater than the opening column and we are
15071 // allowing indentation, then we do not warn.
15072 if (allow_indent && (closing_column > opening_column)) return;
15073
15074 // Otherwise, add a warning.
15075 PM_PARSER_WARN_FORMAT(
15076 parser,
15077 PM_TOKEN_START(parser, closing_token),
15078 PM_TOKEN_LENGTH(closing_token),
15079 PM_WARN_INDENTATION_MISMATCH,
15080 (int) (closing_token->end - closing_token->start),
15081 (const char *) closing_token->start,
15082 (int) (opening_token->end - opening_token->start),
15083 (const char *) opening_token->start,
15084 ((int32_t) opening_newline_index) + parser->start_line
15085 );
15086}
15087
15088typedef enum {
15089 PM_RESCUES_BEGIN = 1,
15090 PM_RESCUES_BLOCK,
15091 PM_RESCUES_CLASS,
15092 PM_RESCUES_DEF,
15093 PM_RESCUES_LAMBDA,
15094 PM_RESCUES_MODULE,
15095 PM_RESCUES_SCLASS
15096} pm_rescues_type_t;
15097
15102static PRISM_INLINE void
15103parse_rescues(pm_parser_t *parser, size_t opening_newline_index, const pm_token_t *opening, pm_begin_node_t *parent_node, pm_rescues_type_t type, uint16_t depth) {
15104 pm_rescue_node_t *current = NULL;
15105
15106 while (match1(parser, PM_TOKEN_KEYWORD_RESCUE)) {
15107 if (opening != NULL) parser_warn_indentation_mismatch(parser, opening_newline_index, opening, false, false);
15108 parser_lex(parser);
15109
15110 pm_rescue_node_t *rescue = pm_rescue_node_create(parser, &parser->previous);
15111
15112 switch (parser->current.type) {
15113 case PM_TOKEN_EQUAL_GREATER: {
15114 // Here we have an immediate => after the rescue keyword, in which case
15115 // we're going to have an empty list of exceptions to rescue (which
15116 // implies StandardError).
15117 parser_lex(parser);
15118 pm_rescue_node_operator_set(parser, rescue, &parser->previous);
15119
15120 pm_node_t *reference = parse_expression(parser, PM_BINDING_POWER_INDEX, PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_RESCUE_VARIABLE, (uint16_t) (depth + 1));
15121 reference = parse_target(parser, reference, false, false);
15122
15123 pm_rescue_node_reference_set(rescue, reference);
15124 break;
15125 }
15126 case PM_TOKEN_NEWLINE:
15127 case PM_TOKEN_SEMICOLON:
15128 case PM_TOKEN_KEYWORD_THEN:
15129 // Here we have a terminator for the rescue keyword, in which
15130 // case we're going to just continue on.
15131 break;
15132 default: {
15133 if (token_begins_expression_p(parser->current.type) || match1(parser, PM_TOKEN_USTAR)) {
15134 // Here we have something that could be an exception expression, so
15135 // we'll attempt to parse it here and any others delimited by commas.
15136
15137 do {
15138 pm_node_t *expression = parse_starred_expression(parser, PM_BINDING_POWER_DEFINED, false, PM_ERR_RESCUE_EXPRESSION, (uint16_t) (depth + 1));
15139 pm_rescue_node_exceptions_append(parser->arena, rescue, expression);
15140
15141 // If we hit a newline, then this is the end of the rescue expression. We
15142 // can continue on to parse the statements.
15143 if (match3(parser, PM_TOKEN_NEWLINE, PM_TOKEN_SEMICOLON, PM_TOKEN_KEYWORD_THEN)) break;
15144
15145 // If we hit a `=>` then we're going to parse the exception variable. Once
15146 // we've done that, we'll break out of the loop and parse the statements.
15147 if (accept1(parser, PM_TOKEN_EQUAL_GREATER)) {
15148 pm_rescue_node_operator_set(parser, rescue, &parser->previous);
15149
15150 pm_node_t *reference = parse_expression(parser, PM_BINDING_POWER_INDEX, PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_RESCUE_VARIABLE, (uint16_t) (depth + 1));
15151 reference = parse_target(parser, reference, false, false);
15152
15153 pm_rescue_node_reference_set(rescue, reference);
15154 break;
15155 }
15156 } while (accept1(parser, PM_TOKEN_COMMA));
15157 }
15158 }
15159 }
15160
15161 if (accept2(parser, PM_TOKEN_NEWLINE, PM_TOKEN_SEMICOLON)) {
15162 if (accept1(parser, PM_TOKEN_KEYWORD_THEN)) {
15163 rescue->then_keyword_loc = TOK2LOC(parser, &parser->previous);
15164 }
15165 } else {
15166 expect1(parser, PM_TOKEN_KEYWORD_THEN, PM_ERR_RESCUE_TERM);
15167 rescue->then_keyword_loc = TOK2LOC(parser, &parser->previous);
15168 }
15169
15170 if (!match3(parser, PM_TOKEN_KEYWORD_ELSE, PM_TOKEN_KEYWORD_ENSURE, PM_TOKEN_KEYWORD_END)) {
15171 pm_accepts_block_stack_push(parser, true);
15172 pm_context_t context;
15173
15174 switch (type) {
15175 case PM_RESCUES_BEGIN: context = PM_CONTEXT_BEGIN_RESCUE; break;
15176 case PM_RESCUES_BLOCK: context = PM_CONTEXT_BLOCK_RESCUE; break;
15177 case PM_RESCUES_CLASS: context = PM_CONTEXT_CLASS_RESCUE; break;
15178 case PM_RESCUES_DEF: context = PM_CONTEXT_DEF_RESCUE; break;
15179 case PM_RESCUES_LAMBDA: context = PM_CONTEXT_LAMBDA_RESCUE; break;
15180 case PM_RESCUES_MODULE: context = PM_CONTEXT_MODULE_RESCUE; break;
15181 case PM_RESCUES_SCLASS: context = PM_CONTEXT_SCLASS_RESCUE; break;
15182 default: assert(false && "unreachable"); context = PM_CONTEXT_BEGIN_RESCUE; break;
15183 }
15184
15185 pm_statements_node_t *statements = parse_statements(parser, context, (uint16_t) (depth + 1));
15186 if (statements != NULL) pm_rescue_node_statements_set(rescue, statements);
15187
15188 pm_accepts_block_stack_pop(parser);
15189 accept2(parser, PM_TOKEN_NEWLINE, PM_TOKEN_SEMICOLON);
15190 }
15191
15192 if (current == NULL) {
15193 pm_begin_node_rescue_clause_set(parent_node, rescue);
15194 } else {
15195 pm_rescue_node_subsequent_set(current, rescue);
15196 }
15197
15198 current = rescue;
15199 }
15200
15201 // The end node locations on rescue nodes will not be set correctly
15202 // since we won't know the end until we've found all subsequent
15203 // clauses. This sets the end location on all rescues once we know it.
15204 if (current != NULL) {
15205 pm_rescue_node_t *clause = parent_node->rescue_clause;
15206
15207 while (clause != NULL) {
15208 PM_NODE_LENGTH_SET_NODE(clause, current);
15209 clause = clause->subsequent;
15210 }
15211 }
15212
15213 pm_token_t else_keyword;
15214 if (match1(parser, PM_TOKEN_KEYWORD_ELSE)) {
15215 if (opening != NULL) parser_warn_indentation_mismatch(parser, opening_newline_index, opening, false, false);
15216 opening_newline_index = token_newline_index(parser);
15217
15218 else_keyword = parser->current;
15219 opening = &else_keyword;
15220
15221 parser_lex(parser);
15222 accept2(parser, PM_TOKEN_NEWLINE, PM_TOKEN_SEMICOLON);
15223
15224 pm_statements_node_t *else_statements = NULL;
15225 if (!match2(parser, PM_TOKEN_KEYWORD_END, PM_TOKEN_KEYWORD_ENSURE)) {
15226 pm_accepts_block_stack_push(parser, true);
15227 pm_context_t context;
15228
15229 switch (type) {
15230 case PM_RESCUES_BEGIN: context = PM_CONTEXT_BEGIN_ELSE; break;
15231 case PM_RESCUES_BLOCK: context = PM_CONTEXT_BLOCK_ELSE; break;
15232 case PM_RESCUES_CLASS: context = PM_CONTEXT_CLASS_ELSE; break;
15233 case PM_RESCUES_DEF: context = PM_CONTEXT_DEF_ELSE; break;
15234 case PM_RESCUES_LAMBDA: context = PM_CONTEXT_LAMBDA_ELSE; break;
15235 case PM_RESCUES_MODULE: context = PM_CONTEXT_MODULE_ELSE; break;
15236 case PM_RESCUES_SCLASS: context = PM_CONTEXT_SCLASS_ELSE; break;
15237 default: assert(false && "unreachable"); context = PM_CONTEXT_BEGIN_ELSE; break;
15238 }
15239
15240 else_statements = parse_statements(parser, context, (uint16_t) (depth + 1));
15241 pm_accepts_block_stack_pop(parser);
15242
15243 accept2(parser, PM_TOKEN_NEWLINE, PM_TOKEN_SEMICOLON);
15244 }
15245
15246 pm_else_node_t *else_clause = pm_else_node_create(parser, &else_keyword, else_statements, &parser->current);
15247 pm_begin_node_else_clause_set(parent_node, else_clause);
15248
15249 // If we don't have a `current` rescue node, then this is a dangling
15250 // else, and it's an error.
15251 if (current == NULL) pm_parser_err_node(parser, UP(else_clause), PM_ERR_BEGIN_LONELY_ELSE);
15252 }
15253
15254 if (match1(parser, PM_TOKEN_KEYWORD_ENSURE)) {
15255 if (opening != NULL) parser_warn_indentation_mismatch(parser, opening_newline_index, opening, false, false);
15256 pm_token_t ensure_keyword = parser->current;
15257
15258 parser_lex(parser);
15259 accept2(parser, PM_TOKEN_NEWLINE, PM_TOKEN_SEMICOLON);
15260
15261 pm_statements_node_t *ensure_statements = NULL;
15262 if (!match1(parser, PM_TOKEN_KEYWORD_END)) {
15263 pm_accepts_block_stack_push(parser, true);
15264 pm_context_t context;
15265
15266 switch (type) {
15267 case PM_RESCUES_BEGIN: context = PM_CONTEXT_BEGIN_ENSURE; break;
15268 case PM_RESCUES_BLOCK: context = PM_CONTEXT_BLOCK_ENSURE; break;
15269 case PM_RESCUES_CLASS: context = PM_CONTEXT_CLASS_ENSURE; break;
15270 case PM_RESCUES_DEF: context = PM_CONTEXT_DEF_ENSURE; break;
15271 case PM_RESCUES_LAMBDA: context = PM_CONTEXT_LAMBDA_ENSURE; break;
15272 case PM_RESCUES_MODULE: context = PM_CONTEXT_MODULE_ENSURE; break;
15273 case PM_RESCUES_SCLASS: context = PM_CONTEXT_SCLASS_ENSURE; break;
15274 default: assert(false && "unreachable"); context = PM_CONTEXT_BEGIN_RESCUE; break;
15275 }
15276
15277 ensure_statements = parse_statements(parser, context, (uint16_t) (depth + 1));
15278 pm_accepts_block_stack_pop(parser);
15279
15280 accept2(parser, PM_TOKEN_NEWLINE, PM_TOKEN_SEMICOLON);
15281 }
15282
15283 pm_ensure_node_t *ensure_clause = pm_ensure_node_create(parser, &ensure_keyword, ensure_statements, &parser->current);
15284 pm_begin_node_ensure_clause_set(parent_node, ensure_clause);
15285 }
15286
15287 if (match1(parser, PM_TOKEN_KEYWORD_END)) {
15288 if (opening != NULL) parser_warn_indentation_mismatch(parser, opening_newline_index, opening, false, false);
15289 pm_begin_node_end_keyword_set(parser, parent_node, &parser->current);
15290 } else {
15291 pm_token_t end_keyword = (pm_token_t) { .type = PM_TOKEN_KEYWORD_END, .start = parser->previous.end, .end = parser->previous.end };
15292 pm_begin_node_end_keyword_set(parser, parent_node, &end_keyword);
15293 }
15294}
15295
15300static pm_begin_node_t *
15301parse_rescues_implicit_begin(pm_parser_t *parser, size_t opening_newline_index, const pm_token_t *opening, const uint8_t *start, pm_statements_node_t *statements, pm_rescues_type_t type, uint16_t depth) {
15302 pm_begin_node_t *node = pm_begin_node_create(parser, NULL, statements);
15303 parse_rescues(parser, opening_newline_index, opening, node, type, (uint16_t) (depth + 1));
15304
15305 node->base.location.start = U32(start - parser->start);
15306 PM_NODE_LENGTH_SET_TOKEN(parser, node, &parser->current);
15307
15308 return node;
15309}
15310
15315parse_block_parameters(
15316 pm_parser_t *parser,
15317 bool allows_trailing_comma,
15318 const pm_token_t *opening,
15319 bool is_lambda_literal,
15320 bool accepts_blocks_in_defaults,
15321 uint16_t depth
15322) {
15323 pm_parameters_node_t *parameters = NULL;
15324 if (!match1(parser, PM_TOKEN_SEMICOLON)) {
15325 if (!is_lambda_literal) {
15326 context_push(parser, PM_CONTEXT_BLOCK_PARAMETERS);
15327 }
15328 parameters = parse_parameters(
15329 parser,
15330 is_lambda_literal ? PM_BINDING_POWER_DEFINED : PM_BINDING_POWER_INDEX,
15331 false,
15332 allows_trailing_comma,
15333 false,
15334 accepts_blocks_in_defaults,
15335 true,
15336 is_lambda_literal ? PM_ERR_ARGUMENT_NO_FORWARDING_ELLIPSES_LAMBDA : PM_ERR_ARGUMENT_NO_FORWARDING_ELLIPSES_BLOCK,
15337 (uint16_t) (depth + 1)
15338 );
15339 if (!is_lambda_literal) {
15340 context_pop(parser);
15341 }
15342 }
15343
15344 pm_block_parameters_node_t *block_parameters = pm_block_parameters_node_create(parser, parameters, opening);
15345 if (opening != NULL) {
15346 accept1(parser, PM_TOKEN_NEWLINE);
15347
15348 if (accept1(parser, PM_TOKEN_SEMICOLON)) {
15349 do {
15350 switch (parser->current.type) {
15351 case PM_TOKEN_CONSTANT:
15352 pm_parser_err_current(parser, PM_ERR_ARGUMENT_FORMAL_CONSTANT);
15353 parser_lex(parser);
15354 break;
15355 case PM_TOKEN_INSTANCE_VARIABLE:
15356 pm_parser_err_current(parser, PM_ERR_ARGUMENT_FORMAL_IVAR);
15357 parser_lex(parser);
15358 break;
15359 case PM_TOKEN_GLOBAL_VARIABLE:
15360 pm_parser_err_current(parser, PM_ERR_ARGUMENT_FORMAL_GLOBAL);
15361 parser_lex(parser);
15362 break;
15363 case PM_TOKEN_CLASS_VARIABLE:
15364 pm_parser_err_current(parser, PM_ERR_ARGUMENT_FORMAL_CLASS);
15365 parser_lex(parser);
15366 break;
15367 default:
15368 expect1(parser, PM_TOKEN_IDENTIFIER, PM_ERR_BLOCK_PARAM_LOCAL_VARIABLE);
15369 break;
15370 }
15371
15372 bool repeated = pm_parser_parameter_name_check(parser, &parser->previous);
15373 pm_parser_local_add_token(parser, &parser->previous, 1);
15374
15375 pm_block_local_variable_node_t *local = pm_block_local_variable_node_create(parser, &parser->previous);
15376 if (repeated) pm_node_flag_set_repeated_parameter(UP(local));
15377
15378 pm_block_parameters_node_append_local(parser->arena, block_parameters, local);
15379 } while (accept1(parser, PM_TOKEN_COMMA));
15380 }
15381 }
15382
15383 return block_parameters;
15384}
15385
15390static bool
15391outer_scope_using_numbered_parameters_p(pm_parser_t *parser) {
15392 for (pm_scope_t *scope = parser->current_scope->previous; scope != NULL && !scope->closed; scope = scope->previous) {
15393 if (scope->parameters & PM_SCOPE_PARAMETERS_NUMBERED_FOUND) return true;
15394 }
15395
15396 return false;
15397}
15398
15404static const char * const pm_numbered_parameter_names[] = {
15405 "_1", "_2", "_3", "_4", "_5", "_6", "_7", "_8", "_9"
15406};
15407
15413static pm_node_t *
15414parse_blocklike_parameters(pm_parser_t *parser, pm_node_t *parameters, const pm_token_t *opening, const pm_token_t *closing) {
15415 pm_node_list_t *implicit_parameters = &parser->current_scope->implicit_parameters;
15416
15417 // If we have ordinary parameters, then we will return them as the set of
15418 // parameters.
15419 if (parameters != NULL) {
15420 // If we also have implicit parameters, then this is an error.
15421 if (implicit_parameters->size > 0) {
15422 pm_node_t *node = implicit_parameters->nodes[0];
15423
15424 if (PM_NODE_TYPE_P(node, PM_LOCAL_VARIABLE_READ_NODE)) {
15425 pm_parser_err_node(parser, node, PM_ERR_NUMBERED_PARAMETER_ORDINARY);
15426 } else if (PM_NODE_TYPE_P(node, PM_IT_LOCAL_VARIABLE_READ_NODE)) {
15427 pm_parser_err_node(parser, node, PM_ERR_IT_NOT_ALLOWED_ORDINARY);
15428 } else {
15429 assert(false && "unreachable");
15430 }
15431 }
15432
15433 return parameters;
15434 }
15435
15436 // If we don't have any implicit parameters, then the set of parameters is
15437 // NULL.
15438 if (implicit_parameters->size == 0) {
15439 return NULL;
15440 }
15441
15442 // If we don't have ordinary parameters, then we now must validate our set
15443 // of implicit parameters. We can only have numbered parameters or it, but
15444 // they cannot be mixed.
15445 uint8_t numbered_parameter = 0;
15446 bool it_parameter = false;
15447
15448 for (size_t index = 0; index < implicit_parameters->size; index++) {
15449 pm_node_t *node = implicit_parameters->nodes[index];
15450
15451 if (PM_NODE_TYPE_P(node, PM_LOCAL_VARIABLE_READ_NODE)) {
15452 if (it_parameter) {
15453 pm_parser_err_node(parser, node, PM_ERR_NUMBERED_PARAMETER_IT);
15454 } else if (outer_scope_using_numbered_parameters_p(parser)) {
15455 pm_parser_err_node(parser, node, PM_ERR_NUMBERED_PARAMETER_OUTER_BLOCK);
15456 } else if (parser->current_scope->parameters & PM_SCOPE_PARAMETERS_NUMBERED_INNER) {
15457 pm_parser_err_node(parser, node, PM_ERR_NUMBERED_PARAMETER_INNER_BLOCK);
15458 } else if (pm_token_is_numbered_parameter(parser, PM_NODE_START(node), PM_NODE_LENGTH(node))) {
15459 numbered_parameter = MAX(numbered_parameter, (uint8_t) (parser->start[node->location.start + 1] - '0'));
15460 } else {
15461 assert(false && "unreachable");
15462 }
15463 } else if (PM_NODE_TYPE_P(node, PM_IT_LOCAL_VARIABLE_READ_NODE)) {
15464 if (numbered_parameter > 0) {
15465 pm_parser_err_node(parser, node, PM_ERR_IT_NOT_ALLOWED_NUMBERED);
15466 } else {
15467 it_parameter = true;
15468 }
15469 }
15470 }
15471
15472 if (numbered_parameter > 0) {
15473 // Go through the parent scopes and mark them as being disallowed from
15474 // using numbered parameters because this inner scope is using them.
15475 for (pm_scope_t *scope = parser->current_scope->previous; scope != NULL && !scope->closed; scope = scope->previous) {
15476 scope->parameters |= PM_SCOPE_PARAMETERS_NUMBERED_INNER;
15477 }
15478 return UP(pm_numbered_parameters_node_create(parser, opening, closing, numbered_parameter));
15479 }
15480
15481 if (it_parameter) {
15482 return UP(pm_it_parameters_node_create(parser, opening, closing));
15483 }
15484
15485 return NULL;
15486}
15487
15491static pm_block_node_t *
15492parse_block(pm_parser_t *parser, uint16_t depth) {
15493 pm_token_t opening = parser->previous;
15494 accept1(parser, PM_TOKEN_NEWLINE);
15495
15496 /* A brace block is delimited by `{`/`}`, whose block-accepting frame is
15497 * managed by the lexer. A `do`/`end` block is delimited by keywords, so we
15498 * push the frame here (covering the block parameters and body) and pop it
15499 * before consuming `end`, mirroring parse.y's `do_body` rule. */
15500 bool do_block = opening.type != PM_TOKEN_BRACE_LEFT && opening.type != PM_TOKEN_BRACE_LEFT_ARGUMENT;
15501 if (do_block) pm_accepts_block_stack_push(parser, true);
15502 pm_parser_scope_push(parser, false);
15503
15504 pm_block_parameters_node_t *block_parameters = NULL;
15505
15506 if (accept1(parser, PM_TOKEN_PIPE)) {
15507 pm_token_t block_parameters_opening = parser->previous;
15508 if (match1(parser, PM_TOKEN_PIPE)) {
15509 block_parameters = pm_block_parameters_node_create(parser, NULL, &block_parameters_opening);
15510 parser->command_start = true;
15511 parser_lex(parser);
15512 } else {
15513 block_parameters = parse_block_parameters(parser, true, &block_parameters_opening, false, true, (uint16_t) (depth + 1));
15514 accept1(parser, PM_TOKEN_NEWLINE);
15515 parser->command_start = true;
15516 expect1(parser, PM_TOKEN_PIPE, PM_ERR_BLOCK_PARAM_PIPE_TERM);
15517 }
15518
15519 pm_block_parameters_node_closing_set(parser, block_parameters, &parser->previous);
15520 }
15521
15522 accept1(parser, PM_TOKEN_NEWLINE);
15523 pm_node_t *statements = NULL;
15524
15525 if (!do_block) {
15526 if (!match1(parser, PM_TOKEN_BRACE_RIGHT)) {
15527 statements = UP(parse_statements(parser, PM_CONTEXT_BLOCK_BRACES, (uint16_t) (depth + 1)));
15528 }
15529
15530 expect1_opening(parser, PM_TOKEN_BRACE_RIGHT, PM_ERR_BLOCK_TERM_BRACE, &opening);
15531 } else {
15532 if (!match1(parser, PM_TOKEN_KEYWORD_END)) {
15533 if (!match3(parser, PM_TOKEN_KEYWORD_RESCUE, PM_TOKEN_KEYWORD_ELSE, PM_TOKEN_KEYWORD_ENSURE)) {
15534 statements = UP(parse_statements(parser, PM_CONTEXT_BLOCK_KEYWORDS, (uint16_t) (depth + 1)));
15535 }
15536
15537 if (match2(parser, PM_TOKEN_KEYWORD_RESCUE, PM_TOKEN_KEYWORD_ENSURE)) {
15538 assert(statements == NULL || PM_NODE_TYPE_P(statements, PM_STATEMENTS_NODE));
15539 statements = UP(parse_rescues_implicit_begin(parser, 0, NULL, opening.start, (pm_statements_node_t *) statements, PM_RESCUES_BLOCK, (uint16_t) (depth + 1)));
15540 }
15541 }
15542
15543 /* Pop the `do`/`end` frame before consuming `end` so the token
15544 * following the block is lexed in the enclosing context. */
15545 pm_accepts_block_stack_pop(parser);
15546 expect1_opening(parser, PM_TOKEN_KEYWORD_END, PM_ERR_BLOCK_TERM_END, &opening);
15547 }
15548
15549 pm_constant_id_list_t locals;
15550 pm_locals_order(parser, &parser->current_scope->locals, &locals, pm_parser_scope_toplevel_p(parser));
15551 pm_node_t *parameters = parse_blocklike_parameters(parser, UP(block_parameters), &opening, &parser->previous);
15552
15553 pm_parser_scope_pop(parser);
15554 return pm_block_node_create(parser, &locals, &opening, parameters, statements, &parser->previous);
15555}
15556
15568static bool
15569parse_arguments_list(pm_parser_t *parser, pm_arguments_t *arguments, bool full_arguments, uint8_t flags, uint16_t depth) {
15570 /* Fast path: if the current token can't begin an expression and isn't
15571 * a parenthesis, block opener, or splat/block-pass operator, there are
15572 * no arguments to parse. */
15573 if (
15574 !token_begins_expression_p(parser->current.type) &&
15575 !match6(parser, PM_TOKEN_PARENTHESIS_LEFT, PM_TOKEN_KEYWORD_DO, PM_TOKEN_KEYWORD_DO_BLOCK, PM_TOKEN_USTAR, PM_TOKEN_USTAR_STAR, PM_TOKEN_UAMPERSAND)
15576 ) {
15577 return false;
15578 }
15579
15580 bool found = false;
15581 bool parsed_command_args = false;
15582
15583 if (accept1(parser, PM_TOKEN_PARENTHESIS_LEFT)) {
15584 found |= true;
15585 arguments->opening_loc = TOK2LOC(parser, &parser->previous);
15586
15587 if (accept1(parser, PM_TOKEN_PARENTHESIS_RIGHT)) {
15588 arguments->closing_loc = TOK2LOC(parser, &parser->previous);
15589 } else {
15590 parse_arguments(parser, arguments, full_arguments, PM_TOKEN_PARENTHESIS_RIGHT, (uint8_t) (flags & ~PM_PARSE_ACCEPTS_DO_BLOCK), (uint16_t) (depth + 1));
15591
15592 // `yield` parses its arguments through the restricted `call_args`
15593 // grammar, which (unlike the `opt_call_args` that method calls and
15594 // `super` use) permits neither a block argument nor a trailing
15595 // comma. `full_arguments` is false only for `yield`, so we use it
15596 // to reject the trailing comma in `yield(a,)` that the arguments
15597 // parser otherwise accepts before the closing parenthesis.
15598 if (!full_arguments && parser->previous.type == PM_TOKEN_COMMA) {
15599 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->previous, PM_ERR_EXPECT_ARGUMENT, pm_token_str(parser->current.type));
15600 }
15601
15602 if (!accept1(parser, PM_TOKEN_PARENTHESIS_RIGHT)) {
15603 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->current, PM_ERR_ARGUMENT_TERM_PAREN, pm_token_str(parser->current.type));
15604 parser->previous.start = parser->previous.end;
15605 parser->previous.type = 0;
15606 }
15607
15608 arguments->closing_loc = TOK2LOC(parser, &parser->previous);
15609 }
15610 } else if ((flags & PM_PARSE_ACCEPTS_COMMAND_CALL) && (token_begins_expression_p(parser->current.type) || match3(parser, PM_TOKEN_USTAR, PM_TOKEN_USTAR_STAR, PM_TOKEN_UAMPERSAND)) && !match1(parser, PM_TOKEN_BRACE_LEFT)) {
15611 found |= true;
15612 parsed_command_args = true;
15613
15614 /* The command-args frame does not accept blocks, so that a trailing
15615 * `do` binds to this command rather than to an argument. Mirroring
15616 * parse.y's `command_args` rule: when the first argument begins with an
15617 * opening delimiter, the lexer has already pushed that delimiter's
15618 * (block-accepting) frame. We must push the command-args frame beneath
15619 * it, so pop the delimiter frame, push the command-args frame, and then
15620 * restore the delimiter frame on top (the delimiter's closing token
15621 * will pop it back off during argument parsing). */
15622 bool lookahead_delimiter = match5(parser, PM_TOKEN_PARENTHESIS_LEFT, PM_TOKEN_PARENTHESIS_LEFT_GROUPING, PM_TOKEN_PARENTHESIS_LEFT_PARENTHESES, PM_TOKEN_BRACKET_LEFT, PM_TOKEN_BRACKET_LEFT_ARRAY);
15623 if (lookahead_delimiter) pm_accepts_block_stack_pop(parser);
15624 pm_accepts_block_stack_push(parser, false);
15625 if (lookahead_delimiter) pm_accepts_block_stack_push(parser, true);
15626
15627 // If we get here, then the subsequent token cannot be used as an infix
15628 // operator. In this case we assume the subsequent token is part of an
15629 // argument to this method call.
15630 parse_arguments(parser, arguments, full_arguments, PM_TOKEN_EOF, flags, (uint16_t) (depth + 1));
15631
15632 // If we have done with the arguments and still not consumed the comma,
15633 // then we have a trailing comma where we need to check whether it is
15634 // allowed or not.
15635 if (parser->previous.type == PM_TOKEN_COMMA && !match1(parser, PM_TOKEN_SEMICOLON)) {
15636 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->previous, PM_ERR_EXPECT_ARGUMENT, pm_token_str(parser->current.type));
15637 }
15638
15639 /* Symmetrically, if the command arguments are followed by a brace block
15640 * (`m args { }`), the lexer has already pushed that block's frame. Pop
15641 * it, pop the command-args frame beneath it, and restore the block
15642 * frame so the block's `}` still pops it. This mirrors the `tLBRACE_ARG`
15643 * lookahead handling in parse.y's `command_args` rule. */
15644 bool lookahead_brace = match2(parser, PM_TOKEN_BRACE_LEFT, PM_TOKEN_BRACE_LEFT_ARGUMENT);
15645 if (lookahead_brace) pm_accepts_block_stack_pop(parser);
15646 pm_accepts_block_stack_pop(parser);
15647 if (lookahead_brace) pm_accepts_block_stack_push(parser, true);
15648 }
15649
15650 // If we're at the end of the arguments, we can now check if there is a block
15651 // node that starts with a {. If there is, then we can parse it and add it to
15652 // the arguments.
15653 if (full_arguments) {
15654 pm_block_node_t *block = NULL;
15655
15656 if (accept2(parser, PM_TOKEN_BRACE_LEFT, PM_TOKEN_BRACE_LEFT_ARGUMENT)) {
15657 found |= true;
15658 block = parse_block(parser, (uint16_t) (depth + 1));
15659 pm_arguments_validate_block(parser, arguments, block);
15660 } else if (pm_accepts_block_stack_p(parser) && accept1(parser, PM_TOKEN_KEYWORD_DO)) {
15661 found |= true;
15662 block = parse_block(parser, (uint16_t) (depth + 1));
15663 } else if (parsed_command_args && pm_accepts_block_stack_p(parser) && (flags & PM_PARSE_ACCEPTS_DO_BLOCK) && accept1(parser, PM_TOKEN_KEYWORD_DO_BLOCK)) {
15664 found |= true;
15665 block = parse_block(parser, (uint16_t) (depth + 1));
15666 }
15667
15668 if (block != NULL) {
15669 if (arguments->block == NULL && !arguments->has_forwarding) {
15670 arguments->block = UP(block);
15671 } else {
15672 pm_parser_err_node(parser, UP(block), PM_ERR_ARGUMENT_BLOCK_MULTI);
15673
15674 if (arguments->block != NULL) {
15675 if (arguments->arguments == NULL) {
15676 arguments->arguments = pm_arguments_node_create(parser);
15677 }
15678 pm_arguments_node_arguments_append(parser->arena, arguments->arguments, arguments->block);
15679 }
15680 arguments->block = UP(block);
15681 }
15682 }
15683 }
15684
15685 return found;
15686}
15687
15692static void
15693parse_return(pm_parser_t *parser, pm_node_t *node) {
15694 bool in_sclass = false;
15695 for (pm_context_node_t *context_node = parser->current_context; context_node != NULL; context_node = context_node->prev) {
15696 switch (context_node->context) {
15697 case PM_CONTEXT_BEGIN_ELSE:
15698 case PM_CONTEXT_BEGIN_ENSURE:
15699 case PM_CONTEXT_BEGIN_RESCUE:
15700 case PM_CONTEXT_BEGIN:
15701 case PM_CONTEXT_CASE_IN:
15702 case PM_CONTEXT_CASE_WHEN:
15703 case PM_CONTEXT_DEFAULT_PARAMS:
15704 case PM_CONTEXT_DEFINED:
15705 case PM_CONTEXT_ELSE:
15706 case PM_CONTEXT_ELSIF:
15707 case PM_CONTEXT_EMBEXPR:
15708 case PM_CONTEXT_FOR_INDEX:
15709 case PM_CONTEXT_FOR:
15710 case PM_CONTEXT_IF:
15711 case PM_CONTEXT_LOOP_PREDICATE:
15712 case PM_CONTEXT_MAIN:
15713 case PM_CONTEXT_MULTI_TARGET:
15714 case PM_CONTEXT_PARENS:
15715 case PM_CONTEXT_POSTEXE:
15716 case PM_CONTEXT_PREDICATE:
15717 case PM_CONTEXT_PREEXE:
15718 case PM_CONTEXT_RESCUE_MODIFIER:
15719 case PM_CONTEXT_TERNARY:
15720 case PM_CONTEXT_UNLESS:
15721 case PM_CONTEXT_UNTIL:
15722 case PM_CONTEXT_WHILE:
15723 // Keep iterating up the lists of contexts, because returns can
15724 // see through these.
15725 continue;
15726 case PM_CONTEXT_SCLASS_ELSE:
15727 case PM_CONTEXT_SCLASS_ENSURE:
15728 case PM_CONTEXT_SCLASS_RESCUE:
15729 case PM_CONTEXT_SCLASS:
15730 in_sclass = true;
15731 continue;
15732 case PM_CONTEXT_CLASS_ELSE:
15733 case PM_CONTEXT_CLASS_ENSURE:
15734 case PM_CONTEXT_CLASS_RESCUE:
15735 case PM_CONTEXT_CLASS:
15736 case PM_CONTEXT_MODULE_ELSE:
15737 case PM_CONTEXT_MODULE_ENSURE:
15738 case PM_CONTEXT_MODULE_RESCUE:
15739 case PM_CONTEXT_MODULE:
15740 // These contexts are invalid for a return.
15741 pm_parser_err_node(parser, node, PM_ERR_RETURN_INVALID);
15742 return;
15743 case PM_CONTEXT_BLOCK_BRACES:
15744 case PM_CONTEXT_BLOCK_ELSE:
15745 case PM_CONTEXT_BLOCK_ENSURE:
15746 case PM_CONTEXT_BLOCK_KEYWORDS:
15747 case PM_CONTEXT_BLOCK_RESCUE:
15748 case PM_CONTEXT_BLOCK_PARAMETERS:
15749 case PM_CONTEXT_DEF_ELSE:
15750 case PM_CONTEXT_DEF_ENSURE:
15751 case PM_CONTEXT_DEF_PARAMS:
15752 case PM_CONTEXT_DEF_RESCUE:
15753 case PM_CONTEXT_DEF:
15754 case PM_CONTEXT_LAMBDA_BRACES:
15755 case PM_CONTEXT_LAMBDA_DO_END:
15756 case PM_CONTEXT_LAMBDA_ELSE:
15757 case PM_CONTEXT_LAMBDA_ENSURE:
15758 case PM_CONTEXT_LAMBDA_RESCUE:
15759 // These contexts are valid for a return, and we should not
15760 // continue to loop.
15761 return;
15762 case PM_CONTEXT_NONE:
15763 case PM_CONTEXT_MAXIMUM:
15764 // This case should never happen.
15765 assert(false && "unreachable");
15766 break;
15767 }
15768 }
15769 if (in_sclass && parser->version >= PM_OPTIONS_VERSION_CRUBY_3_4) {
15770 pm_parser_err_node(parser, node, PM_ERR_RETURN_INVALID);
15771 }
15772}
15773
15778static void
15779parse_block_exit(pm_parser_t *parser, pm_node_t *node) {
15780 for (pm_context_node_t *context_node = parser->current_context; context_node != NULL; context_node = context_node->prev) {
15781 switch (context_node->context) {
15782 case PM_CONTEXT_BLOCK_BRACES:
15783 case PM_CONTEXT_BLOCK_KEYWORDS:
15784 case PM_CONTEXT_BLOCK_ELSE:
15785 case PM_CONTEXT_BLOCK_ENSURE:
15786 case PM_CONTEXT_BLOCK_PARAMETERS:
15787 case PM_CONTEXT_BLOCK_RESCUE:
15788 case PM_CONTEXT_DEFINED:
15789 case PM_CONTEXT_FOR:
15790 case PM_CONTEXT_LAMBDA_BRACES:
15791 case PM_CONTEXT_LAMBDA_DO_END:
15792 case PM_CONTEXT_LAMBDA_ELSE:
15793 case PM_CONTEXT_LAMBDA_ENSURE:
15794 case PM_CONTEXT_LAMBDA_RESCUE:
15795 case PM_CONTEXT_LOOP_PREDICATE:
15796 case PM_CONTEXT_UNTIL:
15797 case PM_CONTEXT_WHILE:
15798 // These are the good cases. We're allowed to have a block exit
15799 // in these contexts.
15800 return;
15801 case PM_CONTEXT_POSTEXE:
15802 // https://bugs.ruby-lang.org/issues/20409
15803 if (context_node->context == PM_CONTEXT_POSTEXE) {
15804 if (parser->version < PM_OPTIONS_VERSION_CRUBY_4_1) {
15805 return;
15806 }
15807 }
15809 case PM_CONTEXT_DEF:
15810 case PM_CONTEXT_DEF_PARAMS:
15811 case PM_CONTEXT_DEF_ELSE:
15812 case PM_CONTEXT_DEF_ENSURE:
15813 case PM_CONTEXT_DEF_RESCUE:
15814 case PM_CONTEXT_MAIN:
15815 case PM_CONTEXT_PREEXE:
15816 case PM_CONTEXT_SCLASS:
15817 case PM_CONTEXT_SCLASS_ELSE:
15818 case PM_CONTEXT_SCLASS_ENSURE:
15819 case PM_CONTEXT_SCLASS_RESCUE:
15820 // These are the bad cases. We're not allowed to have a block
15821 // exit in these contexts.
15822 //
15823 // If we get here, then we're about to mark this block exit
15824 // as invalid. However, it could later _become_ valid if we
15825 // find a trailing while/until on the expression. In this
15826 // case instead of adding the error here, we'll add the
15827 // block exit to the list of exits for the expression, and
15828 // the node parsing will handle validating it instead.
15829 assert(parser->current_block_exits != NULL);
15830 pm_node_list_append(parser->arena, parser->current_block_exits, node);
15831 return;
15832 case PM_CONTEXT_BEGIN_ELSE:
15833 case PM_CONTEXT_BEGIN_ENSURE:
15834 case PM_CONTEXT_BEGIN_RESCUE:
15835 case PM_CONTEXT_BEGIN:
15836 case PM_CONTEXT_CASE_IN:
15837 case PM_CONTEXT_CASE_WHEN:
15838 case PM_CONTEXT_CLASS_ELSE:
15839 case PM_CONTEXT_CLASS_ENSURE:
15840 case PM_CONTEXT_CLASS_RESCUE:
15841 case PM_CONTEXT_CLASS:
15842 case PM_CONTEXT_DEFAULT_PARAMS:
15843 case PM_CONTEXT_ELSE:
15844 case PM_CONTEXT_ELSIF:
15845 case PM_CONTEXT_EMBEXPR:
15846 case PM_CONTEXT_FOR_INDEX:
15847 case PM_CONTEXT_IF:
15848 case PM_CONTEXT_MODULE_ELSE:
15849 case PM_CONTEXT_MODULE_ENSURE:
15850 case PM_CONTEXT_MODULE_RESCUE:
15851 case PM_CONTEXT_MODULE:
15852 case PM_CONTEXT_MULTI_TARGET:
15853 case PM_CONTEXT_PARENS:
15854 case PM_CONTEXT_PREDICATE:
15855 case PM_CONTEXT_RESCUE_MODIFIER:
15856 case PM_CONTEXT_TERNARY:
15857 case PM_CONTEXT_UNLESS:
15858 // In these contexts we should continue walking up the list of
15859 // contexts.
15860 break;
15861 case PM_CONTEXT_NONE:
15862 case PM_CONTEXT_MAXIMUM:
15863 // This case should never happen.
15864 assert(false && "unreachable");
15865 break;
15866 }
15867 }
15868}
15869
15874static pm_node_list_t *
15875push_block_exits(pm_parser_t *parser, pm_node_list_t *current_block_exits) {
15876 pm_node_list_t *previous_block_exits = parser->current_block_exits;
15877 parser->current_block_exits = current_block_exits;
15878 return previous_block_exits;
15879}
15880
15886static void
15887flush_block_exits(pm_parser_t *parser, pm_node_list_t *previous_block_exits) {
15888 pm_node_t *block_exit;
15889 PM_NODE_LIST_FOREACH(parser->current_block_exits, index, block_exit) {
15890 const char *type;
15891
15892 switch (PM_NODE_TYPE(block_exit)) {
15893 case PM_BREAK_NODE: type = "break"; break;
15894 case PM_NEXT_NODE: type = "next"; break;
15895 case PM_REDO_NODE: type = "redo"; break;
15896 default: assert(false && "unreachable"); type = ""; break;
15897 }
15898
15899 PM_PARSER_ERR_NODE_FORMAT(parser, block_exit, PM_ERR_INVALID_BLOCK_EXIT, type);
15900 }
15901
15902 parser->current_block_exits = previous_block_exits;
15903}
15904
15909static void
15910pop_block_exits(pm_parser_t *parser, pm_node_list_t *previous_block_exits) {
15911 if (match2(parser, PM_TOKEN_KEYWORD_WHILE_MODIFIER, PM_TOKEN_KEYWORD_UNTIL_MODIFIER)) {
15912 // If we matched a trailing while/until, then all of the block exits in
15913 // the contained list are valid. In this case we do not need to do
15914 // anything.
15915 parser->current_block_exits = previous_block_exits;
15916 } else if (previous_block_exits != NULL) {
15917 // If we did not matching a trailing while/until, then all of the block
15918 // exits contained in the list are invalid for this specific context.
15919 // However, they could still become valid in a higher level context if
15920 // there is another list above this one. In this case we'll push all of
15921 // the block exits up to the previous list.
15922 pm_node_list_concat(parser->arena, previous_block_exits, parser->current_block_exits);
15923 parser->current_block_exits = previous_block_exits;
15924 } else {
15925 // If we did not match a trailing while/until and this was the last
15926 // chance to do so, then all of the block exits in the list are invalid
15927 // and we need to add an error for each of them.
15928 flush_block_exits(parser, previous_block_exits);
15929 }
15930}
15931
15932static PRISM_INLINE pm_node_t *
15933parse_predicate(pm_parser_t *parser, pm_binding_power_t binding_power, pm_context_t context, pm_token_t *then_keyword, uint16_t depth) {
15934 context_push(parser, PM_CONTEXT_PREDICATE);
15935 pm_diagnostic_id_t error_id = context == PM_CONTEXT_IF ? PM_ERR_CONDITIONAL_IF_PREDICATE : PM_ERR_CONDITIONAL_UNLESS_PREDICATE;
15936 pm_node_t *predicate = parse_value_expression(parser, binding_power, PM_PARSE_ACCEPTS_COMMAND_CALL | PM_PARSE_ACCEPTS_DO_BLOCK, error_id, (uint16_t) (depth + 1));
15937
15938 // Predicates are closed by a term, a "then", or a term and then a "then".
15939 bool predicate_closed = accept2(parser, PM_TOKEN_NEWLINE, PM_TOKEN_SEMICOLON);
15940
15941 if (accept1(parser, PM_TOKEN_KEYWORD_THEN)) {
15942 predicate_closed = true;
15943 *then_keyword = parser->previous;
15944 }
15945
15946 if (!predicate_closed) {
15947 pm_parser_err_current(parser, PM_ERR_CONDITIONAL_PREDICATE_TERM);
15948 }
15949
15950 context_pop(parser);
15951 return predicate;
15952}
15953
15954static PRISM_INLINE pm_node_t *
15955parse_conditional(pm_parser_t *parser, pm_context_t context, size_t opening_newline_index, bool if_after_else, uint16_t depth) {
15956 pm_node_list_t current_block_exits = { 0 };
15957 pm_node_list_t *previous_block_exits = push_block_exits(parser, &current_block_exits);
15958
15959 pm_token_t keyword = parser->previous;
15960 pm_token_t then_keyword = { 0 };
15961
15962 pm_node_t *predicate = parse_predicate(parser, PM_BINDING_POWER_COMPOSITION, context, &then_keyword, (uint16_t) (depth + 1));
15963 pm_statements_node_t *statements = NULL;
15964
15965 if (!match3(parser, PM_TOKEN_KEYWORD_ELSIF, PM_TOKEN_KEYWORD_ELSE, PM_TOKEN_KEYWORD_END)) {
15966 pm_accepts_block_stack_push(parser, true);
15967 statements = parse_statements(parser, context, (uint16_t) (depth + 1));
15968 pm_accepts_block_stack_pop(parser);
15969 accept2(parser, PM_TOKEN_NEWLINE, PM_TOKEN_SEMICOLON);
15970 }
15971
15972 pm_node_t *parent = NULL;
15973
15974 switch (context) {
15975 case PM_CONTEXT_IF:
15976 parent = UP(pm_if_node_create(parser, &keyword, predicate, NTOK2PTR(then_keyword), statements, NULL, NULL));
15977 break;
15978 case PM_CONTEXT_UNLESS:
15979 parent = UP(pm_unless_node_create(parser, &keyword, predicate, NTOK2PTR(then_keyword), statements));
15980 break;
15981 default:
15982 assert(false && "unreachable");
15983 break;
15984 }
15985
15986 pm_node_t *current = parent;
15987
15988 // Parse any number of elsif clauses. This will form a linked list of if
15989 // nodes pointing to each other from the top.
15990 if (context == PM_CONTEXT_IF) {
15991 while (match1(parser, PM_TOKEN_KEYWORD_ELSIF)) {
15992 if (parser_end_of_line_p(parser)) {
15993 PM_PARSER_WARN_TOKEN_FORMAT_CONTENT(parser, &parser->current, PM_WARN_KEYWORD_EOL);
15994 }
15995
15996 parser_warn_indentation_mismatch(parser, opening_newline_index, &keyword, false, false);
15997 pm_token_t elsif_keyword = parser->current;
15998 parser_lex(parser);
15999
16000 pm_node_t *predicate = parse_predicate(parser, PM_BINDING_POWER_COMPOSITION, PM_CONTEXT_ELSIF, &then_keyword, (uint16_t) (depth + 1));
16001 pm_accepts_block_stack_push(parser, true);
16002
16003 pm_statements_node_t *statements = parse_statements(parser, PM_CONTEXT_ELSIF, (uint16_t) (depth + 1));
16004 pm_accepts_block_stack_pop(parser);
16005 accept2(parser, PM_TOKEN_NEWLINE, PM_TOKEN_SEMICOLON);
16006
16007 pm_node_t *elsif = UP(pm_if_node_create(parser, &elsif_keyword, predicate, NTOK2PTR(then_keyword), statements, NULL, NULL));
16008 ((pm_if_node_t *) current)->subsequent = elsif;
16009 current = elsif;
16010 }
16011 }
16012
16013 if (match1(parser, PM_TOKEN_KEYWORD_ELSE)) {
16014 parser_warn_indentation_mismatch(parser, opening_newline_index, &keyword, false, false);
16015 opening_newline_index = token_newline_index(parser);
16016
16017 parser_lex(parser);
16018 pm_token_t else_keyword = parser->previous;
16019
16020 pm_accepts_block_stack_push(parser, true);
16021 pm_statements_node_t *else_statements = parse_statements(parser, PM_CONTEXT_ELSE, (uint16_t) (depth + 1));
16022 pm_accepts_block_stack_pop(parser);
16023
16024 accept2(parser, PM_TOKEN_NEWLINE, PM_TOKEN_SEMICOLON);
16025 parser_warn_indentation_mismatch(parser, opening_newline_index, &else_keyword, false, false);
16026 expect1_opening(parser, PM_TOKEN_KEYWORD_END, PM_ERR_CONDITIONAL_TERM_ELSE, &keyword);
16027
16028 pm_else_node_t *else_node = pm_else_node_create(parser, &else_keyword, else_statements, &parser->previous);
16029
16030 switch (context) {
16031 case PM_CONTEXT_IF:
16032 ((pm_if_node_t *) current)->subsequent = UP(else_node);
16033 break;
16034 case PM_CONTEXT_UNLESS:
16035 ((pm_unless_node_t *) parent)->else_clause = else_node;
16036 break;
16037 default:
16038 assert(false && "unreachable");
16039 break;
16040 }
16041 } else {
16042 parser_warn_indentation_mismatch(parser, opening_newline_index, &keyword, if_after_else, false);
16043 expect1_opening(parser, PM_TOKEN_KEYWORD_END, PM_ERR_CONDITIONAL_TERM, &keyword);
16044 }
16045
16046 // Set the appropriate end location for all of the nodes in the subtree.
16047 switch (context) {
16048 case PM_CONTEXT_IF: {
16049 pm_node_t *current = parent;
16050 bool recursing = true;
16051
16052 while (recursing) {
16053 switch (PM_NODE_TYPE(current)) {
16054 case PM_IF_NODE:
16055 pm_if_node_end_keyword_loc_set(parser, (pm_if_node_t *) current, &parser->previous);
16056 current = ((pm_if_node_t *) current)->subsequent;
16057 recursing = current != NULL;
16058 break;
16059 case PM_ELSE_NODE:
16060 pm_else_node_end_keyword_loc_set(parser, (pm_else_node_t *) current, &parser->previous);
16061 recursing = false;
16062 break;
16063 default: {
16064 recursing = false;
16065 break;
16066 }
16067 }
16068 }
16069 break;
16070 }
16071 case PM_CONTEXT_UNLESS:
16072 pm_unless_node_end_keyword_loc_set(parser, (pm_unless_node_t *) parent, &parser->previous);
16073 break;
16074 default:
16075 assert(false && "unreachable");
16076 break;
16077 }
16078
16079 pop_block_exits(parser, previous_block_exits);
16080 return parent;
16081}
16082
16087#define PM_CASE_KEYWORD PM_TOKEN_KEYWORD___ENCODING__: case PM_TOKEN_KEYWORD___FILE__: case PM_TOKEN_KEYWORD___LINE__: \
16088 case PM_TOKEN_KEYWORD_ALIAS: case PM_TOKEN_KEYWORD_AND: case PM_TOKEN_KEYWORD_BEGIN: case PM_TOKEN_KEYWORD_BEGIN_UPCASE: \
16089 case PM_TOKEN_KEYWORD_BREAK: case PM_TOKEN_KEYWORD_CASE: case PM_TOKEN_KEYWORD_CLASS: case PM_TOKEN_KEYWORD_DEF: \
16090 case PM_TOKEN_KEYWORD_DEFINED: case PM_TOKEN_KEYWORD_DO: case PM_TOKEN_KEYWORD_DO_BLOCK: case PM_TOKEN_KEYWORD_DO_LAMBDA: case PM_TOKEN_KEYWORD_DO_LOOP: case PM_TOKEN_KEYWORD_ELSE: \
16091 case PM_TOKEN_KEYWORD_ELSIF: case PM_TOKEN_KEYWORD_END: case PM_TOKEN_KEYWORD_END_UPCASE: case PM_TOKEN_KEYWORD_ENSURE: \
16092 case PM_TOKEN_KEYWORD_FALSE: case PM_TOKEN_KEYWORD_FOR: case PM_TOKEN_KEYWORD_IF: case PM_TOKEN_KEYWORD_IN: \
16093 case PM_TOKEN_KEYWORD_MODULE: case PM_TOKEN_KEYWORD_NEXT: case PM_TOKEN_KEYWORD_NIL: case PM_TOKEN_KEYWORD_NOT: \
16094 case PM_TOKEN_KEYWORD_OR: case PM_TOKEN_KEYWORD_REDO: case PM_TOKEN_KEYWORD_RESCUE: case PM_TOKEN_KEYWORD_RETRY: \
16095 case PM_TOKEN_KEYWORD_RETURN: case PM_TOKEN_KEYWORD_SELF: case PM_TOKEN_KEYWORD_SUPER: case PM_TOKEN_KEYWORD_THEN: \
16096 case PM_TOKEN_KEYWORD_TRUE: case PM_TOKEN_KEYWORD_UNDEF: case PM_TOKEN_KEYWORD_UNLESS: case PM_TOKEN_KEYWORD_UNTIL: \
16097 case PM_TOKEN_KEYWORD_WHEN: case PM_TOKEN_KEYWORD_WHILE: case PM_TOKEN_KEYWORD_YIELD
16098
16103#define PM_CASE_OPERATOR PM_TOKEN_AMPERSAND: case PM_TOKEN_BACKTICK: case PM_TOKEN_BANG_EQUAL: \
16104 case PM_TOKEN_BANG_TILDE: case PM_TOKEN_BANG: case PM_TOKEN_BRACKET_LEFT_RIGHT_EQUAL: \
16105 case PM_TOKEN_BRACKET_LEFT_RIGHT: case PM_TOKEN_CARET: case PM_TOKEN_EQUAL_EQUAL_EQUAL: case PM_TOKEN_EQUAL_EQUAL: \
16106 case PM_TOKEN_EQUAL_TILDE: case PM_TOKEN_GREATER_EQUAL: case PM_TOKEN_GREATER_GREATER: case PM_TOKEN_GREATER: \
16107 case PM_TOKEN_LESS_EQUAL_GREATER: case PM_TOKEN_LESS_EQUAL: case PM_TOKEN_LESS_LESS: case PM_TOKEN_LESS: \
16108 case PM_TOKEN_MINUS: case PM_TOKEN_PERCENT: case PM_TOKEN_PIPE: case PM_TOKEN_PLUS: case PM_TOKEN_SLASH: \
16109 case PM_TOKEN_STAR_STAR: case PM_TOKEN_STAR: case PM_TOKEN_TILDE: case PM_TOKEN_UAMPERSAND: case PM_TOKEN_UMINUS: \
16110 case PM_TOKEN_UMINUS_NUM: case PM_TOKEN_UPLUS: case PM_TOKEN_USTAR: case PM_TOKEN_USTAR_STAR
16111
16117#define PM_CASE_PRIMITIVE PM_TOKEN_INTEGER: case PM_TOKEN_INTEGER_IMAGINARY: case PM_TOKEN_INTEGER_RATIONAL: \
16118 case PM_TOKEN_INTEGER_RATIONAL_IMAGINARY: case PM_TOKEN_FLOAT: case PM_TOKEN_FLOAT_IMAGINARY: \
16119 case PM_TOKEN_FLOAT_RATIONAL: case PM_TOKEN_FLOAT_RATIONAL_IMAGINARY: case PM_TOKEN_SYMBOL_BEGIN: \
16120 case PM_TOKEN_REGEXP_BEGIN: case PM_TOKEN_XSTRING_BEGIN: case PM_TOKEN_PERCENT_LOWER_X: case PM_TOKEN_PERCENT_LOWER_I: \
16121 case PM_TOKEN_PERCENT_LOWER_W: case PM_TOKEN_PERCENT_UPPER_I: case PM_TOKEN_PERCENT_UPPER_W: \
16122 case PM_TOKEN_STRING_BEGIN: case PM_TOKEN_KEYWORD_NIL: case PM_TOKEN_KEYWORD_SELF: case PM_TOKEN_KEYWORD_TRUE: \
16123 case PM_TOKEN_KEYWORD_FALSE: case PM_TOKEN_KEYWORD___FILE__: case PM_TOKEN_KEYWORD___LINE__: \
16124 case PM_TOKEN_KEYWORD___ENCODING__: case PM_TOKEN_MINUS_GREATER: case PM_TOKEN_HEREDOC_START: \
16125 case PM_TOKEN_UMINUS_NUM: case PM_TOKEN_CHARACTER_LITERAL
16126
16131#define PM_CASE_PARAMETER PM_TOKEN_UAMPERSAND: case PM_TOKEN_AMPERSAND: case PM_TOKEN_UDOT_DOT_DOT: \
16132 case PM_TOKEN_IDENTIFIER: case PM_TOKEN_LABEL: case PM_TOKEN_USTAR: case PM_TOKEN_STAR: case PM_TOKEN_STAR_STAR: \
16133 case PM_TOKEN_USTAR_STAR: case PM_TOKEN_CONSTANT: case PM_TOKEN_INSTANCE_VARIABLE: case PM_TOKEN_GLOBAL_VARIABLE: \
16134 case PM_TOKEN_CLASS_VARIABLE
16135
16140#define PM_CASE_WRITABLE PM_CLASS_VARIABLE_READ_NODE: case PM_CONSTANT_PATH_NODE: \
16141 case PM_CONSTANT_READ_NODE: case PM_GLOBAL_VARIABLE_READ_NODE: case PM_LOCAL_VARIABLE_READ_NODE: \
16142 case PM_INSTANCE_VARIABLE_READ_NODE: case PM_MULTI_TARGET_NODE: case PM_BACK_REFERENCE_READ_NODE: \
16143 case PM_NUMBERED_REFERENCE_READ_NODE: case PM_IT_LOCAL_VARIABLE_READ_NODE
16144
16145// Assert here that the flags are the same so that we can safely switch the type
16146// of the node without having to move the flags.
16147PM_STATIC_ASSERT(__LINE__, ((int) PM_STRING_FLAGS_FORCED_UTF8_ENCODING) == ((int) PM_ENCODING_FLAGS_FORCED_UTF8_ENCODING), "Expected the flags to match.");
16148
16153static PRISM_INLINE pm_node_flags_t
16154parse_unescaped_encoding(const pm_parser_t *parser, const pm_encoding_t *explicit_encoding) {
16155 if (explicit_encoding != NULL) {
16156 if (explicit_encoding == PM_ENCODING_UTF_8_ENTRY) {
16157 // If the there's an explicit encoding and it's using a UTF-8 escape
16158 // sequence, then mark the string as UTF-8.
16159 return PM_STRING_FLAGS_FORCED_UTF8_ENCODING;
16160 } else if (parser->encoding == PM_ENCODING_US_ASCII_ENTRY) {
16161 // If there's a non-UTF-8 escape sequence being used, then the
16162 // string uses the source encoding, unless the source is marked as
16163 // US-ASCII. In that case the string is forced as ASCII-8BIT in
16164 // order to keep the string valid.
16165 return PM_STRING_FLAGS_FORCED_BINARY_ENCODING;
16166 }
16167 }
16168 return 0;
16169}
16170
16175static pm_node_t *
16176parse_string_part(pm_parser_t *parser, uint16_t depth) {
16177 switch (parser->current.type) {
16178 // Here the lexer has returned to us plain string content. In this case
16179 // we'll create a string node that has no opening or closing and return that
16180 // as the part. These kinds of parts look like:
16181 //
16182 // "aaa #{bbb} #@ccc ddd"
16183 // ^^^^ ^ ^^^^
16184 case PM_TOKEN_STRING_CONTENT: {
16185 pm_node_t *node = UP(pm_string_node_create_current_string(parser, NULL, &parser->current, NULL));
16186 pm_node_flag_set(node, parse_unescaped_encoding(parser, parser->explicit_encoding));
16187
16188 parser_lex(parser);
16189 return node;
16190 }
16191 // Here the lexer has returned the beginning of an embedded expression. In
16192 // that case we'll parse the inner statements and return that as the part.
16193 // These kinds of parts look like:
16194 //
16195 // "aaa #{bbb} #@ccc ddd"
16196 // ^^^^^^
16197 case PM_TOKEN_EMBEXPR_BEGIN: {
16198 // Ruby disallows seeing encoding around interpolation in strings,
16199 // even though it is known at parse time.
16200 parser->explicit_encoding = NULL;
16201
16202 pm_lex_state_t state = parser->lex_state;
16203 int brace_nesting = parser->brace_nesting;
16204
16205 parser->brace_nesting = 0;
16206 lex_state_set(parser, PM_LEX_STATE_BEG);
16207 parser_lex(parser);
16208
16209 pm_token_t opening = parser->previous;
16210 pm_statements_node_t *statements = NULL;
16211
16212 if (!match3(parser, PM_TOKEN_EMBEXPR_END, PM_TOKEN_HEREDOC_END, PM_TOKEN_EOF)) {
16213 statements = parse_statements(parser, PM_CONTEXT_EMBEXPR, (uint16_t) (depth + 1));
16214 }
16215
16216 parser->brace_nesting = brace_nesting;
16217 lex_state_set(parser, state);
16218 expect1(parser, PM_TOKEN_EMBEXPR_END, PM_ERR_EMBEXPR_END);
16219
16220 // If this set of embedded statements only contains a single
16221 // statement, then Ruby does not consider it as a possible statement
16222 // that could emit a line event.
16223 if (statements != NULL && statements->body.size == 1) {
16224 pm_node_flag_unset(statements->body.nodes[0], PM_NODE_FLAG_NEWLINE);
16225 }
16226
16227 return UP(pm_embedded_statements_node_create(parser, &opening, statements, &parser->previous));
16228 }
16229
16230 // Here the lexer has returned the beginning of an embedded variable.
16231 // In that case we'll parse the variable and create an appropriate node
16232 // for it and then return that node. These kinds of parts look like:
16233 //
16234 // "aaa #{bbb} #@ccc ddd"
16235 // ^^^^^
16236 case PM_TOKEN_EMBVAR: {
16237 // Ruby disallows seeing encoding around interpolation in strings,
16238 // even though it is known at parse time.
16239 parser->explicit_encoding = NULL;
16240
16241 lex_state_set(parser, PM_LEX_STATE_BEG);
16242 parser_lex(parser);
16243
16244 pm_token_t operator = parser->previous;
16245 pm_node_t *variable;
16246
16247 switch (parser->current.type) {
16248 // In this case a back reference is being interpolated. We'll
16249 // create a global variable read node.
16250 case PM_TOKEN_BACK_REFERENCE:
16251 parser_lex(parser);
16252 variable = UP(pm_back_reference_read_node_create(parser, &parser->previous));
16253 break;
16254 // In this case an nth reference is being interpolated. We'll
16255 // create a global variable read node.
16256 case PM_TOKEN_NUMBERED_REFERENCE:
16257 parser_lex(parser);
16258 variable = UP(pm_numbered_reference_read_node_create(parser, &parser->previous));
16259 break;
16260 // In this case a global variable is being interpolated. We'll
16261 // create a global variable read node.
16262 case PM_TOKEN_GLOBAL_VARIABLE:
16263 parser_lex(parser);
16264 variable = UP(pm_global_variable_read_node_create(parser, &parser->previous));
16265 break;
16266 // In this case an instance variable is being interpolated.
16267 // We'll create an instance variable read node.
16268 case PM_TOKEN_INSTANCE_VARIABLE:
16269 parser_lex(parser);
16270 variable = UP(pm_instance_variable_read_node_create(parser, &parser->previous));
16271 break;
16272 // In this case a class variable is being interpolated. We'll
16273 // create a class variable read node.
16274 case PM_TOKEN_CLASS_VARIABLE:
16275 parser_lex(parser);
16276 variable = UP(pm_class_variable_read_node_create(parser, &parser->previous));
16277 break;
16278 // We can hit here if we got an invalid token. In that case
16279 // we'll not attempt to lex this token and instead just return a
16280 // missing node.
16281 default:
16282 expect1(parser, PM_TOKEN_IDENTIFIER, PM_ERR_EMBVAR_INVALID);
16283 variable = UP(pm_error_recovery_node_create(parser, PM_TOKEN_START(parser, &parser->current), PM_TOKEN_LENGTH(&parser->current)));
16284 break;
16285 }
16286
16287 return UP(pm_embedded_variable_node_create(parser, &operator, variable));
16288 }
16289 default:
16290 parser_lex(parser);
16291 pm_parser_err_previous(parser, PM_ERR_CANNOT_PARSE_STRING_PART);
16292 return NULL;
16293 }
16294}
16295
16301static const uint8_t *
16302parse_operator_symbol_name(const pm_token_t *name) {
16303 switch (name->type) {
16304 case PM_TOKEN_TILDE:
16305 case PM_TOKEN_BANG:
16306 if (name->end[-1] == '@') return name->end - 1;
16308 default:
16309 return name->end;
16310 }
16311}
16312
16313static pm_node_t *
16314parse_operator_symbol(pm_parser_t *parser, const pm_token_t *opening, pm_lex_state_t next_state) {
16315 pm_symbol_node_t *symbol = pm_symbol_node_create(parser, opening, &parser->current, NULL);
16316 const uint8_t *end = parse_operator_symbol_name(&parser->current);
16317
16318 if (next_state != PM_LEX_STATE_NONE) lex_state_set(parser, next_state);
16319 parser_lex(parser);
16320
16321 pm_string_shared_init(&symbol->unescaped, parser->previous.start, end);
16322 pm_node_flag_set(UP(symbol), PM_SYMBOL_FLAGS_FORCED_US_ASCII_ENCODING);
16323
16324 return UP(symbol);
16325}
16326
16332static pm_node_t *
16333parse_symbol(pm_parser_t *parser, pm_lex_mode_t *lex_mode, pm_lex_state_t next_state, uint16_t depth) {
16334 const pm_token_t opening = parser->previous;
16335
16336 if (lex_mode->mode != PM_LEX_STRING) {
16337 if (next_state != PM_LEX_STATE_NONE) lex_state_set(parser, next_state);
16338
16339 switch (parser->current.type) {
16340 case PM_CASE_OPERATOR:
16341 return parse_operator_symbol(parser, &opening, next_state == PM_LEX_STATE_NONE ? PM_LEX_STATE_ENDFN : next_state);
16342 case PM_TOKEN_IDENTIFIER:
16343 case PM_TOKEN_CONSTANT:
16344 case PM_TOKEN_INSTANCE_VARIABLE:
16345 case PM_TOKEN_METHOD_NAME:
16346 case PM_TOKEN_CLASS_VARIABLE:
16347 case PM_TOKEN_GLOBAL_VARIABLE:
16348 case PM_TOKEN_NUMBERED_REFERENCE:
16349 case PM_TOKEN_BACK_REFERENCE:
16350 case PM_CASE_KEYWORD:
16351 parser_lex(parser);
16352 break;
16353 default:
16354 expect2(parser, PM_TOKEN_IDENTIFIER, PM_TOKEN_METHOD_NAME, PM_ERR_SYMBOL_INVALID);
16355 break;
16356 }
16357
16358 pm_symbol_node_t *symbol = pm_symbol_node_create(parser, &opening, &parser->previous, NULL);
16359 pm_string_shared_init(&symbol->unescaped, parser->previous.start, parser->previous.end);
16360 pm_node_flag_set(UP(symbol), parse_symbol_encoding(parser, parser->explicit_encoding, &parser->previous, &symbol->unescaped, false));
16361
16362 return UP(symbol);
16363 }
16364
16365 if (lex_mode->as.string.interpolation) {
16366 // If we have the end of the symbol, then we can return an empty symbol.
16367 if (match1(parser, PM_TOKEN_STRING_END)) {
16368 if (next_state != PM_LEX_STATE_NONE) lex_state_set(parser, next_state);
16369 parser_lex(parser);
16370 pm_token_t content = {
16371 .type = PM_TOKEN_STRING_CONTENT,
16372 .start = parser->previous.start,
16373 .end = parser->previous.start
16374 };
16375
16376 return UP(pm_symbol_node_create(parser, &opening, &content, &parser->previous));
16377 }
16378
16379 // Now we can parse the first part of the symbol.
16380 pm_node_t *part = parse_string_part(parser, (uint16_t) (depth + 1));
16381
16382 // If we got a string part, then it's possible that we could transform
16383 // what looks like an interpolated symbol into a regular symbol.
16384 if (part && PM_NODE_TYPE_P(part, PM_STRING_NODE) && match2(parser, PM_TOKEN_STRING_END, PM_TOKEN_EOF)) {
16385 if (next_state != PM_LEX_STATE_NONE) lex_state_set(parser, next_state);
16386 expect1(parser, PM_TOKEN_STRING_END, PM_ERR_SYMBOL_TERM_INTERPOLATED);
16387
16388 return UP(pm_string_node_to_symbol_node(parser, (pm_string_node_t *) part, &opening, &parser->previous));
16389 }
16390
16391 pm_interpolated_symbol_node_t *symbol = pm_interpolated_symbol_node_create(parser, &opening, NULL, &opening);
16392 if (part) pm_interpolated_symbol_node_append(parser->arena, symbol, part);
16393
16394 while (!match2(parser, PM_TOKEN_STRING_END, PM_TOKEN_EOF)) {
16395 if ((part = parse_string_part(parser, (uint16_t) (depth + 1))) != NULL) {
16396 pm_interpolated_symbol_node_append(parser->arena, symbol, part);
16397 }
16398 }
16399
16400 if (next_state != PM_LEX_STATE_NONE) lex_state_set(parser, next_state);
16401 if (match1(parser, PM_TOKEN_EOF)) {
16402 pm_parser_err_token(parser, &opening, PM_ERR_SYMBOL_TERM_INTERPOLATED);
16403 } else {
16404 expect1(parser, PM_TOKEN_STRING_END, PM_ERR_SYMBOL_TERM_INTERPOLATED);
16405 }
16406
16407 pm_interpolated_symbol_node_closing_loc_set(parser, symbol, &parser->previous);
16408 return UP(symbol);
16409 }
16410
16411 pm_token_t content;
16412 pm_string_t unescaped;
16413
16414 if (match1(parser, PM_TOKEN_STRING_CONTENT)) {
16415 content = parser->current;
16416 unescaped = parser->current_string;
16417 parser_lex(parser);
16418
16419 // If we have two string contents in a row, then the content of this
16420 // symbol is split because of heredoc contents. This looks like:
16421 //
16422 // <<A; :'a
16423 // A
16424 // b'
16425 //
16426 // In this case, the best way we have to represent this is as an
16427 // interpolated string node, so that's what we'll do here.
16428 if (match1(parser, PM_TOKEN_STRING_CONTENT)) {
16429 pm_interpolated_symbol_node_t *symbol = pm_interpolated_symbol_node_create(parser, &opening, NULL, &opening);
16430 pm_node_t *part = UP(pm_string_node_create_unescaped(parser, NULL, &content, NULL, &unescaped));
16431 pm_interpolated_symbol_node_append(parser->arena, symbol, part);
16432
16433 part = UP(pm_string_node_create_unescaped(parser, NULL, &parser->current, NULL, &parser->current_string));
16434 pm_interpolated_symbol_node_append(parser->arena, symbol, part);
16435
16436 if (next_state != PM_LEX_STATE_NONE) {
16437 lex_state_set(parser, next_state);
16438 }
16439
16440 parser_lex(parser);
16441 expect1(parser, PM_TOKEN_STRING_END, PM_ERR_SYMBOL_TERM_DYNAMIC);
16442
16443 pm_interpolated_symbol_node_closing_loc_set(parser, symbol, &parser->previous);
16444 return UP(symbol);
16445 }
16446 } else {
16447 content = (pm_token_t) { .type = PM_TOKEN_STRING_CONTENT, .start = parser->previous.end, .end = parser->previous.end };
16448 pm_string_shared_init(&unescaped, content.start, content.end);
16449 }
16450
16451 if (next_state != PM_LEX_STATE_NONE) {
16452 lex_state_set(parser, next_state);
16453 }
16454
16455 if (match1(parser, PM_TOKEN_EOF)) {
16456 pm_parser_err_token(parser, &opening, PM_ERR_SYMBOL_TERM_DYNAMIC);
16457 } else {
16458 expect1(parser, PM_TOKEN_STRING_END, PM_ERR_SYMBOL_TERM_DYNAMIC);
16459 }
16460
16461 return UP(pm_symbol_node_create_unescaped(parser, &opening, &content, &parser->previous, &unescaped, parse_symbol_encoding(parser, parser->explicit_encoding, &content, &unescaped, false)));
16462}
16463
16468static PRISM_INLINE pm_node_t *
16469parse_undef_argument(pm_parser_t *parser, uint16_t depth) {
16470 switch (parser->current.type) {
16471 case PM_CASE_OPERATOR:
16472 return parse_operator_symbol(parser, NULL, PM_LEX_STATE_NONE);
16473 case PM_CASE_KEYWORD:
16474 case PM_TOKEN_CONSTANT:
16475 case PM_TOKEN_IDENTIFIER:
16476 case PM_TOKEN_METHOD_NAME: {
16477 parser_lex(parser);
16478
16479 pm_symbol_node_t *symbol = pm_symbol_node_create(parser, NULL, &parser->previous, NULL);
16480 pm_string_shared_init(&symbol->unescaped, parser->previous.start, parser->previous.end);
16481 pm_node_flag_set(UP(symbol), parse_symbol_encoding(parser, parser->explicit_encoding, &parser->previous, &symbol->unescaped, false));
16482
16483 return UP(symbol);
16484 }
16485 case PM_TOKEN_SYMBOL_BEGIN: {
16486 pm_lex_mode_t lex_mode = *parser->lex_modes.current;
16487 parser_lex(parser);
16488
16489 return parse_symbol(parser, &lex_mode, PM_LEX_STATE_NONE, (uint16_t) (depth + 1));
16490 }
16491 default:
16492 pm_parser_err_current(parser, PM_ERR_UNDEF_ARGUMENT);
16493 return UP(pm_error_recovery_node_create(parser, PM_TOKEN_START(parser, &parser->current), PM_TOKEN_LENGTH(&parser->current)));
16494 }
16495}
16496
16503static PRISM_INLINE pm_node_t *
16504parse_alias_argument(pm_parser_t *parser, bool first, uint16_t depth) {
16505 switch (parser->current.type) {
16506 case PM_CASE_OPERATOR:
16507 return parse_operator_symbol(parser, NULL, first ? PM_LEX_STATE_FNAME | PM_LEX_STATE_FITEM : PM_LEX_STATE_NONE);
16508 case PM_CASE_KEYWORD:
16509 case PM_TOKEN_CONSTANT:
16510 case PM_TOKEN_IDENTIFIER:
16511 case PM_TOKEN_METHOD_NAME: {
16512 if (first) lex_state_set(parser, PM_LEX_STATE_FNAME | PM_LEX_STATE_FITEM);
16513 parser_lex(parser);
16514
16515 pm_symbol_node_t *symbol = pm_symbol_node_create(parser, NULL, &parser->previous, NULL);
16516 pm_string_shared_init(&symbol->unescaped, parser->previous.start, parser->previous.end);
16517 pm_node_flag_set(UP(symbol), parse_symbol_encoding(parser, parser->explicit_encoding, &parser->previous, &symbol->unescaped, false));
16518
16519 return UP(symbol);
16520 }
16521 case PM_TOKEN_SYMBOL_BEGIN: {
16522 pm_lex_mode_t lex_mode = *parser->lex_modes.current;
16523 parser_lex(parser);
16524
16525 return parse_symbol(parser, &lex_mode, first ? PM_LEX_STATE_FNAME | PM_LEX_STATE_FITEM : PM_LEX_STATE_NONE, (uint16_t) (depth + 1));
16526 }
16527 case PM_TOKEN_BACK_REFERENCE:
16528 parser_lex(parser);
16529 return UP(pm_back_reference_read_node_create(parser, &parser->previous));
16530 case PM_TOKEN_NUMBERED_REFERENCE:
16531 parser_lex(parser);
16532 return UP(pm_numbered_reference_read_node_create(parser, &parser->previous));
16533 case PM_TOKEN_GLOBAL_VARIABLE:
16534 parser_lex(parser);
16535 return UP(pm_global_variable_read_node_create(parser, &parser->previous));
16536 default:
16537 pm_parser_err_current(parser, PM_ERR_ALIAS_ARGUMENT);
16538 return UP(pm_error_recovery_node_create(parser, PM_TOKEN_START(parser, &parser->current), PM_TOKEN_LENGTH(&parser->current)));
16539 }
16540}
16541
16546static pm_node_t *
16547parse_variable(pm_parser_t *parser) {
16548 pm_constant_id_t name_id = pm_parser_constant_id_token(parser, &parser->previous);
16549 int depth;
16550 bool is_numbered_param = pm_token_is_numbered_parameter(parser, PM_TOKEN_START(parser, &parser->previous), PM_TOKEN_LENGTH(&parser->previous));
16551
16552 if (!is_numbered_param && ((depth = pm_parser_local_depth_constant_id(parser, name_id)) != -1)) {
16553 return UP(pm_local_variable_read_node_create_constant_id(parser, &parser->previous, name_id, (uint32_t) depth, false));
16554 }
16555
16556 pm_scope_t *current_scope = parser->current_scope;
16557 if (!current_scope->closed && !(current_scope->parameters & PM_SCOPE_PARAMETERS_IMPLICIT_DISALLOWED)) {
16558 if (is_numbered_param) {
16559 // When you use a numbered parameter, it implies the existence of
16560 // all of the locals that exist before it. For example, referencing
16561 // _2 means that _1 must exist. Therefore here we loop through all
16562 // of the possibilities and add them into the constant pool.
16563 uint8_t maximum = (uint8_t) (parser->previous.start[1] - '0');
16564 for (uint8_t number = 1; number <= maximum; number++) {
16565 pm_parser_local_add_constant(parser, pm_numbered_parameter_names[number - 1], 2);
16566 }
16567
16568 if (!match1(parser, PM_TOKEN_EQUAL)) {
16569 parser->current_scope->parameters |= PM_SCOPE_PARAMETERS_NUMBERED_FOUND;
16570 }
16571
16572 pm_node_t *node = UP(pm_local_variable_read_node_create_constant_id(parser, &parser->previous, name_id, 0, false));
16573 pm_node_list_append(parser->arena, &current_scope->implicit_parameters, node);
16574
16575 return node;
16576 } else if ((parser->version >= PM_OPTIONS_VERSION_CRUBY_3_4) && pm_token_is_it(parser->previous.start, parser->previous.end)) {
16577 pm_node_t *node = UP(pm_it_local_variable_read_node_create(parser, &parser->previous));
16578 pm_node_list_append(parser->arena, &current_scope->implicit_parameters, node);
16579
16580 return node;
16581 }
16582 }
16583
16584 return NULL;
16585}
16586
16590static pm_node_t *
16591parse_variable_call(pm_parser_t *parser) {
16592 pm_node_flags_t flags = 0;
16593
16594 if (!match1(parser, PM_TOKEN_PARENTHESIS_LEFT) && (parser->previous.end[-1] != '!') && (parser->previous.end[-1] != '?')) {
16595 pm_node_t *node = parse_variable(parser);
16596 if (node != NULL) return node;
16597 flags |= PM_CALL_NODE_FLAGS_VARIABLE_CALL;
16598 }
16599
16600 pm_call_node_t *node = pm_call_node_variable_call_create(parser, &parser->previous);
16601 pm_node_flag_set(UP(node), flags);
16602
16603 return UP(node);
16604}
16605
16612parse_method_definition_name(pm_parser_t *parser) {
16613 switch (parser->current.type) {
16614 case PM_CASE_KEYWORD:
16615 case PM_TOKEN_CONSTANT:
16616 case PM_TOKEN_METHOD_NAME:
16617 parser_lex(parser);
16618 return parser->previous;
16619 case PM_TOKEN_IDENTIFIER:
16620 pm_refute_numbered_parameter(parser, PM_TOKEN_START(parser, &parser->current), PM_TOKEN_LENGTH(&parser->current));
16621 parser_lex(parser);
16622 return parser->previous;
16623 case PM_CASE_OPERATOR:
16624 lex_state_set(parser, PM_LEX_STATE_ENDFN);
16625 parser_lex(parser);
16626 return parser->previous;
16627 default:
16628 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->current, PM_ERR_DEF_NAME, pm_token_str(parser->current.type));
16629 return (pm_token_t) { .type = 0, .start = parser->current.start, .end = parser->current.end };
16630 }
16631}
16632
16633static void
16634parse_heredoc_dedent_string(pm_arena_t *arena, pm_string_t *string, size_t common_whitespace) {
16635 // Make a writable copy in the arena if the string isn't already writable.
16636 // We keep a mutable pointer to the arena memory so we can memmove into it
16637 // below without casting away const from the string's source field.
16638 uint8_t *writable;
16639
16640 if (string->type != PM_STRING_OWNED) {
16641 size_t length = pm_string_length(string);
16642 writable = (uint8_t *) pm_arena_memdup(arena, pm_string_source(string), length, PRISM_ALIGNOF(uint8_t));
16643 pm_string_constant_init(string, (const char *) writable, length);
16644 } else {
16645 writable = (uint8_t *) string->source;
16646 }
16647
16648 // Now get the bounds of the existing string. We'll use this as a
16649 // destination to move bytes into. We'll also use it for bounds checking
16650 // since we don't require that these strings be null terminated.
16651 size_t dest_length = pm_string_length(string);
16652 const uint8_t *source_cursor = writable;
16653 const uint8_t *source_end = source_cursor + dest_length;
16654
16655 // We're going to move bytes backward in the string when we get leading
16656 // whitespace, so we'll maintain a pointer to the current position in the
16657 // string that we're writing to.
16658 size_t trimmed_whitespace = 0;
16659
16660 // While we haven't reached the amount of common whitespace that we need to
16661 // trim and we haven't reached the end of the string, we'll keep trimming
16662 // whitespace. Trimming in this context means skipping over these bytes such
16663 // that they aren't copied into the new string.
16664 while ((source_cursor < source_end) && pm_char_is_inline_whitespace(*source_cursor) && trimmed_whitespace < common_whitespace) {
16665 if (*source_cursor == '\t') {
16666 trimmed_whitespace = (trimmed_whitespace / PM_TAB_WHITESPACE_SIZE + 1) * PM_TAB_WHITESPACE_SIZE;
16667 if (trimmed_whitespace > common_whitespace) break;
16668 } else {
16669 trimmed_whitespace++;
16670 }
16671
16672 source_cursor++;
16673 dest_length--;
16674 }
16675
16676 memmove(writable, source_cursor, (size_t) (source_end - source_cursor));
16677 string->length = dest_length;
16678}
16679
16684static PRISM_INLINE bool
16685heredoc_dedent_discard_string_node(pm_parser_t *parser, pm_string_node_t *string_node) {
16686 if (string_node->unescaped.length == 0) {
16687 const uint8_t *cursor = parser->start + PM_LOCATION_START(&string_node->content_loc);
16688 return pm_memchr(cursor, '\\', string_node->content_loc.length, parser->encoding_changed, parser->encoding) == NULL;
16689 }
16690 return false;
16691}
16692
16696static void
16697parse_heredoc_dedent(pm_parser_t *parser, pm_node_list_t *nodes, size_t common_whitespace) {
16698 // The next node should be dedented if it's the first node in the list or if
16699 // it follows a string node.
16700 bool dedent_next = true;
16701
16702 // Iterate over all nodes, and trim whitespace accordingly. We're going to
16703 // keep around two indices: a read and a write.
16704 size_t write_index = 0;
16705
16706 pm_node_t *node;
16707 PM_NODE_LIST_FOREACH(nodes, read_index, node) {
16708 // We're not manipulating child nodes that aren't strings. In this case
16709 // we'll skip past it and indicate that the subsequent node should not
16710 // be dedented.
16711 if (!PM_NODE_TYPE_P(node, PM_STRING_NODE)) {
16712 nodes->nodes[write_index++] = node;
16713 dedent_next = false;
16714 continue;
16715 }
16716
16717 pm_string_node_t *string_node = ((pm_string_node_t *) node);
16718 if (dedent_next) {
16719 parse_heredoc_dedent_string(parser->arena, &string_node->unescaped, common_whitespace);
16720 }
16721
16722 if (heredoc_dedent_discard_string_node(parser, string_node)) {
16723 } else {
16724 nodes->nodes[write_index++] = node;
16725 }
16726
16727 // We always dedent the next node if it follows a string node.
16728 dedent_next = true;
16729 }
16730
16731 nodes->size = write_index;
16732}
16733
16737static pm_token_t
16738parse_strings_empty_content(const uint8_t *location) {
16739 return (pm_token_t) { .type = PM_TOKEN_STRING_CONTENT, .start = location, .end = location };
16740}
16741
16745static PRISM_INLINE pm_node_t *
16746parse_strings(pm_parser_t *parser, pm_node_t *current, bool accepts_label, uint16_t depth) {
16747 assert(parser->current.type == PM_TOKEN_STRING_BEGIN);
16748 bool concating = false;
16749
16750 while (match1(parser, PM_TOKEN_STRING_BEGIN)) {
16751 pm_node_t *node = NULL;
16752
16753 // Here we have found a string literal. We'll parse it and add it to
16754 // the list of strings.
16755 const pm_lex_mode_t *lex_mode = parser->lex_modes.current;
16756 assert(lex_mode->mode == PM_LEX_STRING);
16757 bool lex_interpolation = lex_mode->as.string.interpolation;
16758 bool label_allowed = lex_mode->as.string.label_allowed && accepts_label;
16759
16760 pm_token_t opening = parser->current;
16761 parser_lex(parser);
16762
16763 if (match2(parser, PM_TOKEN_STRING_END, PM_TOKEN_EOF)) {
16764 expect1(parser, PM_TOKEN_STRING_END, PM_ERR_STRING_LITERAL_EOF);
16765 // If we get here, then we have an end immediately after a
16766 // start. In that case we'll create an empty content token and
16767 // return an uninterpolated string.
16768 pm_token_t content = parse_strings_empty_content(parser->previous.start);
16769 pm_string_node_t *string = pm_string_node_create(parser, &opening, &content, &parser->previous);
16770
16771 pm_string_shared_init(&string->unescaped, content.start, content.end);
16772 node = UP(string);
16773 } else if (accept1(parser, PM_TOKEN_LABEL_END)) {
16774 // If we get here, then we have an end of a label immediately
16775 // after a start. In that case we'll create an empty symbol
16776 // node.
16777 pm_symbol_node_t *symbol = pm_symbol_node_create(parser, &opening, NULL, &parser->previous);
16778 pm_string_shared_init(&symbol->unescaped, parser->previous.start, parser->previous.start);
16779 node = UP(symbol);
16780
16781 if (!label_allowed) pm_parser_err_node(parser, node, PM_ERR_UNEXPECTED_LABEL);
16782 } else if (!lex_interpolation) {
16783 // If we don't accept interpolation then we expect the string to
16784 // start with a single string content node.
16785 pm_string_t unescaped;
16786 pm_token_t content;
16787
16788 if (match1(parser, PM_TOKEN_EOF)) {
16789 unescaped = PM_STRING_EMPTY;
16790 content = (pm_token_t) { .type = PM_TOKEN_STRING_CONTENT, .start = parser->start, .end = parser->start };
16791 } else {
16792 unescaped = parser->current_string;
16793 expect1(parser, PM_TOKEN_STRING_CONTENT, PM_ERR_EXPECT_STRING_CONTENT);
16794 content = parser->previous;
16795 }
16796
16797 // It is unfortunately possible to have multiple string content
16798 // nodes in a row in the case that there's heredoc content in
16799 // the middle of the string, like this cursed example:
16800 //
16801 // <<-END+'b
16802 // a
16803 // END
16804 // c'+'d'
16805 //
16806 // In that case we need to switch to an interpolated string to
16807 // be able to contain all of the parts.
16808 if (match1(parser, PM_TOKEN_STRING_CONTENT)) {
16809 pm_node_list_t parts = { 0 };
16810 pm_node_t *part = UP(pm_string_node_create_unescaped(parser, NULL, &content, NULL, &unescaped));
16811 pm_node_list_append(parser->arena, &parts, part);
16812
16813 do {
16814 part = UP(pm_string_node_create_current_string(parser, NULL, &parser->current, NULL));
16815 pm_node_list_append(parser->arena, &parts, part);
16816 parser_lex(parser);
16817 } while (match1(parser, PM_TOKEN_STRING_CONTENT));
16818
16819 expect1(parser, PM_TOKEN_STRING_END, PM_ERR_STRING_LITERAL_EOF);
16820 node = UP(pm_interpolated_string_node_create(parser, &opening, &parts, &parser->previous));
16821 } else if (accept1(parser, PM_TOKEN_LABEL_END)) {
16822 node = UP(pm_symbol_node_create_unescaped(parser, &opening, &content, &parser->previous, &unescaped, parse_symbol_encoding(parser, parser->explicit_encoding, &content, &unescaped, true)));
16823 if (!label_allowed) pm_parser_err_node(parser, node, PM_ERR_UNEXPECTED_LABEL);
16824 } else if (match1(parser, PM_TOKEN_EOF)) {
16825 pm_parser_err_token(parser, &opening, PM_ERR_STRING_LITERAL_EOF);
16826 node = UP(pm_string_node_create_unescaped(parser, &opening, &content, &parser->current, &unescaped));
16827 } else if (accept1(parser, PM_TOKEN_STRING_END)) {
16828 node = UP(pm_string_node_create_unescaped(parser, &opening, &content, &parser->previous, &unescaped));
16829 } else {
16830 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->previous, PM_ERR_STRING_LITERAL_TERM, pm_token_str(parser->previous.type));
16831 parser->previous.start = parser->previous.end;
16832 parser->previous.type = 0;
16833 node = UP(pm_string_node_create_unescaped(parser, &opening, &content, &parser->previous, &unescaped));
16834 }
16835 } else if (match1(parser, PM_TOKEN_STRING_CONTENT)) {
16836 // In this case we've hit string content so we know the string
16837 // at least has something in it. We'll need to check if the
16838 // following token is the end (in which case we can return a
16839 // plain string) or if it's not then it has interpolation.
16840 pm_token_t content = parser->current;
16841 pm_string_t unescaped = parser->current_string;
16842 const pm_encoding_t *explicit_encoding = parser->explicit_encoding;
16843 parser_lex(parser);
16844
16845 if (match2(parser, PM_TOKEN_STRING_END, PM_TOKEN_EOF)) {
16846 node = UP(pm_string_node_create_unescaped(parser, &opening, &content, &parser->current, &unescaped));
16847 pm_node_flag_set(node, parse_unescaped_encoding(parser, explicit_encoding));
16848
16849 // Kind of odd behavior, but basically if we have an
16850 // unterminated string and it ends in a newline, we back up one
16851 // character so that the error message is on the last line of
16852 // content in the string.
16853 if (!accept1(parser, PM_TOKEN_STRING_END)) {
16854 const uint8_t *location = parser->previous.end;
16855 if (location > parser->start && location[-1] == '\n') location--;
16856 pm_parser_err(parser, U32(location - parser->start), 0, PM_ERR_STRING_LITERAL_EOF);
16857
16858 parser->previous.start = parser->previous.end;
16859 parser->previous.type = 0;
16860 }
16861 } else if (accept1(parser, PM_TOKEN_LABEL_END)) {
16862 node = UP(pm_symbol_node_create_unescaped(parser, &opening, &content, &parser->previous, &unescaped, parse_symbol_encoding(parser, explicit_encoding, &content, &unescaped, true)));
16863 if (!label_allowed) pm_parser_err_node(parser, node, PM_ERR_UNEXPECTED_LABEL);
16864 } else {
16865 // If we get here, then we have interpolation so we'll need
16866 // to create a string or symbol node with interpolation.
16867 pm_node_list_t parts = { 0 };
16868 pm_node_t *part = UP(pm_string_node_create_unescaped(parser, NULL, &parser->previous, NULL, &unescaped));
16869 pm_node_flag_set(part, parse_unescaped_encoding(parser, explicit_encoding));
16870 pm_node_list_append(parser->arena, &parts, part);
16871
16872 while (!match3(parser, PM_TOKEN_STRING_END, PM_TOKEN_LABEL_END, PM_TOKEN_EOF)) {
16873 if ((part = parse_string_part(parser, (uint16_t) (depth + 1))) != NULL) {
16874 pm_node_list_append(parser->arena, &parts, part);
16875 }
16876 }
16877
16878 if (accept1(parser, PM_TOKEN_LABEL_END)) {
16879 node = UP(pm_interpolated_symbol_node_create(parser, &opening, &parts, &parser->previous));
16880 if (!label_allowed) pm_parser_err_node(parser, node, PM_ERR_UNEXPECTED_LABEL);
16881 } else if (match1(parser, PM_TOKEN_EOF)) {
16882 pm_parser_err_token(parser, &opening, PM_ERR_STRING_INTERPOLATED_TERM);
16883 node = UP(pm_interpolated_string_node_create(parser, &opening, &parts, &parser->current));
16884 } else {
16885 expect1(parser, PM_TOKEN_STRING_END, PM_ERR_STRING_INTERPOLATED_TERM);
16886 node = UP(pm_interpolated_string_node_create(parser, &opening, &parts, &parser->previous));
16887 }
16888 }
16889 } else {
16890 // If we get here, then the first part of the string is not plain
16891 // string content, in which case we need to parse the string as an
16892 // interpolated string.
16893 pm_node_list_t parts = { 0 };
16894 pm_node_t *part;
16895
16896 while (!match3(parser, PM_TOKEN_STRING_END, PM_TOKEN_LABEL_END, PM_TOKEN_EOF)) {
16897 if ((part = parse_string_part(parser, (uint16_t) (depth + 1))) != NULL) {
16898 pm_node_list_append(parser->arena, &parts, part);
16899 }
16900 }
16901
16902 if (accept1(parser, PM_TOKEN_LABEL_END)) {
16903 node = UP(pm_interpolated_symbol_node_create(parser, &opening, &parts, &parser->previous));
16904 if (!label_allowed) pm_parser_err_node(parser, node, PM_ERR_UNEXPECTED_LABEL);
16905 } else if (match1(parser, PM_TOKEN_EOF)) {
16906 pm_parser_err_token(parser, &opening, PM_ERR_STRING_INTERPOLATED_TERM);
16907 node = UP(pm_interpolated_string_node_create(parser, &opening, &parts, &parser->current));
16908 } else {
16909 expect1(parser, PM_TOKEN_STRING_END, PM_ERR_STRING_INTERPOLATED_TERM);
16910 node = UP(pm_interpolated_string_node_create(parser, &opening, &parts, &parser->previous));
16911 }
16912 }
16913
16914 if (current == NULL) {
16915 // If the node we just parsed is a symbol node, then we can't
16916 // concatenate it with anything else, so we can now return that
16917 // node.
16918 if (PM_NODE_TYPE_P(node, PM_SYMBOL_NODE) || PM_NODE_TYPE_P(node, PM_INTERPOLATED_SYMBOL_NODE)) {
16919 return node;
16920 }
16921
16922 // If we don't already have a node, then it's fine and we can just
16923 // set the result to be the node we just parsed.
16924 current = node;
16925 } else {
16926 // Otherwise we need to check the type of the node we just parsed.
16927 // If it cannot be concatenated with the previous node, then we'll
16928 // need to add a syntax error.
16929 if (!PM_NODE_TYPE_P(node, PM_STRING_NODE) && !PM_NODE_TYPE_P(node, PM_INTERPOLATED_STRING_NODE)) {
16930 pm_parser_err_node(parser, node, PM_ERR_STRING_CONCATENATION);
16931 }
16932
16933 // If we haven't already created our container for concatenation,
16934 // we'll do that now.
16935 if (!concating) {
16936 if (!PM_NODE_TYPE_P(current, PM_STRING_NODE) && !PM_NODE_TYPE_P(current, PM_INTERPOLATED_STRING_NODE)) {
16937 pm_parser_err_node(parser, current, PM_ERR_STRING_CONCATENATION);
16938 }
16939
16940 concating = true;
16941 pm_interpolated_string_node_t *container = pm_interpolated_string_node_create(parser, NULL, NULL, NULL);
16942 pm_interpolated_string_node_append(parser, container, current);
16943 current = UP(container);
16944 }
16945
16946 pm_interpolated_string_node_append(parser, (pm_interpolated_string_node_t *) current, node);
16947 }
16948 }
16949
16950 return current;
16951}
16952
16953#define PM_PARSE_PATTERN_SINGLE 0
16954#define PM_PARSE_PATTERN_TOP 1
16955#define PM_PARSE_PATTERN_MULTI 2
16956
16957static pm_node_t *
16958parse_pattern(pm_parser_t *parser, pm_constant_id_set_t *captures, uint8_t flags, pm_diagnostic_id_t diag_id, uint16_t depth);
16959
16965static void
16966parse_pattern_capture(pm_parser_t *parser, pm_constant_id_set_t *captures, pm_constant_id_t capture, const pm_location_t *location) {
16967 // Skip this capture if it starts with an underscore.
16968 if (peek_at(parser, parser->start + location->start) == '_') return;
16969
16970 if (!pm_constant_id_set_insert(parser->arena, captures, capture)) {
16971 pm_parser_err(parser, location->start, location->length, PM_ERR_PATTERN_CAPTURE_DUPLICATE);
16972 }
16973}
16974
16978static pm_node_t *
16979parse_pattern_constant_path(pm_parser_t *parser, pm_constant_id_set_t *captures, pm_node_t *node, uint16_t depth) {
16980 // Now, if there are any :: operators that follow, parse them as constant
16981 // path nodes.
16982 while (accept1(parser, PM_TOKEN_COLON_COLON)) {
16983 pm_token_t delimiter = parser->previous;
16984 expect1(parser, PM_TOKEN_CONSTANT, PM_ERR_CONSTANT_PATH_COLON_COLON_CONSTANT);
16985 node = UP(pm_constant_path_node_create(parser, node, &delimiter, &parser->previous));
16986 }
16987
16988 // If there is a [ or ( that follows, then this is part of a larger pattern
16989 // expression. We'll parse the inner pattern here, then modify the returned
16990 // inner pattern with our constant path attached.
16991 if (!match2(parser, PM_TOKEN_BRACKET_LEFT, PM_TOKEN_PARENTHESIS_LEFT)) {
16992 return node;
16993 }
16994
16995 pm_token_t opening;
16996 pm_token_t closing;
16997 pm_node_t *inner = NULL;
16998
16999 if (accept1(parser, PM_TOKEN_BRACKET_LEFT)) {
17000 opening = parser->previous;
17001 accept1(parser, PM_TOKEN_NEWLINE);
17002
17003 if (!accept1(parser, PM_TOKEN_BRACKET_RIGHT)) {
17004 inner = parse_pattern(parser, captures, PM_PARSE_PATTERN_TOP | PM_PARSE_PATTERN_MULTI, PM_ERR_PATTERN_EXPRESSION_AFTER_BRACKET, (uint16_t) (depth + 1));
17005 accept1(parser, PM_TOKEN_NEWLINE);
17006 expect1_opening(parser, PM_TOKEN_BRACKET_RIGHT, PM_ERR_PATTERN_TERM_BRACKET, &opening);
17007 }
17008
17009 closing = parser->previous;
17010 } else {
17011 parser_lex(parser);
17012 opening = parser->previous;
17013 accept1(parser, PM_TOKEN_NEWLINE);
17014
17015 if (!accept1(parser, PM_TOKEN_PARENTHESIS_RIGHT)) {
17016 inner = parse_pattern(parser, captures, PM_PARSE_PATTERN_TOP | PM_PARSE_PATTERN_MULTI, PM_ERR_PATTERN_EXPRESSION_AFTER_PAREN, (uint16_t) (depth + 1));
17017 accept1(parser, PM_TOKEN_NEWLINE);
17018 expect1_opening(parser, PM_TOKEN_PARENTHESIS_RIGHT, PM_ERR_PATTERN_TERM_PAREN, &opening);
17019 }
17020
17021 closing = parser->previous;
17022 }
17023
17024 if (!inner) {
17025 // If there was no inner pattern, then we have something like Foo() or
17026 // Foo[]. In that case we'll create an array pattern with no requireds.
17027 return UP(pm_array_pattern_node_constant_create(parser, node, &opening, &closing));
17028 }
17029
17030 // Now that we have the inner pattern, check to see if it's an array, find,
17031 // or hash pattern. If it is, then we'll attach our constant path to it if
17032 // it doesn't already have a constant. If it's not one of those node types
17033 // or it does have a constant, then we'll create an array pattern.
17034 switch (PM_NODE_TYPE(inner)) {
17035 case PM_ARRAY_PATTERN_NODE: {
17036 pm_array_pattern_node_t *pattern_node = (pm_array_pattern_node_t *) inner;
17037
17038 if (pattern_node->constant == NULL && pattern_node->opening_loc.length == 0) {
17039 PM_NODE_START_SET_NODE(pattern_node, node);
17040 PM_NODE_LENGTH_SET_TOKEN(parser, pattern_node, &closing);
17041
17042 pattern_node->constant = node;
17043 pattern_node->opening_loc = TOK2LOC(parser, &opening);
17044 pattern_node->closing_loc = TOK2LOC(parser, &closing);
17045
17046 return UP(pattern_node);
17047 }
17048
17049 break;
17050 }
17051 case PM_FIND_PATTERN_NODE: {
17052 pm_find_pattern_node_t *pattern_node = (pm_find_pattern_node_t *) inner;
17053
17054 if (pattern_node->constant == NULL && pattern_node->opening_loc.length == 0) {
17055 PM_NODE_START_SET_NODE(pattern_node, node);
17056 PM_NODE_LENGTH_SET_TOKEN(parser, pattern_node, &closing);
17057
17058 pattern_node->constant = node;
17059 pattern_node->opening_loc = TOK2LOC(parser, &opening);
17060 pattern_node->closing_loc = TOK2LOC(parser, &closing);
17061
17062 return UP(pattern_node);
17063 }
17064
17065 break;
17066 }
17067 case PM_HASH_PATTERN_NODE: {
17068 pm_hash_pattern_node_t *pattern_node = (pm_hash_pattern_node_t *) inner;
17069
17070 if (pattern_node->constant == NULL && pattern_node->opening_loc.length == 0) {
17071 PM_NODE_START_SET_NODE(pattern_node, node);
17072 PM_NODE_LENGTH_SET_TOKEN(parser, pattern_node, &closing);
17073
17074 pattern_node->constant = node;
17075 pattern_node->opening_loc = TOK2LOC(parser, &opening);
17076 pattern_node->closing_loc = TOK2LOC(parser, &closing);
17077
17078 return UP(pattern_node);
17079 }
17080
17081 break;
17082 }
17083 default:
17084 break;
17085 }
17086
17087 // If we got here, then we didn't return one of the inner patterns by
17088 // attaching its constant. In this case we'll create an array pattern and
17089 // attach our constant to it.
17090 pm_array_pattern_node_t *pattern_node = pm_array_pattern_node_constant_create(parser, node, &opening, &closing);
17091 pm_array_pattern_node_requireds_append(parser->arena, pattern_node, inner);
17092 return UP(pattern_node);
17093}
17094
17098static pm_splat_node_t *
17099parse_pattern_rest(pm_parser_t *parser, pm_constant_id_set_t *captures) {
17100 assert(parser->previous.type == PM_TOKEN_USTAR);
17101 pm_token_t operator = parser->previous;
17102 pm_node_t *name = NULL;
17103
17104 // Rest patterns don't necessarily have a name associated with them. So we
17105 // will check for that here. If they do, then we'll add it to the local
17106 // table since this pattern will cause it to become a local variable.
17107 if (accept1(parser, PM_TOKEN_IDENTIFIER)) {
17108 pm_constant_id_t constant_id = pm_parser_constant_id_token(parser, &parser->previous);
17109
17110 int depth;
17111 if ((depth = pm_parser_local_depth_constant_id(parser, constant_id)) == -1) {
17112 pm_parser_local_add(parser, constant_id, parser->previous.start, parser->previous.end, 0);
17113 }
17114
17115 pm_location_t previous_loc = TOK2LOC(parser, &parser->previous);
17116 parse_pattern_capture(parser, captures, constant_id, &previous_loc);
17117 name = UP(pm_local_variable_target_node_create(
17118 parser,
17119 &previous_loc,
17120 constant_id,
17121 (uint32_t) (depth == -1 ? 0 : depth)
17122 ));
17123 }
17124
17125 // Finally we can return the created node.
17126 return pm_splat_node_create(parser, &operator, name);
17127}
17128
17132static pm_node_t *
17133parse_pattern_keyword_rest(pm_parser_t *parser, pm_constant_id_set_t *captures) {
17134 assert(parser->current.type == PM_TOKEN_USTAR_STAR);
17135 parser_lex(parser);
17136
17137 pm_token_t operator = parser->previous;
17138 pm_node_t *value = NULL;
17139
17140 if (accept1(parser, PM_TOKEN_KEYWORD_NIL)) {
17141 return UP(pm_no_keywords_parameter_node_create(parser, &operator, &parser->previous));
17142 }
17143
17144 if (accept1(parser, PM_TOKEN_IDENTIFIER)) {
17145 pm_constant_id_t constant_id = pm_parser_constant_id_token(parser, &parser->previous);
17146
17147 int depth;
17148 if ((depth = pm_parser_local_depth_constant_id(parser, constant_id)) == -1) {
17149 pm_parser_local_add(parser, constant_id, parser->previous.start, parser->previous.end, 0);
17150 }
17151
17152 pm_location_t previous_loc = TOK2LOC(parser, &parser->previous);
17153 parse_pattern_capture(parser, captures, constant_id, &previous_loc);
17154 value = UP(pm_local_variable_target_node_create(
17155 parser,
17156 &previous_loc,
17157 constant_id,
17158 (uint32_t) (depth == -1 ? 0 : depth)
17159 ));
17160 }
17161
17162 return UP(pm_assoc_splat_node_create(parser, value, &operator));
17163}
17164
17169static bool
17170pm_slice_is_valid_local(const pm_parser_t *parser, const uint8_t *start, const uint8_t *end) {
17171 ptrdiff_t length = end - start;
17172 if (length == 0) return false;
17173
17174 // First ensure that it starts with a valid identifier starting character.
17175 size_t width = char_is_identifier_start(parser, start, end - start);
17176 if (width == 0) return false;
17177
17178 // Next, ensure that it's not an uppercase character.
17179 if (parser->encoding_changed) {
17180 if (parser->encoding->isupper_char(start, length)) return false;
17181 } else {
17182 if (pm_encoding_utf_8_isupper_char(start, length)) return false;
17183 }
17184
17185 // Next, iterate through all of the bytes of the string to ensure that they
17186 // are all valid identifier characters.
17187 const uint8_t *cursor = start + width;
17188 while ((width = char_is_identifier(parser, cursor, end - cursor))) cursor += width;
17189 return cursor == end;
17190}
17191
17196static pm_node_t *
17197parse_pattern_hash_implicit_value(pm_parser_t *parser, pm_constant_id_set_t *captures, pm_symbol_node_t *key) {
17198 const pm_location_t *value_loc = &((pm_symbol_node_t *) key)->content_loc;
17199 const uint8_t *start = parser->start + PM_LOCATION_START(value_loc);
17200 const uint8_t *end = parser->start + PM_LOCATION_END(value_loc);
17201
17202 pm_constant_id_t constant_id = pm_parser_constant_id_raw(parser, start, end);
17203 int depth = -1;
17204
17205 if (pm_slice_is_valid_local(parser, start, end)) {
17206 depth = pm_parser_local_depth_constant_id(parser, constant_id);
17207 } else {
17208 pm_parser_err(parser, PM_NODE_START(key), PM_NODE_LENGTH(key), PM_ERR_PATTERN_HASH_KEY_LOCALS);
17209
17210 if ((end > start) && ((end[-1] == '!') || (end[-1] == '?'))) {
17211 PM_PARSER_ERR_FORMAT(parser, value_loc->start, value_loc->length, PM_ERR_INVALID_LOCAL_VARIABLE_WRITE, (int) (end - start), (const char *) start);
17212 }
17213 }
17214
17215 if (depth == -1) {
17216 pm_parser_local_add(parser, constant_id, start, end, 0);
17217 }
17218
17219 parse_pattern_capture(parser, captures, constant_id, value_loc);
17220 pm_local_variable_target_node_t *target = pm_local_variable_target_node_create(
17221 parser,
17222 value_loc,
17223 constant_id,
17224 (uint32_t) (depth == -1 ? 0 : depth)
17225 );
17226
17227 return UP(pm_implicit_node_create(parser, UP(target)));
17228}
17229
17234static void
17235parse_pattern_hash_key(pm_parser_t *parser, pm_static_literals_t *keys, pm_node_t *node) {
17236 if (pm_static_literals_add(&parser->line_offsets, parser->start, parser->start_line, parser->encoding, keys, node, true) != NULL) {
17237 pm_parser_err_node(parser, node, PM_ERR_PATTERN_HASH_KEY_DUPLICATE);
17238 }
17239}
17240
17245parse_pattern_hash(pm_parser_t *parser, pm_constant_id_set_t *captures, pm_node_t *first_node, uint16_t depth) {
17246 pm_node_list_t assocs = { 0 };
17247 pm_static_literals_t keys = { 0 };
17248 pm_node_t *rest = NULL;
17249
17250 switch (PM_NODE_TYPE(first_node)) {
17251 case PM_ASSOC_SPLAT_NODE:
17252 case PM_NO_KEYWORDS_PARAMETER_NODE:
17253 rest = first_node;
17254 break;
17255 case PM_INTERPOLATED_SYMBOL_NODE:
17256 case PM_SYMBOL_NODE: {
17257 if (pm_symbol_node_label_p(parser, first_node)) {
17258 if (PM_NODE_TYPE_P(first_node, PM_INTERPOLATED_SYMBOL_NODE)) {
17259 pm_parser_err_node(parser, first_node, PM_ERR_PATTERN_HASH_KEY_INTERPOLATED);
17260 } else {
17261 parse_pattern_hash_key(parser, &keys, first_node);
17262 }
17263
17264 pm_node_t *value;
17265
17266 /*
17267 * The label has an implicit value when the next token cannot
17268 * begin a pattern, mirroring the grammar's `p_kw: p_kw_label`
17269 * reduction.
17270 */
17271 if (!token_begins_pattern_p(parser->current.type)) {
17272 if (PM_NODE_TYPE_P(first_node, PM_SYMBOL_NODE)) {
17273 value = parse_pattern_hash_implicit_value(parser, captures, (pm_symbol_node_t *) first_node);
17274 } else {
17275 value = UP(pm_error_recovery_node_create(parser, PM_NODE_END(first_node), 0));
17276 }
17277 } else {
17278 // Here we have a value for the first assoc in the list, so
17279 // we will parse it now.
17280 value = parse_pattern(parser, captures, PM_PARSE_PATTERN_SINGLE, PM_ERR_PATTERN_EXPRESSION_AFTER_KEY, (uint16_t) (depth + 1));
17281 }
17282
17283 pm_node_t *assoc = UP(pm_assoc_node_create(parser, first_node, NULL, value));
17284 pm_node_list_append(parser->arena, &assocs, assoc);
17285 break;
17286 }
17287 }
17289 default: {
17290 // If we get anything else, then this is an error. For this we'll
17291 // create a missing node for the value and create an assoc node for
17292 // the first node in the list.
17293 pm_diagnostic_id_t diag_id = PM_NODE_TYPE_P(first_node, PM_INTERPOLATED_SYMBOL_NODE) ? PM_ERR_PATTERN_HASH_KEY_INTERPOLATED : PM_ERR_PATTERN_HASH_KEY_LABEL;
17294 pm_parser_err_node(parser, first_node, diag_id);
17295
17296 pm_node_t *value = UP(pm_error_recovery_node_create(parser, PM_NODE_START(first_node), PM_NODE_LENGTH(first_node)));
17297 pm_node_t *assoc = UP(pm_assoc_node_create(parser, first_node, NULL, value));
17298
17299 pm_node_list_append(parser->arena, &assocs, assoc);
17300 break;
17301 }
17302 }
17303
17304 // If there are any other assocs, then we'll parse them now.
17305 while (accept1(parser, PM_TOKEN_COMMA)) {
17306 /*
17307 * A trailing comma ends the pattern when the next token cannot begin
17308 * another element, mirroring the grammar's `p_kwargs: p_kwarg ','`
17309 * reduction.
17310 */
17311 if (!token_begins_pattern_p(parser->current.type)) {
17312 // Trailing commas are not allowed to follow a rest pattern.
17313 if (rest != NULL) {
17314 pm_parser_err_token(parser, &parser->current, PM_ERR_PATTERN_EXPRESSION_AFTER_REST);
17315 }
17316
17317 break;
17318 }
17319
17320 if (match1(parser, PM_TOKEN_USTAR_STAR)) {
17321 pm_node_t *assoc = parse_pattern_keyword_rest(parser, captures);
17322
17323 if (rest == NULL) {
17324 rest = assoc;
17325 } else {
17326 pm_parser_err_node(parser, assoc, PM_ERR_PATTERN_EXPRESSION_AFTER_REST);
17327 pm_node_list_append(parser->arena, &assocs, assoc);
17328 }
17329 } else {
17330 pm_node_t *key;
17331
17332 if (match1(parser, PM_TOKEN_STRING_BEGIN)) {
17333 key = parse_strings(parser, NULL, true, (uint16_t) (depth + 1));
17334
17335 if (PM_NODE_TYPE_P(key, PM_INTERPOLATED_SYMBOL_NODE)) {
17336 pm_parser_err_node(parser, key, PM_ERR_PATTERN_HASH_KEY_INTERPOLATED);
17337 } else if (!pm_symbol_node_label_p(parser, key)) {
17338 pm_parser_err_node(parser, key, PM_ERR_PATTERN_LABEL_AFTER_COMMA);
17339 }
17340 } else if (accept1(parser, PM_TOKEN_LABEL)) {
17341 key = UP(pm_symbol_node_label_create(parser, &parser->previous));
17342 } else {
17343 expect1(parser, PM_TOKEN_LABEL, PM_ERR_PATTERN_LABEL_AFTER_COMMA);
17344
17345 pm_token_t label = { .type = PM_TOKEN_LABEL, .start = parser->previous.end, .end = parser->previous.end };
17346 key = UP(pm_symbol_node_create(parser, NULL, &label, NULL));
17347 }
17348
17349 parse_pattern_hash_key(parser, &keys, key);
17350 pm_node_t *value = NULL;
17351
17352 /*
17353 * The label has an implicit value when the next token cannot
17354 * begin a pattern, mirroring the grammar's `p_kw: p_kw_label`
17355 * reduction.
17356 */
17357 if (!token_begins_pattern_p(parser->current.type)) {
17358 if (PM_NODE_TYPE_P(key, PM_SYMBOL_NODE)) {
17359 value = parse_pattern_hash_implicit_value(parser, captures, (pm_symbol_node_t *) key);
17360 } else {
17361 value = UP(pm_error_recovery_node_create(parser, PM_NODE_END(key), 0));
17362 }
17363 } else {
17364 value = parse_pattern(parser, captures, PM_PARSE_PATTERN_SINGLE, PM_ERR_PATTERN_EXPRESSION_AFTER_KEY, (uint16_t) (depth + 1));
17365 }
17366
17367 pm_node_t *assoc = UP(pm_assoc_node_create(parser, key, NULL, value));
17368
17369 if (rest != NULL) {
17370 pm_parser_err_node(parser, assoc, PM_ERR_PATTERN_EXPRESSION_AFTER_REST);
17371 }
17372
17373 pm_node_list_append(parser->arena, &assocs, assoc);
17374 }
17375 }
17376
17377 pm_hash_pattern_node_t *node = pm_hash_pattern_node_node_list_create(parser, &assocs, rest);
17378 // assocs.nodes is arena-allocated; no explicit free needed.
17379
17380 pm_static_literals_free(&keys);
17381 return node;
17382}
17383
17387static pm_node_t *
17388parse_pattern_primitive(pm_parser_t *parser, pm_constant_id_set_t *captures, pm_diagnostic_id_t diag_id, uint16_t depth) {
17389 switch (parser->current.type) {
17390 case PM_TOKEN_IDENTIFIER:
17391 case PM_TOKEN_METHOD_NAME: {
17392 parser_lex(parser);
17393 pm_constant_id_t constant_id = pm_parser_constant_id_token(parser, &parser->previous);
17394
17395 int depth;
17396 if ((depth = pm_parser_local_depth_constant_id(parser, constant_id)) == -1) {
17397 pm_parser_local_add(parser, constant_id, parser->previous.start, parser->previous.end, 0);
17398 }
17399
17400 pm_location_t previous_loc = TOK2LOC(parser, &parser->previous);
17401 parse_pattern_capture(parser, captures, constant_id, &previous_loc);
17402 return UP(pm_local_variable_target_node_create(
17403 parser,
17404 &previous_loc,
17405 constant_id,
17406 (uint32_t) (depth == -1 ? 0 : depth)
17407 ));
17408 }
17409 case PM_TOKEN_BRACKET_LEFT_ARRAY: {
17410 pm_token_t opening = parser->current;
17411 parser_lex(parser);
17412
17413 if (accept1(parser, PM_TOKEN_BRACKET_RIGHT)) {
17414 // If we have an empty array pattern, then we'll just return a new
17415 // array pattern node.
17416 return UP(pm_array_pattern_node_empty_create(parser, &opening, &parser->previous));
17417 }
17418
17419 // Otherwise, we'll parse the inner pattern, then deal with it depending
17420 // on the type it returns.
17421 pm_node_t *inner = parse_pattern(parser, captures, PM_PARSE_PATTERN_MULTI, PM_ERR_PATTERN_EXPRESSION_AFTER_BRACKET, (uint16_t) (depth + 1));
17422
17423 accept1(parser, PM_TOKEN_NEWLINE);
17424 expect1_opening(parser, PM_TOKEN_BRACKET_RIGHT, PM_ERR_PATTERN_TERM_BRACKET, &opening);
17425 pm_token_t closing = parser->previous;
17426
17427 switch (PM_NODE_TYPE(inner)) {
17428 case PM_ARRAY_PATTERN_NODE: {
17429 pm_array_pattern_node_t *pattern_node = (pm_array_pattern_node_t *) inner;
17430 if (pattern_node->opening_loc.length == 0) {
17431 PM_NODE_START_SET_TOKEN(parser, pattern_node, &opening);
17432 PM_NODE_LENGTH_SET_TOKEN(parser, pattern_node, &closing);
17433
17434 pattern_node->opening_loc = TOK2LOC(parser, &opening);
17435 pattern_node->closing_loc = TOK2LOC(parser, &closing);
17436
17437 return UP(pattern_node);
17438 }
17439
17440 break;
17441 }
17442 case PM_FIND_PATTERN_NODE: {
17443 pm_find_pattern_node_t *pattern_node = (pm_find_pattern_node_t *) inner;
17444 if (pattern_node->opening_loc.length == 0) {
17445 PM_NODE_START_SET_TOKEN(parser, pattern_node, &opening);
17446 PM_NODE_LENGTH_SET_TOKEN(parser, pattern_node, &closing);
17447
17448 pattern_node->opening_loc = TOK2LOC(parser, &opening);
17449 pattern_node->closing_loc = TOK2LOC(parser, &closing);
17450
17451 return UP(pattern_node);
17452 }
17453
17454 break;
17455 }
17456 default:
17457 break;
17458 }
17459
17460 pm_array_pattern_node_t *node = pm_array_pattern_node_empty_create(parser, &opening, &closing);
17461 pm_array_pattern_node_requireds_append(parser->arena, node, inner);
17462 return UP(node);
17463 }
17464 case PM_TOKEN_BRACE_LEFT_HASH: {
17465 bool previous_pattern_matching_newlines = parser->pattern_matching_newlines;
17466 parser->pattern_matching_newlines = false;
17467
17469 pm_token_t opening = parser->current;
17470 parser_lex(parser);
17471
17472 if (accept1(parser, PM_TOKEN_BRACE_RIGHT)) {
17473 // If we have an empty hash pattern, then we'll just return a new hash
17474 // pattern node.
17475 node = pm_hash_pattern_node_empty_create(parser, &opening, &parser->previous);
17476 } else {
17477 pm_node_t *first_node;
17478
17479 switch (parser->current.type) {
17480 case PM_TOKEN_LABEL:
17481 parser_lex(parser);
17482 first_node = UP(pm_symbol_node_label_create(parser, &parser->previous));
17483 break;
17484 case PM_TOKEN_USTAR_STAR:
17485 first_node = parse_pattern_keyword_rest(parser, captures);
17486 break;
17487 case PM_TOKEN_STRING_BEGIN:
17488 first_node = parse_expression(parser, PM_BINDING_POWER_MAX, PM_PARSE_ACCEPTS_DO_BLOCK | PM_PARSE_ACCEPTS_LABEL, PM_ERR_PATTERN_HASH_KEY_LABEL, (uint16_t) (depth + 1));
17489 break;
17490 default: {
17491 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->current, PM_ERR_PATTERN_HASH_KEY, pm_token_str(parser->current.type));
17492 parser_lex(parser);
17493
17494 first_node = UP(pm_error_recovery_node_create(parser, PM_TOKEN_START(parser, &parser->previous), PM_TOKEN_LENGTH(&parser->previous)));
17495 break;
17496 }
17497 }
17498
17499 node = parse_pattern_hash(parser, captures, first_node, (uint16_t) (depth + 1));
17500
17501 accept1(parser, PM_TOKEN_NEWLINE);
17502 expect1_opening(parser, PM_TOKEN_BRACE_RIGHT, PM_ERR_PATTERN_TERM_BRACE, &opening);
17503 pm_token_t closing = parser->previous;
17504
17505 PM_NODE_START_SET_TOKEN(parser, node, &opening);
17506 PM_NODE_LENGTH_SET_TOKEN(parser, node, &closing);
17507
17508 node->opening_loc = TOK2LOC(parser, &opening);
17509 node->closing_loc = TOK2LOC(parser, &closing);
17510 }
17511
17512 parser->pattern_matching_newlines = previous_pattern_matching_newlines;
17513 return UP(node);
17514 }
17515 case PM_TOKEN_UDOT_DOT:
17516 case PM_TOKEN_UDOT_DOT_DOT: {
17517 pm_token_t operator = parser->current;
17518 parser_lex(parser);
17519
17520 // Since we have a unary range operator, we need to parse the subsequent
17521 // expression as the right side of the range.
17522 switch (parser->current.type) {
17523 case PM_CASE_PRIMITIVE: {
17524 pm_node_t *right = parse_expression(parser, PM_BINDING_POWER_MAX, PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_PATTERN_EXPRESSION_AFTER_RANGE, (uint16_t) (depth + 1));
17525 return UP(pm_range_node_create(parser, NULL, &operator, right));
17526 }
17527 default: {
17528 pm_parser_err_token(parser, &operator, PM_ERR_PATTERN_EXPRESSION_AFTER_RANGE);
17529 pm_node_t *right = UP(pm_error_recovery_node_create(parser, PM_TOKEN_START(parser, &operator), PM_TOKEN_LENGTH(&operator)));
17530 return UP(pm_range_node_create(parser, NULL, &operator, right));
17531 }
17532 }
17533 }
17534 case PM_CASE_PRIMITIVE: {
17535 pm_node_t *node = parse_expression(parser, PM_BINDING_POWER_MAX, PM_PARSE_ACCEPTS_LABEL | PM_PARSE_ACCEPTS_DO_BLOCK, diag_id, (uint16_t) (depth + 1));
17536
17537 // If we found a label, we need to immediately return to the caller.
17538 if (pm_symbol_node_label_p(parser, node)) return node;
17539
17540 // Call nodes (arithmetic operations) are not allowed in patterns
17541 if (PM_NODE_TYPE(node) == PM_CALL_NODE) {
17542 pm_parser_err_node(parser, node, diag_id);
17543 return UP(pm_error_recovery_node_create_unexpected(parser, node));
17544 }
17545
17546 // Now that we have a primitive, we need to check if it's part of a range.
17547 if (accept2(parser, PM_TOKEN_DOT_DOT, PM_TOKEN_DOT_DOT_DOT)) {
17548 pm_token_t operator = parser->previous;
17549
17550 // Now that we have the operator, we need to check if this is followed
17551 // by another expression. If it is, then we will create a full range
17552 // node. Otherwise, we'll create an endless range.
17553 switch (parser->current.type) {
17554 case PM_CASE_PRIMITIVE: {
17555 pm_node_t *right = parse_expression(parser, PM_BINDING_POWER_MAX, PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_PATTERN_EXPRESSION_AFTER_RANGE, (uint16_t) (depth + 1));
17556 return UP(pm_range_node_create(parser, node, &operator, right));
17557 }
17558 default:
17559 return UP(pm_range_node_create(parser, node, &operator, NULL));
17560 }
17561 }
17562
17563 return node;
17564 }
17565 case PM_TOKEN_CARET: {
17566 parser_lex(parser);
17567 pm_token_t operator = parser->previous;
17568
17569 // At this point we have a pin operator. We need to check the subsequent
17570 // expression to determine if it's a variable or an expression.
17571 switch (parser->current.type) {
17572 case PM_TOKEN_IDENTIFIER: {
17573 parser_lex(parser);
17574 pm_node_t *variable = UP(parse_variable(parser));
17575
17576 if (variable == NULL) {
17577 PM_PARSER_ERR_TOKEN_FORMAT_CONTENT(parser, &parser->previous, PM_ERR_NO_LOCAL_VARIABLE);
17578 variable = UP(pm_local_variable_read_node_missing_create(parser, &parser->previous, 0));
17579 }
17580
17581 return UP(pm_pinned_variable_node_create(parser, &operator, variable));
17582 }
17583 case PM_TOKEN_INSTANCE_VARIABLE: {
17584 parser_lex(parser);
17585 pm_node_t *variable = UP(pm_instance_variable_read_node_create(parser, &parser->previous));
17586
17587 return UP(pm_pinned_variable_node_create(parser, &operator, variable));
17588 }
17589 case PM_TOKEN_CLASS_VARIABLE: {
17590 parser_lex(parser);
17591 pm_node_t *variable = UP(pm_class_variable_read_node_create(parser, &parser->previous));
17592
17593 return UP(pm_pinned_variable_node_create(parser, &operator, variable));
17594 }
17595 case PM_TOKEN_GLOBAL_VARIABLE: {
17596 parser_lex(parser);
17597 pm_node_t *variable = UP(pm_global_variable_read_node_create(parser, &parser->previous));
17598
17599 return UP(pm_pinned_variable_node_create(parser, &operator, variable));
17600 }
17601 case PM_TOKEN_PARENTHESIS_LEFT_GROUPING: {
17602 bool previous_pattern_matching_newlines = parser->pattern_matching_newlines;
17603 parser->pattern_matching_newlines = false;
17604
17605 pm_token_t lparen = parser->current;
17606 parser_lex(parser);
17607
17608 pm_node_t *expression = parse_value_expression(parser, PM_BINDING_POWER_COMPOSITION, PM_PARSE_ACCEPTS_DO_BLOCK | PM_PARSE_ACCEPTS_COMMAND_CALL, PM_ERR_PATTERN_EXPRESSION_AFTER_PIN, (uint16_t) (depth + 1));
17609 parser->pattern_matching_newlines = previous_pattern_matching_newlines;
17610
17611 accept1(parser, PM_TOKEN_NEWLINE);
17612 expect1_opening(parser, PM_TOKEN_PARENTHESIS_RIGHT, PM_ERR_PATTERN_TERM_PAREN, &lparen);
17613 return UP(pm_pinned_expression_node_create(parser, expression, &operator, &lparen, &parser->previous));
17614 }
17615 default: {
17616 // If we get here, then we have a pin operator followed by something
17617 // not understood. We'll create a missing node and return that.
17618 pm_parser_err_token(parser, &operator, PM_ERR_PATTERN_EXPRESSION_AFTER_PIN);
17619 pm_node_t *variable = UP(pm_error_recovery_node_create(parser, PM_TOKEN_START(parser, &operator), PM_TOKEN_LENGTH(&operator)));
17620 return UP(pm_pinned_variable_node_create(parser, &operator, variable));
17621 }
17622 }
17623 }
17624 case PM_TOKEN_UCOLON_COLON: {
17625 pm_token_t delimiter = parser->current;
17626 parser_lex(parser);
17627
17628 expect1(parser, PM_TOKEN_CONSTANT, PM_ERR_CONSTANT_PATH_COLON_COLON_CONSTANT);
17629 pm_constant_path_node_t *node = pm_constant_path_node_create(parser, NULL, &delimiter, &parser->previous);
17630
17631 return parse_pattern_constant_path(parser, captures, UP(node), (uint16_t) (depth + 1));
17632 }
17633 case PM_TOKEN_CONSTANT: {
17634 pm_token_t constant = parser->current;
17635 parser_lex(parser);
17636
17637 pm_node_t *node = UP(pm_constant_read_node_create(parser, &constant));
17638 return parse_pattern_constant_path(parser, captures, node, (uint16_t) (depth + 1));
17639 }
17640 default:
17641 pm_parser_err_current(parser, diag_id);
17642 return UP(pm_error_recovery_node_create(parser, PM_TOKEN_START(parser, &parser->current), PM_TOKEN_LENGTH(&parser->current)));
17643 }
17644}
17645
17653static bool
17654parse_pattern_alternation_error_each(const pm_node_t *node, void *data) {
17655 pm_parser_t *parser = (pm_parser_t *) data;
17656
17657 switch (PM_NODE_TYPE(node)) {
17658 case PM_LOCAL_VARIABLE_TARGET_NODE:
17659 // Underscore-prefixed names are not captures, see
17660 // parse_pattern_capture.
17661 if (peek_at(parser, parser->start + PM_NODE_START(node)) != '_') {
17662 pm_parser_err(parser, PM_NODE_START(node), PM_NODE_LENGTH(node), PM_ERR_PATTERN_CAPTURE_IN_ALTERNATIVE);
17663 }
17664 return false;
17665 case PM_ARRAY_PATTERN_NODE:
17666 case PM_ASSOC_NODE:
17667 case PM_ASSOC_SPLAT_NODE:
17668 case PM_CAPTURE_PATTERN_NODE:
17669 case PM_FIND_PATTERN_NODE:
17670 case PM_HASH_PATTERN_NODE:
17671 case PM_IMPLICIT_NODE:
17672 case PM_PARENTHESES_NODE:
17673 case PM_SPLAT_NODE:
17674 return true;
17675 default:
17676 return false;
17677 }
17678}
17679
17685static void
17686parse_pattern_alternation_error(pm_parser_t *parser, const pm_node_t *node) {
17687 pm_visit_node(node, parse_pattern_alternation_error_each, parser);
17688}
17689
17694static pm_node_t *
17695parse_pattern_primitives(pm_parser_t *parser, pm_constant_id_set_t *captures, pm_node_t *first_node, pm_diagnostic_id_t diag_id, uint16_t depth) {
17696 pm_node_t *node = first_node;
17697 bool alternation = false;
17698
17699 while ((node == NULL) || (alternation = accept1(parser, PM_TOKEN_PIPE))) {
17700 if (alternation && !PM_NODE_TYPE_P(node, PM_ALTERNATION_PATTERN_NODE) && captures->size) {
17701 parse_pattern_alternation_error(parser, node);
17702 }
17703
17704 switch (parser->current.type) {
17705 case PM_TOKEN_IDENTIFIER:
17706 case PM_TOKEN_BRACKET_LEFT_ARRAY:
17707 case PM_TOKEN_BRACE_LEFT_HASH:
17708 case PM_TOKEN_CARET:
17709 case PM_TOKEN_CONSTANT:
17710 case PM_TOKEN_UCOLON_COLON:
17711 case PM_TOKEN_UDOT_DOT:
17712 case PM_TOKEN_UDOT_DOT_DOT:
17713 case PM_CASE_PRIMITIVE: {
17714 if (!alternation) {
17715 node = parse_pattern_primitive(parser, captures, diag_id, (uint16_t) (depth + 1));
17716 } else {
17717 pm_token_t operator = parser->previous;
17718 pm_node_t *right = parse_pattern_primitive(parser, captures, PM_ERR_PATTERN_EXPRESSION_AFTER_PIPE, (uint16_t) (depth + 1));
17719
17720 if (captures->size) parse_pattern_alternation_error(parser, right);
17721 node = UP(pm_alternation_pattern_node_create(parser, node, right, &operator));
17722 }
17723
17724 break;
17725 }
17726 case PM_TOKEN_PARENTHESIS_LEFT_GROUPING:
17727 case PM_TOKEN_PARENTHESIS_LEFT_PARENTHESES: {
17728 pm_token_t operator = parser->previous;
17729 pm_token_t opening = parser->current;
17730 parser_lex(parser);
17731
17732 pm_node_t *body = parse_pattern(parser, captures, PM_PARSE_PATTERN_SINGLE, PM_ERR_PATTERN_EXPRESSION_AFTER_PAREN, (uint16_t) (depth + 1));
17733 accept1(parser, PM_TOKEN_NEWLINE);
17734 expect1_opening(parser, PM_TOKEN_PARENTHESIS_RIGHT, PM_ERR_PATTERN_TERM_PAREN, &opening);
17735 pm_node_t *right = UP(pm_parentheses_node_create(parser, &opening, body, &parser->previous, 0));
17736
17737 if (!alternation) {
17738 node = right;
17739 } else {
17740 if (captures->size) parse_pattern_alternation_error(parser, right);
17741 node = UP(pm_alternation_pattern_node_create(parser, node, right, &operator));
17742 }
17743
17744 break;
17745 }
17746 default: {
17747 pm_parser_err_current(parser, diag_id);
17748 pm_node_t *right = UP(pm_error_recovery_node_create(parser, PM_TOKEN_START(parser, &parser->current), PM_TOKEN_LENGTH(&parser->current)));
17749
17750 if (!alternation) {
17751 node = right;
17752 } else {
17753 if (captures->size) parse_pattern_alternation_error(parser, right);
17754 node = UP(pm_alternation_pattern_node_create(parser, node, right, &parser->previous));
17755 }
17756
17757 break;
17758 }
17759 }
17760 }
17761
17762 // If we have an =>, then we are assigning this pattern to a variable.
17763 // In this case we should create an assignment node.
17764 while (accept1(parser, PM_TOKEN_EQUAL_GREATER)) {
17765 pm_token_t operator = parser->previous;
17766 expect1(parser, PM_TOKEN_IDENTIFIER, PM_ERR_PATTERN_IDENT_AFTER_HROCKET);
17767
17768 pm_constant_id_t constant_id = pm_parser_constant_id_token(parser, &parser->previous);
17769 int depth;
17770
17771 if ((depth = pm_parser_local_depth_constant_id(parser, constant_id)) == -1) {
17772 pm_parser_local_add(parser, constant_id, parser->previous.start, parser->previous.end, 0);
17773 }
17774
17775 pm_location_t previous_loc = TOK2LOC(parser, &parser->previous);
17776 parse_pattern_capture(parser, captures, constant_id, &previous_loc);
17777 pm_local_variable_target_node_t *target = pm_local_variable_target_node_create(
17778 parser,
17779 &previous_loc,
17780 constant_id,
17781 (uint32_t) (depth == -1 ? 0 : depth)
17782 );
17783
17784 node = UP(pm_capture_pattern_node_create(parser, node, target, &operator));
17785 }
17786
17787 return node;
17788}
17789
17793static pm_node_t *
17794parse_pattern(pm_parser_t *parser, pm_constant_id_set_t *captures, uint8_t flags, pm_diagnostic_id_t diag_id, uint16_t depth) {
17795 pm_node_t *node = NULL;
17796
17797 bool leading_rest = false;
17798 bool trailing_rest = false;
17799
17800 switch (parser->current.type) {
17801 case PM_TOKEN_LABEL: {
17802 parser_lex(parser);
17803 pm_node_t *key = UP(pm_symbol_node_label_create(parser, &parser->previous));
17804 node = UP(parse_pattern_hash(parser, captures, key, (uint16_t) (depth + 1)));
17805
17806 if (!(flags & PM_PARSE_PATTERN_TOP)) {
17807 pm_parser_err_node(parser, node, PM_ERR_PATTERN_HASH_IMPLICIT);
17808 }
17809
17810 return node;
17811 }
17812 case PM_TOKEN_USTAR_STAR: {
17813 node = parse_pattern_keyword_rest(parser, captures);
17814 node = UP(parse_pattern_hash(parser, captures, node, (uint16_t) (depth + 1)));
17815
17816 if (!(flags & PM_PARSE_PATTERN_TOP)) {
17817 pm_parser_err_node(parser, node, PM_ERR_PATTERN_HASH_IMPLICIT);
17818 }
17819
17820 return node;
17821 }
17822 case PM_TOKEN_STRING_BEGIN: {
17823 // We need special handling for string beginnings because they could
17824 // be dynamic symbols leading to hash patterns.
17825 node = parse_pattern_primitive(parser, captures, diag_id, (uint16_t) (depth + 1));
17826
17827 if (pm_symbol_node_label_p(parser, node)) {
17828 node = UP(parse_pattern_hash(parser, captures, node, (uint16_t) (depth + 1)));
17829
17830 if (!(flags & PM_PARSE_PATTERN_TOP)) {
17831 pm_parser_err_node(parser, node, PM_ERR_PATTERN_HASH_IMPLICIT);
17832 }
17833
17834 return node;
17835 }
17836
17837 node = parse_pattern_primitives(parser, captures, node, diag_id, (uint16_t) (depth + 1));
17838 break;
17839 }
17840 case PM_TOKEN_USTAR: {
17841 if (flags & (PM_PARSE_PATTERN_TOP | PM_PARSE_PATTERN_MULTI)) {
17842 parser_lex(parser);
17843 node = UP(parse_pattern_rest(parser, captures));
17844 leading_rest = true;
17845 break;
17846 }
17847 }
17849 default:
17850 node = parse_pattern_primitives(parser, captures, NULL, diag_id, (uint16_t) (depth + 1));
17851 break;
17852 }
17853
17854 // If we got a dynamic label symbol, then we need to treat it like the
17855 // beginning of a hash pattern.
17856 if (pm_symbol_node_label_p(parser, node)) {
17857 return UP(parse_pattern_hash(parser, captures, node, (uint16_t) (depth + 1)));
17858 }
17859
17860 if ((flags & PM_PARSE_PATTERN_MULTI) && match1(parser, PM_TOKEN_COMMA)) {
17861 // If we have a comma, then we are now parsing either an array pattern
17862 // or a find pattern. We need to parse all of the patterns, put them
17863 // into a big list, and then determine which type of node we have.
17864 pm_node_list_t nodes = { 0 };
17865 pm_node_list_append(parser->arena, &nodes, node);
17866
17867 // Gather up all of the patterns into the list.
17868 while (accept1(parser, PM_TOKEN_COMMA)) {
17869 /*
17870 * A trailing comma ends the pattern when the next token cannot
17871 * begin another pattern element, leaving the token for the
17872 * enclosing context to accept or reject.
17873 */
17874 if (!token_begins_pattern_p(parser->current.type)) {
17875 // A trailing comma forms an implicit rest pattern (`[a,]` is
17876 // `[a, *]`). If a rest pattern has already been parsed, then
17877 // this is a second rest, which is not allowed (e.g. `[a, *b,]`
17878 // or `x => a, *b,`).
17879 if (trailing_rest) {
17880 pm_parser_err_previous(parser, PM_ERR_PATTERN_REST);
17881 }
17882
17883 node = UP(pm_implicit_rest_node_create(parser, &parser->previous));
17884 pm_node_list_append(parser->arena, &nodes, node);
17885 trailing_rest = true;
17886 break;
17887 }
17888
17889 if (accept1(parser, PM_TOKEN_USTAR)) {
17890 node = UP(parse_pattern_rest(parser, captures));
17891
17892 // If we have already parsed a splat pattern, then this is an
17893 // error. We will continue to parse the rest of the patterns,
17894 // but we will indicate it as an error.
17895 if (trailing_rest) {
17896 pm_parser_err_previous(parser, PM_ERR_PATTERN_REST);
17897 }
17898
17899 trailing_rest = true;
17900 } else {
17901 node = parse_pattern_primitives(parser, captures, NULL, PM_ERR_PATTERN_EXPRESSION_AFTER_COMMA, (uint16_t) (depth + 1));
17902 }
17903
17904 pm_node_list_append(parser->arena, &nodes, node);
17905 }
17906
17907 // If the first pattern and the last pattern are rest patterns, then we
17908 // will call this a find pattern, regardless of how many rest patterns
17909 // are in between because we know we already added the appropriate
17910 // errors. Otherwise we will create an array pattern.
17911 if (leading_rest && PM_NODE_TYPE_P(nodes.nodes[nodes.size - 1], PM_SPLAT_NODE)) {
17912 node = UP(pm_find_pattern_node_create(parser, &nodes));
17913
17914 if (nodes.size == 2) {
17915 pm_parser_err_node(parser, node, PM_ERR_PATTERN_FIND_MISSING_INNER);
17916 }
17917 } else {
17918 node = UP(pm_array_pattern_node_node_list_create(parser, &nodes));
17919
17920 if (leading_rest && trailing_rest) {
17921 pm_parser_err_node(parser, node, PM_ERR_PATTERN_ARRAY_MULTIPLE_RESTS);
17922 }
17923 }
17924
17925 // nodes.nodes is arena-allocated; no explicit free needed.
17926 } else if (leading_rest) {
17927 // Otherwise, if we parsed a single splat pattern, then we know we have
17928 // an array pattern, so we can go ahead and create that node.
17929 node = UP(pm_array_pattern_node_rest_create(parser, node));
17930 }
17931
17932 return node;
17933}
17934
17940static PRISM_INLINE void
17941parse_negative_numeric(pm_node_t *node) {
17942 switch (PM_NODE_TYPE(node)) {
17943 case PM_INTEGER_NODE: {
17944 pm_integer_node_t *cast = (pm_integer_node_t *) node;
17945 cast->base.location.start--;
17946 cast->base.location.length++;
17947 cast->value.negative = true;
17948 break;
17949 }
17950 case PM_FLOAT_NODE: {
17951 pm_float_node_t *cast = (pm_float_node_t *) node;
17952 cast->base.location.start--;
17953 cast->base.location.length++;
17954 cast->value = -cast->value;
17955 break;
17956 }
17957 case PM_RATIONAL_NODE: {
17958 pm_rational_node_t *cast = (pm_rational_node_t *) node;
17959 cast->base.location.start--;
17960 cast->base.location.length++;
17961 cast->numerator.negative = true;
17962 break;
17963 }
17964 case PM_IMAGINARY_NODE:
17965 node->location.start--;
17966 node->location.length++;
17967 parse_negative_numeric(((pm_imaginary_node_t *) node)->numeric);
17968 break;
17969 default:
17970 assert(false && "unreachable");
17971 break;
17972 }
17973}
17974
17980static void
17981pm_parser_err_prefix(pm_parser_t *parser, pm_diagnostic_id_t diag_id) {
17982 switch (diag_id) {
17983 case PM_ERR_HASH_KEY: {
17984 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->previous, diag_id, pm_token_str(parser->previous.type));
17985 break;
17986 }
17987 case PM_ERR_HASH_VALUE:
17988 case PM_ERR_EXPECT_EXPRESSION_AFTER_OPERATOR: {
17989 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->current, diag_id, pm_token_str(parser->current.type));
17990 break;
17991 }
17992 case PM_ERR_UNARY_RECEIVER: {
17993 const char *human = (parser->current.type == PM_TOKEN_EOF ? "end-of-input" : pm_token_str(parser->current.type));
17994 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->previous, diag_id, human, parser->previous.start[0]);
17995 break;
17996 }
17997 case PM_ERR_UNARY_DISALLOWED:
17998 case PM_ERR_EXPECT_ARGUMENT: {
17999 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->current, diag_id, pm_token_str(parser->current.type));
18000 break;
18001 }
18002 default:
18003 pm_parser_err_previous(parser, diag_id);
18004 break;
18005 }
18006}
18007
18011static void
18012parse_retry(pm_parser_t *parser, const pm_node_t *node) {
18013#define CONTEXT_NONE 0
18014#define CONTEXT_THROUGH_ENSURE 1
18015#define CONTEXT_THROUGH_ELSE 2
18016
18017 pm_context_node_t *context_node = parser->current_context;
18018 int context = CONTEXT_NONE;
18019
18020 while (context_node != NULL) {
18021 switch (context_node->context) {
18022 case PM_CONTEXT_BEGIN_RESCUE:
18023 case PM_CONTEXT_BLOCK_RESCUE:
18024 case PM_CONTEXT_CLASS_RESCUE:
18025 case PM_CONTEXT_DEF_RESCUE:
18026 case PM_CONTEXT_LAMBDA_RESCUE:
18027 case PM_CONTEXT_MODULE_RESCUE:
18028 case PM_CONTEXT_SCLASS_RESCUE:
18029 case PM_CONTEXT_DEFINED:
18030 case PM_CONTEXT_RESCUE_MODIFIER:
18031 // These are the good cases. We're allowed to have a retry here.
18032 return;
18033 case PM_CONTEXT_CLASS:
18034 case PM_CONTEXT_DEF:
18035 case PM_CONTEXT_DEF_PARAMS:
18036 case PM_CONTEXT_MAIN:
18037 case PM_CONTEXT_MODULE:
18038 case PM_CONTEXT_PREEXE:
18039 case PM_CONTEXT_SCLASS:
18040 // These are the bad cases. We're not allowed to have a retry in
18041 // these contexts.
18042 if (context == CONTEXT_NONE) {
18043 pm_parser_err_node(parser, node, PM_ERR_INVALID_RETRY_WITHOUT_RESCUE);
18044 } else if (context == CONTEXT_THROUGH_ENSURE) {
18045 pm_parser_err_node(parser, node, PM_ERR_INVALID_RETRY_AFTER_ENSURE);
18046 } else if (context == CONTEXT_THROUGH_ELSE) {
18047 pm_parser_err_node(parser, node, PM_ERR_INVALID_RETRY_AFTER_ELSE);
18048 }
18049 return;
18050 case PM_CONTEXT_BEGIN_ELSE:
18051 case PM_CONTEXT_BLOCK_ELSE:
18052 case PM_CONTEXT_CLASS_ELSE:
18053 case PM_CONTEXT_DEF_ELSE:
18054 case PM_CONTEXT_LAMBDA_ELSE:
18055 case PM_CONTEXT_MODULE_ELSE:
18056 case PM_CONTEXT_SCLASS_ELSE:
18057 // These are also bad cases, but with a more specific error
18058 // message indicating the else.
18059 context = CONTEXT_THROUGH_ELSE;
18060 break;
18061 case PM_CONTEXT_BEGIN_ENSURE:
18062 case PM_CONTEXT_BLOCK_ENSURE:
18063 case PM_CONTEXT_CLASS_ENSURE:
18064 case PM_CONTEXT_DEF_ENSURE:
18065 case PM_CONTEXT_LAMBDA_ENSURE:
18066 case PM_CONTEXT_MODULE_ENSURE:
18067 case PM_CONTEXT_SCLASS_ENSURE:
18068 // These are also bad cases, but with a more specific error
18069 // message indicating the ensure.
18070 context = CONTEXT_THROUGH_ENSURE;
18071 break;
18072 case PM_CONTEXT_NONE:
18073 case PM_CONTEXT_MAXIMUM:
18074 // This case should never happen.
18075 assert(false && "unreachable");
18076 break;
18077 case PM_CONTEXT_BEGIN:
18078 case PM_CONTEXT_BLOCK_BRACES:
18079 case PM_CONTEXT_BLOCK_KEYWORDS:
18080 case PM_CONTEXT_BLOCK_PARAMETERS:
18081 case PM_CONTEXT_CASE_IN:
18082 case PM_CONTEXT_CASE_WHEN:
18083 case PM_CONTEXT_DEFAULT_PARAMS:
18084 case PM_CONTEXT_ELSE:
18085 case PM_CONTEXT_ELSIF:
18086 case PM_CONTEXT_EMBEXPR:
18087 case PM_CONTEXT_FOR_INDEX:
18088 case PM_CONTEXT_FOR:
18089 case PM_CONTEXT_IF:
18090 case PM_CONTEXT_LAMBDA_BRACES:
18091 case PM_CONTEXT_LAMBDA_DO_END:
18092 case PM_CONTEXT_LOOP_PREDICATE:
18093 case PM_CONTEXT_MULTI_TARGET:
18094 case PM_CONTEXT_PARENS:
18095 case PM_CONTEXT_POSTEXE:
18096 case PM_CONTEXT_PREDICATE:
18097 case PM_CONTEXT_TERNARY:
18098 case PM_CONTEXT_UNLESS:
18099 case PM_CONTEXT_UNTIL:
18100 case PM_CONTEXT_WHILE:
18101 // In these contexts we should continue walking up the list of
18102 // contexts.
18103 break;
18104 }
18105
18106 context_node = context_node->prev;
18107 }
18108
18109#undef CONTEXT_NONE
18110#undef CONTEXT_ENSURE
18111#undef CONTEXT_ELSE
18112}
18113
18117static void
18118parse_yield(pm_parser_t *parser, const pm_node_t *node) {
18119 pm_context_node_t *context_node = parser->current_context;
18120
18121 while (context_node != NULL) {
18122 switch (context_node->context) {
18123 case PM_CONTEXT_DEF:
18124 case PM_CONTEXT_DEF_PARAMS:
18125 case PM_CONTEXT_DEFINED:
18126 case PM_CONTEXT_DEF_ENSURE:
18127 case PM_CONTEXT_DEF_RESCUE:
18128 case PM_CONTEXT_DEF_ELSE:
18129 // These are the good cases. We're allowed to have a block exit
18130 // in these contexts.
18131 return;
18132 case PM_CONTEXT_CLASS:
18133 case PM_CONTEXT_CLASS_ENSURE:
18134 case PM_CONTEXT_CLASS_RESCUE:
18135 case PM_CONTEXT_CLASS_ELSE:
18136 case PM_CONTEXT_MAIN:
18137 case PM_CONTEXT_MODULE:
18138 case PM_CONTEXT_MODULE_ENSURE:
18139 case PM_CONTEXT_MODULE_RESCUE:
18140 case PM_CONTEXT_MODULE_ELSE:
18141 case PM_CONTEXT_SCLASS:
18142 case PM_CONTEXT_SCLASS_RESCUE:
18143 case PM_CONTEXT_SCLASS_ENSURE:
18144 case PM_CONTEXT_SCLASS_ELSE:
18145 // These are the bad cases. We're not allowed to have a retry in
18146 // these contexts.
18147 pm_parser_err_node(parser, node, PM_ERR_INVALID_YIELD);
18148 return;
18149 case PM_CONTEXT_NONE:
18150 case PM_CONTEXT_MAXIMUM:
18151 // This case should never happen.
18152 assert(false && "unreachable");
18153 break;
18154 case PM_CONTEXT_BEGIN:
18155 case PM_CONTEXT_BEGIN_ELSE:
18156 case PM_CONTEXT_BEGIN_ENSURE:
18157 case PM_CONTEXT_BEGIN_RESCUE:
18158 case PM_CONTEXT_BLOCK_BRACES:
18159 case PM_CONTEXT_BLOCK_KEYWORDS:
18160 case PM_CONTEXT_BLOCK_ELSE:
18161 case PM_CONTEXT_BLOCK_ENSURE:
18162 case PM_CONTEXT_BLOCK_PARAMETERS:
18163 case PM_CONTEXT_BLOCK_RESCUE:
18164 case PM_CONTEXT_CASE_IN:
18165 case PM_CONTEXT_CASE_WHEN:
18166 case PM_CONTEXT_DEFAULT_PARAMS:
18167 case PM_CONTEXT_ELSE:
18168 case PM_CONTEXT_ELSIF:
18169 case PM_CONTEXT_EMBEXPR:
18170 case PM_CONTEXT_FOR_INDEX:
18171 case PM_CONTEXT_FOR:
18172 case PM_CONTEXT_IF:
18173 case PM_CONTEXT_LAMBDA_BRACES:
18174 case PM_CONTEXT_LAMBDA_DO_END:
18175 case PM_CONTEXT_LAMBDA_ELSE:
18176 case PM_CONTEXT_LAMBDA_ENSURE:
18177 case PM_CONTEXT_LAMBDA_RESCUE:
18178 case PM_CONTEXT_LOOP_PREDICATE:
18179 case PM_CONTEXT_MULTI_TARGET:
18180 case PM_CONTEXT_PARENS:
18181 case PM_CONTEXT_POSTEXE:
18182 case PM_CONTEXT_PREDICATE:
18183 case PM_CONTEXT_PREEXE:
18184 case PM_CONTEXT_RESCUE_MODIFIER:
18185 case PM_CONTEXT_TERNARY:
18186 case PM_CONTEXT_UNLESS:
18187 case PM_CONTEXT_UNTIL:
18188 case PM_CONTEXT_WHILE:
18189 // In these contexts we should continue walking up the list of
18190 // contexts.
18191 break;
18192 }
18193
18194 context_node = context_node->prev;
18195 }
18196}
18197
18202static pm_node_t *
18203parse_case(pm_parser_t *parser, uint8_t flags, uint16_t depth) {
18204 size_t opening_newline_index = token_newline_index(parser);
18205 parser_lex(parser);
18206
18207 pm_token_t case_keyword = parser->previous;
18208 pm_node_t *predicate = NULL;
18209
18210 pm_node_list_t current_block_exits = { 0 };
18211 pm_node_list_t *previous_block_exits = push_block_exits(parser, &current_block_exits);
18212
18213 if (accept2(parser, PM_TOKEN_NEWLINE, PM_TOKEN_SEMICOLON)) {
18214 while (accept2(parser, PM_TOKEN_NEWLINE, PM_TOKEN_SEMICOLON));
18215 predicate = NULL;
18216 } else if (match3(parser, PM_TOKEN_KEYWORD_WHEN, PM_TOKEN_KEYWORD_IN, PM_TOKEN_KEYWORD_END)) {
18217 predicate = NULL;
18218 } else if (!token_begins_expression_p(parser->current.type)) {
18219 predicate = NULL;
18220 } else {
18221 predicate = parse_value_expression(parser, PM_BINDING_POWER_COMPOSITION, (uint8_t) ((flags & PM_PARSE_ACCEPTS_DO_BLOCK) | PM_PARSE_ACCEPTS_COMMAND_CALL), PM_ERR_CASE_EXPRESSION_AFTER_CASE, (uint16_t) (depth + 1));
18222 while (accept2(parser, PM_TOKEN_NEWLINE, PM_TOKEN_SEMICOLON));
18223 }
18224
18225 if (match1(parser, PM_TOKEN_KEYWORD_END)) {
18226 parser_warn_indentation_mismatch(parser, opening_newline_index, &case_keyword, false, false);
18227 parser_lex(parser);
18228 pop_block_exits(parser, previous_block_exits);
18229 pm_parser_err_token(parser, &case_keyword, PM_ERR_CASE_MISSING_CONDITIONS);
18230 return UP(pm_case_node_create(parser, &case_keyword, predicate, &parser->previous));
18231 }
18232
18233 /* At this point we can create a case node, though we don't yet know if it
18234 * is a case-in or case-when node. */
18235 pm_node_t *node;
18236
18237 if (match1(parser, PM_TOKEN_KEYWORD_WHEN)) {
18238 pm_case_node_t *case_node = pm_case_node_create(parser, &case_keyword, predicate, NULL);
18239 pm_static_literals_t literals = { 0 };
18240
18241 /* At this point we've seen a when keyword, so we know this is a
18242 * case-when node. We will continue to parse the when nodes until we hit
18243 * the end of the list. */
18244 while (match1(parser, PM_TOKEN_KEYWORD_WHEN)) {
18245 parser_warn_indentation_mismatch(parser, opening_newline_index, &case_keyword, false, true);
18246 parser_lex(parser);
18247
18248 pm_token_t when_keyword = parser->previous;
18249 pm_when_node_t *when_node = pm_when_node_create(parser, &when_keyword);
18250
18251 do {
18252 if (accept1(parser, PM_TOKEN_USTAR)) {
18253 pm_token_t operator = parser->previous;
18254 pm_node_t *expression = parse_value_expression(parser, PM_BINDING_POWER_DEFINED, flags & PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_EXPECT_EXPRESSION_AFTER_STAR, (uint16_t) (depth + 1));
18255
18256 pm_splat_node_t *splat_node = pm_splat_node_create(parser, &operator, expression);
18257 pm_when_node_conditions_append(parser->arena, when_node, UP(splat_node));
18258
18259 if (PM_NODE_TYPE_P(expression, PM_ERROR_RECOVERY_NODE)) break;
18260 } else {
18261 pm_node_t *condition = parse_value_expression(parser, PM_BINDING_POWER_DEFINED, flags & PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_CASE_EXPRESSION_AFTER_WHEN, (uint16_t) (depth + 1));
18262 pm_when_node_conditions_append(parser->arena, when_node, condition);
18263
18264 /* If we found a missing node, then this is a syntax error
18265 * and we should stop looping. */
18266 if (PM_NODE_TYPE_P(condition, PM_ERROR_RECOVERY_NODE)) break;
18267
18268 /* If this is a string node, then we need to mark it as
18269 * frozen because when clause strings are frozen. */
18270 if (PM_NODE_TYPE_P(condition, PM_STRING_NODE)) {
18271 pm_node_flag_set(condition, PM_STRING_FLAGS_FROZEN | PM_NODE_FLAG_STATIC_LITERAL);
18272 } else if (PM_NODE_TYPE_P(condition, PM_SOURCE_FILE_NODE) && parser->version < PM_OPTIONS_VERSION_CRUBY_4_1) {
18273 pm_node_flag_set(condition, PM_NODE_FLAG_STATIC_LITERAL);
18274 }
18275
18276 pm_when_clause_static_literals_add(parser, &literals, condition);
18277 }
18278 } while (accept1(parser, PM_TOKEN_COMMA));
18279
18280 if (accept2(parser, PM_TOKEN_NEWLINE, PM_TOKEN_SEMICOLON)) {
18281 if (accept1(parser, PM_TOKEN_KEYWORD_THEN)) {
18282 pm_when_node_then_keyword_loc_set(parser, when_node, &parser->previous);
18283 }
18284 } else {
18285 expect1(parser, PM_TOKEN_KEYWORD_THEN, PM_ERR_EXPECT_WHEN_DELIMITER);
18286 pm_when_node_then_keyword_loc_set(parser, when_node, &parser->previous);
18287 }
18288
18289 if (!match3(parser, PM_TOKEN_KEYWORD_WHEN, PM_TOKEN_KEYWORD_ELSE, PM_TOKEN_KEYWORD_END)) {
18290 pm_statements_node_t *statements = parse_statements(parser, PM_CONTEXT_CASE_WHEN, (uint16_t) (depth + 1));
18291 if (statements != NULL) {
18292 pm_when_node_statements_set(when_node, statements);
18293 }
18294 }
18295
18296 pm_case_node_condition_append(parser->arena, case_node, UP(when_node));
18297 }
18298
18299 /* If we didn't parse any conditions (in or when) then we need to
18300 * indicate that we have an error. */
18301 if (case_node->conditions.size == 0) {
18302 pm_parser_err_token(parser, &case_keyword, PM_ERR_CASE_MISSING_CONDITIONS);
18303 }
18304
18305 pm_static_literals_free(&literals);
18306 node = UP(case_node);
18307 } else {
18308 pm_case_match_node_t *case_node = pm_case_match_node_create(parser, &case_keyword, predicate);
18309
18310 /* If this is a case-match node (i.e., it is a pattern matching case
18311 * statement) then we must have a predicate. */
18312 if (predicate == NULL) {
18313 pm_parser_err_token(parser, &case_keyword, PM_ERR_CASE_MATCH_MISSING_PREDICATE);
18314 }
18315
18316 /* At this point we expect that we're parsing a case-in node. We will
18317 * continue to parse the in nodes until we hit the end of the list. */
18318 while (match1(parser, PM_TOKEN_KEYWORD_IN)) {
18319 parser_warn_indentation_mismatch(parser, opening_newline_index, &case_keyword, false, true);
18320
18321 bool previous_pattern_matching_newlines = parser->pattern_matching_newlines;
18322 parser->pattern_matching_newlines = true;
18323
18324 lex_state_set(parser, PM_LEX_STATE_BEG | PM_LEX_STATE_LABEL);
18325 parser->command_start = false;
18326 parser_lex(parser);
18327
18328 pm_token_t in_keyword = parser->previous;
18329
18330 pm_constant_id_set_t captures = { 0 };
18331 pm_node_t *pattern = parse_pattern(parser, &captures, PM_PARSE_PATTERN_TOP | PM_PARSE_PATTERN_MULTI, PM_ERR_PATTERN_EXPRESSION_AFTER_IN, (uint16_t) (depth + 1));
18332
18333 parser->pattern_matching_newlines = previous_pattern_matching_newlines;
18334
18335 /* Since we're in the top-level of the case-in node we need to
18336 * check for guard clauses in the form of `if` or `unless`
18337 * statements. */
18338 if (accept1(parser, PM_TOKEN_KEYWORD_IF_MODIFIER)) {
18339 pm_token_t keyword = parser->previous;
18340 pm_node_t *predicate = parse_value_expression(parser, PM_BINDING_POWER_COMPOSITION, (uint8_t) ((flags & PM_PARSE_ACCEPTS_DO_BLOCK) | PM_PARSE_ACCEPTS_COMMAND_CALL), PM_ERR_CONDITIONAL_IF_PREDICATE, (uint16_t) (depth + 1));
18341 pattern = UP(pm_if_node_modifier_create(parser, pattern, &keyword, predicate));
18342 } else if (accept1(parser, PM_TOKEN_KEYWORD_UNLESS_MODIFIER)) {
18343 pm_token_t keyword = parser->previous;
18344 pm_node_t *predicate = parse_value_expression(parser, PM_BINDING_POWER_COMPOSITION, (uint8_t) ((flags & PM_PARSE_ACCEPTS_DO_BLOCK) | PM_PARSE_ACCEPTS_COMMAND_CALL), PM_ERR_CONDITIONAL_UNLESS_PREDICATE, (uint16_t) (depth + 1));
18345 pattern = UP(pm_unless_node_modifier_create(parser, pattern, &keyword, predicate));
18346 }
18347
18348 /* Now we need to check for the terminator of the in node's pattern.
18349 * It can be a newline or semicolon optionally followed by a `then`
18350 * keyword. */
18351 pm_token_t then_keyword = { 0 };
18352 if (accept2(parser, PM_TOKEN_NEWLINE, PM_TOKEN_SEMICOLON)) {
18353 if (accept1(parser, PM_TOKEN_KEYWORD_THEN)) {
18354 then_keyword = parser->previous;
18355 }
18356 } else {
18357 expect1(parser, PM_TOKEN_KEYWORD_THEN, PM_ERR_EXPECT_IN_DELIMITER);
18358 then_keyword = parser->previous;
18359 }
18360
18361 /* Now we can actually parse the statements associated with the in
18362 * node. */
18363 pm_statements_node_t *statements;
18364 if (match3(parser, PM_TOKEN_KEYWORD_IN, PM_TOKEN_KEYWORD_ELSE, PM_TOKEN_KEYWORD_END)) {
18365 statements = NULL;
18366 } else {
18367 statements = parse_statements(parser, PM_CONTEXT_CASE_IN, (uint16_t) (depth + 1));
18368 }
18369
18370 /* Now that we have the full pattern and statements, we can create
18371 * the node and attach it to the case node. */
18372 pm_node_t *condition = UP(pm_in_node_create(parser, pattern, statements, &in_keyword, NTOK2PTR(then_keyword)));
18373 pm_case_match_node_condition_append(parser->arena, case_node, condition);
18374 }
18375
18376 /* If we didn't parse any conditions (in or when) then we need to
18377 * indicate that we have an error. */
18378 if (case_node->conditions.size == 0) {
18379 pm_parser_err_token(parser, &case_keyword, PM_ERR_CASE_MISSING_CONDITIONS);
18380 }
18381
18382 node = UP(case_node);
18383 }
18384
18385 accept2(parser, PM_TOKEN_NEWLINE, PM_TOKEN_SEMICOLON);
18386 if (accept1(parser, PM_TOKEN_KEYWORD_ELSE)) {
18387 pm_token_t else_keyword = parser->previous;
18388 pm_else_node_t *else_node;
18389
18390 if (!match1(parser, PM_TOKEN_KEYWORD_END)) {
18391 else_node = pm_else_node_create(parser, &else_keyword, parse_statements(parser, PM_CONTEXT_ELSE, (uint16_t) (depth + 1)), &parser->current);
18392 } else {
18393 else_node = pm_else_node_create(parser, &else_keyword, NULL, &parser->current);
18394 }
18395
18396 if (PM_NODE_TYPE_P(node, PM_CASE_NODE)) {
18397 pm_case_node_else_clause_set((pm_case_node_t *) node, else_node);
18398 } else {
18399 pm_case_match_node_else_clause_set((pm_case_match_node_t *) node, else_node);
18400 }
18401 }
18402
18403 parser_warn_indentation_mismatch(parser, opening_newline_index, &case_keyword, false, false);
18404 expect1_opening(parser, PM_TOKEN_KEYWORD_END, PM_ERR_CASE_TERM, &case_keyword);
18405
18406 if (PM_NODE_TYPE_P(node, PM_CASE_NODE)) {
18407 pm_case_node_end_keyword_loc_set(parser, (pm_case_node_t *) node, &parser->previous);
18408 } else {
18409 pm_case_match_node_end_keyword_loc_set(parser, (pm_case_match_node_t *) node, &parser->previous);
18410 }
18411
18412 pop_block_exits(parser, previous_block_exits);
18413 return node;
18414}
18415
18420static pm_node_t *
18421parse_class(pm_parser_t *parser, uint8_t flags, uint16_t depth) {
18422 size_t opening_newline_index = token_newline_index(parser);
18423 parser_lex(parser);
18424
18425 pm_token_t class_keyword = parser->previous;
18426 pm_do_loop_stack_push(parser, false);
18427
18428 pm_node_list_t current_block_exits = { 0 };
18429 pm_node_list_t *previous_block_exits = push_block_exits(parser, &current_block_exits);
18430
18431 if (accept1(parser, PM_TOKEN_LESS_LESS)) {
18432 pm_token_t operator = parser->previous;
18433 pm_node_t *expression = parse_value_expression(parser, PM_BINDING_POWER_COMPOSITION, (uint8_t) ((flags & PM_PARSE_ACCEPTS_DO_BLOCK) | PM_PARSE_ACCEPTS_COMMAND_CALL), PM_ERR_EXPECT_EXPRESSION_AFTER_LESS_LESS, (uint16_t) (depth + 1));
18434
18435 pm_parser_scope_push(parser, true);
18436 if (!match2(parser, PM_TOKEN_NEWLINE, PM_TOKEN_SEMICOLON)) {
18437 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->current, PM_ERR_EXPECT_SINGLETON_CLASS_DELIMITER, pm_token_str(parser->current.type));
18438 }
18439
18440 pm_node_t *statements = NULL;
18441 if (!match4(parser, PM_TOKEN_KEYWORD_RESCUE, PM_TOKEN_KEYWORD_ENSURE, PM_TOKEN_KEYWORD_ELSE, PM_TOKEN_KEYWORD_END)) {
18442 pm_accepts_block_stack_push(parser, true);
18443 statements = UP(parse_statements(parser, PM_CONTEXT_SCLASS, (uint16_t) (depth + 1)));
18444 pm_accepts_block_stack_pop(parser);
18445 }
18446
18447 if (match2(parser, PM_TOKEN_KEYWORD_RESCUE, PM_TOKEN_KEYWORD_ENSURE)) {
18448 assert(statements == NULL || PM_NODE_TYPE_P(statements, PM_STATEMENTS_NODE));
18449 statements = UP(parse_rescues_implicit_begin(parser, opening_newline_index, &class_keyword, class_keyword.start, (pm_statements_node_t *) statements, PM_RESCUES_SCLASS, (uint16_t) (depth + 1)));
18450 } else {
18451 parser_warn_indentation_mismatch(parser, opening_newline_index, &class_keyword, false, false);
18452 }
18453
18454 expect1_opening(parser, PM_TOKEN_KEYWORD_END, PM_ERR_CLASS_TERM, &class_keyword);
18455
18456 pm_constant_id_list_t locals;
18457 pm_locals_order(parser, &parser->current_scope->locals, &locals, false);
18458
18459 pm_parser_scope_pop(parser);
18460 pm_do_loop_stack_pop(parser);
18461
18462 flush_block_exits(parser, previous_block_exits);
18463 return UP(pm_singleton_class_node_create(parser, &locals, &class_keyword, &operator, expression, statements, &parser->previous));
18464 }
18465
18466 pm_node_t *constant_path = parse_expression(parser, PM_BINDING_POWER_INDEX, flags & PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_CLASS_NAME, (uint16_t) (depth + 1));
18467 pm_token_t name = parser->previous;
18468 if (name.type != PM_TOKEN_CONSTANT) {
18469 pm_parser_err_token(parser, &name, PM_ERR_CLASS_NAME);
18470 }
18471
18472 pm_token_t inheritance_operator = { 0 };
18473 pm_node_t *superclass;
18474
18475 if (match1(parser, PM_TOKEN_LESS)) {
18476 inheritance_operator = parser->current;
18477 lex_state_set(parser, PM_LEX_STATE_BEG);
18478
18479 parser->command_start = true;
18480 parser_lex(parser);
18481
18482 superclass = parse_value_expression(parser, PM_BINDING_POWER_COMPOSITION, (uint8_t) ((flags & PM_PARSE_ACCEPTS_DO_BLOCK) | PM_PARSE_ACCEPTS_COMMAND_CALL), PM_ERR_CLASS_SUPERCLASS, (uint16_t) (depth + 1));
18483 } else {
18484 superclass = NULL;
18485 }
18486
18487 pm_parser_scope_push(parser, true);
18488
18489 if (inheritance_operator.start != NULL) {
18490 expect2(parser, PM_TOKEN_NEWLINE, PM_TOKEN_SEMICOLON, PM_ERR_CLASS_UNEXPECTED_END);
18491 } else {
18492 accept2(parser, PM_TOKEN_NEWLINE, PM_TOKEN_SEMICOLON);
18493 }
18494 pm_node_t *statements = NULL;
18495
18496 if (!match4(parser, PM_TOKEN_KEYWORD_RESCUE, PM_TOKEN_KEYWORD_ENSURE, PM_TOKEN_KEYWORD_ELSE, PM_TOKEN_KEYWORD_END)) {
18497 pm_accepts_block_stack_push(parser, true);
18498 statements = UP(parse_statements(parser, PM_CONTEXT_CLASS, (uint16_t) (depth + 1)));
18499 pm_accepts_block_stack_pop(parser);
18500 }
18501
18502 if (match2(parser, PM_TOKEN_KEYWORD_RESCUE, PM_TOKEN_KEYWORD_ENSURE)) {
18503 assert(statements == NULL || PM_NODE_TYPE_P(statements, PM_STATEMENTS_NODE));
18504 statements = UP(parse_rescues_implicit_begin(parser, opening_newline_index, &class_keyword, class_keyword.start, (pm_statements_node_t *) statements, PM_RESCUES_CLASS, (uint16_t) (depth + 1)));
18505 } else {
18506 parser_warn_indentation_mismatch(parser, opening_newline_index, &class_keyword, false, false);
18507 }
18508
18509 expect1_opening(parser, PM_TOKEN_KEYWORD_END, PM_ERR_CLASS_TERM, &class_keyword);
18510
18511 if (context_def_p(parser)) {
18512 pm_parser_err_token(parser, &class_keyword, PM_ERR_CLASS_IN_METHOD);
18513 }
18514
18515 pm_constant_id_list_t locals;
18516 pm_locals_order(parser, &parser->current_scope->locals, &locals, false);
18517
18518 pm_parser_scope_pop(parser);
18519 pm_do_loop_stack_pop(parser);
18520
18521 if (!PM_NODE_TYPE_P(constant_path, PM_CONSTANT_PATH_NODE) && !(PM_NODE_TYPE_P(constant_path, PM_CONSTANT_READ_NODE))) {
18522 pm_parser_err_node(parser, constant_path, PM_ERR_CLASS_NAME);
18523 if (!PM_NODE_TYPE_P(constant_path, PM_ERROR_RECOVERY_NODE)) {
18524 constant_path = UP(pm_error_recovery_node_create_unexpected(parser, constant_path));
18525 }
18526 }
18527
18528 pop_block_exits(parser, previous_block_exits);
18529 return UP(pm_class_node_create(parser, &locals, &class_keyword, constant_path, &name, NTOK2PTR(inheritance_operator), superclass, statements, &parser->previous));
18530}
18531
18535static pm_node_t *
18536parse_def(pm_parser_t *parser, pm_binding_power_t binding_power, uint8_t flags, uint16_t depth) {
18537 pm_node_list_t current_block_exits = { 0 };
18538 pm_node_list_t *previous_block_exits = push_block_exits(parser, &current_block_exits);
18539
18540 pm_token_t def_keyword = parser->current;
18541 size_t opening_newline_index = token_newline_index(parser);
18542
18543 pm_node_t *receiver = NULL;
18544 pm_token_t operator = { 0 };
18545 pm_token_t name;
18546
18547 /* This context is necessary for lexing `...` in a bare params correctly. It
18548 * must be pushed before lexing the first param, so it is here. */
18549 context_push(parser, PM_CONTEXT_DEF_PARAMS);
18550 parser_lex(parser);
18551
18552 /* This will be false if the method name is not a valid identifier but could
18553 * be followed by an operator. */
18554 bool valid_name = true;
18555
18556 switch (parser->current.type) {
18557 case PM_CASE_OPERATOR:
18558 pm_parser_scope_push(parser, true);
18559 lex_state_set(parser, PM_LEX_STATE_ENDFN);
18560 parser_lex(parser);
18561
18562 name = parser->previous;
18563 break;
18564 case PM_TOKEN_IDENTIFIER: {
18565 parser_lex(parser);
18566
18567 if (match2(parser, PM_TOKEN_DOT, PM_TOKEN_COLON_COLON)) {
18568 receiver = parse_variable_call(parser);
18569
18570 pm_parser_scope_push(parser, true);
18571 lex_state_set(parser, PM_LEX_STATE_FNAME);
18572 parser_lex(parser);
18573
18574 operator = parser->previous;
18575 name = parse_method_definition_name(parser);
18576 } else {
18577 pm_refute_numbered_parameter(parser, PM_TOKEN_START(parser, &parser->previous), PM_TOKEN_LENGTH(&parser->previous));
18578 pm_parser_scope_push(parser, true);
18579
18580 name = parser->previous;
18581 }
18582
18583 break;
18584 }
18585 case PM_TOKEN_INSTANCE_VARIABLE:
18586 case PM_TOKEN_CLASS_VARIABLE:
18587 case PM_TOKEN_GLOBAL_VARIABLE:
18588 valid_name = false;
18590 case PM_TOKEN_CONSTANT:
18591 case PM_TOKEN_KEYWORD_NIL:
18592 case PM_TOKEN_KEYWORD_SELF:
18593 case PM_TOKEN_KEYWORD_TRUE:
18594 case PM_TOKEN_KEYWORD_FALSE:
18595 case PM_TOKEN_KEYWORD___FILE__:
18596 case PM_TOKEN_KEYWORD___LINE__:
18597 case PM_TOKEN_KEYWORD___ENCODING__: {
18598 pm_parser_scope_push(parser, true);
18599 parser_lex(parser);
18600
18601 pm_token_t identifier = parser->previous;
18602
18603 if (match2(parser, PM_TOKEN_DOT, PM_TOKEN_COLON_COLON)) {
18604 lex_state_set(parser, PM_LEX_STATE_FNAME);
18605 parser_lex(parser);
18606 operator = parser->previous;
18607
18608 switch (identifier.type) {
18609 case PM_TOKEN_CONSTANT:
18610 receiver = UP(pm_constant_read_node_create(parser, &identifier));
18611 break;
18612 case PM_TOKEN_INSTANCE_VARIABLE:
18613 receiver = UP(pm_instance_variable_read_node_create(parser, &identifier));
18614 break;
18615 case PM_TOKEN_CLASS_VARIABLE:
18616 receiver = UP(pm_class_variable_read_node_create(parser, &identifier));
18617 break;
18618 case PM_TOKEN_GLOBAL_VARIABLE:
18619 receiver = UP(pm_global_variable_read_node_create(parser, &identifier));
18620 break;
18621 case PM_TOKEN_KEYWORD_NIL:
18622 receiver = UP(pm_nil_node_create(parser, &identifier));
18623 break;
18624 case PM_TOKEN_KEYWORD_SELF:
18625 receiver = UP(pm_self_node_create(parser, &identifier));
18626 break;
18627 case PM_TOKEN_KEYWORD_TRUE:
18628 receiver = UP(pm_true_node_create(parser, &identifier));
18629 break;
18630 case PM_TOKEN_KEYWORD_FALSE:
18631 receiver = UP(pm_false_node_create(parser, &identifier));
18632 break;
18633 case PM_TOKEN_KEYWORD___FILE__:
18634 receiver = UP(pm_source_file_node_create(parser, &identifier));
18635 break;
18636 case PM_TOKEN_KEYWORD___LINE__:
18637 receiver = UP(pm_source_line_node_create(parser, &identifier));
18638 break;
18639 case PM_TOKEN_KEYWORD___ENCODING__:
18640 receiver = UP(pm_source_encoding_node_create(parser, &identifier));
18641 break;
18642 default:
18643 break;
18644 }
18645
18646 name = parse_method_definition_name(parser);
18647 } else {
18648 if (!valid_name) {
18649 PM_PARSER_ERR_TOKEN_FORMAT(parser, &identifier, PM_ERR_DEF_NAME, pm_token_str(identifier.type));
18650 }
18651
18652 name = identifier;
18653 }
18654 break;
18655 }
18656 case PM_TOKEN_PARENTHESIS_LEFT: {
18657 /* The current context is `PM_CONTEXT_DEF_PARAMS`, however the inner
18658 * expression of this parenthesis should not be processed under this
18659 * context. Thus, the context is popped here. */
18660 context_pop(parser);
18661 parser_lex(parser);
18662
18663 pm_token_t lparen = parser->previous;
18664 pm_node_t *expression = parse_value_expression(parser, PM_BINDING_POWER_COMPOSITION, (uint8_t) ((flags & PM_PARSE_ACCEPTS_DO_BLOCK) | PM_PARSE_ACCEPTS_COMMAND_CALL), PM_ERR_DEF_RECEIVER, (uint16_t) (depth + 1));
18665
18666 accept1(parser, PM_TOKEN_NEWLINE);
18667 expect1(parser, PM_TOKEN_PARENTHESIS_RIGHT, PM_ERR_EXPECT_RPAREN);
18668 pm_token_t rparen = parser->previous;
18669
18670 lex_state_set(parser, PM_LEX_STATE_FNAME);
18671 expect2(parser, PM_TOKEN_DOT, PM_TOKEN_COLON_COLON, PM_ERR_DEF_RECEIVER_TERM);
18672
18673 operator = parser->previous;
18674 receiver = UP(pm_parentheses_node_create(parser, &lparen, expression, &rparen, 0));
18675
18676 /* To push `PM_CONTEXT_DEF_PARAMS` again is for the same reason as
18677 * described the above. */
18678 pm_parser_scope_push(parser, true);
18679 context_push(parser, PM_CONTEXT_DEF_PARAMS);
18680 name = parse_method_definition_name(parser);
18681 break;
18682 }
18683 default:
18684 pm_parser_scope_push(parser, true);
18685 name = parse_method_definition_name(parser);
18686 break;
18687 }
18688
18689 pm_token_t lparen = { 0 };
18690 pm_token_t rparen = { 0 };
18691 pm_parameters_node_t *params;
18692
18693 bool accept_endless_def = true;
18694 switch (parser->current.type) {
18695 case PM_TOKEN_PARENTHESIS_LEFT: {
18696 parser_lex(parser);
18697 lparen = parser->previous;
18698
18699 if (match1(parser, PM_TOKEN_PARENTHESIS_RIGHT)) {
18700 params = NULL;
18701 } else {
18702 /* https://bugs.ruby-lang.org/issues/19107 */
18703 bool allow_trailing_comma = parser->version >= PM_OPTIONS_VERSION_CRUBY_4_1;
18704 params = parse_parameters(
18705 parser,
18706 PM_BINDING_POWER_DEFINED,
18707 true,
18708 allow_trailing_comma,
18709 true,
18710 true,
18711 false,
18712 PM_ERR_ARGUMENT_NO_FORWARDING_ELLIPSES,
18713 (uint16_t) (depth + 1)
18714 );
18715 }
18716
18717 lex_state_set(parser, PM_LEX_STATE_BEG);
18718 parser->command_start = true;
18719
18720 context_pop(parser);
18721 if (!accept1(parser, PM_TOKEN_PARENTHESIS_RIGHT)) {
18722 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->current, PM_ERR_DEF_PARAMS_TERM_PAREN, pm_token_str(parser->current.type));
18723 parser->previous.start = parser->previous.end;
18724 parser->previous.type = 0;
18725 }
18726
18727 rparen = parser->previous;
18728 break;
18729 }
18730 case PM_CASE_PARAMETER: {
18731 /* If we're about to lex a label, we need to add the label state to
18732 * make sure the next newline is ignored. */
18733 if (parser->current.type == PM_TOKEN_LABEL) {
18734 lex_state_set(parser, parser->lex_state | PM_LEX_STATE_LABEL);
18735 }
18736
18737 params = parse_parameters(
18738 parser,
18739 PM_BINDING_POWER_DEFINED,
18740 false,
18741 false,
18742 true,
18743 true,
18744 false,
18745 PM_ERR_ARGUMENT_NO_FORWARDING_ELLIPSES,
18746 (uint16_t) (depth + 1)
18747 );
18748
18749 /* Reject `def * = 1` and similar. We have to specifically check for
18750 * them because they create ambiguity with optional arguments. */
18751 accept_endless_def = false;
18752
18753 context_pop(parser);
18754 break;
18755 }
18756 default: {
18757 params = NULL;
18758 context_pop(parser);
18759 break;
18760 }
18761 }
18762
18763 pm_node_t *statements = NULL;
18764 pm_token_t equal = { 0 };
18765 pm_token_t end_keyword = { 0 };
18766
18767 if (accept1(parser, PM_TOKEN_EQUAL)) {
18768 if (token_is_setter_name(&name)) {
18769 pm_parser_err_token(parser, &name, PM_ERR_DEF_ENDLESS_SETTER);
18770 }
18771 if (!accept_endless_def) {
18772 pm_parser_err_previous(parser, PM_ERR_DEF_ENDLESS_PARAMETERS);
18773 }
18774 if (
18775 parser->current_context->context == PM_CONTEXT_DEFAULT_PARAMS &&
18776 parser->current_context->prev->context == PM_CONTEXT_BLOCK_PARAMETERS
18777 ) {
18778 PM_PARSER_ERR_FORMAT(parser, PM_TOKEN_START(parser, &def_keyword), PM_TOKENS_LENGTH(&def_keyword, &parser->previous), PM_ERR_UNEXPECTED_PARAMETER_DEFAULT_VALUE, "endless method definition");
18779 }
18780 equal = parser->previous;
18781
18782 context_push(parser, PM_CONTEXT_DEF);
18783 pm_do_loop_stack_push(parser, false);
18784 statements = UP(pm_statements_node_create(parser));
18785
18786 uint8_t allow_flags;
18787 if (parser->version >= PM_OPTIONS_VERSION_CRUBY_4_0) {
18788 allow_flags = flags & PM_PARSE_ACCEPTS_COMMAND_CALL;
18789 } else {
18790 /* Allow `def foo = puts "Hello"` but not
18791 * `private def foo = puts "Hello"` */
18792 allow_flags = (binding_power == PM_BINDING_POWER_ASSIGNMENT || binding_power < PM_BINDING_POWER_COMPOSITION) ? PM_PARSE_ACCEPTS_COMMAND_CALL : 0;
18793 }
18794
18795 /* Inside a def body, we push true onto the accepts_block_stack so that
18796 * `do` is lexed as PM_TOKEN_KEYWORD_DO (which can only start a block
18797 * for primary-level constructs, not commands). During command argument
18798 * parsing, the stack is pushed to false, causing `do` to be lexed as
18799 * PM_TOKEN_KEYWORD_DO_BLOCK, which is not consumed inside the endless
18800 * def body and instead left for the outer context. A method definition
18801 * opens a fresh context all the way through its rescue modifier, so
18802 * this frame spans the rescue modifier value as well: the `do` in
18803 * `baz def f = a rescue z do end` lexes as a plain keyword that
18804 * attaches to `z` rather than to `baz`. */
18805 pm_accepts_block_stack_push(parser, true);
18806 pm_node_t *statement = parse_expression(parser, PM_BINDING_POWER_DEFINED + 1, allow_flags | PM_PARSE_IN_ENDLESS_DEF, PM_ERR_DEF_ENDLESS, (uint16_t) (depth + 1));
18807
18808 /* If an unconsumed PM_TOKEN_KEYWORD_DO follows the body, it is an error
18809 * (e.g., `def f = 1 do end`). PM_TOKEN_KEYWORD_DO_BLOCK is
18810 * intentionally not caught here — it should bubble up to the outer
18811 * context (e.g., `private def f = puts "Hello" do end` where the block
18812 * attaches to `private`). */
18813 if (accept1(parser, PM_TOKEN_KEYWORD_DO)) {
18814 pm_block_node_t *block = parse_block(parser, (uint16_t) (depth + 1));
18815 pm_parser_err_node(parser, UP(block), PM_ERR_DEF_ENDLESS_DO_BLOCK);
18816 }
18817
18818 /* Any number of rescue modifiers chain onto the body within the method
18819 * definition itself, associating to the left: `def f = a rescue b
18820 * rescue c` defines a method whose body is `(a rescue b) rescue c`,
18821 * rather than a rescue modifier guarding the definition. */
18822 while (accept1(parser, PM_TOKEN_KEYWORD_RESCUE_MODIFIER)) {
18823 context_push(parser, PM_CONTEXT_RESCUE_MODIFIER);
18824
18825 pm_token_t rescue_keyword = parser->previous;
18826
18827 /* In the Ruby grammar, the rescue value of an endless method
18828 * command excludes and/or and in/=>. */
18829 pm_node_t *value = parse_expression(parser, PM_BINDING_POWER_MATCH + 1, flags & PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_RESCUE_MODIFIER_VALUE, (uint16_t) (depth + 1));
18830 context_pop(parser);
18831
18832 statement = UP(pm_rescue_modifier_node_create(parser, statement, &rescue_keyword, value));
18833 }
18834
18835 pm_accepts_block_stack_pop(parser);
18836
18837 /* A nested endless def whose body is a command call (e.g.,
18838 * `def f = def g = foo bar`) is a command assignment and cannot appear
18839 * as a def body. */
18840 if (PM_NODE_TYPE_P(statement, PM_DEF_NODE) && pm_command_call_value_p(parser, statement)) {
18841 PM_PARSER_ERR_NODE_FORMAT(parser, statement, PM_ERR_EXPECT_EOL_AFTER_STATEMENT, pm_token_str(parser->current.type));
18842 }
18843
18844 pm_statements_node_body_append(parser, (pm_statements_node_t *) statements, statement, false);
18845 pm_do_loop_stack_pop(parser);
18846 context_pop(parser);
18847 } else {
18848 if (lparen.start == NULL) {
18849 lex_state_set(parser, PM_LEX_STATE_BEG);
18850 parser->command_start = true;
18851 expect2(parser, PM_TOKEN_NEWLINE, PM_TOKEN_SEMICOLON, PM_ERR_DEF_PARAMS_TERM);
18852 } else {
18853 accept2(parser, PM_TOKEN_NEWLINE, PM_TOKEN_SEMICOLON);
18854 }
18855
18856 pm_accepts_block_stack_push(parser, true);
18857 pm_do_loop_stack_push(parser, false);
18858
18859 if (!match4(parser, PM_TOKEN_KEYWORD_RESCUE, PM_TOKEN_KEYWORD_ENSURE, PM_TOKEN_KEYWORD_ELSE, PM_TOKEN_KEYWORD_END)) {
18860 pm_accepts_block_stack_push(parser, true);
18861 statements = UP(parse_statements(parser, PM_CONTEXT_DEF, (uint16_t) (depth + 1)));
18862 pm_accepts_block_stack_pop(parser);
18863 }
18864
18865 if (match3(parser, PM_TOKEN_KEYWORD_RESCUE, PM_TOKEN_KEYWORD_ENSURE, PM_TOKEN_KEYWORD_ELSE)) {
18866 assert(statements == NULL || PM_NODE_TYPE_P(statements, PM_STATEMENTS_NODE));
18867 statements = UP(parse_rescues_implicit_begin(parser, opening_newline_index, &def_keyword, def_keyword.start, (pm_statements_node_t *) statements, PM_RESCUES_DEF, (uint16_t) (depth + 1)));
18868 } else {
18869 parser_warn_indentation_mismatch(parser, opening_newline_index, &def_keyword, false, false);
18870 }
18871
18872 pm_accepts_block_stack_pop(parser);
18873 pm_do_loop_stack_pop(parser);
18874
18875 expect1_opening(parser, PM_TOKEN_KEYWORD_END, PM_ERR_DEF_TERM, &def_keyword);
18876 end_keyword = parser->previous;
18877 }
18878
18879 pm_constant_id_list_t locals;
18880 pm_locals_order(parser, &parser->current_scope->locals, &locals, false);
18881 pm_parser_scope_pop(parser);
18882
18883 /* If the final character is `@` as is the case when defining methods to
18884 * override the unary operators, we should ignore the @ in the same way we
18885 * do for symbols. */
18886 pm_constant_id_t name_id = pm_parser_constant_id_raw(parser, name.start, parse_operator_symbol_name(&name));
18887
18888 flush_block_exits(parser, previous_block_exits);
18889
18890 return UP(pm_def_node_create(
18891 parser,
18892 name_id,
18893 &name,
18894 receiver,
18895 params,
18896 statements,
18897 &locals,
18898 &def_keyword,
18899 NTOK2PTR(operator),
18900 NTOK2PTR(lparen),
18901 NTOK2PTR(rparen),
18902 NTOK2PTR(equal),
18903 NTOK2PTR(end_keyword)
18904 ));
18905}
18906
18910static pm_node_t *
18911parse_module(pm_parser_t *parser, uint8_t flags, uint16_t depth) {
18912 pm_node_list_t current_block_exits = { 0 };
18913 pm_node_list_t *previous_block_exits = push_block_exits(parser, &current_block_exits);
18914
18915 size_t opening_newline_index = token_newline_index(parser);
18916 parser_lex(parser);
18917 pm_token_t module_keyword = parser->previous;
18918
18919 pm_node_t *constant_path = parse_expression(parser, PM_BINDING_POWER_INDEX, flags & PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_MODULE_NAME, (uint16_t) (depth + 1));
18920 pm_token_t name;
18921
18922 /* If we can recover from a syntax error that occurred while parsing the
18923 * name of the module, then we'll handle that here. */
18924 if (PM_NODE_TYPE_P(constant_path, PM_ERROR_RECOVERY_NODE)) {
18925 pop_block_exits(parser, previous_block_exits);
18926
18927 pm_token_t missing = (pm_token_t) { .type = 0, .start = parser->previous.end, .end = parser->previous.end };
18928 return UP(pm_module_node_create(parser, NULL, &module_keyword, constant_path, &missing, NULL, &missing));
18929 }
18930
18931 while (accept1(parser, PM_TOKEN_COLON_COLON)) {
18932 pm_token_t double_colon = parser->previous;
18933
18934 expect1(parser, PM_TOKEN_CONSTANT, PM_ERR_CONSTANT_PATH_COLON_COLON_CONSTANT);
18935 constant_path = UP(pm_constant_path_node_create(parser, constant_path, &double_colon, &parser->previous));
18936 }
18937
18938 /* Here we retrieve the name of the module. If it wasn't a constant, then
18939 * it's possible that `module foo` was passed, which is a syntax error. We
18940 * handle that here as well. */
18941 name = parser->previous;
18942 if (name.type != PM_TOKEN_CONSTANT) {
18943 pm_parser_err_token(parser, &name, PM_ERR_MODULE_NAME);
18944 }
18945
18946 if (!PM_NODE_TYPE_P(constant_path, PM_CONSTANT_READ_NODE) && !PM_NODE_TYPE_P(constant_path, PM_CONSTANT_PATH_NODE) && !PM_NODE_TYPE_P(constant_path, PM_ERROR_RECOVERY_NODE)) {
18947 constant_path = UP(pm_error_recovery_node_create_unexpected(parser, constant_path));
18948 }
18949
18950 pm_parser_scope_push(parser, true);
18951 accept2(parser, PM_TOKEN_SEMICOLON, PM_TOKEN_NEWLINE);
18952 pm_node_t *statements = NULL;
18953
18954 if (!match4(parser, PM_TOKEN_KEYWORD_RESCUE, PM_TOKEN_KEYWORD_ENSURE, PM_TOKEN_KEYWORD_ELSE, PM_TOKEN_KEYWORD_END)) {
18955 pm_accepts_block_stack_push(parser, true);
18956 statements = UP(parse_statements(parser, PM_CONTEXT_MODULE, (uint16_t) (depth + 1)));
18957 pm_accepts_block_stack_pop(parser);
18958 }
18959
18960 if (match3(parser, PM_TOKEN_KEYWORD_RESCUE, PM_TOKEN_KEYWORD_ENSURE, PM_TOKEN_KEYWORD_ELSE)) {
18961 assert(statements == NULL || PM_NODE_TYPE_P(statements, PM_STATEMENTS_NODE));
18962 statements = UP(parse_rescues_implicit_begin(parser, opening_newline_index, &module_keyword, module_keyword.start, (pm_statements_node_t *) statements, PM_RESCUES_MODULE, (uint16_t) (depth + 1)));
18963 } else {
18964 parser_warn_indentation_mismatch(parser, opening_newline_index, &module_keyword, false, false);
18965 }
18966
18967 pm_constant_id_list_t locals;
18968 pm_locals_order(parser, &parser->current_scope->locals, &locals, false);
18969
18970 pm_parser_scope_pop(parser);
18971 expect1_opening(parser, PM_TOKEN_KEYWORD_END, PM_ERR_MODULE_TERM, &module_keyword);
18972
18973 if (context_def_p(parser)) {
18974 pm_parser_err_token(parser, &module_keyword, PM_ERR_MODULE_IN_METHOD);
18975 }
18976
18977 pop_block_exits(parser, previous_block_exits);
18978
18979 return UP(pm_module_node_create(parser, &locals, &module_keyword, constant_path, &name, statements, &parser->previous));
18980}
18981
18985static pm_node_t *
18986parse_string_array(pm_parser_t *parser, uint16_t depth) {
18987 parser_lex(parser);
18988 pm_token_t opening = parser->previous;
18989 pm_array_node_t *array = pm_array_node_create(parser, &opening);
18990
18991 /* This is the current node that we are parsing that will be added to the
18992 * list of elements. */
18993 pm_node_t *current = NULL;
18994
18995 while (!match2(parser, PM_TOKEN_STRING_END, PM_TOKEN_EOF)) {
18996 switch (parser->current.type) {
18997 case PM_TOKEN_WORDS_SEP: {
18998 /* Reset the explicit encoding if we hit a separator since each
18999 * element can have its own encoding. */
19000 parser->explicit_encoding = NULL;
19001
19002 if (current == NULL) {
19003 /* If we hit a separator before we have any content, then we
19004 * don't need to do anything. */
19005 } else {
19006 /* If we hit a separator after we've hit content, then we
19007 * need to append that content to the list and reset the
19008 * current node. */
19009 pm_array_node_elements_append(parser->arena, array, current);
19010 current = NULL;
19011 }
19012
19013 parser_lex(parser);
19014 break;
19015 }
19016 case PM_TOKEN_STRING_CONTENT: {
19017 pm_node_t *string = UP(pm_string_node_create_current_string(parser, NULL, &parser->current, NULL));
19018 pm_node_flag_set(string, parse_unescaped_encoding(parser, parser->explicit_encoding));
19019 parser_lex(parser);
19020
19021 if (current == NULL) {
19022 /* If we hit content and the current node is NULL, then this
19023 * is the first string content we've seen. In that case
19024 * we're going to create a new string node and set that to
19025 * the current. */
19026 current = string;
19027 } else if (PM_NODE_TYPE_P(current, PM_INTERPOLATED_STRING_NODE)) {
19028 /* If we hit string content and the current node is an
19029 * interpolated string, then we need to append the string
19030 * content to the list of child nodes. */
19031 pm_interpolated_string_node_append(parser, (pm_interpolated_string_node_t *) current, string);
19032 } else if (PM_NODE_TYPE_P(current, PM_STRING_NODE)) {
19033 /* If we hit string content and the current node is a string
19034 * node, then we need to convert the current node into an
19035 * interpolated string and add the string content to the
19036 * list of child nodes. */
19037 pm_interpolated_string_node_t *interpolated = pm_interpolated_string_node_create(parser, NULL, NULL, NULL);
19038 pm_interpolated_string_node_append(parser, interpolated, current);
19039 pm_interpolated_string_node_append(parser, interpolated, string);
19040 current = UP(interpolated);
19041 } else {
19042 assert(false && "unreachable");
19043 }
19044
19045 break;
19046 }
19047 case PM_TOKEN_EMBVAR: {
19048 if (current == NULL) {
19049 /* If we hit an embedded variable and the current node is
19050 * NULL, then this is the start of a new string. We'll set
19051 * the current node to a new interpolated string. */
19052 current = UP(pm_interpolated_string_node_create(parser, NULL, NULL, NULL));
19053 } else if (PM_NODE_TYPE_P(current, PM_STRING_NODE)) {
19054 /* If we hit an embedded variable and the current node is a
19055 * string node, then we'll convert the current into an
19056 * interpolated string and add the string node to the list
19057 * of parts. */
19058 pm_interpolated_string_node_t *interpolated = pm_interpolated_string_node_create(parser, NULL, NULL, NULL);
19059 pm_interpolated_string_node_append(parser, interpolated, current);
19060 current = UP(interpolated);
19061 } else {
19062 /* If we hit an embedded variable and the current node is an
19063 * interpolated string, then we'll just add the embedded
19064 * variable. */
19065 }
19066
19067 pm_node_t *part = parse_string_part(parser, (uint16_t) (depth + 1));
19068 pm_interpolated_string_node_append(parser, (pm_interpolated_string_node_t *) current, part);
19069 break;
19070 }
19071 case PM_TOKEN_EMBEXPR_BEGIN: {
19072 if (current == NULL) {
19073 /* If we hit an embedded expression and the current node is
19074 * NULL, then this is the start of a new string. We'll set
19075 * the current node to a new interpolated string. */
19076 current = UP(pm_interpolated_string_node_create(parser, NULL, NULL, NULL));
19077 } else if (PM_NODE_TYPE_P(current, PM_STRING_NODE)) {
19078 /* If we hit an embedded expression and the current node is
19079 * a string node, then we'll convert the current into an
19080 * interpolated string and add the string node to the list
19081 * of parts. */
19082 pm_interpolated_string_node_t *interpolated = pm_interpolated_string_node_create(parser, NULL, NULL, NULL);
19083 pm_interpolated_string_node_append(parser, interpolated, current);
19084 current = UP(interpolated);
19085 } else if (PM_NODE_TYPE_P(current, PM_INTERPOLATED_STRING_NODE)) {
19086 /* If we hit an embedded expression and the current node is
19087 * an interpolated string, then we'll just continue on. */
19088 } else {
19089 assert(false && "unreachable");
19090 }
19091
19092 pm_node_t *part = parse_string_part(parser, (uint16_t) (depth + 1));
19093 pm_interpolated_string_node_append(parser, (pm_interpolated_string_node_t *) current, part);
19094 break;
19095 }
19096 default:
19097 expect1(parser, PM_TOKEN_STRING_CONTENT, PM_ERR_LIST_W_UPPER_ELEMENT);
19098 parser_lex(parser);
19099 break;
19100 }
19101 }
19102
19103 /* If we have a current node, then we need to append it to the list. */
19104 if (current) {
19105 pm_array_node_elements_append(parser->arena, array, current);
19106 }
19107
19108 pm_token_t closing = parser->current;
19109 if (match1(parser, PM_TOKEN_EOF)) {
19110 pm_parser_err_token(parser, &opening, PM_ERR_LIST_W_UPPER_TERM);
19111 closing = (pm_token_t) { .type = 0, .start = parser->previous.end, .end = parser->previous.end };
19112 } else {
19113 expect1(parser, PM_TOKEN_STRING_END, PM_ERR_LIST_W_UPPER_TERM);
19114 }
19115
19116 pm_array_node_close_set(parser, array, &closing);
19117 return UP(array);
19118}
19119
19123static pm_node_t *
19124parse_symbol_array(pm_parser_t *parser, uint16_t depth) {
19125 parser_lex(parser);
19126 pm_token_t opening = parser->previous;
19127 pm_array_node_t *array = pm_array_node_create(parser, &opening);
19128
19129 /* This is the current node that we are parsing that will be added to the
19130 * list of elements. */
19131 pm_node_t *current = NULL;
19132
19133 while (!match2(parser, PM_TOKEN_STRING_END, PM_TOKEN_EOF)) {
19134 switch (parser->current.type) {
19135 case PM_TOKEN_WORDS_SEP: {
19136 /* Reset the explicit encoding if we hit a separator since each
19137 * element can have its own encoding. */
19138 parser->explicit_encoding = NULL;
19139
19140 if (current == NULL) {
19141 /* If we hit a separator before we have any content, then we
19142 * don't need to do anything. */
19143 } else {
19144 /* If we hit a separator after we've hit content, then we
19145 * need to append that content to the list and reset the
19146 * current node. */
19147 pm_array_node_elements_append(parser->arena, array, current);
19148 current = NULL;
19149 }
19150
19151 parser_lex(parser);
19152 break;
19153 }
19154 case PM_TOKEN_STRING_CONTENT: {
19155 if (current == NULL) {
19156 /* If we hit content and the current node is NULL, then this
19157 * is the first string content we've seen. In that case
19158 * we're going to create a new string node and set that to
19159 * the current. */
19160 current = UP(pm_symbol_node_create_current_string(parser, NULL, &parser->current, NULL));
19161 parser_lex(parser);
19162 } else if (PM_NODE_TYPE_P(current, PM_INTERPOLATED_SYMBOL_NODE)) {
19163 /* If we hit string content and the current node is an
19164 * interpolated string, then we need to append the string
19165 * content to the list of child nodes. */
19166 pm_node_t *string = UP(pm_string_node_create_current_string(parser, NULL, &parser->current, NULL));
19167 parser_lex(parser);
19168
19169 pm_interpolated_symbol_node_append(parser->arena, (pm_interpolated_symbol_node_t *) current, string);
19170 } else if (PM_NODE_TYPE_P(current, PM_SYMBOL_NODE)) {
19171 /* If we hit string content and the current node is a symbol
19172 * node, then we need to convert the current node into an
19173 * interpolated string and add the string content to the
19174 * list of child nodes. */
19175 pm_symbol_node_t *cast = (pm_symbol_node_t *) current;
19176 pm_token_t content = {
19177 .type = PM_TOKEN_STRING_CONTENT,
19178 .start = parser->start + cast->content_loc.start,
19179 .end = parser->start + cast->content_loc.start + cast->content_loc.length
19180 };
19181
19182 pm_node_t *first_string = UP(pm_string_node_create_unescaped(parser, NULL, &content, NULL, &cast->unescaped));
19183 pm_node_t *second_string = UP(pm_string_node_create_current_string(parser, NULL, &parser->previous, NULL));
19184 parser_lex(parser);
19185
19186 pm_interpolated_symbol_node_t *interpolated = pm_interpolated_symbol_node_create(parser, NULL, NULL, NULL);
19187 pm_interpolated_symbol_node_append(parser->arena, interpolated, first_string);
19188 pm_interpolated_symbol_node_append(parser->arena, interpolated, second_string);
19189
19190 current = UP(interpolated);
19191 } else {
19192 assert(false && "unreachable");
19193 }
19194
19195 break;
19196 }
19197 case PM_TOKEN_EMBVAR: {
19198 bool start_location_set = false;
19199 if (current == NULL) {
19200 /* If we hit an embedded variable and the current node is
19201 * NULL, then this is the start of a new string. We'll set
19202 * the current node to a new interpolated string. */
19203 current = UP(pm_interpolated_symbol_node_create(parser, NULL, NULL, NULL));
19204 } else if (PM_NODE_TYPE_P(current, PM_SYMBOL_NODE)) {
19205 /* If we hit an embedded variable and the current node is a
19206 * string node, then we'll convert the current into an
19207 * interpolated string and add the string node to the list
19208 * of parts. */
19209 pm_interpolated_symbol_node_t *interpolated = pm_interpolated_symbol_node_create(parser, NULL, NULL, NULL);
19210
19211 current = UP(pm_symbol_node_to_string_node(parser, (pm_symbol_node_t *) current));
19212 pm_interpolated_symbol_node_append(parser->arena, interpolated, current);
19213 PM_NODE_START_SET_NODE(interpolated, current);
19214 start_location_set = true;
19215 current = UP(interpolated);
19216 } else {
19217 /* If we hit an embedded variable and the current node is an
19218 * interpolated string, then we'll just add the embedded
19219 * variable. */
19220 }
19221
19222 pm_node_t *part = parse_string_part(parser, (uint16_t) (depth + 1));
19223 pm_interpolated_symbol_node_append(parser->arena, (pm_interpolated_symbol_node_t *) current, part);
19224 if (!start_location_set) {
19225 PM_NODE_START_SET_NODE(current, part);
19226 }
19227 break;
19228 }
19229 case PM_TOKEN_EMBEXPR_BEGIN: {
19230 bool start_location_set = false;
19231 if (current == NULL) {
19232 /* If we hit an embedded expression and the current node is
19233 * NULL, then this is the start of a new string. We'll set
19234 * the current node to a new interpolated string. */
19235 current = UP(pm_interpolated_symbol_node_create(parser, NULL, NULL, NULL));
19236 } else if (PM_NODE_TYPE_P(current, PM_SYMBOL_NODE)) {
19237 /* If we hit an embedded expression and the current node is
19238 * a string node, then we'll convert the current into an
19239 * interpolated string and add the string node to the list
19240 * of parts. */
19241 pm_interpolated_symbol_node_t *interpolated = pm_interpolated_symbol_node_create(parser, NULL, NULL, NULL);
19242
19243 current = UP(pm_symbol_node_to_string_node(parser, (pm_symbol_node_t *) current));
19244 pm_interpolated_symbol_node_append(parser->arena, interpolated, current);
19245 PM_NODE_START_SET_NODE(interpolated, current);
19246 start_location_set = true;
19247 current = UP(interpolated);
19248 } else if (PM_NODE_TYPE_P(current, PM_INTERPOLATED_SYMBOL_NODE)) {
19249 /* If we hit an embedded expression and the current node is
19250 * an interpolated string, then we'll just continue on. */
19251 } else {
19252 assert(false && "unreachable");
19253 }
19254
19255 pm_node_t *part = parse_string_part(parser, (uint16_t) (depth + 1));
19256 pm_interpolated_symbol_node_append(parser->arena, (pm_interpolated_symbol_node_t *) current, part);
19257 if (!start_location_set) {
19258 PM_NODE_START_SET_NODE(current, part);
19259 }
19260 break;
19261 }
19262 default:
19263 expect1(parser, PM_TOKEN_STRING_CONTENT, PM_ERR_LIST_I_UPPER_ELEMENT);
19264 parser_lex(parser);
19265 break;
19266 }
19267 }
19268
19269 /* If we have a current node, then we need to append it to the list. */
19270 if (current) {
19271 pm_array_node_elements_append(parser->arena, array, current);
19272 }
19273
19274 pm_token_t closing = parser->current;
19275 if (match1(parser, PM_TOKEN_EOF)) {
19276 pm_parser_err_token(parser, &opening, PM_ERR_LIST_I_UPPER_TERM);
19277 closing = (pm_token_t) { .type = 0, .start = parser->previous.end, .end = parser->previous.end };
19278 } else {
19279 expect1(parser, PM_TOKEN_STRING_END, PM_ERR_LIST_I_UPPER_TERM);
19280 }
19281 pm_array_node_close_set(parser, array, &closing);
19282
19283 return UP(array);
19284}
19285
19290static pm_node_t *
19291parse_parentheses(pm_parser_t *parser, pm_binding_power_t binding_power, uint8_t flags, uint16_t depth) {
19292 pm_token_t opening = parser->current;
19293 pm_node_flags_t paren_flags = 0;
19294
19295 pm_node_list_t current_block_exits = { 0 };
19296 pm_node_list_t *previous_block_exits = push_block_exits(parser, &current_block_exits);
19297
19298 parser_lex(parser);
19299 while (true) {
19300 if (accept1(parser, PM_TOKEN_SEMICOLON)) {
19301 paren_flags |= PM_PARENTHESES_NODE_FLAGS_MULTIPLE_STATEMENTS;
19302 } else if (!accept1(parser, PM_TOKEN_NEWLINE)) {
19303 break;
19304 }
19305 }
19306
19307 /* If this is the end of the file or we match a right parenthesis, then we
19308 * have an empty parentheses node, and we can immediately return. */
19309 if (match2(parser, PM_TOKEN_PARENTHESIS_RIGHT, PM_TOKEN_EOF)) {
19310 /* A command argument group sets EXPR_ENDARG before its ')' is
19311 * consumed, even when the group is empty, so that a following '{' is
19312 * scanned as a block brace. */
19313 if (match1(parser, PM_TOKEN_PARENTHESIS_RIGHT) && opening.type == PM_TOKEN_PARENTHESIS_LEFT_PARENTHESES) {
19314 lex_state_set(parser, PM_LEX_STATE_ENDARG);
19315 }
19316
19317 expect1(parser, PM_TOKEN_PARENTHESIS_RIGHT, PM_ERR_EXPECT_RPAREN);
19318 pop_block_exits(parser, previous_block_exits);
19319 return UP(pm_parentheses_node_create(parser, &opening, NULL, &parser->previous, paren_flags));
19320 }
19321
19322 /* Otherwise, we're going to parse the first statement in the list of
19323 * statements within the parentheses. */
19324 context_push(parser, PM_CONTEXT_PARENS);
19325 pm_node_t *statement = parse_expression(parser, PM_BINDING_POWER_STATEMENT, PM_PARSE_ACCEPTS_COMMAND_CALL | PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_CANNOT_PARSE_EXPRESSION, (uint16_t) (depth + 1));
19326 context_pop(parser);
19327
19328 /* Determine if this statement is followed by a terminator. In the case of a
19329 * single statement, this is fine. But in the case of multiple statements
19330 * it's required. */
19331 bool terminator_found = false;
19332
19333 if (accept1(parser, PM_TOKEN_SEMICOLON)) {
19334 terminator_found = true;
19335 paren_flags |= PM_PARENTHESES_NODE_FLAGS_MULTIPLE_STATEMENTS;
19336 } else if (accept1(parser, PM_TOKEN_NEWLINE)) {
19337 terminator_found = true;
19338 }
19339
19340 if (terminator_found) {
19341 while (true) {
19342 if (accept1(parser, PM_TOKEN_SEMICOLON)) {
19343 paren_flags |= PM_PARENTHESES_NODE_FLAGS_MULTIPLE_STATEMENTS;
19344 } else if (!accept1(parser, PM_TOKEN_NEWLINE)) {
19345 break;
19346 }
19347 }
19348 }
19349
19350 /* If we hit a right parenthesis, then we're done parsing the parentheses
19351 * node, and we can check which kind of node we should return. */
19352 if (match1(parser, PM_TOKEN_PARENTHESIS_RIGHT)) {
19353 if (opening.type == PM_TOKEN_PARENTHESIS_LEFT_PARENTHESES) {
19354 lex_state_set(parser, PM_LEX_STATE_ENDARG);
19355 }
19356
19357 parser_lex(parser);
19358 pop_block_exits(parser, previous_block_exits);
19359
19360 if (PM_NODE_TYPE_P(statement, PM_MULTI_TARGET_NODE) || PM_NODE_TYPE_P(statement, PM_SPLAT_NODE)) {
19361 /* If we have a single statement and are ending on a right
19362 * parenthesis, then we need to check if this is possibly a multiple
19363 * target node. */
19364 pm_multi_target_node_t *multi_target;
19365
19366 if (PM_NODE_TYPE_P(statement, PM_MULTI_TARGET_NODE) && ((pm_multi_target_node_t *) statement)->lparen_loc.length == 0) {
19367 multi_target = (pm_multi_target_node_t *) statement;
19368 } else {
19369 multi_target = pm_multi_target_node_create(parser);
19370 pm_multi_target_node_targets_append(parser, multi_target, statement);
19371 }
19372
19373 multi_target->lparen_loc = TOK2LOC(parser, &opening);
19374 multi_target->rparen_loc = TOK2LOC(parser, &parser->previous);
19375 PM_NODE_START_SET_TOKEN(parser, multi_target, &opening);
19376 PM_NODE_LENGTH_SET_TOKEN(parser, multi_target, &parser->previous);
19377
19378 pm_node_t *result;
19379 if (match1(parser, PM_TOKEN_COMMA) && (binding_power == PM_BINDING_POWER_STATEMENT)) {
19380 result = parse_targets(parser, UP(multi_target), PM_BINDING_POWER_INDEX, (uint16_t) (depth + 1));
19381 accept1(parser, PM_TOKEN_NEWLINE);
19382 } else {
19383 result = UP(multi_target);
19384 }
19385
19386 if (context_p(parser, PM_CONTEXT_MULTI_TARGET)) {
19387 /* All set, this is explicitly allowed by the parent context. */
19388 } else if (context_p(parser, PM_CONTEXT_FOR_INDEX) && match2(parser, PM_TOKEN_KEYWORD_IN, PM_TOKEN_COMMA)) {
19389 /* All set, we're inside a for loop and we're parsing multiple
19390 * targets. A comma continues the index target list, as in
19391 * `for (a, b), c in ...`. */
19392 } else if (flags & PM_PARSE_ACCEPTS_STATEMENT) {
19393 /* The rescue-modifier value parser promotes this target on a
19394 * following `=` or comma. Reject any other binary operator that
19395 * would otherwise consume the target list (e.g. `(a, b) + c`). */
19396 if (pm_binding_powers[parser->current.type].binary && !match1(parser, PM_TOKEN_EQUAL)) {
19397 pm_parser_err_node(parser, result, PM_ERR_WRITE_TARGET_UNEXPECTED);
19398 }
19399 } else if (binding_power != PM_BINDING_POWER_STATEMENT) {
19400 /* Multi targets are not allowed when it's not a statement
19401 * level. */
19402 pm_parser_err_node(parser, result, PM_ERR_WRITE_TARGET_UNEXPECTED);
19403 } else if (!match2(parser, PM_TOKEN_EQUAL, PM_TOKEN_PARENTHESIS_RIGHT)) {
19404 /* Multi targets must be followed by an equal sign in order to
19405 * be valid (or a right parenthesis if they are nested). */
19406 pm_parser_err_node(parser, result, PM_ERR_WRITE_TARGET_UNEXPECTED);
19407 }
19408
19409 return result;
19410 }
19411
19412 /* If we have a single statement and are ending on a right parenthesis
19413 * and we didn't return a multiple assignment node, then we can return a
19414 * regular parentheses node now. */
19415 pm_statements_node_t *statements = pm_statements_node_create(parser);
19416 pm_statements_node_body_append(parser, statements, statement, true);
19417
19418 return UP(pm_parentheses_node_create(parser, &opening, UP(statements), &parser->previous, paren_flags));
19419 }
19420
19421 /* If we have more than one statement in the set of parentheses, then we are
19422 * going to parse all of them as a list of statements. We'll do that here.
19423 */
19424 context_push(parser, PM_CONTEXT_PARENS);
19425 paren_flags |= PM_PARENTHESES_NODE_FLAGS_MULTIPLE_STATEMENTS;
19426
19427 pm_statements_node_t *statements = pm_statements_node_create(parser);
19428 pm_statements_node_body_append(parser, statements, statement, true);
19429
19430 /* If we didn't find a terminator and we didn't find a right parenthesis,
19431 * then this is a syntax error. */
19432 if (!terminator_found && !match1(parser, PM_TOKEN_EOF)) {
19433 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->current, PM_ERR_EXPECT_EOL_AFTER_STATEMENT, pm_token_str(parser->current.type));
19434 }
19435
19436 /* Parse each statement within the parentheses. */
19437 while (true) {
19438 pm_node_t *node = parse_expression(parser, PM_BINDING_POWER_STATEMENT, PM_PARSE_ACCEPTS_COMMAND_CALL | PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_CANNOT_PARSE_EXPRESSION, (uint16_t) (depth + 1));
19439 pm_statements_node_body_append(parser, statements, node, true);
19440
19441 /* If we're recovering from a syntax error, then we need to stop parsing
19442 * the statements now. */
19443 if (parser->recovering) {
19444 /* If this is the level of context where the recovery has happened,
19445 * then we can mark the parser as done recovering. */
19446 if (match1(parser, PM_TOKEN_PARENTHESIS_RIGHT)) parser->recovering = false;
19447 break;
19448 }
19449
19450 /* If we couldn't parse an expression at all, then we need to bail out
19451 * of the loop. */
19452 if (PM_NODE_TYPE_P(node, PM_ERROR_RECOVERY_NODE)) break;
19453
19454 /* If we successfully parsed a statement, then we are going to need a
19455 * terminator to delimit them. */
19456 if (accept2(parser, PM_TOKEN_NEWLINE, PM_TOKEN_SEMICOLON)) {
19457 while (accept2(parser, PM_TOKEN_NEWLINE, PM_TOKEN_SEMICOLON));
19458 if (match1(parser, PM_TOKEN_PARENTHESIS_RIGHT)) break;
19459 } else if (match1(parser, PM_TOKEN_PARENTHESIS_RIGHT)) {
19460 break;
19461 } else if (!match1(parser, PM_TOKEN_EOF)) {
19462 /* If we're at the end of the file, then we're going to add an error
19463 * after this for the ) anyway. */
19464 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->current, PM_ERR_EXPECT_EOL_AFTER_STATEMENT, pm_token_str(parser->current.type));
19465 }
19466 }
19467
19468 context_pop(parser);
19469 expect1(parser, PM_TOKEN_PARENTHESIS_RIGHT, PM_ERR_EXPECT_RPAREN);
19470
19471 /* When we're parsing multi targets, we allow them to be followed by a right
19472 * parenthesis if they are at the statement level. This is only possible if
19473 * they are the final statement in a parentheses. We need to explicitly
19474 * reject that here. */
19475 {
19476 pm_node_t *statement = statements->body.nodes[statements->body.size - 1];
19477
19478 if (PM_NODE_TYPE_P(statement, PM_SPLAT_NODE)) {
19479 pm_multi_target_node_t *multi_target = pm_multi_target_node_create(parser);
19480 pm_multi_target_node_targets_append(parser, multi_target, statement);
19481
19482 statement = UP(multi_target);
19483 statements->body.nodes[statements->body.size - 1] = statement;
19484 }
19485
19486 if (PM_NODE_TYPE_P(statement, PM_MULTI_TARGET_NODE)) {
19487 const uint8_t *offset = parser->start + PM_NODE_END(statement);
19488 pm_token_t operator = { .type = PM_TOKEN_EQUAL, .start = offset, .end = offset };
19489 pm_node_t *value = UP(pm_error_recovery_node_create(parser, PM_NODE_END(statement), 0));
19490
19491 statement = UP(pm_multi_write_node_create(parser, (pm_multi_target_node_t *) statement, &operator, value));
19492 statements->body.nodes[statements->body.size - 1] = statement;
19493
19494 pm_parser_err_node(parser, statement, PM_ERR_WRITE_TARGET_UNEXPECTED);
19495 }
19496 }
19497
19498 pop_block_exits(parser, previous_block_exits);
19499 pm_void_statements_check(parser, statements, true);
19500 return UP(pm_parentheses_node_create(parser, &opening, UP(statements), &parser->previous, paren_flags));
19501}
19502
19508static pm_node_t *
19509parse_splat(pm_parser_t *parser, uint8_t flags, uint16_t depth) {
19510 pm_token_t operator = parser->previous;
19511 pm_node_t *name = NULL;
19512
19513 if (token_begins_expression_p(parser->current.type)) {
19514 name = parse_expression(parser, PM_BINDING_POWER_INDEX, flags & PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_EXPECT_EXPRESSION_AFTER_STAR, (uint16_t) (depth + 1));
19515 }
19516
19517 return UP(pm_splat_node_create(parser, &operator, name));
19518}
19519
19523static PRISM_INLINE pm_node_t *
19524parse_expression_prefix(pm_parser_t *parser, pm_binding_power_t binding_power, uint8_t flags, pm_diagnostic_id_t diag_id, uint16_t depth) {
19525 switch (parser->current.type) {
19526 case PM_TOKEN_BRACKET_LEFT_ARRAY: {
19527 parser_lex(parser);
19528
19529 pm_array_node_t *array = pm_array_node_create(parser, &parser->previous);
19530 bool parsed_bare_hash = false;
19531
19532 while (!match2(parser, PM_TOKEN_BRACKET_RIGHT, PM_TOKEN_EOF)) {
19533 bool accepted_newline = accept1(parser, PM_TOKEN_NEWLINE);
19534
19535 // Handle the case where we don't have a comma and we have a
19536 // newline followed by a right bracket.
19537 if (accepted_newline && match1(parser, PM_TOKEN_BRACKET_RIGHT)) {
19538 break;
19539 }
19540
19541 // Ensure that we have a comma between elements in the array.
19542 if (array->elements.size > 0) {
19543 if (accept1(parser, PM_TOKEN_COMMA)) {
19544 // If there was a comma but we also accepts a newline,
19545 // then this is a syntax error.
19546 if (accepted_newline) {
19547 pm_parser_err_previous(parser, PM_ERR_INVALID_COMMA);
19548 }
19549 } else {
19550 // If there was no comma, then we need to add a syntax
19551 // error.
19552 PM_PARSER_ERR_FORMAT(parser, PM_TOKEN_END(parser, &parser->previous), 0, PM_ERR_ARRAY_SEPARATOR, pm_token_str(parser->current.type));
19553 parser->previous.start = parser->previous.end;
19554 parser->previous.type = 0;
19555 }
19556 }
19557
19558 // If we have a right bracket immediately following a comma,
19559 // this is allowed since it's a trailing comma. In this case we
19560 // can break out of the loop.
19561 if (match1(parser, PM_TOKEN_BRACKET_RIGHT)) break;
19562
19563 pm_node_t *element;
19564
19565 if (accept1(parser, PM_TOKEN_USTAR)) {
19566 pm_token_t operator = parser->previous;
19567 pm_node_t *expression = NULL;
19568
19569 if (match3(parser, PM_TOKEN_BRACKET_RIGHT, PM_TOKEN_COMMA, PM_TOKEN_EOF)) {
19570 pm_parser_scope_forwarding_positionals_check(parser, &operator);
19571 } else {
19572 expression = parse_value_expression(parser, PM_BINDING_POWER_DEFINED, flags & PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_ARRAY_EXPRESSION_AFTER_STAR, (uint16_t) (depth + 1));
19573 }
19574
19575 element = UP(pm_splat_node_create(parser, &operator, expression));
19576 } else if (match2(parser, PM_TOKEN_LABEL, PM_TOKEN_USTAR_STAR)) {
19577 if (parsed_bare_hash) {
19578 pm_parser_err_current(parser, PM_ERR_EXPRESSION_BARE_HASH);
19579 }
19580
19581 element = UP(pm_keyword_hash_node_create(parser));
19582 pm_static_literals_t hash_keys = { 0 };
19583
19584 if (!match8(parser, PM_TOKEN_EOF, PM_TOKEN_NEWLINE, PM_TOKEN_SEMICOLON, PM_TOKEN_KEYWORD_DO_BLOCK, PM_TOKEN_BRACE_RIGHT, PM_TOKEN_BRACKET_RIGHT, PM_TOKEN_KEYWORD_DO, PM_TOKEN_PARENTHESIS_RIGHT)) {
19585 parse_assocs(parser, &hash_keys, element, (uint16_t) (depth + 1));
19586 }
19587
19588 pm_static_literals_free(&hash_keys);
19589 parsed_bare_hash = true;
19590 } else {
19591 element = parse_value_expression(parser, PM_BINDING_POWER_DEFINED, (uint8_t) ((flags & PM_PARSE_ACCEPTS_DO_BLOCK) | PM_PARSE_ACCEPTS_LABEL), PM_ERR_ARRAY_EXPRESSION, (uint16_t) (depth + 1));
19592
19593 if (pm_symbol_node_label_p(parser, element) || accept1(parser, PM_TOKEN_EQUAL_GREATER)) {
19594 if (parsed_bare_hash) {
19595 pm_parser_err_previous(parser, PM_ERR_EXPRESSION_BARE_HASH);
19596 }
19597
19598 pm_keyword_hash_node_t *hash = pm_keyword_hash_node_create(parser);
19599 pm_static_literals_t hash_keys = { 0 };
19600 pm_hash_key_static_literals_add(parser, &hash_keys, element);
19601
19602 pm_token_t operator = { 0 };
19603 if (parser->previous.type == PM_TOKEN_EQUAL_GREATER) {
19604 operator = parser->previous;
19605 }
19606
19607 pm_node_t *value = parse_value_expression(parser, PM_BINDING_POWER_DEFINED, flags & PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_HASH_VALUE, (uint16_t) (depth + 1));
19608 pm_node_t *assoc = UP(pm_assoc_node_create(parser, element, NTOK2PTR(operator), value));
19609 pm_keyword_hash_node_elements_append(parser->arena, hash, assoc);
19610
19611 element = UP(hash);
19612 if (accept1(parser, PM_TOKEN_COMMA) && !match1(parser, PM_TOKEN_BRACKET_RIGHT)) {
19613 parse_assocs(parser, &hash_keys, element, (uint16_t) (depth + 1));
19614 }
19615
19616 pm_static_literals_free(&hash_keys);
19617 parsed_bare_hash = true;
19618 }
19619 }
19620
19621 pm_array_node_elements_append(parser->arena, array, element);
19622 if (PM_NODE_TYPE_P(element, PM_ERROR_RECOVERY_NODE)) break;
19623 }
19624
19625 accept1(parser, PM_TOKEN_NEWLINE);
19626
19627 if (!accept1(parser, PM_TOKEN_BRACKET_RIGHT)) {
19628 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->current, PM_ERR_ARRAY_TERM, pm_token_str(parser->current.type));
19629 parser->previous.start = parser->previous.end;
19630 parser->previous.type = 0;
19631 }
19632
19633 pm_array_node_close_set(parser, array, &parser->previous);
19634
19635 return UP(array);
19636 }
19637 case PM_TOKEN_PARENTHESIS_LEFT_GROUPING:
19638 case PM_TOKEN_PARENTHESIS_LEFT_PARENTHESES:
19639 return parse_parentheses(parser, binding_power, flags, depth);
19640 case PM_TOKEN_BRACE_LEFT_HASH: {
19641 parser_lex(parser);
19642
19643 pm_token_t opening = parser->previous;
19644 pm_hash_node_t *node = pm_hash_node_create(parser, &opening);
19645
19646 if (!match2(parser, PM_TOKEN_BRACE_RIGHT, PM_TOKEN_EOF)) {
19647 pm_static_literals_t hash_keys = { 0 };
19648 parse_assocs(parser, &hash_keys, UP(node), (uint16_t) (depth + 1));
19649 pm_static_literals_free(&hash_keys);
19650
19651 accept1(parser, PM_TOKEN_NEWLINE);
19652 }
19653
19654 expect1_opening(parser, PM_TOKEN_BRACE_RIGHT, PM_ERR_HASH_TERM, &opening);
19655 pm_hash_node_closing_loc_set(parser, node, &parser->previous);
19656
19657 return UP(node);
19658 }
19659 case PM_TOKEN_CHARACTER_LITERAL: {
19660 pm_node_t *node = UP(pm_string_node_create_current_string(
19661 parser,
19662 &(pm_token_t) {
19663 .type = PM_TOKEN_STRING_BEGIN,
19664 .start = parser->current.start,
19665 .end = parser->current.start + 1
19666 },
19667 &(pm_token_t) {
19668 .type = PM_TOKEN_STRING_CONTENT,
19669 .start = parser->current.start + 1,
19670 .end = parser->current.end
19671 },
19672 NULL
19673 ));
19674
19675 pm_node_flag_set(node, parse_unescaped_encoding(parser, parser->explicit_encoding));
19676
19677 // Skip past the character literal here, since now we have handled
19678 // parser->explicit_encoding correctly.
19679 parser_lex(parser);
19680
19681 // Characters can be followed by strings in which case they are
19682 // automatically concatenated.
19683 if (match1(parser, PM_TOKEN_STRING_BEGIN)) {
19684 return parse_strings(parser, node, false, (uint16_t) (depth + 1));
19685 }
19686
19687 return node;
19688 }
19689 case PM_TOKEN_CLASS_VARIABLE: {
19690 parser_lex(parser);
19691 pm_node_t *node = UP(pm_class_variable_read_node_create(parser, &parser->previous));
19692
19693 if (binding_power == PM_BINDING_POWER_STATEMENT && match1(parser, PM_TOKEN_COMMA)) {
19694 node = parse_targets_validate(parser, node, PM_BINDING_POWER_INDEX, (uint16_t) (depth + 1));
19695 }
19696
19697 return node;
19698 }
19699 case PM_TOKEN_CONSTANT: {
19700 parser_lex(parser);
19701 pm_token_t constant = parser->previous;
19702
19703 // If a constant is immediately followed by parentheses, then this is in
19704 // fact a method call, not a constant read.
19705 if (
19706 match1(parser, PM_TOKEN_PARENTHESIS_LEFT) ||
19707 ((flags & PM_PARSE_ACCEPTS_COMMAND_CALL) && (token_begins_expression_p(parser->current.type) || match3(parser, PM_TOKEN_UAMPERSAND, PM_TOKEN_USTAR, PM_TOKEN_USTAR_STAR))) ||
19708 (pm_accepts_block_stack_p(parser) && match1(parser, PM_TOKEN_KEYWORD_DO)) ||
19709 match1(parser, PM_TOKEN_BRACE_LEFT)
19710 ) {
19711 pm_arguments_t arguments = { 0 };
19712 parse_arguments_list(parser, &arguments, true, flags, (uint16_t) (depth + 1));
19713 return UP(pm_call_node_fcall_create(parser, &constant, &arguments));
19714 }
19715
19716 pm_node_t *node = UP(pm_constant_read_node_create(parser, &parser->previous));
19717
19718 if ((binding_power == PM_BINDING_POWER_STATEMENT) && match1(parser, PM_TOKEN_COMMA)) {
19719 // If we get here, then we have a comma immediately following a
19720 // constant, so we're going to parse this as a multiple assignment.
19721 node = parse_targets_validate(parser, node, PM_BINDING_POWER_INDEX, (uint16_t) (depth + 1));
19722 }
19723
19724 return node;
19725 }
19726 case PM_TOKEN_UCOLON_COLON: {
19727 parser_lex(parser);
19728 pm_token_t delimiter = parser->previous;
19729
19730 expect1(parser, PM_TOKEN_CONSTANT, PM_ERR_CONSTANT_PATH_COLON_COLON_CONSTANT);
19731 pm_node_t *node = UP(pm_constant_path_node_create(parser, NULL, &delimiter, &parser->previous));
19732
19733 if ((binding_power == PM_BINDING_POWER_STATEMENT) && match1(parser, PM_TOKEN_COMMA)) {
19734 node = parse_targets_validate(parser, node, PM_BINDING_POWER_INDEX, (uint16_t) (depth + 1));
19735 }
19736
19737 return node;
19738 }
19739 case PM_TOKEN_UDOT_DOT:
19740 case PM_TOKEN_UDOT_DOT_DOT: {
19741 pm_token_t operator = parser->current;
19742 parser_lex(parser);
19743
19744 pm_node_t *right = parse_expression(parser, pm_binding_powers[operator.type].left, flags & PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_EXPECT_EXPRESSION_AFTER_OPERATOR, (uint16_t) (depth + 1));
19745
19746 // Unary .. and ... are special because these are non-associative
19747 // operators that can also be unary operators. In this case we need
19748 // to explicitly reject code that has a .. or ... that follows this
19749 // expression.
19750 if (match2(parser, PM_TOKEN_DOT_DOT, PM_TOKEN_DOT_DOT_DOT)) {
19751 pm_parser_err_current(parser, PM_ERR_UNEXPECTED_RANGE_OPERATOR);
19752 }
19753
19754 return UP(pm_range_node_create(parser, NULL, &operator, right));
19755 }
19756 case PM_TOKEN_FLOAT:
19757 parser_lex(parser);
19758 return UP(pm_float_node_create(parser, &parser->previous));
19759 case PM_TOKEN_FLOAT_IMAGINARY:
19760 parser_lex(parser);
19761 return UP(pm_float_node_imaginary_create(parser, &parser->previous));
19762 case PM_TOKEN_FLOAT_RATIONAL:
19763 parser_lex(parser);
19764 return UP(pm_float_node_rational_create(parser, &parser->previous));
19765 case PM_TOKEN_FLOAT_RATIONAL_IMAGINARY:
19766 parser_lex(parser);
19767 return UP(pm_float_node_rational_imaginary_create(parser, &parser->previous));
19768 case PM_TOKEN_NUMBERED_REFERENCE: {
19769 parser_lex(parser);
19770 pm_node_t *node = UP(pm_numbered_reference_read_node_create(parser, &parser->previous));
19771
19772 if (binding_power == PM_BINDING_POWER_STATEMENT && match1(parser, PM_TOKEN_COMMA)) {
19773 node = parse_targets_validate(parser, node, PM_BINDING_POWER_INDEX, (uint16_t) (depth + 1));
19774 }
19775
19776 return node;
19777 }
19778 case PM_TOKEN_GLOBAL_VARIABLE: {
19779 parser_lex(parser);
19780 pm_node_t *node = UP(pm_global_variable_read_node_create(parser, &parser->previous));
19781
19782 if (binding_power == PM_BINDING_POWER_STATEMENT && match1(parser, PM_TOKEN_COMMA)) {
19783 node = parse_targets_validate(parser, node, PM_BINDING_POWER_INDEX, (uint16_t) (depth + 1));
19784 }
19785
19786 return node;
19787 }
19788 case PM_TOKEN_BACK_REFERENCE: {
19789 parser_lex(parser);
19790 pm_node_t *node = UP(pm_back_reference_read_node_create(parser, &parser->previous));
19791
19792 if (binding_power == PM_BINDING_POWER_STATEMENT && match1(parser, PM_TOKEN_COMMA)) {
19793 node = parse_targets_validate(parser, node, PM_BINDING_POWER_INDEX, (uint16_t) (depth + 1));
19794 }
19795
19796 return node;
19797 }
19798 case PM_TOKEN_IDENTIFIER:
19799 case PM_TOKEN_METHOD_NAME: {
19800 parser_lex(parser);
19801 pm_token_t identifier = parser->previous;
19802 pm_node_t *node = parse_variable_call(parser);
19803
19804 if (PM_NODE_TYPE_P(node, PM_CALL_NODE)) {
19805 // If parse_variable_call returned with a call node, then we
19806 // know the identifier is not in the local table. In that case
19807 // we need to check if there are arguments following the
19808 // identifier.
19809 pm_call_node_t *call = (pm_call_node_t *) node;
19810 pm_arguments_t arguments = { 0 };
19811
19812 if (parse_arguments_list(parser, &arguments, true, flags, (uint16_t) (depth + 1))) {
19813 // Since we found arguments, we need to turn off the
19814 // variable call bit in the flags.
19815 pm_node_flag_unset(UP(call), PM_CALL_NODE_FLAGS_VARIABLE_CALL);
19816
19817 call->opening_loc = arguments.opening_loc;
19818 call->arguments = arguments.arguments;
19819 call->closing_loc = arguments.closing_loc;
19820 call->block = arguments.block;
19821
19822 const pm_location_t *end = pm_arguments_end(&arguments);
19823 if (end == NULL) {
19824 PM_NODE_LENGTH_SET_LOCATION(call, &call->message_loc);
19825 } else {
19826 PM_NODE_LENGTH_SET_LOCATION(call, end);
19827 }
19828 }
19829 } else {
19830 // Otherwise, we know the identifier is in the local table. This
19831 // can still be a method call if it is followed by arguments or
19832 // a block, so we need to check for that here.
19833 if (
19834 ((flags & PM_PARSE_ACCEPTS_COMMAND_CALL) && (token_begins_expression_p(parser->current.type) || match3(parser, PM_TOKEN_UAMPERSAND, PM_TOKEN_USTAR, PM_TOKEN_USTAR_STAR))) ||
19835 (pm_accepts_block_stack_p(parser) && match1(parser, PM_TOKEN_KEYWORD_DO)) ||
19836 match1(parser, PM_TOKEN_BRACE_LEFT)
19837 ) {
19838 pm_arguments_t arguments = { 0 };
19839 parse_arguments_list(parser, &arguments, true, flags, (uint16_t) (depth + 1));
19840 pm_call_node_t *fcall = pm_call_node_fcall_create(parser, &identifier, &arguments);
19841
19842 if (PM_NODE_TYPE_P(node, PM_IT_LOCAL_VARIABLE_READ_NODE)) {
19843 // If we're about to convert an 'it' implicit local
19844 // variable read into a method call, we need to remove
19845 // it from the list of implicit local variables.
19846 pm_node_unreference(parser, node);
19847 } else {
19848 // Otherwise, we're about to convert a regular local
19849 // variable read into a method call, in which case we
19850 // need to indicate that this was not a read for the
19851 // purposes of warnings.
19852 assert(PM_NODE_TYPE_P(node, PM_LOCAL_VARIABLE_READ_NODE));
19853
19854 if (pm_token_is_numbered_parameter(parser, PM_TOKEN_START(parser, &identifier), PM_TOKEN_LENGTH(&identifier))) {
19855 pm_node_unreference(parser, node);
19856 } else {
19858 pm_locals_unread(&pm_parser_scope_find(parser, cast->depth)->locals, cast->name);
19859 }
19860 }
19861
19862 return UP(fcall);
19863 }
19864 }
19865
19866 if ((binding_power == PM_BINDING_POWER_STATEMENT) && match1(parser, PM_TOKEN_COMMA)) {
19867 node = parse_targets_validate(parser, node, PM_BINDING_POWER_INDEX, (uint16_t) (depth + 1));
19868 }
19869
19870 return node;
19871 }
19872 case PM_TOKEN_HEREDOC_START: {
19873 // Here we have found a heredoc. We'll parse it and add it to the
19874 // list of strings.
19875 assert(parser->lex_modes.current->mode == PM_LEX_HEREDOC);
19876 pm_heredoc_lex_mode_t lex_mode = parser->lex_modes.current->as.heredoc.base;
19877
19878 size_t common_whitespace = (size_t) -1;
19879 parser->lex_modes.current->as.heredoc.common_whitespace = &common_whitespace;
19880
19881 parser_lex(parser);
19882 pm_token_t opening = parser->previous;
19883
19884 pm_node_t *node;
19885 pm_node_t *part;
19886
19887 if (match2(parser, PM_TOKEN_HEREDOC_END, PM_TOKEN_EOF)) {
19888 // If we get here, then we have an empty heredoc. We'll create
19889 // an empty content token and return an empty string node.
19890 expect1_heredoc_term(parser, lex_mode.ident_start, lex_mode.ident_length);
19891 pm_token_t content = parse_strings_empty_content(parser->previous.start);
19892
19893 if (lex_mode.quote == PM_HEREDOC_QUOTE_BACKTICK) {
19894 node = UP(pm_xstring_node_create_unescaped(parser, &opening, &content, &parser->previous, &PM_STRING_EMPTY));
19895 } else {
19896 node = UP(pm_string_node_create_unescaped(parser, &opening, &content, &parser->previous, &PM_STRING_EMPTY));
19897 }
19898
19899 PM_NODE_LENGTH_SET_TOKEN(parser, node, &opening);
19900 } else if ((part = parse_string_part(parser, (uint16_t) (depth + 1))) == NULL) {
19901 // If we get here, then we tried to find something in the
19902 // heredoc but couldn't actually parse anything, so we'll just
19903 // return a missing node.
19904 //
19905 // parse_string_part handles its own errors, so there is no need
19906 // for us to add one here.
19907 node = UP(pm_error_recovery_node_create(parser, PM_TOKEN_START(parser, &parser->previous), PM_TOKEN_LENGTH(&parser->previous)));
19908 } else if (PM_NODE_TYPE_P(part, PM_STRING_NODE) && match2(parser, PM_TOKEN_HEREDOC_END, PM_TOKEN_EOF)) {
19909 // If we get here, then the part that we parsed was plain string
19910 // content and we're at the end of the heredoc, so we can return
19911 // just a string node with the heredoc opening and closing as
19912 // its opening and closing.
19913 pm_node_flag_set(part, parse_unescaped_encoding(parser, parser->explicit_encoding));
19914 pm_string_node_t *cast = (pm_string_node_t *) part;
19915
19916 cast->opening_loc = TOK2LOC(parser, &opening);
19917 cast->closing_loc = TOK2LOC(parser, &parser->current);
19918 cast->base.location = cast->opening_loc;
19919
19920 if (lex_mode.quote == PM_HEREDOC_QUOTE_BACKTICK) {
19921 assert(sizeof(pm_string_node_t) == sizeof(pm_x_string_node_t));
19922 cast->base.type = PM_X_STRING_NODE;
19923 }
19924
19925 if (lex_mode.indent == PM_HEREDOC_INDENT_TILDE && (common_whitespace != (size_t) -1) && (common_whitespace != 0)) {
19926 parse_heredoc_dedent_string(parser->arena, &cast->unescaped, common_whitespace);
19927 }
19928
19929 node = UP(cast);
19930 expect1_heredoc_term(parser, lex_mode.ident_start, lex_mode.ident_length);
19931 } else {
19932 // If we get here, then we have multiple parts in the heredoc,
19933 // so we'll need to create an interpolated string node to hold
19934 // them all.
19935 pm_node_list_t parts = { 0 };
19936 pm_node_list_append(parser->arena, &parts, part);
19937
19938 while (!match2(parser, PM_TOKEN_HEREDOC_END, PM_TOKEN_EOF)) {
19939 if ((part = parse_string_part(parser, (uint16_t) (depth + 1))) != NULL) {
19940 pm_node_list_append(parser->arena, &parts, part);
19941 }
19942 }
19943
19944 // Now that we have all of the parts, create the correct type of
19945 // interpolated node.
19946 if (lex_mode.quote == PM_HEREDOC_QUOTE_BACKTICK) {
19947 pm_interpolated_x_string_node_t *cast = pm_interpolated_xstring_node_create(parser, &opening, &opening);
19948 cast->parts = parts;
19949
19950 expect1_heredoc_term(parser, lex_mode.ident_start, lex_mode.ident_length);
19951 pm_interpolated_xstring_node_closing_set(parser, cast, &parser->previous);
19952
19953 cast->base.location = cast->opening_loc;
19954 node = UP(cast);
19955 } else {
19956 pm_interpolated_string_node_t *cast = pm_interpolated_string_node_create(parser, &opening, &parts, &opening);
19957
19958 expect1_heredoc_term(parser, lex_mode.ident_start, lex_mode.ident_length);
19959 pm_interpolated_string_node_closing_set(parser, cast, &parser->previous);
19960
19961 cast->base.location = cast->opening_loc;
19962 node = UP(cast);
19963 }
19964
19965 // If this is a heredoc that is indented with a ~, then we need
19966 // to dedent each line by the common leading whitespace.
19967 if (lex_mode.indent == PM_HEREDOC_INDENT_TILDE && (common_whitespace != (size_t) -1) && (common_whitespace != 0)) {
19968 pm_node_list_t *nodes;
19969 if (lex_mode.quote == PM_HEREDOC_QUOTE_BACKTICK) {
19970 nodes = &((pm_interpolated_x_string_node_t *) node)->parts;
19971 } else {
19972 nodes = &((pm_interpolated_string_node_t *) node)->parts;
19973 }
19974
19975 parse_heredoc_dedent(parser, nodes, common_whitespace);
19976 }
19977 }
19978
19979 /* If a missing terminator left this heredoc's lex mode on the
19980 * stack, it still points at our stack-local common_whitespace.
19981 * Clear the pointer so that subsequent lexing cannot read from
19982 * this function's dead stack frame. */
19983 pm_lex_mode_t *whitespace_mode = parser->lex_modes.current;
19984 do {
19985 if (whitespace_mode->mode == PM_LEX_HEREDOC && whitespace_mode->as.heredoc.common_whitespace == &common_whitespace) {
19986 whitespace_mode->as.heredoc.common_whitespace = NULL;
19987 }
19988 whitespace_mode = whitespace_mode->prev;
19989 } while (whitespace_mode != NULL);
19990
19991 if (match1(parser, PM_TOKEN_STRING_BEGIN)) {
19992 return parse_strings(parser, node, false, (uint16_t) (depth + 1));
19993 }
19994
19995 return node;
19996 }
19997 case PM_TOKEN_INSTANCE_VARIABLE: {
19998 parser_lex(parser);
19999 pm_node_t *node = UP(pm_instance_variable_read_node_create(parser, &parser->previous));
20000
20001 if (binding_power == PM_BINDING_POWER_STATEMENT && match1(parser, PM_TOKEN_COMMA)) {
20002 node = parse_targets_validate(parser, node, PM_BINDING_POWER_INDEX, (uint16_t) (depth + 1));
20003 }
20004
20005 return node;
20006 }
20007 case PM_TOKEN_INTEGER: {
20008 pm_node_flags_t base = parser->integer.base;
20009 parser_lex(parser);
20010 return UP(pm_integer_node_create(parser, base, &parser->previous));
20011 }
20012 case PM_TOKEN_INTEGER_IMAGINARY: {
20013 pm_node_flags_t base = parser->integer.base;
20014 parser_lex(parser);
20015 return UP(pm_integer_node_imaginary_create(parser, base, &parser->previous));
20016 }
20017 case PM_TOKEN_INTEGER_RATIONAL: {
20018 pm_node_flags_t base = parser->integer.base;
20019 parser_lex(parser);
20020 return UP(pm_integer_node_rational_create(parser, base, &parser->previous));
20021 }
20022 case PM_TOKEN_INTEGER_RATIONAL_IMAGINARY: {
20023 pm_node_flags_t base = parser->integer.base;
20024 parser_lex(parser);
20025 return UP(pm_integer_node_rational_imaginary_create(parser, base, &parser->previous));
20026 }
20027 case PM_TOKEN_KEYWORD___ENCODING__:
20028 parser_lex(parser);
20029 return UP(pm_source_encoding_node_create(parser, &parser->previous));
20030 case PM_TOKEN_KEYWORD___FILE__:
20031 parser_lex(parser);
20032 return UP(pm_source_file_node_create(parser, &parser->previous));
20033 case PM_TOKEN_KEYWORD___LINE__:
20034 parser_lex(parser);
20035 return UP(pm_source_line_node_create(parser, &parser->previous));
20036 case PM_TOKEN_KEYWORD_ALIAS: {
20037 if (binding_power != PM_BINDING_POWER_STATEMENT && !(flags & PM_PARSE_ACCEPTS_STATEMENT)) {
20038 pm_parser_err_current(parser, PM_ERR_STATEMENT_ALIAS);
20039 }
20040
20041 parser_lex(parser);
20042 pm_token_t keyword = parser->previous;
20043
20044 pm_node_t *new_name = parse_alias_argument(parser, true, (uint16_t) (depth + 1));
20045 pm_node_t *old_name = parse_alias_argument(parser, false, (uint16_t) (depth + 1));
20046
20047 switch (PM_NODE_TYPE(new_name)) {
20048 case PM_BACK_REFERENCE_READ_NODE:
20049 case PM_NUMBERED_REFERENCE_READ_NODE:
20050 case PM_GLOBAL_VARIABLE_READ_NODE: {
20051 if (PM_NODE_TYPE_P(old_name, PM_BACK_REFERENCE_READ_NODE) || PM_NODE_TYPE_P(old_name, PM_NUMBERED_REFERENCE_READ_NODE) || PM_NODE_TYPE_P(old_name, PM_GLOBAL_VARIABLE_READ_NODE)) {
20052 if (PM_NODE_TYPE_P(old_name, PM_NUMBERED_REFERENCE_READ_NODE)) {
20053 pm_parser_err_node(parser, old_name, PM_ERR_ALIAS_ARGUMENT_NUMBERED_REFERENCE);
20054 }
20055 } else if (!PM_NODE_TYPE_P(old_name, PM_ERROR_RECOVERY_NODE)) {
20056 pm_parser_err_node(parser, old_name, PM_ERR_ALIAS_ARGUMENT);
20057 old_name = UP(pm_error_recovery_node_create_unexpected(parser, old_name));
20058 }
20059
20060 return UP(pm_alias_global_variable_node_create(parser, &keyword, new_name, old_name));
20061 }
20062 case PM_SYMBOL_NODE:
20063 case PM_INTERPOLATED_SYMBOL_NODE: {
20064 if (!PM_NODE_TYPE_P(old_name, PM_SYMBOL_NODE) && !PM_NODE_TYPE_P(old_name, PM_INTERPOLATED_SYMBOL_NODE) && !PM_NODE_TYPE_P(old_name, PM_ERROR_RECOVERY_NODE)) {
20065 pm_parser_err_node(parser, old_name, PM_ERR_ALIAS_ARGUMENT);
20066 old_name = UP(pm_error_recovery_node_create_unexpected(parser, old_name));
20067 }
20068 }
20070 default:
20071 return UP(pm_alias_method_node_create(parser, &keyword, new_name, old_name));
20072 }
20073 }
20074 case PM_TOKEN_KEYWORD_CASE:
20075 return parse_case(parser, flags, depth);
20076 case PM_TOKEN_KEYWORD_BEGIN: {
20077 size_t opening_newline_index = token_newline_index(parser);
20078 parser_lex(parser);
20079
20080 pm_token_t begin_keyword = parser->previous;
20081 accept2(parser, PM_TOKEN_NEWLINE, PM_TOKEN_SEMICOLON);
20082
20083 pm_node_list_t current_block_exits = { 0 };
20084 pm_node_list_t *previous_block_exits = push_block_exits(parser, &current_block_exits);
20085 pm_statements_node_t *begin_statements = NULL;
20086
20087 if (!match4(parser, PM_TOKEN_KEYWORD_RESCUE, PM_TOKEN_KEYWORD_ENSURE, PM_TOKEN_KEYWORD_ELSE, PM_TOKEN_KEYWORD_END)) {
20088 pm_accepts_block_stack_push(parser, true);
20089 begin_statements = parse_statements(parser, PM_CONTEXT_BEGIN, (uint16_t) (depth + 1));
20090 pm_accepts_block_stack_pop(parser);
20091 accept2(parser, PM_TOKEN_NEWLINE, PM_TOKEN_SEMICOLON);
20092 }
20093
20094 pm_begin_node_t *begin_node = pm_begin_node_create(parser, &begin_keyword, begin_statements);
20095 parse_rescues(parser, opening_newline_index, &begin_keyword, begin_node, PM_RESCUES_BEGIN, (uint16_t) (depth + 1));
20096 expect1_opening(parser, PM_TOKEN_KEYWORD_END, PM_ERR_BEGIN_TERM, &begin_keyword);
20097
20098 PM_NODE_LENGTH_SET_TOKEN(parser, begin_node, &parser->previous);
20099 pm_begin_node_end_keyword_set(parser, begin_node, &parser->previous);
20100 pop_block_exits(parser, previous_block_exits);
20101 return UP(begin_node);
20102 }
20103 case PM_TOKEN_KEYWORD_BEGIN_UPCASE: {
20104 pm_node_list_t current_block_exits = { 0 };
20105 pm_node_list_t *previous_block_exits = push_block_exits(parser, &current_block_exits);
20106
20107 if (binding_power != PM_BINDING_POWER_STATEMENT) {
20108 pm_parser_err_current(parser, PM_ERR_STATEMENT_PREEXE_BEGIN);
20109 }
20110
20111 parser_lex(parser);
20112 pm_token_t keyword = parser->previous;
20113
20114 expect1(parser, PM_TOKEN_BRACE_LEFT, PM_ERR_BEGIN_UPCASE_BRACE);
20115 pm_token_t opening = parser->previous;
20116 pm_statements_node_t *statements = parse_statements(parser, PM_CONTEXT_PREEXE, (uint16_t) (depth + 1));
20117
20118 expect1_opening(parser, PM_TOKEN_BRACE_RIGHT, PM_ERR_BEGIN_UPCASE_TERM, &opening);
20119 pm_context_t context = parser->current_context->context;
20120 if ((context != PM_CONTEXT_MAIN) && (context != PM_CONTEXT_PREEXE)) {
20121 pm_parser_err_token(parser, &keyword, PM_ERR_BEGIN_UPCASE_TOPLEVEL);
20122 }
20123
20124 flush_block_exits(parser, previous_block_exits);
20125 return UP(pm_pre_execution_node_create(parser, &keyword, &opening, statements, &parser->previous));
20126 }
20127 case PM_TOKEN_KEYWORD_BREAK:
20128 case PM_TOKEN_KEYWORD_NEXT:
20129 case PM_TOKEN_KEYWORD_RETURN: {
20130 parser_lex(parser);
20131
20132 pm_token_t keyword = parser->previous;
20133 pm_arguments_t arguments = { 0 };
20134
20135 if (
20136 token_begins_expression_p(parser->current.type) ||
20137 match2(parser, PM_TOKEN_USTAR, PM_TOKEN_USTAR_STAR)
20138 ) {
20139 pm_binding_power_t binding_power = pm_binding_powers[parser->current.type].left;
20140
20141 if (binding_power == PM_BINDING_POWER_UNSET || binding_power >= PM_BINDING_POWER_RANGE) {
20142 pm_token_t next = parser->current;
20143 parse_arguments(parser, &arguments, false, PM_TOKEN_EOF, flags, (uint16_t) (depth + 1));
20144
20145 // Reject `foo && return bar`.
20146 if (!(flags & PM_PARSE_ACCEPTS_COMMAND_CALL) && arguments.arguments != NULL) {
20147 PM_PARSER_ERR_TOKEN_FORMAT(parser, &next, PM_ERR_EXPECT_EOL_AFTER_STATEMENT, pm_token_str(next.type));
20148 }
20149
20150 // Reject a trailing comma, e.g. `return a,`. The arguments
20151 // parser silently accepts a trailing comma only when it is
20152 // immediately followed by the EOF terminator; in every other
20153 // case (e.g. `return a,;`) it reports the dangling comma
20154 // itself. We reject the accepted case here to stay in line
20155 // with the command call argument parsing above.
20156 if (parser->previous.type == PM_TOKEN_COMMA && match1(parser, PM_TOKEN_EOF)) {
20157 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->previous, PM_ERR_EXPECT_ARGUMENT, pm_token_str(parser->current.type));
20158 }
20159 }
20160
20161 // It's possible that we've parsed a block argument through our
20162 // call to parse_arguments. If we found one, we should mark it
20163 // as invalid and destroy it, as we don't have a place for it.
20164 if (arguments.block != NULL) {
20165 pm_parser_err_node(parser, arguments.block, PM_ERR_UNEXPECTED_BLOCK_ARGUMENT);
20166 pm_node_unreference(parser, arguments.block);
20167 arguments.block = NULL;
20168 }
20169 }
20170
20171 switch (keyword.type) {
20172 case PM_TOKEN_KEYWORD_BREAK: {
20173 pm_node_t *node = UP(pm_break_node_create(parser, &keyword, arguments.arguments));
20174 if (!parser->partial_script) parse_block_exit(parser, node);
20175 return node;
20176 }
20177 case PM_TOKEN_KEYWORD_NEXT: {
20178 pm_node_t *node = UP(pm_next_node_create(parser, &keyword, arguments.arguments));
20179 if (!parser->partial_script) parse_block_exit(parser, node);
20180 return node;
20181 }
20182 case PM_TOKEN_KEYWORD_RETURN: {
20183 pm_node_t *node = UP(pm_return_node_create(parser, &keyword, arguments.arguments));
20184 parse_return(parser, node);
20185 return node;
20186 }
20187 default:
20188 assert(false && "unreachable");
20189 return UP(pm_error_recovery_node_create(parser, PM_TOKEN_START(parser, &parser->previous), PM_TOKEN_LENGTH(&parser->previous)));
20190 }
20191 }
20192 case PM_TOKEN_KEYWORD_SUPER: {
20193 parser_lex(parser);
20194
20195 pm_token_t keyword = parser->previous;
20196 pm_arguments_t arguments = { 0 };
20197 parse_arguments_list(parser, &arguments, true, flags, (uint16_t) (depth + 1));
20198
20199 if (
20200 arguments.opening_loc.length == 0 &&
20201 arguments.arguments == NULL &&
20202 ((arguments.block == NULL) || PM_NODE_TYPE_P(arguments.block, PM_BLOCK_NODE))
20203 ) {
20204 return UP(pm_forwarding_super_node_create(parser, &keyword, &arguments));
20205 }
20206
20207 return UP(pm_super_node_create(parser, &keyword, &arguments));
20208 }
20209 case PM_TOKEN_KEYWORD_YIELD: {
20210 parser_lex(parser);
20211
20212 pm_token_t keyword = parser->previous;
20213 pm_arguments_t arguments = { 0 };
20214 parse_arguments_list(parser, &arguments, false, flags, (uint16_t) (depth + 1));
20215
20216 // It's possible that we've parsed a block argument through our
20217 // call to parse_arguments_list. If we found one, we should mark it
20218 // as invalid and destroy it, as we don't have a place for it on the
20219 // yield node.
20220 if (arguments.block != NULL) {
20221 pm_parser_err_node(parser, arguments.block, PM_ERR_UNEXPECTED_BLOCK_ARGUMENT);
20222 pm_node_unreference(parser, arguments.block);
20223 arguments.block = NULL;
20224 }
20225
20226 pm_node_t *node = UP(pm_yield_node_create(parser, &keyword, &arguments.opening_loc, arguments.arguments, &arguments.closing_loc));
20227 if (!parser->parsing_eval && !parser->partial_script) parse_yield(parser, node);
20228
20229 return node;
20230 }
20231 case PM_TOKEN_KEYWORD_CLASS:
20232 return parse_class(parser, flags, depth);
20233 case PM_TOKEN_KEYWORD_DEF:
20234 return parse_def(parser, binding_power, flags, depth);
20235 case PM_TOKEN_KEYWORD_DEFINED: {
20236 parser_lex(parser);
20237
20238 pm_token_t keyword = parser->previous;
20239 pm_token_t lparen = { 0 };
20240 pm_token_t rparen = { 0 };
20241 pm_node_t *expression;
20242
20243 context_push(parser, PM_CONTEXT_DEFINED);
20244 bool newline = accept1(parser, PM_TOKEN_NEWLINE);
20245
20246 if (accept2(parser, PM_TOKEN_PARENTHESIS_LEFT, PM_TOKEN_PARENTHESIS_LEFT_GROUPING)) {
20247 lparen = parser->previous;
20248
20249 if (newline && accept1(parser, PM_TOKEN_PARENTHESIS_RIGHT)) {
20250 expression = UP(pm_parentheses_node_create(parser, &lparen, NULL, &parser->previous, 0));
20251 lparen = (pm_token_t) { 0 };
20252 } else {
20253 expression = parse_expression(parser, PM_BINDING_POWER_COMPOSITION, PM_PARSE_ACCEPTS_COMMAND_CALL | PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_DEFINED_EXPRESSION, (uint16_t) (depth + 1));
20254
20255 if (!parser->recovering) {
20256 accept1(parser, PM_TOKEN_NEWLINE);
20257 expect1(parser, PM_TOKEN_PARENTHESIS_RIGHT, PM_ERR_EXPECT_RPAREN);
20258 rparen = parser->previous;
20259 }
20260 }
20261 } else {
20262 expression = parse_expression(parser, PM_BINDING_POWER_DEFINED, flags & PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_DEFINED_EXPRESSION, (uint16_t) (depth + 1));
20263 }
20264
20265 context_pop(parser);
20266 return UP(pm_defined_node_create(
20267 parser,
20268 NTOK2PTR(lparen),
20269 expression,
20270 NTOK2PTR(rparen),
20271 &keyword
20272 ));
20273 }
20274 case PM_TOKEN_KEYWORD_END_UPCASE: {
20275 if (binding_power != PM_BINDING_POWER_STATEMENT && !(flags & PM_PARSE_ACCEPTS_STATEMENT)) {
20276 pm_parser_err_current(parser, PM_ERR_STATEMENT_POSTEXE_END);
20277 }
20278
20279 parser_lex(parser);
20280 pm_token_t keyword = parser->previous;
20281
20282 if (context_def_p(parser)) {
20283 pm_parser_warn_token(parser, &keyword, PM_WARN_END_IN_METHOD);
20284 }
20285
20286 expect1(parser, PM_TOKEN_BRACE_LEFT, PM_ERR_END_UPCASE_BRACE);
20287 pm_token_t opening = parser->previous;
20288 pm_statements_node_t *statements = parse_statements(parser, PM_CONTEXT_POSTEXE, (uint16_t) (depth + 1));
20289
20290 expect1_opening(parser, PM_TOKEN_BRACE_RIGHT, PM_ERR_END_UPCASE_TERM, &opening);
20291 return UP(pm_post_execution_node_create(parser, &keyword, &opening, statements, &parser->previous));
20292 }
20293 case PM_TOKEN_KEYWORD_FALSE:
20294 parser_lex(parser);
20295 return UP(pm_false_node_create(parser, &parser->previous));
20296 case PM_TOKEN_KEYWORD_FOR: {
20297 size_t opening_newline_index = token_newline_index(parser);
20298 parser_lex(parser);
20299
20300 pm_token_t for_keyword = parser->previous;
20301 pm_node_t *index;
20302
20303 context_push(parser, PM_CONTEXT_FOR_INDEX);
20304
20305 // First, parse out the first index expression.
20306 if (accept1(parser, PM_TOKEN_USTAR)) {
20307 index = parse_splat(parser, flags, depth);
20308 } else if (token_begins_expression_p(parser->current.type)) {
20309 index = parse_expression(parser, PM_BINDING_POWER_INDEX, flags & PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_EXPECT_EXPRESSION_AFTER_COMMA, (uint16_t) (depth + 1));
20310 } else {
20311 pm_parser_err_token(parser, &for_keyword, PM_ERR_FOR_INDEX);
20312 index = UP(pm_error_recovery_node_create(parser, PM_TOKEN_START(parser, &for_keyword), PM_TOKEN_LENGTH(&for_keyword)));
20313 }
20314
20315 // Now, if there are multiple index expressions, parse them out.
20316 if (match1(parser, PM_TOKEN_COMMA)) {
20317 index = parse_targets(parser, index, PM_BINDING_POWER_INDEX, (uint16_t) (depth + 1));
20318 } else {
20319 index = parse_target(parser, index, false, false);
20320 }
20321
20322 context_pop(parser);
20323 pm_do_loop_stack_push(parser, true);
20324
20325 expect1(parser, PM_TOKEN_KEYWORD_IN, PM_ERR_FOR_IN);
20326 pm_token_t in_keyword = parser->previous;
20327
20328 pm_node_t *collection = parse_value_expression(parser, PM_BINDING_POWER_COMPOSITION, (uint8_t) ((flags & PM_PARSE_ACCEPTS_DO_BLOCK) | PM_PARSE_ACCEPTS_COMMAND_CALL), PM_ERR_FOR_COLLECTION, (uint16_t) (depth + 1));
20329 pm_do_loop_stack_pop(parser);
20330
20331 pm_token_t do_keyword = { 0 };
20332 if (accept1(parser, PM_TOKEN_KEYWORD_DO_LOOP)) {
20333 do_keyword = parser->previous;
20334 } else {
20335 if (!match2(parser, PM_TOKEN_SEMICOLON, PM_TOKEN_NEWLINE)) {
20336 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->current, PM_ERR_EXPECT_FOR_DELIMITER, pm_token_str(parser->current.type));
20337 }
20338 }
20339
20340 pm_statements_node_t *statements = NULL;
20341 if (!match1(parser, PM_TOKEN_KEYWORD_END)) {
20342 statements = parse_statements(parser, PM_CONTEXT_FOR, (uint16_t) (depth + 1));
20343 }
20344
20345 parser_warn_indentation_mismatch(parser, opening_newline_index, &for_keyword, false, false);
20346 expect1_opening(parser, PM_TOKEN_KEYWORD_END, PM_ERR_FOR_TERM, &for_keyword);
20347
20348 return UP(pm_for_node_create(parser, index, collection, statements, &for_keyword, &in_keyword, NTOK2PTR(do_keyword), &parser->previous));
20349 }
20350 case PM_TOKEN_KEYWORD_IF:
20351 if (parser_end_of_line_p(parser)) {
20352 PM_PARSER_WARN_TOKEN_FORMAT_CONTENT(parser, &parser->current, PM_WARN_KEYWORD_EOL);
20353 }
20354
20355 size_t opening_newline_index = token_newline_index(parser);
20356 bool if_after_else = parser->previous.type == PM_TOKEN_KEYWORD_ELSE;
20357 parser_lex(parser);
20358
20359 return parse_conditional(parser, PM_CONTEXT_IF, opening_newline_index, if_after_else, (uint16_t) (depth + 1));
20360 case PM_TOKEN_KEYWORD_UNDEF: {
20361 if (binding_power != PM_BINDING_POWER_STATEMENT && !(flags & PM_PARSE_ACCEPTS_STATEMENT)) {
20362 pm_parser_err_current(parser, PM_ERR_STATEMENT_UNDEF);
20363 }
20364
20365 parser_lex(parser);
20366 pm_undef_node_t *undef = pm_undef_node_create(parser, &parser->previous);
20367 pm_node_t *name = parse_undef_argument(parser, (uint16_t) (depth + 1));
20368
20369 if (PM_NODE_TYPE_P(name, PM_ERROR_RECOVERY_NODE)) {
20370 } else {
20371 pm_undef_node_append(parser->arena, undef, name);
20372
20373 while (match1(parser, PM_TOKEN_COMMA)) {
20374 lex_state_set(parser, PM_LEX_STATE_FNAME | PM_LEX_STATE_FITEM);
20375 parser_lex(parser);
20376 name = parse_undef_argument(parser, (uint16_t) (depth + 1));
20377
20378 if (PM_NODE_TYPE_P(name, PM_ERROR_RECOVERY_NODE)) {
20379 break;
20380 }
20381
20382 pm_undef_node_append(parser->arena, undef, name);
20383 }
20384 }
20385
20386 return UP(undef);
20387 }
20388 case PM_TOKEN_KEYWORD_NOT: {
20389 parser_lex(parser);
20390
20391 pm_token_t message = parser->previous;
20392 pm_arguments_t arguments = { 0 };
20393 pm_node_t *receiver = NULL;
20394
20395 // The `not` keyword without parentheses is only valid in contexts
20396 // where it would be parsed as an expression (i.e., at or below
20397 // the `not` binding power level). In other contexts (e.g., method
20398 // arguments, array elements, assignment right-hand sides),
20399 // parentheses are required: `not(x)`. An exception is made for
20400 // endless def bodies, where `not` is valid as both `arg` and
20401 // `command` (e.g., `def f = not 1`, `def f = not foo bar`).
20402 if (binding_power > PM_BINDING_POWER_NOT && !(flags & PM_PARSE_IN_ENDLESS_DEF) && !match1(parser, PM_TOKEN_PARENTHESIS_LEFT)) {
20403 if (match1(parser, PM_TOKEN_PARENTHESIS_LEFT_PARENTHESES)) {
20404 pm_parser_err(parser, PM_TOKEN_END(parser, &parser->previous), 1, PM_ERR_EXPECT_LPAREN_AFTER_NOT_LPAREN);
20405 } else {
20406 accept1(parser, PM_TOKEN_NEWLINE);
20407 pm_parser_err_current(parser, PM_ERR_EXPECT_LPAREN_AFTER_NOT_OTHER);
20408 }
20409
20410 return UP(pm_error_recovery_node_create(parser, PM_TOKEN_START(parser, &parser->current), PM_TOKEN_LENGTH(&parser->current)));
20411 }
20412
20413 accept1(parser, PM_TOKEN_NEWLINE);
20414
20415 if (accept2(parser, PM_TOKEN_PARENTHESIS_LEFT, PM_TOKEN_PARENTHESIS_LEFT_GROUPING)) {
20416 pm_token_t lparen = parser->previous;
20417
20418 if (accept1(parser, PM_TOKEN_PARENTHESIS_RIGHT)) {
20419 receiver = UP(pm_parentheses_node_create(parser, &lparen, NULL, &parser->previous, 0));
20420 } else {
20421 arguments.opening_loc = TOK2LOC(parser, &lparen);
20422 receiver = parse_expression(parser, PM_BINDING_POWER_COMPOSITION, PM_PARSE_ACCEPTS_COMMAND_CALL | PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_NOT_EXPRESSION, (uint16_t) (depth + 1));
20423
20424 if (!parser->recovering) {
20425 accept1(parser, PM_TOKEN_NEWLINE);
20426 expect1(parser, PM_TOKEN_PARENTHESIS_RIGHT, PM_ERR_EXPECT_RPAREN);
20427 arguments.closing_loc = TOK2LOC(parser, &parser->previous);
20428 }
20429 }
20430 } else {
20431 receiver = parse_expression(parser, PM_BINDING_POWER_NOT, (uint8_t) ((flags & PM_PARSE_ACCEPTS_DO_BLOCK) | PM_PARSE_ACCEPTS_COMMAND_CALL), PM_ERR_NOT_EXPRESSION, (uint16_t) (depth + 1));
20432 }
20433
20434 return UP(pm_call_node_not_create(parser, receiver, &message, &arguments));
20435 }
20436 case PM_TOKEN_KEYWORD_UNLESS: {
20437 size_t opening_newline_index = token_newline_index(parser);
20438 parser_lex(parser);
20439
20440 return parse_conditional(parser, PM_CONTEXT_UNLESS, opening_newline_index, false, (uint16_t) (depth + 1));
20441 }
20442 case PM_TOKEN_KEYWORD_MODULE:
20443 return parse_module(parser, flags, depth);
20444 case PM_TOKEN_KEYWORD_NIL:
20445 parser_lex(parser);
20446 return UP(pm_nil_node_create(parser, &parser->previous));
20447 case PM_TOKEN_KEYWORD_REDO: {
20448 parser_lex(parser);
20449
20450 pm_node_t *node = UP(pm_redo_node_create(parser, &parser->previous));
20451 if (!parser->partial_script) parse_block_exit(parser, node);
20452
20453 return node;
20454 }
20455 case PM_TOKEN_KEYWORD_RETRY: {
20456 parser_lex(parser);
20457
20458 pm_node_t *node = UP(pm_retry_node_create(parser, &parser->previous));
20459 parse_retry(parser, node);
20460
20461 return node;
20462 }
20463 case PM_TOKEN_KEYWORD_SELF:
20464 parser_lex(parser);
20465 return UP(pm_self_node_create(parser, &parser->previous));
20466 case PM_TOKEN_KEYWORD_TRUE:
20467 parser_lex(parser);
20468 return UP(pm_true_node_create(parser, &parser->previous));
20469 case PM_TOKEN_KEYWORD_UNTIL: {
20470 size_t opening_newline_index = token_newline_index(parser);
20471
20472 context_push(parser, PM_CONTEXT_LOOP_PREDICATE);
20473 pm_do_loop_stack_push(parser, true);
20474
20475 parser_lex(parser);
20476 pm_token_t keyword = parser->previous;
20477 pm_node_t *predicate = parse_value_expression(parser, PM_BINDING_POWER_COMPOSITION, (uint8_t) ((flags & PM_PARSE_ACCEPTS_DO_BLOCK) | PM_PARSE_ACCEPTS_COMMAND_CALL), PM_ERR_CONDITIONAL_UNTIL_PREDICATE, (uint16_t) (depth + 1));
20478
20479 pm_do_loop_stack_pop(parser);
20480 context_pop(parser);
20481
20482 pm_token_t do_keyword = { 0 };
20483 if (accept1(parser, PM_TOKEN_KEYWORD_DO_LOOP)) {
20484 do_keyword = parser->previous;
20485 } else {
20486 expect2(parser, PM_TOKEN_NEWLINE, PM_TOKEN_SEMICOLON, PM_ERR_CONDITIONAL_UNTIL_PREDICATE);
20487 }
20488
20489 pm_statements_node_t *statements = NULL;
20490 if (!match1(parser, PM_TOKEN_KEYWORD_END)) {
20491 pm_accepts_block_stack_push(parser, true);
20492 statements = parse_statements(parser, PM_CONTEXT_UNTIL, (uint16_t) (depth + 1));
20493 pm_accepts_block_stack_pop(parser);
20494 accept2(parser, PM_TOKEN_NEWLINE, PM_TOKEN_SEMICOLON);
20495 }
20496
20497 parser_warn_indentation_mismatch(parser, opening_newline_index, &keyword, false, false);
20498 expect1_opening(parser, PM_TOKEN_KEYWORD_END, PM_ERR_UNTIL_TERM, &keyword);
20499
20500 return UP(pm_until_node_create(parser, &keyword, NTOK2PTR(do_keyword), &parser->previous, predicate, statements, 0));
20501 }
20502 case PM_TOKEN_KEYWORD_WHILE: {
20503 size_t opening_newline_index = token_newline_index(parser);
20504
20505 context_push(parser, PM_CONTEXT_LOOP_PREDICATE);
20506 pm_do_loop_stack_push(parser, true);
20507
20508 parser_lex(parser);
20509 pm_token_t keyword = parser->previous;
20510 pm_node_t *predicate = parse_value_expression(parser, PM_BINDING_POWER_COMPOSITION, (uint8_t) ((flags & PM_PARSE_ACCEPTS_DO_BLOCK) | PM_PARSE_ACCEPTS_COMMAND_CALL), PM_ERR_CONDITIONAL_WHILE_PREDICATE, (uint16_t) (depth + 1));
20511
20512 pm_do_loop_stack_pop(parser);
20513 context_pop(parser);
20514
20515 pm_token_t do_keyword = { 0 };
20516 if (accept1(parser, PM_TOKEN_KEYWORD_DO_LOOP)) {
20517 do_keyword = parser->previous;
20518 } else {
20519 expect2(parser, PM_TOKEN_NEWLINE, PM_TOKEN_SEMICOLON, PM_ERR_CONDITIONAL_WHILE_PREDICATE);
20520 }
20521
20522 pm_statements_node_t *statements = NULL;
20523 if (!match1(parser, PM_TOKEN_KEYWORD_END)) {
20524 pm_accepts_block_stack_push(parser, true);
20525 statements = parse_statements(parser, PM_CONTEXT_WHILE, (uint16_t) (depth + 1));
20526 pm_accepts_block_stack_pop(parser);
20527 accept2(parser, PM_TOKEN_NEWLINE, PM_TOKEN_SEMICOLON);
20528 }
20529
20530 parser_warn_indentation_mismatch(parser, opening_newline_index, &keyword, false, false);
20531 expect1_opening(parser, PM_TOKEN_KEYWORD_END, PM_ERR_WHILE_TERM, &keyword);
20532
20533 return UP(pm_while_node_create(parser, &keyword, NTOK2PTR(do_keyword), &parser->previous, predicate, statements, 0));
20534 }
20535 case PM_TOKEN_PERCENT_LOWER_I: {
20536 parser_lex(parser);
20537 pm_token_t opening = parser->previous;
20538 pm_array_node_t *array = pm_array_node_create(parser, &opening);
20539 pm_node_t *current = NULL;
20540
20541 while (!match2(parser, PM_TOKEN_STRING_END, PM_TOKEN_EOF)) {
20542 accept1(parser, PM_TOKEN_WORDS_SEP);
20543 if (match1(parser, PM_TOKEN_STRING_END)) break;
20544
20545 // Interpolation is not possible but nested heredocs can still lead to
20546 // consecutive (disjoint) string tokens when the final newline is escaped.
20547 while (match1(parser, PM_TOKEN_STRING_CONTENT)) {
20548 // Record the string node, moving to interpolation if needed.
20549 if (current == NULL) {
20550 current = UP(pm_symbol_node_create_current_string(parser, NULL, &parser->current, NULL));
20551 parser_lex(parser);
20552 } else if (PM_NODE_TYPE_P(current, PM_INTERPOLATED_SYMBOL_NODE)) {
20553 pm_node_t *string = UP(pm_string_node_create_current_string(parser, NULL, &parser->current, NULL));
20554 parser_lex(parser);
20555 pm_interpolated_symbol_node_append(parser->arena, (pm_interpolated_symbol_node_t *) current, string);
20556 } else if (PM_NODE_TYPE_P(current, PM_SYMBOL_NODE)) {
20557 pm_symbol_node_t *cast = (pm_symbol_node_t *) current;
20558 pm_token_t content = { .type = PM_TOKEN_STRING_CONTENT, .start = parser->start + cast->content_loc.start, .end = parser->start + cast->content_loc.start + cast->content_loc.length };
20559 pm_node_t *first_string = UP(pm_string_node_create_unescaped(parser, NULL, &content, NULL, &cast->unescaped));
20560 pm_node_t *second_string = UP(pm_string_node_create_current_string(parser, NULL, &parser->previous, NULL));
20561 parser_lex(parser);
20562
20563 pm_interpolated_symbol_node_t *interpolated = pm_interpolated_symbol_node_create(parser, NULL, NULL, NULL);
20564 pm_interpolated_symbol_node_append(parser->arena, interpolated, first_string);
20565 pm_interpolated_symbol_node_append(parser->arena, interpolated, second_string);
20566
20567 // current is arena-allocated so no explicit free is needed.
20568 current = UP(interpolated);
20569 } else {
20570 assert(false && "unreachable");
20571 }
20572 }
20573
20574 if (current) {
20575 pm_array_node_elements_append(parser->arena, array, current);
20576 current = NULL;
20577 } else {
20578 expect1(parser, PM_TOKEN_STRING_CONTENT, PM_ERR_LIST_I_LOWER_ELEMENT);
20579 }
20580 }
20581
20582 pm_token_t closing = parser->current;
20583 if (match1(parser, PM_TOKEN_EOF)) {
20584 pm_parser_err_token(parser, &opening, PM_ERR_LIST_I_LOWER_TERM);
20585 closing = (pm_token_t) { .type = 0, .start = parser->previous.end, .end = parser->previous.end };
20586 } else {
20587 expect1(parser, PM_TOKEN_STRING_END, PM_ERR_LIST_I_LOWER_TERM);
20588 }
20589 pm_array_node_close_set(parser, array, &closing);
20590
20591 return UP(array);
20592 }
20593 case PM_TOKEN_PERCENT_UPPER_I:
20594 return parse_symbol_array(parser, depth);
20595 case PM_TOKEN_PERCENT_LOWER_W: {
20596 parser_lex(parser);
20597 pm_token_t opening = parser->previous;
20598 pm_array_node_t *array = pm_array_node_create(parser, &opening);
20599 pm_node_t *current = NULL;
20600
20601 while (!match2(parser, PM_TOKEN_STRING_END, PM_TOKEN_EOF)) {
20602 accept1(parser, PM_TOKEN_WORDS_SEP);
20603 if (match1(parser, PM_TOKEN_STRING_END)) break;
20604
20605 // Interpolation is not possible but nested heredocs can still lead to
20606 // consecutive (disjoint) string tokens when the final newline is escaped.
20607 while (match1(parser, PM_TOKEN_STRING_CONTENT)) {
20608 pm_node_t *string = UP(pm_string_node_create_current_string(parser, NULL, &parser->current, NULL));
20609
20610 // Record the string node, moving to interpolation if needed.
20611 if (current == NULL) {
20612 current = string;
20613 } else if (PM_NODE_TYPE_P(current, PM_INTERPOLATED_STRING_NODE)) {
20614 pm_interpolated_string_node_append(parser, (pm_interpolated_string_node_t *) current, string);
20615 } else if (PM_NODE_TYPE_P(current, PM_STRING_NODE)) {
20616 pm_interpolated_string_node_t *interpolated = pm_interpolated_string_node_create(parser, NULL, NULL, NULL);
20617 pm_interpolated_string_node_append(parser, interpolated, current);
20618 pm_interpolated_string_node_append(parser, interpolated, string);
20619 current = UP(interpolated);
20620 } else {
20621 assert(false && "unreachable");
20622 }
20623 parser_lex(parser);
20624 }
20625
20626 if (current) {
20627 pm_array_node_elements_append(parser->arena, array, current);
20628 current = NULL;
20629 } else {
20630 expect1(parser, PM_TOKEN_STRING_CONTENT, PM_ERR_LIST_W_LOWER_ELEMENT);
20631 }
20632 }
20633
20634 pm_token_t closing = parser->current;
20635 if (match1(parser, PM_TOKEN_EOF)) {
20636 pm_parser_err_token(parser, &opening, PM_ERR_LIST_W_LOWER_TERM);
20637 closing = (pm_token_t) { .type = 0, .start = parser->previous.end, .end = parser->previous.end };
20638 } else {
20639 expect1(parser, PM_TOKEN_STRING_END, PM_ERR_LIST_W_LOWER_TERM);
20640 }
20641
20642 pm_array_node_close_set(parser, array, &closing);
20643 return UP(array);
20644 }
20645 case PM_TOKEN_PERCENT_UPPER_W:
20646 return parse_string_array(parser, depth);
20647 case PM_TOKEN_REGEXP_BEGIN: {
20648 pm_token_t opening = parser->current;
20649 parser_lex(parser);
20650
20651 if (match1(parser, PM_TOKEN_REGEXP_END)) {
20652 // If we get here, then we have an end immediately after a start. In
20653 // that case we'll create an empty content token and return an
20654 // uninterpolated regular expression.
20655 pm_token_t content = (pm_token_t) {
20656 .type = PM_TOKEN_STRING_CONTENT,
20657 .start = parser->previous.end,
20658 .end = parser->previous.end
20659 };
20660
20661 parser_lex(parser);
20662
20663 pm_regular_expression_node_t *node = pm_regular_expression_node_create(parser, &opening, &content, &parser->previous);
20664 pm_node_flag_set(UP(node), pm_regexp_parse(parser, node, NULL, NULL));
20665 return UP(node);
20666 }
20667
20669
20670 if (match1(parser, PM_TOKEN_STRING_CONTENT)) {
20671 // In this case we've hit string content so we know the regular
20672 // expression at least has something in it. We'll need to check if the
20673 // following token is the end (in which case we can return a plain
20674 // regular expression) or if it's not then it has interpolation.
20675 pm_string_t unescaped = parser->current_string;
20676 pm_token_t content = parser->current;
20677 parser_lex(parser);
20678
20679 // If we hit an end, then we can create a regular expression
20680 // node without interpolation, which can be represented more
20681 // succinctly and more easily compiled.
20682 if (accept1(parser, PM_TOKEN_REGEXP_END)) {
20683 pm_regular_expression_node_t *node = (pm_regular_expression_node_t *) pm_regular_expression_node_create_unescaped(parser, &opening, &content, &parser->previous, &unescaped);
20684
20685 // If we're not immediately followed by a =~, then we
20686 // parse and validate now. If it is followed by a =~,
20687 // then it will get parsed in the =~ handler where
20688 // named captures can also be extracted.
20689 if (!match1(parser, PM_TOKEN_EQUAL_TILDE)) {
20690 pm_node_flag_set(UP(node), pm_regexp_parse(parser, node, NULL, NULL));
20691 }
20692
20693 return UP(node);
20694 }
20695
20696 // If we get here, then we have interpolation so we'll need to create
20697 // a regular expression node with interpolation.
20698 interpolated = pm_interpolated_regular_expression_node_create(parser, &opening);
20699
20700 pm_node_t *part = UP(pm_string_node_create_unescaped(parser, NULL, &parser->previous, NULL, &unescaped));
20701 if (parser->encoding == PM_ENCODING_US_ASCII_ENTRY) {
20702 // This is extremely strange, but the first string part of a
20703 // regular expression will always be tagged as binary if we
20704 // are in a US-ASCII file, no matter its contents.
20705 pm_node_flag_set(part, PM_STRING_FLAGS_FORCED_BINARY_ENCODING);
20706 }
20707
20708 pm_interpolated_regular_expression_node_append(parser->arena, interpolated, part);
20709 } else {
20710 // If the first part of the body of the regular expression is not a
20711 // string content, then we have interpolation and we need to create an
20712 // interpolated regular expression node.
20713 interpolated = pm_interpolated_regular_expression_node_create(parser, &opening);
20714 }
20715
20716 // Now that we're here and we have interpolation, we'll parse all of the
20717 // parts into the list.
20718 pm_node_t *part;
20719 while (!match2(parser, PM_TOKEN_REGEXP_END, PM_TOKEN_EOF)) {
20720 if ((part = parse_string_part(parser, (uint16_t) (depth + 1))) != NULL) {
20721 pm_interpolated_regular_expression_node_append(parser->arena, interpolated, part);
20722 }
20723 }
20724
20725 pm_token_t closing = parser->current;
20726 if (match1(parser, PM_TOKEN_EOF)) {
20727 pm_parser_err_token(parser, &opening, PM_ERR_REGEXP_TERM);
20728 closing = (pm_token_t) { .type = 0, .start = parser->previous.end, .end = parser->previous.end };
20729 } else {
20730 expect1(parser, PM_TOKEN_REGEXP_END, PM_ERR_REGEXP_TERM);
20731 }
20732
20733 pm_interpolated_regular_expression_node_closing_set(parser, interpolated, &closing);
20734 return UP(interpolated);
20735 }
20736 case PM_TOKEN_XSTRING_BEGIN:
20737 case PM_TOKEN_PERCENT_LOWER_X: {
20738 parser_lex(parser);
20739 pm_token_t opening = parser->previous;
20740
20741 // When we get here, we don't know if this string is going to have
20742 // interpolation or not, even though it is allowed. Still, we want to be
20743 // able to return a string node without interpolation if we can since
20744 // it'll be faster.
20745 if (match1(parser, PM_TOKEN_STRING_END)) {
20746 // If we get here, then we have an end immediately after a start. In
20747 // that case we'll create an empty content token and return an
20748 // uninterpolated string.
20749 pm_token_t content = (pm_token_t) {
20750 .type = PM_TOKEN_STRING_CONTENT,
20751 .start = parser->previous.end,
20752 .end = parser->previous.end
20753 };
20754
20755 parser_lex(parser);
20756 return UP(pm_xstring_node_create(parser, &opening, &content, &parser->previous));
20757 }
20758
20760
20761 if (match1(parser, PM_TOKEN_STRING_CONTENT)) {
20762 // In this case we've hit string content so we know the string
20763 // at least has something in it. We'll need to check if the
20764 // following token is the end (in which case we can return a
20765 // plain string) or if it's not then it has interpolation.
20766 pm_string_t unescaped = parser->current_string;
20767 pm_token_t content = parser->current;
20768 parser_lex(parser);
20769
20770 if (match1(parser, PM_TOKEN_STRING_END)) {
20771 pm_node_t *node = UP(pm_xstring_node_create_unescaped(parser, &opening, &content, &parser->current, &unescaped));
20772 pm_node_flag_set(node, parse_unescaped_encoding(parser, parser->explicit_encoding));
20773 parser_lex(parser);
20774 return node;
20775 }
20776
20777 // If we get here, then we have interpolation so we'll need to
20778 // create a string node with interpolation.
20779 node = pm_interpolated_xstring_node_create(parser, &opening, &opening);
20780
20781 pm_node_t *part = UP(pm_string_node_create_unescaped(parser, NULL, &parser->previous, NULL, &unescaped));
20782 pm_node_flag_set(part, parse_unescaped_encoding(parser, parser->explicit_encoding));
20783
20784 pm_interpolated_xstring_node_append(parser->arena, node, part);
20785 } else {
20786 // If the first part of the body of the string is not a string
20787 // content, then we have interpolation and we need to create an
20788 // interpolated string node.
20789 node = pm_interpolated_xstring_node_create(parser, &opening, &opening);
20790 }
20791
20792 pm_node_t *part;
20793 while (!match2(parser, PM_TOKEN_STRING_END, PM_TOKEN_EOF)) {
20794 if ((part = parse_string_part(parser, (uint16_t) (depth + 1))) != NULL) {
20795 pm_interpolated_xstring_node_append(parser->arena, node, part);
20796 }
20797 }
20798
20799 pm_token_t closing = parser->current;
20800 if (match1(parser, PM_TOKEN_EOF)) {
20801 pm_parser_err_token(parser, &opening, PM_ERR_XSTRING_TERM);
20802 closing = (pm_token_t) { .type = 0, .start = parser->previous.end, .end = parser->previous.end };
20803 } else {
20804 expect1(parser, PM_TOKEN_STRING_END, PM_ERR_XSTRING_TERM);
20805 }
20806 pm_interpolated_xstring_node_closing_set(parser, node, &closing);
20807
20808 return UP(node);
20809 }
20810 case PM_TOKEN_USTAR: {
20811 parser_lex(parser);
20812
20813 // * operators at the beginning of expressions are only valid in the
20814 // context of a multiple assignment. We enforce that here. We'll
20815 // still lex past it though and create a missing node place.
20816 if (binding_power != PM_BINDING_POWER_STATEMENT) {
20817 pm_parser_err_prefix(parser, diag_id);
20818 return UP(pm_error_recovery_node_create(parser, PM_TOKEN_START(parser, &parser->previous), PM_TOKEN_LENGTH(&parser->previous)));
20819 }
20820
20821 pm_node_t *splat = parse_splat(parser, flags, depth);
20822
20823 if (match1(parser, PM_TOKEN_COMMA)) {
20824 return parse_targets_validate(parser, splat, PM_BINDING_POWER_INDEX, (uint16_t) (depth + 1));
20825 } else {
20826 return parse_target_validate(parser, splat, true);
20827 }
20828 }
20829 case PM_TOKEN_BANG: {
20830 if (binding_power > PM_BINDING_POWER_UNARY) {
20831 pm_parser_err_prefix(parser, PM_ERR_UNARY_DISALLOWED);
20832 }
20833
20834 parser_lex(parser);
20835
20836 pm_token_t operator = parser->previous;
20837 pm_node_t *receiver = parse_expression(parser, pm_binding_powers[parser->previous.type].right, (uint8_t) ((flags & PM_PARSE_ACCEPTS_DO_BLOCK) | (binding_power < PM_BINDING_POWER_MATCH ? PM_PARSE_ACCEPTS_COMMAND_CALL : 0)), PM_ERR_UNARY_RECEIVER, (uint16_t) (depth + 1));
20838 pm_call_node_t *node = pm_call_node_unary_create(parser, &operator, receiver, "!");
20839
20840 pm_conditional_predicate(parser, receiver, PM_CONDITIONAL_PREDICATE_TYPE_NOT);
20841 return UP(node);
20842 }
20843 case PM_TOKEN_TILDE: {
20844 if (binding_power > PM_BINDING_POWER_UNARY) {
20845 pm_parser_err_prefix(parser, PM_ERR_UNARY_DISALLOWED);
20846 }
20847 parser_lex(parser);
20848
20849 pm_token_t operator = parser->previous;
20850 pm_node_t *receiver = parse_expression(parser, pm_binding_powers[parser->previous.type].right, flags & PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_UNARY_RECEIVER, (uint16_t) (depth + 1));
20851 pm_call_node_t *node = pm_call_node_unary_create(parser, &operator, receiver, "~");
20852
20853 return UP(node);
20854 }
20855 case PM_TOKEN_UMINUS: {
20856 if (binding_power > PM_BINDING_POWER_UNARY) {
20857 pm_parser_err_prefix(parser, PM_ERR_UNARY_DISALLOWED);
20858 }
20859 parser_lex(parser);
20860
20861 pm_token_t operator = parser->previous;
20862 pm_node_t *receiver = parse_expression(parser, pm_binding_powers[parser->previous.type].right, flags & PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_UNARY_RECEIVER, (uint16_t) (depth + 1));
20863 pm_call_node_t *node = pm_call_node_unary_create(parser, &operator, receiver, "-@");
20864
20865 return UP(node);
20866 }
20867 case PM_TOKEN_UMINUS_NUM: {
20868 parser_lex(parser);
20869
20870 pm_token_t operator = parser->previous;
20871 pm_node_t *node = parse_expression(parser, pm_binding_powers[parser->previous.type].right, flags & PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_UNARY_RECEIVER, (uint16_t) (depth + 1));
20872
20873 if (accept1(parser, PM_TOKEN_STAR_STAR)) {
20874 pm_token_t exponent_operator = parser->previous;
20875 pm_node_t *exponent = parse_expression(parser, pm_binding_powers[exponent_operator.type].right, flags & PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_EXPECT_ARGUMENT, (uint16_t) (depth + 1));
20876 node = UP(pm_call_node_binary_create(parser, node, &exponent_operator, exponent, 0));
20877 node = UP(pm_call_node_unary_create(parser, &operator, node, "-@"));
20878 } else {
20879 switch (PM_NODE_TYPE(node)) {
20880 case PM_INTEGER_NODE:
20881 case PM_FLOAT_NODE:
20882 case PM_RATIONAL_NODE:
20883 case PM_IMAGINARY_NODE:
20884 parse_negative_numeric(node);
20885 break;
20886 default:
20887 node = UP(pm_call_node_unary_create(parser, &operator, node, "-@"));
20888 break;
20889 }
20890 }
20891
20892 return node;
20893 }
20894 case PM_TOKEN_MINUS_GREATER: {
20895 int previous_lambda_enclosure_nesting = parser->lambda_enclosure_nesting;
20896 parser->lambda_enclosure_nesting = parser->enclosure_nesting;
20897
20898 size_t opening_newline_index = token_newline_index(parser);
20899 parser_lex(parser);
20900
20901 pm_token_t operator = parser->previous;
20902 pm_parser_scope_push(parser, false);
20903
20904 pm_block_parameters_node_t *block_parameters;
20905
20906 switch (parser->current.type) {
20907 case PM_TOKEN_PARENTHESIS_LEFT: {
20908 pm_token_t opening = parser->current;
20909 parser_lex(parser);
20910
20911 if (match1(parser, PM_TOKEN_PARENTHESIS_RIGHT)) {
20912 block_parameters = pm_block_parameters_node_create(parser, NULL, &opening);
20913 } else {
20914 block_parameters = parse_block_parameters(parser, false, &opening, true, true, (uint16_t) (depth + 1));
20915 }
20916
20917 accept1(parser, PM_TOKEN_NEWLINE);
20918 expect1(parser, PM_TOKEN_PARENTHESIS_RIGHT, PM_ERR_EXPECT_RPAREN);
20919
20920 pm_block_parameters_node_closing_set(parser, block_parameters, &parser->previous);
20921 break;
20922 }
20923 case PM_CASE_PARAMETER: {
20924 block_parameters = parse_block_parameters(parser, false, NULL, true, false, (uint16_t) (depth + 1));
20925 break;
20926 }
20927 default: {
20928 block_parameters = NULL;
20929 break;
20930 }
20931 }
20932
20933 pm_token_t opening;
20934 pm_node_t *body = NULL;
20935
20936 if (accept1(parser, PM_TOKEN_LAMBDA_BEGIN)) {
20937 opening = parser->previous;
20938
20939 if (!match1(parser, PM_TOKEN_BRACE_RIGHT)) {
20940 body = UP(parse_statements(parser, PM_CONTEXT_LAMBDA_BRACES, (uint16_t) (depth + 1)));
20941 }
20942
20943 parser_warn_indentation_mismatch(parser, opening_newline_index, &operator, false, false);
20944
20945 /* Restore the enclosing lambda's nesting now that the body has
20946 * been parsed, so that the token following the closing `}` is
20947 * lexed in the enclosing context. During the body the nesting
20948 * held this lambda's own level, which every token inside the
20949 * braces sits above. This mirrors parse.y restoring
20950 * `p->lex.lpar_beg` after `lambda_body`. */
20951 parser->lambda_enclosure_nesting = previous_lambda_enclosure_nesting;
20952 expect1_opening(parser, PM_TOKEN_BRACE_RIGHT, PM_ERR_LAMBDA_TERM_BRACE, &opening);
20953 } else {
20954 /* A `-> { }` body is delimited by `{`/`}`, whose block-accepting
20955 * frame the lexer manages. A `-> do end` body is delimited by
20956 * keywords, so push the frame here and pop it before `end`. The
20957 * push must precede consuming the `do`, which lexes the first
20958 * token of the body; this matches parse.y's CMDARG_PUSH(0)
20959 * before `lambda_body`. */
20960 pm_accepts_block_stack_push(parser, true);
20961 expect1(parser, PM_TOKEN_KEYWORD_DO_LAMBDA, PM_ERR_LAMBDA_OPEN);
20962 opening = parser->previous;
20963
20964 /* The lexer cleared the nesting when it produced the `do`. If
20965 * it was missing entirely, clear it here so that the body is
20966 * recovered the same way it would have been parsed: no token
20967 * within it sits at the beginning of a lambda. */
20968 parser->lambda_enclosure_nesting = -1;
20969
20970 if (!match3(parser, PM_TOKEN_KEYWORD_END, PM_TOKEN_KEYWORD_RESCUE, PM_TOKEN_KEYWORD_ENSURE)) {
20971 body = UP(parse_statements(parser, PM_CONTEXT_LAMBDA_DO_END, (uint16_t) (depth + 1)));
20972 }
20973
20974 if (match2(parser, PM_TOKEN_KEYWORD_RESCUE, PM_TOKEN_KEYWORD_ENSURE)) {
20975 assert(body == NULL || PM_NODE_TYPE_P(body, PM_STATEMENTS_NODE));
20976 body = UP(parse_rescues_implicit_begin(parser, opening_newline_index, &operator, opening.start, (pm_statements_node_t *) body, PM_RESCUES_LAMBDA, (uint16_t) (depth + 1)));
20977 } else {
20978 parser_warn_indentation_mismatch(parser, opening_newline_index, &operator, false, false);
20979 }
20980
20981 pm_accepts_block_stack_pop(parser);
20982
20983 /* As with the brace branch above, restore the nesting before
20984 * consuming the closing `end`, which lexes the token that
20985 * follows it. */
20986 parser->lambda_enclosure_nesting = previous_lambda_enclosure_nesting;
20987 expect1_opening(parser, PM_TOKEN_KEYWORD_END, PM_ERR_LAMBDA_TERM_END, &operator);
20988 }
20989
20990 pm_constant_id_list_t locals;
20991 pm_locals_order(parser, &parser->current_scope->locals, &locals, pm_parser_scope_toplevel_p(parser));
20992 pm_node_t *parameters = parse_blocklike_parameters(parser, UP(block_parameters), &operator, &parser->previous);
20993
20994 pm_parser_scope_pop(parser);
20995
20996 return UP(pm_lambda_node_create(parser, &locals, &operator, &opening, &parser->previous, parameters, body));
20997 }
20998 case PM_TOKEN_UPLUS: {
20999 if (binding_power > PM_BINDING_POWER_UNARY) {
21000 pm_parser_err_prefix(parser, PM_ERR_UNARY_DISALLOWED);
21001 }
21002 parser_lex(parser);
21003
21004 pm_token_t operator = parser->previous;
21005 pm_node_t *receiver = parse_expression(parser, pm_binding_powers[parser->previous.type].right, flags & PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_UNARY_RECEIVER, (uint16_t) (depth + 1));
21006 pm_call_node_t *node = pm_call_node_unary_create(parser, &operator, receiver, "+@");
21007
21008 return UP(node);
21009 }
21010 case PM_TOKEN_STRING_BEGIN:
21011 return parse_strings(parser, NULL, flags & PM_PARSE_ACCEPTS_LABEL, (uint16_t) (depth + 1));
21012 case PM_TOKEN_SYMBOL_BEGIN: {
21013 pm_lex_mode_t lex_mode = *parser->lex_modes.current;
21014 parser_lex(parser);
21015
21016 return parse_symbol(parser, &lex_mode, PM_LEX_STATE_END, (uint16_t) (depth + 1));
21017 }
21018 default: {
21019 pm_context_t recoverable = context_recoverable(parser, &parser->current);
21020
21021 if (recoverable != PM_CONTEXT_NONE) {
21022 parser->recovering = true;
21023
21024 // If the given error is not the generic one, then we'll add it
21025 // here because it will provide more context in addition to the
21026 // recoverable error that we will also add.
21027 if (diag_id != PM_ERR_CANNOT_PARSE_EXPRESSION) {
21028 pm_parser_err_prefix(parser, diag_id);
21029 }
21030
21031 // If we get here, then we are assuming this token is closing a
21032 // parent context, so we'll indicate that to the user so that
21033 // they know how we behaved.
21034 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->current, PM_ERR_UNEXPECTED_TOKEN_CLOSE_CONTEXT, pm_token_str(parser->current.type), context_human(recoverable));
21035 } else if (diag_id == PM_ERR_CANNOT_PARSE_EXPRESSION) {
21036 // We're going to make a special case here, because "cannot
21037 // parse expression" is pretty generic, and we know here that we
21038 // have an unexpected token.
21039 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->current, PM_ERR_UNEXPECTED_TOKEN_IGNORE, pm_token_str(parser->current.type));
21040 } else {
21041 pm_parser_err_prefix(parser, diag_id);
21042 }
21043
21044 return UP(pm_error_recovery_node_create(parser, PM_TOKEN_START(parser, &parser->previous), PM_TOKEN_LENGTH(&parser->previous)));
21045 }
21046 }
21047}
21048
21049static pm_node_t *
21050parse_rescue_modifier_value(pm_parser_t *parser, uint8_t flags, bool statement, uint16_t depth);
21051
21059static void
21060parse_rescue_modifier_terminator(pm_parser_t *parser, uint8_t flags, uint16_t depth) {
21061 if (pm_binding_powers[parser->current.type].left > PM_BINDING_POWER_MODIFIER) {
21062 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->current, PM_ERR_EXPECT_EOL_AFTER_STATEMENT, pm_token_str(parser->current.type));
21063 parser_lex(parser);
21064 parse_expression(parser, pm_binding_powers[parser->previous.type].right, flags & PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_EXPECT_EXPRESSION_AFTER_OPERATOR, (uint16_t) (depth + 1));
21065 }
21066}
21067
21077static pm_node_t *
21078parse_assignment_value(pm_parser_t *parser, pm_binding_power_t previous_binding_power, pm_binding_power_t binding_power, uint8_t flags, pm_diagnostic_id_t diag_id, uint16_t depth) {
21079 pm_node_t *value = parse_value_expression(parser, binding_power, (uint8_t) ((flags & PM_PARSE_ACCEPTS_DO_BLOCK) | (previous_binding_power == PM_BINDING_POWER_ASSIGNMENT ? (flags & PM_PARSE_ACCEPTS_COMMAND_CALL) : (previous_binding_power < PM_BINDING_POWER_MATCH ? PM_PARSE_ACCEPTS_COMMAND_CALL : 0))), diag_id, (uint16_t) (depth + 1));
21080
21081 // Assignments whose value is a command call (e.g., a = b c) can only
21082 // be followed by modifiers (if/unless/while/until/rescue) and not by
21083 // operators with higher binding power. If we find one, emit an error
21084 // and skip the operator and its right-hand side.
21085 if (pm_binding_powers[parser->current.type].left > PM_BINDING_POWER_MODIFIER && (pm_command_call_value_p(parser, value) || pm_block_call_p(value))) {
21086 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->current, PM_ERR_EXPECT_EOL_AFTER_STATEMENT, pm_token_str(parser->current.type));
21087 parser_lex(parser);
21088 parse_expression(parser, pm_binding_powers[parser->previous.type].right, flags & PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_EXPECT_EXPRESSION_AFTER_OPERATOR, (uint16_t) (depth + 1));
21089 }
21090
21091 // Contradicting binding powers, the right-hand-side value of the assignment
21092 // allows the `rescue` modifier.
21093 if (match1(parser, PM_TOKEN_KEYWORD_RESCUE_MODIFIER)) {
21094 context_push(parser, PM_CONTEXT_RESCUE_MODIFIER);
21095
21096 pm_token_t rescue = parser->current;
21097 parser_lex(parser);
21098
21099 // As in parse_assignment_values, the resbody is a `stmt` (permitting a
21100 // multiple assignment / command call) when the rescued value is itself a
21101 // command call, and a plain `arg` otherwise.
21102 bool statement_value = pm_command_call_value_p(parser, value) || pm_block_call_p(value);
21103 uint8_t rescue_flags = (uint8_t) ((flags & PM_PARSE_ACCEPTS_DO_BLOCK) | (statement_value ? PM_PARSE_ACCEPTS_COMMAND_CALL : 0));
21104
21105 pm_node_t *right = parse_rescue_modifier_value(parser, rescue_flags, statement_value, (uint16_t) (depth + 1));
21106 context_pop(parser);
21107
21108 // A pattern-match resbody is a statement, but here the rescue is nested
21109 // in an assignment value where parse_expression_terminator cannot see
21110 // it, so reject a trailing operator above the modifier level directly.
21111 if (PM_NODE_TYPE_P(right, PM_MATCH_REQUIRED_NODE) || PM_NODE_TYPE_P(right, PM_MATCH_PREDICATE_NODE)) {
21112 parse_rescue_modifier_terminator(parser, flags, depth);
21113 }
21114
21115 return UP(pm_rescue_modifier_node_create(parser, value, &rescue, right));
21116 }
21117
21118 return value;
21119}
21120
21125static void
21126parse_assignment_value_local(pm_parser_t *parser, const pm_node_t *node) {
21127 switch (PM_NODE_TYPE(node)) {
21128 case PM_BEGIN_NODE: {
21129 const pm_begin_node_t *cast = (const pm_begin_node_t *) node;
21130 if (cast->statements != NULL) parse_assignment_value_local(parser, (const pm_node_t *) cast->statements);
21131 break;
21132 }
21133 case PM_LOCAL_VARIABLE_WRITE_NODE: {
21135 pm_locals_read(&pm_parser_scope_find(parser, cast->depth)->locals, cast->name);
21136 break;
21137 }
21138 case PM_PARENTHESES_NODE: {
21139 const pm_parentheses_node_t *cast = (const pm_parentheses_node_t *) node;
21140 if (cast->body != NULL) parse_assignment_value_local(parser, cast->body);
21141 break;
21142 }
21143 case PM_STATEMENTS_NODE: {
21144 const pm_statements_node_t *cast = (const pm_statements_node_t *) node;
21145 const pm_node_t *statement;
21146
21147 PM_NODE_LIST_FOREACH(&cast->body, index, statement) {
21148 parse_assignment_value_local(parser, statement);
21149 }
21150 break;
21151 }
21152 default:
21153 break;
21154 }
21155}
21156
21169static pm_node_t *
21170parse_assignment_values(pm_parser_t *parser, pm_binding_power_t previous_binding_power, pm_binding_power_t binding_power, uint8_t flags, pm_diagnostic_id_t diag_id, uint16_t depth) {
21171 bool statement_level = (previous_binding_power == PM_BINDING_POWER_STATEMENT) || (flags & PM_PARSE_ACCEPTS_STATEMENT);
21172
21173 bool permitted = true;
21174 if (!statement_level && match1(parser, PM_TOKEN_USTAR)) permitted = false;
21175
21176 // A command call (e.g. `x = y z`) is permitted as the value when assigning
21177 // directly (carrying the caller's flag), or in any statement-level context
21178 // — which includes a rescue modifier value via the flag.
21179 uint8_t command_call_flag = (previous_binding_power == PM_BINDING_POWER_ASSIGNMENT)
21180 ? (uint8_t) (flags & PM_PARSE_ACCEPTS_COMMAND_CALL)
21181 : ((previous_binding_power < PM_BINDING_POWER_MODIFIER || statement_level) ? PM_PARSE_ACCEPTS_COMMAND_CALL : 0);
21182
21183 pm_node_t *value = parse_starred_expression(parser, binding_power, (uint8_t) ((flags & PM_PARSE_ACCEPTS_DO_BLOCK) | command_call_flag), diag_id, (uint16_t) (depth + 1));
21184 if (!permitted) pm_parser_err_node(parser, value, PM_ERR_UNEXPECTED_MULTI_WRITE);
21185
21186 parse_assignment_value_local(parser, value);
21187 bool single_value = true;
21188
21189 // Block calls (command call + do block, e.g., `foo bar do end`) cannot
21190 // be followed by a comma to form a multi-value RHS because each element
21191 // of a multi-value assignment must be an `arg`, not a `block_call`.
21192 if (statement_level && !pm_block_call_p(value) && (PM_NODE_TYPE_P(value, PM_SPLAT_NODE) || match1(parser, PM_TOKEN_COMMA))) {
21193 single_value = false;
21194
21195 pm_array_node_t *array = pm_array_node_create(parser, NULL);
21196 pm_array_node_elements_append(parser->arena, array, value);
21197 value = UP(array);
21198
21199 while (accept1(parser, PM_TOKEN_COMMA)) {
21200 pm_node_t *element = parse_starred_expression(parser, binding_power, false, PM_ERR_ARRAY_ELEMENT, (uint16_t) (depth + 1));
21201
21202 pm_array_node_elements_append(parser->arena, array, element);
21203 if (PM_NODE_TYPE_P(element, PM_ERROR_RECOVERY_NODE)) break;
21204
21205 parse_assignment_value_local(parser, element);
21206 }
21207 }
21208
21209 // Assignments whose value is a command call (e.g., a = b c) can only
21210 // be followed by modifiers (if/unless/while/until/rescue) and not by
21211 // operators with higher binding power. If we find one, emit an error
21212 // and skip the operator and its right-hand side.
21213 if (single_value && pm_binding_powers[parser->current.type].left > PM_BINDING_POWER_MODIFIER && (pm_command_call_value_p(parser, value) || pm_block_call_p(value))) {
21214 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->current, PM_ERR_EXPECT_EOL_AFTER_STATEMENT, pm_token_str(parser->current.type));
21215 parser_lex(parser);
21216 parse_expression(parser, pm_binding_powers[parser->previous.type].right, flags & PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_EXPECT_EXPRESSION_AFTER_OPERATOR, (uint16_t) (depth + 1));
21217 }
21218
21219 // Contradicting binding powers, the right-hand-side value of the assignment
21220 // allows the `rescue` modifier.
21221 bool multiple_assignment = (binding_power == (PM_BINDING_POWER_MULTI_ASSIGNMENT + 1));
21222 if ((single_value || multiple_assignment) && match1(parser, PM_TOKEN_KEYWORD_RESCUE_MODIFIER)) {
21223 bool command_value = pm_command_call_value_p(parser, value) || pm_block_call_p(value);
21224
21225 // A multiple assignment whose value is a command call (`x, y = foo
21226 // bar`) is a complete statement (parse.y: `mlhs '='
21227 // command_call_value`, which has no rescue), so a trailing `rescue`
21228 // modifies the whole assignment rather than the value. Leave it for the
21229 // statement-level rescue instead of binding it to the value here. For a
21230 // non-command value the rescue does bind to the value (parse.y:
21231 // `mlhs '=' mrhs_arg modifier_rescue stmt`).
21232 if (multiple_assignment && command_value) return value;
21233
21234 context_push(parser, PM_CONTEXT_RESCUE_MODIFIER);
21235
21236 pm_token_t rescue = parser->current;
21237 parser_lex(parser);
21238
21239 // The resbody is a `stmt` (parse.y: `command_rhs`/`mlhs '=' mrhs_arg`),
21240 // which permits a multiple assignment and a command call, when this is a
21241 // multiple assignment or the rescued value is itself a command call.
21242 // Otherwise it is a plain `arg` (parse.y: `arg_rhs`).
21243 bool statement_value = multiple_assignment || command_value;
21244 uint8_t rescue_flags = (uint8_t) ((flags & PM_PARSE_ACCEPTS_DO_BLOCK) | (statement_value ? PM_PARSE_ACCEPTS_COMMAND_CALL : 0));
21245
21246 pm_node_t *right = parse_rescue_modifier_value(parser, rescue_flags, statement_value, (uint16_t) (depth + 1));
21247 context_pop(parser);
21248
21249 // A pattern-match resbody is a statement, but here the rescue is nested
21250 // in an assignment value where parse_expression_terminator cannot see
21251 // it, so reject a trailing operator above the modifier level directly.
21252 if (PM_NODE_TYPE_P(right, PM_MATCH_REQUIRED_NODE) || PM_NODE_TYPE_P(right, PM_MATCH_PREDICATE_NODE)) {
21253 parse_rescue_modifier_terminator(parser, flags, depth);
21254 }
21255
21256 return UP(pm_rescue_modifier_node_create(parser, value, &rescue, right));
21257 }
21258
21259 return value;
21260}
21261
21274static pm_node_t *
21275parse_rescue_modifier_value(pm_parser_t *parser, uint8_t flags, bool statement, uint16_t depth) {
21276 if (statement) {
21277 pm_node_t *value;
21278 bool multiple;
21279
21280 if (match1(parser, PM_TOKEN_USTAR)) {
21281 // A leading splat can only begin a multiple assignment target list.
21282 parser_lex(parser);
21283 value = parse_splat(parser, flags, depth);
21284 multiple = true;
21285 } else {
21286 // The flag lets a single-target assignment take a multiple-value or
21287 // splat right-hand side (`b = c, d` / `b = *c`); a comma _before_ an
21288 // `=`, or a parenthesized target list (`(b, c), d = 1`), instead
21289 // promotes to a multiple assignment target list below.
21290 value = parse_expression(parser, pm_binding_powers[PM_TOKEN_KEYWORD_RESCUE_MODIFIER].right, flags | PM_PARSE_ACCEPTS_STATEMENT, PM_ERR_RESCUE_MODIFIER_VALUE, (uint16_t) (depth + 1));
21291 multiple = match1(parser, PM_TOKEN_COMMA) || PM_NODE_TYPE_P(value, PM_MULTI_TARGET_NODE);
21292 }
21293
21294 if (multiple) {
21295 pm_node_t *target = parse_targets_validate(parser, value, PM_BINDING_POWER_INDEX, (uint16_t) (depth + 1));
21296
21297 // A promoted target list is only a valid rescue value as part of a
21298 // complete `targets = values`. parse_targets_validate already
21299 // reports a missing `=` for every terminator except `)` (which it
21300 // permits for an enclosing mlhs paren that does not apply here), so
21301 // reject that case.
21302 if (!match1(parser, PM_TOKEN_EQUAL)) {
21303 if (match1(parser, PM_TOKEN_PARENTHESIS_RIGHT)) pm_parser_err_node(parser, target, PM_ERR_WRITE_TARGET_UNEXPECTED);
21304 return target;
21305 }
21306
21307 pm_token_t operator = parser->current;
21308 parser_lex(parser);
21309
21310 pm_node_t *values = parse_assignment_values(parser, PM_BINDING_POWER_STATEMENT, PM_BINDING_POWER_MULTI_ASSIGNMENT + 1, flags, PM_ERR_EXPECT_EXPRESSION_AFTER_EQUAL, (uint16_t) (depth + 1));
21311 value = parse_write(parser, target, &operator, values);
21312 }
21313
21314 // Reject a trailing operator that cannot follow a statement resbody.
21315 // Pattern-match handlers are statements too, but for the bare statement
21316 // form they are reported by parse_expression_terminator instead (which
21317 // keeps its existing error-recovery), so they are excluded here.
21318 if (!PM_NODE_TYPE_P(value, PM_MATCH_REQUIRED_NODE) && !PM_NODE_TYPE_P(value, PM_MATCH_PREDICATE_NODE)) {
21319 parse_rescue_modifier_terminator(parser, flags, depth);
21320 }
21321
21322 return value;
21323 }
21324
21325 // Otherwise the resbody is a plain `arg` (parse.y: `arg modifier_rescue
21326 // arg`), parsed above the `and`/`or`/`not` level so those stay outside it.
21327 return parse_expression(parser, PM_BINDING_POWER_DEFINED, flags, PM_ERR_RESCUE_MODIFIER_VALUE, (uint16_t) (depth + 1));
21328}
21329
21337static void
21338parse_call_operator_write(pm_parser_t *parser, pm_call_node_t *call_node, const pm_token_t *operator) {
21339 if (call_node->arguments != NULL) {
21340 pm_parser_err_token(parser, operator, PM_ERR_OPERATOR_WRITE_ARGUMENTS);
21341 pm_node_unreference(parser, UP(call_node->arguments));
21342 call_node->arguments = NULL;
21343 }
21344
21345 if (call_node->block != NULL) {
21346 pm_parser_err_token(parser, operator, PM_ERR_OPERATOR_WRITE_BLOCK);
21347 pm_node_unreference(parser, UP(call_node->block));
21348 call_node->block = NULL;
21349 }
21350}
21351
21352static PRISM_INLINE const uint8_t *
21353pm_named_capture_escape_hex(pm_buffer_t *unescaped, const uint8_t *cursor, const uint8_t *end) {
21354 cursor++;
21355
21356 if (cursor < end && pm_char_is_hexadecimal_digit(*cursor)) {
21357 uint8_t value = escape_hexadecimal_digit(*cursor);
21358 cursor++;
21359
21360 if (cursor < end && pm_char_is_hexadecimal_digit(*cursor)) {
21361 value = (uint8_t) ((value << 4) | escape_hexadecimal_digit(*cursor));
21362 cursor++;
21363 }
21364
21365 pm_buffer_append_byte(unescaped, value);
21366 } else {
21367 pm_buffer_append_string(unescaped, "\\x", 2);
21368 }
21369
21370 return cursor;
21371}
21372
21373static PRISM_INLINE const uint8_t *
21374pm_named_capture_escape_octal(pm_buffer_t *unescaped, const uint8_t *cursor, const uint8_t *end) {
21375 uint8_t value = (uint8_t) (*cursor - '0');
21376 cursor++;
21377
21378 if (cursor < end && pm_char_is_octal_digit(*cursor)) {
21379 value = ((uint8_t) (value << 3)) | ((uint8_t) (*cursor - '0'));
21380 cursor++;
21381
21382 if (cursor < end && pm_char_is_octal_digit(*cursor)) {
21383 value = ((uint8_t) (value << 3)) | ((uint8_t) (*cursor - '0'));
21384 cursor++;
21385 }
21386 }
21387
21388 pm_buffer_append_byte(unescaped, value);
21389 return cursor;
21390}
21391
21392static PRISM_INLINE const uint8_t *
21393pm_named_capture_escape_unicode(pm_parser_t *parser, pm_buffer_t *unescaped, const uint8_t *cursor, const uint8_t *end, const pm_location_t *error_location) {
21394 const uint8_t *start = cursor - 1;
21395 cursor++;
21396
21397 if (cursor >= end) {
21398 pm_buffer_append_string(unescaped, "\\u", 2);
21399 return cursor;
21400 }
21401
21402 if (*cursor != '{') {
21403 size_t length = pm_strspn_hexadecimal_digit(cursor, MIN(end - cursor, 4));
21404 uint32_t value = escape_unicode(parser, cursor, length, error_location, 0);
21405
21406 if (!pm_buffer_append_unicode_codepoint(unescaped, value)) {
21407 pm_buffer_append_string(unescaped, (const char *) start, (size_t) ((cursor + length) - start));
21408 }
21409
21410 return cursor + length;
21411 }
21412
21413 cursor++;
21414 for (;;) {
21415 while (cursor < end && *cursor == ' ') cursor++;
21416
21417 if (cursor >= end) break;
21418 if (*cursor == '}') {
21419 cursor++;
21420 break;
21421 }
21422
21423 size_t length = pm_strspn_hexadecimal_digit(cursor, end - cursor);
21424 if (length == 0) {
21425 break;
21426 }
21427 uint32_t value = escape_unicode(parser, cursor, length, error_location, 0);
21428
21429 (void) pm_buffer_append_unicode_codepoint(unescaped, value);
21430 cursor += length;
21431 }
21432
21433 return cursor;
21434}
21435
21436static void
21437pm_named_capture_escape(pm_parser_t *parser, pm_buffer_t *unescaped, const uint8_t *source, const size_t length, const uint8_t *cursor, const pm_location_t *error_location) {
21438 const uint8_t *end = source + length;
21439 pm_buffer_append_string(unescaped, (const char *) source, (size_t) (cursor - source));
21440
21441 for (;;) {
21442 if (++cursor >= end) {
21443 pm_buffer_append_byte(unescaped, '\\');
21444 return;
21445 }
21446
21447 switch (*cursor) {
21448 case 'x':
21449 cursor = pm_named_capture_escape_hex(unescaped, cursor, end);
21450 break;
21451 case '0': case '1': case '2': case '3': case '4': case '5': case '6': case '7':
21452 cursor = pm_named_capture_escape_octal(unescaped, cursor, end);
21453 break;
21454 case 'u':
21455 cursor = pm_named_capture_escape_unicode(parser, unescaped, cursor, end, error_location);
21456 break;
21457 default:
21458 pm_buffer_append_byte(unescaped, '\\');
21459 break;
21460 }
21461
21462 const uint8_t *next_cursor = pm_memchr(cursor, '\\', (size_t) (end - cursor), parser->encoding_changed, parser->encoding);
21463 if (next_cursor == NULL) break;
21464
21465 pm_buffer_append_string(unescaped, (const char *) cursor, (size_t) (next_cursor - cursor));
21466 cursor = next_cursor;
21467 }
21468
21469 pm_buffer_append_string(unescaped, (const char *) cursor, (size_t) (end - cursor));
21470}
21471
21476static void
21477parse_regular_expression_named_capture(pm_parser_t *parser, const pm_string_t *capture, bool shared, pm_regexp_name_data_t *callback_data) {
21478 pm_call_node_t *call = callback_data->call;
21479 pm_constant_id_set_t *names = &callback_data->names;
21480
21481 const uint8_t *source = pm_string_source(capture);
21482 size_t length = pm_string_length(capture);
21483 pm_buffer_t unescaped = { 0 };
21484
21485 // First, we need to handle escapes within the name of the capture group.
21486 // This is because regular expressions have three different representations
21487 // in prism. The first is the plain source code. The second is the
21488 // representation that will be sent to the regular expression engine, which
21489 // is the value of the "unescaped" field. This is poorly named, because it
21490 // actually still contains escapes, just a subset of them that the regular
21491 // expression engine knows how to handle. The third representation is fully
21492 // unescaped, which is what we need.
21493 const uint8_t *cursor = pm_memchr(source, '\\', length, parser->encoding_changed, parser->encoding);
21494 if (PRISM_UNLIKELY(cursor != NULL)) {
21495 pm_named_capture_escape(parser, &unescaped, source, length, cursor, shared ? NULL : &call->receiver->location);
21496 source = (const uint8_t *) pm_buffer_value(&unescaped);
21497 length = pm_buffer_length(&unescaped);
21498 }
21499
21500 const uint8_t *start;
21501 const uint8_t *end;
21502 pm_constant_id_t name;
21503
21504 // If the name of the capture group isn't a valid identifier, we do
21505 // not add it to the local table.
21506 if (!pm_slice_is_valid_local(parser, source, source + length)) {
21507 pm_buffer_cleanup(&unescaped);
21508 return;
21509 }
21510
21511 if (shared) {
21512 // If the unescaped string is a slice of the source, then we can
21513 // copy the names directly. The pointers will line up.
21514 start = source;
21515 end = source + length;
21516 name = pm_parser_constant_id_raw(parser, start, end);
21517 } else {
21518 // Otherwise, the name is a slice of the malloc-ed owned string,
21519 // in which case we need to copy it out into a new string.
21520 start = parser->start + PM_NODE_START(call->receiver);
21521 end = parser->start + PM_NODE_END(call->receiver);
21522
21523 uint8_t *memory = (uint8_t *) pm_arena_alloc(parser->arena, length, 1);
21524 memcpy(memory, source, length);
21525 name = pm_parser_constant_id_owned(parser, memory, length);
21526 }
21527
21528 /*
21529 * Add this name to the set of constants if it is valid, not duplicated,
21530 * and not a keyword.
21531 */
21532 if (name != 0 && pm_constant_id_set_insert(parser->arena, names, name)) {
21533
21534 int depth;
21535 if ((depth = pm_parser_local_depth_constant_id(parser, name)) == -1) {
21536 // If the local is not already a local but it is a keyword, then we
21537 // do not want to add a capture for this.
21538 if (pm_local_is_keyword((const char *) source, length)) {
21539 pm_buffer_cleanup(&unescaped);
21540 return;
21541 }
21542
21543 // If the identifier is not already a local, then we will add it to
21544 // the local table.
21545 pm_parser_local_add(parser, name, start, end, 0);
21546 }
21547
21548 // Here we lazily create the MatchWriteNode since we know we're
21549 // about to add a target.
21550 if (callback_data->match == NULL) {
21551 callback_data->match = pm_match_write_node_create(parser, call);
21552 }
21553
21554 // Next, create the local variable target and add it to the list of
21555 // targets for the match.
21556 pm_token_t token = { .type = 0, .start = start, .end = end };
21557 pm_location_t token_loc = TOK2LOC(parser, &token);
21558 pm_node_t *target = UP(pm_local_variable_target_node_create(parser, &token_loc, name, depth == -1 ? 0 : (uint32_t) depth));
21559 pm_node_list_append(parser->arena, &callback_data->match->targets, target);
21560 }
21561
21562 pm_buffer_cleanup(&unescaped);
21563}
21564
21570static pm_node_t *
21571parse_interpolated_regular_expression_named_captures(pm_parser_t *parser, const pm_string_t *content, pm_call_node_t *call, bool extended_mode) {
21572 pm_regexp_name_data_t callback_data = {
21573 .call = call,
21574 .match = NULL,
21575 .names = { 0 },
21576 };
21577
21578 pm_regexp_parse_named_captures(parser, pm_string_source(content), pm_string_length(content), false, extended_mode, parse_regular_expression_named_capture, &callback_data);
21579
21580 if (callback_data.match != NULL) {
21581 return UP(callback_data.match);
21582 } else {
21583 return UP(call);
21584 }
21585}
21586
21587static PRISM_INLINE pm_node_t *
21588parse_expression_infix(pm_parser_t *parser, pm_node_t *node, pm_binding_power_t previous_binding_power, pm_binding_power_t binding_power, uint8_t flags, uint16_t depth) {
21589 pm_token_t token = parser->current;
21590
21591 switch (token.type) {
21592 case PM_TOKEN_EQUAL: {
21593 switch (PM_NODE_TYPE(node)) {
21594 case PM_CALL_NODE: {
21595 // If we have no arguments to the call node and we need this
21596 // to be a target then this is either a method call or a
21597 // local variable write. This _must_ happen before the value
21598 // is parsed because it could be referenced in the value.
21599 pm_call_node_t *call_node = (pm_call_node_t *) node;
21600 if (PM_NODE_FLAG_P(call_node, PM_CALL_NODE_FLAGS_VARIABLE_CALL)) {
21601 pm_parser_local_add_location(parser, &call_node->message_loc, 0);
21602 }
21603 }
21605 case PM_CASE_WRITABLE: {
21606 // When we have `it = value`, we need to add `it` as a local
21607 // variable before parsing the value, in case the value
21608 // references the variable.
21609 if (PM_NODE_TYPE_P(node, PM_IT_LOCAL_VARIABLE_READ_NODE)) {
21610 pm_parser_local_add_location(parser, &node->location, 0);
21611 }
21612
21613 parser_lex(parser);
21614 pm_node_t *value = parse_assignment_values(parser, previous_binding_power, PM_NODE_TYPE_P(node, PM_MULTI_TARGET_NODE) ? PM_BINDING_POWER_MULTI_ASSIGNMENT + 1 : binding_power, flags, PM_ERR_EXPECT_EXPRESSION_AFTER_EQUAL, (uint16_t) (depth + 1));
21615
21616 if (PM_NODE_TYPE_P(node, PM_MULTI_TARGET_NODE) && previous_binding_power != PM_BINDING_POWER_STATEMENT && !(flags & PM_PARSE_ACCEPTS_STATEMENT)) {
21617 pm_parser_err_node(parser, node, PM_ERR_UNEXPECTED_MULTI_WRITE);
21618 }
21619
21620 return parse_write(parser, node, &token, value);
21621 }
21622 case PM_SPLAT_NODE: {
21623 pm_multi_target_node_t *multi_target = pm_multi_target_node_create(parser);
21624 pm_multi_target_node_targets_append(parser, multi_target, node);
21625
21626 parser_lex(parser);
21627 pm_node_t *value = parse_assignment_values(parser, previous_binding_power, PM_BINDING_POWER_MULTI_ASSIGNMENT + 1, flags, PM_ERR_EXPECT_EXPRESSION_AFTER_EQUAL, (uint16_t) (depth + 1));
21628 return parse_write(parser, UP(multi_target), &token, value);
21629 }
21630 case PM_SOURCE_ENCODING_NODE:
21631 case PM_FALSE_NODE:
21632 case PM_SOURCE_FILE_NODE:
21633 case PM_SOURCE_LINE_NODE:
21634 case PM_NIL_NODE:
21635 case PM_SELF_NODE:
21636 case PM_TRUE_NODE: {
21637 // In these special cases, we have specific error messages
21638 // and we will replace them with local variable writes.
21639 parser_lex(parser);
21640 pm_node_t *value = parse_assignment_values(parser, previous_binding_power, binding_power, flags, PM_ERR_EXPECT_EXPRESSION_AFTER_EQUAL, (uint16_t) (depth + 1));
21641 return parse_unwriteable_write(parser, node, &token, value);
21642 }
21643 default:
21644 // In this case we have an = sign, but we don't know what
21645 // it's for. We need to treat it as an error. We'll mark it
21646 // as an error and skip past it.
21647 parser_lex(parser);
21648 pm_parser_err_token(parser, &token, PM_ERR_EXPRESSION_NOT_WRITABLE);
21649 return node;
21650 }
21651 }
21652 case PM_TOKEN_AMPERSAND_AMPERSAND_EQUAL: {
21653 switch (PM_NODE_TYPE(node)) {
21654 case PM_BACK_REFERENCE_READ_NODE:
21655 case PM_NUMBERED_REFERENCE_READ_NODE:
21656 PM_PARSER_ERR_NODE_FORMAT_CONTENT(parser, node, PM_ERR_WRITE_TARGET_READONLY);
21658 case PM_GLOBAL_VARIABLE_READ_NODE: {
21659 parser_lex(parser);
21660
21661 pm_node_t *value = parse_assignment_value(parser, previous_binding_power, binding_power, flags, PM_ERR_EXPECT_EXPRESSION_AFTER_AMPAMPEQ, (uint16_t) (depth + 1));
21662 pm_node_t *result = UP(pm_global_variable_and_write_node_create(parser, node, &token, value));
21663
21664 return result;
21665 }
21666 case PM_CLASS_VARIABLE_READ_NODE: {
21667 parser_lex(parser);
21668
21669 pm_node_t *value = parse_assignment_value(parser, previous_binding_power, binding_power, flags, PM_ERR_EXPECT_EXPRESSION_AFTER_AMPAMPEQ, (uint16_t) (depth + 1));
21670 pm_node_t *result = UP(pm_class_variable_and_write_node_create(parser, (pm_class_variable_read_node_t *) node, &token, value));
21671
21672 return result;
21673 }
21674 case PM_CONSTANT_PATH_NODE: {
21675 parser_lex(parser);
21676
21677 pm_node_t *value = parse_assignment_value(parser, previous_binding_power, binding_power, flags, PM_ERR_EXPECT_EXPRESSION_AFTER_AMPAMPEQ, (uint16_t) (depth + 1));
21678 pm_node_t *write = UP(pm_constant_path_and_write_node_create(parser, (pm_constant_path_node_t *) node, &token, value));
21679
21680 return parse_shareable_constant_write(parser, write);
21681 }
21682 case PM_CONSTANT_READ_NODE: {
21683 parser_lex(parser);
21684
21685 pm_node_t *value = parse_assignment_value(parser, previous_binding_power, binding_power, flags, PM_ERR_EXPECT_EXPRESSION_AFTER_AMPAMPEQ, (uint16_t) (depth + 1));
21686 pm_node_t *write = UP(pm_constant_and_write_node_create(parser, (pm_constant_read_node_t *) node, &token, value));
21687
21688 if (context_def_p(parser)) {
21689 pm_parser_err_node(parser, write, PM_ERR_WRITE_TARGET_IN_METHOD);
21690 }
21691
21692 return parse_shareable_constant_write(parser, write);
21693 }
21694 case PM_INSTANCE_VARIABLE_READ_NODE: {
21695 parser_lex(parser);
21696
21697 pm_node_t *value = parse_assignment_value(parser, previous_binding_power, binding_power, flags, PM_ERR_EXPECT_EXPRESSION_AFTER_AMPAMPEQ, (uint16_t) (depth + 1));
21698 pm_node_t *result = UP(pm_instance_variable_and_write_node_create(parser, (pm_instance_variable_read_node_t *) node, &token, value));
21699
21700 return result;
21701 }
21702 case PM_IT_LOCAL_VARIABLE_READ_NODE: {
21703 pm_constant_id_t name = pm_parser_local_add_constant(parser, "it", 2);
21704 parser_lex(parser);
21705
21706 pm_node_t *value = parse_assignment_value(parser, previous_binding_power, binding_power, flags, PM_ERR_EXPECT_EXPRESSION_AFTER_AMPAMPEQ, (uint16_t) (depth + 1));
21707 pm_node_t *result = UP(pm_local_variable_and_write_node_create(parser, node, &token, value, name, 0));
21708
21709 pm_node_unreference(parser, node);
21710 return result;
21711 }
21712 case PM_LOCAL_VARIABLE_READ_NODE: {
21713 if (pm_token_is_numbered_parameter(parser, PM_NODE_START(node), PM_NODE_LENGTH(node))) {
21714 PM_PARSER_ERR_FORMAT(parser, node->location.start, node->location.length, PM_ERR_PARAMETER_NUMBERED_RESERVED, parser->start + node->location.start);
21715 pm_node_unreference(parser, node);
21716 }
21717
21719 parser_lex(parser);
21720
21721 pm_node_t *value = parse_assignment_value(parser, previous_binding_power, binding_power, flags, PM_ERR_EXPECT_EXPRESSION_AFTER_AMPAMPEQ, (uint16_t) (depth + 1));
21722 pm_node_t *result = UP(pm_local_variable_and_write_node_create(parser, node, &token, value, cast->name, cast->depth));
21723
21724 return result;
21725 }
21726 case PM_CALL_NODE: {
21727 pm_call_node_t *cast = (pm_call_node_t *) node;
21728
21729 // If we have a vcall (a method with no arguments and no
21730 // receiver that could have been a local variable) then we
21731 // will transform it into a local variable write.
21732 if (PM_NODE_FLAG_P(cast, PM_CALL_NODE_FLAGS_VARIABLE_CALL)) {
21733 pm_refute_numbered_parameter(parser, cast->message_loc.start, cast->message_loc.length);
21734 pm_constant_id_t constant_id = pm_parser_local_add_location(parser, &cast->message_loc, 1);
21735 parser_lex(parser);
21736
21737 pm_node_t *value = parse_assignment_value(parser, previous_binding_power, binding_power, flags, PM_ERR_EXPECT_EXPRESSION_AFTER_AMPAMPEQ, (uint16_t) (depth + 1));
21738 pm_node_t *result = UP(pm_local_variable_and_write_node_create(parser, UP(cast), &token, value, constant_id, 0));
21739
21740 return result;
21741 }
21742
21743 // Move past the token here so that we have already added
21744 // the local variable by this point.
21745 parser_lex(parser);
21746
21747 // If there is no call operator and the message is "[]" then
21748 // this is an aref expression, and we can transform it into
21749 // an aset expression.
21750 if (PM_NODE_FLAG_P(cast, PM_CALL_NODE_FLAGS_INDEX)) {
21751 pm_node_t *value = parse_assignment_value(parser, previous_binding_power, binding_power, flags, PM_ERR_EXPECT_EXPRESSION_AFTER_AMPAMPEQ, (uint16_t) (depth + 1));
21752 return UP(pm_index_and_write_node_create(parser, cast, &token, value));
21753 }
21754
21755 // If this node cannot be writable, then we have an error.
21756 if (pm_call_node_writable_p(parser, cast)) {
21757 parse_write_name(parser, &cast->name);
21758 } else {
21759 pm_parser_err_node(parser, node, PM_ERR_WRITE_TARGET_UNEXPECTED);
21760 }
21761
21762 parse_call_operator_write(parser, cast, &token);
21763 pm_node_t *value = parse_assignment_value(parser, previous_binding_power, binding_power, flags, PM_ERR_EXPECT_EXPRESSION_AFTER_AMPAMPEQ, (uint16_t) (depth + 1));
21764 return UP(pm_call_and_write_node_create(parser, cast, &token, value));
21765 }
21766 case PM_MULTI_WRITE_NODE: {
21767 parser_lex(parser);
21768 pm_parser_err_token(parser, &token, PM_ERR_AMPAMPEQ_MULTI_ASSIGN);
21769 return node;
21770 }
21771 default:
21772 parser_lex(parser);
21773
21774 // In this case we have an &&= sign, but we don't know what it's for.
21775 // We need to treat it as an error. For now, we'll mark it as an error
21776 // and just skip right past it.
21777 pm_parser_err_token(parser, &token, PM_ERR_EXPECT_EXPRESSION_AFTER_AMPAMPEQ);
21778 return node;
21779 }
21780 }
21781 case PM_TOKEN_PIPE_PIPE_EQUAL: {
21782 switch (PM_NODE_TYPE(node)) {
21783 case PM_BACK_REFERENCE_READ_NODE:
21784 case PM_NUMBERED_REFERENCE_READ_NODE:
21785 PM_PARSER_ERR_NODE_FORMAT_CONTENT(parser, node, PM_ERR_WRITE_TARGET_READONLY);
21787 case PM_GLOBAL_VARIABLE_READ_NODE: {
21788 parser_lex(parser);
21789
21790 pm_node_t *value = parse_assignment_value(parser, previous_binding_power, binding_power, flags, PM_ERR_EXPECT_EXPRESSION_AFTER_PIPEPIPEEQ, (uint16_t) (depth + 1));
21791 pm_node_t *result = UP(pm_global_variable_or_write_node_create(parser, node, &token, value));
21792
21793 return result;
21794 }
21795 case PM_CLASS_VARIABLE_READ_NODE: {
21796 parser_lex(parser);
21797
21798 pm_node_t *value = parse_assignment_value(parser, previous_binding_power, binding_power, flags, PM_ERR_EXPECT_EXPRESSION_AFTER_PIPEPIPEEQ, (uint16_t) (depth + 1));
21799 pm_node_t *result = UP(pm_class_variable_or_write_node_create(parser, (pm_class_variable_read_node_t *) node, &token, value));
21800
21801 return result;
21802 }
21803 case PM_CONSTANT_PATH_NODE: {
21804 parser_lex(parser);
21805
21806 pm_node_t *value = parse_assignment_value(parser, previous_binding_power, binding_power, flags, PM_ERR_EXPECT_EXPRESSION_AFTER_PIPEPIPEEQ, (uint16_t) (depth + 1));
21807 pm_node_t *write = UP(pm_constant_path_or_write_node_create(parser, (pm_constant_path_node_t *) node, &token, value));
21808
21809 return parse_shareable_constant_write(parser, write);
21810 }
21811 case PM_CONSTANT_READ_NODE: {
21812 parser_lex(parser);
21813
21814 pm_node_t *value = parse_assignment_value(parser, previous_binding_power, binding_power, flags, PM_ERR_EXPECT_EXPRESSION_AFTER_PIPEPIPEEQ, (uint16_t) (depth + 1));
21815 pm_node_t *write = UP(pm_constant_or_write_node_create(parser, (pm_constant_read_node_t *) node, &token, value));
21816
21817 if (context_def_p(parser)) {
21818 pm_parser_err_node(parser, write, PM_ERR_WRITE_TARGET_IN_METHOD);
21819 }
21820
21821 return parse_shareable_constant_write(parser, write);
21822 }
21823 case PM_INSTANCE_VARIABLE_READ_NODE: {
21824 parser_lex(parser);
21825
21826 pm_node_t *value = parse_assignment_value(parser, previous_binding_power, binding_power, flags, PM_ERR_EXPECT_EXPRESSION_AFTER_PIPEPIPEEQ, (uint16_t) (depth + 1));
21827 pm_node_t *result = UP(pm_instance_variable_or_write_node_create(parser, (pm_instance_variable_read_node_t *) node, &token, value));
21828
21829 return result;
21830 }
21831 case PM_IT_LOCAL_VARIABLE_READ_NODE: {
21832 pm_constant_id_t name = pm_parser_local_add_constant(parser, "it", 2);
21833 parser_lex(parser);
21834
21835 pm_node_t *value = parse_assignment_value(parser, previous_binding_power, binding_power, flags, PM_ERR_EXPECT_EXPRESSION_AFTER_PIPEPIPEEQ, (uint16_t) (depth + 1));
21836 pm_node_t *result = UP(pm_local_variable_or_write_node_create(parser, node, &token, value, name, 0));
21837
21838 pm_node_unreference(parser, node);
21839 return result;
21840 }
21841 case PM_LOCAL_VARIABLE_READ_NODE: {
21842 if (pm_token_is_numbered_parameter(parser, PM_NODE_START(node), PM_NODE_LENGTH(node))) {
21843 PM_PARSER_ERR_FORMAT(parser, PM_NODE_START(node), PM_NODE_LENGTH(node), PM_ERR_PARAMETER_NUMBERED_RESERVED, parser->start + PM_NODE_START(node));
21844 pm_node_unreference(parser, node);
21845 }
21846
21848 parser_lex(parser);
21849
21850 pm_node_t *value = parse_assignment_value(parser, previous_binding_power, binding_power, flags, PM_ERR_EXPECT_EXPRESSION_AFTER_PIPEPIPEEQ, (uint16_t) (depth + 1));
21851 pm_node_t *result = UP(pm_local_variable_or_write_node_create(parser, node, &token, value, cast->name, cast->depth));
21852
21853 return result;
21854 }
21855 case PM_CALL_NODE: {
21856 pm_call_node_t *cast = (pm_call_node_t *) node;
21857
21858 // If we have a vcall (a method with no arguments and no
21859 // receiver that could have been a local variable) then we
21860 // will transform it into a local variable write.
21861 if (PM_NODE_FLAG_P(cast, PM_CALL_NODE_FLAGS_VARIABLE_CALL)) {
21862 pm_refute_numbered_parameter(parser, cast->message_loc.start, cast->message_loc.length);
21863 pm_constant_id_t constant_id = pm_parser_local_add_location(parser, &cast->message_loc, 1);
21864 parser_lex(parser);
21865
21866 pm_node_t *value = parse_assignment_value(parser, previous_binding_power, binding_power, flags, PM_ERR_EXPECT_EXPRESSION_AFTER_PIPEPIPEEQ, (uint16_t) (depth + 1));
21867 pm_node_t *result = UP(pm_local_variable_or_write_node_create(parser, UP(cast), &token, value, constant_id, 0));
21868
21869 return result;
21870 }
21871
21872 // Move past the token here so that we have already added
21873 // the local variable by this point.
21874 parser_lex(parser);
21875
21876 // If there is no call operator and the message is "[]" then
21877 // this is an aref expression, and we can transform it into
21878 // an aset expression.
21879 if (PM_NODE_FLAG_P(cast, PM_CALL_NODE_FLAGS_INDEX)) {
21880 pm_node_t *value = parse_assignment_value(parser, previous_binding_power, binding_power, flags, PM_ERR_EXPECT_EXPRESSION_AFTER_PIPEPIPEEQ, (uint16_t) (depth + 1));
21881 return UP(pm_index_or_write_node_create(parser, cast, &token, value));
21882 }
21883
21884 // If this node cannot be writable, then we have an error.
21885 if (pm_call_node_writable_p(parser, cast)) {
21886 parse_write_name(parser, &cast->name);
21887 } else {
21888 pm_parser_err_node(parser, node, PM_ERR_WRITE_TARGET_UNEXPECTED);
21889 }
21890
21891 parse_call_operator_write(parser, cast, &token);
21892 pm_node_t *value = parse_assignment_value(parser, previous_binding_power, binding_power, flags, PM_ERR_EXPECT_EXPRESSION_AFTER_PIPEPIPEEQ, (uint16_t) (depth + 1));
21893 return UP(pm_call_or_write_node_create(parser, cast, &token, value));
21894 }
21895 case PM_MULTI_WRITE_NODE: {
21896 parser_lex(parser);
21897 pm_parser_err_token(parser, &token, PM_ERR_PIPEPIPEEQ_MULTI_ASSIGN);
21898 return node;
21899 }
21900 default:
21901 parser_lex(parser);
21902
21903 // In this case we have an ||= sign, but we don't know what it's for.
21904 // We need to treat it as an error. For now, we'll mark it as an error
21905 // and just skip right past it.
21906 pm_parser_err_token(parser, &token, PM_ERR_EXPECT_EXPRESSION_AFTER_PIPEPIPEEQ);
21907 return node;
21908 }
21909 }
21910 case PM_TOKEN_AMPERSAND_EQUAL:
21911 case PM_TOKEN_CARET_EQUAL:
21912 case PM_TOKEN_GREATER_GREATER_EQUAL:
21913 case PM_TOKEN_LESS_LESS_EQUAL:
21914 case PM_TOKEN_MINUS_EQUAL:
21915 case PM_TOKEN_PERCENT_EQUAL:
21916 case PM_TOKEN_PIPE_EQUAL:
21917 case PM_TOKEN_PLUS_EQUAL:
21918 case PM_TOKEN_SLASH_EQUAL:
21919 case PM_TOKEN_STAR_EQUAL:
21920 case PM_TOKEN_STAR_STAR_EQUAL: {
21921 switch (PM_NODE_TYPE(node)) {
21922 case PM_BACK_REFERENCE_READ_NODE:
21923 case PM_NUMBERED_REFERENCE_READ_NODE:
21924 PM_PARSER_ERR_NODE_FORMAT_CONTENT(parser, node, PM_ERR_WRITE_TARGET_READONLY);
21926 case PM_GLOBAL_VARIABLE_READ_NODE: {
21927 parser_lex(parser);
21928
21929 pm_node_t *value = parse_assignment_value(parser, previous_binding_power, binding_power, flags, PM_ERR_EXPECT_EXPRESSION_AFTER_OPERATOR, (uint16_t) (depth + 1));
21930 pm_node_t *result = UP(pm_global_variable_operator_write_node_create(parser, node, &token, value));
21931
21932 return result;
21933 }
21934 case PM_CLASS_VARIABLE_READ_NODE: {
21935 parser_lex(parser);
21936
21937 pm_node_t *value = parse_assignment_value(parser, previous_binding_power, binding_power, flags, PM_ERR_EXPECT_EXPRESSION_AFTER_OPERATOR, (uint16_t) (depth + 1));
21938 pm_node_t *result = UP(pm_class_variable_operator_write_node_create(parser, (pm_class_variable_read_node_t *) node, &token, value));
21939
21940 return result;
21941 }
21942 case PM_CONSTANT_PATH_NODE: {
21943 parser_lex(parser);
21944
21945 pm_node_t *value = parse_assignment_value(parser, previous_binding_power, binding_power, flags, PM_ERR_EXPECT_EXPRESSION_AFTER_OPERATOR, (uint16_t) (depth + 1));
21946 pm_node_t *write = UP(pm_constant_path_operator_write_node_create(parser, (pm_constant_path_node_t *) node, &token, value));
21947
21948 return parse_shareable_constant_write(parser, write);
21949 }
21950 case PM_CONSTANT_READ_NODE: {
21951 parser_lex(parser);
21952
21953 pm_node_t *value = parse_assignment_value(parser, previous_binding_power, binding_power, flags, PM_ERR_EXPECT_EXPRESSION_AFTER_OPERATOR, (uint16_t) (depth + 1));
21954 pm_node_t *write = UP(pm_constant_operator_write_node_create(parser, (pm_constant_read_node_t *) node, &token, value));
21955
21956 if (context_def_p(parser)) {
21957 pm_parser_err_node(parser, write, PM_ERR_WRITE_TARGET_IN_METHOD);
21958 }
21959
21960 return parse_shareable_constant_write(parser, write);
21961 }
21962 case PM_INSTANCE_VARIABLE_READ_NODE: {
21963 parser_lex(parser);
21964
21965 pm_node_t *value = parse_assignment_value(parser, previous_binding_power, binding_power, flags, PM_ERR_EXPECT_EXPRESSION_AFTER_OPERATOR, (uint16_t) (depth + 1));
21966 pm_node_t *result = UP(pm_instance_variable_operator_write_node_create(parser, (pm_instance_variable_read_node_t *) node, &token, value));
21967
21968 return result;
21969 }
21970 case PM_IT_LOCAL_VARIABLE_READ_NODE: {
21971 pm_constant_id_t name = pm_parser_local_add_constant(parser, "it", 2);
21972 parser_lex(parser);
21973
21974 pm_node_t *value = parse_assignment_value(parser, previous_binding_power, binding_power, flags, PM_ERR_EXPECT_EXPRESSION_AFTER_OPERATOR, (uint16_t) (depth + 1));
21975 pm_node_t *result = UP(pm_local_variable_operator_write_node_create(parser, node, &token, value, name, 0));
21976
21977 pm_node_unreference(parser, node);
21978 return result;
21979 }
21980 case PM_LOCAL_VARIABLE_READ_NODE: {
21981 if (pm_token_is_numbered_parameter(parser, PM_NODE_START(node), PM_NODE_LENGTH(node))) {
21982 PM_PARSER_ERR_FORMAT(parser, PM_NODE_START(node), PM_NODE_LENGTH(node), PM_ERR_PARAMETER_NUMBERED_RESERVED, parser->start + PM_NODE_START(node));
21983 pm_node_unreference(parser, node);
21984 }
21985
21987 parser_lex(parser);
21988
21989 pm_node_t *value = parse_assignment_value(parser, previous_binding_power, binding_power, flags, PM_ERR_EXPECT_EXPRESSION_AFTER_OPERATOR, (uint16_t) (depth + 1));
21990 pm_node_t *result = UP(pm_local_variable_operator_write_node_create(parser, node, &token, value, cast->name, cast->depth));
21991
21992 return result;
21993 }
21994 case PM_CALL_NODE: {
21995 parser_lex(parser);
21996 pm_call_node_t *cast = (pm_call_node_t *) node;
21997
21998 // If we have a vcall (a method with no arguments and no
21999 // receiver that could have been a local variable) then we
22000 // will transform it into a local variable write.
22001 if (PM_NODE_FLAG_P(cast, PM_CALL_NODE_FLAGS_VARIABLE_CALL)) {
22002 pm_refute_numbered_parameter(parser, cast->message_loc.start, cast->message_loc.length);
22003 pm_constant_id_t constant_id = pm_parser_local_add_location(parser, &cast->message_loc, 1);
22004 pm_node_t *value = parse_assignment_value(parser, previous_binding_power, binding_power, flags, PM_ERR_EXPECT_EXPRESSION_AFTER_OPERATOR, (uint16_t) (depth + 1));
22005 pm_node_t *result = UP(pm_local_variable_operator_write_node_create(parser, UP(cast), &token, value, constant_id, 0));
22006
22007 return result;
22008 }
22009
22010 // If there is no call operator and the message is "[]" then
22011 // this is an aref expression, and we can transform it into
22012 // an aset expression.
22013 if (PM_NODE_FLAG_P(cast, PM_CALL_NODE_FLAGS_INDEX)) {
22014 pm_node_t *value = parse_assignment_value(parser, previous_binding_power, binding_power, flags, PM_ERR_EXPECT_EXPRESSION_AFTER_OPERATOR, (uint16_t) (depth + 1));
22015 return UP(pm_index_operator_write_node_create(parser, cast, &token, value));
22016 }
22017
22018 // If this node cannot be writable, then we have an error.
22019 if (pm_call_node_writable_p(parser, cast)) {
22020 parse_write_name(parser, &cast->name);
22021 } else {
22022 pm_parser_err_node(parser, node, PM_ERR_WRITE_TARGET_UNEXPECTED);
22023 }
22024
22025 parse_call_operator_write(parser, cast, &token);
22026 pm_node_t *value = parse_assignment_value(parser, previous_binding_power, binding_power, flags, PM_ERR_EXPECT_EXPRESSION_AFTER_OPERATOR, (uint16_t) (depth + 1));
22027 return UP(pm_call_operator_write_node_create(parser, cast, &token, value));
22028 }
22029 case PM_MULTI_WRITE_NODE: {
22030 parser_lex(parser);
22031 pm_parser_err_token(parser, &token, PM_ERR_OPERATOR_MULTI_ASSIGN);
22032 return node;
22033 }
22034 default:
22035 parser_lex(parser);
22036
22037 // In this case we have an operator but we don't know what it's for.
22038 // We need to treat it as an error. For now, we'll mark it as an error
22039 // and just skip right past it.
22040 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->previous, PM_ERR_EXPECT_EXPRESSION_AFTER_OPERATOR, pm_token_str(parser->current.type));
22041 return node;
22042 }
22043 }
22044 case PM_TOKEN_AMPERSAND_AMPERSAND:
22045 case PM_TOKEN_KEYWORD_AND: {
22046 parser_lex(parser);
22047
22048 pm_node_t *right = parse_expression(parser, binding_power, (uint8_t) ((flags & PM_PARSE_ACCEPTS_DO_BLOCK) | (parser->previous.type == PM_TOKEN_KEYWORD_AND ? PM_PARSE_ACCEPTS_COMMAND_CALL : 0)), PM_ERR_EXPECT_EXPRESSION_AFTER_OPERATOR, (uint16_t) (depth + 1));
22049 return UP(pm_and_node_create(parser, node, &token, right));
22050 }
22051 case PM_TOKEN_KEYWORD_OR:
22052 case PM_TOKEN_PIPE_PIPE: {
22053 parser_lex(parser);
22054
22055 pm_node_t *right = parse_expression(parser, binding_power, (uint8_t) ((flags & PM_PARSE_ACCEPTS_DO_BLOCK) | (parser->previous.type == PM_TOKEN_KEYWORD_OR ? PM_PARSE_ACCEPTS_COMMAND_CALL : 0)), PM_ERR_EXPECT_EXPRESSION_AFTER_OPERATOR, (uint16_t) (depth + 1));
22056 return UP(pm_or_node_create(parser, node, &token, right));
22057 }
22058 case PM_TOKEN_EQUAL_TILDE: {
22059 // Note that we _must_ parse the value before adding the local
22060 // variables in order to properly mirror the behavior of Ruby. For
22061 // example,
22062 //
22063 // /(?<foo>bar)/ =~ foo
22064 //
22065 // In this case, `foo` should be a method call and not a local yet.
22066 parser_lex(parser);
22067 pm_node_t *argument = parse_expression(parser, binding_power, flags & PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_EXPECT_EXPRESSION_AFTER_OPERATOR, (uint16_t) (depth + 1));
22068
22069 // By default, we're going to create a call node and then return it.
22070 pm_call_node_t *call = pm_call_node_binary_create(parser, node, &token, argument, 0);
22071 pm_node_t *result = UP(call);
22072
22073 // If the receiver of this =~ is a regular expression node, then we
22074 // need to introduce local variables for it based on its named
22075 // capture groups.
22076 if (PM_NODE_TYPE_P(node, PM_INTERPOLATED_REGULAR_EXPRESSION_NODE)) {
22077 // It's possible to have an interpolated regular expression node
22078 // that only contains strings. This is because it can be split
22079 // up by a heredoc. In this case we need to concat the unescaped
22080 // strings together and then parse them as a regular expression.
22082
22083 bool interpolated = false;
22084 size_t total_length = 0;
22085
22086 pm_node_t *part;
22087 PM_NODE_LIST_FOREACH(parts, index, part) {
22088 if (PM_NODE_TYPE_P(part, PM_STRING_NODE)) {
22089 total_length += pm_string_length(&((pm_string_node_t *) part)->unescaped);
22090 } else {
22091 interpolated = true;
22092 break;
22093 }
22094 }
22095
22096 if (!interpolated && total_length > 0) {
22097 void *memory = xmalloc(total_length);
22098 if (!memory) abort();
22099
22100 uint8_t *cursor = memory;
22101 PM_NODE_LIST_FOREACH(parts, index, part) {
22102 pm_string_t *unescaped = &((pm_string_node_t *) part)->unescaped;
22103 size_t length = pm_string_length(unescaped);
22104
22105 memcpy(cursor, pm_string_source(unescaped), length);
22106 cursor += length;
22107 }
22108
22109 pm_string_t owned;
22110 pm_string_owned_init(&owned, (uint8_t *) memory, total_length);
22111
22112 result = parse_interpolated_regular_expression_named_captures(parser, &owned, call, PM_NODE_FLAG_P(node, PM_REGULAR_EXPRESSION_FLAGS_EXTENDED));
22113 pm_string_cleanup(&owned);
22114 }
22115 } else if (PM_NODE_TYPE_P(node, PM_REGULAR_EXPRESSION_NODE)) {
22116 // If we have a regular expression node, then we can parse
22117 // the named captures and validate encoding in one pass.
22119
22120 pm_regexp_name_data_t name_data = {
22121 .call = call,
22122 .match = NULL,
22123 .names = { 0 },
22124 };
22125
22126 pm_node_flag_set(UP(regexp), pm_regexp_parse(parser, regexp, parse_regular_expression_named_capture, &name_data));
22127
22128 if (name_data.match != NULL) {
22129 result = UP(name_data.match);
22130 }
22131 }
22132
22133 return result;
22134 }
22135 case PM_TOKEN_UAMPERSAND:
22136 case PM_TOKEN_USTAR:
22137 case PM_TOKEN_USTAR_STAR:
22138 // The only times this will occur are when we are in an error state,
22139 // but we'll put them in here so that errors can propagate.
22140 case PM_TOKEN_BANG_EQUAL:
22141 case PM_TOKEN_BANG_TILDE:
22142 case PM_TOKEN_EQUAL_EQUAL:
22143 case PM_TOKEN_EQUAL_EQUAL_EQUAL:
22144 case PM_TOKEN_LESS_EQUAL_GREATER:
22145 case PM_TOKEN_CARET:
22146 case PM_TOKEN_PIPE:
22147 case PM_TOKEN_AMPERSAND:
22148 case PM_TOKEN_GREATER_GREATER:
22149 case PM_TOKEN_LESS_LESS:
22150 case PM_TOKEN_MINUS:
22151 case PM_TOKEN_PLUS:
22152 case PM_TOKEN_PERCENT:
22153 case PM_TOKEN_SLASH:
22154 case PM_TOKEN_STAR:
22155 case PM_TOKEN_STAR_STAR: {
22156 parser_lex(parser);
22157 pm_token_t operator = parser->previous;
22158 switch (PM_NODE_TYPE(node)) {
22159 case PM_RESCUE_MODIFIER_NODE: {
22161 if (PM_NODE_TYPE_P(cast->rescue_expression, PM_MATCH_PREDICATE_NODE) || PM_NODE_TYPE_P(cast->rescue_expression, PM_MATCH_REQUIRED_NODE)) {
22162 PM_PARSER_ERR_TOKEN_FORMAT(parser, &operator, PM_ERR_EXPECT_EOL_AFTER_STATEMENT, pm_token_str(operator.type));
22163 }
22164 break;
22165 }
22166 case PM_AND_NODE: {
22167 pm_and_node_t *cast = (pm_and_node_t *) node;
22168 if (PM_NODE_TYPE_P(cast->right, PM_MATCH_PREDICATE_NODE) || PM_NODE_TYPE_P(cast->right, PM_MATCH_REQUIRED_NODE)) {
22169 PM_PARSER_ERR_TOKEN_FORMAT(parser, &operator, PM_ERR_EXPECT_EOL_AFTER_STATEMENT, pm_token_str(operator.type));
22170 }
22171 break;
22172 }
22173 case PM_OR_NODE: {
22174 pm_or_node_t *cast = (pm_or_node_t *) node;
22175 if (PM_NODE_TYPE_P(cast->right, PM_MATCH_PREDICATE_NODE) || PM_NODE_TYPE_P(cast->right, PM_MATCH_REQUIRED_NODE)) {
22176 PM_PARSER_ERR_TOKEN_FORMAT(parser, &operator, PM_ERR_EXPECT_EOL_AFTER_STATEMENT, pm_token_str(operator.type));
22177 }
22178 break;
22179 }
22180 default:
22181 break;
22182 }
22183
22184 pm_node_t *argument = parse_expression(parser, binding_power, flags & PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_EXPECT_EXPRESSION_AFTER_OPERATOR, (uint16_t) (depth + 1));
22185 return UP(pm_call_node_binary_create(parser, node, &token, argument, 0));
22186 }
22187 case PM_TOKEN_GREATER:
22188 case PM_TOKEN_GREATER_EQUAL:
22189 case PM_TOKEN_LESS:
22190 case PM_TOKEN_LESS_EQUAL: {
22191 if (PM_NODE_TYPE_P(node, PM_CALL_NODE) && PM_NODE_FLAG_P(node, PM_CALL_NODE_FLAGS_COMPARISON)) {
22192 PM_PARSER_WARN_TOKEN_FORMAT_CONTENT(parser, &parser->current, PM_WARN_COMPARISON_AFTER_COMPARISON);
22193 }
22194
22195 parser_lex(parser);
22196 pm_node_t *argument = parse_expression(parser, binding_power, flags & PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_EXPECT_EXPRESSION_AFTER_OPERATOR, (uint16_t) (depth + 1));
22197 return UP(pm_call_node_binary_create(parser, node, &token, argument, PM_CALL_NODE_FLAGS_COMPARISON));
22198 }
22199 case PM_TOKEN_AMPERSAND_DOT:
22200 case PM_TOKEN_DOT: {
22201 parser_lex(parser);
22202 pm_token_t operator = parser->previous;
22203 pm_arguments_t arguments = { 0 };
22204
22205 // This if statement handles the foo.() syntax.
22206 if (match1(parser, PM_TOKEN_PARENTHESIS_LEFT)) {
22207 parse_arguments_list(parser, &arguments, true, false, (uint16_t) (depth + 1));
22208 return UP(pm_call_node_shorthand_create(parser, node, &operator, &arguments));
22209 }
22210
22211 switch (PM_NODE_TYPE(node)) {
22212 case PM_RESCUE_MODIFIER_NODE: {
22214 if (PM_NODE_TYPE_P(cast->rescue_expression, PM_MATCH_PREDICATE_NODE) || PM_NODE_TYPE_P(cast->rescue_expression, PM_MATCH_REQUIRED_NODE)) {
22215 PM_PARSER_ERR_TOKEN_FORMAT(parser, &operator, PM_ERR_EXPECT_EOL_AFTER_STATEMENT, pm_token_str(operator.type));
22216 }
22217 break;
22218 }
22219 case PM_AND_NODE: {
22220 pm_and_node_t *cast = (pm_and_node_t *) node;
22221 if (PM_NODE_TYPE_P(cast->right, PM_MATCH_PREDICATE_NODE) || PM_NODE_TYPE_P(cast->right, PM_MATCH_REQUIRED_NODE)) {
22222 PM_PARSER_ERR_TOKEN_FORMAT(parser, &operator, PM_ERR_EXPECT_EOL_AFTER_STATEMENT, pm_token_str(operator.type));
22223 }
22224 break;
22225 }
22226 case PM_OR_NODE: {
22227 pm_or_node_t *cast = (pm_or_node_t *) node;
22228 if (PM_NODE_TYPE_P(cast->right, PM_MATCH_PREDICATE_NODE) || PM_NODE_TYPE_P(cast->right, PM_MATCH_REQUIRED_NODE)) {
22229 PM_PARSER_ERR_TOKEN_FORMAT(parser, &operator, PM_ERR_EXPECT_EOL_AFTER_STATEMENT, pm_token_str(operator.type));
22230 }
22231 break;
22232 }
22233 default:
22234 break;
22235 }
22236
22237 pm_token_t message;
22238
22239 switch (parser->current.type) {
22240 case PM_CASE_OPERATOR:
22241 case PM_CASE_KEYWORD:
22242 case PM_TOKEN_CONSTANT:
22243 case PM_TOKEN_IDENTIFIER:
22244 case PM_TOKEN_METHOD_NAME: {
22245 parser_lex(parser);
22246 message = parser->previous;
22247 break;
22248 }
22249 default: {
22250 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->current, PM_ERR_EXPECT_MESSAGE, pm_token_str(parser->current.type));
22251 message = (pm_token_t) { .type = 0, .start = parser->previous.end, .end = parser->previous.end };
22252 }
22253 }
22254
22255 parse_arguments_list(parser, &arguments, true, flags, (uint16_t) (depth + 1));
22256 pm_call_node_t *call = pm_call_node_call_create(parser, node, &operator, &message, &arguments);
22257
22258 if (
22259 (previous_binding_power == PM_BINDING_POWER_STATEMENT) &&
22260 arguments.arguments == NULL &&
22261 arguments.opening_loc.length == 0 &&
22262 match1(parser, PM_TOKEN_COMMA)
22263 ) {
22264 return parse_targets_validate(parser, UP(call), PM_BINDING_POWER_INDEX, (uint16_t) (depth + 1));
22265 } else {
22266 return UP(call);
22267 }
22268 }
22269 case PM_TOKEN_DOT_DOT:
22270 case PM_TOKEN_DOT_DOT_DOT: {
22271 parser_lex(parser);
22272
22273 pm_node_t *right = NULL;
22274 if (token_begins_expression_p(parser->current.type)) {
22275 right = parse_expression(parser, binding_power, flags & PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_EXPECT_EXPRESSION_AFTER_OPERATOR, (uint16_t) (depth + 1));
22276 }
22277
22278 return UP(pm_range_node_create(parser, node, &token, right));
22279 }
22280 case PM_TOKEN_KEYWORD_IF_MODIFIER: {
22281 pm_token_t keyword = parser->current;
22282 parser_lex(parser);
22283
22284 pm_node_t *predicate = parse_value_expression(parser, binding_power, (uint8_t) ((flags & PM_PARSE_ACCEPTS_DO_BLOCK) | PM_PARSE_ACCEPTS_COMMAND_CALL), PM_ERR_CONDITIONAL_IF_PREDICATE, (uint16_t) (depth + 1));
22285 return UP(pm_if_node_modifier_create(parser, node, &keyword, predicate));
22286 }
22287 case PM_TOKEN_KEYWORD_UNLESS_MODIFIER: {
22288 pm_token_t keyword = parser->current;
22289 parser_lex(parser);
22290
22291 pm_node_t *predicate = parse_value_expression(parser, binding_power, (uint8_t) ((flags & PM_PARSE_ACCEPTS_DO_BLOCK) | PM_PARSE_ACCEPTS_COMMAND_CALL), PM_ERR_CONDITIONAL_UNLESS_PREDICATE, (uint16_t) (depth + 1));
22292 return UP(pm_unless_node_modifier_create(parser, node, &keyword, predicate));
22293 }
22294 case PM_TOKEN_KEYWORD_UNTIL_MODIFIER: {
22295 parser_lex(parser);
22296 pm_statements_node_t *statements = pm_statements_node_create(parser);
22297 pm_statements_node_body_append(parser, statements, node, true);
22298
22299 pm_node_t *predicate = parse_value_expression(parser, binding_power, (uint8_t) ((flags & PM_PARSE_ACCEPTS_DO_BLOCK) | PM_PARSE_ACCEPTS_COMMAND_CALL), PM_ERR_CONDITIONAL_UNTIL_PREDICATE, (uint16_t) (depth + 1));
22300 return UP(pm_until_node_modifier_create(parser, &token, predicate, statements, PM_NODE_TYPE_P(node, PM_BEGIN_NODE) ? PM_LOOP_FLAGS_BEGIN_MODIFIER : 0));
22301 }
22302 case PM_TOKEN_KEYWORD_WHILE_MODIFIER: {
22303 parser_lex(parser);
22304 pm_statements_node_t *statements = pm_statements_node_create(parser);
22305 pm_statements_node_body_append(parser, statements, node, true);
22306
22307 pm_node_t *predicate = parse_value_expression(parser, binding_power, (uint8_t) ((flags & PM_PARSE_ACCEPTS_DO_BLOCK) | PM_PARSE_ACCEPTS_COMMAND_CALL), PM_ERR_CONDITIONAL_WHILE_PREDICATE, (uint16_t) (depth + 1));
22308 return UP(pm_while_node_modifier_create(parser, &token, predicate, statements, PM_NODE_TYPE_P(node, PM_BEGIN_NODE) ? PM_LOOP_FLAGS_BEGIN_MODIFIER : 0));
22309 }
22310 case PM_TOKEN_QUESTION_MARK: {
22311 context_push(parser, PM_CONTEXT_TERNARY);
22312 pm_node_list_t current_block_exits = { 0 };
22313 pm_node_list_t *previous_block_exits = push_block_exits(parser, &current_block_exits);
22314
22315 pm_token_t qmark = parser->current;
22316 parser_lex(parser);
22317
22318 pm_node_t *true_expression = parse_expression(parser, PM_BINDING_POWER_DEFINED, flags & PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_TERNARY_EXPRESSION_TRUE, (uint16_t) (depth + 1));
22319
22320 if (parser->recovering) {
22321 // If parsing the true expression of this ternary resulted in a syntax
22322 // error that we can recover from, then we're going to put missing nodes
22323 // and tokens into the remaining places. We want to be sure to do this
22324 // before the `expect` function call to make sure it doesn't
22325 // accidentally move past a ':' token that occurs after the syntax
22326 // error.
22327 pm_token_t colon = (pm_token_t) { .type = 0, .start = parser->previous.end, .end = parser->previous.end };
22328 pm_node_t *false_expression = UP(pm_error_recovery_node_create(parser, PM_TOKEN_START(parser, &colon), PM_TOKEN_LENGTH(&colon)));
22329
22330 context_pop(parser);
22331 pop_block_exits(parser, previous_block_exits);
22332 return UP(pm_if_node_ternary_create(parser, node, &qmark, true_expression, &colon, false_expression));
22333 }
22334
22335 accept1(parser, PM_TOKEN_NEWLINE);
22336 expect1(parser, PM_TOKEN_COLON, PM_ERR_TERNARY_COLON);
22337
22338 pm_token_t colon = parser->previous;
22339 pm_node_t *false_expression = parse_expression(parser, PM_BINDING_POWER_DEFINED, flags & PM_PARSE_ACCEPTS_DO_BLOCK, PM_ERR_TERNARY_EXPRESSION_FALSE, (uint16_t) (depth + 1));
22340
22341 context_pop(parser);
22342 pop_block_exits(parser, previous_block_exits);
22343 return UP(pm_if_node_ternary_create(parser, node, &qmark, true_expression, &colon, false_expression));
22344 }
22345 case PM_TOKEN_COLON_COLON: {
22346 parser_lex(parser);
22347 pm_token_t delimiter = parser->previous;
22348
22349 switch (parser->current.type) {
22350 case PM_TOKEN_CONSTANT: {
22351 parser_lex(parser);
22352 pm_node_t *path;
22353
22354 if (
22355 (parser->current.type == PM_TOKEN_PARENTHESIS_LEFT) ||
22356 ((flags & PM_PARSE_ACCEPTS_COMMAND_CALL) && (token_begins_expression_p(parser->current.type) || match3(parser, PM_TOKEN_UAMPERSAND, PM_TOKEN_USTAR, PM_TOKEN_USTAR_STAR)))
22357 ) {
22358 // If we have a constant immediately following a '::' operator, then
22359 // this can either be a constant path or a method call, depending on
22360 // what follows the constant.
22361 //
22362 // If we have parentheses, then this is a method call. That would
22363 // look like Foo::Bar().
22364 pm_token_t message = parser->previous;
22365 pm_arguments_t arguments = { 0 };
22366
22367 parse_arguments_list(parser, &arguments, true, flags, (uint16_t) (depth + 1));
22368 path = UP(pm_call_node_call_create(parser, node, &delimiter, &message, &arguments));
22369 } else {
22370 // Otherwise, this is a constant path. That would look like Foo::Bar.
22371 path = UP(pm_constant_path_node_create(parser, node, &delimiter, &parser->previous));
22372 }
22373
22374 // If this is followed by a comma then it is a multiple assignment.
22375 if (previous_binding_power == PM_BINDING_POWER_STATEMENT && match1(parser, PM_TOKEN_COMMA)) {
22376 return parse_targets_validate(parser, path, PM_BINDING_POWER_INDEX, (uint16_t) (depth + 1));
22377 }
22378
22379 return path;
22380 }
22381 case PM_CASE_OPERATOR:
22382 case PM_CASE_KEYWORD:
22383 case PM_TOKEN_IDENTIFIER:
22384 case PM_TOKEN_METHOD_NAME: {
22385 parser_lex(parser);
22386 pm_token_t message = parser->previous;
22387
22388 // If we have an identifier following a '::' operator, then it is for
22389 // sure a method call.
22390 pm_arguments_t arguments = { 0 };
22391 parse_arguments_list(parser, &arguments, true, flags, (uint16_t) (depth + 1));
22392 pm_call_node_t *call = pm_call_node_call_create(parser, node, &delimiter, &message, &arguments);
22393
22394 // If this is followed by a comma then it is a multiple assignment.
22395 if (previous_binding_power == PM_BINDING_POWER_STATEMENT && match1(parser, PM_TOKEN_COMMA)) {
22396 return parse_targets_validate(parser, UP(call), PM_BINDING_POWER_INDEX, (uint16_t) (depth + 1));
22397 }
22398
22399 return UP(call);
22400 }
22401 case PM_TOKEN_PARENTHESIS_LEFT: {
22402 // If we have a parenthesis following a '::' operator, then it is the
22403 // method call shorthand. That would look like Foo::(bar).
22404 pm_arguments_t arguments = { 0 };
22405 parse_arguments_list(parser, &arguments, true, false, (uint16_t) (depth + 1));
22406
22407 return UP(pm_call_node_shorthand_create(parser, node, &delimiter, &arguments));
22408 }
22409 default: {
22410 expect1(parser, PM_TOKEN_CONSTANT, PM_ERR_CONSTANT_PATH_COLON_COLON_CONSTANT);
22411 return UP(pm_constant_path_node_create(parser, node, &delimiter, &parser->previous));
22412 }
22413 }
22414 }
22415 case PM_TOKEN_KEYWORD_RESCUE_MODIFIER: {
22416 context_push(parser, PM_CONTEXT_RESCUE_MODIFIER);
22417 parser_lex(parser);
22418 accept1(parser, PM_TOKEN_NEWLINE);
22419
22420 pm_node_t *value = parse_rescue_modifier_value(parser, (uint8_t) ((flags & PM_PARSE_ACCEPTS_DO_BLOCK) | PM_PARSE_ACCEPTS_COMMAND_CALL), previous_binding_power == PM_BINDING_POWER_STATEMENT, (uint16_t) (depth + 1));
22421 context_pop(parser);
22422
22423 return UP(pm_rescue_modifier_node_create(parser, node, &token, value));
22424 }
22425 case PM_TOKEN_BRACKET_LEFT: {
22426 parser_lex(parser);
22427
22428 pm_arguments_t arguments = { 0 };
22429 arguments.opening_loc = TOK2LOC(parser, &parser->previous);
22430
22431 if (!accept1(parser, PM_TOKEN_BRACKET_RIGHT)) {
22432 parse_arguments(parser, &arguments, false, PM_TOKEN_BRACKET_RIGHT, (uint8_t) (flags & ~PM_PARSE_ACCEPTS_DO_BLOCK), (uint16_t) (depth + 1));
22433 expect1(parser, PM_TOKEN_BRACKET_RIGHT, PM_ERR_EXPECT_RBRACKET);
22434 }
22435
22436 arguments.closing_loc = TOK2LOC(parser, &parser->previous);
22437
22438 // If we have a comma after the closing bracket then this is a multiple
22439 // assignment and we should parse the targets.
22440 if (previous_binding_power == PM_BINDING_POWER_STATEMENT && match1(parser, PM_TOKEN_COMMA)) {
22441 pm_call_node_t *aref = pm_call_node_aref_create(parser, node, &arguments);
22442 return parse_targets_validate(parser, UP(aref), PM_BINDING_POWER_INDEX, (uint16_t) (depth + 1));
22443 }
22444
22445 // If we're at the end of the arguments, we can now check if there is a
22446 // block node that starts with a {. If there is, then we can parse it and
22447 // add it to the arguments.
22448 pm_block_node_t *block = NULL;
22449 if (accept1(parser, PM_TOKEN_BRACE_LEFT)) {
22450 block = parse_block(parser, (uint16_t) (depth + 1));
22451 pm_arguments_validate_block(parser, &arguments, block);
22452 } else if (pm_accepts_block_stack_p(parser) && accept1(parser, PM_TOKEN_KEYWORD_DO)) {
22453 block = parse_block(parser, (uint16_t) (depth + 1));
22454 }
22455
22456 if (block != NULL) {
22457 if (arguments.block != NULL) {
22458 pm_parser_err_node(parser, UP(block), PM_ERR_ARGUMENT_AFTER_BLOCK);
22459 if (arguments.arguments == NULL) {
22460 arguments.arguments = pm_arguments_node_create(parser);
22461 }
22462 pm_arguments_node_arguments_append(parser->arena, arguments.arguments, arguments.block);
22463 }
22464
22465 arguments.block = UP(block);
22466 }
22467
22468 return UP(pm_call_node_aref_create(parser, node, &arguments));
22469 }
22470 case PM_TOKEN_KEYWORD_IN: {
22471 bool previous_pattern_matching_newlines = parser->pattern_matching_newlines;
22472 parser->pattern_matching_newlines = true;
22473
22474 pm_token_t operator = parser->current;
22475 parser->command_start = false;
22476 lex_state_set(parser, PM_LEX_STATE_BEG | PM_LEX_STATE_LABEL);
22477 parser_lex(parser);
22478
22479 pm_constant_id_set_t captures = { 0 };
22480 pm_node_t *pattern = parse_pattern(parser, &captures, PM_PARSE_PATTERN_TOP | PM_PARSE_PATTERN_MULTI, PM_ERR_PATTERN_EXPRESSION_AFTER_IN, (uint16_t) (depth + 1));
22481
22482 parser->pattern_matching_newlines = previous_pattern_matching_newlines;
22483
22484 return UP(pm_match_predicate_node_create(parser, node, pattern, &operator));
22485 }
22486 case PM_TOKEN_EQUAL_GREATER: {
22487 bool previous_pattern_matching_newlines = parser->pattern_matching_newlines;
22488 parser->pattern_matching_newlines = true;
22489
22490 pm_token_t operator = parser->current;
22491 parser->command_start = false;
22492 lex_state_set(parser, PM_LEX_STATE_BEG | PM_LEX_STATE_LABEL);
22493 parser_lex(parser);
22494
22495 pm_constant_id_set_t captures = { 0 };
22496 pm_node_t *pattern = parse_pattern(parser, &captures, PM_PARSE_PATTERN_TOP | PM_PARSE_PATTERN_MULTI, PM_ERR_PATTERN_EXPRESSION_AFTER_HROCKET, (uint16_t) (depth + 1));
22497
22498 parser->pattern_matching_newlines = previous_pattern_matching_newlines;
22499
22500 return UP(pm_match_required_node_create(parser, node, pattern, &operator));
22501 }
22502 default:
22503 assert(false && "unreachable");
22504 return NULL;
22505 }
22506}
22507
22508#undef PM_PARSE_PATTERN_SINGLE
22509#undef PM_PARSE_PATTERN_TOP
22510#undef PM_PARSE_PATTERN_MULTI
22511
22524static bool
22525parse_expression_terminator(pm_parser_t *parser, pm_node_t *node) {
22526 pm_binding_power_t left = pm_binding_powers[parser->current.type].left;
22527
22528 switch (PM_NODE_TYPE(node)) {
22529 case PM_MULTI_WRITE_NODE:
22530 case PM_RETURN_NODE:
22531 case PM_BREAK_NODE:
22532 case PM_NEXT_NODE:
22533 return left > PM_BINDING_POWER_MODIFIER;
22534 case PM_CLASS_VARIABLE_WRITE_NODE:
22535 case PM_CONSTANT_PATH_WRITE_NODE:
22536 case PM_CONSTANT_WRITE_NODE:
22537 case PM_GLOBAL_VARIABLE_WRITE_NODE:
22538 case PM_INSTANCE_VARIABLE_WRITE_NODE:
22539 case PM_LOCAL_VARIABLE_WRITE_NODE:
22540 return PM_NODE_FLAG_P(node, PM_WRITE_NODE_FLAGS_IMPLICIT_ARRAY) && left > PM_BINDING_POWER_MODIFIER;
22541 case PM_CALL_NODE: {
22542 // Calls with an implicit array on the right-hand side are
22543 // statements and can only be followed by modifiers.
22544 if (PM_NODE_FLAG_P(node, PM_CALL_NODE_FLAGS_IMPLICIT_ARRAY)) {
22545 return left > PM_BINDING_POWER_MODIFIER;
22546 }
22547
22548 // Command-style calls (including block commands like
22549 // `foo bar do end`) can only be followed by composition
22550 // (and/or) and modifier (if/unless/etc.) operators.
22551 if (pm_command_call_value_p(parser, node)) {
22552 return left > PM_BINDING_POWER_COMPOSITION;
22553 }
22554
22555 // A block call (command with do-block, or any call chained
22556 // from one) can only be followed by call chaining (., ::,
22557 // &.), composition (and/or), and modifier operators.
22558 return left > PM_BINDING_POWER_COMPOSITION && left < PM_BINDING_POWER_CALL && pm_block_call_p(node);
22559 }
22560 case PM_SUPER_NODE:
22561 case PM_YIELD_NODE:
22562 // Command-style super/yield (without parens) can only be followed
22563 // by composition and modifier operators.
22564 if (pm_command_call_value_p(parser, node)) {
22565 return left > PM_BINDING_POWER_COMPOSITION;
22566 }
22567
22568 /* A super carrying a do-block is a block call, so it may also be
22569 * followed by call chaining (`.`, `::`, `&.`). */
22570 return left > PM_BINDING_POWER_COMPOSITION && left < PM_BINDING_POWER_CALL && pm_block_call_p(node);
22571 case PM_DEF_NODE:
22572 // An endless method whose body is a command-style call (e.g.,
22573 // `def f = foo bar`) is a command assignment and can only be
22574 // followed by modifiers.
22575 return left > PM_BINDING_POWER_MODIFIER && pm_command_call_value_p(parser, node);
22576 case PM_RESCUE_MODIFIER_NODE:
22577 // A rescue modifier whose handler is a pattern match (=> or in)
22578 // produces a statement and cannot be followed by operators above
22579 // the modifier level.
22580 if (left > PM_BINDING_POWER_MODIFIER) {
22582 pm_node_t *rescue_expression = cast->rescue_expression;
22583 return PM_NODE_TYPE_P(rescue_expression, PM_MATCH_REQUIRED_NODE) || PM_NODE_TYPE_P(rescue_expression, PM_MATCH_PREDICATE_NODE);
22584 }
22585 return false;
22586 default:
22587 return false;
22588 }
22589}
22590
22599static pm_node_t *
22600parse_expression(pm_parser_t *parser, pm_binding_power_t binding_power, uint8_t flags, pm_diagnostic_id_t diag_id, uint16_t depth) {
22601 if (PRISM_UNLIKELY(depth >= PRISM_DEPTH_MAXIMUM)) {
22602 pm_parser_err_current(parser, PM_ERR_NESTING_TOO_DEEP);
22603 return UP(pm_error_recovery_node_create(parser, PM_TOKEN_START(parser, &parser->current), PM_TOKEN_LENGTH(&parser->current)));
22604 }
22605
22606 pm_node_t *node = parse_expression_prefix(parser, binding_power, flags, diag_id, depth);
22607
22608 // Some prefix nodes are statements and can only be followed by modifiers
22609 // (if/unless/while/until/rescue) or nothing at all. We check these cheaply
22610 // here before entering the infix loop.
22611 switch (PM_NODE_TYPE(node)) {
22612 case PM_ERROR_RECOVERY_NODE:
22613 return node;
22614 case PM_PRE_EXECUTION_NODE:
22615 return node;
22616 case PM_POST_EXECUTION_NODE:
22617 case PM_ALIAS_GLOBAL_VARIABLE_NODE:
22618 case PM_ALIAS_METHOD_NODE:
22619 case PM_UNDEF_NODE:
22620 if (pm_binding_powers[parser->current.type].left > PM_BINDING_POWER_MODIFIER) {
22621 return node;
22622 }
22623 break;
22624 case PM_CALL_NODE:
22625 case PM_SUPER_NODE:
22626 case PM_YIELD_NODE:
22627 case PM_DEF_NODE:
22628 if (parse_expression_terminator(parser, node)) {
22629 return node;
22630 }
22631 break;
22632 case PM_SYMBOL_NODE:
22633 if (pm_symbol_node_label_p(parser, node)) {
22634 return node;
22635 }
22636 break;
22637 default:
22638 break;
22639 }
22640
22641 // Look and see if the next token can be parsed as an infix operator. If it
22642 // can, then we'll parse it using parse_expression_infix.
22643 pm_binding_powers_t current_binding_powers;
22644 pm_token_type_t current_token_type;
22645
22646 while (
22647 current_token_type = parser->current.type,
22648 current_binding_powers = pm_binding_powers[current_token_type],
22649 binding_power <= current_binding_powers.left &&
22650 current_binding_powers.binary
22651 ) {
22652 node = parse_expression_infix(parser, node, binding_power, current_binding_powers.right, flags, (uint16_t) (depth + 1));
22653 if (parse_expression_terminator(parser, node)) return node;
22654
22655 // If the operator is nonassoc and we should not be able to parse the
22656 // upcoming infix operator, break.
22657 if (current_binding_powers.nonassoc) {
22658 // If we are about to parse another non-associative operator at the
22659 // same precedence as the one we just parsed, then we need to add an
22660 // error. This covers chaining the same operator (`1 == 2 == 3`) as
22661 // well as different operators that share a precedence, since they
22662 // are equally non-associative with one another (`1 == 2 != 3`,
22663 // `1...2..3`).
22664 pm_binding_powers_t next_binding_powers = pm_binding_powers[parser->current.type];
22665 if (next_binding_powers.nonassoc && next_binding_powers.left == current_binding_powers.left) {
22666 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->current, PM_ERR_NON_ASSOCIATIVE_OPERATOR, pm_token_str(parser->current.type), pm_token_str(current_token_type));
22667 break;
22668 }
22669
22670 // If this is an endless range, then we need to reject a couple of
22671 // additional operators because it violates the normal operator
22672 // precedence rules. Those patterns are:
22673 //
22674 // 1.. & 2
22675 // 1.. * 2
22676 //
22677 if (PM_NODE_TYPE_P(node, PM_RANGE_NODE) && ((pm_range_node_t *) node)->right == NULL) {
22678 if (match4(parser, PM_TOKEN_UAMPERSAND, PM_TOKEN_USTAR, PM_TOKEN_DOT, PM_TOKEN_AMPERSAND_DOT)) {
22679 PM_PARSER_ERR_TOKEN_FORMAT(parser, &parser->current, PM_ERR_NON_ASSOCIATIVE_OPERATOR, pm_token_str(parser->current.type), pm_token_str(current_token_type));
22680 break;
22681 }
22682
22683 if (PM_BINDING_POWER_TERM <= next_binding_powers.left) {
22684 break;
22685 }
22686 } else if (current_binding_powers.left <= next_binding_powers.left) {
22687 break;
22688 }
22689 }
22690
22691 if (flags & PM_PARSE_ACCEPTS_COMMAND_CALL) {
22692 // A command-style method call is only accepted on method chains.
22693 // Thus, we check whether the parsed node can continue method chains.
22694 // The method chain can continue if the parsed node is one of the following five kinds:
22695 // (1) index access: foo[1]
22696 // (2) attribute access: foo.bar
22697 // (3) method call with parenthesis: foo.bar(1)
22698 // (4) method call with a block: foo.bar do end
22699 // (5) constant path: foo::Bar
22700 switch (node->type) {
22701 case PM_CALL_NODE: {
22702 pm_call_node_t *cast = (pm_call_node_t *)node;
22703 if (
22704 // (1) foo[1]
22705 !(
22706 cast->call_operator_loc.length == 0 &&
22707 cast->message_loc.length > 0 &&
22708 parser->start[cast->message_loc.start] == '[' &&
22709 parser->start[cast->message_loc.start + cast->message_loc.length - 1] == ']'
22710 ) &&
22711 // (2) foo.bar
22712 !(
22713 cast->call_operator_loc.length > 0 &&
22714 cast->arguments == NULL &&
22715 cast->block == NULL &&
22716 cast->opening_loc.length == 0
22717 ) &&
22718 // (3) foo.bar(1)
22719 !(
22720 cast->call_operator_loc.length > 0 &&
22721 cast->opening_loc.length > 0
22722 ) &&
22723 // (4) foo.bar do end
22724 !(
22725 cast->block != NULL && PM_NODE_TYPE_P(cast->block, PM_BLOCK_NODE)
22726 )
22727 ) {
22728 flags &= (uint8_t) ~PM_PARSE_ACCEPTS_COMMAND_CALL;
22729 }
22730 break;
22731 }
22732 // (5) foo::Bar
22733 case PM_CONSTANT_PATH_NODE:
22734 break;
22735 default:
22736 flags &= (uint8_t) ~PM_PARSE_ACCEPTS_COMMAND_CALL;
22737 break;
22738 }
22739 }
22740
22741 if (context_terminator(parser->current_context->context, &parser->current)) {
22742 pm_binding_powers_t next_binding_powers = pm_binding_powers[parser->current.type];
22743 if (
22744 !next_binding_powers.binary ||
22745 binding_power > next_binding_powers.left ||
22746 (PM_NODE_TYPE_P(node, PM_CALL_NODE) && pm_call_node_command_p((pm_call_node_t *) node))
22747 ) {
22748 return node;
22749 }
22750 }
22751 }
22752
22753 return node;
22754}
22755
22760static pm_statements_node_t *
22761wrap_statements(pm_parser_t *parser, pm_statements_node_t *statements) {
22762 if (PM_PARSER_COMMAND_LINE_OPTION_P(parser)) {
22763 if (statements == NULL) {
22764 statements = pm_statements_node_create(parser);
22765 }
22766
22767 pm_arguments_node_t *arguments = pm_arguments_node_create(parser);
22768 pm_arguments_node_arguments_append(
22769 parser->arena,
22770 arguments,
22771 UP(pm_global_variable_read_node_synthesized_create(parser, pm_parser_constant_id_constant(parser, "$_", 2)))
22772 );
22773
22774 pm_statements_node_body_append(parser, statements, UP(pm_call_node_fcall_synthesized_create(
22775 parser,
22776 arguments,
22777 pm_parser_constant_id_constant(parser, "print", 5)
22778 )), true);
22779 }
22780
22781 if (PM_PARSER_COMMAND_LINE_OPTION_N(parser)) {
22782 if (PM_PARSER_COMMAND_LINE_OPTION_A(parser)) {
22783 if (statements == NULL) {
22784 statements = pm_statements_node_create(parser);
22785 }
22786
22787 pm_arguments_node_t *arguments = pm_arguments_node_create(parser);
22788 pm_arguments_node_arguments_append(
22789 parser->arena,
22790 arguments,
22791 UP(pm_global_variable_read_node_synthesized_create(parser, pm_parser_constant_id_constant(parser, "$;", 2)))
22792 );
22793
22794 pm_global_variable_read_node_t *receiver = pm_global_variable_read_node_synthesized_create(parser, pm_parser_constant_id_constant(parser, "$_", 2));
22795 pm_call_node_t *call = pm_call_node_call_synthesized_create(parser, UP(receiver), "split", arguments);
22796
22797 pm_global_variable_write_node_t *write = pm_global_variable_write_node_synthesized_create(
22798 parser,
22799 pm_parser_constant_id_constant(parser, "$F", 2),
22800 UP(call)
22801 );
22802
22803 pm_statements_node_body_prepend(parser->arena, statements, UP(write));
22804 }
22805
22806 pm_arguments_node_t *arguments = pm_arguments_node_create(parser);
22807 pm_arguments_node_arguments_append(
22808 parser->arena,
22809 arguments,
22810 UP(pm_global_variable_read_node_synthesized_create(parser, pm_parser_constant_id_constant(parser, "$/", 2)))
22811 );
22812
22813 if (PM_PARSER_COMMAND_LINE_OPTION_L(parser)) {
22814 pm_keyword_hash_node_t *keywords = pm_keyword_hash_node_create(parser);
22815 pm_keyword_hash_node_elements_append(parser->arena, keywords, UP(pm_assoc_node_create(
22816 parser,
22817 UP(pm_symbol_node_synthesized_create(parser, "chomp")),
22818 NULL,
22819 UP(pm_true_node_synthesized_create(parser))
22820 )));
22821
22822 pm_arguments_node_arguments_append(parser->arena, arguments, UP(keywords));
22823 pm_node_flag_set(UP(arguments), PM_ARGUMENTS_NODE_FLAGS_CONTAINS_KEYWORDS);
22824 }
22825
22826 pm_statements_node_t *wrapped_statements = pm_statements_node_create(parser);
22827 pm_statements_node_body_append(parser, wrapped_statements, UP(pm_while_node_synthesized_create(
22828 parser,
22829 UP(pm_call_node_fcall_synthesized_create(parser, arguments, pm_parser_constant_id_constant(parser, "gets", 4))),
22830 statements
22831 )), true);
22832
22833 statements = wrapped_statements;
22834 }
22835
22836 return statements;
22837}
22838
22842static pm_node_t *
22843parse_program(pm_parser_t *parser) {
22844 // If the current scope is NULL, then we want to push a new top level scope.
22845 // The current scope could exist in the event that we are parsing an eval
22846 // and the user has passed into scopes that already exist.
22847 if (parser->current_scope == NULL) {
22848 pm_parser_scope_push(parser, true);
22849 }
22850
22851 pm_node_list_t current_block_exits = { 0 };
22852 pm_node_list_t *previous_block_exits = push_block_exits(parser, &current_block_exits);
22853
22854 parser_lex(parser);
22855 pm_statements_node_t *statements = parse_statements(parser, PM_CONTEXT_MAIN, 0);
22856
22857 if (statements != NULL && !parser->parsing_eval) {
22858 // If we have statements, then the top-level statement should be
22859 // explicitly checked as well. We have to do this here because
22860 // everywhere else we check all but the last statement.
22861 assert(statements->body.size > 0);
22862 pm_void_statement_check(parser, statements->body.nodes[statements->body.size - 1]);
22863 }
22864
22865 pm_constant_id_list_t locals;
22866 pm_locals_order(parser, &parser->current_scope->locals, &locals, true);
22867 pm_parser_scope_pop(parser);
22868
22869 // At the top level, see if we need to wrap the statements in a program
22870 // node with a while loop based on the options.
22871 if (parser->command_line & (PM_OPTIONS_COMMAND_LINE_P | PM_OPTIONS_COMMAND_LINE_N)) {
22872 statements = wrap_statements(parser, statements);
22873 } else {
22874 flush_block_exits(parser, previous_block_exits);
22875 }
22876
22877 // If this is an empty file, then we're still going to parse all of the
22878 // statements in order to gather up all of the comments and such. Here we'll
22879 // correct the location information.
22880 if (statements == NULL) {
22881 statements = pm_statements_node_create(parser);
22882 statements->base.location = (pm_location_t) { 0 };
22883 }
22884
22885 return UP(pm_program_node_create(parser, &locals, statements));
22886}
22887
22888/******************************************************************************/
22889/* External functions */
22890/******************************************************************************/
22891
22901static const char *
22902pm_strnstr(const char *big, const char *little, size_t big_length) {
22903 size_t little_length = strlen(little);
22904
22905 for (const char *max = big + big_length - little_length; big <= max; big++) {
22906 if (*big == *little && memcmp(big, little, little_length) == 0) return big;
22907 }
22908
22909 return NULL;
22910}
22911
22912#ifdef _WIN32
22913#define pm_parser_warn_shebang_carriage_return(parser, start, length) ((void) 0)
22914#else
22920static void
22921pm_parser_warn_shebang_carriage_return(pm_parser_t *parser, const uint8_t *start, size_t length) {
22922 if (length > 2 && start[length - 2] == '\r' && start[length - 1] == '\n') {
22923 pm_parser_warn(parser, U32(start - parser->start), U32(length), PM_WARN_SHEBANG_CARRIAGE_RETURN);
22924 }
22925}
22926#endif
22927
22932static void
22933pm_parser_init_shebang(pm_parser_t *parser, const pm_options_t *options, const char *engine, size_t length) {
22934 const char *switches = pm_strnstr(engine, " -", length);
22935 if (switches == NULL) return;
22936
22937 pm_options_t next_options = *options;
22938 options->shebang_callback(
22939 &next_options,
22940 (const uint8_t *) (switches + 1),
22941 length - ((size_t) (switches - engine)) - 1,
22942 options->shebang_callback_data
22943 );
22944
22945 size_t encoding_length;
22946 if ((encoding_length = pm_string_length(&next_options.encoding)) > 0) {
22947 const uint8_t *encoding_source = pm_string_source(&next_options.encoding);
22948 parser_lex_magic_comment_encoding_value(parser, encoding_source, encoding_source + encoding_length);
22949 }
22950
22951 parser->command_line = next_options.command_line;
22952 parser->frozen_string_literal = next_options.frozen_string_literal;
22953}
22954
22958void
22959pm_parser_init(pm_arena_t *arena, pm_parser_t *parser, const uint8_t *source, size_t size, const pm_options_t *options) {
22960 assert(arena != NULL);
22961 assert(source != NULL);
22962
22963 *parser = (pm_parser_t) {
22964 .arena = arena,
22965 .metadata_arena = { 0 },
22966 .node_id = 0,
22967 .lex_state = PM_LEX_STATE_BEG,
22968 .enclosure_nesting = 0,
22969 .lambda_enclosure_nesting = -1,
22970 .brace_nesting = 0,
22971 .do_loop_stack = 0,
22972 .accepts_block_stack = 0,
22973 .lex_modes = {
22974 .index = 0,
22975 .stack = {{ .mode = PM_LEX_DEFAULT }},
22976 .current = &parser->lex_modes.stack[0],
22977 },
22978 .start = source,
22979 .end = source + size,
22980 .previous = { .type = PM_TOKEN_EOF, .start = source, .end = source },
22981 .current = { .type = PM_TOKEN_EOF, .start = source, .end = source },
22982 .next_start = NULL,
22983 .heredoc_end = NULL,
22984 .data_loc = { 0 },
22985 .comment_list = { 0 },
22986 .magic_comment_list = { 0 },
22987 .warning_list = { 0 },
22988 .error_list = { 0 },
22989 .current_scope = NULL,
22990 .current_context = NULL,
22991 .encoding = PM_ENCODING_UTF_8_ENTRY,
22992 .encoding_changed_callback = NULL,
22993 .encoding_comment_start = source,
22994 .lex_callback = { 0 },
22995 .filepath = { 0 },
22996 .constant_pool = { 0 },
22997 .line_offsets = { 0 },
22998 .integer = { 0 },
22999 .current_string = PM_STRING_EMPTY,
23000 .start_line = 1,
23001 .explicit_encoding = NULL,
23002 .command_line = 0,
23003 .parsing_eval = false,
23004 .partial_script = false,
23005 .command_start = true,
23006 .recovering = false,
23007 .continuable = true,
23008 .encoding_locked = false,
23009 .encoding_changed = false,
23010 .pattern_matching_newlines = false,
23011 .in_keyword_arg = false,
23012 .current_block_exits = NULL,
23013 .semantic_token_seen = false,
23014 .frozen_string_literal = PM_OPTIONS_FROZEN_STRING_LITERAL_UNSET,
23015 .warn_mismatched_indentation = true
23016 };
23017
23018 /* Pre-size the arenas based on input size to reduce the number of block
23019 * allocations (and the kernel page zeroing they trigger). The ratios were
23020 * measured empirically: AST arena ~3.3x input, metadata arena ~1.1x input.
23021 * The reserve call is a no-op when the capacity is at or below the default
23022 * arena block size, so small inputs don't waste an extra allocation. */
23023 if (size <= SIZE_MAX / 4) pm_arena_reserve(arena, size * 4);
23024 if (size <= SIZE_MAX / 5 * 4) pm_arena_reserve(&parser->metadata_arena, size + size / 4);
23025
23026 /* Initialize the constant pool. Measured across 1532 Ruby stdlib files, the
23027 * bytes/constant ratio has a median of ~56 and a 90th percentile of ~135.
23028 * We use 120 as a balance between over-allocation waste and resize
23029 * frequency. Resizes are cheap with arena allocation, so we lean toward
23030 * under-estimating. */
23031 uint32_t constant_size = ((uint32_t) size) / 120;
23032 pm_constant_pool_init(&parser->metadata_arena, &parser->constant_pool, constant_size < 4 ? 4 : constant_size);
23033
23034 /* Initialize the line offset list. Similar to the constant pool, we are
23035 * going to estimate the number of newlines that we will need based on the
23036 * size of the input. */
23037 size_t newline_size = size / 22;
23038 pm_line_offset_list_init(&parser->metadata_arena, &parser->line_offsets, newline_size < 4 ? 4 : newline_size);
23039
23040 // If options were provided to this parse, establish them here.
23041 if (options != NULL) {
23042 // filepath option
23043 parser->filepath = options->filepath;
23044
23045 // line option
23046 parser->start_line = options->line;
23047
23048 // encoding option
23049 size_t encoding_length = pm_string_length(&options->encoding);
23050 if (encoding_length > 0) {
23051 const uint8_t *encoding_source = pm_string_source(&options->encoding);
23052 parser_lex_magic_comment_encoding_value(parser, encoding_source, encoding_source + encoding_length);
23053 }
23054
23055 // encoding_locked option
23056 parser->encoding_locked = options->encoding_locked;
23057
23058 // frozen_string_literal option
23059 parser->frozen_string_literal = options->frozen_string_literal;
23060
23061 // command_line option
23062 parser->command_line = options->command_line;
23063
23064 // version option
23065 parser->version = options->version;
23066
23067 // partial_script
23068 parser->partial_script = options->partial_script;
23069
23070 // scopes option
23071 parser->parsing_eval = options->scopes_count > 0;
23072 if (parser->parsing_eval) parser->warn_mismatched_indentation = false;
23073
23074 for (size_t scope_index = 0; scope_index < options->scopes_count; scope_index++) {
23075 const pm_options_scope_t *scope = pm_options_scope(options, scope_index);
23076 pm_parser_scope_push(parser, scope_index == 0);
23077
23078 // Scopes given from the outside are not allowed to have numbered
23079 // parameters.
23080 parser->current_scope->parameters = ((pm_scope_parameters_t) scope->forwarding) | PM_SCOPE_PARAMETERS_IMPLICIT_DISALLOWED;
23081
23082 for (size_t local_index = 0; local_index < scope->locals_count; local_index++) {
23083 const pm_string_t *local = pm_options_scope_local(scope, local_index);
23084
23085 const uint8_t *source = pm_string_source(local);
23086 size_t length = pm_string_length(local);
23087
23088 uint8_t *allocated = (uint8_t *) pm_arena_alloc(&parser->metadata_arena, length, 1);
23089 memcpy(allocated, source, length);
23090 pm_parser_local_add_owned(parser, allocated, length);
23091 }
23092 }
23093 }
23094
23095 // Now that we have established the user-provided options, check if
23096 // a version was given and parse as the latest version otherwise.
23097 if (parser->version == PM_OPTIONS_VERSION_UNSET) {
23098 parser->version = PM_OPTIONS_VERSION_LATEST;
23099 }
23100
23101 pm_accepts_block_stack_push(parser, true);
23102
23103 // Skip past the UTF-8 BOM if it exists.
23104 if (size >= 3 && source[0] == 0xef && source[1] == 0xbb && source[2] == 0xbf) {
23105 parser->current.end += 3;
23106 parser->encoding_comment_start += 3;
23107
23108 if (parser->encoding != PM_ENCODING_UTF_8_ENTRY) {
23109 parser->encoding = PM_ENCODING_UTF_8_ENTRY;
23110 if (parser->encoding_changed_callback != NULL) parser->encoding_changed_callback(parser);
23111 }
23112 }
23113
23114 // If the -x command line flag is set, or the first shebang of the file does
23115 // not include "ruby", then we'll search for a shebang that does include
23116 // "ruby" and start parsing from there.
23117 bool search_shebang = PM_PARSER_COMMAND_LINE_OPTION_X(parser);
23118
23119 // If the first two bytes of the source are a shebang, then we will do a bit
23120 // of extra processing.
23121 //
23122 // First, we'll indicate that the encoding comment is at the end of the
23123 // shebang. This means that when a shebang is present the encoding comment
23124 // can begin on the second line.
23125 //
23126 // Second, we will check if the shebang includes "ruby". If it does, then we
23127 // we will start parsing from there. We will also potentially warning the
23128 // user if there is a carriage return at the end of the shebang. We will
23129 // also potentially call the shebang callback if this is the main script to
23130 // allow the caller to parse the shebang and find any command-line options.
23131 // If the shebang does not include "ruby" and this is the main script being
23132 // parsed, then we will start searching the file for a shebang that does
23133 // contain "ruby" as if -x were passed on the command line.
23134 const uint8_t *newline = next_newline(parser->current.end, parser->end - parser->current.end);
23135 size_t length = (size_t) ((newline != NULL ? newline : parser->end) - parser->current.end);
23136
23137 if (length > 2 && parser->current.end[0] == '#' && parser->current.end[1] == '!') {
23138 const char *engine;
23139
23140 if ((engine = pm_strnstr((const char *) parser->start, "ruby", length)) != NULL) {
23141 if (newline != NULL) {
23142 parser->encoding_comment_start = newline + 1;
23143
23144 if (options == NULL || options->main_script) {
23145 pm_parser_warn_shebang_carriage_return(parser, parser->start, length + 1);
23146 }
23147 }
23148
23149 if (options != NULL && options->main_script && options->shebang_callback != NULL) {
23150 pm_parser_init_shebang(parser, options, engine, length - ((size_t) (engine - (const char *) parser->start)));
23151 }
23152
23153 search_shebang = false;
23154 } else if (options != NULL && options->main_script && !parser->parsing_eval) {
23155 search_shebang = true;
23156 }
23157 }
23158
23159 // Here we're going to find the first shebang that includes "ruby" and start
23160 // parsing from there.
23161 if (search_shebang) {
23162 // If a shebang that includes "ruby" is not found, then we're going to a
23163 // a load error to the list of errors on the parser.
23164 bool found_shebang = false;
23165
23166 // This is going to point to the start of each line as we check it.
23167 // We'll maintain a moving window looking at each line at they come.
23168 const uint8_t *cursor = parser->start;
23169
23170 // The newline pointer points to the end of the current line that we're
23171 // considering. If it is NULL, then we're at the end of the file.
23172 const uint8_t *newline = next_newline(cursor, parser->end - cursor);
23173
23174 while (newline != NULL) {
23175 pm_line_offset_list_append(&parser->metadata_arena, &parser->line_offsets, U32(newline - parser->start + 1));
23176
23177 cursor = newline + 1;
23178 newline = next_newline(cursor, parser->end - cursor);
23179
23180 size_t length = (size_t) ((newline != NULL ? newline : parser->end) - cursor);
23181 if (length > 2 && cursor[0] == '#' && cursor[1] == '!') {
23182 const char *engine;
23183 if ((engine = pm_strnstr((const char *) cursor, "ruby", length)) != NULL) {
23184 found_shebang = true;
23185
23186 if (newline != NULL) {
23187 pm_parser_warn_shebang_carriage_return(parser, cursor, length + 1);
23188 parser->encoding_comment_start = newline + 1;
23189 }
23190
23191 if (options != NULL && options->shebang_callback != NULL) {
23192 pm_parser_init_shebang(parser, options, engine, length - ((size_t) (engine - (const char *) cursor)));
23193 }
23194
23195 break;
23196 }
23197 }
23198 }
23199
23200 if (found_shebang) {
23201 parser->previous = (pm_token_t) { .type = PM_TOKEN_EOF, .start = cursor, .end = cursor };
23202 parser->current = (pm_token_t) { .type = PM_TOKEN_EOF, .start = cursor, .end = cursor };
23203 } else {
23204 pm_parser_err(parser, 0, 0, PM_ERR_SCRIPT_NOT_FOUND);
23205 pm_line_offset_list_clear(&parser->line_offsets);
23206 }
23207 }
23208
23209 // The encoding comment can start after any amount of inline whitespace, so
23210 // here we'll advance it to the first non-inline-whitespace character so
23211 // that it is ready for future comparisons.
23212 parser->encoding_comment_start += pm_strspn_inline_whitespace(parser->encoding_comment_start, parser->end - parser->encoding_comment_start);
23213}
23214
23223pm_parser_new(pm_arena_t *arena, const uint8_t *source, size_t size, const pm_options_t *options) {
23224 pm_parser_t *parser = (pm_parser_t *) xmalloc(sizeof(pm_parser_t));
23225 if (parser == NULL) abort();
23226
23227 pm_parser_init(arena, parser, source, size, options);
23228 return parser;
23229}
23230
23234void
23235pm_parser_cleanup(pm_parser_t *parser) {
23236 pm_string_cleanup(&parser->filepath);
23237 pm_arena_cleanup(&parser->metadata_arena);
23238
23239 while (parser->current_scope != NULL) {
23240 // Normally, popping the scope doesn't free the locals since it is
23241 // assumed that ownership has transferred to the AST. However if we have
23242 // scopes while we're freeing the parser, it's likely they came from
23243 // eval scopes and we need to free them explicitly here.
23244 pm_parser_scope_pop(parser);
23245 }
23246
23247 while (parser->lex_modes.index >= PM_LEX_STACK_SIZE) {
23248 lex_mode_pop(parser);
23249 }
23250}
23251
23255void
23257 pm_parser_cleanup(parser);
23258 xfree_sized(parser, sizeof(pm_parser_t));
23259}
23260
23266static bool
23267pm_parse_err_is_fatal(pm_diagnostic_id_t diag_id) {
23268 switch (diag_id) {
23269 case PM_ERR_ARRAY_EXPRESSION_AFTER_STAR:
23270 case PM_ERR_BEGIN_UPCASE_BRACE:
23271 case PM_ERR_CLASS_VARIABLE_BARE:
23272 case PM_ERR_END_UPCASE_BRACE:
23273 case PM_ERR_ESCAPE_INVALID_HEXADECIMAL:
23274 case PM_ERR_ESCAPE_INVALID_UNICODE_LIST:
23275 case PM_ERR_ESCAPE_INVALID_UNICODE_SHORT:
23276 case PM_ERR_EXPRESSION_NOT_WRITABLE:
23277 case PM_ERR_EXPRESSION_NOT_WRITABLE_SELF:
23278 case PM_ERR_FLOAT_PARSE:
23279 case PM_ERR_GLOBAL_VARIABLE_BARE:
23280 case PM_ERR_HASH_KEY:
23281 case PM_ERR_HEREDOC_IDENTIFIER:
23282 case PM_ERR_INSTANCE_VARIABLE_BARE:
23283 case PM_ERR_INVALID_BLOCK_EXIT:
23284 case PM_ERR_INVALID_ENCODING_MAGIC_COMMENT:
23285 case PM_ERR_INVALID_FLOAT_EXPONENT:
23286 case PM_ERR_INVALID_NUMBER_BINARY:
23287 case PM_ERR_INVALID_NUMBER_DECIMAL:
23288 case PM_ERR_INVALID_NUMBER_HEXADECIMAL:
23289 case PM_ERR_INVALID_NUMBER_OCTAL:
23290 case PM_ERR_INVALID_NUMBER_UNDERSCORE_TRAILING:
23291 case PM_ERR_NO_LOCAL_VARIABLE:
23292 case PM_ERR_PARAMETER_ORDER:
23293 case PM_ERR_STATEMENT_UNDEF:
23294 case PM_ERR_VOID_EXPRESSION:
23295 return true;
23296 default:
23297 return false;
23298 }
23299}
23300
23334static void
23335pm_parse_continuable(pm_parser_t *parser) {
23336 // If there are no errors then there is nothing to continue.
23337 if (parser->error_list.size == 0) {
23338 parser->continuable = false;
23339 return;
23340 }
23341
23342 if (!parser->continuable) return;
23343
23344 size_t source_length = (size_t) (parser->end - parser->start);
23345
23346 // First pass: check if there are any non-stray, non-fatal errors.
23347 bool has_non_stray_error = false;
23348 for (pm_diagnostic_t *error = (pm_diagnostic_t *) parser->error_list.head; error != NULL; error = (pm_diagnostic_t *) error->node.next) {
23349 if (error->diag_id != PM_ERR_UNEXPECTED_TOKEN_IGNORE && error->diag_id != PM_ERR_UNEXPECTED_TOKEN_CLOSE_CONTEXT && !pm_parse_err_is_fatal(error->diag_id)) {
23350 has_non_stray_error = true;
23351 break;
23352 }
23353 }
23354
23355 // Second pass: check each error. We track the minimum source position
23356 // among non-stray, non-fatal errors seen so far in list order, which
23357 // lets us detect cascade stray tokens.
23358 size_t non_stray_min_start = SIZE_MAX;
23359
23360 for (pm_diagnostic_t *error = (pm_diagnostic_t *) parser->error_list.head; error != NULL; error = (pm_diagnostic_t *) error->node.next) {
23361 size_t error_start = (size_t) error->location.start;
23362 size_t error_end = error_start + (size_t) error->location.length;
23363 bool at_eof = error_end >= source_length;
23364
23365 // Fatal errors are non-continuable unless they occur at EOF.
23366 if (pm_parse_err_is_fatal(error->diag_id) && !at_eof) {
23367 parser->continuable = false;
23368 return;
23369 }
23370
23371 // Track non-stray, non-fatal error positions in list order.
23372 if (error->diag_id != PM_ERR_UNEXPECTED_TOKEN_IGNORE &&
23373 error->diag_id != PM_ERR_UNEXPECTED_TOKEN_CLOSE_CONTEXT) {
23374 if (error_start < non_stray_min_start) non_stray_min_start = error_start;
23375 continue;
23376 }
23377
23378 // This is a stray token. Determine if it is a cascade effect
23379 // of a preceding error or genuinely stray.
23380
23381 // Rule (a): a non-stray error was seen earlier in the list at a
23382 // strictly earlier position — this stray is a cascade effect.
23383 if (non_stray_min_start < error_start) continue;
23384
23385 // Rule (b): this stray is at EOF with valid code before it.
23386 // Single-byte stray tokens at EOF (like `\` for line continuation)
23387 // are likely truncated tokens. Multi-byte stray tokens (like the
23388 // keyword `end`) need additional evidence that they are cascade
23389 // effects (i.e. non-stray errors exist elsewhere).
23390 if (at_eof && error_start > 0) {
23391 // Exception: closing delimiters at EOF are genuinely stray.
23392 if (error->location.length == 1) {
23393 const uint8_t *byte = parser->start + error_start;
23394 if (*byte == ')' || *byte == ']' || *byte == '}') {
23395 parser->continuable = false;
23396 return;
23397 }
23398
23399 // Single-byte non-delimiter stray at EOF: cascade.
23400 continue;
23401 }
23402
23403 // Multi-byte stray at EOF: cascade only if there are
23404 // non-stray errors (evidence of a preceding parse failure).
23405 if (has_non_stray_error) continue;
23406 }
23407
23408 // Rule (c): a stray `=` at the start of a line could be the
23409 // beginning of an embedded document (`=begin`). The remaining
23410 // bytes after `=` parse as an identifier, so the error is not
23411 // at EOF, but the construct is genuinely incomplete.
23412 if (error->location.length == 1) {
23413 const uint8_t *byte = parser->start + error_start;
23414 if (*byte == '=' && (error_start == 0 || *(byte - 1) == '\n')) continue;
23415 }
23416
23417 // This stray token is genuinely non-continuable.
23418 parser->continuable = false;
23419 return;
23420 }
23421}
23422
23426pm_node_t *
23428 pm_node_t *node = parse_program(parser);
23429 pm_parse_continuable(parser);
23430 return node;
23431}
23432
23439pm_node_t *
23440pm_parse_stream(pm_parser_t **parser, pm_arena_t *arena, pm_source_t *source, const pm_options_t *options) {
23441 bool eof = pm_source_stream_read(source);
23442
23443 pm_parser_t *tmp = pm_parser_new(arena, pm_source_source(source), pm_source_length(source), options);
23444 pm_node_t *node = pm_parse(tmp);
23445
23446 while (!eof && tmp->error_list.size > 0) {
23447 eof = pm_source_stream_read(source);
23448
23449 pm_parser_free(tmp);
23450 pm_arena_cleanup(arena);
23451
23452 tmp = pm_parser_new(arena, pm_source_source(source), pm_source_length(source), options);
23453 node = pm_parse(tmp);
23454 }
23455
23456 *parser = tmp;
23457 return node;
23458}
23459
23460#undef PM_CASE_KEYWORD
23461#undef PM_CASE_OPERATOR
23462#undef PM_CASE_WRITABLE
23463#undef PM_STRING_EMPTY
23464
23465// We optionally support serializing to a binary string. For systems that don't
23466// want or need this functionality, it can be turned off with the
23467// PRISM_EXCLUDE_SERIALIZATION define.
23468#ifndef PRISM_EXCLUDE_SERIALIZATION
23469
23470static PRISM_INLINE void
23471pm_serialize_header(pm_buffer_t *buffer) {
23472 pm_buffer_append_string(buffer, "PRISM", 5);
23473 pm_buffer_append_byte(buffer, PRISM_VERSION_MAJOR);
23474 pm_buffer_append_byte(buffer, PRISM_VERSION_MINOR);
23475 pm_buffer_append_byte(buffer, PRISM_VERSION_PATCH);
23476 pm_buffer_append_byte(buffer, PRISM_SERIALIZE_ONLY_SEMANTICS_FIELDS ? 1 : 0);
23477}
23478
23482void
23483pm_serialize(pm_parser_t *parser, pm_node_t *node, pm_buffer_t *buffer) {
23484 pm_serialize_header(buffer);
23485 pm_serialize_content(parser, node, buffer);
23486 pm_buffer_append_byte(buffer, '\0');
23487}
23488
23493void
23494pm_serialize_parse(pm_buffer_t *buffer, const uint8_t *source, size_t size, const char *data) {
23495 pm_options_t options = { 0 };
23496 pm_options_read(&options, data);
23497
23498 pm_arena_t arena = { 0 };
23499 pm_parser_t parser;
23500 pm_parser_init(&arena, &parser, source, size, &options);
23501
23502 pm_node_t *node = pm_parse(&parser);
23503
23504 pm_serialize_header(buffer);
23505 pm_serialize_content(&parser, node, buffer);
23506 pm_buffer_append_byte(buffer, '\0');
23507
23508 pm_parser_cleanup(&parser);
23509 pm_arena_cleanup(&arena);
23510 pm_options_cleanup(&options);
23511}
23512
23517void
23518pm_serialize_parse_stream(pm_buffer_t *buffer, pm_source_t *source, const char *data) {
23519 pm_arena_t arena = { 0 };
23520 pm_parser_t *parser;
23521 pm_options_t options = { 0 };
23522 pm_options_read(&options, data);
23523
23524 pm_node_t *node = pm_parse_stream(&parser, &arena, source, &options);
23525 pm_serialize_header(buffer);
23526 pm_serialize_content(parser, node, buffer);
23527 pm_buffer_append_byte(buffer, '\0');
23528
23529 pm_parser_free(parser);
23530 pm_arena_cleanup(&arena);
23531 pm_options_cleanup(&options);
23532}
23533
23542int8_t
23543pm_serialize_parse_errors_format(pm_buffer_t *buffer, const uint8_t *source, size_t size, const char *data, pm_errors_format_type_t format_type) {
23544 pm_options_t options = { 0 };
23545 pm_options_read(&options, data);
23546
23547 pm_arena_t arena = { 0 };
23548 pm_parser_t parser;
23549 pm_parser_init(&arena, &parser, source, size, &options);
23550
23551 pm_parse(&parser);
23552
23553 int8_t result = -1;
23554 if (parser.error_list.size > 0) {
23555 const char *encoding_name = parser.encoding->name;
23556 pm_buffer_append_string(buffer, encoding_name, strlen(encoding_name));
23557 pm_buffer_append_byte(buffer, '\0');
23558
23559 result = (int8_t) pm_errors_format(&parser, buffer, format_type);
23560 }
23561
23562 pm_parser_cleanup(&parser);
23563 pm_arena_cleanup(&arena);
23564 pm_options_cleanup(&options);
23565
23566 return result;
23567}
23568
23572void
23573pm_serialize_parse_comments(pm_buffer_t *buffer, const uint8_t *source, size_t size, const char *data) {
23574 pm_options_t options = { 0 };
23575 pm_options_read(&options, data);
23576
23577 pm_arena_t arena = { 0 };
23578 pm_parser_t parser;
23579 pm_parser_init(&arena, &parser, source, size, &options);
23580
23581 pm_parse(&parser);
23582 pm_serialize_header(buffer);
23583 pm_serialize_encoding(parser.encoding, buffer);
23584 pm_buffer_append_varsint(buffer, parser.start_line);
23585 pm_serialize_line_offset_list(&parser.line_offsets, buffer);
23586 pm_serialize_comment_list(&parser.comment_list, buffer);
23587
23588 pm_parser_cleanup(&parser);
23589 pm_arena_cleanup(&arena);
23590 pm_options_cleanup(&options);
23591}
23592
23593#endif
#define PRISM_ALIGNOF
Get the alignment requirement of a type.
Definition align.h:15
pm_comment_type_t
This is the type of a comment that we've found while parsing.
Definition comments.h:18
uint32_t pm_constant_id_t
A constant id is a unique identifier for a constant in the constant pool.
pm_errors_format_type_t
The type of formatting to use when formatting errors.
A header file that defines macros to exclude certain features of the prism library.
#define PRISM_FALLTHROUGH
We use -Wimplicit-fallthrough to guard potentially unintended fall-through between cases of a switch.
Definition fallthrough.h:15
#define xmalloc
Old name of ruby_xmalloc.
Definition xmalloc.h:53
#define xcalloc
Old name of ruby_xcalloc.
Definition xmalloc.h:55
int len
Length of the buffer.
Definition io.h:8
#define PRISM_INLINE
Old Visual Studio versions do not support the inline keyword, so we need to define it to be __inline.
Definition inline.h:12
VALUE type(ANYARGS)
ANYARGS-ed function type.
static const uint8_t PM_OPTIONS_COMMAND_LINE_N
A bit representing whether or not the command line -n option was set.
Definition options.h:96
#define PM_OPTIONS_FROZEN_STRING_LITERAL_DISABLED
String literals should not be frozen.
Definition options.h:31
#define PM_OPTIONS_FROZEN_STRING_LITERAL_ENABLED
String literals should be made frozen.
Definition options.h:42
#define PM_OPTIONS_FROZEN_STRING_LITERAL_UNSET
String literals may be frozen or mutable depending on the implementation default.
Definition options.h:37
static const uint8_t PM_OPTIONS_COMMAND_LINE_P
A bit representing whether or not the command line -p option was set.
Definition options.h:102
PRISM_NODISCARD PRISM_EXPORTED_FUNCTION pm_parser_t * pm_parser_new(pm_arena_t *arena, const uint8_t *source, size_t size, const pm_options_t *options) PRISM_NONNULL(1)
Allocate and initialize a parser with the given start and end pointers.
Definition prism.c:23223
PRISM_EXPORTED_FUNCTION void pm_parser_free(pm_parser_t *parser) PRISM_NONNULL(1)
Free both the memory held by the given parser and the parser itself.
Definition prism.c:23256
PRISM_EXPORTED_FUNCTION pm_node_t * pm_parse(pm_parser_t *parser) PRISM_NONNULL(1)
Initiate the parser with the given parser.
Definition prism.c:23427
#define PM_NODE_LIST_FOREACH(list, index, node)
Loop through each node in the node list, writing each node to the given pm_node_t pointer.
Definition node.h:18
The version of the Prism library.
#define PRISM_VERSION
The version of the Prism library as a constant string.
Definition version.h:29
#define PRISM_VERSION_PATCH
The patch version of the Prism library as an int.
Definition version.h:24
#define PRISM_VERSION_MINOR
The minor version of the Prism library as an int.
Definition version.h:19
#define PRISM_VERSION_MAJOR
The major version of the Prism library as an int.
Definition version.h:14
The functions related to serializing the AST to a binary format.
Functions for parsing streams.
AndNode.
Definition ast.h:1312
PM_NODE_ALIGNAS struct pm_node * left
AndNode::left.
Definition ast.h:1327
PM_NODE_ALIGNAS struct pm_node * right
AndNode::right.
Definition ast.h:1340
ArgumentsNode.
Definition ast.h:1372
pm_node_t base
The embedded base node.
Definition ast.h:1374
struct pm_node_list arguments
ArgumentsNode::arguments.
Definition ast.h:1384
This is a special out parameter to the parse_arguments_list function that includes opening and closin...
Definition prism.c:1764
pm_node_t * block
The optional block attached to the call.
Definition prism.c:1775
bool has_forwarding
The flag indicating whether this arguments list has forwarding argument.
Definition prism.c:1778
pm_location_t opening_loc
The optional location of the opening parenthesis or bracket.
Definition prism.c:1766
pm_arguments_node_t * arguments
The lazily-allocated optional arguments node.
Definition prism.c:1769
pm_location_t closing_loc
The optional location of the closing parenthesis or bracket.
Definition prism.c:1772
ArrayNode.
Definition ast.h:1402
struct pm_node_list elements
ArrayNode::elements.
Definition ast.h:1411
ArrayPatternNode.
Definition ast.h:1462
PM_NODE_ALIGNAS struct pm_node * constant
ArrayPatternNode::constant.
Definition ast.h:1480
pm_location_t opening_loc
ArrayPatternNode::opening_loc.
Definition ast.h:1520
pm_location_t closing_loc
ArrayPatternNode::closing_loc.
Definition ast.h:1530
AssocNode.
Definition ast.h:1545
PM_NODE_ALIGNAS struct pm_node * value
AssocNode::value.
Definition ast.h:1576
PM_NODE_ALIGNAS struct pm_node * key
AssocNode::key.
Definition ast.h:1563
AssocSplatNode.
Definition ast.h:1601
BeginNode.
Definition ast.h:1668
PM_NODE_ALIGNAS struct pm_else_node * else_clause
BeginNode::else_clause.
Definition ast.h:1710
PM_NODE_ALIGNAS struct pm_ensure_node * ensure_clause
BeginNode::ensure_clause.
Definition ast.h:1720
PM_NODE_ALIGNAS struct pm_statements_node * statements
BeginNode::statements.
Definition ast.h:1690
PM_NODE_ALIGNAS struct pm_rescue_node * rescue_clause
BeginNode::rescue_clause.
Definition ast.h:1700
pm_node_t base
The embedded base node.
Definition ast.h:1670
This struct represents a set of binding powers used for a given token.
Definition prism.c:12630
bool binary
Whether or not this token can be used as a binary operator.
Definition prism.c:12638
pm_binding_power_t left
The left binding power.
Definition prism.c:12632
bool nonassoc
Whether or not this token can be used as non-associative binary operator.
Definition prism.c:12644
pm_binding_power_t right
The right binding power.
Definition prism.c:12635
BlockLocalVariableNode.
Definition ast.h:1785
BlockNode.
Definition ast.h:1812
BlockParametersNode.
Definition ast.h:1940
CallNode.
Definition ast.h:2164
pm_location_t opening_loc
CallNode::opening_loc.
Definition ast.h:2225
pm_location_t closing_loc
CallNode::closing_loc.
Definition ast.h:2245
pm_constant_id_t name
CallNode::name.
Definition ast.h:2205
PM_NODE_ALIGNAS struct pm_arguments_node * arguments
CallNode::arguments.
Definition ast.h:2235
pm_location_t equal_loc
CallNode::equal_loc.
Definition ast.h:2258
pm_location_t call_operator_loc
CallNode::call_operator_loc.
Definition ast.h:2195
pm_location_t message_loc
CallNode::message_loc.
Definition ast.h:2215
PM_NODE_ALIGNAS struct pm_node * block
CallNode::block.
Definition ast.h:2268
PM_NODE_ALIGNAS struct pm_node * receiver
CallNode::receiver.
Definition ast.h:2182
CaseMatchNode.
Definition ast.h:2599
struct pm_node_list conditions
CaseMatchNode::conditions.
Definition ast.h:2621
PM_NODE_ALIGNAS struct pm_else_node * else_clause
CaseMatchNode::else_clause.
Definition ast.h:2631
CaseNode.
Definition ast.h:2668
PM_NODE_ALIGNAS struct pm_else_node * else_clause
CaseNode::else_clause.
Definition ast.h:2700
struct pm_node_list conditions
CaseNode::conditions.
Definition ast.h:2690
ClassVariableReadNode.
Definition ast.h:2957
ClassVariableTargetNode.
Definition ast.h:2985
ClassVariableWriteNode.
Definition ast.h:3007
A list of constant IDs.
ConstantPathNode.
Definition ast.h:3216
ConstantPathTargetNode.
Definition ast.h:3351
ConstantReadNode.
Definition ast.h:3444
ConstantTargetNode.
Definition ast.h:3472
ConstantWriteNode.
Definition ast.h:3494
DefNode.
Definition ast.h:3556
pm_location_t equal_loc
DefNode::equal_loc.
Definition ast.h:3613
PM_NODE_ALIGNAS struct pm_node * body
DefNode::body.
Definition ast.h:3583
ElseNode.
Definition ast.h:3670
PM_NODE_ALIGNAS struct pm_statements_node * statements
ElseNode::statements.
Definition ast.h:3682
EnsureNode.
Definition ast.h:3765
PM_NODE_ALIGNAS struct pm_statements_node * statements
EnsureNode::statements.
Definition ast.h:3777
FindPatternNode.
Definition ast.h:3844
pm_location_t opening_loc
FindPatternNode::opening_loc.
Definition ast.h:3908
PM_NODE_ALIGNAS struct pm_node * constant
FindPatternNode::constant.
Definition ast.h:3856
pm_location_t closing_loc
FindPatternNode::closing_loc.
Definition ast.h:3921
FlipFlopNode.
Definition ast.h:3939
FloatNode.
Definition ast.h:3971
double value
FloatNode::value.
Definition ast.h:3980
pm_node_t base
The embedded base node.
Definition ast.h:3973
ForwardingParameterNode.
Definition ast.h:4104
GlobalVariableReadNode.
Definition ast.h:4277
GlobalVariableTargetNode.
Definition ast.h:4305
GlobalVariableWriteNode.
Definition ast.h:4327
HashNode.
Definition ast.h:4388
struct pm_node_list elements
HashNode::elements.
Definition ast.h:4413
HashPatternNode.
Definition ast.h:4447
PM_NODE_ALIGNAS struct pm_node * constant
HashPatternNode::constant.
Definition ast.h:4462
pm_location_t opening_loc
HashPatternNode::opening_loc.
Definition ast.h:4501
pm_location_t closing_loc
HashPatternNode::closing_loc.
Definition ast.h:4514
IfNode.
Definition ast.h:4535
PM_NODE_ALIGNAS struct pm_statements_node * statements
IfNode::statements.
Definition ast.h:4594
PM_NODE_ALIGNAS struct pm_node * subsequent
IfNode::subsequent.
Definition ast.h:4613
ImaginaryNode.
Definition ast.h:4640
InNode.
Definition ast.h:4716
PM_NODE_ALIGNAS struct pm_statements_node * statements
InNode::statements.
Definition ast.h:4728
InstanceVariableReadNode.
Definition ast.h:5119
InstanceVariableTargetNode.
Definition ast.h:5147
InstanceVariableWriteNode.
Definition ast.h:5169
IntegerNode.
Definition ast.h:5236
pm_integer_t value
IntegerNode::value.
Definition ast.h:5245
pm_node_t base
The embedded base node.
Definition ast.h:5238
bool negative
Whether or not the integer is negative.
Definition integer.h:38
InterpolatedMatchLastLineNode.
Definition ast.h:5273
InterpolatedRegularExpressionNode.
Definition ast.h:5318
InterpolatedStringNode.
Definition ast.h:5354
pm_node_t base
The embedded base node.
Definition ast.h:5356
pm_location_t opening_loc
InterpolatedStringNode::opening_loc.
Definition ast.h:5361
InterpolatedSymbolNode.
Definition ast.h:5386
InterpolatedXStringNode.
Definition ast.h:5418
pm_location_t opening_loc
InterpolatedXStringNode::opening_loc.
Definition ast.h:5425
pm_node_t base
The embedded base node.
Definition ast.h:5420
struct pm_node_list parts
InterpolatedXStringNode::parts.
Definition ast.h:5430
KeywordHashNode.
Definition ast.h:5487
int32_t line
The line number.
uint32_t * offsets
The list of offsets.
size_t size
The number of offsets in the list.
LocalVariableReadNode.
Definition ast.h:5723
uint32_t depth
LocalVariableReadNode::depth.
Definition ast.h:5753
pm_constant_id_t name
LocalVariableReadNode::name.
Definition ast.h:5740
LocalVariableTargetNode.
Definition ast.h:5771
LocalVariableWriteNode.
Definition ast.h:5798
uint32_t depth
LocalVariableWriteNode::depth.
Definition ast.h:5824
pm_constant_id_t name
LocalVariableWriteNode::name.
Definition ast.h:5811
This struct represents a slice in the source code, defined by an offset and a length.
Definition ast.h:575
uint32_t start
The offset of the location from the start of the source.
Definition ast.h:577
uint32_t length
The length of the location.
Definition ast.h:580
MatchLastLineNode.
Definition ast.h:5889
struct pm_node_list targets
MatchWriteNode::targets.
Definition ast.h:6056
MultiTargetNode.
Definition ast.h:6123
pm_location_t lparen_loc
MultiTargetNode::lparen_loc.
Definition ast.h:6180
struct pm_node_list lefts
MultiTargetNode::lefts.
Definition ast.h:6140
pm_location_t rparen_loc
MultiTargetNode::rparen_loc.
Definition ast.h:6190
MultiWriteNode.
Definition ast.h:6205
A list of nodes in the source, most often used for lists of children.
Definition ast.h:588
size_t size
The number of nodes in the list.
Definition ast.h:590
struct pm_node ** nodes
The nodes in the list.
Definition ast.h:596
This is the base structure that represents a node in the syntax tree.
Definition ast.h:1086
pm_node_type_t type
This represents the type of the node.
Definition ast.h:1091
pm_location_t location
This is the location of the node in the source.
Definition ast.h:1109
OptionalParameterNode.
Definition ast.h:6499
OrNode.
Definition ast.h:6536
PM_NODE_ALIGNAS struct pm_node * right
OrNode::right.
Definition ast.h:6564
PM_NODE_ALIGNAS struct pm_node * left
OrNode::left.
Definition ast.h:6551
ParametersNode.
Definition ast.h:6590
PM_NODE_ALIGNAS struct pm_node * block
ParametersNode::block.
Definition ast.h:6627
PM_NODE_ALIGNAS struct pm_node * rest
ParametersNode::rest.
Definition ast.h:6607
PM_NODE_ALIGNAS struct pm_node * keyword_rest
ParametersNode::keyword_rest.
Definition ast.h:6622
ParenthesesNode.
Definition ast.h:6645
PM_NODE_ALIGNAS struct pm_node * body
ParenthesesNode::body.
Definition ast.h:6652
RangeNode.
Definition ast.h:6875
PM_NODE_ALIGNAS struct pm_node * right
RangeNode::right.
Definition ast.h:6904
PM_NODE_ALIGNAS struct pm_node * left
RangeNode::left.
Definition ast.h:6890
RationalNode.
Definition ast.h:6932
pm_node_t base
The embedded base node.
Definition ast.h:6934
pm_integer_t numerator
RationalNode::numerator.
Definition ast.h:6943
In order to properly set a regular expression's encoding and to validate the byte sequence for the un...
Definition prism.c:9832
pm_buffer_t regexp_buffer
The buffer holding the regexp source.
Definition prism.c:9837
pm_token_buffer_t base
The embedded base buffer.
Definition prism.c:9834
RegularExpressionNode.
Definition ast.h:6997
RequiredParameterNode.
Definition ast.h:7069
RescueModifierNode.
Definition ast.h:7091
PM_NODE_ALIGNAS struct pm_node * rescue_expression
RescueModifierNode::rescue_expression.
Definition ast.h:7108
RescueNode.
Definition ast.h:7128
PM_NODE_ALIGNAS struct pm_rescue_node * subsequent
RescueNode::subsequent.
Definition ast.h:7165
pm_location_t then_keyword_loc
RescueNode::then_keyword_loc.
Definition ast.h:7155
SplatNode.
Definition ast.h:7418
PM_NODE_ALIGNAS struct pm_node * expression
SplatNode::expression.
Definition ast.h:7430
StatementsNode.
Definition ast.h:7445
struct pm_node_list body
StatementsNode::body.
Definition ast.h:7452
pm_node_t base
The embedded base node.
Definition ast.h:7447
StringNode.
Definition ast.h:7479
pm_node_t base
The embedded base node.
Definition ast.h:7481
pm_string_t unescaped
StringNode::unescaped.
Definition ast.h:7501
pm_location_t content_loc
StringNode::content_loc.
Definition ast.h:7491
pm_location_t closing_loc
StringNode::closing_loc.
Definition ast.h:7496
pm_location_t opening_loc
StringNode::opening_loc.
Definition ast.h:7486
A generic string type that can have various ownership semantics.
Definition stringy.h:18
const uint8_t * source
A pointer to the start of the string.
Definition stringy.h:20
enum pm_string_t::@118 type
The type of the string.
size_t length
The length of the string in bytes of memory.
Definition stringy.h:23
SuperNode.
Definition ast.h:7521
PM_NODE_ALIGNAS struct pm_arguments_node * arguments
SuperNode::arguments.
Definition ast.h:7540
pm_location_t lparen_loc
SuperNode::lparen_loc.
Definition ast.h:7533
PM_NODE_ALIGNAS struct pm_node * block
SuperNode::block.
Definition ast.h:7550
SymbolNode.
Definition ast.h:7573
pm_location_t content_loc
SymbolNode::content_loc.
Definition ast.h:7585
pm_string_t unescaped
SymbolNode::unescaped.
Definition ast.h:7595
When we're lexing certain types (strings, symbols, lists, etc.) we have string content associated wit...
Definition prism.c:9806
pm_buffer_t buffer
The buffer that we're using to keep track of the string content.
Definition prism.c:9811
const uint8_t * cursor
The cursor into the source string that points to how far we have currently copied into the buffer.
Definition prism.c:9817
This struct represents a token in the Ruby source.
Definition ast.h:547
const uint8_t * end
A pointer to the end location of the token in the source.
Definition ast.h:555
const uint8_t * start
A pointer to the start location of the token in the source.
Definition ast.h:552
pm_token_type_t type
The type of the token.
Definition ast.h:549
UndefNode.
Definition ast.h:7627
UnlessNode.
Definition ast.h:7657
PM_NODE_ALIGNAS struct pm_statements_node * statements
UnlessNode::statements.
Definition ast.h:7706
PM_NODE_ALIGNAS struct pm_else_node * else_clause
UnlessNode::else_clause.
Definition ast.h:7716
WhenNode.
Definition ast.h:7791
PM_NODE_ALIGNAS struct pm_statements_node * statements
WhenNode::statements.
Definition ast.h:7813
XStringNode.
Definition ast.h:7880
YieldNode.
Definition ast.h:7917
pm_location_t lparen_loc
YieldNode::lparen_loc.
Definition ast.h:7929
PM_NODE_ALIGNAS struct pm_arguments_node * arguments
YieldNode::arguments.
Definition ast.h:7934
#define PRISM_UNUSED
GCC will warn if you specify a function or parameter that is unused at runtime.
Definition unused.h:13