Ruby 4.1.0dev (2026-10-02 revision 2192dca2007ff501eb33e87c0ae404579524a7eb)
parser.h
1#ifndef PRISM_INTERNAL_PARSER_H
2#define PRISM_INTERNAL_PARSER_H
3
5
6#include "prism/internal/arena.h"
7#include "prism/internal/constant_pool.h"
8#include "prism/internal/encoding.h"
9#include "prism/internal/list.h"
10#include "prism/internal/options.h"
11#include "prism/internal/static_literals.h"
12#include "prism/internal/strpbrk.h"
13
14#include "prism/ast.h"
16#include "prism/parser.h"
17
18#include <stdbool.h>
19#include <stddef.h>
20#include <stdint.h>
21
22/*
23 * This enum provides various bits that represent different kinds of states that
24 * the lexer can track. This is used to determine which kind of token to return
25 * based on the context of the parser.
26 */
27typedef enum {
28 PM_LEX_STATE_BIT_BEG,
29 PM_LEX_STATE_BIT_END,
30 PM_LEX_STATE_BIT_ENDARG,
31 PM_LEX_STATE_BIT_ENDFN,
32 PM_LEX_STATE_BIT_ARG,
33 PM_LEX_STATE_BIT_CMDARG,
34 PM_LEX_STATE_BIT_MID,
35 PM_LEX_STATE_BIT_FNAME,
36 PM_LEX_STATE_BIT_DOT,
37 PM_LEX_STATE_BIT_CLASS,
38 PM_LEX_STATE_BIT_LABEL,
39 PM_LEX_STATE_BIT_LABELED,
40 PM_LEX_STATE_BIT_FITEM
41} pm_lex_state_bit_t;
42
43/*
44 * This enum combines the various bits from the above enum into individual
45 * values that represent the various states of the lexer.
46 */
47typedef enum {
48 PM_LEX_STATE_NONE = 0,
49 PM_LEX_STATE_BEG = (1 << PM_LEX_STATE_BIT_BEG),
50 PM_LEX_STATE_END = (1 << PM_LEX_STATE_BIT_END),
51 PM_LEX_STATE_ENDARG = (1 << PM_LEX_STATE_BIT_ENDARG),
52 PM_LEX_STATE_ENDFN = (1 << PM_LEX_STATE_BIT_ENDFN),
53 PM_LEX_STATE_ARG = (1 << PM_LEX_STATE_BIT_ARG),
54 PM_LEX_STATE_CMDARG = (1 << PM_LEX_STATE_BIT_CMDARG),
55 PM_LEX_STATE_MID = (1 << PM_LEX_STATE_BIT_MID),
56 PM_LEX_STATE_FNAME = (1 << PM_LEX_STATE_BIT_FNAME),
57 PM_LEX_STATE_DOT = (1 << PM_LEX_STATE_BIT_DOT),
58 PM_LEX_STATE_CLASS = (1 << PM_LEX_STATE_BIT_CLASS),
59 PM_LEX_STATE_LABEL = (1 << PM_LEX_STATE_BIT_LABEL),
60 PM_LEX_STATE_LABELED = (1 << PM_LEX_STATE_BIT_LABELED),
61 PM_LEX_STATE_FITEM = (1 << PM_LEX_STATE_BIT_FITEM),
62 PM_LEX_STATE_BEG_ANY = PM_LEX_STATE_BEG | PM_LEX_STATE_MID | PM_LEX_STATE_CLASS,
63 PM_LEX_STATE_ARG_ANY = PM_LEX_STATE_ARG | PM_LEX_STATE_CMDARG,
64 PM_LEX_STATE_END_ANY = PM_LEX_STATE_END | PM_LEX_STATE_ENDARG | PM_LEX_STATE_ENDFN
65} pm_lex_state_t;
66
67/*
68 * The type of quote that a heredoc uses.
69 */
70typedef enum {
71 PM_HEREDOC_QUOTE_NONE,
72 PM_HEREDOC_QUOTE_SINGLE = '\'',
73 PM_HEREDOC_QUOTE_DOUBLE = '"',
74 PM_HEREDOC_QUOTE_BACKTICK = '`',
75} pm_heredoc_quote_t;
76
77/*
78 * The type of indentation that a heredoc uses.
79 */
80typedef enum {
81 PM_HEREDOC_INDENT_NONE,
82 PM_HEREDOC_INDENT_DASH,
83 PM_HEREDOC_INDENT_TILDE,
84} pm_heredoc_indent_t;
85
86/*
87 * All of the information necessary to store to lexing a heredoc.
88 */
89typedef struct {
90 /* A pointer to the start of the heredoc identifier. */
91 const uint8_t *ident_start;
92
93 /* The length of the heredoc identifier. */
94 size_t ident_length;
95
96 /* The type of quote that the heredoc uses. */
97 pm_heredoc_quote_t quote;
98
99 /* The type of indentation that the heredoc uses. */
100 pm_heredoc_indent_t indent;
102
103/*
104 * When lexing Ruby source, the lexer has a small amount of state to tell which
105 * kind of token it is currently lexing. For example, when we find the start of
106 * a string, the first token that we return is a TOKEN_STRING_BEGIN token. After
107 * that the lexer is now in the PM_LEX_STRING mode, and will return tokens that
108 * are found as part of a string.
109 */
110typedef struct pm_lex_mode {
111 /* The type of this lex mode. */
112 enum {
113 /* This state is used when any given token is being lexed. */
114 PM_LEX_DEFAULT,
115
116 /*
117 * This state is used when we're lexing as normal but inside an embedded
118 * expression of a string.
119 */
120 PM_LEX_EMBEXPR,
121
122 /*
123 * This state is used when we're lexing a variable that is embedded
124 * directly inside of a string with the # shorthand.
125 */
126 PM_LEX_EMBVAR,
127
128 /* This state is used when you are inside the content of a heredoc. */
129 PM_LEX_HEREDOC,
130
131 /*
132 * This state is used when we are lexing a list of tokens, as in a %w
133 * word list literal or a %i symbol list literal.
134 */
135 PM_LEX_LIST,
136
137 /*
138 * This state is used when a regular expression has been begun and we
139 * are looking for the terminator.
140 */
141 PM_LEX_REGEXP,
142
143 /*
144 * This state is used when we are lexing a string or a string-like
145 * token, as in string content with either quote or an xstring.
146 */
147 PM_LEX_STRING
148 } mode;
149
150 /* The data associated with this type of lex mode. */
151 union {
152 struct {
153 /* This keeps track of the nesting level of the list. */
154 size_t nesting;
155
156 /* Whether or not interpolation is allowed in this list. */
157 bool interpolation;
158
159 /*
160 * Whether any token has been emitted from this list. A word
161 * separator delimits the opener from the first word, so one
162 * without source characters is emitted when the list does not
163 * start with whitespace.
164 */
165 bool started;
166
167 /*
168 * Whether the previously emitted token was a word separator. A
169 * word separator delimits the last word from the terminator, so
170 * one without source characters is emitted when the list does
171 * not end with whitespace.
172 */
173 bool separated;
174
175 /*
176 * When lexing a list, it takes into account balancing the
177 * terminator if the terminator is one of (), [], {}, or <>.
178 */
179 uint8_t incrementor;
180
181 /* This is the terminator of the list literal. */
182 uint8_t terminator;
183
184 /*
185 * This is the character set that should be used to delimit the
186 * tokens within the list.
187 */
188 uint8_t breakpoints[PM_STRPBRK_CACHE_SIZE];
189 } list;
190
191 struct {
192 /*
193 * This keeps track of the nesting level of the regular expression.
194 */
195 size_t nesting;
196
197 /*
198 * When lexing a regular expression, it takes into account balancing
199 * the terminator if the terminator is one of (), [], {}, or <>.
200 */
201 uint8_t incrementor;
202
203 /* This is the terminator of the regular expression. */
204 uint8_t terminator;
205
206 /*
207 * This is the character set that should be used to delimit the
208 * tokens within the regular expression.
209 */
210 uint8_t breakpoints[PM_STRPBRK_CACHE_SIZE];
211 } regexp;
212
213 struct {
214 /* This keeps track of the nesting level of the string. */
215 size_t nesting;
216
217 /* Whether or not interpolation is allowed in this string. */
218 bool interpolation;
219
220 /*
221 * Whether or not at the end of the string we should allow a :,
222 * which would indicate this was a dynamic symbol instead of a
223 * string.
224 */
225 bool label_allowed;
226
227 /*
228 * When lexing a string, it takes into account balancing the
229 * terminator if the terminator is one of (), [], {}, or <>.
230 */
231 uint8_t incrementor;
232
233 /*
234 * This is the terminator of the string. It is typically either a
235 * single or double quote.
236 */
237 uint8_t terminator;
238
239 /*
240 * This is the character set that should be used to delimit the
241 * tokens within the string.
242 */
243 uint8_t breakpoints[PM_STRPBRK_CACHE_SIZE];
244 } string;
245
246 struct {
247 /*
248 * All of the data necessary to lex a heredoc.
249 */
251
252 /*
253 * This is the pointer to the character where lexing should resume
254 * once the heredoc has been completely processed.
255 */
256 const uint8_t *next_start;
257
258 /*
259 * This is used to track the amount of common whitespace on each
260 * line so that we know how much to dedent each line in the case of
261 * a tilde heredoc.
262 */
263 size_t *common_whitespace;
264
265 /* True if the previous token ended with a line continuation. */
266 bool line_continuation;
267 } heredoc;
268 } as;
269
270 /* The previous lex state so that it knows how to pop. */
271 struct pm_lex_mode *prev;
273
274/*
275 * We pre-allocate a certain number of lex states in order to avoid having to
276 * call malloc too many times while parsing. You really shouldn't need more than
277 * this because you only really nest deeply when doing string interpolation.
278 */
279#define PM_LEX_STACK_SIZE 4
280
281/*
282 * While parsing, we keep track of a stack of contexts. This is helpful for
283 * error recovery so that we can pop back to a previous context when we hit a
284 * token that is understood by a parent context but not by the current context.
285 */
286typedef enum {
287 /* a null context, used for returning a value from a function */
288 PM_CONTEXT_NONE = 0,
289
290 /* a begin statement */
291 PM_CONTEXT_BEGIN,
292
293 /* an ensure statement with an explicit begin */
294 PM_CONTEXT_BEGIN_ENSURE,
295
296 /* a rescue else statement with an explicit begin */
297 PM_CONTEXT_BEGIN_ELSE,
298
299 /* a rescue statement with an explicit begin */
300 PM_CONTEXT_BEGIN_RESCUE,
301
302 /* expressions in block arguments using braces */
303 PM_CONTEXT_BLOCK_BRACES,
304
305 /* expressions in block arguments using do..end */
306 PM_CONTEXT_BLOCK_KEYWORDS,
307
308 /* an ensure statement within a do..end block */
309 PM_CONTEXT_BLOCK_ENSURE,
310
311 /* a rescue else statement within a do..end block */
312 PM_CONTEXT_BLOCK_ELSE,
313
314 /* expressions in block parameters `foo do |...| end ` */
315 PM_CONTEXT_BLOCK_PARAMETERS,
316
317 /* a rescue statement within a do..end block */
318 PM_CONTEXT_BLOCK_RESCUE,
319
320 /* a case when statements */
321 PM_CONTEXT_CASE_WHEN,
322
323 /* a case in statements */
324 PM_CONTEXT_CASE_IN,
325
326 /* a class declaration */
327 PM_CONTEXT_CLASS,
328
329 /* an ensure statement within a class statement */
330 PM_CONTEXT_CLASS_ENSURE,
331
332 /* a rescue else statement within a class statement */
333 PM_CONTEXT_CLASS_ELSE,
334
335 /* a rescue statement within a class statement */
336 PM_CONTEXT_CLASS_RESCUE,
337
338 /* a method definition */
339 PM_CONTEXT_DEF,
340
341 /* an ensure statement within a method definition */
342 PM_CONTEXT_DEF_ENSURE,
343
344 /* a rescue else statement within a method definition */
345 PM_CONTEXT_DEF_ELSE,
346
347 /* a rescue statement within a method definition */
348 PM_CONTEXT_DEF_RESCUE,
349
350 /* a method definition's parameters */
351 PM_CONTEXT_DEF_PARAMS,
352
353 /* a defined? expression */
354 PM_CONTEXT_DEFINED,
355
356 /* a method definition's default parameter */
357 PM_CONTEXT_DEFAULT_PARAMS,
358
359 /* an else clause */
360 PM_CONTEXT_ELSE,
361
362 /* an elsif clause */
363 PM_CONTEXT_ELSIF,
364
365 /* an interpolated expression */
366 PM_CONTEXT_EMBEXPR,
367
368 /* a for loop */
369 PM_CONTEXT_FOR,
370
371 /* a for loop's index */
372 PM_CONTEXT_FOR_INDEX,
373
374 /* an if statement */
375 PM_CONTEXT_IF,
376
377 /* a lambda expression with braces */
378 PM_CONTEXT_LAMBDA_BRACES,
379
380 /* a lambda expression with do..end */
381 PM_CONTEXT_LAMBDA_DO_END,
382
383 /* an ensure statement within a lambda expression */
384 PM_CONTEXT_LAMBDA_ENSURE,
385
386 /* a rescue else statement within a lambda expression */
387 PM_CONTEXT_LAMBDA_ELSE,
388
389 /* a rescue statement within a lambda expression */
390 PM_CONTEXT_LAMBDA_RESCUE,
391
392 /* the predicate clause of a loop statement */
393 PM_CONTEXT_LOOP_PREDICATE,
394
395 /* the top level context */
396 PM_CONTEXT_MAIN,
397
398 /* a module declaration */
399 PM_CONTEXT_MODULE,
400
401 /* an ensure statement within a module statement */
402 PM_CONTEXT_MODULE_ENSURE,
403
404 /* a rescue else statement within a module statement */
405 PM_CONTEXT_MODULE_ELSE,
406
407 /* a rescue statement within a module statement */
408 PM_CONTEXT_MODULE_RESCUE,
409
410 /* a multiple target expression */
411 PM_CONTEXT_MULTI_TARGET,
412
413 /* a parenthesized expression */
414 PM_CONTEXT_PARENS,
415
416 /* an END block */
417 PM_CONTEXT_POSTEXE,
418
419 /* a predicate inside an if/elsif/unless statement */
420 PM_CONTEXT_PREDICATE,
421
422 /* a BEGIN block */
423 PM_CONTEXT_PREEXE,
424
425 /* a modifier rescue clause */
426 PM_CONTEXT_RESCUE_MODIFIER,
427
428 /* a singleton class definition */
429 PM_CONTEXT_SCLASS,
430
431 /* an ensure statement with a singleton class */
432 PM_CONTEXT_SCLASS_ENSURE,
433
434 /* a rescue else statement with a singleton class */
435 PM_CONTEXT_SCLASS_ELSE,
436
437 /* a rescue statement with a singleton class */
438 PM_CONTEXT_SCLASS_RESCUE,
439
440 /* a ternary expression */
441 PM_CONTEXT_TERNARY,
442
443 /* an unless statement */
444 PM_CONTEXT_UNLESS,
445
446 /* an until statement */
447 PM_CONTEXT_UNTIL,
448
449 /* a while statement */
450 PM_CONTEXT_WHILE,
451
452 /* the number of contexts, which is not itself a context */
453 PM_CONTEXT_MAXIMUM,
454} pm_context_t;
455
456/* This is a node in a linked list of contexts. */
457typedef struct pm_context_node {
458 /* The context that this node represents. */
459 pm_context_t context;
460
461 /* A pointer to the previous context in the linked list. */
462 struct pm_context_node *prev;
463
464 /* One bit set per context in this list, including one for this node. */
465 uint64_t mask;
467
468/* The type of shareable constant value that can be set. */
469typedef uint8_t pm_shareable_constant_value_t;
470static const pm_shareable_constant_value_t PM_SCOPE_SHAREABLE_CONSTANT_NONE = 0x0;
471static const pm_shareable_constant_value_t PM_SCOPE_SHAREABLE_CONSTANT_LITERAL = PM_SHAREABLE_CONSTANT_NODE_FLAGS_LITERAL;
472static const pm_shareable_constant_value_t PM_SCOPE_SHAREABLE_CONSTANT_EXPERIMENTAL_EVERYTHING = PM_SHAREABLE_CONSTANT_NODE_FLAGS_EXPERIMENTAL_EVERYTHING;
473static const pm_shareable_constant_value_t PM_SCOPE_SHAREABLE_CONSTANT_EXPERIMENTAL_COPY = PM_SHAREABLE_CONSTANT_NODE_FLAGS_EXPERIMENTAL_COPY;
474
475/*
476 * This tracks an individual local variable in a certain lexical context, as
477 * well as the number of times is it read.
478 */
479typedef struct {
480 /* The name of the local variable. */
481 pm_constant_id_t name;
482
483 /* The location of the local variable in the source. */
484 pm_location_t location;
485
486 /* The index of the local variable in the local table. */
487 uint32_t index;
488
489 /* The number of times the local variable is read. */
490 uint32_t reads;
491
492 /* The hash of the local variable. */
493 uint32_t hash;
494} pm_local_t;
495
496/*
497 * This is a set of local variables in a certain lexical context (method, class,
498 * module, etc.). We need to track how many times these variables are read in
499 * order to warn if they only get written.
500 */
501typedef struct pm_locals {
502 /* The number of local variables in the set. */
503 uint32_t size;
504
505 /* The capacity of the local variables set. */
506 uint32_t capacity;
507
508 /*
509 * A bloom filter over constant IDs stored in this set. Used to quickly
510 * reject lookups for names that are definitely not present, avoiding the
511 * cost of a linear scan or hash probe.
512 */
513 uint32_t bloom;
514
515 /* The nullable allocated memory for the local variables in the set. */
516 pm_local_t *locals;
518
519/* The flags about scope parameters that can be set. */
520typedef uint8_t pm_scope_parameters_t;
521static const pm_scope_parameters_t PM_SCOPE_PARAMETERS_NONE = 0x0;
522static const pm_scope_parameters_t PM_SCOPE_PARAMETERS_FORWARDING_POSITIONALS = 0x1;
523static const pm_scope_parameters_t PM_SCOPE_PARAMETERS_FORWARDING_KEYWORDS = 0x2;
524static const pm_scope_parameters_t PM_SCOPE_PARAMETERS_FORWARDING_BLOCK = 0x4;
525static const pm_scope_parameters_t PM_SCOPE_PARAMETERS_FORWARDING_ALL = 0x8;
526static const pm_scope_parameters_t PM_SCOPE_PARAMETERS_IMPLICIT_DISALLOWED = 0x10;
527static const pm_scope_parameters_t PM_SCOPE_PARAMETERS_NUMBERED_INNER = 0x20;
528static const pm_scope_parameters_t PM_SCOPE_PARAMETERS_NUMBERED_FOUND = 0x40;
529
530/*
531 * This struct represents a node in a linked list of scopes. Some scopes can see
532 * into their parent scopes, while others cannot.
533 */
534typedef struct pm_scope {
535 /* A pointer to the previous scope in the linked list. */
536 struct pm_scope *previous;
537
538 /* The IDs of the locals in the given scope. */
539 pm_locals_t locals;
540
541 /*
542 * This is a list of the implicit parameters contained within the block.
543 * These will be processed after the block is parsed to determine the kind
544 * of parameters node that should be used and to check if any errors need to
545 * be added.
546 */
547 pm_node_list_t implicit_parameters;
548
549 /*
550 * This is a bitfield that indicates the parameters that are being used in
551 * this scope. It is a combination of the PM_SCOPE_PARAMETERS_* constants.
552 * There are three different kinds of parameters that can be used in a
553 * scope:
554 *
555 * - Ordinary parameters (e.g., def foo(bar); end)
556 * - Numbered parameters (e.g., def foo; _1; end)
557 * - The it parameter (e.g., def foo; it; end)
558 *
559 * If ordinary parameters are being used, then certain parameters can be
560 * forwarded to another method/structure. Those are indicated by four
561 * additional bits in the params field. For example, some combinations of:
562 *
563 * - def foo(*); end
564 * - def foo(**); end
565 * - def foo(&); end
566 * - def foo(...); end
567 */
568 pm_scope_parameters_t parameters;
569
570 /*
571 * The current state of constant shareability for this scope. This is
572 * changed by magic shareable_constant_value comments.
573 */
574 pm_shareable_constant_value_t shareable_constant;
575
576 /*
577 * A boolean indicating whether or not this scope can see into its parent.
578 * If closed is true, then the scope cannot see into its parent.
579 */
580 bool closed;
581} pm_scope_t;
582
583/*
584 * A struct that represents a stack of boolean values.
585 */
586typedef uint32_t pm_state_stack_t;
587
588/*
589 * This struct represents the overall parser. It contains a reference to the
590 * source file, as well as pointers that indicate where in the source it's
591 * currently parsing. It also contains the most recent and current token that
592 * it's considering.
593 */
595 /* The arena used for all AST-lifetime allocations. Caller-owned. */
596 pm_arena_t *arena;
597
598 /* The arena used for parser metadata (comments, diagnostics, etc.). */
599 pm_arena_t metadata_arena;
600
601 /*
602 * The next node identifier that will be assigned. This is a unique
603 * identifier used to track nodes such that the syntax tree can be dropped
604 * but the node can be found through another parse.
605 */
606 uint32_t node_id;
607
608 /*
609 * A single-entry cache for pm_parser_constant_id_raw. Avoids redundant
610 * constant pool lookups when the same token is resolved multiple times
611 * (e.g., once during lexing for local variable detection, and again
612 * during parsing for node creation).
613 */
614 struct {
615 const uint8_t *start;
616 const uint8_t *end;
618 } constant_cache;
619
620 /* The current state of the lexer. */
621 pm_lex_state_t lex_state;
622
623 /* Tracks the current nesting of (), [], and {}. */
624 int enclosure_nesting;
625
626 /*
627 * Used to temporarily track the nesting of enclosures to determine if a {
628 * is the beginning of a lambda following the parameters of a lambda.
629 */
630 int lambda_enclosure_nesting;
631
632 /*
633 * Used to track the nesting of braces to ensure we get the correct value
634 * when we are interpolating blocks with braces.
635 */
636 int brace_nesting;
637
638 /*
639 * The stack used to determine if a do keyword belongs to the predicate of a
640 * while, until, or for loop.
641 */
642 pm_state_stack_t do_loop_stack;
643
644 /*
645 * The stack used to determine if a do keyword belongs to the beginning of a
646 * block.
647 */
648 pm_state_stack_t accepts_block_stack;
649
650 /* A stack of lex modes. */
651 struct {
652 /* The current mode of the lexer. */
653 pm_lex_mode_t *current;
654
655 /* The stack of lexer modes. */
656 pm_lex_mode_t stack[PM_LEX_STACK_SIZE];
657
658 /* The current index into the lexer mode stack. */
659 size_t index;
660 } lex_modes;
661
662 /* The pointer to the start of the source. */
663 const uint8_t *start;
664
665 /* The pointer to the end of the source. */
666 const uint8_t *end;
667
668 /* The previous token we were considering. */
669 pm_token_t previous;
670
671 /* The current token we're considering. */
672 pm_token_t current;
673
674 /*
675 * This is a special field set on the parser when we need the parser to jump
676 * to a specific location when lexing the next token, as opposed to just
677 * using the end of the previous token. Normally this is NULL.
678 */
679 const uint8_t *next_start;
680
681 /*
682 * This field indicates the end of a heredoc whose identifier was found on
683 * the current line. If another heredoc is found on the same line, then this
684 * will be moved forward to the end of that heredoc. If no heredocs are
685 * found on a line then this is NULL.
686 */
687 const uint8_t *heredoc_end;
688
689 /* The list of comments that have been found while parsing. */
690 pm_list_t comment_list;
691
692 /* The list of magic comments that have been found while parsing. */
693 pm_list_t magic_comment_list;
694
695 /*
696 * An optional location that represents the location of the __END__ marker
697 * and the rest of the content of the file. This content is loaded into the
698 * DATA constant when the file being parsed is the main file being executed.
699 */
700 pm_location_t data_loc;
701
702 /* The list of warnings that have been found while parsing. */
703 pm_list_t warning_list;
704
705 /* The list of errors that have been found while parsing. */
706 pm_list_t error_list;
707
708 /* The current local scope. */
709 pm_scope_t *current_scope;
710
711 /* The current parsing context. */
712 pm_context_node_t *current_context;
713
714 /*
715 * The encoding functions for the current file is attached to the parser as
716 * it's parsing so that it can change with a magic comment.
717 */
718 const pm_encoding_t *encoding;
719
720 /*
721 * When the encoding that is being used to parse the source is changed by
722 * prism, we provide the ability here to call out to a user-defined
723 * function.
724 */
725 pm_encoding_changed_callback_t encoding_changed_callback;
726
727 /*
728 * This pointer indicates where a comment must start if it is to be
729 * considered an encoding comment.
730 */
731 const uint8_t *encoding_comment_start;
732
733 /*
734 * When you are lexing through a file, the lexer needs all of the information
735 * that the parser additionally provides (for example, the local table). So if
736 * you want to properly lex Ruby, you need to actually lex it in the context of
737 * the parser. In order to provide this functionality, we optionally allow a
738 * struct to be attached to the parser that calls back out to a user-provided
739 * callback when each token is lexed.
740 */
741 struct {
742 /*
743 * This is the callback that is called when a token is lexed. It is
744 * passed the opaque data pointer, the parser, and the token that was
745 * lexed.
746 */
747 pm_lex_callback_t callback;
748
749 /*
750 * This opaque pointer is used to provide whatever information the user
751 * deemed necessary to the callback. In our case we use it to pass the
752 * array that the tokens get appended into.
753 */
754 void *data;
755 } lex_callback;
756
757 /*
758 * This is the path of the file being parsed. We use the filepath when
759 * constructing SourceFileNodes.
760 */
761 pm_string_t filepath;
762
763 /*
764 * This constant pool keeps all of the constants defined throughout the file
765 * so that we can reference them later.
766 */
767 pm_constant_pool_t constant_pool;
768
769 /* This is the list of line offsets in the source file. */
770 pm_line_offset_list_t line_offsets;
771
772 /*
773 * State communicated from the lexer to the parser for integer tokens.
774 */
775 struct {
776 /*
777 * A flag indicating the base of the integer (binary, octal, decimal,
778 * hexadecimal). Set during lexing and read during node creation.
779 */
780 pm_node_flags_t base;
781
782 /*
783 * When lexing a decimal integer that fits in a uint32_t, we compute
784 * the value during lexing to avoid re-scanning the digits during
785 * parsing. If lexed is true, this holds the result and
786 * pm_integer_parse can be skipped.
787 */
788 uint32_t value;
789
790 /* Whether value holds a valid pre-computed integer. */
791 bool lexed;
792 } integer;
793
794 /*
795 * This string is used to pass information from the lexer to the parser. It
796 * is particularly necessary because of escape sequences.
797 */
798 pm_string_t current_string;
799
800 /*
801 * The line number at the start of the parse. This will be used to offset
802 * the line numbers of all of the locations.
803 */
804 int32_t start_line;
805
806 /*
807 * When a string-like expression is being lexed, any byte or escape sequence
808 * that resolves to a value whose top bit is set (i.e., >= 0x80) will
809 * explicitly set the encoding to the same encoding as the source.
810 * Alternatively, if a unicode escape sequence is used (e.g., \\u{80}) that
811 * resolves to a value whose top bit is set, then the encoding will be
812 * explicitly set to UTF-8.
813 *
814 * The _next_ time this happens, if the encoding that is about to become the
815 * explicitly set encoding does not match the previously set explicit
816 * encoding, a mixed encoding error will be emitted.
817 *
818 * When the expression is finished being lexed, the explicit encoding
819 * controls the encoding of the expression. For the most part this means
820 * that the expression will either be encoded in the source encoding or
821 * UTF-8. This holds for all encodings except US-ASCII. If the source is
822 * US-ASCII and an explicit encoding was set that was _not_ UTF-8, then the
823 * expression will be encoded as ASCII-8BIT.
824 *
825 * Note that if the expression is a list, different elements within the same
826 * list can have different encodings, so this will get reset between each
827 * element. Furthermore all of this only applies to lists that support
828 * interpolation, because otherwise escapes that could change the encoding
829 * are ignored.
830 *
831 * At first glance, it may make more sense for this to live on the lexer
832 * mode, but we need it here to communicate back to the parser for character
833 * literals that do not push a new lexer mode.
834 */
835 const pm_encoding_t *explicit_encoding;
836
837 /*
838 * When parsing block exits (e.g., break, next, redo), we need to validate
839 * that they are in correct contexts. For the most part we can do this by
840 * looking at our parent contexts. However, modifier while and until
841 * expressions can change that context to make block exits valid. In these
842 * cases, we need to keep track of the block exits and then validate them
843 * after the expression has been parsed.
844 *
845 * We use a pointer here because we don't want to keep a whole list attached
846 * since this will only be used in the context of begin/end expressions.
847 */
848 pm_node_list_t *current_block_exits;
849
850 /* The version of prism that we should use to parse. */
851 pm_options_version_t version;
852
853 /* The command line flags given from the options. */
854 uint8_t command_line;
855
856 /*
857 * Whether or not we have found a frozen_string_literal magic comment with
858 * a true or false value.
859 * May be:
860 * - PM_OPTIONS_FROZEN_STRING_LITERAL_DISABLED
861 * - PM_OPTIONS_FROZEN_STRING_LITERAL_ENABLED
862 * - PM_OPTIONS_FROZEN_STRING_LITERAL_UNSET
863 */
864 int8_t frozen_string_literal;
865
866 /*
867 * Whether or not we are parsing an eval string. This impacts whether or not
868 * we should evaluate if block exits/yields are valid.
869 */
870 bool parsing_eval;
871
872 /*
873 * Whether or not we are parsing a "partial" script, which is a script that
874 * will be evaluated in the context of another script, so we should not
875 * check jumps (next/break/etc.) for validity.
876 */
877 bool partial_script;
878
879 /* Whether or not we're at the beginning of a command. */
880 bool command_start;
881
882 /*
883 * Whether or not we're currently parsing the body of an endless method
884 * definition. In this context, PM_TOKEN_KEYWORD_DO_BLOCK should not be
885 * consumed by commands (it should bubble up to the outer context).
886 */
887 bool in_endless_def_body;
888
889 /* Whether or not we're currently recovering from a syntax error. */
890 bool recovering;
891
892 /*
893 * Whether or not the source being parsed could become valid if more input
894 * were appended. This is set to false when the parser encounters a token
895 * that is definitively wrong (e.g., a stray `end` or `]`) as opposed to
896 * merely incomplete.
897 */
898 bool continuable;
899
900 /*
901 * This is very specialized behavior for when you want to parse in a context
902 * that does not respect encoding comments. Its main use case is translating
903 * into the whitequark/parser AST which re-encodes source files in UTF-8
904 * before they are parsed and ignores encoding comments.
905 */
906 bool encoding_locked;
907
908 /*
909 * Whether or not the encoding has been changed by a magic comment. We use
910 * this to provide a fast path for the lexer instead of going through the
911 * function pointer.
912 */
913 bool encoding_changed;
914
915 /*
916 * This flag indicates that we are currently parsing a pattern matching
917 * expression and impacts that calculation of newlines.
918 */
919 bool pattern_matching_newlines;
920
921 /* This flag indicates that we are currently parsing a keyword argument. */
922 bool in_keyword_arg;
923
924 /*
925 * Whether or not the parser has seen a token that has semantic meaning
926 * (i.e., a token that is not a comment or whitespace).
927 */
928 bool semantic_token_seen;
929
930 /*
931 * By default, Ruby always warns about mismatched indentation. This can be
932 * toggled with a magic comment.
933 */
934 bool warn_mismatched_indentation;
935
936#if defined(PRISM_HAS_NEON) || defined(PRISM_HAS_SSSE3) || defined(PRISM_HAS_SWAR)
937 /*
938 * Cached lookup tables for pm_strpbrk's SIMD fast path. Avoids rebuilding
939 * the nibble-based tables on every call when the charset hasn't changed
940 * (which is the common case during string/regex/list lexing).
941 */
942 struct {
943 /* The cached charset (null-terminated, max 11 chars + NUL). */
944 uint8_t charset[12];
945
946 /* Nibble-based low lookup table for SIMD matching. */
947 uint8_t low_lut[16];
948
949 /* Nibble-based high lookup table for SIMD matching. */
950 uint8_t high_lut[16];
951
952 /* Scalar fallback table (4 x 64-bit bitmasks covering all ASCII). */
953 uint64_t table[4];
954 } strpbrk_cache;
955#endif
956};
957
958/*
959 * Initialize a parser with the given start and end pointers.
960 */
961void pm_parser_init(pm_arena_t *arena, pm_parser_t *parser, const uint8_t *source, size_t size, const pm_options_t *options);
962
963/*
964 * Free the memory held by the given parser.
965 *
966 * This does not free the `pm_options_t` object that was used to initialize the
967 * parser.
968 */
969void pm_parser_cleanup(pm_parser_t *parser);
970
971#endif
uint32_t pm_constant_id_t
A constant id is a unique identifier for a constant in the constant pool.
A list of byte offsets of newlines in a string.
The parser used to parse Ruby source.
void(* pm_lex_callback_t)(pm_parser_t *parser, pm_token_t *token, void *data)
This is the callback that is called when a token is lexed.
Definition parser.h:55
void(* pm_encoding_changed_callback_t)(pm_parser_t *parser)
When the encoding that is being used to parse the source is changed by prism, we provide the ability ...
Definition parser.h:49
C99 shim for <stdbool.h>
A list of offsets of the start of lines in a string.
This struct represents a slice in the source code, defined by an offset and a length.
Definition ast.h:575
A list of nodes in the source, most often used for lists of children.
Definition ast.h:588
A generic string type that can have various ownership semantics.
Definition stringy.h:18
This struct represents a token in the Ruby source.
Definition ast.h:547