Ruby 4.1.0dev (2026-09-29 revision 5fe7d61b2050ab8e130316fee458320cba1c0cfe)
string.c (5fe7d61b2050ab8e130316fee458320cba1c0cfe)
1/**********************************************************************
2
3 string.c -
4
5 $Author$
6 created at: Mon Aug 9 17:12:58 JST 1993
7
8 Copyright (C) 1993-2007 Yukihiro Matsumoto
9 Copyright (C) 2000 Network Applied Communication Laboratory, Inc.
10 Copyright (C) 2000 Information-technology Promotion Agency, Japan
11
12**********************************************************************/
13
14#include "ruby/internal/config.h"
15
16#include <ctype.h>
17#include <errno.h>
18#include <math.h>
19
20#ifdef HAVE_UNISTD_H
21# include <unistd.h>
22#endif
23
24#include "debug_counter.h"
25#include "encindex.h"
26#include "id.h"
27#include "internal.h"
28#include "internal/array.h"
29#include "internal/bits.h"
30#include "internal/compar.h"
31#include "internal/compilers.h"
32#include "internal/concurrent_set.h"
33#include "internal/encoding.h"
34#include "internal/error.h"
35#include "internal/gc.h"
36#include "internal/hash.h"
37#include "internal/numeric.h"
38#include "internal/object.h"
39#include "internal/proc.h"
40#include "internal/re.h"
41#include "internal/sanitizers.h"
42#include "internal/simd.h"
43#include "internal/string.h"
44#include "internal/transcode.h"
45#include "probes.h"
46#include "ruby/encoding.h"
47#include "ruby/re.h"
48#include "ruby/thread.h"
49#include "ruby/util.h"
50#include "ruby/ractor.h"
51#include "ruby_assert.h"
52#include "shape.h"
53#include "vm_core.h"
54#include "vm_sync.h"
55#include "zjit.h"
57
58#if defined HAVE_CRYPT_R
59# if defined HAVE_CRYPT_H
60# include <crypt.h>
61# endif
62#elif !defined HAVE_CRYPT
63# include "missing/crypt.h"
64# define HAVE_CRYPT_R 1
65#endif
66
67#undef rb_str_new
68#undef rb_usascii_str_new
69#undef rb_utf8_str_new
70#undef rb_enc_str_new
71#undef rb_str_new_cstr
72#undef rb_usascii_str_new_cstr
73#undef rb_utf8_str_new_cstr
74#undef rb_enc_str_new_cstr
75#undef rb_external_str_new_cstr
76#undef rb_locale_str_new_cstr
77#undef rb_str_dup_frozen
78#undef rb_str_buf_new_cstr
79#undef rb_str_buf_cat
80#undef rb_str_buf_cat2
81#undef rb_str_cat2
82#undef rb_str_cat_cstr
83#undef rb_fstring_cstr
84
87
88/* Flags of RString
89 *
90 * 0: STR_SHARED (equal to ELTS_SHARED)
91 * The string is shared. The buffer this string points to is owned by
92 * another string (the shared root).
93 * 1: RSTRING_NOEMBED
94 * The string is not embedded. When a string is embedded, the contents
95 * follow the header. When a string is not embedded, the contents is
96 * on a separately allocated buffer.
97 * 2: STR_CHILLED (will be frozen in a future version)
98 * The string was allocated as a literal in a file without an explicit `frozen_string_literal` comment.
99 * It emits a deprecation warning when mutated for the first time.
100 * 4: STR_PRECOMPUTED_HASH
101 * The string is embedded and has its precomputed hashcode stored
102 * after the terminator.
103 * 5: STR_SHARED_ROOT
104 * Other strings may point to the contents of this string. When this
105 * flag is set, STR_SHARED must not be set.
106 * 6: STR_BORROWED
107 * When RSTRING_NOEMBED is set and klass is 0, this string is unsafe
108 * to be unshared by rb_str_tmp_frozen_release.
109 * 7: STR_TMPLOCK
110 * The pointer to the buffer is passed to a system call such as
111 * read(2). Any modification and realloc is prohibited.
112 * 8-9: ENC_CODERANGE
113 * Stores the coderange of the string.
114 * 10-16: ENCODING
115 * Stores the encoding of the string.
116 * 17: RSTRING_FSTR
117 * The string is a fstring. The string is deduplicated in the fstring
118 * table.
119 * 18: STR_NOFREE
120 * Do not free this string's buffer when the string is reclaimed
121 * by the garbage collector. Used for when the string buffer is a C
122 * string literal.
123 * 19: STR_FAKESTR
124 * The string is not allocated or managed by the garbage collector.
125 * Typically, the string object header (struct RString) is temporarily
126 * allocated on C stack.
127 */
128
129#define RUBY_MAX_CHAR_LEN 16
130#define STR_PRECOMPUTED_HASH FL_USER4
131#define STR_SHARED_ROOT FL_USER5
132#define STR_BORROWED FL_USER6
133#define STR_TMPLOCK FL_USER7
134#define STR_NOFREE FL_USER18
135
136#define STR_SET_NOEMBED(str) do {\
137 FL_SET((str), STR_NOEMBED);\
138 FL_UNSET((str), STR_SHARED | STR_SHARED_ROOT | STR_BORROWED);\
139} while (0)
140#define STR_SET_EMBED(str) FL_UNSET((str), STR_NOEMBED | STR_SHARED | STR_NOFREE)
141
142#define STR_SET_LEN(str, n) do { \
143 RSTRING(str)->len = (n); \
144} while (0)
145
146#define TERM_LEN(str) (rb_str_enc_fastpath(str) ? 1 : rb_enc_mbminlen(rb_enc_from_index(ENCODING_GET(str))))
147#define TERM_FILL(ptr, termlen) do {\
148 char *const term_fill_ptr = (ptr);\
149 const int term_fill_len = (termlen);\
150 *term_fill_ptr = '\0';\
151 if (UNLIKELY(term_fill_len > 1))\
152 memset(term_fill_ptr, 0, term_fill_len);\
153} while (0)
154
155#define RESIZE_CAPA(str,capacity) do {\
156 const int termlen = TERM_LEN(str);\
157 RESIZE_CAPA_TERM(str,capacity,termlen);\
158} while (0)
159#define RESIZE_CAPA_TERM(str,capacity,termlen) do {\
160 if (STR_EMBED_P(str)) {\
161 if (str_embed_capa(str) < capacity + termlen) {\
162 char *const tmp = ALLOC_N(char, (size_t)(capacity) + (termlen));\
163 const long tlen = RSTRING_LEN(str);\
164 memcpy(tmp, RSTRING_PTR(str), str_embed_capa(str));\
165 RSTRING(str)->as.heap.ptr = tmp;\
166 RSTRING(str)->len = tlen;\
167 STR_SET_NOEMBED(str);\
168 RSTRING(str)->as.heap.aux.capa = (capacity);\
169 }\
170 }\
171 else {\
172 RUBY_ASSERT(!FL_TEST((str), STR_SHARED)); \
173 SIZED_REALLOC_N(RSTRING(str)->as.heap.ptr, char, \
174 (size_t)(capacity) + (termlen), STR_HEAP_SIZE(str)); \
175 RSTRING(str)->as.heap.aux.capa = (capacity);\
176 }\
177} while (0)
178
179#define STR_SET_SHARED(str, shared_str) do { \
180 if (!FL_TEST(str, STR_FAKESTR)) { \
181 RUBY_ASSERT(RSTRING_PTR(shared_str) <= RSTRING_PTR(str)); \
182 RUBY_ASSERT(RSTRING_PTR(str) <= RSTRING_PTR(shared_str) + RSTRING_LEN(shared_str)); \
183 RB_OBJ_WRITE((str), &RSTRING(str)->as.heap.aux.shared, (shared_str)); \
184 FL_SET((str), STR_SHARED); \
185 rb_gc_register_pinning_obj(str); \
186 FL_SET((shared_str), STR_SHARED_ROOT); \
187 if (RBASIC_CLASS((shared_str)) == 0) /* for CoW-friendliness */ \
188 FL_SET_RAW((shared_str), STR_BORROWED); \
189 } \
190} while (0)
191
192#define STR_HEAP_PTR(str) (RSTRING(str)->as.heap.ptr)
193#define STR_HEAP_SIZE(str) ((size_t)RSTRING(str)->as.heap.aux.capa + TERM_LEN(str))
194/* TODO: include the terminator size in capa. */
195
196#define STR_ENC_GET(str) get_encoding(str)
197
198static inline bool
199zero_filled(const char *s, int n)
200{
201 for (; n > 0; --n) {
202 if (*s++) return false;
203 }
204 return true;
205}
206
207#if !defined SHARABLE_MIDDLE_SUBSTRING
208# define SHARABLE_MIDDLE_SUBSTRING 0
209#endif
210
211static inline bool
212SHARABLE_SUBSTRING_P(VALUE str, long beg, long len)
213{
214#if SHARABLE_MIDDLE_SUBSTRING
215 return true;
216#else
217 long end = beg + len;
218 long source_len = RSTRING_LEN(str);
219 return end == source_len || zero_filled(RSTRING_PTR(str) + end, TERM_LEN(str));
220#endif
221}
222
223static inline long
224str_embed_capa(VALUE str)
225{
226 return rb_obj_shape_slot_size(str) - offsetof(struct RString, as.embed.ary);
227}
228
229bool
230rb_str_reembeddable_p(VALUE str)
231{
232 return !FL_TEST(str, STR_NOFREE|STR_SHARED_ROOT|STR_SHARED);
233}
234
235/* True when other strings read this string's bytes out of its own slot, so the slot
236 * contents must stay valid for as long as the object does. */
237bool
238rb_str_embedded_shared_root_p(VALUE str)
239{
240 return STR_EMBED_P(str) && FL_TEST(str, STR_SHARED_ROOT);
241}
242
243static inline size_t
244rb_str_embed_size(long capa, long termlen)
245{
246 size_t size = offsetof(struct RString, as.embed.ary) + capa + termlen;
247 if (size < sizeof(struct RString)) size = sizeof(struct RString);
248 return size;
249}
250
251size_t
252rb_str_size_as_embedded(VALUE str)
253{
254 size_t real_size;
255 if (STR_EMBED_P(str)) {
256 size_t capa = RSTRING(str)->len;
257 if (FL_TEST_RAW(str, STR_PRECOMPUTED_HASH)) capa += sizeof(st_index_t);
258
259 real_size = rb_str_embed_size(capa, TERM_LEN(str));
260 }
261 /* if the string is not currently embedded, but it can be embedded, how
262 * much space would it require */
263 else if (rb_str_reembeddable_p(str)) {
264 size_t capa = RSTRING(str)->as.heap.aux.capa;
265 if (FL_TEST_RAW(str, STR_PRECOMPUTED_HASH)) capa += sizeof(st_index_t);
266
267 real_size = rb_str_embed_size(capa, TERM_LEN(str));
268 }
269 else {
270 real_size = sizeof(struct RString);
271 }
272
273 return real_size;
274}
275
276static inline bool
277STR_EMBEDDABLE_P(long len, long termlen)
278{
279 return rb_gc_size_allocatable_p(rb_str_embed_size(len, termlen));
280}
281
282/* Substrings and duplicated strings that need a slot larger than this are shared
283 * instead of copied. Larger slots hold fewer objects per page and trigger GC
284 * more often, which outweighs the copy they save; see [Feature #22186] for the
285 * benchmarks. */
286#define STR_COPY_MAX_EMBED_SIZE 256
287
288static VALUE str_replace_shared_without_enc(VALUE str2, VALUE str);
289static VALUE str_new_frozen(VALUE klass, VALUE orig);
290static VALUE str_new_frozen_buffer(VALUE klass, VALUE orig, int copy_encoding);
291static VALUE str_new_static(VALUE klass, const char *ptr, long len, int encindex);
292static VALUE str_new(VALUE klass, const char *ptr, long len);
293static void str_make_independent_expand(VALUE str, long len, long expand, const int termlen);
294static inline void str_modifiable(VALUE str);
295static VALUE rb_str_downcase(int argc, VALUE *argv, VALUE str);
296static inline VALUE str_alloc_embed(VALUE klass, size_t capa);
297
298static inline void
299str_make_independent(VALUE str)
300{
301 long len = RSTRING_LEN(str);
302 int termlen = TERM_LEN(str);
303 str_make_independent_expand((str), len, 0L, termlen);
304}
305
306static inline int str_dependent_p(VALUE str);
307
308void
309rb_str_make_independent(VALUE str)
310{
311 if (str_dependent_p(str)) {
312 str_make_independent(str);
313 }
314}
315
316void
317rb_str_make_embedded(VALUE str)
318{
319 RUBY_ASSERT(rb_str_reembeddable_p(str));
320 RUBY_ASSERT(!STR_EMBED_P(str));
321
322 int termlen = TERM_LEN(str);
323 char *buf = RSTRING(str)->as.heap.ptr;
324 long old_capa = RSTRING(str)->as.heap.aux.capa + termlen;
325 long len = RSTRING(str)->len;
326
327 STR_SET_EMBED(str);
328 STR_SET_LEN(str, len);
329
330 if (len > 0) {
331 memcpy(RSTRING_PTR(str), buf, len);
332 SIZED_FREE_N(buf, old_capa);
333 }
334
335 TERM_FILL(RSTRING(str)->as.embed.ary + len, termlen);
336}
337
338void
339rb_debug_rstring_null_ptr(const char *func)
340{
341 fprintf(stderr, "%s is returning NULL!! "
342 "SIGSEGV is highly expected to follow immediately.\n"
343 "If you could reproduce, attach your debugger here, "
344 "and look at the passed string.\n",
345 func);
346}
347
348/* symbols for [up|down|swap]case/capitalize options */
349static VALUE sym_ascii, sym_turkic, sym_lithuanian, sym_fold;
350
351static rb_encoding *
352get_encoding(VALUE str)
353{
354 return rb_enc_from_index(ENCODING_GET(str));
355}
356
357static void
358mustnot_broken(VALUE str)
359{
360 if (is_broken_string(str)) {
361 rb_raise(rb_eArgError, "invalid byte sequence in %s", rb_enc_name(STR_ENC_GET(str)));
362 }
363}
364
365static void
366mustnot_wchar(VALUE str)
367{
368 rb_encoding *enc = STR_ENC_GET(str);
369 if (rb_enc_mbminlen(enc) > 1) {
370 rb_raise(rb_eArgError, "wide char encoding: %s", rb_enc_name(enc));
371 }
372}
373
374static VALUE register_fstring(VALUE str, bool copy, bool force_precompute_hash);
375
376#if SIZEOF_LONG == SIZEOF_VOIDP
377#define PRECOMPUTED_FAKESTR_HASH 1
378#else
379#endif
380
381static inline bool
382BARE_STRING_P(VALUE str)
383{
384 return RBASIC_CLASS(str) == rb_cString && !rb_obj_shape_has_ivars(str);
385}
386
387static inline st_index_t
388str_do_hash(VALUE str)
389{
390 st_index_t h = rb_memhash((const void *)RSTRING_PTR(str), RSTRING_LEN(str));
391 int e = RSTRING_LEN(str) ? ENCODING_GET(str) : 0;
392 if (e && !is_ascii_string(str)) {
393 h = rb_hash_end(rb_hash_uint32(h, (uint32_t)e));
394 }
395 return h;
396}
397
398static VALUE
399str_store_precomputed_hash(VALUE str, st_index_t hash)
400{
401 RUBY_ASSERT(!FL_TEST_RAW(str, STR_PRECOMPUTED_HASH));
402 RUBY_ASSERT(STR_EMBED_P(str));
403
404#if RUBY_DEBUG
405 size_t used_bytes = (RSTRING_LEN(str) + TERM_LEN(str));
406 size_t free_bytes = str_embed_capa(str) - used_bytes;
407 RUBY_ASSERT(free_bytes >= sizeof(st_index_t));
408#endif
409
410 memcpy(RSTRING_END(str) + TERM_LEN(str), &hash, sizeof(hash));
411
412 FL_SET(str, STR_PRECOMPUTED_HASH);
413
414 return str;
415}
416
417VALUE
418rb_fstring(VALUE str)
419{
420 VALUE fstr;
421 int bare;
422
423 Check_Type(str, T_STRING);
424
425 if (FL_TEST(str, RSTRING_FSTR))
426 return str;
427
428 bare = BARE_STRING_P(str);
429 if (!bare) {
430 if (STR_EMBED_P(str)) {
431 OBJ_FREEZE(str);
432 return str;
433 }
434
435 if (FL_TEST_RAW(str, STR_SHARED_ROOT | STR_SHARED) == STR_SHARED_ROOT) {
437 return str;
438 }
439 }
440
441 if (!FL_TEST_RAW(str, FL_FREEZE | STR_NOFREE | STR_CHILLED))
442 rb_str_resize(str, RSTRING_LEN(str));
443
444 fstr = register_fstring(str, false, false);
445
446 if (!bare) {
447 str_replace_shared_without_enc(str, fstr);
448 OBJ_FREEZE(str);
449 return str;
450 }
451 return fstr;
452}
453
454static VALUE fstring_table_obj;
455
456static VALUE
457fstring_concurrent_set_hash(VALUE str)
458{
459#ifdef PRECOMPUTED_FAKESTR_HASH
460 st_index_t h;
461 if (FL_TEST_RAW(str, STR_FAKESTR)) {
462 // register_fstring precomputes the hash and stores it in capa for fake strings
463 h = (st_index_t)RSTRING(str)->as.heap.aux.capa;
464 }
465 else {
466 h = rb_str_hash(str);
467 }
468 // rb_str_hash doesn't include the encoding for ascii only strings, so
469 // we add it to avoid common collisions between `:sym.name` (ASCII) and `"sym"` (UTF-8)
470 return (VALUE)rb_hash_end(rb_hash_uint32(h, (uint32_t)ENCODING_GET_INLINED(str)));
471#else
472 return (VALUE)rb_str_hash(str);
473#endif
474}
475
476static bool
477fstring_concurrent_set_cmp(VALUE a, VALUE b)
478{
479 long alen, blen;
480 const char *aptr, *bptr;
481
484
485 RSTRING_GETMEM(a, aptr, alen);
486 RSTRING_GETMEM(b, bptr, blen);
487 return (alen == blen &&
488 ENCODING_GET(a) == ENCODING_GET(b) &&
489 memcmp(aptr, bptr, alen) == 0);
490}
491
493 bool copy;
494 bool force_precompute_hash;
495};
496
497static VALUE
498fstring_concurrent_set_create(VALUE str, void *data)
499{
500 struct fstr_create_arg *arg = data;
501
502 // Unless the string is empty or binary, its coderange has been precomputed.
503 int coderange = ENC_CODERANGE(str);
504
505 if (FL_TEST_RAW(str, STR_FAKESTR)) {
506 if (arg->copy) {
507 VALUE new_str;
508 long len = RSTRING_LEN(str);
509 long capa = len + sizeof(st_index_t);
510 int term_len = TERM_LEN(str);
511
512 if (arg->force_precompute_hash && STR_EMBEDDABLE_P(capa, term_len)) {
513 new_str = str_alloc_embed(rb_cString, capa + term_len);
514 memcpy(RSTRING_PTR(new_str), RSTRING_PTR(str), len);
515 STR_SET_LEN(new_str, RSTRING_LEN(str));
516 TERM_FILL(RSTRING_END(new_str), TERM_LEN(str));
517 rb_enc_copy(new_str, str);
518 str_store_precomputed_hash(new_str, str_do_hash(str));
519 }
520 else {
521 new_str = str_new(rb_cString, RSTRING(str)->as.heap.ptr, RSTRING(str)->len);
522 rb_enc_copy(new_str, str);
523#ifdef PRECOMPUTED_FAKESTR_HASH
524 if (rb_str_capacity(new_str) >= RSTRING_LEN(str) + term_len + sizeof(st_index_t)) {
525 str_store_precomputed_hash(new_str, (st_index_t)RSTRING(str)->as.heap.aux.capa);
526 }
527#endif
528 }
529 str = new_str;
530 }
531 else {
532 str = str_new_static(rb_cString, RSTRING(str)->as.heap.ptr,
533 RSTRING(str)->len,
534 ENCODING_GET(str));
535 }
536 OBJ_FREEZE(str);
537 }
538 else {
539 if (!OBJ_FROZEN(str) || CHILLED_STRING_P(str)) {
540 str = str_new_frozen(rb_cString, str);
541 }
542 if (STR_SHARED_P(str)) { /* str should not be shared */
543 /* shared substring */
544 str_make_independent(str);
546 }
547 if (!BARE_STRING_P(str)) {
548 str = str_new_frozen(rb_cString, str);
549 }
550 }
551
552 ENC_CODERANGE_SET(str, coderange);
553 RBASIC(str)->flags |= RSTRING_FSTR;
554 if (!RB_OBJ_SHAREABLE_P(str)) {
556 }
557 RUBY_ASSERT((rb_gc_verify_shareable(str), 1));
560 RUBY_ASSERT(!FL_TEST_RAW(str, STR_FAKESTR));
561 RUBY_ASSERT(!rb_obj_shape_has_ivars(str));
563 RUBY_ASSERT(!rb_objspace_garbage_object_p(str));
564
565 return str;
566}
567
568static const struct rb_concurrent_set_funcs fstring_concurrent_set_funcs = {
569 .hash = fstring_concurrent_set_hash,
570 .cmp = fstring_concurrent_set_cmp,
571 .create = fstring_concurrent_set_create,
572 .free = NULL,
573};
574
575void
576Init_fstring_table(void)
577{
578 fstring_table_obj = rb_concurrent_set_new(&fstring_concurrent_set_funcs, 8192);
579 rb_gc_register_address(&fstring_table_obj);
580}
581
582static VALUE
583register_fstring(VALUE str, bool copy, bool force_precompute_hash)
584{
585 struct fstr_create_arg args = {
586 .copy = copy,
587 .force_precompute_hash = force_precompute_hash
588 };
589
590#if SIZEOF_VOIDP == SIZEOF_LONG
591 if (FL_TEST_RAW(str, STR_FAKESTR)) {
592 // if the string hasn't been interned, we'll need the hash twice, so we
593 // compute it once and store it in capa
594 RSTRING(str)->as.heap.aux.capa = (long)str_do_hash(str);
595 }
596#endif
597
598 VALUE result = rb_concurrent_set_find_or_insert(&fstring_table_obj, str, &args);
599
600 RUBY_ASSERT(!rb_objspace_garbage_object_p(result));
602 RUBY_ASSERT(OBJ_FROZEN(result));
604 RUBY_ASSERT((rb_gc_verify_shareable(result), 1));
605 RUBY_ASSERT(!FL_TEST_RAW(result, STR_FAKESTR));
607
608 return result;
609}
610
611bool
612rb_obj_is_fstring_table(VALUE obj)
613{
614 ASSERT_vm_locking();
615
616 return obj == fstring_table_obj;
617}
618
619void
620rb_gc_free_fstring(VALUE obj)
621{
622 ASSERT_vm_locking_with_barrier();
623
624 RUBY_ASSERT(FL_TEST(obj, RSTRING_FSTR));
626 RUBY_ASSERT(!FL_TEST(obj, STR_SHARED));
627
628 rb_concurrent_set_delete_by_identity(fstring_table_obj, obj);
629
630 RB_DEBUG_COUNTER_INC(obj_str_fstr);
631
632 FL_UNSET(obj, RSTRING_FSTR);
633}
634
635void
636rb_fstring_foreach_with_replace(int (*callback)(VALUE *str, void *data), void *data)
637{
638 if (fstring_table_obj) {
639 rb_concurrent_set_foreach_with_replace(fstring_table_obj, callback, data);
640 }
641}
642
643static VALUE
644setup_fake_str(struct RString *fake_str, const char *name, long len, int encidx)
645{
646 fake_str->basic.flags = T_STRING|RSTRING_NOEMBED|STR_NOFREE|STR_FAKESTR;
647 RBASIC_SET_FULL_SHAPE_ID((VALUE)fake_str, ROOT_SHAPE_ID | SHAPE_ID_LAYOUT_OTHER);
648
649 if (!name) {
651 name = "";
652 }
653
654 ENCODING_SET_INLINED((VALUE)fake_str, encidx);
655
656 RBASIC_SET_CLASS_RAW((VALUE)fake_str, rb_cString);
657 fake_str->len = len;
658 fake_str->as.heap.ptr = (char *)name;
659 fake_str->as.heap.aux.capa = len;
660 return (VALUE)fake_str;
661}
662
663/*
664 * set up a fake string which refers a static string literal.
665 */
666VALUE
667rb_setup_fake_str(struct RString *fake_str, const char *name, long len, rb_encoding *enc)
668{
669 return setup_fake_str(fake_str, name, len, rb_enc_to_index(enc));
670}
671
672/*
673 * rb_fstring_new and rb_fstring_cstr family create or lookup a frozen
674 * shared string which refers a static string literal. `ptr` must
675 * point a constant string.
676 */
677VALUE
678rb_fstring_new(const char *ptr, long len)
679{
680 struct RString fake_str = {RBASIC_INIT};
681 return register_fstring(setup_fake_str(&fake_str, ptr, len, ENCINDEX_US_ASCII), false, false);
682}
683
684VALUE
685rb_fstring_enc_new(const char *ptr, long len, rb_encoding *enc)
686{
687 struct RString fake_str = {RBASIC_INIT};
688 return register_fstring(rb_setup_fake_str(&fake_str, ptr, len, enc), false, false);
689}
690
691VALUE
692rb_fstring_cstr(const char *ptr)
693{
694 return rb_fstring_new(ptr, strlen(ptr));
695}
696
697static inline bool
698single_byte_optimizable(VALUE str)
699{
700 int encindex = ENCODING_GET(str);
701 switch (encindex) {
702 case ENCINDEX_ASCII_8BIT:
703 case ENCINDEX_US_ASCII:
704 return true;
705 case ENCINDEX_UTF_8:
706 // For UTF-8 it's worth scanning the string coderange when unknown.
707 return rb_enc_str_coderange(str) == ENC_CODERANGE_7BIT;
708 }
709 /* Conservative. It may be ENC_CODERANGE_UNKNOWN. */
710 if (ENC_CODERANGE(str) == ENC_CODERANGE_7BIT) {
711 return true;
712 }
713
714 if (rb_enc_mbmaxlen(rb_enc_from_index(encindex)) == 1) {
715 return true;
716 }
717
718 /* Conservative. Possibly single byte.
719 * "\xa1" in Shift_JIS for example. */
720 return false;
721}
722
724
725static inline const char *
726search_nonascii(const char *p, const char *e)
727{
728 const char *s, *t;
729
730 if (p < e && !ISASCII(*p)) {
731 return p;
732 }
733
734#if defined(__STDC_VERSION__) && (__STDC_VERSION__ >= 199901L)
735# if SIZEOF_UINTPTR_T == 8
736# define NONASCII_MASK UINT64_C(0x8080808080808080)
737# elif SIZEOF_UINTPTR_T == 4
738# define NONASCII_MASK UINT32_C(0x80808080)
739# else
740# error "don't know what to do."
741# endif
742#else
743# if SIZEOF_UINTPTR_T == 8
744# define NONASCII_MASK ((uintptr_t)0x80808080UL << 32 | (uintptr_t)0x80808080UL)
745# elif SIZEOF_UINTPTR_T == 4
746# define NONASCII_MASK 0x80808080UL /* or...? */
747# else
748# error "don't know what to do."
749# endif
750#endif
751
752 if (UNALIGNED_WORD_ACCESS || e - p >= SIZEOF_VOIDP) {
753#if !UNALIGNED_WORD_ACCESS
754 if ((uintptr_t)p % SIZEOF_VOIDP) {
755 int l = SIZEOF_VOIDP - (uintptr_t)p % SIZEOF_VOIDP;
756 p += l;
757 switch (l) {
758 default: UNREACHABLE;
759#if SIZEOF_VOIDP > 4
760 case 7: if (p[-7]&0x80) return p-7;
761 case 6: if (p[-6]&0x80) return p-6;
762 case 5: if (p[-5]&0x80) return p-5;
763 case 4: if (p[-4]&0x80) return p-4;
764#endif
765 case 3: if (p[-3]&0x80) return p-3;
766 case 2: if (p[-2]&0x80) return p-2;
767 case 1: if (p[-1]&0x80) return p-1;
768 case 0: break;
769 }
770 }
771#endif
772#if defined(HAVE_BUILTIN___BUILTIN_ASSUME_ALIGNED) &&! UNALIGNED_WORD_ACCESS
773#define aligned_ptr(value) \
774 __builtin_assume_aligned((value), sizeof(uintptr_t))
775#else
776#define aligned_ptr(value) (value)
777#endif
778 s = aligned_ptr(p);
779 t = (e - (SIZEOF_VOIDP-1));
780#undef aligned_ptr
781 for (;s < t; s += sizeof(uintptr_t)) {
782 uintptr_t word;
783 memcpy(&word, s, sizeof(word));
784 if (word & NONASCII_MASK) {
785#ifdef WORDS_BIGENDIAN
786 return (const char *)s + (nlz_intptr(word&NONASCII_MASK)>>3);
787#else
788 return (const char *)s + (ntz_intptr(word&NONASCII_MASK)>>3);
789#endif
790 }
791 }
792 p = (const char *)s;
793 }
794
795 switch (e - p) {
796 default: UNREACHABLE;
797#if SIZEOF_VOIDP > 4
798 case 7: if (e[-7]&0x80) return e-7;
799 case 6: if (e[-6]&0x80) return e-6;
800 case 5: if (e[-5]&0x80) return e-5;
801 case 4: if (e[-4]&0x80) return e-4;
802#endif
803 case 3: if (e[-3]&0x80) return e-3;
804 case 2: if (e[-2]&0x80) return e-2;
805 case 1: if (e[-1]&0x80) return e-1;
806 case 0: return NULL;
807 }
808}
809
810static int
811coderange_scan(const char *p, long len, rb_encoding *enc)
812{
813 const char *e = p + len;
814
815 if (rb_enc_to_index(enc) == rb_ascii8bit_encindex()) {
816 /* enc is ASCII-8BIT. ASCII-8BIT string never be broken. */
817 p = search_nonascii(p, e);
819 }
820
821 if (rb_enc_asciicompat(enc)) {
822 p = search_nonascii(p, e);
823 if (!p) return ENC_CODERANGE_7BIT;
824 for (;;) {
825 int ret = rb_enc_precise_mbclen(p, e, enc);
827 p += MBCLEN_CHARFOUND_LEN(ret);
828 if (p == e) break;
829 p = search_nonascii(p, e);
830 if (!p) break;
831 }
832 }
833 else {
834 while (p < e) {
835 int ret = rb_enc_precise_mbclen(p, e, enc);
837 p += MBCLEN_CHARFOUND_LEN(ret);
838 }
839 }
840 return ENC_CODERANGE_VALID;
841}
842
843long
844rb_str_coderange_scan_restartable(const char *s, const char *e, rb_encoding *enc, int *cr)
845{
846 const char *p = s;
847
848 if (*cr == ENC_CODERANGE_BROKEN)
849 return e - s;
850
851 if (rb_enc_to_index(enc) == rb_ascii8bit_encindex()) {
852 /* enc is ASCII-8BIT. ASCII-8BIT string never be broken. */
853 if (*cr == ENC_CODERANGE_VALID) return e - s;
854 p = search_nonascii(p, e);
856 return e - s;
857 }
858 else if (rb_enc_asciicompat(enc)) {
859 p = search_nonascii(p, e);
860 if (!p) {
861 if (*cr != ENC_CODERANGE_VALID) *cr = ENC_CODERANGE_7BIT;
862 return e - s;
863 }
864 for (;;) {
865 int ret = rb_enc_precise_mbclen(p, e, enc);
866 if (!MBCLEN_CHARFOUND_P(ret)) {
868 return p - s;
869 }
870 p += MBCLEN_CHARFOUND_LEN(ret);
871 if (p == e) break;
872 p = search_nonascii(p, e);
873 if (!p) break;
874 }
875 }
876 else {
877 while (p < e) {
878 int ret = rb_enc_precise_mbclen(p, e, enc);
879 if (!MBCLEN_CHARFOUND_P(ret)) {
881 return p - s;
882 }
883 p += MBCLEN_CHARFOUND_LEN(ret);
884 }
885 }
887 return e - s;
888}
889
890static inline void
891str_enc_copy(VALUE str1, VALUE str2)
892{
893 rb_enc_set_index(str1, ENCODING_GET(str2));
894}
895
896/* Like str_enc_copy, but does not check frozen status of str1.
897 * You should use this only if you're certain that str1 is not frozen. */
898static inline void
899str_enc_copy_direct(VALUE str1, VALUE str2)
900{
901 int inlined_encoding = RB_ENCODING_GET_INLINED(str2);
902 if (inlined_encoding == ENCODING_INLINE_MAX) {
903 rb_enc_set_index(str1, rb_enc_get_index(str2));
904 }
905 else {
906 ENCODING_SET_INLINED(str1, inlined_encoding);
907 }
908}
909
910static void
911rb_enc_cr_str_copy_for_substr(VALUE dest, VALUE src)
912{
913 /* this function is designed for copying encoding and coderange
914 * from src to new string "dest" which is made from the part of src.
915 */
916 str_enc_copy(dest, src);
917 if (RSTRING_LEN(dest) == 0) {
918 if (!rb_enc_asciicompat(STR_ENC_GET(src)))
920 else
922 return;
923 }
924 switch (ENC_CODERANGE(src)) {
927 break;
929 if (!rb_enc_asciicompat(STR_ENC_GET(src)) ||
930 search_nonascii(RSTRING_PTR(dest), RSTRING_END(dest)))
932 else
934 break;
935 default:
936 break;
937 }
938}
939
940static void
941rb_enc_cr_str_exact_copy(VALUE dest, VALUE src)
942{
943 str_enc_copy(dest, src);
945}
946
947static int
948enc_coderange_scan(VALUE str, rb_encoding *enc)
949{
950 return coderange_scan(RSTRING_PTR(str), RSTRING_LEN(str), enc);
951}
952
953int
954rb_enc_str_coderange_scan(VALUE str, rb_encoding *enc)
955{
956 return enc_coderange_scan(str, enc);
957}
958
959int
960rbimpl_enc_str_coderange_scan(VALUE str)
961{
962 int cr = enc_coderange_scan(str, get_encoding(str));
963 ENC_CODERANGE_SET(str, cr);
964 return cr;
965}
966
967#undef rb_enc_str_coderange
968int
969rb_enc_str_coderange(VALUE str)
970{
971 int cr = ENC_CODERANGE(str);
972
973 if (cr == ENC_CODERANGE_UNKNOWN) {
974 cr = rbimpl_enc_str_coderange_scan(str);
975 }
976 return cr;
977}
978#define rb_enc_str_coderange rb_enc_str_coderange_inline
979
980static inline bool
981rb_enc_str_asciicompat(VALUE str)
982{
983 int encindex = ENCODING_GET_INLINED(str);
984 return rb_str_encindex_fastpath(encindex) || rb_enc_asciicompat(rb_enc_get_from_index(encindex));
985}
986
987int
989{
990 switch(ENC_CODERANGE(str)) {
992 return rb_enc_str_asciicompat(str) && is_ascii_string(str);
994 return true;
995 default:
996 return false;
997 }
998}
999
1000static inline void
1001str_mod_check(VALUE s, const char *p, long len)
1002{
1003 if (RSTRING_PTR(s) != p || RSTRING_LEN(s) != len){
1004 rb_raise(rb_eRuntimeError, "string modified");
1005 }
1006}
1007
1008static size_t
1009str_capacity(VALUE str, const int termlen)
1010{
1011 if (STR_EMBED_P(str)) {
1012 return str_embed_capa(str) - termlen;
1013 }
1014 else if (FL_ANY_RAW(str, STR_SHARED|STR_NOFREE)) {
1015 return RSTRING(str)->len;
1016 }
1017 else {
1018 return RSTRING(str)->as.heap.aux.capa;
1019 }
1020}
1021
1022size_t
1024{
1025 return str_capacity(str, TERM_LEN(str));
1026}
1027
1028static inline void
1029must_not_null(const char *ptr)
1030{
1031 if (!ptr) {
1032 rb_raise(rb_eArgError, "NULL pointer given");
1033 }
1034}
1035
1036static inline VALUE
1037str_alloc_embed(VALUE klass, size_t capa)
1038{
1039 size_t size = rb_str_embed_size(capa, 0);
1040 RUBY_ASSERT(size > 0);
1041 RUBY_ASSERT(rb_gc_size_allocatable_p(size));
1042
1043 NEWOBJ_OF(str, struct RString, klass, T_STRING, size);
1044
1045 str->len = 0;
1046 str->as.embed.ary[0] = 0;
1047
1048 return (VALUE)str;
1049}
1050
1051static inline VALUE
1052str_alloc_heap(VALUE klass)
1053{
1054 NEWOBJ_OF(str, struct RString, klass, T_STRING | STR_NOEMBED, sizeof(struct RString));
1055
1056 str->len = 0;
1057 str->as.heap.aux.capa = 0;
1058 str->as.heap.ptr = NULL;
1059
1060 return (VALUE)str;
1061}
1062
1063static inline VALUE
1064empty_str_alloc(VALUE klass)
1065{
1066 RUBY_DTRACE_CREATE_HOOK(STRING, 0);
1067 VALUE str = str_alloc_embed(klass, 0);
1068 memset(RSTRING(str)->as.embed.ary, 0, str_embed_capa(str));
1070 return str;
1071}
1072
1073static VALUE
1074str_enc_new(VALUE klass, const char *ptr, long len, rb_encoding *enc)
1075{
1076 VALUE str;
1077
1078 if (len < 0) {
1079 rb_raise(rb_eArgError, "negative string size (or size too big)");
1080 }
1081
1082 if (enc == NULL) {
1083 enc = rb_ascii8bit_encoding();
1084 }
1085
1086 RUBY_DTRACE_CREATE_HOOK(STRING, len);
1087
1088 int termlen = rb_enc_mbminlen(enc);
1089
1090 if (STR_EMBEDDABLE_P(len, termlen)) {
1091 str = str_alloc_embed(klass, len + termlen);
1092 if (len == 0) {
1093 ENC_CODERANGE_SET(str, rb_enc_asciicompat(enc) ? ENC_CODERANGE_7BIT : ENC_CODERANGE_VALID);
1094 }
1095 }
1096 else {
1097 str = str_alloc_heap(klass);
1098 RSTRING(str)->as.heap.aux.capa = len;
1099 /* :FIXME: @shyouhei guesses `len + termlen` is guaranteed to never
1100 * integer overflow. If we can STATIC_ASSERT that, the following
1101 * mul_add_mul can be reverted to a simple ALLOC_N. */
1102 RSTRING(str)->as.heap.ptr =
1103 rb_xmalloc_mul_add_mul(sizeof(char), len, sizeof(char), termlen);
1104 }
1105
1106 rb_enc_raw_set(str, enc);
1107
1108 if (ptr) {
1109 memcpy(RSTRING_PTR(str), ptr, len);
1110 }
1111 else {
1112 memset(RSTRING_PTR(str), 0, len);
1113 }
1114
1115 STR_SET_LEN(str, len);
1116 TERM_FILL(RSTRING_PTR(str) + len, termlen);
1117 return str;
1118}
1119
1120static VALUE
1121str_new(VALUE klass, const char *ptr, long len)
1122{
1123 return str_enc_new(klass, ptr, len, rb_ascii8bit_encoding());
1124}
1125
1126VALUE
1127rb_str_new(const char *ptr, long len)
1128{
1129 return str_new(rb_cString, ptr, len);
1130}
1131
1132VALUE
1133rb_usascii_str_new(const char *ptr, long len)
1134{
1135 return str_enc_new(rb_cString, ptr, len, rb_usascii_encoding());
1136}
1137
1138VALUE
1139rb_utf8_str_new(const char *ptr, long len)
1140{
1141 return str_enc_new(rb_cString, ptr, len, rb_utf8_encoding());
1142}
1143
1144VALUE
1145rb_enc_str_new(const char *ptr, long len, rb_encoding *enc)
1146{
1147 return str_enc_new(rb_cString, ptr, len, enc);
1148}
1149
1150VALUE
1152{
1153 must_not_null(ptr);
1154 /* rb_str_new_cstr() can take pointer from non-malloc-generated
1155 * memory regions, and that cannot be detected by the MSAN. Just
1156 * trust the programmer that the argument passed here is a sane C
1157 * string. */
1158 __msan_unpoison_string(ptr);
1159 return rb_str_new(ptr, strlen(ptr));
1160}
1161
1162VALUE
1164{
1165 return rb_enc_str_new_cstr(ptr, rb_usascii_encoding());
1166}
1167
1168VALUE
1170{
1171 return rb_enc_str_new_cstr(ptr, rb_utf8_encoding());
1172}
1173
1174VALUE
1176{
1177 must_not_null(ptr);
1178 if (rb_enc_mbminlen(enc) != 1) {
1179 rb_raise(rb_eArgError, "wchar encoding given");
1180 }
1181 return rb_enc_str_new(ptr, strlen(ptr), enc);
1182}
1183
1184static VALUE
1185str_new_static(VALUE klass, const char *ptr, long len, int encindex)
1186{
1187 VALUE str;
1188
1189 if (len < 0) {
1190 rb_raise(rb_eArgError, "negative string size (or size too big)");
1191 }
1192
1193 if (!ptr) {
1194 str = str_enc_new(klass, ptr, len, rb_enc_from_index(encindex));
1195 }
1196 else {
1197 RUBY_DTRACE_CREATE_HOOK(STRING, len);
1198 str = str_alloc_heap(klass);
1199 RSTRING(str)->len = len;
1200 RSTRING(str)->as.heap.ptr = (char *)ptr;
1201 RSTRING(str)->as.heap.aux.capa = len;
1202 RBASIC(str)->flags |= STR_NOFREE;
1203 rb_enc_associate_index(str, encindex);
1204 }
1205 return str;
1206}
1207
1208VALUE
1209rb_str_new_static(const char *ptr, long len)
1210{
1211 return str_new_static(rb_cString, ptr, len, 0);
1212}
1213
1214/* Take an xmalloc'd buffer as the String's body without copying it; the String owns it
1215 * from here and frees it like any other heap string. ptr must hold capa bytes plus the
1216 * terminator for encindex, which is what a Ractor courier's string node carries. */
1217VALUE
1218rb_str_new_owned(char *ptr, long len, long capa, int encindex)
1219{
1220 RUBY_DTRACE_CREATE_HOOK(STRING, len);
1221 VALUE str = str_alloc_heap(rb_cString);
1222 RSTRING(str)->len = len;
1223 RSTRING(str)->as.heap.ptr = ptr;
1224 /* Freed by size (STR_HEAP_SIZE = capa + terminator), so capa must describe the
1225 * allocation the caller made, not just the bytes in use. */
1226 RSTRING(str)->as.heap.aux.capa = capa;
1227 rb_enc_associate_index(str, encindex);
1228 return str;
1229}
1230
1231VALUE
1233{
1234 return str_new_static(rb_cString, ptr, len, ENCINDEX_US_ASCII);
1235}
1236
1237VALUE
1239{
1240 return str_new_static(rb_cString, ptr, len, ENCINDEX_UTF_8);
1241}
1242
1243VALUE
1245{
1246 return str_new_static(rb_cString, ptr, len, rb_enc_to_index(enc));
1247}
1248
1249static VALUE str_cat_conv_enc_opts(VALUE newstr, long ofs, const char *ptr, long len,
1250 rb_encoding *from, rb_encoding *to,
1251 int ecflags, VALUE ecopts);
1252
1253static inline bool
1254is_enc_ascii_string(VALUE str, rb_encoding *enc)
1255{
1256 int encidx = rb_enc_to_index(enc);
1257 if (rb_enc_get_index(str) == encidx)
1258 return is_ascii_string(str);
1259 return enc_coderange_scan(str, enc) == ENC_CODERANGE_7BIT;
1260}
1261
1262VALUE
1263rb_str_conv_enc_opts(VALUE str, rb_encoding *from, rb_encoding *to, int ecflags, VALUE ecopts)
1264{
1265 long len;
1266 const char *ptr;
1267 VALUE newstr;
1268
1269 if (!to) return str;
1270 if (!from) from = rb_enc_get(str);
1271 if (from == to) return str;
1272 if ((rb_enc_asciicompat(to) && is_enc_ascii_string(str, from)) ||
1273 rb_is_ascii8bit_enc(to)) {
1274 if (STR_ENC_GET(str) != to) {
1275 str = rb_str_dup(str);
1276 rb_enc_associate(str, to);
1277 }
1278 return str;
1279 }
1280
1281 RSTRING_GETMEM(str, ptr, len);
1282 newstr = str_cat_conv_enc_opts(rb_str_buf_new(len), 0, ptr, len,
1283 from, to, ecflags, ecopts);
1284 if (NIL_P(newstr)) {
1285 /* some error, return original */
1286 return str;
1287 }
1288 return newstr;
1289}
1290
1291VALUE
1292rb_str_cat_conv_enc_opts(VALUE newstr, long ofs, const char *ptr, long len,
1293 rb_encoding *from, int ecflags, VALUE ecopts)
1294{
1295 long olen;
1296
1297 olen = RSTRING_LEN(newstr);
1298 if (ofs < -olen || olen < ofs)
1299 rb_raise(rb_eIndexError, "index %ld out of string", ofs);
1300 if (ofs < 0) ofs += olen;
1301 if (!from) {
1302 STR_SET_LEN(newstr, ofs);
1303 return rb_str_cat(newstr, ptr, len);
1304 }
1305
1306 rb_str_modify(newstr);
1307 return str_cat_conv_enc_opts(newstr, ofs, ptr, len, from,
1308 rb_enc_get(newstr),
1309 ecflags, ecopts);
1310}
1311
1312VALUE
1313rb_str_initialize(VALUE str, const char *ptr, long len, rb_encoding *enc)
1314{
1315 STR_SET_LEN(str, 0);
1316 rb_enc_associate(str, enc);
1317 rb_str_cat(str, ptr, len);
1318 return str;
1319}
1320
1321static VALUE
1322str_cat_conv_enc_opts(VALUE newstr, long ofs, const char *ptr, long len,
1323 rb_encoding *from, rb_encoding *to,
1324 int ecflags, VALUE ecopts)
1325{
1326 rb_econv_t *ec;
1328 long olen;
1329 VALUE econv_wrapper;
1330 const unsigned char *start, *sp;
1331 unsigned char *dest, *dp;
1332 size_t converted_output = (size_t)ofs;
1333
1334 olen = rb_str_capacity(newstr);
1335
1336 econv_wrapper = rb_obj_alloc(rb_cEncodingConverter);
1337 RBASIC_CLEAR_CLASS(econv_wrapper);
1338 ec = rb_econv_open_opts(from->name, to->name, ecflags, ecopts);
1339 if (!ec) return Qnil;
1340 DATA_PTR(econv_wrapper) = ec;
1341
1342 sp = (unsigned char*)ptr;
1343 start = sp;
1344 while ((dest = (unsigned char*)RSTRING_PTR(newstr)),
1345 (dp = dest + converted_output),
1346 (ret = rb_econv_convert(ec, &sp, start + len, &dp, dest + olen, 0)),
1348 /* destination buffer short */
1349 size_t converted_input = sp - start;
1350 size_t rest = len - converted_input;
1351 converted_output = dp - dest;
1352 rb_str_set_len(newstr, converted_output);
1353 if (converted_input && converted_output &&
1354 rest < (LONG_MAX / converted_output)) {
1355 rest = (rest * converted_output) / converted_input;
1356 }
1357 else {
1358 rest = olen;
1359 }
1360 olen += rest < 2 ? 2 : rest;
1361 rb_str_resize(newstr, olen);
1362 }
1363 DATA_PTR(econv_wrapper) = 0;
1364 RB_GC_GUARD(econv_wrapper);
1365 rb_econv_close(ec);
1366 switch (ret) {
1367 case econv_finished:
1368 len = dp - (unsigned char*)RSTRING_PTR(newstr);
1369 rb_str_set_len(newstr, len);
1370 rb_enc_associate(newstr, to);
1371 return newstr;
1372
1373 default:
1374 return Qnil;
1375 }
1376}
1377
1378VALUE
1380{
1381 return rb_str_conv_enc_opts(str, from, to, 0, Qnil);
1382}
1383
1384VALUE
1386{
1387 rb_encoding *ienc;
1388 VALUE str;
1389 const int eidx = rb_enc_to_index(eenc);
1390
1391 if (!ptr) {
1392 return rb_enc_str_new(ptr, len, eenc);
1393 }
1394
1395 /* ASCII-8BIT case, no conversion */
1396 if ((eidx == rb_ascii8bit_encindex()) ||
1397 (eidx == rb_usascii_encindex() && search_nonascii(ptr, ptr + len))) {
1398 return rb_str_new(ptr, len);
1399 }
1400 /* no default_internal or same encoding, no conversion */
1401 ienc = rb_default_internal_encoding();
1402 if (!ienc || eenc == ienc) {
1403 return rb_enc_str_new(ptr, len, eenc);
1404 }
1405 /* ASCII compatible, and ASCII only string, no conversion in
1406 * default_internal */
1407 if ((eidx == rb_ascii8bit_encindex()) ||
1408 (eidx == rb_usascii_encindex()) ||
1409 (rb_enc_asciicompat(eenc) && !search_nonascii(ptr, ptr + len))) {
1410 return rb_enc_str_new(ptr, len, ienc);
1411 }
1412 /* convert from the given encoding to default_internal */
1413 str = rb_enc_str_new(NULL, 0, ienc);
1414 /* when the conversion failed for some reason, just ignore the
1415 * default_internal and result in the given encoding as-is. */
1416 if (NIL_P(rb_str_cat_conv_enc_opts(str, 0, ptr, len, eenc, 0, Qnil))) {
1417 rb_str_initialize(str, ptr, len, eenc);
1418 }
1419 return str;
1420}
1421
1422VALUE
1423rb_external_str_with_enc(VALUE str, rb_encoding *eenc)
1424{
1425 int eidx = rb_enc_to_index(eenc);
1426 if (eidx == rb_usascii_encindex() &&
1427 !is_ascii_string(str)) {
1428 rb_enc_associate_index(str, rb_ascii8bit_encindex());
1429 return str;
1430 }
1431 rb_enc_associate_index(str, eidx);
1432 return rb_str_conv_enc(str, eenc, rb_default_internal_encoding());
1433}
1434
1435VALUE
1436rb_external_str_new(const char *ptr, long len)
1437{
1438 return rb_external_str_new_with_enc(ptr, len, rb_default_external_encoding());
1439}
1440
1441VALUE
1443{
1444 return rb_external_str_new_with_enc(ptr, strlen(ptr), rb_default_external_encoding());
1445}
1446
1447VALUE
1448rb_locale_str_new(const char *ptr, long len)
1449{
1450 return rb_external_str_new_with_enc(ptr, len, rb_locale_encoding());
1451}
1452
1453VALUE
1455{
1456 return rb_external_str_new_with_enc(ptr, strlen(ptr), rb_locale_encoding());
1457}
1458
1459VALUE
1461{
1462 return rb_external_str_new_with_enc(ptr, len, rb_filesystem_encoding());
1463}
1464
1465VALUE
1467{
1468 return rb_external_str_new_with_enc(ptr, strlen(ptr), rb_filesystem_encoding());
1469}
1470
1471VALUE
1473{
1474 return rb_str_export_to_enc(str, rb_default_external_encoding());
1475}
1476
1477VALUE
1479{
1480 return rb_str_export_to_enc(str, rb_locale_encoding());
1481}
1482
1483VALUE
1485{
1486 return rb_str_conv_enc(str, STR_ENC_GET(str), enc);
1487}
1488
1489static VALUE
1490str_replace_shared_without_enc(VALUE str2, VALUE str)
1491{
1492 const int termlen = TERM_LEN(str);
1493 char *ptr;
1494 long len;
1495
1496 RSTRING_GETMEM(str, ptr, len);
1497 if (str_embed_capa(str2) >= len + termlen) {
1498 char *ptr2 = RSTRING(str2)->as.embed.ary;
1499 STR_SET_EMBED(str2);
1500 memcpy(ptr2, RSTRING_PTR(str), len);
1501 TERM_FILL(ptr2+len, termlen);
1502 }
1503 else {
1504 VALUE root;
1505 if (STR_SHARED_P(str)) {
1506 root = RSTRING(str)->as.heap.aux.shared;
1507 RSTRING_GETMEM(str, ptr, len);
1508 }
1509 else {
1510 root = rb_str_new_frozen(str);
1511 RSTRING_GETMEM(root, ptr, len);
1512 }
1513 RUBY_ASSERT(OBJ_FROZEN(root));
1514
1515 if (!STR_EMBED_P(str2) && !FL_TEST_RAW(str2, STR_SHARED|STR_NOFREE)) {
1516 if (FL_TEST_RAW(str2, STR_SHARED_ROOT)) {
1517 rb_fatal("about to free a possible shared root");
1518 }
1519 char *ptr2 = STR_HEAP_PTR(str2);
1520 if (ptr2 != ptr) {
1521 SIZED_FREE_N(ptr2, STR_HEAP_SIZE(str2));
1522 }
1523 }
1524 FL_SET(str2, STR_NOEMBED);
1525 RSTRING(str2)->as.heap.ptr = ptr;
1526 STR_SET_SHARED(str2, root);
1527 }
1528
1529 STR_SET_LEN(str2, len);
1530
1531 return str2;
1532}
1533
1534static VALUE
1535str_replace_shared(VALUE str2, VALUE str)
1536{
1537 str_replace_shared_without_enc(str2, str);
1538 rb_enc_cr_str_exact_copy(str2, str);
1539 return str2;
1540}
1541
1542static VALUE
1543str_new_shared(VALUE klass, VALUE str)
1544{
1545 return str_replace_shared(str_alloc_heap(klass), str);
1546}
1547
1548VALUE
1550{
1551 return str_new_shared(rb_obj_class(str), str);
1552}
1553
1554VALUE
1556{
1557 if (RB_FL_TEST_RAW(orig, FL_FREEZE | STR_CHILLED) == FL_FREEZE) return orig;
1558 return str_new_frozen(rb_obj_class(orig), orig);
1559}
1560
1561static VALUE
1562rb_str_new_frozen_String(VALUE orig)
1563{
1564 if (OBJ_FROZEN(orig) && rb_obj_class(orig) == rb_cString) return orig;
1565 return str_new_frozen(rb_cString, orig);
1566}
1567
1568
1569VALUE
1570rb_str_frozen_bare_string(VALUE orig)
1571{
1572 if (RB_LIKELY(BARE_STRING_P(orig) && OBJ_FROZEN_RAW(orig))) return orig;
1573 return str_new_frozen(rb_cString, orig);
1574}
1575
1576VALUE
1577rb_str_tmp_frozen_acquire(VALUE orig)
1578{
1579 if (OBJ_FROZEN_RAW(orig)) return orig;
1580 return str_new_frozen_buffer(0, orig, FALSE);
1581}
1582
1583VALUE
1584rb_str_tmp_frozen_no_embed_acquire(VALUE orig)
1585{
1586 if (OBJ_FROZEN_RAW(orig) && !STR_EMBED_P(orig) && !rb_str_reembeddable_p(orig)) return orig;
1587 if (STR_SHARED_P(orig) && !STR_EMBED_P(RSTRING(orig)->as.heap.aux.shared)) return rb_str_tmp_frozen_acquire(orig);
1588
1589 VALUE str = str_alloc_heap(0);
1590 OBJ_FREEZE(str);
1591 /* Always set the STR_SHARED_ROOT to ensure it does not get re-embedded. */
1592 FL_SET(str, STR_SHARED_ROOT);
1593
1594 size_t capa = str_capacity(orig, TERM_LEN(orig));
1595
1596 /* If the string is embedded then we want to create a copy that is heap
1597 * allocated. If the string is shared then the shared root must be
1598 * embedded, so we want to create a copy. If the string is a shared root
1599 * then it must be embedded, so we want to create a copy. */
1600 if (STR_EMBED_P(orig) || FL_TEST_RAW(orig, STR_SHARED | STR_SHARED_ROOT | RSTRING_FSTR)) {
1601 RSTRING(str)->as.heap.ptr = rb_xmalloc_mul_add_mul(sizeof(char), capa, sizeof(char), TERM_LEN(orig));
1602 memcpy(RSTRING(str)->as.heap.ptr, RSTRING_PTR(orig), capa);
1603 }
1604 else {
1605 /* orig must be heap allocated and not shared, so we can safely transfer
1606 * the pointer to str. */
1607 RSTRING(str)->as.heap.ptr = RSTRING(orig)->as.heap.ptr;
1608 RBASIC(str)->flags |= RBASIC(orig)->flags & STR_NOFREE;
1609 RBASIC(orig)->flags &= ~STR_NOFREE;
1610 STR_SET_SHARED(orig, str);
1611 /* str was just allocated here, so orig is its only child and it is
1612 * safe for rb_str_tmp_frozen_release to give the buffer back. */
1613 FL_UNSET_RAW(str, STR_BORROWED);
1614 if (RB_OBJ_SHAREABLE_P(orig)) {
1616 RUBY_ASSERT((rb_gc_verify_shareable(str), 1));
1617 }
1618 }
1619
1620 RSTRING(str)->len = RSTRING(orig)->len;
1621 RSTRING(str)->as.heap.aux.capa = capa + (TERM_LEN(orig) - TERM_LEN(str));
1622
1623 return str;
1624}
1625
1626void
1627rb_str_tmp_frozen_release(VALUE orig, VALUE tmp)
1628{
1629 if (RBASIC_CLASS(tmp) != 0)
1630 return;
1631
1632 if (STR_EMBED_P(tmp)) {
1634 }
1635 else if (FL_TEST_RAW(orig, STR_SHARED | STR_TMPLOCK) == STR_SHARED &&
1636 !OBJ_FROZEN_RAW(orig)) {
1637 VALUE shared = RSTRING(orig)->as.heap.aux.shared;
1638
1639 if (shared == tmp && !FL_TEST_RAW(tmp, STR_BORROWED)) {
1640 RUBY_ASSERT(RSTRING(orig)->as.heap.ptr == RSTRING(tmp)->as.heap.ptr);
1641 RUBY_ASSERT(RSTRING_LEN(orig) == RSTRING_LEN(tmp));
1642
1643 /* Unshare orig since the root (tmp) only has this one child. */
1644 FL_UNSET_RAW(orig, STR_SHARED);
1645 RSTRING(orig)->as.heap.aux.capa = RSTRING(tmp)->as.heap.aux.capa + TERM_LEN(tmp) - TERM_LEN(orig);
1646 RBASIC(orig)->flags |= RBASIC(tmp)->flags & STR_NOFREE;
1648
1649 /* Make tmp embedded and empty so it is safe for sweeping. */
1650 STR_SET_EMBED(tmp);
1651 STR_SET_LEN(tmp, 0);
1652 }
1653 }
1654}
1655
1656static VALUE
1657str_new_frozen(VALUE klass, VALUE orig)
1658{
1659 return str_new_frozen_buffer(klass, orig, TRUE);
1660}
1661
1662/* Transfers ownership of orig's buffer to a new shared root string.
1663 * termlen is the terminator length of the returned string, which may differ
1664 * from orig's terminator length when the caller does not copy the encoding.
1665 * The capacity is stored without the terminator, so it must be adjusted for
1666 * the difference to keep the buffer size (capa + termlen) unchanged. */
1667static VALUE
1668heap_str_make_shared(VALUE klass, VALUE orig, int termlen)
1669{
1670 RUBY_ASSERT(!STR_EMBED_P(orig));
1671 RUBY_ASSERT(!STR_SHARED_P(orig));
1673
1674 VALUE str = str_alloc_heap(klass);
1675 STR_SET_LEN(str, RSTRING_LEN(orig));
1676 RSTRING(str)->as.heap.ptr = RSTRING_PTR(orig);
1677 RSTRING(str)->as.heap.aux.capa = RSTRING(orig)->as.heap.aux.capa + TERM_LEN(orig) - termlen;
1678 RBASIC(str)->flags |= RBASIC(orig)->flags & STR_NOFREE;
1679 RBASIC(orig)->flags &= ~STR_NOFREE;
1680 STR_SET_SHARED(orig, str);
1681 if (klass == 0)
1682 FL_UNSET_RAW(str, STR_BORROWED);
1683 return str;
1684}
1685
1686static VALUE
1687str_new_frozen_buffer(VALUE klass, VALUE orig, int copy_encoding)
1688{
1689 VALUE str;
1690
1691 long len = RSTRING_LEN(orig);
1692 rb_encoding *enc = copy_encoding ? STR_ENC_GET(orig) : rb_ascii8bit_encoding();
1693 int termlen = copy_encoding ? TERM_LEN(orig) : 1;
1694
1695 if (STR_EMBED_P(orig) || STR_EMBEDDABLE_P(len, termlen)) {
1696 str = str_enc_new(klass, RSTRING_PTR(orig), len, enc);
1697 RUBY_ASSERT(STR_EMBED_P(str));
1698 }
1699 else {
1700 if (FL_TEST_RAW(orig, STR_SHARED)) {
1701 VALUE shared = RSTRING(orig)->as.heap.aux.shared;
1702 long ofs = RSTRING(orig)->as.heap.ptr - RSTRING_PTR(shared);
1703 long rest = RSTRING_LEN(shared) - ofs - RSTRING_LEN(orig);
1704 RUBY_ASSERT(ofs >= 0);
1705 RUBY_ASSERT(rest >= 0);
1706 RUBY_ASSERT(ofs + rest <= RSTRING_LEN(shared));
1708
1709 if ((ofs > 0) || (rest > 0) ||
1710 (klass != RBASIC(shared)->klass) ||
1711 ENCODING_GET(shared) != ENCODING_GET(orig)) {
1712 str = str_new_shared(klass, shared);
1713 RUBY_ASSERT(!STR_EMBED_P(str));
1714 RSTRING(str)->as.heap.ptr += ofs;
1715 STR_SET_LEN(str, RSTRING_LEN(str) - (ofs + rest));
1716 }
1717 else {
1718 if (RBASIC_CLASS(shared) == 0)
1719 FL_SET_RAW(shared, STR_BORROWED);
1720 return shared;
1721 }
1722 }
1723 else if (STR_EMBEDDABLE_P(RSTRING_LEN(orig), TERM_LEN(orig))) {
1724 str = str_alloc_embed(klass, RSTRING_LEN(orig) + TERM_LEN(orig));
1725 STR_SET_EMBED(str);
1726 memcpy(RSTRING_PTR(str), RSTRING_PTR(orig), RSTRING_LEN(orig));
1727 STR_SET_LEN(str, RSTRING_LEN(orig));
1728 ENC_CODERANGE_SET(str, ENC_CODERANGE(orig));
1729 TERM_FILL(RSTRING_END(str), TERM_LEN(orig));
1730 }
1731 else {
1732 if (RB_OBJ_SHAREABLE_P(orig)) {
1733 str = str_new(klass, RSTRING_PTR(orig), RSTRING_LEN(orig));
1734 }
1735 else {
1736 str = heap_str_make_shared(klass, orig, termlen);
1737 }
1738 }
1739 }
1740
1741 if (copy_encoding) rb_enc_cr_str_exact_copy(str, orig);
1742 OBJ_FREEZE(str);
1743 return str;
1744}
1745
1746VALUE
1747rb_str_new_with_class(VALUE obj, const char *ptr, long len)
1748{
1749 return str_enc_new(rb_obj_class(obj), ptr, len, STR_ENC_GET(obj));
1750}
1751
1752static VALUE
1753str_new_empty_String(VALUE str)
1754{
1755 VALUE v = rb_str_new(0, 0);
1756 rb_enc_copy(v, str);
1757 return v;
1758}
1759
1760#define STR_BUF_MIN_SIZE 63
1761
1762VALUE
1764{
1765 if (STR_EMBEDDABLE_P(capa, 1)) {
1766 return str_alloc_embed(rb_cString, capa + 1);
1767 }
1768
1769 VALUE str = str_alloc_heap(rb_cString);
1770
1771 RSTRING(str)->as.heap.aux.capa = capa;
1772 RSTRING(str)->as.heap.ptr = ALLOC_N(char, (size_t)capa + 1);
1773 RSTRING(str)->as.heap.ptr[0] = '\0';
1774
1775 return str;
1776}
1777
1778VALUE
1780{
1781 VALUE str;
1782 long len = strlen(ptr);
1783
1784 str = rb_str_buf_new(len);
1785 rb_str_buf_cat(str, ptr, len);
1786
1787 return str;
1788}
1789
1790VALUE
1792{
1793 return str_new(0, 0, len);
1794}
1795
1796void
1798{
1799 if (STR_EMBED_P(str)) {
1800 RB_DEBUG_COUNTER_INC(obj_str_embed);
1801 }
1802 else if (FL_TEST(str, STR_SHARED | STR_NOFREE)) {
1803 (void)RB_DEBUG_COUNTER_INC_IF(obj_str_shared, FL_TEST(str, STR_SHARED));
1804 (void)RB_DEBUG_COUNTER_INC_IF(obj_str_shared, FL_TEST(str, STR_NOFREE));
1805 }
1806 else {
1807 RB_DEBUG_COUNTER_INC(obj_str_ptr);
1808 SIZED_FREE_N(STR_HEAP_PTR(str), STR_HEAP_SIZE(str));
1809 }
1810}
1811
1812size_t
1813rb_str_memsize(VALUE str)
1814{
1815 if (FL_TEST(str, STR_NOEMBED|STR_SHARED|STR_NOFREE) == STR_NOEMBED) {
1816 return STR_HEAP_SIZE(str);
1817 }
1818 else {
1819 return 0;
1820 }
1821}
1822
1823VALUE
1825{
1826 return rb_convert_type_with_id(str, T_STRING, "String", idTo_str);
1827}
1828
1829static inline void str_discard(VALUE str);
1830static void str_shared_replace(VALUE str, VALUE str2);
1831
1832void
1834{
1835 if (str != str2) str_shared_replace(str, str2);
1836}
1837
1838static void
1839str_shared_replace(VALUE str, VALUE str2)
1840{
1841 rb_encoding *enc;
1842 int cr;
1843 int termlen;
1844
1845 RUBY_ASSERT(str2 != str);
1846 enc = STR_ENC_GET(str2);
1847 cr = ENC_CODERANGE(str2);
1848 str_discard(str);
1849 termlen = rb_enc_mbminlen(enc);
1850
1851 STR_SET_LEN(str, RSTRING_LEN(str2));
1852
1853 if (str_embed_capa(str) >= RSTRING_LEN(str2) + termlen) {
1854 STR_SET_EMBED(str);
1855 memcpy(RSTRING_PTR(str), RSTRING_PTR(str2), (size_t)RSTRING_LEN(str2) + termlen);
1856 rb_enc_associate(str, enc);
1857 ENC_CODERANGE_SET(str, cr);
1858 }
1859 else {
1860 if (STR_EMBED_P(str2)) {
1861 RUBY_ASSERT(!FL_TEST(str2, STR_SHARED));
1862 long len = RSTRING_LEN(str2);
1863 RUBY_ASSERT(len + termlen <= str_embed_capa(str2));
1864
1865 char *new_ptr = ALLOC_N(char, len + termlen);
1866 memcpy(new_ptr, RSTRING(str2)->as.embed.ary, len + termlen);
1867 RSTRING(str2)->as.heap.ptr = new_ptr;
1868 STR_SET_LEN(str2, len);
1869 RSTRING(str2)->as.heap.aux.capa = len;
1870 STR_SET_NOEMBED(str2);
1871 }
1872
1873 STR_SET_NOEMBED(str);
1874 FL_UNSET(str, STR_SHARED);
1875 RSTRING(str)->as.heap.ptr = RSTRING_PTR(str2);
1876
1877 if (FL_TEST(str2, STR_SHARED)) {
1878 VALUE shared = RSTRING(str2)->as.heap.aux.shared;
1879 STR_SET_SHARED(str, shared);
1880 }
1881 else {
1882 RSTRING(str)->as.heap.aux.capa = RSTRING(str2)->as.heap.aux.capa;
1883 }
1884
1885 /* abandon str2 */
1886 STR_SET_EMBED(str2);
1887 RSTRING_PTR(str2)[0] = 0;
1888 STR_SET_LEN(str2, 0);
1889 rb_enc_associate(str, enc);
1890 ENC_CODERANGE_SET(str, cr);
1891 }
1892}
1893
1894VALUE
1896{
1897 VALUE str;
1898
1899 if (RB_TYPE_P(obj, T_STRING)) {
1900 return obj;
1901 }
1902 str = rb_funcall(obj, idTo_s, 0);
1903 return rb_obj_as_string_result(str, obj);
1904}
1905
1906VALUE
1907rb_obj_as_string_result(VALUE str, VALUE obj)
1908{
1909 if (!RB_TYPE_P(str, T_STRING))
1910 return rb_any_to_s(obj);
1911 return str;
1912}
1913
1914static VALUE
1915str_replace(VALUE str, VALUE str2)
1916{
1917 long len;
1918
1919 len = RSTRING_LEN(str2);
1920 if (STR_SHARED_P(str2)) {
1921 VALUE shared = RSTRING(str2)->as.heap.aux.shared;
1923 STR_SET_NOEMBED(str);
1924 STR_SET_LEN(str, len);
1925 RSTRING(str)->as.heap.ptr = RSTRING_PTR(str2);
1926 STR_SET_SHARED(str, shared);
1927 rb_enc_cr_str_exact_copy(str, str2);
1928 }
1929 else {
1930 str_replace_shared(str, str2);
1931 }
1932
1933 return str;
1934}
1935
1936static inline VALUE
1937ec_str_alloc_embed(struct rb_execution_context_struct *ec, VALUE klass, size_t capa)
1938{
1939 size_t size = rb_str_embed_size(capa, 0);
1940 RUBY_ASSERT(size > 0);
1941 RUBY_ASSERT(rb_gc_size_allocatable_p(size));
1942
1943 EC_NEWOBJ_OF(str, struct RString, klass, T_STRING, size, ec);
1944
1945 str->len = 0;
1946
1947 return (VALUE)str;
1948}
1949
1950static inline VALUE
1951ec_str_alloc_heap(struct rb_execution_context_struct *ec, VALUE klass)
1952{
1953 EC_NEWOBJ_OF(str, struct RString, klass, T_STRING | STR_NOEMBED, sizeof(struct RString), ec);
1954
1955 str->as.heap.aux.capa = 0;
1956 str->as.heap.ptr = NULL;
1957
1958 return (VALUE)str;
1959}
1960
1961static inline void
1962str_duplicate_setup_encoding(VALUE str, VALUE dup, VALUE flags)
1963{
1964 int encidx = 0;
1965 if ((flags & ENCODING_MASK) == (ENCODING_INLINE_MAX<<ENCODING_SHIFT)) {
1966 encidx = rb_enc_get_index(str);
1967 flags &= ~ENCODING_MASK;
1968 }
1969 FL_SET_RAW(dup, flags & ~FL_FREEZE);
1970 if (encidx) rb_enc_associate_index(dup, encidx);
1971}
1972
1973static const VALUE flag_mask = ENC_CODERANGE_MASK | ENCODING_MASK | FL_FREEZE;
1974
1975static inline void
1976str_duplicate_setup_embed(VALUE klass, VALUE str, VALUE dup)
1977{
1978 VALUE flags = FL_TEST_RAW(str, flag_mask);
1979 long len = RSTRING_LEN(str);
1980
1981 RUBY_ASSERT(STR_EMBED_P(dup));
1982 RUBY_ASSERT(str_embed_capa(dup) >= len + TERM_LEN(str));
1983 MEMCPY(RSTRING(dup)->as.embed.ary, RSTRING(str)->as.embed.ary, char, len + TERM_LEN(str));
1984 STR_SET_LEN(dup, RSTRING_LEN(str));
1985 str_duplicate_setup_encoding(str, dup, flags);
1986}
1987
1988static inline void
1989str_duplicate_setup_heap(VALUE klass, VALUE str, VALUE dup)
1990{
1991 VALUE flags = FL_TEST_RAW(str, flag_mask);
1992 VALUE root = str;
1993 if (FL_TEST_RAW(str, STR_SHARED)) {
1994 root = RSTRING(str)->as.heap.aux.shared;
1995 }
1996 else if (UNLIKELY(!OBJ_FROZEN_RAW(str))) {
1997 root = str = str_new_frozen(klass, str);
1998 flags = FL_TEST_RAW(str, flag_mask);
1999 }
2000 RUBY_ASSERT(!STR_SHARED_P(root));
2002
2003 RSTRING(dup)->as.heap.ptr = RSTRING_PTR(str);
2004 FL_SET_RAW(dup, RSTRING_NOEMBED);
2005 STR_SET_SHARED(dup, root);
2006 flags |= RSTRING_NOEMBED | STR_SHARED;
2007
2008 STR_SET_LEN(dup, RSTRING_LEN(str));
2009 str_duplicate_setup_encoding(str, dup, flags);
2010}
2011
2012static inline VALUE
2013str_duplicate(VALUE klass, VALUE str)
2014{
2015 VALUE dup;
2016 if (STR_EMBED_P(str) && rb_str_embed_size(RSTRING_LEN(str), 1) <= STR_COPY_MAX_EMBED_SIZE) {
2017 dup = str_alloc_embed(klass, RSTRING_LEN(str) + TERM_LEN(str));
2018
2019 str_duplicate_setup_embed(klass, str, dup);
2020 }
2021 else {
2022 dup = str_alloc_heap(klass);
2023
2024 str_duplicate_setup_heap(klass, str, dup);
2025 }
2026
2027 return dup;
2028}
2029
2030VALUE
2032{
2033 return str_duplicate(rb_obj_class(str), str);
2034}
2035
2036/* :nodoc: */
2037VALUE
2038rb_str_dup_m(VALUE str)
2039{
2040 if (LIKELY(BARE_STRING_P(str))) {
2041 return str_duplicate(rb_cString, str);
2042 }
2043 else {
2044 return rb_obj_dup(str);
2045 }
2046}
2047
2048VALUE
2050{
2051 RUBY_DTRACE_CREATE_HOOK(STRING, RSTRING_LEN(str));
2052 return str_duplicate(rb_cString, str);
2053}
2054
2055VALUE
2056rb_ec_str_resurrect(struct rb_execution_context_struct *ec, VALUE str, bool chilled)
2057{
2058 RUBY_DTRACE_CREATE_HOOK(STRING, RSTRING_LEN(str));
2059 VALUE new_str, klass = rb_cString;
2060
2061 if (!(chilled && RTEST(rb_ivar_defined(str, id_debug_created_info))) && STR_EMBED_P(str)) {
2062 new_str = ec_str_alloc_embed(ec, klass, RSTRING_LEN(str) + TERM_LEN(str));
2063 str_duplicate_setup_embed(klass, str, new_str);
2064 }
2065 else {
2066 new_str = ec_str_alloc_heap(ec, klass);
2067 str_duplicate_setup_heap(klass, str, new_str);
2068 }
2069 if (chilled) {
2070 FL_SET_RAW(new_str, STR_CHILLED);
2071 }
2072 return new_str;
2073}
2074
2075#if USE_ZJIT
2076bool
2077rb_zjit_str_resurrect_fastpath(VALUE str, bool chilled, size_t *size_out,
2078 VALUE *flags_out,
2079 long *len_out, size_t *byte_size_out)
2080{
2081 if (chilled && RTEST(rb_ivar_defined(str, id_debug_created_info))) return false;
2082
2083 if (!STR_EMBED_P(str)) return false;
2084
2085 long len = RSTRING_LEN(str);
2086 long termlen = TERM_LEN(str);
2087 size_t size = rb_str_embed_size(len + termlen, 0);
2088 if (!rb_gc_size_allocatable_p(size)) return false;
2089
2090 VALUE flags = FL_TEST_RAW(str, flag_mask);
2091
2092 if ((flags & ENCODING_MASK) == ((VALUE)ENCODING_INLINE_MAX << ENCODING_SHIFT)) {
2093 return false;
2094 }
2095
2096 flags &= ~FL_FREEZE;
2097 flags |= T_STRING;
2098 if (chilled) flags |= STR_CHILLED;
2099
2100 *size_out = size;
2101 *flags_out = flags;
2102 *len_out = len;
2103 *byte_size_out = (size_t)(len + termlen);
2104 return true;
2105}
2106#endif
2107
2108VALUE
2109rb_str_with_debug_created_info(VALUE str, VALUE path, int line)
2110{
2111 VALUE debug_info = rb_ary_new_from_args(2, path, INT2FIX(line));
2112 if (OBJ_FROZEN_RAW(str)) str = rb_str_dup(str);
2113 rb_ivar_set(str, id_debug_created_info, rb_ary_freeze(debug_info));
2114 FL_SET_RAW(str, STR_CHILLED);
2115 return rb_str_freeze(str);
2116}
2117
2118/*
2119 * The documentation block below uses an include (instead of inline text)
2120 * because the included text has non-ASCII characters (which are not allowed in a C file).
2121 */
2122
2123/*
2124 *
2125 * call-seq:
2126 * String.new(string = ''.encode(Encoding::ASCII_8BIT) , **options) -> new_string
2127 *
2128 * :include: doc/string/new.rdoc
2129 *
2130 */
2131
2132static VALUE
2133rb_str_init(int argc, VALUE *argv, VALUE str)
2134{
2135 static ID keyword_ids[2];
2136 VALUE orig, opt, venc, vcapa;
2137 VALUE kwargs[2];
2138 rb_encoding *enc = 0;
2139 int n;
2140
2141 if (!keyword_ids[0]) {
2142 keyword_ids[0] = rb_id_encoding();
2143 CONST_ID(keyword_ids[1], "capacity");
2144 }
2145
2146 n = rb_scan_args(argc, argv, "01:", &orig, &opt);
2147 if (!NIL_P(opt)) {
2148 rb_get_kwargs(opt, keyword_ids, 0, 2, kwargs);
2149 venc = kwargs[0];
2150 vcapa = kwargs[1];
2151 if (!UNDEF_P(venc) && !NIL_P(venc)) {
2152 enc = rb_to_encoding(venc);
2153 }
2154 if (!UNDEF_P(vcapa) && !NIL_P(vcapa)) {
2155 long capa = NUM2LONG(vcapa);
2156 long len = 0;
2157 int termlen = enc ? rb_enc_mbminlen(enc) : 1;
2158
2159 if (capa < STR_BUF_MIN_SIZE) {
2160 capa = STR_BUF_MIN_SIZE;
2161 }
2162 if (n == 1) {
2163 StringValue(orig);
2164 len = RSTRING_LEN(orig);
2165 if (capa < len) {
2166 capa = len;
2167 }
2168 if (orig == str) n = 0;
2169 }
2170 str_modifiable(str);
2171 if (STR_EMBED_P(str) || FL_TEST(str, STR_SHARED|STR_NOFREE)) {
2172 /* make noembed always */
2173 const size_t size = (size_t)capa + termlen;
2174 const char *const old_ptr = RSTRING_PTR(str);
2175 const size_t osize = RSTRING_LEN(str) + TERM_LEN(str);
2176 char *new_ptr = ALLOC_N(char, size);
2177 if (STR_EMBED_P(str)) RUBY_ASSERT((long)osize <= str_embed_capa(str));
2178 memcpy(new_ptr, old_ptr, osize < size ? osize : size);
2179 FL_UNSET_RAW(str, STR_SHARED|STR_NOFREE);
2180 RSTRING(str)->as.heap.ptr = new_ptr;
2181 }
2182 else if (STR_HEAP_SIZE(str) != (size_t)capa + termlen) {
2183 SIZED_REALLOC_N(RSTRING(str)->as.heap.ptr, char,
2184 (size_t)capa + termlen, STR_HEAP_SIZE(str));
2185 }
2186 STR_SET_LEN(str, len);
2187 TERM_FILL(&RSTRING(str)->as.heap.ptr[len], termlen);
2188 if (n == 1) {
2189 memcpy(RSTRING(str)->as.heap.ptr, RSTRING_PTR(orig), len);
2190 rb_enc_cr_str_exact_copy(str, orig);
2191 }
2192 FL_SET(str, STR_NOEMBED);
2193 RSTRING(str)->as.heap.aux.capa = capa;
2194 }
2195 else if (n == 1) {
2196 rb_str_replace(str, orig);
2197 }
2198 if (enc) {
2199 rb_enc_associate(str, enc);
2201 }
2202 }
2203 else if (n == 1) {
2204 rb_str_replace(str, orig);
2205 }
2206 return str;
2207}
2208
2209/* :nodoc: */
2210static VALUE
2211rb_str_s_new(int argc, VALUE *argv, VALUE klass)
2212{
2213 if (klass != rb_cString) {
2214 return rb_class_new_instance_pass_kw(argc, argv, klass);
2215 }
2216
2217 static ID keyword_ids[2];
2218 VALUE orig, opt, encoding = Qnil, capacity = Qnil;
2219 VALUE kwargs[2];
2220 rb_encoding *enc = NULL;
2221
2222 int n = rb_scan_args(argc, argv, "01:", &orig, &opt);
2223 if (NIL_P(opt)) {
2224 return rb_class_new_instance_pass_kw(argc, argv, klass);
2225 }
2226
2227 keyword_ids[0] = rb_id_encoding();
2228 CONST_ID(keyword_ids[1], "capacity");
2229 rb_get_kwargs(opt, keyword_ids, 0, 2, kwargs);
2230 encoding = kwargs[0];
2231 capacity = kwargs[1];
2232
2233 if (n == 1) {
2234 orig = StringValue(orig);
2235 }
2236 else {
2237 orig = Qnil;
2238 }
2239
2240 if (UNDEF_P(encoding)) {
2241 if (!NIL_P(orig)) {
2242 encoding = rb_obj_encoding(orig);
2243 }
2244 }
2245
2246 if (!UNDEF_P(encoding)) {
2247 enc = rb_to_encoding(encoding);
2248 }
2249
2250 // If capacity is nil, we're basically just duping `orig`.
2251 if (UNDEF_P(capacity)) {
2252 if (NIL_P(orig)) {
2253 VALUE empty_str = str_new(klass, "", 0);
2254 if (enc) {
2255 rb_enc_associate(empty_str, enc);
2256 }
2257 return empty_str;
2258 }
2259 VALUE copy = str_duplicate(klass, orig);
2260 rb_enc_associate(copy, enc);
2261 ENC_CODERANGE_CLEAR(copy);
2262 return copy;
2263 }
2264
2265 long capa = 0;
2266 capa = NUM2LONG(capacity);
2267 if (capa < 0) {
2268 capa = 0;
2269 }
2270
2271 if (!NIL_P(orig)) {
2272 long orig_capa = rb_str_capacity(orig);
2273 if (orig_capa > capa) {
2274 capa = orig_capa;
2275 }
2276 }
2277
2278 VALUE str = str_enc_new(klass, NULL, capa, enc);
2279 STR_SET_LEN(str, 0);
2280 TERM_FILL(RSTRING_PTR(str), enc ? rb_enc_mbmaxlen(enc) : 1);
2281
2282 if (!NIL_P(orig)) {
2283 rb_str_buf_append(str, orig);
2284 }
2285
2286 return str;
2287}
2288
2289#ifdef NONASCII_MASK
2290#define is_utf8_lead_byte(c) (((c)&0xC0) != 0x80)
2291
2292/*
2293 * UTF-8 leading bytes have either 0xxxxxxx or 11xxxxxx
2294 * bit representation. (see https://en.wikipedia.org/wiki/UTF-8)
2295 * Therefore, the following pseudocode can detect UTF-8 leading bytes.
2296 *
2297 * if (!(byte & 0x80))
2298 * byte |= 0x40; // turn on bit6
2299 * return ((byte>>6) & 1); // bit6 represent whether this byte is leading or not.
2300 *
2301 * This function calculates whether a byte is leading or not for all bytes
2302 * in the argument word by concurrently using the above logic, and then
2303 * adds up the number of leading bytes in the word.
2304 */
2305static inline uintptr_t
2306count_utf8_lead_bytes_with_word(const uintptr_t *s)
2307{
2308 uintptr_t d = *s;
2309
2310 /* Transform so that bit0 indicates whether we have a UTF-8 leading byte or not. */
2311 d = (d>>6) | (~d>>7);
2312 d &= NONASCII_MASK >> 7;
2313
2314 /* Gather all bytes. */
2315#if defined(HAVE_BUILTIN___BUILTIN_POPCOUNT) && defined(__POPCNT__)
2316 /* use only if it can use POPCNT */
2317 return rb_popcount_intptr(d);
2318#else
2319 d += (d>>8);
2320 d += (d>>16);
2321# if SIZEOF_VOIDP == 8
2322 d += (d>>32);
2323# endif
2324 return (d&0xF);
2325#endif
2326}
2327#endif
2328
2329static inline long
2330enc_strlen(const char *p, const char *e, rb_encoding *enc, int cr)
2331{
2332 long c;
2333 const char *q;
2334
2335 if (rb_enc_mbmaxlen(enc) == rb_enc_mbminlen(enc)) {
2336 long diff = (long)(e - p);
2337 return diff / rb_enc_mbminlen(enc) + !!(diff % rb_enc_mbminlen(enc));
2338 }
2339#ifdef NONASCII_MASK
2340 else if (cr == ENC_CODERANGE_VALID && enc == rb_utf8_encoding()) {
2341 uintptr_t len = 0;
2342 if ((int)sizeof(uintptr_t) * 2 < e - p) {
2343 const uintptr_t *s, *t;
2344 const uintptr_t lowbits = sizeof(uintptr_t) - 1;
2345 s = (const uintptr_t*)(~lowbits & ((uintptr_t)p + lowbits));
2346 t = (const uintptr_t*)(~lowbits & (uintptr_t)e);
2347 while (p < (const char *)s) {
2348 if (is_utf8_lead_byte(*p)) len++;
2349 p++;
2350 }
2351 while (s < t) {
2352 len += count_utf8_lead_bytes_with_word(s);
2353 s++;
2354 }
2355 p = (const char *)s;
2356 }
2357 while (p < e) {
2358 if (is_utf8_lead_byte(*p)) len++;
2359 p++;
2360 }
2361 return (long)len;
2362 }
2363#endif
2364 else if (rb_enc_asciicompat(enc)) {
2365 c = 0;
2366 if (ENC_CODERANGE_CLEAN_P(cr)) {
2367 while (p < e) {
2368 q = search_nonascii(p, e);
2369 if (!q)
2370 return c + (e - p);
2371 c += q - p;
2372 p = q;
2373 p += rb_enc_fast_mbclen(p, e, enc);
2374 c++;
2375 }
2376 }
2377 else {
2378 while (p < e) {
2379 q = search_nonascii(p, e);
2380 if (!q)
2381 return c + (e - p);
2382 c += q - p;
2383 p = q;
2384 p += rb_enc_mbclen(p, e, enc);
2385 c++;
2386 }
2387 }
2388 return c;
2389 }
2390
2391 for (c=0; p<e; c++) {
2392 p += rb_enc_mbclen(p, e, enc);
2393 }
2394 return c;
2395}
2396
2397long
2398rb_enc_strlen(const char *p, const char *e, rb_encoding *enc)
2399{
2400 return enc_strlen(p, e, enc, ENC_CODERANGE_UNKNOWN);
2401}
2402
2403/* To get strlen with cr
2404 * Note that given cr is not used.
2405 */
2406long
2407rb_enc_strlen_cr(const char *p, const char *e, rb_encoding *enc, int *cr)
2408{
2409 long c;
2410 const char *q;
2411 int ret;
2412
2413 *cr = 0;
2414 if (rb_enc_mbmaxlen(enc) == rb_enc_mbminlen(enc)) {
2415 long diff = (long)(e - p);
2416 return diff / rb_enc_mbminlen(enc) + !!(diff % rb_enc_mbminlen(enc));
2417 }
2418 else if (rb_enc_asciicompat(enc)) {
2419 c = 0;
2420 while (p < e) {
2421 q = search_nonascii(p, e);
2422 if (!q) {
2423 if (!*cr) *cr = ENC_CODERANGE_7BIT;
2424 return c + (e - p);
2425 }
2426 c += q - p;
2427 p = q;
2428 ret = rb_enc_precise_mbclen(p, e, enc);
2429 if (MBCLEN_CHARFOUND_P(ret)) {
2430 *cr |= ENC_CODERANGE_VALID;
2431 p += MBCLEN_CHARFOUND_LEN(ret);
2432 }
2433 else {
2435 p++;
2436 }
2437 c++;
2438 }
2439 if (!*cr) *cr = ENC_CODERANGE_7BIT;
2440 return c;
2441 }
2442
2443 for (c=0; p<e; c++) {
2444 ret = rb_enc_precise_mbclen(p, e, enc);
2445 if (MBCLEN_CHARFOUND_P(ret)) {
2446 *cr |= ENC_CODERANGE_VALID;
2447 p += MBCLEN_CHARFOUND_LEN(ret);
2448 }
2449 else {
2451 if (p + rb_enc_mbminlen(enc) <= e)
2452 p += rb_enc_mbminlen(enc);
2453 else
2454 p = e;
2455 }
2456 }
2457 if (!*cr) *cr = ENC_CODERANGE_7BIT;
2458 return c;
2459}
2460
2461/* enc must be str's enc or rb_enc_check(str, str2) */
2462static long
2463str_strlen(VALUE str, rb_encoding *enc)
2464{
2465 const char *p, *e;
2466 int cr;
2467
2468 if (single_byte_optimizable(str)) return RSTRING_LEN(str);
2469 if (!enc) enc = STR_ENC_GET(str);
2470 p = RSTRING_PTR(str);
2471 e = RSTRING_END(str);
2472 cr = ENC_CODERANGE(str);
2473
2474 if (cr == ENC_CODERANGE_UNKNOWN) {
2475 long n = rb_enc_strlen_cr(p, e, enc, &cr);
2476 if (cr) ENC_CODERANGE_SET(str, cr);
2477 return n;
2478 }
2479 else {
2480 return enc_strlen(p, e, enc, cr);
2481 }
2482}
2483
2484long
2486{
2487 return str_strlen(str, NULL);
2488}
2489
2490/*
2491 * call-seq:
2492 * length -> integer
2493 *
2494 * :include: doc/string/length.rdoc
2495 *
2496 */
2497
2498VALUE
2500{
2501 return LONG2NUM(str_strlen(str, NULL));
2502}
2503
2504/*
2505 * call-seq:
2506 * bytesize -> integer
2507 *
2508 * :include: doc/string/bytesize.rdoc
2509 *
2510 */
2511
2512VALUE
2513rb_str_bytesize(VALUE str)
2514{
2515 return LONG2NUM(RSTRING_LEN(str));
2516}
2517
2518/*
2519 * call-seq:
2520 * empty? -> true or false
2521 *
2522 * Returns whether the length of +self+ is zero:
2523 *
2524 * 'hello'.empty? # => false
2525 * ' '.empty? # => false
2526 * ''.empty? # => true
2527 *
2528 * Related: see {Querying}[rdoc-ref:String@Querying].
2529 */
2530
2531static VALUE
2532rb_str_empty(VALUE str)
2533{
2534 return RBOOL(RSTRING_LEN(str) == 0);
2535}
2536
2537/*
2538 * call-seq:
2539 * self + other_string -> new_string
2540 *
2541 * Returns a new string containing +other_string+ concatenated to +self+:
2542 *
2543 * 'Hello from ' + self.to_s # => "Hello from main"
2544 *
2545 * Related: see {Converting to New String}[rdoc-ref:String@Converting+to+New+String].
2546 */
2547
2548VALUE
2550{
2551 VALUE str3;
2552 rb_encoding *enc;
2553 const char *ptr1, *ptr2;
2554 char *ptr3;
2555 long len1, len2;
2556 int termlen;
2557
2558 StringValue(str2);
2559 enc = rb_enc_check_str(str1, str2);
2560 RSTRING_GETMEM(str1, ptr1, len1);
2561 RSTRING_GETMEM(str2, ptr2, len2);
2562 termlen = rb_enc_mbminlen(enc);
2563 if (len1 > LONG_MAX - len2) {
2564 rb_raise(rb_eArgError, "string size too big");
2565 }
2566 str3 = str_enc_new(rb_cString, 0, len1+len2, enc);
2567 ptr3 = RSTRING_PTR(str3);
2568 memcpy(ptr3, ptr1, len1);
2569 memcpy(ptr3+len1, ptr2, len2);
2570 TERM_FILL(&ptr3[len1+len2], termlen);
2571
2572 ENCODING_CODERANGE_SET(str3, rb_enc_to_index(enc),
2574 RB_GC_GUARD(str1);
2575 RB_GC_GUARD(str2);
2576 return str3;
2577}
2578
2579/* A variant of rb_str_plus that does not raise but return Qundef instead. */
2580VALUE
2581rb_str_opt_plus(VALUE str1, VALUE str2)
2582{
2585 long len1, len2;
2586 MAYBE_UNUSED(char) *ptr1, *ptr2;
2587 RSTRING_GETMEM(str1, ptr1, len1);
2588 RSTRING_GETMEM(str2, ptr2, len2);
2589 int enc1 = rb_enc_get_index(str1);
2590 int enc2 = rb_enc_get_index(str2);
2591
2592 if (enc1 < 0) {
2593 return Qundef;
2594 }
2595 else if (enc2 < 0) {
2596 return Qundef;
2597 }
2598 else if (enc1 != enc2) {
2599 return Qundef;
2600 }
2601 else if (len1 > LONG_MAX - len2) {
2602 return Qundef;
2603 }
2604 else {
2605 return rb_str_plus(str1, str2);
2606 }
2607
2608}
2609
2610/*
2611 * call-seq:
2612 * self * n -> new_string
2613 *
2614 * Returns a new string containing +n+ copies of +self+:
2615 *
2616 * 'Ho!' * 3 # => "Ho!Ho!Ho!"
2617 * 'No!' * 0 # => ""
2618 *
2619 * Related: see {Converting to New String}[rdoc-ref:String@Converting+to+New+String].
2620 */
2621
2622VALUE
2624{
2625 VALUE str2;
2626 long n, len;
2627 char *ptr2;
2628 int termlen;
2629
2630 if (times == INT2FIX(1)) {
2631 return str_duplicate(rb_cString, str);
2632 }
2633 if (times == INT2FIX(0)) {
2634 str2 = str_alloc_embed(rb_cString, 0);
2635 rb_enc_copy(str2, str);
2636 return str2;
2637 }
2638 len = NUM2LONG(times);
2639 if (len < 0) {
2640 rb_raise(rb_eArgError, "negative argument");
2641 }
2642 if (RSTRING_LEN(str) == 1 && RSTRING_PTR(str)[0] == 0) {
2643 if (STR_EMBEDDABLE_P(len, 1)) {
2644 str2 = str_alloc_embed(rb_cString, len + 1);
2645 memset(RSTRING_PTR(str2), 0, len + 1);
2646 }
2647 else {
2648 str2 = str_alloc_heap(rb_cString);
2649 RSTRING(str2)->as.heap.aux.capa = len;
2650 RSTRING(str2)->as.heap.ptr = ZALLOC_N(char, (size_t)len + 1);
2651 }
2652 STR_SET_LEN(str2, len);
2653 rb_enc_copy(str2, str);
2654 return str2;
2655 }
2656 if (len && LONG_MAX/len < RSTRING_LEN(str)) {
2657 rb_raise(rb_eArgError, "argument too big");
2658 }
2659
2660 len *= RSTRING_LEN(str);
2661 termlen = TERM_LEN(str);
2662 str2 = str_enc_new(rb_cString, 0, len, STR_ENC_GET(str));
2663 ptr2 = RSTRING_PTR(str2);
2664 if (len) {
2665 n = RSTRING_LEN(str);
2666 memcpy(ptr2, RSTRING_PTR(str), n);
2667 while (n <= len/2) {
2668 memcpy(ptr2 + n, ptr2, n);
2669 n *= 2;
2670 }
2671 memcpy(ptr2 + n, ptr2, len-n);
2672 }
2673 STR_SET_LEN(str2, len);
2674 TERM_FILL(&ptr2[len], termlen);
2675 rb_enc_cr_str_copy_for_substr(str2, str);
2676
2677 return str2;
2678}
2679
2680/*
2681 * call-seq:
2682 * self % object -> new_string
2683 *
2684 * Returns the result of formatting +object+ into the format specifications
2685 * contained in +self+
2686 * (see {Format Specifications}[rdoc-ref:language/format_specifications.rdoc]):
2687 *
2688 * '%05d' % 123 # => "00123"
2689 *
2690 * If +self+ contains multiple format specifications,
2691 * +object+ must be an array or hash containing the objects to be formatted:
2692 *
2693 * '%-5s: %016x' % [ 'ID', self.object_id ] # => "ID : 00002b054ec93168"
2694 * 'foo = %{foo}' % {foo: 'bar'} # => "foo = bar"
2695 * 'foo = %{foo}, baz = %{baz}' % {foo: 'bar', baz: 'bat'} # => "foo = bar, baz = bat"
2696 *
2697 * Related: see {Converting to New String}[rdoc-ref:String@Converting+to+New+String].
2698 */
2699
2700static VALUE
2701rb_str_format_m(VALUE str, VALUE arg)
2702{
2703 VALUE tmp = rb_check_array_type(arg);
2704
2705 if (!NIL_P(tmp)) {
2706 VALUE result = rb_str_format_ary(RARRAY_LENINT(tmp), RARRAY_CONST_PTR(tmp), str, tmp);
2707 RB_GC_GUARD(tmp);
2708 return result;
2709 }
2710 return rb_str_format(1, &arg, str);
2711}
2712
2713static inline void
2714rb_check_lockedtmp(VALUE str)
2715{
2716 if (FL_TEST(str, STR_TMPLOCK)) {
2717 rb_raise(rb_eRuntimeError, "can't modify string; temporarily locked");
2718 }
2719}
2720
2721// If none of these flags are set, we know we have an modifiable string.
2722// If any is set, we need to do more detailed checks.
2723#define STR_UNMODIFIABLE_MASK (FL_FREEZE | STR_TMPLOCK | STR_CHILLED)
2724static inline void
2725str_modifiable(VALUE str)
2726{
2727 RUBY_ASSERT(ruby_thread_has_gvl_p());
2728
2729 if (RB_UNLIKELY(FL_ANY_RAW(str, STR_UNMODIFIABLE_MASK))) {
2730 if (CHILLED_STRING_P(str)) {
2731 CHILLED_STRING_MUTATED(str);
2732 }
2733 rb_check_lockedtmp(str);
2734 rb_check_frozen(str);
2735 }
2736}
2737
2738static inline int
2739str_dependent_p(VALUE str)
2740{
2741 if (STR_EMBED_P(str) || !FL_TEST(str, STR_SHARED|STR_NOFREE)) {
2742 return FALSE;
2743 }
2744 else {
2745 return TRUE;
2746 }
2747}
2748
2749// If none of these flags are set, we know we have an independent string.
2750// If any is set, we need to do more detailed checks.
2751#define STR_DEPENDANT_MASK (STR_UNMODIFIABLE_MASK | STR_SHARED | STR_NOFREE)
2752static inline int
2753str_independent(VALUE str)
2754{
2755 RUBY_ASSERT(ruby_thread_has_gvl_p());
2756
2757 if (RB_UNLIKELY(FL_ANY_RAW(str, STR_DEPENDANT_MASK))) {
2758 str_modifiable(str);
2759 return !str_dependent_p(str);
2760 }
2761 return TRUE;
2762}
2763
2764static void
2765str_make_independent_expand(VALUE str, long len, long expand, const int termlen)
2766{
2767 RUBY_ASSERT(ruby_thread_has_gvl_p());
2768
2769 char *ptr;
2770 char *oldptr;
2771 long capa = len + expand;
2772
2773 if (len > capa) len = capa;
2774
2775 if (!STR_EMBED_P(str) && str_embed_capa(str) >= capa + termlen) {
2776 ptr = RSTRING(str)->as.heap.ptr;
2777 STR_SET_EMBED(str);
2778 memcpy(RSTRING(str)->as.embed.ary, ptr, len);
2779 TERM_FILL(RSTRING(str)->as.embed.ary + len, termlen);
2780 STR_SET_LEN(str, len);
2781 return;
2782 }
2783
2784 ptr = ALLOC_N(char, (size_t)capa + termlen);
2785 oldptr = RSTRING_PTR(str);
2786 if (oldptr) {
2787 memcpy(ptr, oldptr, len);
2788 }
2789 if (FL_TEST_RAW(str, STR_NOEMBED|STR_NOFREE|STR_SHARED) == STR_NOEMBED) {
2790 SIZED_FREE_N(oldptr, STR_HEAP_SIZE(str));
2791 }
2792 STR_SET_NOEMBED(str);
2793 FL_UNSET(str, STR_SHARED|STR_NOFREE);
2794 TERM_FILL(ptr + len, termlen);
2795 RSTRING(str)->as.heap.ptr = ptr;
2796 STR_SET_LEN(str, len);
2797 RSTRING(str)->as.heap.aux.capa = capa;
2798}
2799
2800void
2801rb_str_modify(VALUE str)
2802{
2803 if (!str_independent(str))
2804 str_make_independent(str);
2806}
2807
2808void
2810{
2811 RUBY_ASSERT(ruby_thread_has_gvl_p());
2812
2813 int termlen = TERM_LEN(str);
2814 long len = RSTRING_LEN(str);
2815
2816 if (expand < 0) {
2817 rb_raise(rb_eArgError, "negative expanding string size");
2818 }
2819 if (expand >= LONG_MAX - len) {
2820 rb_raise(rb_eArgError, "string size too big");
2821 }
2822
2823 if (!str_independent(str)) {
2824 str_make_independent_expand(str, len, expand, termlen);
2825 }
2826 else if (expand > 0) {
2827 RESIZE_CAPA_TERM(str, len + expand, termlen);
2828 }
2830}
2831
2832/* As rb_str_modify(), but don't clear coderange */
2833static void
2834str_modify_keep_cr(VALUE str)
2835{
2836 if (!str_independent(str))
2837 str_make_independent(str);
2839 /* Force re-scan later */
2841}
2842
2843static inline void
2844str_discard(VALUE str)
2845{
2846 str_modifiable(str);
2847 if (!STR_EMBED_P(str) && !FL_TEST(str, STR_SHARED|STR_NOFREE)) {
2848 SIZED_FREE_N(STR_HEAP_PTR(str), STR_HEAP_SIZE(str));
2849 RSTRING(str)->as.heap.ptr = 0;
2850 STR_SET_LEN(str, 0);
2851 }
2852}
2853
2854void
2856{
2857 int encindex = rb_enc_get_index(str);
2858
2859 if (RB_UNLIKELY(encindex == -1)) {
2860 rb_raise(rb_eTypeError, "not encoding capable object");
2861 }
2862
2863 if (RB_LIKELY(rb_str_encindex_fastpath(encindex))) {
2864 return;
2865 }
2866
2867 rb_encoding *enc = rb_enc_from_index(encindex);
2868 if (!rb_enc_asciicompat(enc)) {
2869 rb_raise(rb_eEncCompatError, "ASCII incompatible encoding: %s", rb_enc_name(enc));
2870 }
2871}
2872
2873VALUE
2875{
2876 RUBY_ASSERT(ruby_thread_has_gvl_p());
2877
2878 VALUE s = *ptr;
2879 if (!RB_TYPE_P(s, T_STRING)) {
2880 s = rb_str_to_str(s);
2881 *ptr = s;
2882 }
2883 return s;
2884}
2885
2886char *
2888{
2889 VALUE str = rb_string_value(ptr);
2890 return RSTRING_PTR(str);
2891}
2892
2893static const char *
2894str_null_char(const char *s, long len, const int minlen, rb_encoding *enc)
2895{
2896 const char *e = s + len;
2897
2898 for (; s + minlen <= e; s += rb_enc_mbclen(s, e, enc)) {
2899 if (zero_filled(s, minlen)) return s;
2900 }
2901 return 0;
2902}
2903
2904static char *
2905str_fill_term(VALUE str, char *s, long len, int termlen)
2906{
2907 /* This function assumes that (capa + termlen) bytes of memory
2908 * is allocated, like many other functions in this file.
2909 */
2910 if (str_dependent_p(str)) {
2911 if (!zero_filled(s + len, termlen))
2912 str_make_independent_expand(str, len, 0L, termlen);
2913 }
2914 else {
2915 TERM_FILL(s + len, termlen);
2916 return s;
2917 }
2918 return RSTRING_PTR(str);
2919}
2920
2921void
2922rb_str_change_terminator_length(VALUE str, const int oldtermlen, const int termlen)
2923{
2924 long capa = str_capacity(str, oldtermlen) + oldtermlen;
2925 long len = RSTRING_LEN(str);
2926
2927 RUBY_ASSERT(capa >= len);
2928 if (capa - len < termlen) {
2929 rb_check_lockedtmp(str);
2930 str_make_independent_expand(str, len, 0L, termlen);
2931 }
2932 else if (str_dependent_p(str)) {
2933 if (termlen > oldtermlen)
2934 str_make_independent_expand(str, len, 0L, termlen);
2935 }
2936 else {
2937 if (!STR_EMBED_P(str)) {
2938 /* modify capa instead of realloc */
2939 RUBY_ASSERT(!FL_TEST((str), STR_SHARED));
2940 RSTRING(str)->as.heap.aux.capa = capa - termlen;
2941 }
2942 if (termlen > oldtermlen) {
2943 TERM_FILL(RSTRING_PTR(str) + len, termlen);
2944 }
2945 }
2946
2947 return;
2948}
2949
2950static char *
2951str_null_check(VALUE str, int *w)
2952{
2953 char *s = RSTRING_PTR(str);
2954 long len = RSTRING_LEN(str);
2955 int minlen = 1;
2956
2957 if (RB_UNLIKELY(!rb_str_enc_fastpath(str))) {
2958 rb_encoding *enc = rb_str_enc_get(str);
2959 minlen = rb_enc_mbminlen(enc);
2960
2961 if (minlen > 1) {
2962 *w = 1;
2963 if (str_null_char(s, len, minlen, enc)) {
2964 return NULL;
2965 }
2966 return str_fill_term(str, s, len, minlen);
2967 }
2968 }
2969
2970 *w = 0;
2971 if (!s || memchr(s, 0, len)) {
2972 return NULL;
2973 }
2974 if (s[len]) {
2975 s = str_fill_term(str, s, len, minlen);
2976 }
2977 return s;
2978}
2979
2980static char *str_to_cstr(VALUE str);
2981
2982const char *
2983rb_str_null_check(VALUE str)
2984{
2986
2987 const char *s;
2988 long len;
2989 RSTRING_GETMEM(str, s, len);
2990
2991 if (RB_LIKELY(rb_str_enc_fastpath(str))) {
2992 if (!s || memchr(s, 0, len)) {
2993 rb_raise(rb_eArgError, "string contains null byte");
2994 }
2995 }
2996 else {
2997 str_to_cstr(str);
2998 }
2999
3000 return s;
3001}
3002
3003char *
3004rb_str_to_cstr(VALUE str)
3005{
3006 int w;
3007 return str_null_check(str, &w);
3008}
3009
3010char *
3012{
3013 VALUE str = rb_string_value(ptr);
3014 return str_to_cstr(str);
3015}
3016
3017static char *
3018str_to_cstr(VALUE str)
3019{
3020 int w;
3021 char *s = str_null_check(str, &w);
3022 if (!s) {
3023 if (w) {
3024 rb_raise(rb_eArgError, "string contains null char");
3025 }
3026 rb_raise(rb_eArgError, "string contains null byte");
3027 }
3028 return s;
3029}
3030
3031char *
3032rb_str_fill_terminator(VALUE str, const int newminlen)
3033{
3034 char *s = RSTRING_PTR(str);
3035 long len = RSTRING_LEN(str);
3036 return str_fill_term(str, s, len, newminlen);
3037}
3038
3039VALUE
3041{
3042 str = rb_check_convert_type_with_id(str, T_STRING, "String", idTo_str);
3043 return str;
3044}
3045
3046/*
3047 * call-seq:
3048 * String.try_convert(object) -> object, new_string, or nil
3049 *
3050 * Attempts to convert the given +object+ to a string.
3051 *
3052 * If +object+ is already a string, returns +object+, unmodified.
3053 *
3054 * Otherwise if +object+ responds to <tt>:to_str</tt>,
3055 * calls <tt>object.to_str</tt> and returns the result.
3056 *
3057 * Returns +nil+ if +object+ does not respond to <tt>:to_str</tt>.
3058 *
3059 * Raises an exception unless <tt>object.to_str</tt> returns a string.
3060 */
3061static VALUE
3062rb_str_s_try_convert(VALUE dummy, VALUE str)
3063{
3064 return rb_check_string_type(str);
3065}
3066
3067static char*
3068str_nth_len(const char *p, const char *e, long *nthp, rb_encoding *enc)
3069{
3070 long nth = *nthp;
3071 if (rb_enc_mbmaxlen(enc) == 1) {
3072 p += nth;
3073 }
3074 else if (rb_enc_mbmaxlen(enc) == rb_enc_mbminlen(enc)) {
3075 p += nth * rb_enc_mbmaxlen(enc);
3076 }
3077 else if (rb_enc_asciicompat(enc)) {
3078 const char *p2, *e2;
3079 int n;
3080
3081 while (p < e && 0 < nth) {
3082 e2 = p + nth;
3083 if (e < e2) {
3084 *nthp = nth;
3085 return (char *)e;
3086 }
3087 p2 = search_nonascii(p, e2);
3088 if (!p2) {
3089 nth -= e2 - p;
3090 *nthp = nth;
3091 return (char *)e2;
3092 }
3093 nth -= p2 - p;
3094 p = p2;
3095 n = rb_enc_mbclen(p, e, enc);
3096 p += n;
3097 nth--;
3098 }
3099 *nthp = nth;
3100 if (nth != 0) {
3101 return (char *)e;
3102 }
3103 return (char *)p;
3104 }
3105 else {
3106 while (p < e && nth--) {
3107 p += rb_enc_mbclen(p, e, enc);
3108 }
3109 }
3110 if (p > e) p = e;
3111 *nthp = nth;
3112 return (char*)p;
3113}
3114
3115char*
3116rb_enc_nth(const char *p, const char *e, long nth, rb_encoding *enc)
3117{
3118 return str_nth_len(p, e, &nth, enc);
3119}
3120
3121static char*
3122str_nth(const char *p, const char *e, long nth, rb_encoding *enc, int singlebyte)
3123{
3124 if (singlebyte)
3125 p += nth;
3126 else {
3127 p = str_nth_len(p, e, &nth, enc);
3128 }
3129 if (!p) return 0;
3130 if (p > e) p = e;
3131 return (char *)p;
3132}
3133
3134/* char offset to byte offset */
3135static long
3136str_offset(const char *p, const char *e, long nth, rb_encoding *enc, int singlebyte)
3137{
3138 const char *pp = str_nth(p, e, nth, enc, singlebyte);
3139 if (!pp) return e - p;
3140 return pp - p;
3141}
3142
3143long
3144rb_str_offset(VALUE str, long pos)
3145{
3146 return str_offset(RSTRING_PTR(str), RSTRING_END(str), pos,
3147 STR_ENC_GET(str), single_byte_optimizable(str));
3148}
3149
3150#ifdef NONASCII_MASK
3151static char *
3152str_utf8_nth(const char *p, const char *e, long *nthp)
3153{
3154 long nth = *nthp;
3155 if ((int)SIZEOF_VOIDP * 2 < e - p && (int)SIZEOF_VOIDP * 2 < nth) {
3156 const uintptr_t *s, *t;
3157 const uintptr_t lowbits = SIZEOF_VOIDP - 1;
3158 s = (const uintptr_t*)(~lowbits & ((uintptr_t)p + lowbits));
3159 t = (const uintptr_t*)(~lowbits & (uintptr_t)e);
3160 while (p < (const char *)s) {
3161 if (is_utf8_lead_byte(*p)) nth--;
3162 p++;
3163 }
3164 do {
3165 nth -= count_utf8_lead_bytes_with_word(s);
3166 s++;
3167 } while (s < t && (int)SIZEOF_VOIDP <= nth);
3168 p = (char *)s;
3169 }
3170 while (p < e) {
3171 if (is_utf8_lead_byte(*p)) {
3172 if (nth == 0) break;
3173 nth--;
3174 }
3175 p++;
3176 }
3177 *nthp = nth;
3178 return (char *)p;
3179}
3180
3181static long
3182str_utf8_offset(const char *p, const char *e, long nth)
3183{
3184 const char *pp = str_utf8_nth(p, e, &nth);
3185 return pp - p;
3186}
3187#endif
3188
3189/* byte offset to char offset */
3190long
3191rb_str_sublen(VALUE str, long pos)
3192{
3193 if (single_byte_optimizable(str) || pos < 0)
3194 return pos;
3195 else {
3196 const char *p = RSTRING_PTR(str);
3197 return enc_strlen(p, p + pos, STR_ENC_GET(str), ENC_CODERANGE(str));
3198 }
3199}
3200
3201static VALUE
3202str_subseq(VALUE str, long beg, long len)
3203{
3204 VALUE str2;
3205
3206 RUBY_ASSERT(beg >= 0);
3207 RUBY_ASSERT(len >= 0);
3208 RUBY_ASSERT(beg+len <= RSTRING_LEN(str));
3209
3210 const int termlen = TERM_LEN(str);
3211 if (!SHARABLE_SUBSTRING_P(str, beg, len)) {
3212 str2 = rb_enc_str_new(RSTRING_PTR(str) + beg, len, rb_str_enc_get(str));
3213 if (ENC_CODERANGE(str) == ENC_CODERANGE_7BIT) {
3215 }
3216 RB_GC_GUARD(str);
3217 return str2;
3218 }
3219
3220 /* Sharing allocates a shared root as well unless str can be one itself, so
3221 * a copy is worth a larger slot only when it saves that second object. */
3222 const bool root_available = STR_SHARED_P(str) ||
3223 RB_FL_TEST_RAW(str, FL_FREEZE | STR_CHILLED) == FL_FREEZE;
3224 const size_t max_embed_size = root_available ?
3225 rb_gc_size_slot_size(sizeof(struct RString)) : STR_COPY_MAX_EMBED_SIZE;
3226 const size_t embed_size = rb_str_embed_size(len, termlen);
3227
3228 if (embed_size <= max_embed_size && rb_gc_size_allocatable_p(embed_size)) {
3229 str2 = str_alloc_embed(rb_cString, len + termlen);
3230 char *ptr2 = RSTRING(str2)->as.embed.ary;
3231 memcpy(ptr2, RSTRING_PTR(str) + beg, len);
3232 TERM_FILL(ptr2 + len, termlen);
3233
3234 STR_SET_LEN(str2, len);
3235 if (ENC_CODERANGE(str) == ENC_CODERANGE_7BIT) {
3237 }
3238
3239 RB_GC_GUARD(str);
3240 }
3241 else {
3242 str2 = str_alloc_heap(rb_cString);
3243 str_replace_shared(str2, str);
3244 RUBY_ASSERT(!STR_EMBED_P(str2));
3245 if (ENC_CODERANGE(str) != ENC_CODERANGE_7BIT) {
3246 ENC_CODERANGE_CLEAR(str2);
3247 }
3248
3249 RSTRING(str2)->as.heap.ptr += beg;
3250 if (RSTRING_LEN(str2) > len) {
3251 STR_SET_LEN(str2, len);
3252 }
3253 }
3254
3255 return str2;
3256}
3257
3258VALUE
3259rb_str_subseq(VALUE str, long beg, long len)
3260{
3261 VALUE str2 = str_subseq(str, beg, len);
3262 rb_enc_cr_str_copy_for_substr(str2, str);
3263 return str2;
3264}
3265
3266char *
3267rb_str_subpos(VALUE str, long beg, long *lenp)
3268{
3269 long len = *lenp;
3270 long slen = -1L;
3271 const long blen = RSTRING_LEN(str);
3272 rb_encoding *enc = STR_ENC_GET(str);
3273 const char *p, *s = RSTRING_PTR(str), *e = s + blen;
3274
3275 if (len < 0) return 0;
3276 if (beg < 0 && -beg < 0) return 0;
3277 if (!blen) {
3278 len = 0;
3279 }
3280 if (single_byte_optimizable(str)) {
3281 if (beg > blen) return 0;
3282 if (beg < 0) {
3283 beg += blen;
3284 if (beg < 0) return 0;
3285 }
3286 if (len > blen - beg)
3287 len = blen - beg;
3288 if (len < 0) return 0;
3289 p = s + beg;
3290 goto end;
3291 }
3292 if (beg < 0) {
3293 if (len > -beg) len = -beg;
3294 if ((ENC_CODERANGE(str) == ENC_CODERANGE_VALID) &&
3295 (-beg * rb_enc_mbmaxlen(enc) < blen / 8)) {
3296 beg = -beg;
3297 while (beg-- > len && (e = rb_enc_prev_char(s, e, e, enc)) != 0);
3298 p = e;
3299 if (!p) return 0;
3300 while (len-- > 0 && (p = rb_enc_prev_char(s, p, e, enc)) != 0);
3301 if (!p) return 0;
3302 len = e - p;
3303 goto end;
3304 }
3305 else {
3306 slen = str_strlen(str, enc);
3307 beg += slen;
3308 if (beg < 0) return 0;
3309 p = s + beg;
3310 if (len == 0) goto end;
3311 }
3312 }
3313 else if (beg > 0 && beg > blen) {
3314 return 0;
3315 }
3316 if (len == 0) {
3317 if (beg > str_strlen(str, enc)) return 0; /* str's enc */
3318 p = s + beg;
3319 }
3320#ifdef NONASCII_MASK
3321 else if (ENC_CODERANGE(str) == ENC_CODERANGE_VALID &&
3322 enc == rb_utf8_encoding()) {
3323 p = str_utf8_nth(s, e, &beg);
3324 if (beg > 0) return 0;
3325 len = str_utf8_offset(p, e, len);
3326 }
3327#endif
3328 else if (rb_enc_mbmaxlen(enc) == rb_enc_mbminlen(enc)) {
3329 int char_sz = rb_enc_mbmaxlen(enc);
3330
3331 p = s + beg * char_sz;
3332 if (p > e) {
3333 return 0;
3334 }
3335 else if (len * char_sz > e - p)
3336 len = e - p;
3337 else
3338 len *= char_sz;
3339 }
3340 else if ((p = str_nth_len(s, e, &beg, enc)) == e) {
3341 if (beg > 0) return 0;
3342 len = 0;
3343 }
3344 else {
3345 len = str_offset(p, e, len, enc, 0);
3346 }
3347 end:
3348 *lenp = len;
3349 RB_GC_GUARD(str);
3350 return (char *)p;
3351}
3352
3353static VALUE str_substr(VALUE str, long beg, long len, int empty);
3354
3355VALUE
3356rb_str_substr(VALUE str, long beg, long len)
3357{
3358 return str_substr(str, beg, len, TRUE);
3359}
3360
3361VALUE
3362rb_str_substr_two_fixnums(VALUE str, VALUE beg, VALUE len, int empty)
3363{
3364 return str_substr(str, NUM2LONG(beg), NUM2LONG(len), empty);
3365}
3366
3367static VALUE
3368str_substr(VALUE str, long beg, long len, int empty)
3369{
3370 const char *p = rb_str_subpos(str, beg, &len);
3371
3372 if (!p) return Qnil;
3373 if (!len && !empty) return Qnil;
3374
3375 beg = p - RSTRING_PTR(str);
3376
3377 VALUE str2 = str_subseq(str, beg, len);
3378 rb_enc_cr_str_copy_for_substr(str2, str);
3379 return str2;
3380}
3381
3382/* :nodoc: */
3383VALUE
3385{
3386 if (CHILLED_STRING_P(str)) {
3387 FL_UNSET_RAW(str, STR_CHILLED);
3388 }
3389
3390 if (OBJ_FROZEN(str)) return str;
3391 rb_str_resize(str, RSTRING_LEN(str));
3392 return rb_obj_freeze(str);
3393}
3394
3395/*
3396 * call-seq:
3397 * +string -> new_string or self
3398 *
3399 * Returns +self+ if +self+ is not frozen and can be mutated
3400 * without warning issuance.
3401 *
3402 * Otherwise returns <tt>self.dup</tt>, which is not frozen.
3403 *
3404 * Related: see {Freezing/Unfreezing}[rdoc-ref:String@FreezingUnfreezing].
3405 */
3406static VALUE
3407str_uplus(VALUE str)
3408{
3409 if (OBJ_FROZEN(str) || CHILLED_STRING_P(str)) {
3410 return rb_str_dup(str);
3411 }
3412 else {
3413 return str;
3414 }
3415}
3416
3417/*
3418 * call-seq:
3419 * -self -> frozen_string
3420 *
3421 * Returns a frozen string equal to +self+.
3422 *
3423 * The returned string is +self+ if and only if all of the following are true:
3424 *
3425 * - +self+ is already frozen.
3426 * - +self+ is an instance of \String (rather than of a subclass of \String)
3427 * - +self+ has no instance variables set on it.
3428 *
3429 * Otherwise, the returned string is a frozen copy of +self+.
3430 *
3431 * Returning +self+, when possible, saves duplicating +self+;
3432 * see {Data deduplication}[https://en.wikipedia.org/wiki/Data_deduplication].
3433 *
3434 * It may also save duplicating other, already-existing, strings:
3435 *
3436 * s0 = 'foo'
3437 * s1 = 'foo'
3438 * s0.object_id == s1.object_id # => false
3439 * (-s0).object_id == (-s1).object_id # => true
3440 *
3441 * Note that method #-@ is convenient for defining a constant:
3442 *
3443 * FileName = -'config/database.yml'
3444 *
3445 * While its alias #dedup is better suited for chaining:
3446 *
3447 * 'foo'.dedup.gsub!('o')
3448 *
3449 * Related: see {Freezing/Unfreezing}[rdoc-ref:String@FreezingUnfreezing].
3450 */
3451static VALUE
3452str_uminus(VALUE str)
3453{
3454 if (!BARE_STRING_P(str) && !rb_obj_frozen_p(str)) {
3455 str = rb_str_dup(str);
3456 }
3457 return rb_fstring(str);
3458}
3459
3460RUBY_ALIAS_FUNCTION(rb_str_dup_frozen(VALUE str), rb_str_new_frozen, (str))
3461#define rb_str_dup_frozen rb_str_new_frozen
3462
3463VALUE
3465{
3466 rb_check_frozen(str);
3467 if (FL_TEST(str, STR_TMPLOCK)) {
3468 rb_raise(rb_eRuntimeError, "temporal locking already locked string");
3469 }
3470 FL_SET(str, STR_TMPLOCK);
3471 return str;
3472}
3473
3474VALUE
3476{
3477 rb_check_frozen(str);
3478 if (!FL_TEST(str, STR_TMPLOCK)) {
3479 rb_raise(rb_eRuntimeError, "temporal unlocking already unlocked string");
3480 }
3481 FL_UNSET(str, STR_TMPLOCK);
3482 return str;
3483}
3484
3485VALUE
3486rb_str_locktmp_ensure(VALUE str, VALUE (*func)(VALUE), VALUE arg)
3487{
3488 rb_str_locktmp(str);
3489 return rb_ensure(func, arg, rb_str_unlocktmp, str);
3490}
3491
3492void
3494{
3495 RUBY_ASSERT(ruby_thread_has_gvl_p());
3496
3497 long capa;
3498 const int termlen = TERM_LEN(str);
3499
3500 str_modifiable(str);
3501 if (STR_SHARED_P(str)) {
3502 rb_raise(rb_eRuntimeError, "can't set length of shared string");
3503 }
3504 if (len > (capa = (long)str_capacity(str, termlen)) || len < 0) {
3505 rb_bug("probable buffer overflow: %ld for %ld", len, capa);
3506 }
3507
3508 int cr = ENC_CODERANGE(str);
3509 if (len == 0) {
3510 /* Empty string does not contain non-ASCII */
3512 }
3513 else if (cr == ENC_CODERANGE_UNKNOWN) {
3514 /* Leave unknown. */
3515 }
3516 else if (len > RSTRING_LEN(str)) {
3517 if (ENC_CODERANGE_CLEAN_P(cr)) {
3518 /* Update the coderange regarding the extended part. */
3519 const char *const prev_end = RSTRING_END(str);
3520 const char *const new_end = RSTRING_PTR(str) + len;
3521 rb_encoding *enc = rb_enc_get(str);
3522 rb_str_coderange_scan_restartable(prev_end, new_end, enc, &cr);
3523 ENC_CODERANGE_SET(str, cr);
3524 }
3525 else if (cr == ENC_CODERANGE_BROKEN) {
3526 /* May be valid now, by appended part. */
3528 }
3529 }
3530 else if (len < RSTRING_LEN(str)) {
3531 if (cr != ENC_CODERANGE_7BIT) {
3532 /* ASCII-only string is keeping after truncated. Valid
3533 * and broken may be invalid or valid, leave unknown. */
3535 }
3536 }
3537
3538 STR_SET_LEN(str, len);
3539 TERM_FILL(&RSTRING_PTR(str)[len], termlen);
3540}
3541
3542VALUE
3543rb_str_resize(VALUE str, long len)
3544{
3545 if (len < 0) {
3546 rb_raise(rb_eArgError, "negative string size (or size too big)");
3547 }
3548
3549 int independent = str_independent(str);
3550 long slen = RSTRING_LEN(str);
3551 const int termlen = TERM_LEN(str);
3552
3553 if (slen > len || (termlen != 1 && slen < len)) {
3555 }
3556
3557 {
3558 long capa;
3559 if (STR_EMBED_P(str)) {
3560 if (len == slen) return str;
3561 if (str_embed_capa(str) >= len + termlen) {
3562 STR_SET_LEN(str, len);
3563 TERM_FILL(RSTRING(str)->as.embed.ary + len, termlen);
3564 return str;
3565 }
3566 str_make_independent_expand(str, slen, len - slen, termlen);
3567 }
3568 else if (str_embed_capa(str) >= len + termlen) {
3569 capa = RSTRING(str)->as.heap.aux.capa;
3570 char *ptr = STR_HEAP_PTR(str);
3571 STR_SET_EMBED(str);
3572 if (slen > len) slen = len;
3573 if (slen > 0) MEMCPY(RSTRING(str)->as.embed.ary, ptr, char, slen);
3574 TERM_FILL(RSTRING(str)->as.embed.ary + len, termlen);
3575 STR_SET_LEN(str, len);
3576 if (independent) {
3577 SIZED_FREE_N(ptr, capa + termlen);
3578 }
3579 return str;
3580 }
3581 else if (!independent) {
3582 if (len == slen) return str;
3583 str_make_independent_expand(str, slen, len - slen, termlen);
3584 }
3585 else if ((capa = RSTRING(str)->as.heap.aux.capa) < len ||
3586 (capa - len) > (len < 1024 ? len : 1024)) {
3587 SIZED_REALLOC_N(RSTRING(str)->as.heap.ptr, char,
3588 (size_t)len + termlen, STR_HEAP_SIZE(str));
3589 RSTRING(str)->as.heap.aux.capa = len;
3590 }
3591 else if (len == slen) return str;
3592 STR_SET_LEN(str, len);
3593 TERM_FILL(RSTRING(str)->as.heap.ptr + len, termlen); /* sentinel */
3594 }
3595 return str;
3596}
3597
3598static void
3599str_ensure_available_capa(VALUE str, long len)
3600{
3601 str_modify_keep_cr(str);
3602
3603 const int termlen = TERM_LEN(str);
3604 long olen = RSTRING_LEN(str);
3605
3606 if (RB_UNLIKELY(olen > LONG_MAX - len)) {
3607 rb_raise(rb_eArgError, "string sizes too big");
3608 }
3609
3610 long total = olen + len;
3611 long capa = str_capacity(str, termlen);
3612
3613 if (capa < total) {
3614 if (total >= LONG_MAX / 2) {
3615 capa = total;
3616 }
3617 while (total > capa) {
3618 capa = 2 * capa + termlen; /* == 2*(capa+termlen)-termlen */
3619 }
3620 RESIZE_CAPA_TERM(str, capa, termlen);
3621 }
3622}
3623
3624static VALUE
3625str_buf_cat4(VALUE str, const char *ptr, long len, bool keep_cr)
3626{
3627 if (keep_cr) {
3628 str_modify_keep_cr(str);
3629 }
3630 else {
3631 rb_str_modify(str);
3632 }
3633 if (len == 0) return 0;
3634
3635 long total, olen, off = -1;
3636 char *sptr;
3637 const int termlen = TERM_LEN(str);
3638
3639 RSTRING_GETMEM(str, sptr, olen);
3640 if (ptr >= sptr && ptr <= sptr + olen) {
3641 off = ptr - sptr;
3642 }
3643
3644 long capa = str_capacity(str, termlen);
3645
3646 if (olen > LONG_MAX - len) {
3647 rb_raise(rb_eArgError, "string sizes too big");
3648 }
3649 total = olen + len;
3650 if (capa < total) {
3651 if (total >= LONG_MAX / 2) {
3652 capa = total;
3653 }
3654 while (total > capa) {
3655 capa = 2 * capa + termlen; /* == 2*(capa+termlen)-termlen */
3656 }
3657 RESIZE_CAPA_TERM(str, capa, termlen);
3658 sptr = RSTRING_PTR(str);
3659 }
3660 if (off != -1) {
3661 ptr = sptr + off;
3662 }
3663 memcpy(sptr + olen, ptr, len);
3664 STR_SET_LEN(str, total);
3665 TERM_FILL(sptr + total, termlen); /* sentinel */
3666
3667 return str;
3668}
3669
3670#define str_buf_cat(str, ptr, len) str_buf_cat4((str), (ptr), len, false)
3671#define str_buf_cat2(str, ptr) str_buf_cat4((str), (ptr), rb_strlen_lit(ptr), false)
3672
3673VALUE
3674rb_str_cat(VALUE str, const char *ptr, long len)
3675{
3676 if (len == 0) return str;
3677 if (len < 0) {
3678 rb_raise(rb_eArgError, "negative string size (or size too big)");
3679 }
3680 return str_buf_cat(str, ptr, len);
3681}
3682
3683VALUE
3684rb_str_cat_cstr(VALUE str, const char *ptr)
3685{
3686 must_not_null(ptr);
3687 return rb_str_buf_cat(str, ptr, strlen(ptr));
3688}
3689
3690static void
3691rb_str_buf_cat_byte(VALUE str, unsigned char byte)
3692{
3693 RUBY_ASSERT(RB_ENCODING_GET_INLINED(str) == ENCINDEX_ASCII_8BIT || RB_ENCODING_GET_INLINED(str) == ENCINDEX_US_ASCII);
3694
3695 // We can't write directly to shared strings without impacting others, so we must make the string independent.
3696 if (UNLIKELY(!str_independent(str))) {
3697 str_make_independent(str);
3698 }
3699
3700 long string_length = -1;
3701 const int null_terminator_length = 1;
3702 char *sptr;
3703 RSTRING_GETMEM(str, sptr, string_length);
3704
3705 // Ensure the resulting string wouldn't be too long.
3706 if (UNLIKELY(string_length > LONG_MAX - 1)) {
3707 rb_raise(rb_eArgError, "string sizes too big");
3708 }
3709
3710 long string_capacity = str_capacity(str, null_terminator_length);
3711
3712 // Get the code range before any modifications since those might clear the code range.
3713 int cr = ENC_CODERANGE(str);
3714
3715 // Check if the string has spare string_capacity to write the new byte.
3716 if (LIKELY(string_capacity >= string_length + 1)) {
3717 // In fast path we can write the new byte and note the string's new length.
3718 sptr[string_length] = byte;
3719 STR_SET_LEN(str, string_length + 1);
3720 TERM_FILL(sptr + string_length + 1, null_terminator_length);
3721 }
3722 else {
3723 // If there's not enough string_capacity, make a call into the general string concatenation function.
3724 str_buf_cat(str, (char *)&byte, 1);
3725 }
3726
3727 // If the code range is already known, we can derive the resulting code range cheaply by looking at the byte we
3728 // just appended. If the code range is unknown, but the string was empty, then we can also derive the code range
3729 // by looking at the byte we just appended. Otherwise, we'd have to scan the bytes to determine the code range so
3730 // we leave it as unknown. It cannot be broken for binary strings so we don't need to handle that option.
3731 if (cr == ENC_CODERANGE_7BIT || string_length == 0) {
3732 if (ISASCII(byte)) {
3734 }
3735 else {
3737
3738 // Promote a US-ASCII string to ASCII-8BIT when a non-ASCII byte is appended.
3739 if (UNLIKELY(RB_ENCODING_GET_INLINED(str) == ENCINDEX_US_ASCII)) {
3740 rb_enc_associate_index(str, ENCINDEX_ASCII_8BIT);
3741 }
3742 }
3743 }
3744}
3745
3746RUBY_ALIAS_FUNCTION(rb_str_buf_cat(VALUE str, const char *ptr, long len), rb_str_cat, (str, ptr, len))
3747RUBY_ALIAS_FUNCTION(rb_str_buf_cat2(VALUE str, const char *ptr), rb_str_cat_cstr, (str, ptr))
3748RUBY_ALIAS_FUNCTION(rb_str_cat2(VALUE str, const char *ptr), rb_str_cat_cstr, (str, ptr))
3749
3750static VALUE
3751rb_enc_cr_str_buf_cat(VALUE str, const char *ptr, long len,
3752 int ptr_encindex, int ptr_cr, int *ptr_cr_ret)
3753{
3754 int str_encindex = ENCODING_GET(str);
3755 int res_encindex;
3756 int str_cr, res_cr;
3757 rb_encoding *str_enc, *ptr_enc;
3758
3759 str_cr = RSTRING_LEN(str) ? ENC_CODERANGE(str) : ENC_CODERANGE_7BIT;
3760
3761 if (str_encindex == ptr_encindex) {
3762 if (str_cr != ENC_CODERANGE_UNKNOWN && ptr_cr == ENC_CODERANGE_UNKNOWN) {
3763 ptr_cr = coderange_scan(ptr, len, rb_enc_from_index(ptr_encindex));
3764 }
3765 }
3766 else {
3767 str_enc = rb_enc_from_index(str_encindex);
3768 ptr_enc = rb_enc_from_index(ptr_encindex);
3769 if (!rb_enc_asciicompat(str_enc) || !rb_enc_asciicompat(ptr_enc)) {
3770 if (len == 0)
3771 return str;
3772 if (RSTRING_LEN(str) == 0) {
3773 rb_str_buf_cat(str, ptr, len);
3774 ENCODING_CODERANGE_SET(str, ptr_encindex, ptr_cr);
3775 rb_str_change_terminator_length(str, rb_enc_mbminlen(str_enc), rb_enc_mbminlen(ptr_enc));
3776 return str;
3777 }
3778 goto incompatible;
3779 }
3780 if (ptr_cr == ENC_CODERANGE_UNKNOWN) {
3781 ptr_cr = coderange_scan(ptr, len, ptr_enc);
3782 }
3783 if (str_cr == ENC_CODERANGE_UNKNOWN) {
3784 if (ENCODING_IS_ASCII8BIT(str) || ptr_cr != ENC_CODERANGE_7BIT) {
3785 str_cr = rb_enc_str_coderange(str);
3786 }
3787 }
3788 }
3789 if (ptr_cr_ret)
3790 *ptr_cr_ret = ptr_cr;
3791
3792 if (str_encindex != ptr_encindex &&
3793 str_cr != ENC_CODERANGE_7BIT &&
3794 ptr_cr != ENC_CODERANGE_7BIT) {
3795 str_enc = rb_enc_from_index(str_encindex);
3796 ptr_enc = rb_enc_from_index(ptr_encindex);
3797 goto incompatible;
3798 }
3799
3800 if (str_cr == ENC_CODERANGE_UNKNOWN) {
3801 res_encindex = str_encindex;
3802 res_cr = ENC_CODERANGE_UNKNOWN;
3803 }
3804 else if (str_cr == ENC_CODERANGE_7BIT) {
3805 if (ptr_cr == ENC_CODERANGE_7BIT) {
3806 res_encindex = str_encindex;
3807 res_cr = ENC_CODERANGE_7BIT;
3808 }
3809 else {
3810 res_encindex = ptr_encindex;
3811 res_cr = ptr_cr;
3812 }
3813 }
3814 else if (str_cr == ENC_CODERANGE_VALID) {
3815 res_encindex = str_encindex;
3816 if (ENC_CODERANGE_CLEAN_P(ptr_cr))
3817 res_cr = str_cr;
3818 else
3819 res_cr = ptr_cr;
3820 }
3821 else { /* str_cr == ENC_CODERANGE_BROKEN */
3822 res_encindex = str_encindex;
3823 res_cr = str_cr;
3824 if (0 < len) res_cr = ENC_CODERANGE_UNKNOWN;
3825 }
3826
3827 if (len < 0) {
3828 rb_raise(rb_eArgError, "negative string size (or size too big)");
3829 }
3830 str_buf_cat(str, ptr, len);
3831 ENCODING_CODERANGE_SET(str, res_encindex, res_cr);
3832 return str;
3833
3834 incompatible:
3835 rb_raise(rb_eEncCompatError, "incompatible character encodings: %s and %s",
3836 rb_enc_inspect_name(str_enc), rb_enc_inspect_name(ptr_enc));
3838}
3839
3840VALUE
3841rb_enc_str_buf_cat(VALUE str, const char *ptr, long len, rb_encoding *ptr_enc)
3842{
3843 return rb_enc_cr_str_buf_cat(str, ptr, len,
3844 rb_enc_to_index(ptr_enc), ENC_CODERANGE_UNKNOWN, NULL);
3845}
3846
3847VALUE
3849{
3850 /* ptr must reference NUL terminated ASCII string. */
3851 int encindex = ENCODING_GET(str);
3852 rb_encoding *enc = rb_enc_from_index(encindex);
3853 if (rb_enc_asciicompat(enc)) {
3854 return rb_enc_cr_str_buf_cat(str, ptr, strlen(ptr),
3855 encindex, ENC_CODERANGE_7BIT, 0);
3856 }
3857 else {
3858 char *buf = ALLOCA_N(char, rb_enc_mbmaxlen(enc));
3859 while (*ptr) {
3860 unsigned int c = (unsigned char)*ptr;
3861 int len = rb_enc_codelen(c, enc);
3862 rb_enc_mbcput(c, buf, enc);
3863 rb_enc_cr_str_buf_cat(str, buf, len,
3864 encindex, ENC_CODERANGE_VALID, 0);
3865 ptr++;
3866 }
3867 return str;
3868 }
3869}
3870
3871VALUE
3873{
3874 int str2_cr = rb_enc_str_coderange(str2);
3875
3876 if (rb_str_enc_fastpath(str)) {
3877 switch (str2_cr) {
3878 case ENC_CODERANGE_7BIT:
3879 // If RHS is 7bit we can do simple concatenation
3880 str_buf_cat4(str, RSTRING_PTR(str2), RSTRING_LEN(str2), true);
3881 RB_GC_GUARD(str2);
3882 return str;
3884 // If RHS is valid, we can do simple concatenation if encodings are the same
3885 if (ENCODING_GET_INLINED(str) == ENCODING_GET_INLINED(str2)) {
3886 str_buf_cat4(str, RSTRING_PTR(str2), RSTRING_LEN(str2), true);
3887 int str_cr = ENC_CODERANGE(str);
3888 if (UNLIKELY(str_cr != ENC_CODERANGE_VALID)) {
3889 ENC_CODERANGE_SET(str, RB_ENC_CODERANGE_AND(str_cr, str2_cr));
3890 }
3891 RB_GC_GUARD(str2);
3892 return str;
3893 }
3894 }
3895 }
3896
3897 rb_enc_cr_str_buf_cat(str, RSTRING_PTR(str2), RSTRING_LEN(str2),
3898 ENCODING_GET(str2), str2_cr, &str2_cr);
3899
3900 ENC_CODERANGE_SET(str2, str2_cr);
3901
3902 return str;
3903}
3904
3905VALUE
3907{
3908 StringValue(str2);
3909 return rb_str_buf_append(str, str2);
3910}
3911
3912VALUE
3913rb_str_concat_literals(size_t num, const VALUE *strary)
3914{
3915 VALUE str;
3916 size_t i, s = 0;
3917 unsigned long len = 1;
3918
3919 if (UNLIKELY(!num)) return rb_str_new(0, 0);
3920 if (UNLIKELY(num == 1)) return rb_str_resurrect(strary[0]);
3921
3922 for (i = 0; i < num; ++i) { len += RSTRING_LEN(strary[i]); }
3923 str = rb_str_buf_new(len);
3924 str_enc_copy_direct(str, strary[0]);
3925
3926 for (i = s; i < num; ++i) {
3927 const VALUE v = strary[i];
3928 int encidx = ENCODING_GET(v);
3929
3930 rb_str_buf_append(str, v);
3931 if (encidx != ENCINDEX_US_ASCII) {
3932 if (ENCODING_GET_INLINED(str) == ENCINDEX_US_ASCII)
3933 rb_enc_set_index(str, encidx);
3934 }
3935 }
3936 return str;
3937}
3938
3939/*
3940 * call-seq:
3941 * concat(*objects) -> string
3942 *
3943 * :include: doc/string/concat.rdoc
3944 */
3945static VALUE
3946rb_str_concat_multi(int argc, VALUE *argv, VALUE str)
3947{
3948 str_modifiable(str);
3949
3950 if (argc == 1) {
3951 return rb_str_concat(str, argv[0]);
3952 }
3953 else if (argc > 1) {
3954 int i;
3955 VALUE arg_str = rb_str_tmp_new(0);
3956 rb_enc_copy(arg_str, str);
3957 for (i = 0; i < argc; i++) {
3958 rb_str_concat(arg_str, argv[i]);
3959 }
3960 rb_str_buf_append(str, arg_str);
3961 }
3962
3963 return str;
3964}
3965
3966/*
3967 * call-seq:
3968 * append_as_bytes(*objects) -> self
3969 *
3970 * Concatenates each object in +objects+ into +self+; returns +self+;
3971 * performs no encoding validation or conversion:
3972 *
3973 * s = 'foo'
3974 * s.append_as_bytes(" \xE2\x82") # => "foo \xE2\x82"
3975 * s.valid_encoding? # => false
3976 * s.append_as_bytes("\xAC 12")
3977 * s.valid_encoding? # => true
3978 *
3979 * When a given object is an integer,
3980 * the value is considered an 8-bit byte;
3981 * if the integer occupies more than one byte (i.e,. is greater than 255),
3982 * appends only the low-order byte (similar to String#setbyte):
3983 *
3984 * s = ""
3985 * s.append_as_bytes(0, 257) # => "\u0000\u0001"
3986 * s.bytesize # => 2
3987 *
3988 * Related: see {Modifying}[rdoc-ref:String@Modifying].
3989 */
3990
3991VALUE
3992rb_str_append_as_bytes(int argc, VALUE *argv, VALUE str)
3993{
3994 long needed_capacity = 0;
3995 volatile VALUE t0;
3996 enum ruby_value_type *types = ALLOCV_N(enum ruby_value_type, t0, argc);
3997
3998 for (int index = 0; index < argc; index++) {
3999 VALUE obj = argv[index];
4000 enum ruby_value_type type = types[index] = rb_type(obj);
4001 switch (type) {
4002 case T_FIXNUM:
4003 case T_BIGNUM:
4004 needed_capacity++;
4005 break;
4006 case T_STRING:
4007 needed_capacity += RSTRING_LEN(obj);
4008 break;
4009 default:
4010 rb_raise(
4012 "wrong argument type %"PRIsVALUE" (expected String or Integer)",
4013 rb_obj_class(obj)
4014 );
4015 break;
4016 }
4017 }
4018
4019 str_ensure_available_capa(str, needed_capacity);
4020 char *sptr = RSTRING_END(str);
4021
4022 for (int index = 0; index < argc; index++) {
4023 VALUE obj = argv[index];
4024 enum ruby_value_type type = types[index];
4025 switch (type) {
4026 case T_FIXNUM:
4027 case T_BIGNUM: {
4028 argv[index] = obj = rb_int_and(obj, INT2FIX(0xff));
4029 char byte = (char)(NUM2INT(obj) & 0xFF);
4030 *sptr = byte;
4031 sptr++;
4032 break;
4033 }
4034 case T_STRING: {
4035 const char *ptr;
4036 long len;
4037 RSTRING_GETMEM(obj, ptr, len);
4038 memcpy(sptr, ptr, len);
4039 sptr += len;
4040 break;
4041 }
4042 default:
4043 rb_bug("append_as_bytes arguments should have been validated");
4044 }
4045 }
4046
4047 STR_SET_LEN(str, RSTRING_LEN(str) + needed_capacity);
4048 TERM_FILL(sptr, TERM_LEN(str)); /* sentinel */
4049
4050 int cr = ENC_CODERANGE(str);
4051 switch (cr) {
4052 case ENC_CODERANGE_7BIT: {
4053 for (int index = 0; index < argc; index++) {
4054 VALUE obj = argv[index];
4055 enum ruby_value_type type = types[index];
4056 switch (type) {
4057 case T_FIXNUM:
4058 case T_BIGNUM: {
4059 if (!ISASCII(NUM2INT(obj))) {
4060 goto clear_cr;
4061 }
4062 break;
4063 }
4064 case T_STRING: {
4065 if (ENC_CODERANGE(obj) != ENC_CODERANGE_7BIT) {
4066 goto clear_cr;
4067 }
4068 break;
4069 }
4070 default:
4071 rb_bug("append_as_bytes arguments should have been validated");
4072 }
4073 }
4074 break;
4075 }
4077 if (ENCODING_GET_INLINED(str) == ENCINDEX_ASCII_8BIT) {
4078 goto keep_cr;
4079 }
4080 else {
4081 goto clear_cr;
4082 }
4083 break;
4084 default:
4085 goto clear_cr;
4086 break;
4087 }
4088
4089 RB_GC_GUARD(t0);
4090
4091 clear_cr:
4092 // If no fast path was hit, we clear the coderange.
4093 // append_as_bytes is predominantly meant to be used in
4094 // buffering situation, hence it's likely the coderange
4095 // will never be scanned, so it's not worth spending time
4096 // precomputing the coderange except for simple and common
4097 // situations.
4099 keep_cr:
4100 return str;
4101}
4102
4103/*
4104 * call-seq:
4105 * self << object -> self
4106 *
4107 * Appends a string representation of +object+ to +self+;
4108 * returns +self+.
4109 *
4110 * If +object+ is a string, appends it to +self+:
4111 *
4112 * s = 'foo'
4113 * s << 'bar' # => "foobar"
4114 * s # => "foobar"
4115 *
4116 * If +object+ is an integer,
4117 * its value is considered a codepoint;
4118 * converts the value to a character before concatenating:
4119 *
4120 * s = 'foo'
4121 * s << 33 # => "foo!"
4122 *
4123 * Additionally, if the codepoint is in range <tt>0..0xff</tt>
4124 * and the encoding of +self+ is Encoding::US_ASCII,
4125 * changes the encoding to Encoding::ASCII_8BIT:
4126 *
4127 * s = 'foo'.encode(Encoding::US_ASCII)
4128 * s.encoding # => #<Encoding:US-ASCII>
4129 * s << 0xff # => "foo\xFF"
4130 * s.encoding # => #<Encoding:BINARY (ASCII-8BIT)>
4131 *
4132 * Raises RangeError if that codepoint is not representable in the encoding of +self+:
4133 *
4134 * s = 'foo'
4135 * s.encoding # => <Encoding:UTF-8>
4136 * s << 0x00110000 # 1114112 out of char range (RangeError)
4137 * s = 'foo'.encode(Encoding::EUC_JP)
4138 * s << 0x00800080 # invalid codepoint 0x800080 in EUC-JP (RangeError)
4139 *
4140 * Related: see {Modifying}[rdoc-ref:String@Modifying].
4141 */
4142VALUE
4144{
4145 unsigned int code;
4146 rb_encoding *enc = STR_ENC_GET(str1);
4147 int encidx;
4148
4149 if (RB_INTEGER_TYPE_P(str2)) {
4150 if (rb_num_to_uint(str2, &code) == 0) {
4151 }
4152 else if (FIXNUM_P(str2)) {
4153 rb_raise(rb_eRangeError, "%ld out of char range", FIX2LONG(str2));
4154 }
4155 else {
4156 rb_raise(rb_eRangeError, "bignum out of char range");
4157 }
4158 }
4159 else {
4160 return rb_str_append(str1, str2);
4161 }
4162
4163 encidx = rb_ascii8bit_appendable_encoding_index(enc, code);
4164
4165 if (encidx >= 0) {
4166 rb_str_buf_cat_byte(str1, (unsigned char)code);
4167 }
4168 else {
4169 long pos = RSTRING_LEN(str1);
4170 int cr = ENC_CODERANGE(str1);
4171 int len;
4172 char *buf;
4173
4174 switch (len = rb_enc_codelen(code, enc)) {
4175 case ONIGERR_INVALID_CODE_POINT_VALUE:
4176 rb_raise(rb_eRangeError, "invalid codepoint 0x%X in %s", code, rb_enc_name(enc));
4177 break;
4178 case ONIGERR_TOO_BIG_WIDE_CHAR_VALUE:
4179 case 0:
4180 rb_raise(rb_eRangeError, "%u out of char range", code);
4181 break;
4182 }
4183 buf = ALLOCA_N(char, len + 1);
4184 rb_enc_mbcput(code, buf, enc);
4185 if (rb_enc_precise_mbclen(buf, buf + len + 1, enc) != len) {
4186 rb_raise(rb_eRangeError, "invalid codepoint 0x%X in %s", code, rb_enc_name(enc));
4187 }
4188 rb_str_resize(str1, pos+len);
4189 memcpy(RSTRING_PTR(str1) + pos, buf, len);
4190 if (cr == ENC_CODERANGE_7BIT && code > 127) {
4192 }
4193 else if (cr == ENC_CODERANGE_BROKEN) {
4195 }
4196 ENC_CODERANGE_SET(str1, cr);
4197 }
4198 return str1;
4199}
4200
4201int
4202rb_ascii8bit_appendable_encoding_index(rb_encoding *enc, unsigned int code)
4203{
4204 int encidx = rb_enc_to_index(enc);
4205
4206 if (encidx == ENCINDEX_ASCII_8BIT || encidx == ENCINDEX_US_ASCII) {
4207 /* US-ASCII automatically extended to ASCII-8BIT */
4208 if (code > 0xFF) {
4209 rb_raise(rb_eRangeError, "%u out of char range", code);
4210 }
4211 if (encidx == ENCINDEX_US_ASCII && code > 127) {
4212 return ENCINDEX_ASCII_8BIT;
4213 }
4214 return encidx;
4215 }
4216 else {
4217 return -1;
4218 }
4219}
4220
4221/*
4222 * call-seq:
4223 * prepend(*other_strings) -> new_string
4224 *
4225 * Prefixes to +self+ the concatenation of the given +other_strings+; returns +self+:
4226 *
4227 * 'baz'.prepend('foo', 'bar') # => "foobarbaz"
4228 *
4229 * Related: see {Modifying}[rdoc-ref:String@Modifying].
4230 *
4231 */
4232
4233static VALUE
4234rb_str_prepend_multi(int argc, VALUE *argv, VALUE str)
4235{
4236 str_modifiable(str);
4237
4238 if (argc == 1) {
4239 rb_str_update(str, 0L, 0L, argv[0]);
4240 }
4241 else if (argc > 1) {
4242 int i;
4243 VALUE arg_str = rb_str_tmp_new(0);
4244 rb_enc_copy(arg_str, str);
4245 for (i = 0; i < argc; i++) {
4246 rb_str_append(arg_str, argv[i]);
4247 }
4248 rb_str_update(str, 0L, 0L, arg_str);
4249 }
4250
4251 return str;
4252}
4253
4254st_index_t
4256{
4257 if (FL_TEST_RAW(str, STR_PRECOMPUTED_HASH)) {
4258 st_index_t precomputed_hash;
4259 memcpy(&precomputed_hash, RSTRING_END(str) + TERM_LEN(str), sizeof(precomputed_hash));
4260
4261 RUBY_ASSERT(precomputed_hash == str_do_hash(str));
4262 return precomputed_hash;
4263 }
4264
4265 return str_do_hash(str);
4266}
4267
4268int
4270{
4271 long len1, len2;
4272 const char *ptr1, *ptr2;
4273 RSTRING_GETMEM(str1, ptr1, len1);
4274 RSTRING_GETMEM(str2, ptr2, len2);
4275 return (len1 != len2 ||
4276 !rb_str_comparable(str1, str2) ||
4277 memcmp(ptr1, ptr2, len1) != 0);
4278}
4279
4280/*
4281 * call-seq:
4282 * hash -> integer
4283 *
4284 * :include: doc/string/hash.rdoc
4285 *
4286 */
4287
4288static VALUE
4289rb_str_hash_m(VALUE str)
4290{
4291 st_index_t hval = rb_str_hash(str);
4292 return ST2FIX(hval);
4293}
4294
4295#define lesser(a,b) (((a)>(b))?(b):(a))
4296
4297int
4299{
4300 int idx1, idx2;
4301 int rc1, rc2;
4302
4303 if (RSTRING_LEN(str1) == 0) return TRUE;
4304 if (RSTRING_LEN(str2) == 0) return TRUE;
4305 idx1 = ENCODING_GET(str1);
4306 idx2 = ENCODING_GET(str2);
4307 if (idx1 == idx2) return TRUE;
4308 rc1 = rb_enc_str_coderange(str1);
4309 rc2 = rb_enc_str_coderange(str2);
4310 if (rc1 == ENC_CODERANGE_7BIT) {
4311 if (rc2 == ENC_CODERANGE_7BIT) return TRUE;
4312 if (rb_enc_asciicompat(rb_enc_from_index(idx2)))
4313 return TRUE;
4314 }
4315 if (rc2 == ENC_CODERANGE_7BIT) {
4316 if (rb_enc_asciicompat(rb_enc_from_index(idx1)))
4317 return TRUE;
4318 }
4319 return FALSE;
4320}
4321
4322int
4324{
4325 long len1, len2;
4326 const char *ptr1, *ptr2;
4327 int retval;
4328
4329 if (str1 == str2) return 0;
4330 RSTRING_GETMEM(str1, ptr1, len1);
4331 RSTRING_GETMEM(str2, ptr2, len2);
4332 if (ptr1 == ptr2 || (retval = memcmp(ptr1, ptr2, lesser(len1, len2))) == 0) {
4333 if (len1 == len2) {
4334 if (!rb_str_comparable(str1, str2)) {
4335 if (ENCODING_GET(str1) > ENCODING_GET(str2))
4336 return 1;
4337 return -1;
4338 }
4339 return 0;
4340 }
4341 if (len1 > len2) return 1;
4342 return -1;
4343 }
4344 if (retval > 0) return 1;
4345 return -1;
4346}
4347
4348/*
4349 * call-seq:
4350 * self == other -> true or false
4351 *
4352 * Returns whether +other+ is equal to +self+.
4353 *
4354 * When +other+ is a string, returns whether +other+ has the same length and content as +self+:
4355 *
4356 * s = 'foo'
4357 * s == 'foo' # => true
4358 * s == 'food' # => false
4359 * s == 'FOO' # => false
4360 *
4361 * Returns +false+ if the two strings' encodings are not compatible:
4362 *
4363 * "\u{e4 f6 fc}".encode(Encoding::ISO_8859_1) == ("\u{c4 d6 dc}") # => false
4364 *
4365 * When +other+ is not a string:
4366 *
4367 * - If +other+ responds to method <tt>to_str</tt>,
4368 * <tt>other == self</tt> is called and its return value is returned.
4369 * - If +other+ does not respond to <tt>to_str</tt>,
4370 * +false+ is returned.
4371 *
4372 * Related: {Comparing}[rdoc-ref:String@Comparing].
4373 */
4374
4375VALUE
4377{
4378 if (str1 == str2) return Qtrue;
4379 if (!RB_TYPE_P(str2, T_STRING)) {
4380 if (!rb_respond_to(str2, idTo_str)) {
4381 return Qfalse;
4382 }
4383 return rb_equal(str2, str1);
4384 }
4385 return rb_str_eql_internal(str1, str2);
4386}
4387
4388/*
4389 * call-seq:
4390 * eql?(object) -> true or false
4391 *
4392 * :include: doc/string/eql_p.rdoc
4393 *
4394 */
4395
4396VALUE
4397rb_str_eql(VALUE str1, VALUE str2)
4398{
4399 if (str1 == str2) return Qtrue;
4400 if (!RB_TYPE_P(str2, T_STRING)) return Qfalse;
4401 return rb_str_eql_internal(str1, str2);
4402}
4403
4404/*
4405 * call-seq:
4406 * self <=> other -> -1, 0, 1, or nil
4407 *
4408 * Compares +self+ and +other+,
4409 * evaluating their _contents_, not their _lengths_.
4410 *
4411 * Returns:
4412 *
4413 * - +-1+, if +self+ is smaller.
4414 * - +0+, if the two are equal.
4415 * - +1+, if +self+ is larger.
4416 * - +nil+, if the two are incomparable.
4417 *
4418 * Examples:
4419 *
4420 * 'a' <=> 'b' # => -1
4421 * 'a' <=> 'ab' # => -1
4422 * 'a' <=> 'a' # => 0
4423 * 'b' <=> 'a' # => 1
4424 * 'ab' <=> 'a' # => 1
4425 * 'a' <=> :a # => nil
4426 *
4427 * \Class \String includes module Comparable,
4428 * each of whose methods uses String#<=> for comparison.
4429 *
4430 * Related: see {Comparing}[rdoc-ref:String@Comparing].
4431 */
4432
4433static VALUE
4434rb_str_cmp_m(VALUE str1, VALUE str2)
4435{
4436 int result;
4437 VALUE s = rb_check_string_type(str2);
4438 if (NIL_P(s)) {
4439 return rb_invcmp(str1, str2);
4440 }
4441 result = rb_str_cmp(str1, s);
4442 return INT2FIX(result);
4443}
4444
4445static VALUE str_casecmp(VALUE str1, VALUE str2);
4446static VALUE str_casecmp_p(VALUE str1, VALUE str2);
4447
4448/*
4449 * call-seq:
4450 * casecmp(other_string) -> -1, 0, 1, or nil
4451 *
4452 * Ignoring case, compares +self+ and +other_string+; returns:
4453 *
4454 * - -1 if <tt>self.downcase</tt> is smaller than <tt>other_string.downcase</tt>.
4455 * - 0 if the two are equal.
4456 * - 1 if <tt>self.downcase</tt> is larger than <tt>other_string.downcase</tt>.
4457 * - +nil+ if the two are incomparable.
4458 *
4459 * See {Case Mapping}[rdoc-ref:case_mapping.rdoc].
4460 *
4461 * Examples:
4462 *
4463 * 'foo'.casecmp('goo') # => -1
4464 * 'goo'.casecmp('foo') # => 1
4465 * 'foo'.casecmp('food') # => -1
4466 * 'food'.casecmp('foo') # => 1
4467 * 'FOO'.casecmp('foo') # => 0
4468 * 'foo'.casecmp('FOO') # => 0
4469 * 'foo'.casecmp(1) # => nil
4470 *
4471 * Related: see {Comparing}[rdoc-ref:String@Comparing].
4472 */
4473
4474VALUE
4475rb_str_casecmp(VALUE str1, VALUE str2)
4476{
4477 VALUE s = rb_check_string_type(str2);
4478 if (NIL_P(s)) {
4479 return Qnil;
4480 }
4481 return str_casecmp(str1, s);
4482}
4483
4484static VALUE
4485str_casecmp(VALUE str1, VALUE str2)
4486{
4487 long len;
4488 rb_encoding *enc;
4489 const char *p1, *p1end, *p2, *p2end;
4490
4491 enc = rb_enc_compatible(str1, str2);
4492 if (!enc) {
4493 return Qnil;
4494 }
4495
4496 p1 = RSTRING_PTR(str1); p1end = RSTRING_END(str1);
4497 p2 = RSTRING_PTR(str2); p2end = RSTRING_END(str2);
4498 if (single_byte_optimizable(str1) && single_byte_optimizable(str2)) {
4499 while (p1 < p1end && p2 < p2end) {
4500 if (*p1 != *p2) {
4501 unsigned int c1 = TOLOWER(*p1 & 0xff);
4502 unsigned int c2 = TOLOWER(*p2 & 0xff);
4503 if (c1 != c2)
4504 return INT2FIX(c1 < c2 ? -1 : 1);
4505 }
4506 p1++;
4507 p2++;
4508 }
4509 }
4510 else {
4511 while (p1 < p1end && p2 < p2end) {
4512 int l1, c1 = rb_enc_ascget(p1, p1end, &l1, enc);
4513 int l2, c2 = rb_enc_ascget(p2, p2end, &l2, enc);
4514
4515 if (0 <= c1 && 0 <= c2) {
4516 c1 = TOLOWER(c1);
4517 c2 = TOLOWER(c2);
4518 if (c1 != c2)
4519 return INT2FIX(c1 < c2 ? -1 : 1);
4520 }
4521 else {
4522 int r;
4523 l1 = rb_enc_mbclen(p1, p1end, enc);
4524 l2 = rb_enc_mbclen(p2, p2end, enc);
4525 len = l1 < l2 ? l1 : l2;
4526 r = memcmp(p1, p2, len);
4527 if (r != 0)
4528 return INT2FIX(r < 0 ? -1 : 1);
4529 if (l1 != l2)
4530 return INT2FIX(l1 < l2 ? -1 : 1);
4531 }
4532 p1 += l1;
4533 p2 += l2;
4534 }
4535 }
4536 if (p1 == p1end && p2 == p2end) return INT2FIX(0);
4537 if (p1 == p1end) return INT2FIX(-1);
4538 return INT2FIX(1);
4539}
4540
4541/*
4542 * call-seq:
4543 * casecmp?(other_string) -> true, false, or nil
4544 *
4545 * Returns +true+ if +self+ and +other_string+ are equal after
4546 * Unicode case folding, +false+ if unequal, +nil+ if incomparable.
4547 *
4548 * See {Case Mapping}[rdoc-ref:case_mapping.rdoc].
4549 *
4550 * Examples:
4551 *
4552 * 'foo'.casecmp?('goo') # => false
4553 * 'goo'.casecmp?('foo') # => false
4554 * 'foo'.casecmp?('food') # => false
4555 * 'food'.casecmp?('foo') # => false
4556 * 'FOO'.casecmp?('foo') # => true
4557 * 'foo'.casecmp?('FOO') # => true
4558 * 'foo'.casecmp?(1) # => nil
4559 *
4560 * Related: see {Comparing}[rdoc-ref:String@Comparing].
4561 */
4562
4563static VALUE
4564rb_str_casecmp_p(VALUE str1, VALUE str2)
4565{
4566 VALUE s = rb_check_string_type(str2);
4567 if (NIL_P(s)) {
4568 return Qnil;
4569 }
4570 return str_casecmp_p(str1, s);
4571}
4572
4573static VALUE
4574str_casecmp_p(VALUE str1, VALUE str2)
4575{
4576 rb_encoding *enc;
4577 VALUE folded_str1, folded_str2;
4578 VALUE fold_opt = sym_fold;
4579
4580 enc = rb_enc_compatible(str1, str2);
4581 if (!enc) {
4582 return Qnil;
4583 }
4584
4585 if (is_ascii_string(str1) && is_ascii_string(str2)) {
4586 if (RSTRING_LEN(str1) != RSTRING_LEN(str2)) return Qfalse;
4587 const char *p1 = RSTRING_PTR(str1), *p1end = RSTRING_END(str1);
4588 const char *p2 = RSTRING_PTR(str2);
4589 while (p1 < p1end) {
4590 if (*p1 != *p2 && TOLOWER((unsigned char)*p1) != TOLOWER((unsigned char)*p2)) {
4591 return Qfalse;
4592 }
4593 p1++;
4594 p2++;
4595 }
4596 return Qtrue;
4597 }
4598
4599 folded_str1 = rb_str_downcase(1, &fold_opt, str1);
4600 folded_str2 = rb_str_downcase(1, &fold_opt, str2);
4601
4602 return rb_str_eql(folded_str1, folded_str2);
4603}
4604
4605static long
4606strseq_core(const char *str_ptr, const char *str_ptr_end, long str_len,
4607 const char *sub_ptr, long sub_len, long offset, rb_encoding *enc)
4608{
4609 const char *search_start = str_ptr;
4610 long pos, search_len = str_len - offset;
4611
4612 for (;;) {
4613 const char *t;
4614 pos = rb_memsearch(sub_ptr, sub_len, search_start, search_len, enc);
4615 if (pos < 0) return pos;
4616 t = rb_enc_right_char_head(search_start, search_start+pos, str_ptr_end, enc);
4617 if (t == search_start + pos) break;
4618 search_len -= t - search_start;
4619 if (search_len <= 0) return -1;
4620 offset += t - search_start;
4621 search_start = t;
4622 }
4623 return pos + offset;
4624}
4625
4626/* found index in byte */
4627#define rb_str_index(str, sub, offset) rb_strseq_index(str, sub, offset, 0)
4628#define rb_str_byteindex(str, sub, offset) rb_strseq_index(str, sub, offset, 1)
4629
4630static long
4631rb_strseq_index(VALUE str, VALUE sub, long offset, int in_byte)
4632{
4633 const char *str_ptr, *str_ptr_end, *sub_ptr;
4634 long str_len, sub_len;
4635 rb_encoding *enc;
4636
4637 enc = rb_enc_check(str, sub);
4638 if (is_broken_string(sub)) return -1;
4639
4640 str_ptr = RSTRING_PTR(str);
4641 str_ptr_end = RSTRING_END(str);
4642 str_len = RSTRING_LEN(str);
4643 sub_ptr = RSTRING_PTR(sub);
4644 sub_len = RSTRING_LEN(sub);
4645
4646 if (str_len < sub_len) return -1;
4647
4648 if (offset != 0) {
4649 long str_len_char, sub_len_char;
4650 int single_byte = single_byte_optimizable(str);
4651 str_len_char = (in_byte || single_byte) ? str_len : str_strlen(str, enc);
4652 sub_len_char = in_byte ? sub_len : str_strlen(sub, enc);
4653 if (offset < 0) {
4654 offset += str_len_char;
4655 if (offset < 0) return -1;
4656 }
4657 if (str_len_char - offset < sub_len_char) return -1;
4658 if (!in_byte) offset = str_offset(str_ptr, str_ptr_end, offset, enc, single_byte);
4659 str_ptr += offset;
4660 }
4661 if (sub_len == 0) return offset;
4662
4663 /* need proceed one character at a time */
4664 return strseq_core(str_ptr, str_ptr_end, str_len, sub_ptr, sub_len, offset, enc);
4665}
4666
4667
4668/*
4669 * call-seq:
4670 * index(pattern, offset = 0) -> integer or nil
4671 *
4672 * :include: doc/string/index.rdoc
4673 *
4674 */
4675
4676static VALUE
4677rb_str_index_m(int argc, VALUE *argv, VALUE str)
4678{
4679 VALUE sub;
4680 VALUE initpos;
4681 rb_encoding *enc = STR_ENC_GET(str);
4682 long pos;
4683
4684 if (rb_scan_args(argc, argv, "11", &sub, &initpos) == 2) {
4685 long slen = str_strlen(str, enc); /* str's enc */
4686 pos = NUM2LONG(initpos);
4687 if (pos < 0 ? (pos += slen) < 0 : pos > slen) {
4688 if (RB_TYPE_P(sub, T_REGEXP)) {
4690 }
4691 return Qnil;
4692 }
4693 }
4694 else {
4695 pos = 0;
4696 }
4697
4698 if (RB_TYPE_P(sub, T_REGEXP)) {
4699 pos = str_offset(RSTRING_PTR(str), RSTRING_END(str), pos,
4700 enc, single_byte_optimizable(str));
4701
4702 if (rb_reg_search(sub, str, pos, 0) >= 0) {
4703 VALUE match = rb_backref_get();
4704 pos = rb_str_sublen(str, RMATCH_BEG(match, 0));
4705 return LONG2NUM(pos);
4706 }
4707 }
4708 else {
4709 StringValue(sub);
4710 pos = rb_str_index(str, sub, pos);
4711 if (pos >= 0) {
4712 pos = rb_str_sublen(str, pos);
4713 return LONG2NUM(pos);
4714 }
4715 }
4716 return Qnil;
4717}
4718
4719/* Ensure that the given pos is a valid character boundary.
4720 * Note that in this function, "character" means a code point
4721 * (Unicode scalar value), not a grapheme cluster.
4722 */
4723static void
4724str_ensure_byte_pos(VALUE str, long pos)
4725{
4726 if (!single_byte_optimizable(str)) {
4727 const char *s = RSTRING_PTR(str);
4728 const char *e = RSTRING_END(str);
4729 const char *p = s + pos;
4730 if (!at_char_boundary(s, p, e, rb_enc_get(str))) {
4731 rb_raise(rb_eIndexError,
4732 "offset %ld does not land on character boundary", pos);
4733 }
4734 }
4735}
4736
4737/*
4738 * call-seq:
4739 * byteindex(object, offset = 0) -> integer or nil
4740 *
4741 * Returns the 0-based integer index of a substring of +self+
4742 * specified by +object+ (a string or Regexp) and +offset+,
4743 * or +nil+ if there is no such substring;
4744 * the returned index is the count of _bytes_ (not characters).
4745 *
4746 * When +object+ is a string,
4747 * returns the index of the first found substring equal to +object+:
4748 *
4749 * s = 'foo' # => "foo"
4750 * s.size # => 3 # Three 1-byte characters.
4751 * s.bytesize # => 3 # Three bytes.
4752 * s.byteindex('f') # => 0
4753 * s.byteindex('o') # => 1
4754 * s.byteindex('oo') # => 1
4755 * s.byteindex('ooo') # => nil
4756 *
4757 * When +object+ is a Regexp,
4758 * returns the index of the first found substring matching +object+;
4759 * updates {Regexp-related global variables}[rdoc-ref:Regexp@Global+Variables]:
4760 *
4761 * s = 'foo'
4762 * s.byteindex(/f/) # => 0
4763 * $~ # => #<MatchData "f">
4764 * s.byteindex(/o/) # => 1
4765 * s.byteindex(/oo/) # => 1
4766 * s.byteindex(/ooo/) # => nil
4767 * $~ # => nil
4768 *
4769 * \Integer argument +offset+, if given, specifies the 0-based index
4770 * of the byte where searching is to begin.
4771 *
4772 * When +offset+ is non-negative,
4773 * searching begins at byte position +offset+:
4774 *
4775 * s = 'foo'
4776 * s.byteindex('o', 1) # => 1
4777 * s.byteindex('o', 2) # => 2
4778 * s.byteindex('o', 3) # => nil
4779 *
4780 * When +offset+ is negative, counts backward from the end of +self+:
4781 *
4782 * s = 'foo'
4783 * s.byteindex('o', -1) # => 2
4784 * s.byteindex('o', -2) # => 1
4785 * s.byteindex('o', -3) # => 1
4786 * s.byteindex('o', -4) # => nil
4787 *
4788 * Raises IndexError if the byte at +offset+ is not the first byte of a character:
4789 *
4790 * s = "\uFFFF\uFFFF" # => "\uFFFF\uFFFF"
4791 * s.size # => 2 # Two 3-byte characters.
4792 * s.bytesize # => 6 # Six bytes.
4793 * s.byteindex("\uFFFF") # => 0
4794 * s.byteindex("\uFFFF", 1) # Raises IndexError
4795 * s.byteindex("\uFFFF", 2) # Raises IndexError
4796 * s.byteindex("\uFFFF", 3) # => 3
4797 * s.byteindex("\uFFFF", 4) # Raises IndexError
4798 * s.byteindex("\uFFFF", 5) # Raises IndexError
4799 * s.byteindex("\uFFFF", 6) # => nil
4800 *
4801 * Related: see {Querying}[rdoc-ref:String@Querying].
4802 */
4803
4804static VALUE
4805rb_str_byteindex_m(int argc, VALUE *argv, VALUE str)
4806{
4807 VALUE sub;
4808 VALUE initpos;
4809 long pos;
4810
4811 if (rb_scan_args(argc, argv, "11", &sub, &initpos) == 2) {
4812 pos = NUM2LONG(initpos);
4813 long slen = RSTRING_LEN(str);
4814 if (pos < 0 ? (pos += slen) < 0 : pos > slen) {
4815 if (RB_TYPE_P(sub, T_REGEXP)) {
4817 }
4818 return Qnil;
4819 }
4820 }
4821 else {
4822 pos = 0;
4823 }
4824
4825 str_ensure_byte_pos(str, pos);
4826
4827 if (RB_TYPE_P(sub, T_REGEXP)) {
4828 if (rb_reg_search(sub, str, pos, 0) >= 0) {
4829 VALUE match = rb_backref_get();
4830 pos = RMATCH_BEG(match, 0);
4831 return LONG2NUM(pos);
4832 }
4833 }
4834 else {
4835 StringValue(sub);
4836 pos = rb_str_byteindex(str, sub, pos);
4837 if (pos >= 0) return LONG2NUM(pos);
4838 }
4839 return Qnil;
4840}
4841
4842static long
4843str_rindex(VALUE str, VALUE sub, const char *s, rb_encoding *enc)
4844{
4845 const char *hit, *adjusted, *sbeg, *e, *t;
4846 int c;
4847 long slen, searchlen;
4848
4849 sbeg = RSTRING_PTR(str);
4850 slen = RSTRING_LEN(sub);
4851 if (slen == 0) return s - sbeg;
4852 e = RSTRING_END(str);
4853 t = RSTRING_PTR(sub);
4854 c = *t & 0xff;
4855 searchlen = s - sbeg + 1;
4856
4857 if (s + slen <= e && memcmp(s, t, slen) == 0) {
4858 return s - sbeg;
4859 }
4860
4861 do {
4862 hit = memrchr(sbeg, c, searchlen);
4863 if (!hit) break;
4864 adjusted = rb_enc_left_char_head(sbeg, hit, e, enc);
4865 if (hit != adjusted) {
4866 searchlen = adjusted - sbeg;
4867 continue;
4868 }
4869 if (hit + slen <= e && memcmp(hit, t, slen) == 0)
4870 return hit - sbeg;
4871 searchlen = adjusted - sbeg;
4872 } while (searchlen > 0);
4873
4874 return -1;
4875}
4876
4877/* found index in byte */
4878static long
4879rb_str_rindex(VALUE str, VALUE sub, long pos)
4880{
4881 long len, slen;
4882 const char *sbeg, *s;
4883 rb_encoding *enc;
4884 int singlebyte;
4885
4886 enc = rb_enc_check(str, sub);
4887 if (is_broken_string(sub)) return -1;
4888 singlebyte = single_byte_optimizable(str);
4889 len = singlebyte ? RSTRING_LEN(str) : str_strlen(str, enc); /* rb_enc_check */
4890 slen = str_strlen(sub, enc); /* rb_enc_check */
4891
4892 /* substring longer than string */
4893 if (len < slen) return -1;
4894 /* character counts, so the byte tail can still be shorter than sub */
4895 if (len - pos < slen) pos = len - slen;
4896 if (len == 0) return pos;
4897
4898 sbeg = RSTRING_PTR(str);
4899
4900 if (pos == 0) {
4901 if (RSTRING_LEN(sub) <= RSTRING_LEN(str) &&
4902 memcmp(sbeg, RSTRING_PTR(sub), RSTRING_LEN(sub)) == 0) {
4903 return 0;
4904 }
4905 else {
4906 return -1;
4907 }
4908 }
4909
4910 s = str_nth(sbeg, RSTRING_END(str), pos, enc, singlebyte);
4911 return str_rindex(str, sub, s, enc);
4912}
4913
4914/*
4915 * call-seq:
4916 * rindex(pattern, offset = self.length) -> integer or nil
4917 *
4918 * :include:doc/string/rindex.rdoc
4919 *
4920 */
4921
4922static VALUE
4923rb_str_rindex_m(int argc, VALUE *argv, VALUE str)
4924{
4925 VALUE sub;
4926 VALUE initpos;
4927 rb_encoding *enc = STR_ENC_GET(str);
4928 long pos, len = str_strlen(str, enc); /* str's enc */
4929
4930 if (rb_scan_args(argc, argv, "11", &sub, &initpos) == 2) {
4931 pos = NUM2LONG(initpos);
4932 if (pos < 0 && (pos += len) < 0) {
4933 if (RB_TYPE_P(sub, T_REGEXP)) {
4935 }
4936 return Qnil;
4937 }
4938 if (pos > len) pos = len;
4939 }
4940 else {
4941 pos = len;
4942 }
4943
4944 if (RB_TYPE_P(sub, T_REGEXP)) {
4945 /* enc = rb_enc_check(str, sub); */
4946 pos = str_offset(RSTRING_PTR(str), RSTRING_END(str), pos,
4947 enc, single_byte_optimizable(str));
4948
4949 if (rb_reg_search(sub, str, pos, 1) >= 0) {
4950 VALUE match = rb_backref_get();
4951 pos = rb_str_sublen(str, RMATCH_BEG(match, 0));
4952 return LONG2NUM(pos);
4953 }
4954 }
4955 else {
4956 StringValue(sub);
4957 pos = rb_str_rindex(str, sub, pos);
4958 if (pos >= 0) {
4959 pos = rb_str_sublen(str, pos);
4960 return LONG2NUM(pos);
4961 }
4962 }
4963 return Qnil;
4964}
4965
4966static long
4967rb_str_byterindex(VALUE str, VALUE sub, long pos)
4968{
4969 long len, slen;
4970 const char *sbeg, *s;
4971 rb_encoding *enc;
4972
4973 enc = rb_enc_check(str, sub);
4974 if (is_broken_string(sub)) return -1;
4975 len = RSTRING_LEN(str);
4976 slen = RSTRING_LEN(sub);
4977
4978 /* substring longer than string */
4979 if (len < slen) return -1;
4980 if (len - pos < slen) pos = len - slen;
4981 if (len == 0) return pos;
4982
4983 sbeg = RSTRING_PTR(str);
4984
4985 if (pos == 0) {
4986 if (memcmp(sbeg, RSTRING_PTR(sub), RSTRING_LEN(sub)) == 0)
4987 return 0;
4988 else
4989 return -1;
4990 }
4991
4992 s = sbeg + pos;
4993 return str_rindex(str, sub, s, enc);
4994}
4995
4996/*
4997 * call-seq:
4998 * byterindex(object, offset = self.bytesize) -> integer or nil
4999 *
5000 * Returns the 0-based integer index of a substring of +self+
5001 * that is the _last_ match for the given +object+ (a string or Regexp) and +offset+,
5002 * or +nil+ if there is no such substring;
5003 * the returned index is the count of _bytes_ (not characters).
5004 *
5005 * When +object+ is a string,
5006 * returns the index of the _last_ found substring equal to +object+:
5007 *
5008 * s = 'foo' # => "foo"
5009 * s.size # => 3 # Three 1-byte characters.
5010 * s.bytesize # => 3 # Three bytes.
5011 * s.byterindex('f') # => 0
5012 * s.byterindex('o') # => 2
5013 * s.byterindex('oo') # => 1
5014 * s.byterindex('ooo') # => nil
5015 *
5016 * When +object+ is a Regexp,
5017 * returns the index of the last found substring matching +object+;
5018 * updates {Regexp-related global variables}[rdoc-ref:Regexp@Global+Variables]:
5019 *
5020 * s = 'foo'
5021 * s.byterindex(/f/) # => 0
5022 * $~ # => #<MatchData "f">
5023 * s.byterindex(/o/) # => 2
5024 * s.byterindex(/oo/) # => 1
5025 * s.byterindex(/ooo/) # => nil
5026 * $~ # => nil
5027 *
5028 * The last match means starting at the possible last position,
5029 * not the last of the longest matches:
5030 *
5031 * s = 'foo'
5032 * s.byterindex(/o+/) # => 2
5033 * $~ #=> #<MatchData "o">
5034 *
5035 * To get the last longest match, use a negative lookbehind:
5036 *
5037 * s = 'foo'
5038 * s.byterindex(/(?<!o)o+/) # => 1
5039 * $~ # => #<MatchData "oo">
5040 *
5041 * Or use method #byteindex with negative lookahead:
5042 *
5043 * s = 'foo'
5044 * s.byteindex(/o+(?!.*o)/) # => 1
5045 * $~ #=> #<MatchData "oo">
5046 *
5047 * \Integer argument +offset+, if given, specifies the 0-based index
5048 * of the byte where searching is to end.
5049 *
5050 * When +offset+ is non-negative,
5051 * searching ends at byte position +offset+:
5052 *
5053 * s = 'foo'
5054 * s.byterindex('o', 0) # => nil
5055 * s.byterindex('o', 1) # => 1
5056 * s.byterindex('o', 2) # => 2
5057 * s.byterindex('o', 3) # => 2
5058 *
5059 * When +offset+ is negative, counts backward from the end of +self+:
5060 *
5061 * s = 'foo'
5062 * s.byterindex('o', -1) # => 2
5063 * s.byterindex('o', -2) # => 1
5064 * s.byterindex('o', -3) # => nil
5065 *
5066 * Raises IndexError if the byte at +offset+ is not the first byte of a character:
5067 *
5068 * s = "\uFFFF\uFFFF" # => "\uFFFF\uFFFF"
5069 * s.size # => 2 # Two 3-byte characters.
5070 * s.bytesize # => 6 # Six bytes.
5071 * s.byterindex("\uFFFF") # => 3
5072 * s.byterindex("\uFFFF", 1) # Raises IndexError
5073 * s.byterindex("\uFFFF", 2) # Raises IndexError
5074 * s.byterindex("\uFFFF", 3) # => 3
5075 * s.byterindex("\uFFFF", 4) # Raises IndexError
5076 * s.byterindex("\uFFFF", 5) # Raises IndexError
5077 * s.byterindex("\uFFFF", 6) # => nil
5078 *
5079 * Related: see {Querying}[rdoc-ref:String@Querying].
5080 */
5081
5082static VALUE
5083rb_str_byterindex_m(int argc, VALUE *argv, VALUE str)
5084{
5085 VALUE sub;
5086 VALUE initpos;
5087 long pos;
5088
5089 if (rb_scan_args(argc, argv, "11", &sub, &initpos) == 2) {
5090 pos = NUM2LONG(initpos);
5091 long len = RSTRING_LEN(str);
5092 if (pos < 0 && (pos += len) < 0) {
5093 if (RB_TYPE_P(sub, T_REGEXP)) {
5095 }
5096 return Qnil;
5097 }
5098 if (pos > len) pos = len;
5099 }
5100 else {
5101 pos = RSTRING_LEN(str);
5102 }
5103
5104 str_ensure_byte_pos(str, pos);
5105
5106 if (RB_TYPE_P(sub, T_REGEXP)) {
5107 if (rb_reg_search(sub, str, pos, 1) >= 0) {
5108 VALUE match = rb_backref_get();
5109 pos = RMATCH_BEG(match, 0);
5110 return LONG2NUM(pos);
5111 }
5112 }
5113 else {
5114 StringValue(sub);
5115 pos = rb_str_byterindex(str, sub, pos);
5116 if (pos >= 0) return LONG2NUM(pos);
5117 }
5118 return Qnil;
5119}
5120
5121/*
5122 * call-seq:
5123 * self =~ other -> integer or nil
5124 *
5125 * When +other+ is a Regexp:
5126 *
5127 * - Returns the integer index (in characters) of the first match
5128 * for +self+ and +other+, or +nil+ if none;
5129 * - Updates {Regexp-related global variables}[rdoc-ref:Regexp@Global+Variables].
5130 *
5131 * Examples:
5132 *
5133 * 'foo' =~ /f/ # => 0
5134 * $~ # => #<MatchData "f">
5135 * 'foo' =~ /o/ # => 1
5136 * $~ # => #<MatchData "o">
5137 * 'foo' =~ /x/ # => nil
5138 * $~ # => nil
5139 *
5140 * Note that <tt>string =~ regexp</tt> is different from <tt>regexp =~ string</tt>
5141 * (see Regexp#=~):
5142 *
5143 * number = nil
5144 * 'no. 9' =~ /(?<number>\d+)/ # => 4
5145 * number # => nil # Not assigned.
5146 * /(?<number>\d+)/ =~ 'no. 9' # => 4
5147 * number # => "9" # Assigned.
5148 *
5149 * When +other+ is not a Regexp, returns the value
5150 * returned by <tt>other =~ self</tt>.
5151 *
5152 * Related: see {Querying}[rdoc-ref:String@Querying].
5153 */
5154
5155static VALUE
5156rb_str_match(VALUE x, VALUE y)
5157{
5158 switch (OBJ_BUILTIN_TYPE(y)) {
5159 case T_STRING:
5160 rb_raise(rb_eTypeError, "type mismatch: String given");
5161
5162 case T_REGEXP:
5163 return rb_reg_match(y, x);
5164
5165 default:
5166 return rb_funcall(y, idEqTilde, 1, x);
5167 }
5168}
5169
5170
5171static VALUE get_pat(VALUE);
5172
5173
5174/*
5175 * call-seq:
5176 * match(pattern, offset = 0) -> matchdata or nil
5177 * match(pattern, offset = 0) {|matchdata| ... } -> object
5178 *
5179 * Creates a MatchData object based on +self+ and the given arguments;
5180 * updates {Regexp Global Variables}[rdoc-ref:Regexp@Global+Variables].
5181 *
5182 * - Computes +regexp+ by converting +pattern+ (if not already a Regexp).
5183 *
5184 * regexp = Regexp.new(pattern)
5185 *
5186 * - Calls <tt>regexp.match</tt> with +self+ to compute +matchdata+.
5187 * If +offset+ is given, it is also passed (see Regexp#match).
5188 *
5189 * With no block given, returns the computed +matchdata+ or +nil+:
5190 *
5191 * 'foo'.match('f') # => #<MatchData "f">
5192 * 'foo'.match('o') # => #<MatchData "o">
5193 * 'foo'.match('x') # => nil
5194 * 'foo'.match('f', 1) # => nil
5195 * 'foo'.match('o', 1) # => #<MatchData "o">
5196 *
5197 * With a block given and computed +matchdata+ non-nil, calls the block with +matchdata+;
5198 * returns the block's return value:
5199 *
5200 * 'foo'.match(/o/) {|matchdata| matchdata } # => #<MatchData "o">
5201 *
5202 * With a block given and +nil+ +matchdata+, does not call the block:
5203 *
5204 * 'foo'.match(/x/) {|matchdata| fail 'Cannot happen' } # => nil
5205 *
5206 * Related: see {Querying}[rdoc-ref:String@Querying].
5207 */
5208
5209static VALUE
5210rb_str_match_m(int argc, VALUE *argv, VALUE str)
5211{
5212 VALUE re, result;
5213 if (argc < 1)
5214 rb_check_arity(argc, 1, 2);
5215 re = argv[0];
5216 argv[0] = str;
5217 result = rb_funcallv(get_pat(re), rb_intern("match"), argc, argv);
5218 if (!NIL_P(result) && rb_block_given_p()) {
5219 return rb_yield(result);
5220 }
5221 return result;
5222}
5223
5224/*
5225 * call-seq:
5226 * match?(pattern, offset = 0) -> true or false
5227 *
5228 * Returns whether a match is found for +self+ and the given arguments;
5229 * does not update {Regexp Global Variables}[rdoc-ref:Regexp@Global+Variables].
5230 *
5231 * Computes +regexp+ by converting +pattern+ (if not already a Regexp):
5232 *
5233 * regexp = Regexp.new(pattern)
5234 *
5235 * The search for +regexp+ in +self+ begins at the given character +offset+.
5236 * Returns +true+ if a match is found, +false+ otherwise:
5237 *
5238 * 'foo'.match?(/o/) # => true
5239 * 'foo'.match?('o') # => true
5240 * 'foo'.match?(/x/) # => false
5241 * 'foo'.match?('f', 1) # => false
5242 * 'foo'.match?('o', 1) # => true
5243 *
5244 * Related: see {Querying}[rdoc-ref:String@Querying].
5245 */
5246
5247static VALUE
5248rb_str_match_m_p(int argc, VALUE *argv, VALUE str)
5249{
5250 VALUE re;
5251 rb_check_arity(argc, 1, 2);
5252 re = get_pat(argv[0]);
5253 return rb_reg_match_p(re, str, argc > 1 ? NUM2LONG(argv[1]) : 0);
5254}
5255
5256enum neighbor_char {
5257 NEIGHBOR_NOT_CHAR,
5258 NEIGHBOR_FOUND,
5259 NEIGHBOR_WRAPPED
5260};
5261
5262static enum neighbor_char
5263enc_succ_char(char *p, long len, rb_encoding *enc)
5264{
5265 long i;
5266 int l;
5267
5268 if (rb_enc_mbminlen(enc) > 1) {
5269 /* wchar, trivial case */
5270 int r = rb_enc_precise_mbclen(p, p + len, enc), c;
5271 if (!MBCLEN_CHARFOUND_P(r)) {
5272 return NEIGHBOR_NOT_CHAR;
5273 }
5274 c = rb_enc_mbc_to_codepoint(p, p + len, enc) + 1;
5275 l = rb_enc_code_to_mbclen(c, enc);
5276 if (!l) return NEIGHBOR_NOT_CHAR;
5277 if (l != len) return NEIGHBOR_WRAPPED;
5278 rb_enc_mbcput(c, p, enc);
5279 r = rb_enc_precise_mbclen(p, p + len, enc);
5280 if (!MBCLEN_CHARFOUND_P(r)) {
5281 return NEIGHBOR_NOT_CHAR;
5282 }
5283 return NEIGHBOR_FOUND;
5284 }
5285 while (1) {
5286 for (i = len-1; 0 <= i && (unsigned char)p[i] == 0xff; i--)
5287 p[i] = '\0';
5288 if (i < 0)
5289 return NEIGHBOR_WRAPPED;
5290 ++((unsigned char*)p)[i];
5291 l = rb_enc_precise_mbclen(p, p+len, enc);
5292 if (MBCLEN_CHARFOUND_P(l)) {
5293 l = MBCLEN_CHARFOUND_LEN(l);
5294 if (l == len) {
5295 return NEIGHBOR_FOUND;
5296 }
5297 else {
5298 memset(p+l, 0xff, len-l);
5299 }
5300 }
5301 if (MBCLEN_INVALID_P(l) && i < len-1) {
5302 long len2;
5303 int l2;
5304 for (len2 = len-1; 0 < len2; len2--) {
5305 l2 = rb_enc_precise_mbclen(p, p+len2, enc);
5306 if (!MBCLEN_INVALID_P(l2))
5307 break;
5308 }
5309 memset(p+len2+1, 0xff, len-(len2+1));
5310 }
5311 }
5312}
5313
5314static enum neighbor_char
5315enc_pred_char(char *p, long len, rb_encoding *enc)
5316{
5317 long i;
5318 int l;
5319 if (rb_enc_mbminlen(enc) > 1) {
5320 /* wchar, trivial case */
5321 int r = rb_enc_precise_mbclen(p, p + len, enc), c;
5322 if (!MBCLEN_CHARFOUND_P(r)) {
5323 return NEIGHBOR_NOT_CHAR;
5324 }
5325 c = rb_enc_mbc_to_codepoint(p, p + len, enc);
5326 if (!c) return NEIGHBOR_NOT_CHAR;
5327 --c;
5328 l = rb_enc_code_to_mbclen(c, enc);
5329 if (!l) return NEIGHBOR_NOT_CHAR;
5330 if (l != len) return NEIGHBOR_WRAPPED;
5331 rb_enc_mbcput(c, p, enc);
5332 r = rb_enc_precise_mbclen(p, p + len, enc);
5333 if (!MBCLEN_CHARFOUND_P(r)) {
5334 return NEIGHBOR_NOT_CHAR;
5335 }
5336 return NEIGHBOR_FOUND;
5337 }
5338 while (1) {
5339 for (i = len-1; 0 <= i && (unsigned char)p[i] == 0; i--)
5340 p[i] = '\xff';
5341 if (i < 0)
5342 return NEIGHBOR_WRAPPED;
5343 --((unsigned char*)p)[i];
5344 l = rb_enc_precise_mbclen(p, p+len, enc);
5345 if (MBCLEN_CHARFOUND_P(l)) {
5346 l = MBCLEN_CHARFOUND_LEN(l);
5347 if (l == len) {
5348 return NEIGHBOR_FOUND;
5349 }
5350 else {
5351 memset(p+l, 0, len-l);
5352 }
5353 }
5354 if (MBCLEN_INVALID_P(l) && i < len-1) {
5355 long len2;
5356 int l2;
5357 for (len2 = len-1; 0 < len2; len2--) {
5358 l2 = rb_enc_precise_mbclen(p, p+len2, enc);
5359 if (!MBCLEN_INVALID_P(l2))
5360 break;
5361 }
5362 memset(p+len2+1, 0, len-(len2+1));
5363 }
5364 }
5365}
5366
5367/*
5368 overwrite +p+ by succeeding letter in +enc+ and returns
5369 NEIGHBOR_FOUND or NEIGHBOR_WRAPPED.
5370 When NEIGHBOR_WRAPPED, carried-out letter is stored into carry.
5371 assuming each ranges are successive, and mbclen
5372 never change in each ranges.
5373 NEIGHBOR_NOT_CHAR is returned if invalid character or the range has only one
5374 character.
5375 */
5376static enum neighbor_char
5377enc_succ_alnum_char(char *p, long len, rb_encoding *enc, char *carry)
5378{
5379 enum neighbor_char ret;
5380 unsigned int c;
5381 int ctype;
5382 int range;
5383 char save[ONIGENC_CODE_TO_MBC_MAXLEN];
5384
5385 /* skip 03A2, invalid char between GREEK CAPITAL LETTERS */
5386 int try;
5387 const int max_gaps = 1;
5388
5389 c = rb_enc_mbc_to_codepoint(p, p+len, enc);
5390 if (rb_enc_isctype(c, ONIGENC_CTYPE_DIGIT, enc))
5391 ctype = ONIGENC_CTYPE_DIGIT;
5392 else if (rb_enc_isctype(c, ONIGENC_CTYPE_ALPHA, enc))
5393 ctype = ONIGENC_CTYPE_ALPHA;
5394 else
5395 return NEIGHBOR_NOT_CHAR;
5396
5397 MEMCPY(save, p, char, len);
5398 for (try = 0; try <= max_gaps; ++try) {
5399 ret = enc_succ_char(p, len, enc);
5400 if (ret == NEIGHBOR_FOUND) {
5401 c = rb_enc_mbc_to_codepoint(p, p+len, enc);
5402 if (rb_enc_isctype(c, ctype, enc))
5403 return NEIGHBOR_FOUND;
5404 }
5405 }
5406 MEMCPY(p, save, char, len);
5407 range = 1;
5408 while (1) {
5409 MEMCPY(save, p, char, len);
5410 ret = enc_pred_char(p, len, enc);
5411 if (ret == NEIGHBOR_FOUND) {
5412 c = rb_enc_mbc_to_codepoint(p, p+len, enc);
5413 if (!rb_enc_isctype(c, ctype, enc)) {
5414 MEMCPY(p, save, char, len);
5415 break;
5416 }
5417 }
5418 else {
5419 MEMCPY(p, save, char, len);
5420 break;
5421 }
5422 range++;
5423 }
5424 if (range == 1) {
5425 return NEIGHBOR_NOT_CHAR;
5426 }
5427
5428 if (ctype != ONIGENC_CTYPE_DIGIT) {
5429 MEMCPY(carry, p, char, len);
5430 return NEIGHBOR_WRAPPED;
5431 }
5432
5433 MEMCPY(carry, p, char, len);
5434 enc_succ_char(carry, len, enc);
5435 return NEIGHBOR_WRAPPED;
5436}
5437
5438
5439static VALUE str_succ(VALUE str);
5440
5441/*
5442 * call-seq:
5443 * succ -> new_str
5444 *
5445 * :include: doc/string/succ.rdoc
5446 *
5447 */
5448
5449VALUE
5451{
5452 VALUE str;
5453 str = rb_str_new(RSTRING_PTR(orig), RSTRING_LEN(orig));
5454 rb_enc_cr_str_copy_for_substr(str, orig);
5455 return str_succ(str);
5456}
5457
5458static VALUE
5459str_succ(VALUE str)
5460{
5461 rb_encoding *enc;
5462 char *sbeg, *s, *e, *last_alnum = 0;
5463 int found_alnum = 0;
5464 long l, slen;
5465 char carry[ONIGENC_CODE_TO_MBC_MAXLEN] = "\1";
5466 long carry_pos = 0, carry_len = 1;
5467 enum neighbor_char neighbor = NEIGHBOR_FOUND;
5468
5469 slen = RSTRING_LEN(str);
5470 if (slen == 0) return str;
5471
5472 enc = STR_ENC_GET(str);
5473 sbeg = RSTRING_PTR(str);
5474 s = e = sbeg + slen;
5475
5476 while ((s = rb_enc_prev_char(sbeg, s, e, enc)) != 0) {
5477 if (neighbor == NEIGHBOR_NOT_CHAR && last_alnum) {
5478 if (ISALPHA(*last_alnum) ? ISDIGIT(*s) :
5479 ISDIGIT(*last_alnum) ? ISALPHA(*s) : 0) {
5480 break;
5481 }
5482 }
5483 l = rb_enc_precise_mbclen(s, e, enc);
5484 if (!ONIGENC_MBCLEN_CHARFOUND_P(l)) continue;
5485 l = ONIGENC_MBCLEN_CHARFOUND_LEN(l);
5486 neighbor = enc_succ_alnum_char(s, l, enc, carry);
5487 switch (neighbor) {
5488 case NEIGHBOR_NOT_CHAR:
5489 continue;
5490 case NEIGHBOR_FOUND:
5491 return str;
5492 case NEIGHBOR_WRAPPED:
5493 last_alnum = s;
5494 break;
5495 }
5496 found_alnum = 1;
5497 carry_pos = s - sbeg;
5498 carry_len = l;
5499 }
5500 if (!found_alnum) { /* str contains no alnum */
5501 s = e;
5502 while ((s = rb_enc_prev_char(sbeg, s, e, enc)) != 0) {
5503 enum neighbor_char neighbor;
5504 char tmp[ONIGENC_CODE_TO_MBC_MAXLEN];
5505 l = rb_enc_precise_mbclen(s, e, enc);
5506 if (!ONIGENC_MBCLEN_CHARFOUND_P(l)) continue;
5507 l = ONIGENC_MBCLEN_CHARFOUND_LEN(l);
5508 MEMCPY(tmp, s, char, l);
5509 neighbor = enc_succ_char(tmp, l, enc);
5510 switch (neighbor) {
5511 case NEIGHBOR_FOUND:
5512 MEMCPY(s, tmp, char, l);
5513 return str;
5514 break;
5515 case NEIGHBOR_WRAPPED:
5516 MEMCPY(s, tmp, char, l);
5517 break;
5518 case NEIGHBOR_NOT_CHAR:
5519 break;
5520 }
5521 if (rb_enc_precise_mbclen(s, s+l, enc) != l) {
5522 /* wrapped to \0...\0. search next valid char. */
5523 enc_succ_char(s, l, enc);
5524 }
5525 if (!rb_enc_asciicompat(enc)) {
5526 MEMCPY(carry, s, char, l);
5527 carry_len = l;
5528 }
5529 carry_pos = s - sbeg;
5530 }
5532 }
5533 RESIZE_CAPA(str, slen + carry_len);
5534 sbeg = RSTRING_PTR(str);
5535 s = sbeg + carry_pos;
5536 memmove(s + carry_len, s, slen - carry_pos);
5537 memmove(s, carry, carry_len);
5538 slen += carry_len;
5539 STR_SET_LEN(str, slen);
5540 TERM_FILL(&sbeg[slen], rb_enc_mbminlen(enc));
5541 rb_enc_str_coderange(str);
5542 return str;
5543}
5544
5545
5546/*
5547 * call-seq:
5548 * succ! -> self
5549 *
5550 * Like String#succ, but modifies +self+ in place; returns +self+.
5551 *
5552 * Related: see {Modifying}[rdoc-ref:String@Modifying].
5553 */
5554
5555static VALUE
5556rb_str_succ_bang(VALUE str)
5557{
5558 rb_str_modify(str);
5559 str_succ(str);
5560 return str;
5561}
5562
5563static int
5564all_digits_p(const char *s, long len)
5565{
5566 while (len-- > 0) {
5567 if (!ISDIGIT(*s)) return 0;
5568 s++;
5569 }
5570 return 1;
5571}
5572
5573static int
5574str_upto_i(VALUE str, VALUE arg)
5575{
5576 rb_yield(str);
5577 return 0;
5578}
5579
5580/*
5581 * call-seq:
5582 * upto(other_string, exclusive = false) {|string| ... } -> self
5583 * upto(other_string, exclusive = false) -> new_enumerator
5584 *
5585 * :include: doc/string/upto.rdoc
5586 *
5587 */
5588
5589static VALUE
5590rb_str_upto(int argc, VALUE *argv, VALUE beg)
5591{
5592 VALUE end, exclusive;
5593
5594 rb_scan_args(argc, argv, "11", &end, &exclusive);
5595 RETURN_ENUMERATOR(beg, argc, argv);
5596 return rb_str_upto_each(beg, end, RTEST(exclusive), str_upto_i, Qnil);
5597}
5598
5599VALUE
5600rb_str_upto_each(VALUE beg, VALUE end, int excl, int (*each)(VALUE, VALUE), VALUE arg)
5601{
5602 VALUE current, after_end;
5603 ID succ;
5604 int n, ascii;
5605 rb_encoding *enc;
5606
5607 CONST_ID(succ, "succ");
5608 StringValue(end);
5609 enc = rb_enc_check(beg, end);
5610 ascii = (is_ascii_string(beg) && is_ascii_string(end));
5611 /* single character */
5612 if (RSTRING_LEN(beg) == 1 && RSTRING_LEN(end) == 1 && ascii) {
5613 char c = RSTRING_PTR(beg)[0];
5614 char e = RSTRING_PTR(end)[0];
5615
5616 if (c > e || (excl && c == e)) return beg;
5617 for (;;) {
5618 VALUE str = rb_enc_str_new(&c, 1, enc);
5620 if ((*each)(str, arg)) break;
5621 if (!excl && c == e) break;
5622 c++;
5623 if (excl && c == e) break;
5624 }
5625 return beg;
5626 }
5627 /* both edges are all digits */
5628 if (ascii && ISDIGIT(RSTRING_PTR(beg)[0]) && ISDIGIT(RSTRING_PTR(end)[0]) &&
5629 all_digits_p(RSTRING_PTR(beg), RSTRING_LEN(beg)) &&
5630 all_digits_p(RSTRING_PTR(end), RSTRING_LEN(end))) {
5631 VALUE b, e;
5632 int width;
5633
5634 width = RSTRING_LENINT(beg);
5635 b = rb_str_to_inum(beg, 10, FALSE);
5636 e = rb_str_to_inum(end, 10, FALSE);
5637 if (FIXNUM_P(b) && FIXNUM_P(e)) {
5638 long bi = FIX2LONG(b);
5639 long ei = FIX2LONG(e);
5640 rb_encoding *usascii = rb_usascii_encoding();
5641
5642 while (bi <= ei) {
5643 if (excl && bi == ei) break;
5644 if ((*each)(rb_enc_sprintf(usascii, "%.*ld", width, bi), arg)) break;
5645 bi++;
5646 }
5647 }
5648 else {
5649 ID op = excl ? '<' : idLE;
5650 VALUE args[2], fmt = rb_fstring_lit("%.*d");
5651
5652 args[0] = INT2FIX(width);
5653 while (rb_funcall(b, op, 1, e)) {
5654 args[1] = b;
5655 if ((*each)(rb_str_format(numberof(args), args, fmt), arg)) break;
5656 b = rb_funcallv(b, succ, 0, 0);
5657 }
5658 }
5659 return beg;
5660 }
5661 /* normal case */
5662 n = rb_str_cmp(beg, end);
5663 if (n > 0 || (excl && n == 0)) return beg;
5664
5665 after_end = rb_funcallv(end, succ, 0, 0);
5666 current = str_duplicate(rb_cString, beg);
5667 while (!rb_str_equal(current, after_end)) {
5668 VALUE next = Qnil;
5669 if (excl || !rb_str_equal(current, end))
5670 next = rb_funcallv(current, succ, 0, 0);
5671 if ((*each)(current, arg)) break;
5672 if (NIL_P(next)) break;
5673 current = next;
5674 StringValue(current);
5675 if (excl && rb_str_equal(current, end)) break;
5676 if (RSTRING_LEN(current) > RSTRING_LEN(end) || RSTRING_LEN(current) == 0)
5677 break;
5678 }
5679
5680 return beg;
5681}
5682
5683VALUE
5684rb_str_upto_endless_each(VALUE beg, int (*each)(VALUE, VALUE), VALUE arg)
5685{
5686 VALUE current;
5687 ID succ;
5688
5689 CONST_ID(succ, "succ");
5690 /* both edges are all digits */
5691 if (is_ascii_string(beg) && ISDIGIT(RSTRING_PTR(beg)[0]) &&
5692 all_digits_p(RSTRING_PTR(beg), RSTRING_LEN(beg))) {
5693 VALUE b, args[2], fmt = rb_fstring_lit("%.*d");
5694 int width = RSTRING_LENINT(beg);
5695 b = rb_str_to_inum(beg, 10, FALSE);
5696 if (FIXNUM_P(b)) {
5697 long bi = FIX2LONG(b);
5698 rb_encoding *usascii = rb_usascii_encoding();
5699
5700 while (FIXABLE(bi)) {
5701 if ((*each)(rb_enc_sprintf(usascii, "%.*ld", width, bi), arg)) break;
5702 bi++;
5703 }
5704 b = LONG2NUM(bi);
5705 }
5706 args[0] = INT2FIX(width);
5707 while (1) {
5708 args[1] = b;
5709 if ((*each)(rb_str_format(numberof(args), args, fmt), arg)) break;
5710 b = rb_funcallv(b, succ, 0, 0);
5711 }
5712 }
5713 /* normal case */
5714 current = str_duplicate(rb_cString, beg);
5715 while (1) {
5716 VALUE next = rb_funcallv(current, succ, 0, 0);
5717 if ((*each)(current, arg)) break;
5718 current = next;
5719 StringValue(current);
5720 if (RSTRING_LEN(current) == 0)
5721 break;
5722 }
5723
5724 return beg;
5725}
5726
5727static int
5728include_range_i(VALUE str, VALUE arg)
5729{
5730 VALUE *argp = (VALUE *)arg;
5731 if (!rb_equal(str, *argp)) return 0;
5732 *argp = Qnil;
5733 return 1;
5734}
5735
5736VALUE
5737rb_str_include_range_p(VALUE beg, VALUE end, VALUE val, VALUE exclusive)
5738{
5739 beg = rb_str_new_frozen(beg);
5740 StringValue(end);
5741 end = rb_str_new_frozen(end);
5742 if (NIL_P(val)) return Qfalse;
5743 val = rb_check_string_type(val);
5744 if (NIL_P(val)) return Qfalse;
5745 if (rb_enc_asciicompat(STR_ENC_GET(beg)) &&
5746 rb_enc_asciicompat(STR_ENC_GET(end)) &&
5747 rb_enc_asciicompat(STR_ENC_GET(val))) {
5748 const char *bp = RSTRING_PTR(beg);
5749 const char *ep = RSTRING_PTR(end);
5750 const char *vp = RSTRING_PTR(val);
5751 if (RSTRING_LEN(beg) == 1 && RSTRING_LEN(end) == 1) {
5752 if (RSTRING_LEN(val) == 0 || RSTRING_LEN(val) > 1)
5753 return Qfalse;
5754 else {
5755 char b = *bp;
5756 char e = *ep;
5757 char v = *vp;
5758
5759 if (ISASCII(b) && ISASCII(e) && ISASCII(v)) {
5760 if (b <= v && v < e) return Qtrue;
5761 return RBOOL(!RTEST(exclusive) && v == e);
5762 }
5763 }
5764 }
5765#if 0
5766 /* both edges are all digits */
5767 if (ISDIGIT(*bp) && ISDIGIT(*ep) &&
5768 all_digits_p(bp, RSTRING_LEN(beg)) &&
5769 all_digits_p(ep, RSTRING_LEN(end))) {
5770 /* TODO */
5771 }
5772#endif
5773 }
5774 rb_str_upto_each(beg, end, RTEST(exclusive), include_range_i, (VALUE)&val);
5775
5776 return RBOOL(NIL_P(val));
5777}
5778
5779static VALUE
5780rb_str_subpat(VALUE str, VALUE re, VALUE backref)
5781{
5782 if (rb_reg_search(re, str, 0, 0) >= 0) {
5783 VALUE match = rb_backref_get();
5784 int nth = rb_reg_backref_number(match, backref);
5785 return rb_reg_nth_match(nth, match);
5786 }
5787 return Qnil;
5788}
5789
5790static VALUE
5791rb_str_aref(VALUE str, VALUE indx)
5792{
5793 long idx;
5794
5795 if (FIXNUM_P(indx)) {
5796 idx = FIX2LONG(indx);
5797 }
5798 else if (RB_TYPE_P(indx, T_REGEXP)) {
5799 return rb_str_subpat(str, indx, INT2FIX(0));
5800 }
5801 else if (RB_TYPE_P(indx, T_STRING)) {
5802 if (rb_str_index(str, indx, 0) != -1)
5803 return str_duplicate(rb_cString, indx);
5804 return Qnil;
5805 }
5806 else {
5807 /* check if indx is Range */
5808 long beg, len = str_strlen(str, NULL);
5809 switch (rb_range_beg_len(indx, &beg, &len, len, 0)) {
5810 case Qfalse:
5811 break;
5812 case Qnil:
5813 return Qnil;
5814 default:
5815 return rb_str_substr(str, beg, len);
5816 }
5817 idx = NUM2LONG(indx);
5818 }
5819
5820 return str_substr(str, idx, 1, FALSE);
5821}
5822
5823
5824/*
5825 * call-seq:
5826 * self[offset] -> new_string or nil
5827 * self[offset, size] -> new_string or nil
5828 * self[range] -> new_string or nil
5829 * self[regexp, capture = 0] -> new_string or nil
5830 * self[substring] -> new_string or nil
5831 *
5832 * :include: doc/string/aref.rdoc
5833 *
5834 */
5835
5836static VALUE
5837rb_str_aref_m(int argc, VALUE *argv, VALUE str)
5838{
5839 if (argc == 2) {
5840 if (RB_TYPE_P(argv[0], T_REGEXP)) {
5841 return rb_str_subpat(str, argv[0], argv[1]);
5842 }
5843 else {
5844 return rb_str_substr_two_fixnums(str, argv[0], argv[1], TRUE);
5845 }
5846 }
5847 rb_check_arity(argc, 1, 2);
5848 return rb_str_aref(str, argv[0]);
5849}
5850
5851VALUE
5853{
5854 char *ptr = RSTRING_PTR(str);
5855 long olen = RSTRING_LEN(str), nlen;
5856
5857 str_modifiable(str);
5858 if (len > olen) len = olen;
5859 nlen = olen - len;
5860 if (str_embed_capa(str) >= nlen + TERM_LEN(str)) {
5861 char *oldptr = ptr;
5862 size_t old_capa = RSTRING(str)->as.heap.aux.capa + TERM_LEN(str);
5863 int fl = (int)(RBASIC(str)->flags & (STR_NOEMBED|STR_SHARED|STR_NOFREE));
5864 STR_SET_EMBED(str);
5865 ptr = RSTRING(str)->as.embed.ary;
5866 memmove(ptr, oldptr + len, nlen);
5867 if (fl == STR_NOEMBED) {
5868 SIZED_FREE_N(oldptr, old_capa);
5869 }
5870 }
5871 else {
5872 if (!STR_SHARED_P(str)) {
5873 VALUE shared = heap_str_make_shared(rb_obj_class(str), str, TERM_LEN(str));
5874 rb_enc_cr_str_exact_copy(shared, str);
5876 }
5877 ptr = RSTRING(str)->as.heap.ptr += len;
5878 }
5879 STR_SET_LEN(str, nlen);
5880
5881 if (!SHARABLE_MIDDLE_SUBSTRING) {
5882 TERM_FILL(ptr + nlen, TERM_LEN(str));
5883 }
5885 return str;
5886}
5887
5888static void
5889rb_str_update_1(VALUE str, long beg, long len, VALUE val, long vbeg, long vlen)
5890{
5891 char *sptr;
5892 long slen;
5893 int cr;
5894
5895 if (beg == 0 && vlen == 0) {
5896 rb_str_drop_bytes(str, len);
5897 return;
5898 }
5899
5900 str_modify_keep_cr(str);
5901 RSTRING_GETMEM(str, sptr, slen);
5902 if (len < vlen) {
5903 /* expand string */
5904 RESIZE_CAPA(str, slen + vlen - len);
5905 sptr = RSTRING_PTR(str);
5906 }
5907
5909 cr = rb_enc_str_coderange(val);
5910 else
5912
5913 if (vlen != len) {
5914 memmove(sptr + beg + vlen,
5915 sptr + beg + len,
5916 slen - (beg + len));
5917 }
5918 if (vlen < beg && len < 0) {
5919 MEMZERO(sptr + slen, char, -len);
5920 }
5921 if (vlen > 0) {
5922 memmove(sptr + beg, RSTRING_PTR(val) + vbeg, vlen);
5923 }
5924 slen += vlen - len;
5925 STR_SET_LEN(str, slen);
5926 TERM_FILL(&sptr[slen], TERM_LEN(str));
5927 ENC_CODERANGE_SET(str, cr);
5928}
5929
5930static inline void
5931rb_str_update_0(VALUE str, long beg, long len, VALUE val)
5932{
5933 rb_str_update_1(str, beg, len, val, 0, RSTRING_LEN(val));
5934}
5935
5936void
5937rb_str_update(VALUE str, long beg, long len, VALUE val)
5938{
5939 long slen;
5940 char *p, *e;
5941 rb_encoding *enc;
5942 int singlebyte = single_byte_optimizable(str);
5943 int cr;
5944
5945 if (len < 0) rb_raise(rb_eIndexError, "negative length %ld", len);
5946
5947 StringValue(val);
5948 enc = rb_enc_check(str, val);
5949 slen = str_strlen(str, enc); /* rb_enc_check */
5950
5951 if ((slen < beg) || ((beg < 0) && (beg + slen < 0))) {
5952 rb_raise(rb_eIndexError, "index %ld out of string", beg);
5953 }
5954 if (beg < 0) {
5955 beg += slen;
5956 }
5957 RUBY_ASSERT(beg >= 0);
5958 RUBY_ASSERT(beg <= slen);
5959
5960 if (len > slen - beg) {
5961 len = slen - beg;
5962 }
5963 p = str_nth(RSTRING_PTR(str), RSTRING_END(str), beg, enc, singlebyte);
5964 if (!p) p = RSTRING_END(str);
5965 e = str_nth(p, RSTRING_END(str), len, enc, singlebyte);
5966 if (!e) e = RSTRING_END(str);
5967 /* error check */
5968 beg = p - RSTRING_PTR(str); /* physical position */
5969 len = e - p; /* physical length */
5970 rb_str_update_0(str, beg, len, val);
5971 rb_enc_associate(str, enc);
5973 if (cr != ENC_CODERANGE_BROKEN)
5974 ENC_CODERANGE_SET(str, cr);
5975}
5976
5977static void
5978rb_str_subpat_set(VALUE str, VALUE re, VALUE backref, VALUE val)
5979{
5980 int nth;
5981 VALUE match;
5982 long start, end, len;
5983 rb_encoding *enc;
5984
5985 if (rb_reg_search(re, str, 0, 0) < 0) {
5986 rb_raise(rb_eIndexError, "regexp not matched");
5987 }
5988 match = rb_backref_get();
5989 nth = rb_reg_backref_number(match, backref);
5990 int num_regs = RMATCH_NREGS(match);
5991 if ((nth >= num_regs) || ((nth < 0) && (-nth >= num_regs))) {
5992 rb_raise(rb_eIndexError, "index %d out of regexp", nth);
5993 }
5994 if (nth < 0) {
5995 nth += num_regs;
5996 }
5997
5998 start = RMATCH_BEG(match, nth);
5999 if (start == -1) {
6000 rb_raise(rb_eIndexError, "regexp group %d not matched", nth);
6001 }
6002 end = RMATCH_END(match, nth);
6003 len = end - start;
6004
6005 StringValue(val);
6006 if (start + len > RSTRING_LEN(str)) {
6007 rb_raise(rb_eRuntimeError, "string modified");
6008 }
6009
6010 enc = rb_enc_check_str(str, val);
6011 rb_str_update_0(str, start, len, val);
6012 rb_enc_associate(str, enc);
6013}
6014
6015static VALUE
6016rb_str_aset(VALUE str, VALUE indx, VALUE val)
6017{
6018 long idx, beg;
6019
6020 switch (TYPE(indx)) {
6021 case T_REGEXP:
6022 rb_str_subpat_set(str, indx, INT2FIX(0), val);
6023 return val;
6024
6025 case T_STRING:
6026 beg = rb_str_index(str, indx, 0);
6027 if (beg < 0) {
6028 rb_raise(rb_eIndexError, "string not matched");
6029 }
6030 beg = rb_str_sublen(str, beg);
6031 rb_str_update(str, beg, str_strlen(indx, NULL), val);
6032 return val;
6033
6034 default:
6035 /* check if indx is Range */
6036 {
6037 long beg, len;
6038 if (rb_range_beg_len(indx, &beg, &len, str_strlen(str, NULL), 2)) {
6039 rb_str_update(str, beg, len, val);
6040 return val;
6041 }
6042 }
6043 /* FALLTHROUGH */
6044
6045 case T_FIXNUM:
6046 idx = NUM2LONG(indx);
6047 rb_str_update(str, idx, 1, val);
6048 return val;
6049 }
6050}
6051
6052/*
6053 * call-seq:
6054 * self[index] = other_string -> new_string
6055 * self[start, length] = other_string -> new_string
6056 * self[range] = other_string -> new_string
6057 * self[regexp, capture = 0] = other_string -> new_string
6058 * self[substring] = other_string -> new_string
6059 *
6060 * :include: doc/string/aset.rdoc
6061 *
6062 */
6063
6064static VALUE
6065rb_str_aset_m(int argc, VALUE *argv, VALUE str)
6066{
6067 if (argc == 3) {
6068 if (RB_TYPE_P(argv[0], T_REGEXP)) {
6069 rb_str_subpat_set(str, argv[0], argv[1], argv[2]);
6070 }
6071 else {
6072 rb_str_update(str, NUM2LONG(argv[0]), NUM2LONG(argv[1]), argv[2]);
6073 }
6074 return argv[2];
6075 }
6076 rb_check_arity(argc, 2, 3);
6077 return rb_str_aset(str, argv[0], argv[1]);
6078}
6079
6080/*
6081 * call-seq:
6082 * insert(offset, other_string) -> self
6083 *
6084 * :include: doc/string/insert.rdoc
6085 *
6086 */
6087
6088static VALUE
6089rb_str_insert(VALUE str, VALUE idx, VALUE str2)
6090{
6091 long pos = NUM2LONG(idx);
6092
6093 if (pos == -1) {
6094 return rb_str_append(str, str2);
6095 }
6096 else if (pos < 0) {
6097 pos++;
6098 }
6099 rb_str_update(str, pos, 0, str2);
6100 return str;
6101}
6102
6103
6104/*
6105 * call-seq:
6106 * slice!(index) -> new_string or nil
6107 * slice!(start, length) -> new_string or nil
6108 * slice!(range) -> new_string or nil
6109 * slice!(regexp, capture = 0) -> new_string or nil
6110 * slice!(substring) -> new_string or nil
6111 *
6112 * Like String#[] (and its alias String#slice), except that:
6113 *
6114 * - Performs substitutions in +self+ (not in a copy of +self+).
6115 * - Returns the removed substring if any modifications were made, +nil+ otherwise.
6116 *
6117 * A few examples:
6118 *
6119 * s = 'hello'
6120 * s.slice!('e') # => "e"
6121 * s # => "hllo"
6122 * s.slice!('e') # => nil
6123 * s # => "hllo"
6124 *
6125 * Related: see {Modifying}[rdoc-ref:String@Modifying].
6126 */
6127
6128static VALUE
6129rb_str_slice_bang(int argc, VALUE *argv, VALUE str)
6130{
6131 VALUE result = Qnil;
6132 VALUE indx;
6133 long beg, len = 1;
6134 char *p;
6135
6136 rb_check_arity(argc, 1, 2);
6137 str_modify_keep_cr(str);
6138 indx = argv[0];
6139 if (RB_TYPE_P(indx, T_REGEXP)) {
6140 if (rb_reg_search(indx, str, 0, 0) < 0) return Qnil;
6141 VALUE match = rb_backref_get();
6142 int num_regs = RMATCH_NREGS(match);
6143 int nth = 0;
6144 if (argc > 1 && (nth = rb_reg_backref_number(match, argv[1])) < 0) {
6145 if ((nth += num_regs) <= 0) return Qnil;
6146 }
6147 else if (nth >= num_regs) return Qnil;
6148 beg = RMATCH_BEG(match, nth);
6149 len = RMATCH_END(match, nth) - beg;
6150 /* Converting the backref may have modified the string. */
6151 if (beg > RSTRING_LEN(str)) return Qnil;
6152 if (len > RSTRING_LEN(str) - beg) len = RSTRING_LEN(str) - beg;
6153 goto subseq;
6154 }
6155 else if (argc == 2) {
6156 beg = NUM2LONG(indx);
6157 len = NUM2LONG(argv[1]);
6158 goto num_index;
6159 }
6160 else if (FIXNUM_P(indx)) {
6161 beg = FIX2LONG(indx);
6162 if (!(p = rb_str_subpos(str, beg, &len))) return Qnil;
6163 if (!len) return Qnil;
6164 beg = p - RSTRING_PTR(str);
6165 goto subseq;
6166 }
6167 else if (RB_TYPE_P(indx, T_STRING)) {
6168 beg = rb_str_index(str, indx, 0);
6169 if (beg == -1) return Qnil;
6170 len = RSTRING_LEN(indx);
6171 result = str_duplicate(rb_cString, indx);
6172 goto squash;
6173 }
6174 else {
6175 switch (rb_range_beg_len(indx, &beg, &len, str_strlen(str, NULL), 0)) {
6176 case Qnil:
6177 return Qnil;
6178 case Qfalse:
6179 beg = NUM2LONG(indx);
6180 if (!(p = rb_str_subpos(str, beg, &len))) return Qnil;
6181 if (!len) return Qnil;
6182 beg = p - RSTRING_PTR(str);
6183 goto subseq;
6184 default:
6185 goto num_index;
6186 }
6187 }
6188
6189 num_index:
6190 if (!(p = rb_str_subpos(str, beg, &len))) return Qnil;
6191 beg = p - RSTRING_PTR(str);
6192
6193 subseq:
6194 result = rb_str_new(RSTRING_PTR(str)+beg, len);
6195 rb_enc_cr_str_copy_for_substr(result, str);
6196
6197 squash:
6198 if (len > 0) {
6199 if (beg == 0) {
6200 rb_str_drop_bytes(str, len);
6201 }
6202 else {
6203 char *sptr = RSTRING_PTR(str);
6204 long slen = RSTRING_LEN(str);
6205 if (beg + len > slen) /* pathological check */
6206 len = slen - beg;
6207 memmove(sptr + beg,
6208 sptr + beg + len,
6209 slen - (beg + len));
6210 slen -= len;
6211 STR_SET_LEN(str, slen);
6212 TERM_FILL(&sptr[slen], TERM_LEN(str));
6213 }
6214 }
6215 return result;
6216}
6217
6218static VALUE
6219get_pat(VALUE pat)
6220{
6221 VALUE val;
6222
6223 switch (OBJ_BUILTIN_TYPE(pat)) {
6224 case T_REGEXP:
6225 return pat;
6226
6227 case T_STRING:
6228 break;
6229
6230 default:
6231 val = rb_check_string_type(pat);
6232 if (NIL_P(val)) {
6233 Check_Type(pat, T_REGEXP);
6234 }
6235 pat = val;
6236 }
6237
6238 return rb_reg_regcomp(pat);
6239}
6240
6241static VALUE
6242get_pat_quoted(VALUE pat, int check)
6243{
6244 VALUE val;
6245
6246 switch (OBJ_BUILTIN_TYPE(pat)) {
6247 case T_REGEXP:
6248 return pat;
6249
6250 case T_STRING:
6251 break;
6252
6253 default:
6254 val = rb_check_string_type(pat);
6255 if (NIL_P(val)) {
6256 Check_Type(pat, T_REGEXP);
6257 }
6258 pat = val;
6259 }
6260 if (check && is_broken_string(pat)) {
6261 rb_exc_raise(rb_reg_check_preprocess(pat));
6262 }
6263 return pat;
6264}
6265
6266static long
6267rb_pat_search0(VALUE pat, VALUE str, long pos, int set_backref_str, VALUE *match)
6268{
6269 if (BUILTIN_TYPE(pat) == T_STRING) {
6270 pos = rb_str_byteindex(str, pat, pos);
6271 if (set_backref_str) {
6272 if (pos >= 0) {
6273 str = rb_str_new_frozen_String(str);
6274 VALUE match_data = rb_backref_set_string(str, pos, RSTRING_LEN(pat));
6275 if (match) {
6276 *match = match_data;
6277 }
6278 }
6279 else {
6281 }
6282 }
6283 return pos;
6284 }
6285 else {
6286 return rb_reg_search0(pat, str, pos, 0, set_backref_str, match);
6287 }
6288}
6289
6290static long
6291rb_pat_search(VALUE pat, VALUE str, long pos, int set_backref_str)
6292{
6293 return rb_pat_search0(pat, str, pos, set_backref_str, NULL);
6294}
6295
6296
6297/*
6298 * call-seq:
6299 * sub!(pattern, replacement) -> self or nil
6300 * sub!(pattern) {|match| ... } -> self or nil
6301 *
6302 * Like String#sub, except that:
6303 *
6304 * - Changes are made to +self+, not to copy of +self+.
6305 * - Returns +self+ if any changes are made, +nil+ otherwise.
6306 *
6307 * Related: see {Modifying}[rdoc-ref:String@Modifying].
6308 */
6309
6310static VALUE
6311rb_str_sub_bang(int argc, VALUE *argv, VALUE str)
6312{
6313 VALUE pat, repl, hash = Qnil;
6314 int iter = 0;
6315 long plen;
6316 int min_arity = rb_block_given_p() ? 1 : 2;
6317 long beg;
6318
6319 rb_check_arity(argc, min_arity, 2);
6320 if (argc == 1) {
6321 iter = 1;
6322 }
6323 else {
6324 repl = argv[1];
6325 if (!RB_TYPE_P(repl, T_STRING)) {
6326 hash = rb_check_hash_type(repl);
6327 if (NIL_P(hash)) {
6328 StringValue(repl);
6329 }
6330 }
6331 }
6332
6333 pat = get_pat_quoted(argv[0], 1);
6334
6335 str_modifiable(str);
6336 beg = rb_pat_search(pat, str, 0, 1);
6337 if (beg >= 0) {
6338 rb_encoding *enc;
6339 int cr = ENC_CODERANGE(str);
6340 long beg0, end0;
6341 VALUE match, match0 = Qnil;
6342 char *p, *rp;
6343 long len, rlen;
6344
6345 match = rb_backref_get();
6346 if (RB_TYPE_P(pat, T_STRING)) {
6347 beg0 = beg;
6348 end0 = beg0 + RSTRING_LEN(pat);
6349 match0 = pat;
6350 }
6351 else {
6352 beg0 = RMATCH_BEG(match, 0);
6353 end0 = RMATCH_END(match, 0);
6354 if (iter) match0 = rb_reg_nth_match(0, match);
6355 }
6356
6357 if (iter || !NIL_P(hash)) {
6358 p = RSTRING_PTR(str); len = RSTRING_LEN(str);
6359
6360 if (iter) {
6361 repl = rb_obj_as_string(rb_yield(match0));
6362 }
6363 else {
6364 repl = rb_hash_aref(hash, rb_str_subseq(str, beg0, end0 - beg0));
6365 repl = rb_obj_as_string(repl);
6366 }
6367 str_mod_check(str, p, len);
6368 rb_check_frozen(str);
6369 }
6370 else {
6371 repl = rb_reg_regsub_match(repl, str, match);
6372 }
6373
6374 enc = rb_enc_compatible(str, repl);
6375 if (!enc) {
6376 rb_encoding *str_enc = STR_ENC_GET(str);
6377 p = RSTRING_PTR(str); len = RSTRING_LEN(str);
6378 if (coderange_scan(p, beg0, str_enc) != ENC_CODERANGE_7BIT ||
6379 coderange_scan(p+end0, len-end0, str_enc) != ENC_CODERANGE_7BIT) {
6380 rb_raise(rb_eEncCompatError, "incompatible character encodings: %s and %s",
6381 rb_enc_inspect_name(str_enc),
6382 rb_enc_inspect_name(STR_ENC_GET(repl)));
6383 }
6384 enc = STR_ENC_GET(repl);
6385 }
6386 rb_str_modify(str);
6387 rb_enc_associate(str, enc);
6389 int cr2 = ENC_CODERANGE(repl);
6390 if (cr2 == ENC_CODERANGE_BROKEN ||
6391 (cr == ENC_CODERANGE_VALID && cr2 == ENC_CODERANGE_7BIT))
6393 else
6394 cr = cr2;
6395 }
6396 plen = end0 - beg0;
6397 rlen = RSTRING_LEN(repl);
6398 len = RSTRING_LEN(str);
6399 if (rlen > plen) {
6400 RESIZE_CAPA(str, len + rlen - plen);
6401 }
6402 p = RSTRING_PTR(str);
6403 if (rlen != plen) {
6404 memmove(p + beg0 + rlen, p + beg0 + plen, len - beg0 - plen);
6405 }
6406 rp = RSTRING_PTR(repl);
6407 memmove(p + beg0, rp, rlen);
6408 len += rlen - plen;
6409 STR_SET_LEN(str, len);
6410 TERM_FILL(&RSTRING_PTR(str)[len], TERM_LEN(str));
6411 ENC_CODERANGE_SET(str, cr);
6412
6413 RB_GC_GUARD(match);
6414
6415 return str;
6416 }
6417 return Qnil;
6418}
6419
6420
6421/*
6422 * call-seq:
6423 * sub(pattern, replacement) -> new_string
6424 * sub(pattern) {|match| ... } -> new_string
6425 *
6426 * :include: doc/string/sub.rdoc
6427 */
6428
6429static VALUE
6430rb_str_sub(int argc, VALUE *argv, VALUE str)
6431{
6432 str = str_duplicate(rb_cString, str);
6433 rb_str_sub_bang(argc, argv, str);
6434 return str;
6435}
6436
6437static VALUE
6438str_gsub(int argc, VALUE *argv, VALUE str, int bang)
6439{
6440 VALUE pat, val = Qnil, repl, match0 = Qnil, dest, hash = Qnil, match = Qnil;
6441 long beg, beg0, end0;
6442 long offset, blen, slen, len, last;
6443 enum {STR, ITER, FAST_MAP, MAP} mode = STR;
6444 char *sp, *cp;
6445 int need_backref_str = -1;
6446 rb_encoding *str_enc;
6447
6448 switch (argc) {
6449 case 1:
6450 RETURN_ENUMERATOR(str, argc, argv);
6451 mode = ITER;
6452 break;
6453 case 2:
6454 repl = argv[1];
6455 if (!RB_TYPE_P(repl, T_STRING)) {
6456 hash = rb_check_hash_type(repl);
6457 if (NIL_P(hash)) {
6458 StringValue(repl);
6459 }
6460 else if (rb_hash_default_unredefined(hash) && !FL_TEST_RAW(hash, RHASH_PROC_DEFAULT)) {
6461 mode = FAST_MAP;
6462 }
6463 else {
6464 mode = MAP;
6465 }
6466 }
6467 break;
6468 default:
6469 rb_error_arity(argc, 1, 2);
6470 }
6471
6472 pat = get_pat_quoted(argv[0], 1);
6473 beg = rb_pat_search0(pat, str, 0, need_backref_str, &match);
6474
6475 if (beg < 0) {
6476 if (bang) return Qnil; /* no match, no substitution */
6477 return str_duplicate(rb_cString, str);
6478 }
6479 if (bang) str_modify_keep_cr(str);
6480
6481 offset = 0;
6482 blen = RSTRING_LEN(str) + 30; /* len + margin */
6483 dest = rb_str_buf_new(blen);
6484 sp = RSTRING_PTR(str);
6485 slen = RSTRING_LEN(str);
6486 cp = sp;
6487 str_enc = STR_ENC_GET(str);
6488 rb_enc_associate(dest, str_enc);
6489 ENC_CODERANGE_SET(dest, rb_enc_asciicompat(str_enc) ? ENC_CODERANGE_7BIT : ENC_CODERANGE_VALID);
6490
6491 do {
6492 if (RB_TYPE_P(pat, T_STRING)) {
6493 beg0 = beg;
6494 end0 = beg0 + RSTRING_LEN(pat);
6495 match0 = pat;
6496 }
6497 else {
6498 beg0 = RMATCH_BEG(match, 0);
6499 end0 = RMATCH_END(match, 0);
6500 if (mode == ITER) match0 = rb_reg_nth_match(0, match);
6501 }
6502
6503 if (mode != STR) {
6504 if (mode == ITER) {
6505 val = rb_obj_as_string(rb_yield(match0));
6506 }
6507 else {
6508 struct RString fake_str = {RBASIC_INIT};
6509 VALUE key;
6510 if (mode == FAST_MAP) {
6511 // It is safe to use a fake_str here because we established that it won't escape,
6512 // as it's only used for `rb_hash_aref` and we checked the hash doesn't have a
6513 // default proc.
6514 key = setup_fake_str(&fake_str, sp + beg0, end0 - beg0, ENCODING_GET_INLINED(str));
6515 }
6516 else {
6517 key = rb_str_subseq(str, beg0, end0 - beg0);
6518 }
6519 val = rb_hash_aref(hash, key);
6520 val = rb_obj_as_string(val);
6521 }
6522 str_mod_check(str, sp, slen);
6523 if (val == dest) { /* paranoid check [ruby-dev:24827] */
6524 rb_raise(rb_eRuntimeError, "block should not cheat");
6525 }
6526 }
6527 else if (need_backref_str) {
6528 val = rb_reg_regsub_match(repl, str, match);
6529 if (need_backref_str < 0) {
6530 need_backref_str = val != repl;
6531 }
6532 }
6533 else {
6534 val = repl;
6535 }
6536
6537 len = beg0 - offset; /* copy pre-match substr */
6538 if (len) {
6539 rb_enc_str_buf_cat(dest, cp, len, str_enc);
6540 }
6541
6542 rb_str_buf_append(dest, val);
6543
6544 last = offset;
6545 offset = end0;
6546 if (beg0 == end0) {
6547 /*
6548 * Always consume at least one character of the input string
6549 * in order to prevent infinite loops.
6550 */
6551 if (RSTRING_LEN(str) <= end0) break;
6552 len = rb_enc_fast_mbclen(RSTRING_PTR(str)+end0, RSTRING_END(str), str_enc);
6553 rb_enc_str_buf_cat(dest, RSTRING_PTR(str)+end0, len, str_enc);
6554 offset = end0 + len;
6555 }
6556 cp = RSTRING_PTR(str) + offset;
6557 if (offset > RSTRING_LEN(str)) break;
6558
6559 // In FAST_MAP and STR mode the backref can't escape so we can re-use the MatchData safely.
6560 if (mode != FAST_MAP && mode != STR) {
6561 match = Qnil;
6562 }
6563 beg = rb_pat_search0(pat, str, offset, need_backref_str, &match);
6564
6565 RB_GC_GUARD(match);
6566 } while (beg >= 0);
6567
6568 if (RSTRING_LEN(str) > offset) {
6569 rb_enc_str_buf_cat(dest, cp, RSTRING_LEN(str) - offset, str_enc);
6570 }
6571 rb_pat_search0(pat, str, last, 1, &match);
6572 if (bang) {
6573 str_shared_replace(str, dest);
6574 }
6575 else {
6576 str = dest;
6577 }
6578
6579 return str;
6580}
6581
6582
6583/*
6584 * call-seq:
6585 * gsub!(pattern, replacement) -> self or nil
6586 * gsub!(pattern) {|match| ... } -> self or nil
6587 * gsub!(pattern) -> an_enumerator
6588 *
6589 * Like String#gsub, except that:
6590 *
6591 * - Performs substitutions in +self+ (not in a copy of +self+).
6592 * - Returns +self+ if any substitutions were performed, +nil+ otherwise.
6593 *
6594 * Related: see {Modifying}[rdoc-ref:String@Modifying].
6595 */
6596
6597static VALUE
6598rb_str_gsub_bang(int argc, VALUE *argv, VALUE str)
6599{
6600 str_modifiable(str);
6601 return str_gsub(argc, argv, str, 1);
6602}
6603
6604
6605/*
6606 * call-seq:
6607 * gsub(pattern, replacement) -> new_string
6608 * gsub(pattern) {|match| ... } -> new_string
6609 * gsub(pattern) -> enumerator
6610 *
6611 * Returns a copy of +self+ with zero or more substrings replaced.
6612 *
6613 * Argument +pattern+ may be a string or a Regexp;
6614 * argument +replacement+ may be a string or a Hash.
6615 * Varying types for the argument values makes this method very versatile.
6616 *
6617 * Below are some simple examples;
6618 * for many more examples, see {Substitution Methods}[rdoc-ref:String@Substitution+Methods].
6619 *
6620 * With arguments +pattern+ and string +replacement+ given,
6621 * replaces each matching substring with the given +replacement+ string:
6622 *
6623 * s = 'abracadabra'
6624 * s.gsub('ab', 'AB') # => "ABracadABra"
6625 * s.gsub(/[a-c]/, 'X') # => "XXrXXXdXXrX"
6626 *
6627 * With arguments +pattern+ and hash +replacement+ given,
6628 * replaces each matching substring with a value from the given +replacement+ hash,
6629 * or removes it:
6630 *
6631 * h = {'a' => 'A', 'b' => 'B', 'c' => 'C'}
6632 * s.gsub(/[a-c]/, h) # => "ABrACAdABrA" # 'a', 'b', 'c' replaced.
6633 * s.gsub(/[a-d]/, h) # => "ABrACAABrA" # 'd' removed.
6634 *
6635 * With argument +pattern+ and a block given,
6636 * calls the block with each matching substring;
6637 * replaces that substring with the block's return value:
6638 *
6639 * s.gsub(/[a-d]/) {|substring| substring.upcase }
6640 * # => "ABrACADABrA"
6641 *
6642 * With argument +pattern+ and no block given,
6643 * returns a new Enumerator.
6644 *
6645 * Related: see {Converting to New String}[rdoc-ref:String@Converting+to+New+String].
6646 */
6647
6648static VALUE
6649rb_str_gsub(int argc, VALUE *argv, VALUE str)
6650{
6651 return str_gsub(argc, argv, str, 0);
6652}
6653
6654
6655/*
6656 * call-seq:
6657 * replace(other_string) -> self
6658 *
6659 * Replaces the contents of +self+ with the contents of +other_string+;
6660 * returns +self+:
6661 *
6662 * s = 'foo' # => "foo"
6663 * s.replace('bar') # => "bar"
6664 *
6665 * Related: see {Modifying}[rdoc-ref:String@Modifying].
6666 */
6667
6668VALUE
6670{
6671 str_modifiable(str);
6672 if (str == str2) return str;
6673
6674 StringValue(str2);
6675 str_discard(str);
6676 return str_replace(str, str2);
6677}
6678
6679/*
6680 * call-seq:
6681 * clear -> self
6682 *
6683 * Removes the contents of +self+:
6684 *
6685 * s = 'foo'
6686 * s.clear # => ""
6687 * s # => ""
6688 *
6689 * Related: see {Modifying}[rdoc-ref:String@Modifying].
6690 */
6691
6692static VALUE
6693rb_str_clear(VALUE str)
6694{
6695 str_discard(str);
6696 STR_SET_EMBED(str);
6697 STR_SET_LEN(str, 0);
6698 RSTRING_PTR(str)[0] = 0;
6699 if (rb_enc_asciicompat(STR_ENC_GET(str)))
6701 else
6703 return str;
6704}
6705
6706/*
6707 * call-seq:
6708 * chr -> string
6709 *
6710 * :include: doc/string/chr.rdoc
6711 *
6712 */
6713
6714static VALUE
6715rb_str_chr(VALUE str)
6716{
6717 return rb_str_substr(str, 0, 1);
6718}
6719
6720/*
6721 * call-seq:
6722 * getbyte(index) -> integer or nil
6723 *
6724 * :include: doc/string/getbyte.rdoc
6725 *
6726 */
6727VALUE
6728rb_str_getbyte(VALUE str, VALUE index)
6729{
6730 long pos = NUM2LONG(index);
6731
6732 if (pos < 0)
6733 pos += RSTRING_LEN(str);
6734 if (pos < 0 || RSTRING_LEN(str) <= pos)
6735 return Qnil;
6736
6737 return INT2FIX((unsigned char)RSTRING_PTR(str)[pos]);
6738}
6739
6740/*
6741 * call-seq:
6742 * setbyte(index, integer) -> integer
6743 *
6744 * Sets the byte at zero-based offset +index+ to the value of the given +integer+;
6745 * returns +integer+:
6746 *
6747 * s = 'xyzzy'
6748 * s.setbyte(2, 129) # => 129
6749 * s # => "xy\x81zy"
6750 *
6751 * Related: see {Modifying}[rdoc-ref:String@Modifying].
6752 */
6753VALUE
6754rb_str_setbyte(VALUE str, VALUE index, VALUE value)
6755{
6756 long pos = NUM2LONG(index);
6757 char *ptr, *head, *left = 0;
6758 rb_encoding *enc;
6759 int cr = ENC_CODERANGE_UNKNOWN, width, nlen;
6760
6761 VALUE v = rb_to_int(value);
6762 VALUE w = rb_int_and(v, INT2FIX(0xff));
6763 char byte = (char)(NUM2INT(w) & 0xFF);
6764
6765 long len = RSTRING_LEN(str);
6766 if (pos < -len || len <= pos)
6767 rb_raise(rb_eIndexError, "index %ld out of string", pos);
6768 if (pos < 0)
6769 pos += len;
6770
6771 if (!str_independent(str))
6772 str_make_independent(str);
6773 enc = STR_ENC_GET(str);
6774 head = RSTRING_PTR(str);
6775 ptr = &head[pos];
6776 if (!STR_EMBED_P(str)) {
6777 cr = ENC_CODERANGE(str);
6778 switch (cr) {
6779 case ENC_CODERANGE_7BIT:
6780 left = ptr;
6781 *ptr = byte;
6782 if (ISASCII(byte)) goto end;
6783 nlen = rb_enc_precise_mbclen(left, head+len, enc);
6784 if (!MBCLEN_CHARFOUND_P(nlen))
6786 else
6788 goto end;
6790 left = rb_enc_left_char_head(head, ptr, head+len, enc);
6791 width = rb_enc_precise_mbclen(left, head+len, enc);
6792 *ptr = byte;
6793 nlen = rb_enc_precise_mbclen(left, head+len, enc);
6794 if (!MBCLEN_CHARFOUND_P(nlen))
6796 else if (MBCLEN_CHARFOUND_LEN(nlen) != width || ISASCII(byte))
6798 goto end;
6799 }
6800 }
6802 *ptr = byte;
6803
6804 end:
6805 return value;
6806}
6807
6808static inline bool
6809str_bit_offset_out_of_range(long byte_len, uint64_t bit_offset)
6810{
6811 /* Compare byte indexes to avoid overflowing byte_len * CHAR_BIT. */
6812 return bit_offset / CHAR_BIT >= (uint64_t)byte_len;
6813}
6814
6815/*
6816 * Keep both the full bit offset and its long representation. Most calls use a
6817 * Fixnum-sized offset and can stay on the original long fast path; only large
6818 * Bignum offsets need the uint64_t path below. This matters on platforms
6819 * where long is narrower than the address space, such as 32-bit and LLP64.
6820 */
6822 uint64_t value;
6823 long long_value;
6824 bool fits_long;
6825};
6826
6827static inline struct str_bit_offset
6828str_bit_offset_from_index(VALUE index)
6829{
6830 VALUE integer = rb_to_int(index);
6831 struct str_bit_offset offset;
6832
6833 /*
6834 * FIXNUM_P only decides whether the common long path is immediately usable.
6835 * This covers practically all offsets on LP64 platforms; Bignum offsets
6836 * are still accepted below when they fit in uint64_t, mainly for platforms
6837 * with 32-bit long where large strings can have Bignum bit offsets.
6838 */
6839 if (FIXNUM_P(integer)) {
6840 offset.long_value = FIX2LONG(integer);
6841 if (offset.long_value < 0) {
6842 rb_raise(rb_eIndexError, "bit index out of range");
6843 }
6844 offset.value = (uint64_t)offset.long_value;
6845 offset.fits_long = true;
6846 return offset;
6847 }
6848
6849 RUBY_ASSERT(RB_TYPE_P(integer, T_BIGNUM));
6850 if (rb_int_negative_p(integer)) {
6851 rb_raise(rb_eIndexError, "bit index out of range");
6852 }
6853 if (rb_cmpint(rb_int_cmp(integer, ULL2NUM(UINT64_MAX)), integer, ULL2NUM(UINT64_MAX)) > 0) {
6854 rb_raise(rb_eArgError, "bit index out of representable range");
6855 }
6856
6857 offset.value = (uint64_t)NUM2ULL(integer);
6858 if (offset.value <= (uint64_t)LONG_MAX) {
6859 offset.long_value = (long)offset.value;
6860 offset.fits_long = true;
6861 }
6862 else {
6863 offset.long_value = 0;
6864 offset.fits_long = false;
6865 }
6866 return offset;
6867}
6868
6869/*
6870 * Bit lengths share the offset's representable range.
6871 * A negative length is an ArgumentError rather than an IndexError.
6872 */
6873static uint64_t
6874str_bit_length_from_index(VALUE index)
6875{
6876 VALUE integer = rb_to_int(index);
6877
6878 if (FIXNUM_P(integer)) {
6879 long value = FIX2LONG(integer);
6880 if (value < 0) {
6881 rb_raise(rb_eArgError, "negative bit length");
6882 }
6883 return (uint64_t)value;
6884 }
6885
6886 RUBY_ASSERT(RB_TYPE_P(integer, T_BIGNUM));
6887 if (rb_int_negative_p(integer)) {
6888 rb_raise(rb_eArgError, "negative bit length");
6889 }
6890 if (rb_cmpint(rb_int_cmp(integer, ULL2NUM(UINT64_MAX)), integer, ULL2NUM(UINT64_MAX)) > 0) {
6891 rb_raise(rb_eArgError, "bit length out of representable range");
6892 }
6893 return (uint64_t)NUM2ULL(integer);
6894}
6895
6896static inline uint64_t
6897str_bit_size(long byte_len)
6898{
6899 /*
6900 * byte_len * CHAR_BIT overflows uint64_t only for byte_len >= 2**61 which cannot
6901 * be allocated. Saturate so that unreachable cases cannot wrap.
6902 */
6903 if ((uint64_t)byte_len > UINT64_MAX / CHAR_BIT) return UINT64_MAX;
6904 return (uint64_t)byte_len * CHAR_BIT;
6905}
6906
6908 uint64_t beg;
6909 uint64_t end_exclusive; /* meaningful only when end_open is false */
6910 bool end_open; /* a nil end: the region runs to the end of self */
6911};
6912
6913/*
6914 * Coerce a bit Range's endpoints to bit offsets. This may run arbitrary Ruby
6915 * (Integer#to_int on the endpoints), so it does NOT read the string's length:
6916 * The caller must resolve the length only after this returns, otherwise
6917 * to_int that reallocates self would leave a stale size.
6918 */
6919static void
6920str_bit_range_to_offsets(VALUE range, struct str_bit_range *out)
6921{
6922 VALUE beg_v, end_v;
6923 int excl;
6924
6925 /*
6926 * We don't use rb_range_beg_len: it counts negative endpoints from the end,
6927 * which is an IndexError for bit positions, and it is limited to long instead
6928 * of uint64_t.
6929 */
6930 rb_range_values(range, &beg_v, &end_v, &excl);
6931
6932 out->beg = NIL_P(beg_v) ? 0 : str_bit_offset_from_index(beg_v).value;
6933 if (NIL_P(end_v)) {
6934 out->end_open = true;
6935 out->end_exclusive = 0;
6936 }
6937 else {
6938 uint64_t end = str_bit_offset_from_index(end_v).value;
6939 out->end_open = false;
6940 /*
6941 * The saturation loses one position only for an inclusive end of
6942 * 2**64-1, which lies beyond any real string either way.
6943 */
6944 out->end_exclusive = (excl || end == UINT64_MAX) ? end : end + 1;
6945 }
6946}
6947
6948/*
6949 * Turn a coerced Range into (beg, len) against the now-current total bit size.
6950 * The length is deliberately not clamped to the bits available, so a reading
6951 * caller can clamp while a writing caller detects the overrun and raises.
6952 */
6953static bool
6954str_bit_range_resolve(const struct str_bit_range *range, uint64_t total_bits, uint64_t *begp, uint64_t *lenp)
6955{
6956 uint64_t beg = range->beg;
6957 if (beg > total_bits) return false;
6958
6959 uint64_t end_exclusive = range->end_open ? total_bits : range->end_exclusive;
6960 if (end_exclusive < beg) end_exclusive = beg;
6961
6962 *begp = beg;
6963 *lenp = end_exclusive - beg;
6964 return true;
6965}
6966
6967static bool
6968str_lsb_first_from_opts(VALUE opts)
6969{
6970 static ID keywords[1];
6971 VALUE vlsb_first;
6972
6973 if (!keywords[0]) {
6974 keywords[0] = rb_intern_const("lsb_first");
6975 }
6976
6977 rb_get_kwargs(opts, keywords, 0, 1, &vlsb_first);
6978 if (vlsb_first == Qundef || vlsb_first == Qtrue) {
6979 return true;
6980 }
6981 if (vlsb_first == Qfalse) {
6982 return false;
6983 }
6984 rb_raise(rb_eArgError, "lsb_first must be true or false");
6985 UNREACHABLE_RETURN(false);
6986}
6987
6988static bool
6989str_lsb_first(int argc, VALUE *argv, VALUE *index)
6990{
6991 VALUE opts;
6992
6993 rb_scan_args(argc, argv, "1:", index, &opts);
6994 return str_lsb_first_from_opts(opts);
6995}
6996
6997static inline uint64_t
6998str_logical_to_physical_bit64(uint64_t logical, bool lsb_first)
6999{
7000 return lsb_first ? logical : ((logical & ~(uint64_t)7) | (7 - (logical & 7)));
7001}
7002
7003static inline long
7004str_logical_to_physical_bit(long logical, bool lsb_first)
7005{
7006 return lsb_first ? logical : ((logical & ~7L) | (7 - (logical & 7L)));
7007}
7008
7010 long byte_index;
7011 unsigned int bit_offset;
7012};
7013
7014static inline struct str_bit_location
7015str_bit_location_from_offset(uint64_t logical, bool lsb_first)
7016{
7017 /*
7018 * When long is 32-bit, a bit offset for a large string can be a Bignum
7019 * while the byte index still fits in long, which is RSTRING_LEN's type.
7020 */
7021 uint64_t physical = str_logical_to_physical_bit64(logical, lsb_first);
7022 struct str_bit_location location;
7023 location.byte_index = (long)(physical / CHAR_BIT);
7024 location.bit_offset = (unsigned int)(physical % CHAR_BIT);
7025 return location;
7026}
7027
7028static inline int
7029str_get_bit(const char *ptr, long bit_index)
7030{
7031 return (((unsigned char)ptr[bit_index / CHAR_BIT]) >> (bit_index % CHAR_BIT)) & 1;
7032}
7033
7034static inline int
7035str_get_bit_location(const char *ptr, struct str_bit_location location)
7036{
7037 return (((unsigned char)ptr[location.byte_index]) >> location.bit_offset) & 1;
7038}
7039
7040static int
7041str_bit_get(int argc, VALUE *argv, VALUE str)
7042{
7043 VALUE index;
7044 bool lsb_first = str_lsb_first(argc, argv, &index);
7045 struct str_bit_offset offset = str_bit_offset_from_index(index);
7046
7047 if (str_bit_offset_out_of_range(RSTRING_LEN(str), offset.value)) {
7048 return -1;
7049 }
7050
7051 if (offset.fits_long) {
7052 return str_get_bit(RSTRING_PTR(str), str_logical_to_physical_bit(offset.long_value, lsb_first));
7053 }
7054 else {
7055 return str_get_bit_location(RSTRING_PTR(str), str_bit_location_from_offset(offset.value, lsb_first));
7056 }
7057}
7058
7059/*
7060 * call-seq:
7061 * bit_get(offset, lsb_first: true) -> 0, 1, or nil
7062 *
7063 * :include: doc/string/bit_get.rdoc
7064 *
7065 */
7066static VALUE
7067rb_str_bit_get(int argc, VALUE *argv, VALUE str)
7068{
7069 int bit = str_bit_get(argc, argv, str);
7070 return bit < 0 ? Qnil : INT2FIX(bit);
7071}
7072
7073/*
7074 * call-seq:
7075 * bit_set?(offset, lsb_first: true) -> true, false, or nil
7076 *
7077 * :include: doc/string/bit_set_p.rdoc
7078 *
7079 */
7080static VALUE
7081rb_str_bit_set_p(int argc, VALUE *argv, VALUE str)
7082{
7083 int bit = str_bit_get(argc, argv, str);
7084 return bit < 0 ? Qnil : RBOOL(bit);
7085}
7086
7087enum str_bit_mutation {
7088 STR_BIT_SET,
7089 STR_BIT_CLEAR,
7090 STR_BIT_FLIP
7091};
7092
7093/*
7094 * Mask for the logical in-byte positions lo..hi (0 <= lo <= hi <= 7) of one
7095 * byte. A contiguous logical run stays contiguous within a byte under both
7096 * numbering conventions; MSB-first only mirrors it.
7097 */
7098static inline unsigned char
7099str_bit_region_byte_mask(unsigned int lo, unsigned int hi, bool lsb_first)
7100{
7101 if (lsb_first) {
7102 return (unsigned char)((0xFFu >> (7 - hi)) & (0xFFu << lo));
7103 }
7104 else {
7105 return (unsigned char)((0xFFu >> lo) & (0xFFu << (7 - hi)));
7106 }
7107}
7108
7109static inline void
7110str_apply_bit_mask(unsigned char *byte, unsigned char mask, enum str_bit_mutation mutation)
7111{
7112 switch (mutation) {
7113 case STR_BIT_SET:
7114 *byte |= mask;
7115 break;
7116 case STR_BIT_CLEAR:
7117 *byte &= (unsigned char)~mask;
7118 break;
7119 case STR_BIT_FLIP:
7120 *byte ^= mask;
7121 break;
7122 }
7123}
7124
7125/* The caller has bounds-checked [beg, beg+len) and called rb_str_modify. */
7126static void
7127str_mutate_bit_region(unsigned char *ptr, uint64_t beg, uint64_t len, bool lsb_first, enum str_bit_mutation mutation)
7128{
7129 uint64_t first_bit = beg;
7130 uint64_t last_bit = beg + len - 1;
7131 long first_byte = (long)(first_bit / CHAR_BIT);
7132 long last_byte = (long)(last_bit / CHAR_BIT);
7133 unsigned int first_off = (unsigned int)(first_bit % CHAR_BIT);
7134 unsigned int last_off = (unsigned int)(last_bit % CHAR_BIT);
7135
7136 if (first_byte == last_byte) {
7137 str_apply_bit_mask(ptr + first_byte, str_bit_region_byte_mask(first_off, last_off, lsb_first), mutation);
7138 return;
7139 }
7140
7141 str_apply_bit_mask(ptr + first_byte, str_bit_region_byte_mask(first_off, 7, lsb_first), mutation);
7142 long middle_len = last_byte - first_byte - 1;
7143 if (middle_len > 0) {
7144 unsigned char *middle = ptr + first_byte + 1;
7145 switch (mutation) {
7146 case STR_BIT_SET:
7147 memset(middle, 0xFF, middle_len);
7148 break;
7149 case STR_BIT_CLEAR:
7150 memset(middle, 0, middle_len);
7151 break;
7152 case STR_BIT_FLIP:
7153 /*
7154 * Byte loop on purpose: the compiler auto-vectorizes it (verified on gcc 13.3
7155 * and clang 18.1 with x86_64), and being read-modify-write, the flip is memory-bound,
7156 * so a manual word-at-a-time XOR loop was measured to be no faster.
7157 */
7158 for (long i = 0; i < middle_len; i++) {
7159 middle[i] ^= 0xFF;
7160 }
7161 break;
7162 }
7163 }
7164 str_apply_bit_mask(ptr + last_byte, str_bit_region_byte_mask(0, last_off, lsb_first), mutation);
7165}
7166
7167static VALUE
7168str_mutate_single_bit(VALUE str, VALUE index, bool lsb_first, enum str_bit_mutation mutation)
7169{
7170 struct str_bit_offset offset = str_bit_offset_from_index(index);
7171 struct str_bit_location location;
7172 long bit_index;
7173 unsigned char *ptr;
7174 unsigned char mask;
7175
7176 rb_check_frozen(str);
7177
7178 if (str_bit_offset_out_of_range(RSTRING_LEN(str), offset.value)) {
7179 rb_raise(rb_eIndexError, "bit index out of range");
7180 }
7181
7182 rb_str_modify(str);
7183 ptr = (unsigned char *)RSTRING_PTR(str);
7184 if (offset.fits_long) {
7185 bit_index = str_logical_to_physical_bit(offset.long_value, lsb_first);
7186 mask = (unsigned char)(1u << (bit_index % CHAR_BIT));
7187 location.byte_index = bit_index / CHAR_BIT;
7188 }
7189 else {
7190 location = str_bit_location_from_offset(offset.value, lsb_first);
7191 mask = (unsigned char)(1u << location.bit_offset);
7192 }
7193
7194 str_apply_bit_mask(ptr + location.byte_index, mask, mutation);
7195 return str;
7196}
7197
7198static VALUE
7199str_mutate_bit(int argc, VALUE *argv, VALUE str, enum str_bit_mutation mutation)
7200{
7201 VALUE target, length_v, opts;
7202 uint64_t beg = 0, len = 0;
7203
7204 /* Count positional arguments so that an explicit nil is not mistaken for an omitted one. */
7205 int nargs = rb_scan_args(argc, argv, "11:", &target, &length_v, &opts);
7206 bool lsb_first = str_lsb_first_from_opts(opts);
7207
7208 bool is_range = rb_obj_is_kind_of(target, rb_cRange);
7209 if (nargs == 1 && !is_range) {
7210 return str_mutate_single_bit(str, target, lsb_first, mutation);
7211 }
7212
7213 struct str_bit_range range = {0};
7214 struct str_bit_offset offset;
7215 if (is_range) {
7216 if (nargs == 2) {
7217 rb_raise(rb_eArgError, "bit length not allowed with a Range");
7218 }
7219 str_bit_range_to_offsets(target, &range);
7220 }
7221 else {
7222 offset = str_bit_offset_from_index(target);
7223 len = str_bit_length_from_index(length_v);
7224 }
7225
7226 /* Even a zero-length write requires a mutable receiver. */
7227 rb_check_frozen(str);
7228
7229 /*
7230 * A region that begins past the end is out of range even when it is
7231 * empty, and one that runs past the end is not allowed to silently
7232 * shrink: both are errors for a mutation, unlike the clamping reads.
7233 * An empty region whose start is within 0..bitsize writes nothing.
7234 */
7235 uint64_t total_bits = str_bit_size(RSTRING_LEN(str));
7236 if (is_range) {
7237 if (!str_bit_range_resolve(&range, total_bits, &beg, &len) || len > total_bits - beg) {
7238 rb_raise(rb_eIndexError, "bit range out of range");
7239 }
7240 }
7241 else {
7242 beg = offset.value;
7243 if (beg > total_bits || len > total_bits - beg) {
7244 rb_raise(rb_eIndexError, "bit range out of range");
7245 }
7246 }
7247
7248 if (len == 0) return str;
7249
7250 rb_str_modify(str);
7251 str_mutate_bit_region((unsigned char *)RSTRING_PTR(str), beg, len, lsb_first, mutation);
7252 return str;
7253}
7254
7255/*
7256 * call-seq:
7257 * bit_set(offset, lsb_first: true) -> self
7258 * bit_set(offset, length, lsb_first: true) -> self
7259 * bit_set(range, lsb_first: true) -> self
7260 *
7261 * :include: doc/string/bit_set.rdoc
7262 *
7263 */
7264static VALUE
7265rb_str_bit_set(int argc, VALUE *argv, VALUE str)
7266{
7267 return str_mutate_bit(argc, argv, str, STR_BIT_SET);
7268}
7269
7270/*
7271 * call-seq:
7272 * bit_clear(offset, lsb_first: true) -> self
7273 * bit_clear(offset, length, lsb_first: true) -> self
7274 * bit_clear(range, lsb_first: true) -> self
7275 *
7276 * :include: doc/string/bit_clear.rdoc
7277 *
7278 */
7279static VALUE
7280rb_str_bit_clear(int argc, VALUE *argv, VALUE str)
7281{
7282 return str_mutate_bit(argc, argv, str, STR_BIT_CLEAR);
7283}
7284
7285/*
7286 * call-seq:
7287 * bit_flip(offset, lsb_first: true) -> self
7288 * bit_flip(offset, length, lsb_first: true) -> self
7289 * bit_flip(range, lsb_first: true) -> self
7290 *
7291 * :include: doc/string/bit_flip.rdoc
7292 *
7293 */
7294static VALUE
7295rb_str_bit_flip(int argc, VALUE *argv, VALUE str)
7296{
7297 return str_mutate_bit(argc, argv, str, STR_BIT_FLIP);
7298}
7299
7300static uint64_t
7301str_count_bits(const unsigned char *ptr, long len)
7302{
7303 uint64_t count = 0;
7304 long off = 0;
7305 long unrolled_end = len & ~31L;
7306 long aligned_end = len & ~7L;
7307
7308 // 32 bytes (256 bits) at a time
7309 for (; off < unrolled_end; off += 32) {
7310 uint64_t w0, w1, w2, w3;
7311 memcpy(&w0, ptr + off, 8);
7312 memcpy(&w1, ptr + off + 8, 8);
7313 memcpy(&w2, ptr + off + 16, 8);
7314 memcpy(&w3, ptr + off + 24, 8);
7315 count += rb_popcount64(w0);
7316 count += rb_popcount64(w1);
7317 count += rb_popcount64(w2);
7318 count += rb_popcount64(w3);
7319 }
7320
7321 // 8 bytes (64 bits) at a time
7322 for (; off < aligned_end; off += 8) {
7323 uint64_t word;
7324 memcpy(&word, ptr + off, 8);
7325 count += rb_popcount64(word);
7326 }
7327
7328 // remaining bytes
7329 if (off < len) {
7330 uint64_t word = 0;
7331 int shift = 0;
7332 for (; off < len; off++, shift += CHAR_BIT) {
7333 word |= (uint64_t)ptr[off] << shift;
7334 }
7335 count += rb_popcount64(word);
7336 }
7337
7338 return count;
7339}
7340
7341static uint64_t
7342str_count_bits_region(const unsigned char *ptr, uint64_t beg, uint64_t len, bool lsb_first)
7343{
7344 uint64_t first_bit = beg;
7345 uint64_t last_bit = beg + len - 1;
7346 long first_byte = (long)(first_bit / CHAR_BIT);
7347 long last_byte = (long)(last_bit / CHAR_BIT);
7348 unsigned int first_off = (unsigned int)(first_bit % CHAR_BIT);
7349 unsigned int last_off = (unsigned int)(last_bit % CHAR_BIT);
7350
7351 if (first_byte == last_byte) {
7352 return rb_popcount32((uint32_t)(ptr[first_byte] & str_bit_region_byte_mask(first_off, last_off, lsb_first)));
7353 }
7354
7355 uint64_t count = rb_popcount32((uint32_t)(ptr[first_byte] & str_bit_region_byte_mask(first_off, 7, lsb_first)));
7356 count += str_count_bits(ptr + first_byte + 1, last_byte - first_byte - 1);
7357 count += rb_popcount32((uint32_t)(ptr[last_byte] & str_bit_region_byte_mask(0, last_off, lsb_first)));
7358 return count;
7359}
7360
7361/*
7362 * call-seq:
7363 * bit_count -> integer
7364 * bit_count(offset, length, lsb_first: true) -> integer
7365 * bit_count(range, lsb_first: true) -> integer
7366 *
7367 * :include: doc/string/bit_count.rdoc
7368 *
7369 */
7370static VALUE
7371rb_str_bit_count(int argc, VALUE *argv, VALUE str)
7372{
7373 VALUE v0, v1, opts;
7374 uint64_t beg = 0, len = 0;
7375
7376 /* Count positional arguments so that an explicit nil is not mistaken for an omitted one. */
7377 int nargs = rb_scan_args(argc, argv, "02:", &v0, &v1, &opts);
7378 /*
7379 * A whole-string popcount is independent of bit numbering.
7380 * no-(offset|range)-argument form only validates lsb_first.
7381 */
7382 bool lsb_first = str_lsb_first_from_opts(opts);
7383
7384 if (nargs == 0) {
7385 return ULL2NUM(str_count_bits((const unsigned char *)RSTRING_PTR(str), RSTRING_LEN(str)));
7386 }
7387
7388 bool is_range = rb_obj_is_kind_of(v0, rb_cRange);
7389 struct str_bit_range range = {0};
7390 if (is_range) {
7391 if (nargs == 2) {
7392 rb_raise(rb_eArgError, "bit length not allowed with a Range");
7393 }
7394 str_bit_range_to_offsets(v0, &range);
7395 }
7396 else if (nargs == 1) {
7397 rb_raise(rb_eArgError, "no bit length given");
7398 }
7399 else {
7400 beg = str_bit_offset_from_index(v0).value;
7401 len = str_bit_length_from_index(v1);
7402 }
7403
7404 const unsigned char *ptr = (const unsigned char *)RSTRING_PTR(str);
7405 uint64_t total_bits = str_bit_size(RSTRING_LEN(str));
7406 if (is_range) {
7407 if (!str_bit_range_resolve(&range, total_bits, &beg, &len)) {
7408 return INT2FIX(0);
7409 }
7410 }
7411 else if (beg >= total_bits) {
7412 return INT2FIX(0);
7413 }
7414
7415 /* Reads clamp: only the part of the region that exists is counted. */
7416 if (len > total_bits - beg) len = total_bits - beg;
7417 if (len == 0) return INT2FIX(0);
7418 return ULL2NUM(str_count_bits_region(ptr, beg, len, lsb_first));
7419}
7420
7421static void
7422str_check_bitwise_length(VALUE str, VALUE other)
7423{
7424 if (RSTRING_LEN(str) != RSTRING_LEN(other)) {
7425 rb_raise(rb_eArgError, "operands must have the same length (%ld vs %ld)",
7426 RSTRING_LEN(str), RSTRING_LEN(other));
7427 }
7428}
7429
7430static VALUE
7431str_bitwise_result(VALUE str)
7432{
7433 long len = RSTRING_LEN(str);
7434 VALUE result = rb_str_buf_new(len);
7435 rb_str_resize(result, len);
7436 rb_enc_associate(result, rb_ascii8bit_encoding());
7437 ENC_CODERANGE_CLEAR(result);
7438 return result;
7439}
7440
7441#define STR_DEFINE_UNARY_BITWISE_KERNEL(name, expr_word, expr_byte) \
7442 static void \
7443 name(unsigned char *dst, const unsigned char *src, long len) \
7444 { \
7445 long off = 0; \
7446 long unrolled_end = len & ~31L; \
7447 long aligned_end = len & ~7L; \
7448 for (; off < unrolled_end; off += 32) { \
7449 uint64_t s0, s1, s2, s3; \
7450 memcpy(&s0, src + off, 8); \
7451 memcpy(&s1, src + off + 8, 8); \
7452 memcpy(&s2, src + off + 16, 8); \
7453 memcpy(&s3, src + off + 24, 8); \
7454 s0 = (expr_word(s0)); \
7455 s1 = (expr_word(s1)); \
7456 s2 = (expr_word(s2)); \
7457 s3 = (expr_word(s3)); \
7458 memcpy(dst + off, &s0, 8); \
7459 memcpy(dst + off + 8, &s1, 8); \
7460 memcpy(dst + off + 16, &s2, 8); \
7461 memcpy(dst + off + 24, &s3, 8); \
7462 } \
7463 for (; off < aligned_end; off += 8) { \
7464 uint64_t word; \
7465 memcpy(&word, src + off, 8); \
7466 word = (expr_word(word)); \
7467 memcpy(dst + off, &word, 8); \
7468 } \
7469 for (; off < len; off++) dst[off] = (expr_byte(src[off])); \
7470 }
7471
7472#define STR_DEFINE_BINARY_BITWISE_KERNEL(name, expr_word, expr_byte) \
7473 static void \
7474 name(unsigned char *dst, const unsigned char *lhs, \
7475 const unsigned char *rhs, long len) \
7476 { \
7477 long off = 0; \
7478 long unrolled_end = len & ~31L; \
7479 long aligned_end = len & ~7L; \
7480 for (; off < unrolled_end; off += 32) { \
7481 uint64_t l0, l1, l2, l3, r0, r1, r2, r3; \
7482 memcpy(&l0, lhs + off, 8); memcpy(&r0, rhs + off, 8); \
7483 memcpy(&l1, lhs + off + 8, 8); memcpy(&r1, rhs + off + 8, 8); \
7484 memcpy(&l2, lhs + off + 16, 8); memcpy(&r2, rhs + off + 16, 8); \
7485 memcpy(&l3, lhs + off + 24, 8); memcpy(&r3, rhs + off + 24, 8); \
7486 l0 = expr_word(l0, r0); \
7487 l1 = expr_word(l1, r1); \
7488 l2 = expr_word(l2, r2); \
7489 l3 = expr_word(l3, r3); \
7490 memcpy(dst + off, &l0, 8); \
7491 memcpy(dst + off + 8, &l1, 8); \
7492 memcpy(dst + off + 16, &l2, 8); \
7493 memcpy(dst + off + 24, &l3, 8); \
7494 } \
7495 for (; off < aligned_end; off += 8) { \
7496 uint64_t lhs_word, rhs_word; \
7497 memcpy(&lhs_word, lhs + off, 8); \
7498 memcpy(&rhs_word, rhs + off, 8); \
7499 lhs_word = expr_word(lhs_word, rhs_word); \
7500 memcpy(dst + off, &lhs_word, 8); \
7501 } \
7502 for (; off < len; off++) dst[off] = expr_byte(lhs[off], rhs[off]); \
7503 }
7504
7505#define STR_BITWISE_NOT_WORD(x) (~(x))
7506#define STR_BITWISE_NOT_BYTE(x) ((unsigned char)~(x))
7507#define STR_BITWISE_AND_WORD(x, y) ((x) & (y))
7508#define STR_BITWISE_AND_BYTE(x, y) ((unsigned char)((x) & (y)))
7509#define STR_BITWISE_OR_WORD(x, y) ((x) | (y))
7510#define STR_BITWISE_OR_BYTE(x, y) ((unsigned char)((x) | (y)))
7511#define STR_BITWISE_XOR_WORD(x, y) ((x) ^ (y))
7512#define STR_BITWISE_XOR_BYTE(x, y) ((unsigned char)((x) ^ (y)))
7513
7514STR_DEFINE_UNARY_BITWISE_KERNEL(str_bitwise_not, STR_BITWISE_NOT_WORD, STR_BITWISE_NOT_BYTE)
7515STR_DEFINE_BINARY_BITWISE_KERNEL(str_bitwise_and, STR_BITWISE_AND_WORD, STR_BITWISE_AND_BYTE)
7516STR_DEFINE_BINARY_BITWISE_KERNEL(str_bitwise_or, STR_BITWISE_OR_WORD, STR_BITWISE_OR_BYTE)
7517STR_DEFINE_BINARY_BITWISE_KERNEL(str_bitwise_xor, STR_BITWISE_XOR_WORD, STR_BITWISE_XOR_BYTE)
7518
7519/*
7520 * call-seq:
7521 * bitwise_not -> string
7522 *
7523 * :include: doc/string/bitwise_not.rdoc
7524 *
7525 */
7526static VALUE
7527rb_str_bitwise_not(VALUE str)
7528{
7529 long len = RSTRING_LEN(str);
7530 VALUE result = str_bitwise_result(str);
7531 str_bitwise_not((unsigned char *)RSTRING_PTR(result),
7532 (const unsigned char *)RSTRING_PTR(str), len);
7533 return result;
7534}
7535
7536/*
7537 * call-seq:
7538 * bitwise_not! -> self
7539 *
7540 * :include: doc/string/bitwise_not_bang.rdoc
7541 *
7542 */
7543static VALUE
7544rb_str_bitwise_not_bang(VALUE str)
7545{
7546 long len;
7547 unsigned char *ptr;
7548
7549 rb_str_modify(str);
7550 len = RSTRING_LEN(str);
7551 ptr = (unsigned char *)RSTRING_PTR(str);
7552 str_bitwise_not(ptr, ptr, len);
7553 return str;
7554}
7555
7556#define STR_DEFINE_BINARY_BITWISE_METHOD(name) \
7557 static VALUE \
7558 rb_str_bitwise_##name(VALUE str, VALUE other) \
7559 { \
7560 long len; \
7561 VALUE result; \
7562 StringValue(other); \
7563 str_check_bitwise_length(str, other); \
7564 len = RSTRING_LEN(str); \
7565 result = str_bitwise_result(str); \
7566 str_bitwise_##name((unsigned char *)RSTRING_PTR(result), \
7567 (const unsigned char *)RSTRING_PTR(str), \
7568 (const unsigned char *)RSTRING_PTR(other), len); \
7569 return result; \
7570 } \
7571 static VALUE \
7572 rb_str_bitwise_##name##_bang(VALUE str, VALUE other) \
7573 { \
7574 long len; \
7575 unsigned char *ptr; \
7576 StringValue(other); \
7577 str_check_bitwise_length(str, other); \
7578 rb_str_modify(str); \
7579 len = RSTRING_LEN(str); \
7580 ptr = (unsigned char *)RSTRING_PTR(str); \
7581 str_bitwise_##name(ptr, ptr, \
7582 (const unsigned char *)RSTRING_PTR(other), len); \
7583 return str; \
7584 }
7585
7586STR_DEFINE_BINARY_BITWISE_METHOD(and)
7587STR_DEFINE_BINARY_BITWISE_METHOD(or)
7588STR_DEFINE_BINARY_BITWISE_METHOD(xor)
7589
7590static VALUE
7591str_byte_substr(VALUE str, long beg, long len, int empty)
7592{
7593 long n = RSTRING_LEN(str);
7594
7595 if (beg > n || len < 0) return Qnil;
7596 if (beg < 0) {
7597 beg += n;
7598 if (beg < 0) return Qnil;
7599 }
7600 if (len > n - beg)
7601 len = n - beg;
7602 if (len <= 0) {
7603 if (!empty) return Qnil;
7604 len = 0;
7605 }
7606
7607 VALUE str2 = str_subseq(str, beg, len);
7608
7609 str_enc_copy_direct(str2, str);
7610
7611 if (RSTRING_LEN(str2) == 0) {
7612 if (!rb_enc_asciicompat(STR_ENC_GET(str)))
7614 else
7616 }
7617 else {
7618 switch (ENC_CODERANGE(str)) {
7619 case ENC_CODERANGE_7BIT:
7621 break;
7622 default:
7624 break;
7625 }
7626 }
7627
7628 return str2;
7629}
7630
7631VALUE
7632rb_str_byte_substr(VALUE str, VALUE beg, VALUE len)
7633{
7634 return str_byte_substr(str, NUM2LONG(beg), NUM2LONG(len), TRUE);
7635}
7636
7637static VALUE
7638str_byte_aref(VALUE str, VALUE indx)
7639{
7640 long idx;
7641 if (FIXNUM_P(indx)) {
7642 idx = FIX2LONG(indx);
7643 }
7644 else {
7645 /* check if indx is Range */
7646 long beg, len = RSTRING_LEN(str);
7647
7648 switch (rb_range_beg_len(indx, &beg, &len, len, 0)) {
7649 case Qfalse:
7650 break;
7651 case Qnil:
7652 return Qnil;
7653 default:
7654 return str_byte_substr(str, beg, len, TRUE);
7655 }
7656
7657 idx = NUM2LONG(indx);
7658 }
7659 return str_byte_substr(str, idx, 1, FALSE);
7660}
7661
7662/*
7663 * call-seq:
7664 * byteslice(offset, length = 1) -> string or nil
7665 * byteslice(range) -> string or nil
7666 *
7667 * :include: doc/string/byteslice.rdoc
7668 */
7669
7670static VALUE
7671rb_str_byteslice(int argc, VALUE *argv, VALUE str)
7672{
7673 if (argc == 2) {
7674 long beg = NUM2LONG(argv[0]);
7675 long len = NUM2LONG(argv[1]);
7676 return str_byte_substr(str, beg, len, TRUE);
7677 }
7678 rb_check_arity(argc, 1, 2);
7679 return str_byte_aref(str, argv[0]);
7680}
7681
7682static void
7683str_check_beg_len(VALUE str, long *beg, long *len)
7684{
7685 long end, slen = RSTRING_LEN(str);
7686
7687 if (*len < 0) rb_raise(rb_eIndexError, "negative length %ld", *len);
7688 if ((slen < *beg) || ((*beg < 0) && (*beg + slen < 0))) {
7689 rb_raise(rb_eIndexError, "index %ld out of string", *beg);
7690 }
7691 if (*beg < 0) {
7692 *beg += slen;
7693 }
7694 RUBY_ASSERT(*beg >= 0);
7695 RUBY_ASSERT(*beg <= slen);
7696
7697 if (*len > slen - *beg) {
7698 *len = slen - *beg;
7699 }
7700 end = *beg + *len;
7701 str_ensure_byte_pos(str, *beg);
7702 str_ensure_byte_pos(str, end);
7703}
7704
7705/*
7706 * call-seq:
7707 * bytesplice(offset, length, str) -> self
7708 * bytesplice(offset, length, str, str_offset, str_length) -> self
7709 * bytesplice(range, str) -> self
7710 * bytesplice(range, str, str_range) -> self
7711 *
7712 * :include: doc/string/bytesplice.rdoc
7713 */
7714
7715static VALUE
7716rb_str_bytesplice(int argc, VALUE *argv, VALUE str)
7717{
7718 long beg, len, vbeg, vlen;
7719 VALUE val;
7720 int cr;
7721
7722 rb_check_arity(argc, 2, 5);
7723 if (!(argc == 2 || argc == 3 || argc == 5)) {
7724 rb_raise(rb_eArgError, "wrong number of arguments (given %d, expected 2, 3, or 5)", argc);
7725 }
7726 if (argc == 2 || (argc == 3 && !RB_INTEGER_TYPE_P(argv[0]))) {
7727 if (!rb_range_beg_len(argv[0], &beg, &len, RSTRING_LEN(str), 2)) {
7728 rb_raise(rb_eTypeError, "wrong argument type %s (expected Range)",
7729 rb_builtin_class_name(argv[0]));
7730 }
7731 val = argv[1];
7732 StringValue(val);
7733 if (argc == 2) {
7734 /* bytesplice(range, str) */
7735 vbeg = 0;
7736 vlen = RSTRING_LEN(val);
7737 }
7738 else {
7739 /* bytesplice(range, str, str_range) */
7740 if (!rb_range_beg_len(argv[2], &vbeg, &vlen, RSTRING_LEN(val), 2)) {
7741 rb_raise(rb_eTypeError, "wrong argument type %s (expected Range)",
7742 rb_builtin_class_name(argv[2]));
7743 }
7744 }
7745 }
7746 else {
7747 beg = NUM2LONG(argv[0]);
7748 len = NUM2LONG(argv[1]);
7749 val = argv[2];
7750 StringValue(val);
7751 if (argc == 3) {
7752 /* bytesplice(index, length, str) */
7753 vbeg = 0;
7754 vlen = RSTRING_LEN(val);
7755 }
7756 else {
7757 /* bytesplice(index, length, str, str_index, str_length) */
7758 vbeg = NUM2LONG(argv[3]);
7759 vlen = NUM2LONG(argv[4]);
7760 }
7761 }
7762 str_check_beg_len(str, &beg, &len);
7763 str_check_beg_len(val, &vbeg, &vlen);
7764 str_modify_keep_cr(str);
7765
7766 if (RB_UNLIKELY(ENCODING_GET_INLINED(str) != ENCODING_GET_INLINED(val))) {
7767 rb_enc_associate(str, rb_enc_check(str, val));
7768 }
7769
7770 rb_str_update_1(str, beg, len, val, vbeg, vlen);
7772 if (cr != ENC_CODERANGE_BROKEN)
7773 ENC_CODERANGE_SET(str, cr);
7774 return str;
7775}
7776
7777/*
7778 * call-seq:
7779 * reverse -> new_string
7780 *
7781 * Returns a new string with the characters from +self+ in reverse order.
7782 *
7783 * 'drawer'.reverse # => "reward"
7784 * 'reviled'.reverse # => "deliver"
7785 * 'stressed'.reverse # => "desserts"
7786 * 'semordnilaps'.reverse # => "spalindromes"
7787 *
7788 * Related: see {Converting to New String}[rdoc-ref:String@Converting+to+New+String].
7789 */
7790
7791static VALUE
7792rb_str_reverse(VALUE str)
7793{
7794 rb_encoding *enc;
7795 VALUE rev;
7796 char *s, *e, *p;
7797 int cr;
7798
7799 if (RSTRING_LEN(str) <= 1) return str_duplicate(rb_cString, str);
7800 enc = STR_ENC_GET(str);
7801 rev = rb_str_new(0, RSTRING_LEN(str));
7802 s = RSTRING_PTR(str); e = RSTRING_END(str);
7803 p = RSTRING_END(rev);
7804 cr = ENC_CODERANGE(str);
7805
7806 if (RSTRING_LEN(str) > 1) {
7807 if (single_byte_optimizable(str)) {
7808 while (s < e) {
7809 *--p = *s++;
7810 }
7811 }
7812 else if (cr == ENC_CODERANGE_VALID) {
7813 while (s < e) {
7814 int clen = rb_enc_fast_mbclen(s, e, enc);
7815
7816 p -= clen;
7817 memcpy(p, s, clen);
7818 s += clen;
7819 }
7820 }
7821 else {
7822 cr = rb_enc_asciicompat(enc) ?
7824 while (s < e) {
7825 int clen = rb_enc_mbclen(s, e, enc);
7826
7827 if (clen > 1 || (*s & 0x80)) cr = ENC_CODERANGE_UNKNOWN;
7828 p -= clen;
7829 memcpy(p, s, clen);
7830 s += clen;
7831 }
7832 }
7833 }
7834 STR_SET_LEN(rev, RSTRING_LEN(str));
7835 str_enc_copy_direct(rev, str);
7836 ENC_CODERANGE_SET(rev, cr);
7837
7838 return rev;
7839}
7840
7841
7842/*
7843 * call-seq:
7844 * reverse! -> self
7845 *
7846 * Returns +self+ with its characters reversed:
7847 *
7848 * 'drawer'.reverse! # => "reward"
7849 * 'reviled'.reverse! # => "deliver"
7850 * 'stressed'.reverse! # => "desserts"
7851 * 'semordnilaps'.reverse! # => "spalindromes"
7852 *
7853 * Related: see {Modifying}[rdoc-ref:String@Modifying].
7854 */
7855
7856static VALUE
7857rb_str_reverse_bang(VALUE str)
7858{
7859 if (RSTRING_LEN(str) > 1) {
7860 if (single_byte_optimizable(str)) {
7861 char *s, *e, c;
7862
7863 str_modify_keep_cr(str);
7864 s = RSTRING_PTR(str);
7865 e = RSTRING_END(str) - 1;
7866 while (s < e) {
7867 c = *s;
7868 *s++ = *e;
7869 *e-- = c;
7870 }
7871 }
7872 else {
7873 str_shared_replace(str, rb_str_reverse(str));
7874 }
7875 }
7876 else {
7877 str_modify_keep_cr(str);
7878 }
7879 return str;
7880}
7881
7882
7883/*
7884 * call-seq:
7885 * include?(other_string) -> true or false
7886 *
7887 * Returns whether +self+ contains +other_string+:
7888 *
7889 * s = 'bar'
7890 * s.include?('ba') # => true
7891 * s.include?('ar') # => true
7892 * s.include?('bar') # => true
7893 * s.include?('a') # => true
7894 * s.include?('') # => true
7895 * s.include?('foo') # => false
7896 *
7897 * Related: see {Querying}[rdoc-ref:String@Querying].
7898 */
7899
7900VALUE
7901rb_str_include(VALUE str, VALUE arg)
7902{
7903 long i;
7904
7905 StringValue(arg);
7906 i = rb_str_index(str, arg, 0);
7907
7908 return RBOOL(i != -1);
7909}
7910
7911
7912/*
7913 * call-seq:
7914 * to_i(base = 10) -> integer
7915 *
7916 * Returns the result of interpreting leading characters in +self+
7917 * as an integer in the given +base+;
7918 * +base+ must be either +0+ or in range <tt>(2..36)</tt>:
7919 *
7920 * '123456'.to_i # => 123456
7921 * '123def'.to_i(16) # => 1195503
7922 *
7923 * With +base+ zero given, string +object+ may contain leading characters
7924 * to specify the actual base:
7925 *
7926 * '123def'.to_i(0) # => 123
7927 * '0123def'.to_i(0) # => 83
7928 * '0b123def'.to_i(0) # => 1
7929 * '0o123def'.to_i(0) # => 83
7930 * '0d123def'.to_i(0) # => 123
7931 * '0x123def'.to_i(0) # => 1195503
7932 *
7933 * Characters past a leading valid number (in the given +base+) are ignored:
7934 *
7935 * '12.345'.to_i # => 12
7936 * '12345'.to_i(2) # => 1
7937 *
7938 * Returns zero if there is no leading valid number:
7939 *
7940 * 'abcdef'.to_i # => 0
7941 * '2'.to_i(2) # => 0
7942 *
7943 * Related: see {Converting to Non-String}[rdoc-ref:String@Converting+to+Non-String].
7944 */
7945
7946static VALUE
7947rb_str_to_i(int argc, VALUE *argv, VALUE str)
7948{
7949 int base = 10;
7950
7951 if (rb_check_arity(argc, 0, 1) && (base = NUM2INT(argv[0])) < 0) {
7952 rb_raise(rb_eArgError, "invalid radix %d", base);
7953 }
7954 return rb_str_to_inum(str, base, FALSE);
7955}
7956
7957
7958/*
7959 * call-seq:
7960 * to_f -> float
7961 *
7962 * Returns the result of interpreting leading characters in +self+ as a Float:
7963 *
7964 * '3.14159'.to_f # => 3.14159
7965 * '1.234e-2'.to_f # => 0.01234
7966 *
7967 * Characters past a leading valid number are ignored:
7968 *
7969 * '3.14 (pi to two places)'.to_f # => 3.14
7970 *
7971 * Returns zero if there is no leading valid number:
7972 *
7973 * 'abcdef'.to_f # => 0.0
7974 *
7975 * See {Converting to Non-String}[rdoc-ref:String@Converting+to+Non-String].
7976 */
7977
7978static VALUE
7979rb_str_to_f(VALUE str)
7980{
7981 return DBL2NUM(rb_str_to_dbl(str, FALSE));
7982}
7983
7984
7985/*
7986 * call-seq:
7987 * to_s -> self or new_string
7988 *
7989 * Returns +self+ if +self+ is a +String+,
7990 * or +self+ converted to a +String+ if +self+ is a subclass of +String+.
7991 *
7992 * Related: see {Converting to New String}[rdoc-ref:String@Converting+to+New+String].
7993 */
7994
7995static VALUE
7996rb_str_to_s(VALUE str)
7997{
7998 if (rb_obj_class(str) != rb_cString) {
7999 return str_duplicate(rb_cString, str);
8000 }
8001 return str;
8002}
8003
8004#if 0
8005static void
8006str_cat_char(VALUE str, unsigned int c, rb_encoding *enc)
8007{
8008 char s[RUBY_MAX_CHAR_LEN];
8009 int n = rb_enc_codelen(c, enc);
8010
8011 rb_enc_mbcput(c, s, enc);
8012 rb_enc_str_buf_cat(str, s, n, enc);
8013}
8014#endif
8015
8016#define CHAR_ESC_LEN 13 /* sizeof(\x{ hex of 32bit unsigned int } \0) */
8017
8018int
8019rb_str_buf_cat_escaped_char(VALUE result, unsigned int c, int unicode_p)
8020{
8021 char buf[CHAR_ESC_LEN + 1];
8022 int l;
8023
8024#if SIZEOF_INT > 4
8025 c &= 0xffffffff;
8026#endif
8027 if (unicode_p) {
8028 if (c < 0x7F && ISPRINT(c)) {
8029 snprintf(buf, CHAR_ESC_LEN, "%c", c);
8030 }
8031 else if (c < 0x10000) {
8032 snprintf(buf, CHAR_ESC_LEN, "\\u%04X", c);
8033 }
8034 else {
8035 snprintf(buf, CHAR_ESC_LEN, "\\u{%X}", c);
8036 }
8037 }
8038 else {
8039 if (c < 0x100) {
8040 snprintf(buf, CHAR_ESC_LEN, "\\x%02X", c);
8041 }
8042 else {
8043 snprintf(buf, CHAR_ESC_LEN, "\\x{%X}", c);
8044 }
8045 }
8046 l = (int)strlen(buf); /* CHAR_ESC_LEN cannot exceed INT_MAX */
8047 rb_str_buf_cat(result, buf, l);
8048 return l;
8049}
8050
8051const char *
8052ruby_escaped_char(int c)
8053{
8054 switch (c) {
8055 case '\0': return "\\0";
8056 case '\n': return "\\n";
8057 case '\r': return "\\r";
8058 case '\t': return "\\t";
8059 case '\f': return "\\f";
8060 case '\013': return "\\v";
8061 case '\010': return "\\b";
8062 case '\007': return "\\a";
8063 case '\033': return "\\e";
8064 case '\x7f': return "\\c?";
8065 }
8066 return NULL;
8067}
8068
8069VALUE
8070rb_str_escape(VALUE str)
8071{
8072 int encidx = ENCODING_GET(str);
8073 rb_encoding *enc = rb_enc_from_index(encidx);
8074 const char *p = RSTRING_PTR(str);
8075 const char *pend = RSTRING_END(str);
8076 const char *prev = p;
8077 char buf[CHAR_ESC_LEN + 1];
8078 VALUE result = rb_str_buf_new(0);
8079 int unicode_p = rb_enc_unicode_p(enc);
8080 int asciicompat = rb_enc_asciicompat(enc);
8081
8082 while (p < pend) {
8083 unsigned int c;
8084 const char *cc;
8085 int n = rb_enc_precise_mbclen(p, pend, enc);
8086 if (!MBCLEN_CHARFOUND_P(n)) {
8087 if (p > prev) str_buf_cat(result, prev, p - prev);
8088 n = rb_enc_mbminlen(enc);
8089 if (pend < p + n)
8090 n = (int)(pend - p);
8091 while (n--) {
8092 snprintf(buf, CHAR_ESC_LEN, "\\x%02X", *p & 0377);
8093 str_buf_cat(result, buf, strlen(buf));
8094 prev = ++p;
8095 }
8096 continue;
8097 }
8098 n = MBCLEN_CHARFOUND_LEN(n);
8099 c = rb_enc_mbc_to_codepoint(p, pend, enc);
8100 p += n;
8101 cc = ruby_escaped_char(c);
8102 if (cc) {
8103 if (p - n > prev) str_buf_cat(result, prev, p - n - prev);
8104 str_buf_cat(result, cc, strlen(cc));
8105 prev = p;
8106 }
8107 else if (asciicompat && rb_enc_isascii(c, enc) && ISPRINT(c)) {
8108 }
8109 else {
8110 if (p - n > prev) str_buf_cat(result, prev, p - n - prev);
8111 rb_str_buf_cat_escaped_char(result, c, unicode_p);
8112 prev = p;
8113 }
8114 }
8115 if (p > prev) str_buf_cat(result, prev, p - prev);
8116 ENCODING_CODERANGE_SET(result, rb_usascii_encindex(), ENC_CODERANGE_7BIT);
8117
8118 return result;
8119}
8120
8121/* Lookup table for the inspect fast path. 1 marks bytes that need
8122 * no escaping. 0 marks bytes that need escape inspection: 0x00-0x1F
8123 * (control), 0x22 ("), 0x23 (#), 0x5C (\‍), 0x7F (DEL), 0x80-0xFF
8124 * (non-ASCII). */
8125static const bool inspect_no_escape[256] = {
8126 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, /* 0x00-0x0F */
8127 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, /* 0x10-0x1F */
8128 1, 1, 0, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, /* 0x20-0x2F */
8129 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, /* 0x30-0x3F */
8130 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, /* 0x40-0x4F */
8131 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 1, 1, 1, /* 0x50-0x5F */
8132 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, /* 0x60-0x6F */
8133 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, /* 0x70-0x7F */
8134};
8135
8136/*
8137 * call-seq:
8138 * inspect -> string
8139 *
8140 * :include: doc/string/inspect.rdoc
8141 *
8142 */
8143
8144VALUE
8146{
8147 int encidx = ENCODING_GET(str);
8148 rb_encoding *enc = rb_enc_from_index(encidx);
8149 const char *p, *pend, *prev;
8150 char buf[CHAR_ESC_LEN + 1];
8151 VALUE result = rb_str_buf_new(RSTRING_LEN(str) + 2); /* string content + surrounding quotes */
8152 rb_encoding *resenc = rb_default_internal_encoding();
8153 int unicode_p = rb_enc_unicode_p(enc);
8154 int asciicompat = rb_enc_asciicompat(enc);
8155 int cr = rb_enc_str_coderange(str);
8156
8157 if (resenc == NULL) resenc = rb_default_external_encoding();
8158 if (!rb_enc_asciicompat(resenc)) resenc = rb_usascii_encoding();
8159 rb_enc_associate(result, resenc);
8160 str_buf_cat2(result, "\"");
8161
8162 p = RSTRING_PTR(str); pend = RSTRING_END(str);
8163 prev = p;
8164 while (p < pend) {
8165 unsigned int c, cc;
8166 int n;
8167
8168 /* Fast path: bulk-skip runs of safe ASCII bytes via a lookup table.
8169 * Only well-formed strings (CR=7BIT for any encoding, or UTF-8 VALID)
8170 * are eligible. */
8171 if (cr == ENC_CODERANGE_7BIT ||
8172 (encidx == ENCINDEX_UTF_8 && cr == ENC_CODERANGE_VALID)) {
8173 while (p < pend && inspect_no_escape[(unsigned char)*p]) p++;
8174 if (p >= pend) break;
8175 }
8176
8177 n = rb_enc_precise_mbclen(p, pend, enc);
8178 if (!MBCLEN_CHARFOUND_P(n)) {
8179 if (p > prev) str_buf_cat(result, prev, p - prev);
8180 n = rb_enc_mbminlen(enc);
8181 if (pend < p + n)
8182 n = (int)(pend - p);
8183 while (n--) {
8184 snprintf(buf, CHAR_ESC_LEN, "\\x%02X", *p & 0377);
8185 str_buf_cat(result, buf, strlen(buf));
8186 prev = ++p;
8187 }
8188 continue;
8189 }
8190 n = MBCLEN_CHARFOUND_LEN(n);
8191 c = rb_enc_mbc_to_codepoint(p, pend, enc);
8192 p += n;
8193 if ((asciicompat || unicode_p) &&
8194 (c == '"'|| c == '\\' ||
8195 (c == '#' &&
8196 p < pend &&
8197 MBCLEN_CHARFOUND_P(rb_enc_precise_mbclen(p,pend,enc)) &&
8198 (cc = rb_enc_codepoint(p,pend,enc),
8199 (cc == '$' || cc == '@' || cc == '{'))))) {
8200 if (p - n > prev) str_buf_cat(result, prev, p - n - prev);
8201 str_buf_cat2(result, "\\");
8202 if (asciicompat || enc == resenc) {
8203 prev = p - n;
8204 continue;
8205 }
8206 }
8207 switch (c) {
8208 case '\n': cc = 'n'; break;
8209 case '\r': cc = 'r'; break;
8210 case '\t': cc = 't'; break;
8211 case '\f': cc = 'f'; break;
8212 case '\013': cc = 'v'; break;
8213 case '\010': cc = 'b'; break;
8214 case '\007': cc = 'a'; break;
8215 case 033: cc = 'e'; break;
8216 default: cc = 0; break;
8217 }
8218 if (cc) {
8219 if (p - n > prev) str_buf_cat(result, prev, p - n - prev);
8220 buf[0] = '\\';
8221 buf[1] = (char)cc;
8222 str_buf_cat(result, buf, 2);
8223 prev = p;
8224 continue;
8225 }
8226 /* The special casing of 0x85 (NEXT_LINE) here is because
8227 * Oniguruma historically treats it as printable, but it
8228 * doesn't match the print POSIX bracket class or character
8229 * property in regexps.
8230 *
8231 * See Ruby Bug #16842 for details:
8232 * https://bugs.ruby-lang.org/issues/16842
8233 */
8234 if ((enc == resenc && rb_enc_isprint(c, enc) && c != 0x85) ||
8235 (asciicompat && rb_enc_isascii(c, enc) && ISPRINT(c))) {
8236 continue;
8237 }
8238 else {
8239 if (p - n > prev) str_buf_cat(result, prev, p - n - prev);
8240 rb_str_buf_cat_escaped_char(result, c, unicode_p);
8241 prev = p;
8242 continue;
8243 }
8244 }
8245 if (p > prev) str_buf_cat(result, prev, p - prev);
8246 str_buf_cat2(result, "\"");
8247
8248 return result;
8249}
8250
8251#define IS_EVSTR(p,e) ((p) < (e) && (*(p) == '$' || *(p) == '@' || *(p) == '{'))
8252
8253/*
8254 * call-seq:
8255 * dump -> new_string
8256 *
8257 * :include: doc/string/dump.rdoc
8258 *
8259 */
8260
8261VALUE
8263{
8264 int encidx = rb_enc_get_index(str);
8265 rb_encoding *enc = rb_enc_from_index(encidx);
8266 long len;
8267 const char *p, *pend;
8268 char *q, *qend;
8269 VALUE result;
8270 int u8 = (encidx == rb_utf8_encindex());
8271 static const char nonascii_suffix[] = ".dup.force_encoding(\"%s\")";
8272
8273 len = 2; /* "" */
8274 if (!rb_enc_asciicompat(enc)) {
8275 len += strlen(nonascii_suffix) - rb_strlen_lit("%s");
8276 len += strlen(enc->name);
8277 }
8278
8279 p = RSTRING_PTR(str); pend = p + RSTRING_LEN(str);
8280 while (p < pend) {
8281 int clen;
8282 unsigned char c = *p++;
8283
8284 switch (c) {
8285 case '"': case '\\':
8286 case '\n': case '\r':
8287 case '\t': case '\f':
8288 case '\013': case '\010': case '\007': case '\033':
8289 clen = 2;
8290 break;
8291
8292 case '#':
8293 clen = IS_EVSTR(p, pend) ? 2 : 1;
8294 break;
8295
8296 default:
8297 if (ISPRINT(c)) {
8298 clen = 1;
8299 }
8300 else {
8301 if (u8 && c > 0x7F) { /* \u notation */
8302 int n = rb_enc_precise_mbclen(p-1, pend, enc);
8303 if (MBCLEN_CHARFOUND_P(n)) {
8304 unsigned int cc = rb_enc_mbc_to_codepoint(p-1, pend, enc);
8305 if (cc <= 0xFFFF)
8306 clen = 6; /* \uXXXX */
8307 else if (cc <= 0xFFFFF)
8308 clen = 9; /* \u{XXXXX} */
8309 else
8310 clen = 10; /* \u{XXXXXX} */
8311 p += MBCLEN_CHARFOUND_LEN(n)-1;
8312 break;
8313 }
8314 }
8315 clen = 4; /* \xNN */
8316 }
8317 break;
8318 }
8319
8320 if (clen > LONG_MAX - len) {
8321 rb_raise(rb_eRuntimeError, "string size too big");
8322 }
8323 len += clen;
8324 }
8325
8326 result = rb_str_new(0, len);
8327 p = RSTRING_PTR(str); pend = p + RSTRING_LEN(str);
8328 q = RSTRING_PTR(result); qend = q + len + 1;
8329
8330 *q++ = '"';
8331 while (p < pend) {
8332 unsigned char c = *p++;
8333
8334 if (c == '"' || c == '\\') {
8335 *q++ = '\\';
8336 *q++ = c;
8337 }
8338 else if (c == '#') {
8339 if (IS_EVSTR(p, pend)) *q++ = '\\';
8340 *q++ = '#';
8341 }
8342 else if (c == '\n') {
8343 *q++ = '\\';
8344 *q++ = 'n';
8345 }
8346 else if (c == '\r') {
8347 *q++ = '\\';
8348 *q++ = 'r';
8349 }
8350 else if (c == '\t') {
8351 *q++ = '\\';
8352 *q++ = 't';
8353 }
8354 else if (c == '\f') {
8355 *q++ = '\\';
8356 *q++ = 'f';
8357 }
8358 else if (c == '\013') {
8359 *q++ = '\\';
8360 *q++ = 'v';
8361 }
8362 else if (c == '\010') {
8363 *q++ = '\\';
8364 *q++ = 'b';
8365 }
8366 else if (c == '\007') {
8367 *q++ = '\\';
8368 *q++ = 'a';
8369 }
8370 else if (c == '\033') {
8371 *q++ = '\\';
8372 *q++ = 'e';
8373 }
8374 else if (ISPRINT(c)) {
8375 *q++ = c;
8376 }
8377 else {
8378 *q++ = '\\';
8379 if (u8) {
8380 int n = rb_enc_precise_mbclen(p-1, pend, enc) - 1;
8381 if (MBCLEN_CHARFOUND_P(n)) {
8382 int cc = rb_enc_mbc_to_codepoint(p-1, pend, enc);
8383 p += n;
8384 if (cc <= 0xFFFF)
8385 snprintf(q, qend-q, "u%04X", cc); /* \uXXXX */
8386 else
8387 snprintf(q, qend-q, "u{%X}", cc); /* \u{XXXXX} or \u{XXXXXX} */
8388 q += strlen(q);
8389 continue;
8390 }
8391 }
8392 snprintf(q, qend-q, "x%02X", c);
8393 q += 3;
8394 }
8395 }
8396 *q++ = '"';
8397 *q = '\0';
8398 if (!rb_enc_asciicompat(enc)) {
8399 snprintf(q, qend-q, nonascii_suffix, enc->name);
8400 encidx = rb_ascii8bit_encindex();
8401 }
8402 /* result from dump is ASCII */
8403 rb_enc_associate_index(result, encidx);
8405 return result;
8406}
8407
8408static int
8409unescape_ascii(unsigned int c)
8410{
8411 switch (c) {
8412 case 'n':
8413 return '\n';
8414 case 'r':
8415 return '\r';
8416 case 't':
8417 return '\t';
8418 case 'f':
8419 return '\f';
8420 case 'v':
8421 return '\13';
8422 case 'b':
8423 return '\010';
8424 case 'a':
8425 return '\007';
8426 case 'e':
8427 return 033;
8428 }
8430}
8431
8432static void
8433undump_after_backslash(VALUE undumped, const char **ss, const char *s_end, rb_encoding **penc, bool *utf8, bool *binary)
8434{
8435 const char *s = *ss;
8436 unsigned int c;
8437 int codelen;
8438 size_t hexlen;
8439 unsigned char buf[6];
8440 static rb_encoding *enc_utf8 = NULL;
8441
8442 switch (*s) {
8443 case '\\':
8444 case '"':
8445 case '#':
8446 rb_str_cat(undumped, s, 1); /* cat itself */
8447 s++;
8448 break;
8449 case 'n':
8450 case 'r':
8451 case 't':
8452 case 'f':
8453 case 'v':
8454 case 'b':
8455 case 'a':
8456 case 'e':
8457 *buf = unescape_ascii(*s);
8458 rb_str_cat(undumped, (char *)buf, 1);
8459 s++;
8460 break;
8461 case 'u':
8462 if (*binary) {
8463 rb_raise(rb_eRuntimeError, "hex escape and Unicode escape are mixed");
8464 }
8465 *utf8 = true;
8466 if (++s >= s_end) {
8467 rb_raise(rb_eRuntimeError, "invalid Unicode escape");
8468 }
8469 if (enc_utf8 == NULL) enc_utf8 = rb_utf8_encoding();
8470 if (*penc != enc_utf8) {
8471 *penc = enc_utf8;
8472 rb_enc_associate(undumped, enc_utf8);
8473 }
8474 if (*s == '{') { /* handle \u{...} form */
8475 s++;
8476 for (;;) {
8477 if (s >= s_end) {
8478 rb_raise(rb_eRuntimeError, "unterminated Unicode escape");
8479 }
8480 if (*s == '}') {
8481 s++;
8482 break;
8483 }
8484 if (ISSPACE(*s)) {
8485 s++;
8486 continue;
8487 }
8488 c = scan_hex(s, s_end-s, &hexlen);
8489 if (hexlen == 0 || hexlen > 6) {
8490 rb_raise(rb_eRuntimeError, "invalid Unicode escape");
8491 }
8492 if (c > 0x10ffff) {
8493 rb_raise(rb_eRuntimeError, "invalid Unicode codepoint (too large)");
8494 }
8495 if (0xd800 <= c && c <= 0xdfff) {
8496 rb_raise(rb_eRuntimeError, "invalid Unicode codepoint");
8497 }
8498 codelen = rb_enc_mbcput(c, (char *)buf, *penc);
8499 rb_str_cat(undumped, (char *)buf, codelen);
8500 s += hexlen;
8501 }
8502 }
8503 else { /* handle \uXXXX form */
8504 c = scan_hex(s, 4, &hexlen);
8505 if (hexlen != 4) {
8506 rb_raise(rb_eRuntimeError, "invalid Unicode escape");
8507 }
8508 if (0xd800 <= c && c <= 0xdfff) {
8509 rb_raise(rb_eRuntimeError, "invalid Unicode codepoint");
8510 }
8511 codelen = rb_enc_mbcput(c, (char *)buf, *penc);
8512 rb_str_cat(undumped, (char *)buf, codelen);
8513 s += hexlen;
8514 }
8515 break;
8516 case 'x':
8517 if (++s >= s_end) {
8518 rb_raise(rb_eRuntimeError, "invalid hex escape");
8519 }
8520 *buf = scan_hex(s, 2, &hexlen);
8521 if (hexlen != 2) {
8522 rb_raise(rb_eRuntimeError, "invalid hex escape");
8523 }
8524 if (!ISASCII(*buf)) {
8525 if (*utf8) {
8526 rb_raise(rb_eRuntimeError, "hex escape and Unicode escape are mixed");
8527 }
8528 *binary = true;
8529 }
8530 rb_str_cat(undumped, (char *)buf, 1);
8531 s += hexlen;
8532 break;
8533 default:
8534 rb_str_cat(undumped, s-1, 2);
8535 s++;
8536 }
8537
8538 *ss = s;
8539}
8540
8541static VALUE rb_str_is_ascii_only_p(VALUE str);
8542
8543/*
8544 * call-seq:
8545 * undump -> new_string
8546 *
8547 * Inverse of String#dump; returns a copy of +self+ with changes of the kinds made by String#dump "undone."
8548 *
8549 * Related: see {Converting to New String}[rdoc-ref:String@Converting+to+New+String].
8550 */
8551
8552static VALUE
8553str_undump(VALUE str)
8554{
8555 const char *s = RSTRING_PTR(str);
8556 const char *s_end = RSTRING_END(str);
8557 rb_encoding *enc = rb_enc_get(str);
8558 VALUE undumped = rb_enc_str_new(s, 0L, enc);
8559 bool utf8 = false;
8560 bool binary = false;
8561 int w;
8562
8564 if (rb_str_is_ascii_only_p(str) == Qfalse) {
8565 rb_raise(rb_eRuntimeError, "non-ASCII character detected");
8566 }
8567 if (!str_null_check(str, &w)) {
8568 rb_raise(rb_eRuntimeError, "string contains null byte");
8569 }
8570 if (RSTRING_LEN(str) < 2) goto invalid_format;
8571 if (*s != '"') goto invalid_format;
8572
8573 /* strip '"' at the start */
8574 s++;
8575
8576 for (;;) {
8577 if (s >= s_end) {
8578 rb_raise(rb_eRuntimeError, "unterminated dumped string");
8579 }
8580
8581 if (*s == '"') {
8582 /* epilogue */
8583 s++;
8584 if (s == s_end) {
8585 /* ascii compatible dumped string */
8586 break;
8587 }
8588 else {
8589 static const char force_encoding_suffix[] = ".force_encoding(\""; /* "\")" */
8590 static const char dup_suffix[] = ".dup";
8591 const char *encname;
8592 int encidx;
8593 ptrdiff_t size;
8594
8595 /* check separately for strings dumped by older versions */
8596 size = sizeof(dup_suffix) - 1;
8597 if (s_end - s > size && memcmp(s, dup_suffix, size) == 0) s += size;
8598
8599 size = sizeof(force_encoding_suffix) - 1;
8600 if (s_end - s <= size) goto invalid_format;
8601 if (memcmp(s, force_encoding_suffix, size) != 0) goto invalid_format;
8602 s += size;
8603
8604 if (utf8) {
8605 rb_raise(rb_eRuntimeError, "dumped string contained Unicode escape but used force_encoding");
8606 }
8607
8608 encname = s;
8609 s = memchr(s, '"', s_end-s);
8610 size = s - encname;
8611 if (!s) goto invalid_format;
8612 if (s_end - s != 2) goto invalid_format;
8613 if (s[0] != '"' || s[1] != ')') goto invalid_format;
8614
8615 encidx = rb_enc_find_index2(encname, (long)size);
8616 if (encidx < 0) {
8617 rb_raise(rb_eRuntimeError, "dumped string has unknown encoding name");
8618 }
8619 rb_enc_associate_index(undumped, encidx);
8620 }
8621 break;
8622 }
8623
8624 if (*s == '\\') {
8625 s++;
8626 if (s >= s_end) {
8627 rb_raise(rb_eRuntimeError, "invalid escape");
8628 }
8629 undump_after_backslash(undumped, &s, s_end, &enc, &utf8, &binary);
8630 }
8631 else {
8632 rb_str_cat(undumped, s++, 1);
8633 }
8634 }
8635
8636 RB_GC_GUARD(str);
8637
8638 return undumped;
8639invalid_format:
8640 rb_raise(rb_eRuntimeError, "invalid dumped string; not wrapped with '\"' nor '\"...\".force_encoding(\"...\")' form");
8641}
8642
8643static void
8644rb_str_check_dummy_enc(rb_encoding *enc)
8645{
8646 if (rb_enc_dummy_p(enc)) {
8647 rb_raise(rb_eEncCompatError, "incompatible encoding with this operation: %s",
8648 rb_enc_name(enc));
8649 }
8650}
8651
8652static rb_encoding *
8653str_true_enc(VALUE str)
8654{
8655 rb_encoding *enc = STR_ENC_GET(str);
8656 rb_str_check_dummy_enc(enc);
8657 return enc;
8658}
8659
8660static OnigCaseFoldType
8661check_case_options(int argc, VALUE *argv, OnigCaseFoldType flags)
8662{
8663 if (argc==0)
8664 return flags;
8665 if (argc>2)
8666 rb_raise(rb_eArgError, "too many options");
8667 if (argv[0]==sym_turkic) {
8668 flags |= ONIGENC_CASE_FOLD_TURKISH_AZERI;
8669 if (argc==2) {
8670 if (argv[1]==sym_lithuanian)
8671 flags |= ONIGENC_CASE_FOLD_LITHUANIAN;
8672 else
8673 rb_raise(rb_eArgError, "invalid second option");
8674 }
8675 }
8676 else if (argv[0]==sym_lithuanian) {
8677 flags |= ONIGENC_CASE_FOLD_LITHUANIAN;
8678 if (argc==2) {
8679 if (argv[1]==sym_turkic)
8680 flags |= ONIGENC_CASE_FOLD_TURKISH_AZERI;
8681 else
8682 rb_raise(rb_eArgError, "invalid second option");
8683 }
8684 }
8685 else if (argc>1)
8686 rb_raise(rb_eArgError, "too many options");
8687 else if (argv[0]==sym_ascii)
8688 flags |= ONIGENC_CASE_ASCII_ONLY;
8689 else if (argv[0]==sym_fold) {
8690 if ((flags & (ONIGENC_CASE_UPCASE|ONIGENC_CASE_DOWNCASE)) == ONIGENC_CASE_DOWNCASE)
8691 flags ^= ONIGENC_CASE_FOLD|ONIGENC_CASE_DOWNCASE;
8692 else
8693 rb_raise(rb_eArgError, "option :fold only allowed for downcasing");
8694 }
8695 else
8696 rb_raise(rb_eArgError, "invalid option");
8697 return flags;
8698}
8699
8700static inline bool
8701case_option_single_p(OnigCaseFoldType flags, rb_encoding *enc, VALUE str)
8702{
8703 if ((flags & ONIGENC_CASE_ASCII_ONLY) && (enc==rb_utf8_encoding() || rb_enc_mbmaxlen(enc) == 1))
8704 return true;
8705 return !(flags & ONIGENC_CASE_FOLD_TURKISH_AZERI) &&
8706 (ENC_CODERANGE(str) == ENC_CODERANGE_7BIT || rb_is_ascii8bit_enc(enc));
8707}
8708
8709/* 16 should be long enough to absorb any kind of single character length increase */
8710#define CASE_MAPPING_ADDITIONAL_LENGTH 20
8711#ifndef CASEMAP_DEBUG
8712# define CASEMAP_DEBUG 0
8713#endif
8714
8715struct mapping_buffer;
8716typedef struct mapping_buffer {
8717 size_t capa;
8718 size_t used;
8719 struct mapping_buffer *next;
8720 OnigUChar space[FLEX_ARY_LEN];
8722
8723static void
8724mapping_buffer_free(void *p)
8725{
8726 mapping_buffer *previous_buffer;
8727 mapping_buffer *current_buffer = p;
8728 while (current_buffer) {
8729 previous_buffer = current_buffer;
8730 current_buffer = current_buffer->next;
8731 ruby_xfree_sized(previous_buffer, offsetof(mapping_buffer, space) + previous_buffer->capa);
8732 }
8733}
8734
8735static const rb_data_type_t mapping_buffer_type = {
8736 "mapping_buffer",
8737 {0, mapping_buffer_free,},
8738 0, 0, RUBY_TYPED_THREAD_SAFE_FREE | RUBY_TYPED_WB_PROTECTED
8739};
8740
8741static VALUE
8742rb_str_casemap(VALUE source, OnigCaseFoldType *flags, rb_encoding *enc)
8743{
8744 VALUE target;
8745
8746 const OnigUChar *source_current, *source_end;
8747 int target_length = 0;
8748 VALUE buffer_anchor;
8749 mapping_buffer *current_buffer = 0;
8750 mapping_buffer **pre_buffer;
8751 size_t buffer_count = 0;
8752 int buffer_length_or_invalid;
8753
8754 if (RSTRING_LEN(source) == 0) return str_duplicate(rb_cString, source);
8755
8756 source_current = (OnigUChar*)RSTRING_PTR(source);
8757 source_end = (OnigUChar*)RSTRING_END(source);
8758
8759 buffer_anchor = TypedData_Wrap_Struct(0, &mapping_buffer_type, 0);
8760 pre_buffer = (mapping_buffer **)&DATA_PTR(buffer_anchor);
8761 while (source_current < source_end) {
8762 /* increase multiplier using buffer count to converge quickly */
8763 size_t capa = (size_t)(source_end-source_current)*++buffer_count + CASE_MAPPING_ADDITIONAL_LENGTH;
8764 if (CASEMAP_DEBUG) {
8765 fprintf(stderr, "Buffer allocation, capa is %"PRIuSIZE"\n", capa); /* for tuning */
8766 }
8767 current_buffer = xmalloc(offsetof(mapping_buffer, space) + capa);
8768 *pre_buffer = current_buffer;
8769 pre_buffer = &current_buffer->next;
8770 current_buffer->next = NULL;
8771 current_buffer->capa = capa;
8772 buffer_length_or_invalid = enc->case_map(flags,
8773 &source_current, source_end,
8774 current_buffer->space,
8775 current_buffer->space+current_buffer->capa,
8776 enc);
8777 if (buffer_length_or_invalid < 0) {
8778 current_buffer = DATA_PTR(buffer_anchor);
8779 DATA_PTR(buffer_anchor) = 0;
8780 mapping_buffer_free(current_buffer);
8781 rb_raise(rb_eArgError, "input string invalid");
8782 }
8783 target_length += current_buffer->used = buffer_length_or_invalid;
8784 }
8785 if (CASEMAP_DEBUG) {
8786 fprintf(stderr, "Buffer count is %"PRIuSIZE"\n", buffer_count); /* for tuning */
8787 }
8788
8789 if (buffer_count==1) {
8790 target = rb_str_new((const char*)current_buffer->space, target_length);
8791 }
8792 else {
8793 char *target_current;
8794
8795 target = rb_str_new(0, target_length);
8796 target_current = RSTRING_PTR(target);
8797 current_buffer = DATA_PTR(buffer_anchor);
8798 while (current_buffer) {
8799 memcpy(target_current, current_buffer->space, current_buffer->used);
8800 target_current += current_buffer->used;
8801 current_buffer = current_buffer->next;
8802 }
8803 }
8804 current_buffer = DATA_PTR(buffer_anchor);
8805 DATA_PTR(buffer_anchor) = 0;
8806 mapping_buffer_free(current_buffer);
8807
8808 RB_GC_GUARD(buffer_anchor);
8809
8810 /* TODO: check about string terminator character */
8811 str_enc_copy_direct(target, source);
8812 /*ENC_CODERANGE_SET(mapped, cr);*/
8813
8814 return target;
8815}
8816
8817static VALUE
8818rb_str_ascii_casemap(VALUE source, VALUE target, OnigCaseFoldType *flags, rb_encoding *enc)
8819{
8820 const OnigUChar *source_current, *source_end;
8821 OnigUChar *target_current, *target_end;
8822 long old_length = RSTRING_LEN(source);
8823 int length_or_invalid;
8824
8825 if (old_length == 0) return Qnil;
8826
8827 source_current = (OnigUChar*)RSTRING_PTR(source);
8828 source_end = (OnigUChar*)RSTRING_END(source);
8829 if (source == target) {
8830 target_current = (OnigUChar*)source_current;
8831 target_end = (OnigUChar*)source_end;
8832 }
8833 else {
8834 target_current = (OnigUChar*)RSTRING_PTR(target);
8835 target_end = (OnigUChar*)RSTRING_END(target);
8836 }
8837
8838 length_or_invalid = onigenc_ascii_only_case_map(flags,
8839 &source_current, source_end,
8840 target_current, target_end, enc);
8841 if (length_or_invalid < 0)
8842 rb_raise(rb_eArgError, "input string invalid");
8843 if (CASEMAP_DEBUG && length_or_invalid != old_length) {
8844 fprintf(stderr, "problem with rb_str_ascii_casemap"
8845 "; old_length=%ld, new_length=%d\n", old_length, length_or_invalid);
8846 rb_raise(rb_eArgError, "internal problem with rb_str_ascii_casemap"
8847 "; old_length=%ld, new_length=%d\n", old_length, length_or_invalid);
8848 }
8849
8850 str_enc_copy(target, source);
8851
8852 return target;
8853}
8854
8855static bool
8856upcase_single(VALUE str)
8857{
8858 char *s = RSTRING_PTR(str), *send = RSTRING_END(str);
8859 bool modified = false;
8860
8861 while (s < send) {
8862 unsigned int c = *(unsigned char*)s;
8863
8864 if ('a' <= c && c <= 'z') {
8865 *s = 'A' + (c - 'a');
8866 modified = true;
8867 }
8868 s++;
8869 }
8870 return modified;
8871}
8872
8873/*
8874 * call-seq:
8875 * upcase!(mapping) -> self or nil
8876 *
8877 * Like String#upcase, except that:
8878 *
8879 * - Changes character casings in +self+ (not in a copy of +self+).
8880 * - Returns +self+ if any changes are made, +nil+ otherwise.
8881 *
8882 * Related: See {Modifying}[rdoc-ref:String@Modifying].
8883 */
8884
8885static VALUE
8886rb_str_upcase_bang(int argc, VALUE *argv, VALUE str)
8887{
8888 rb_encoding *enc;
8889 OnigCaseFoldType flags = ONIGENC_CASE_UPCASE;
8890
8891 flags = check_case_options(argc, argv, flags);
8892 str_modify_keep_cr(str);
8893 enc = str_true_enc(str);
8894 if (case_option_single_p(flags, enc, str)) {
8895 if (upcase_single(str))
8896 flags |= ONIGENC_CASE_MODIFIED;
8897 }
8898 else if (flags&ONIGENC_CASE_ASCII_ONLY)
8899 rb_str_ascii_casemap(str, str, &flags, enc);
8900 else
8901 str_shared_replace(str, rb_str_casemap(str, &flags, enc));
8902
8903 if (ONIGENC_CASE_MODIFIED&flags) return str;
8904 return Qnil;
8905}
8906
8907
8908/*
8909 * call-seq:
8910 * upcase(mapping = :ascii) -> new_string
8911 *
8912 * :include: doc/string/upcase.rdoc
8913 */
8914
8915static VALUE
8916rb_str_upcase(int argc, VALUE *argv, VALUE str)
8917{
8918 rb_encoding *enc;
8919 OnigCaseFoldType flags = ONIGENC_CASE_UPCASE;
8920 VALUE ret;
8921
8922 flags = check_case_options(argc, argv, flags);
8923 enc = str_true_enc(str);
8924 if (case_option_single_p(flags, enc, str)) {
8925 ret = rb_str_new(RSTRING_PTR(str), RSTRING_LEN(str));
8926 str_enc_copy_direct(ret, str);
8927 upcase_single(ret);
8928 }
8929 else if (flags&ONIGENC_CASE_ASCII_ONLY) {
8930 ret = rb_str_new(0, RSTRING_LEN(str));
8931 rb_str_ascii_casemap(str, ret, &flags, enc);
8932 }
8933 else {
8934 ret = rb_str_casemap(str, &flags, enc);
8935 }
8936
8937 return ret;
8938}
8939
8940static bool
8941downcase_single(VALUE str)
8942{
8943 char *s = RSTRING_PTR(str), *send = RSTRING_END(str);
8944 bool modified = false;
8945
8946 while (s < send) {
8947 unsigned int c = *(unsigned char*)s;
8948
8949 if ('A' <= c && c <= 'Z') {
8950 *s = 'a' + (c - 'A');
8951 modified = true;
8952 }
8953 s++;
8954 }
8955
8956 return modified;
8957}
8958
8959/*
8960 * call-seq:
8961 * downcase!(mapping) -> self or nil
8962 *
8963 * Like String#downcase, except that:
8964 *
8965 * - Changes character casings in +self+ (not in a copy of +self+).
8966 * - Returns +self+ if any changes are made, +nil+ otherwise.
8967 *
8968 * Related: See {Modifying}[rdoc-ref:String@Modifying].
8969 */
8970
8971static VALUE
8972rb_str_downcase_bang(int argc, VALUE *argv, VALUE str)
8973{
8974 rb_encoding *enc;
8975 OnigCaseFoldType flags = ONIGENC_CASE_DOWNCASE;
8976
8977 flags = check_case_options(argc, argv, flags);
8978 str_modify_keep_cr(str);
8979 enc = str_true_enc(str);
8980 if (case_option_single_p(flags, enc, str)) {
8981 if (downcase_single(str))
8982 flags |= ONIGENC_CASE_MODIFIED;
8983 }
8984 else if (flags&ONIGENC_CASE_ASCII_ONLY)
8985 rb_str_ascii_casemap(str, str, &flags, enc);
8986 else
8987 str_shared_replace(str, rb_str_casemap(str, &flags, enc));
8988
8989 if (ONIGENC_CASE_MODIFIED&flags) return str;
8990 return Qnil;
8991}
8992
8993
8994/*
8995 * call-seq:
8996 * downcase(mapping = :ascii) -> new_string
8997 *
8998 * :include: doc/string/downcase.rdoc
8999 *
9000 */
9001
9002static VALUE
9003rb_str_downcase(int argc, VALUE *argv, VALUE str)
9004{
9005 rb_encoding *enc;
9006 OnigCaseFoldType flags = ONIGENC_CASE_DOWNCASE;
9007 VALUE ret;
9008
9009 flags = check_case_options(argc, argv, flags);
9010 enc = str_true_enc(str);
9011 if (case_option_single_p(flags, enc, str)) {
9012 ret = rb_str_new(RSTRING_PTR(str), RSTRING_LEN(str));
9013 str_enc_copy_direct(ret, str);
9014 downcase_single(ret);
9015 }
9016 else if (flags&ONIGENC_CASE_ASCII_ONLY) {
9017 ret = rb_str_new(0, RSTRING_LEN(str));
9018 rb_str_ascii_casemap(str, ret, &flags, enc);
9019 }
9020 else {
9021 ret = rb_str_casemap(str, &flags, enc);
9022 }
9023
9024 return ret;
9025}
9026
9027static bool
9028capitalize_single(VALUE str)
9029{
9030 char *s = RSTRING_PTR(str), *send = RSTRING_END(str);
9031 bool modified = false;
9032
9033 if (s < send) {
9034 unsigned int c = (unsigned char)*s;
9035
9036 if ('a' <= c && c <= 'z') {
9037 *s = 'A' + (c - 'a');
9038 modified = true;
9039 }
9040 s++;
9041 }
9042 while (s < send) {
9043 unsigned int c = (unsigned char)*s;
9044
9045 if ('A' <= c && c <= 'Z') {
9046 *s = 'a' + (c - 'A');
9047 modified = true;
9048 }
9049 s++;
9050 }
9051
9052 return modified;
9053}
9054
9055/*
9056 * call-seq:
9057 * capitalize!(mapping = :ascii) -> self or nil
9058 *
9059 * Like String#capitalize, except that:
9060 *
9061 * - Changes character casings in +self+ (not in a copy of +self+).
9062 * - Returns +self+ if any changes are made, +nil+ otherwise.
9063 *
9064 * Related: See {Modifying}[rdoc-ref:String@Modifying].
9065 */
9066
9067static VALUE
9068rb_str_capitalize_bang(int argc, VALUE *argv, VALUE str)
9069{
9070 rb_encoding *enc;
9071 OnigCaseFoldType flags = ONIGENC_CASE_UPCASE | ONIGENC_CASE_TITLECASE;
9072
9073 flags = check_case_options(argc, argv, flags);
9074 str_modify_keep_cr(str);
9075 enc = str_true_enc(str);
9076 if (RSTRING_LEN(str) == 0 || !RSTRING_PTR(str)) return Qnil;
9077 if (case_option_single_p(flags, enc, str)) {
9078 if (capitalize_single(str))
9079 flags |= ONIGENC_CASE_MODIFIED;
9080 }
9081 else if (flags&ONIGENC_CASE_ASCII_ONLY)
9082 rb_str_ascii_casemap(str, str, &flags, enc);
9083 else
9084 str_shared_replace(str, rb_str_casemap(str, &flags, enc));
9085
9086 if (ONIGENC_CASE_MODIFIED&flags) return str;
9087 return Qnil;
9088}
9089
9090
9091/*
9092 * call-seq:
9093 * capitalize(mapping = :ascii) -> new_string
9094 *
9095 * :include: doc/string/capitalize.rdoc
9096 *
9097 */
9098
9099static VALUE
9100rb_str_capitalize(int argc, VALUE *argv, VALUE str)
9101{
9102 rb_encoding *enc;
9103 OnigCaseFoldType flags = ONIGENC_CASE_UPCASE | ONIGENC_CASE_TITLECASE;
9104 VALUE ret;
9105
9106 flags = check_case_options(argc, argv, flags);
9107 enc = str_true_enc(str);
9108 if (RSTRING_LEN(str) == 0 || !RSTRING_PTR(str)) return str;
9109 if (case_option_single_p(flags, enc, str)) {
9110 ret = rb_str_new(RSTRING_PTR(str), RSTRING_LEN(str));
9111 str_enc_copy_direct(ret, str);
9112 capitalize_single(ret);
9113 }
9114 else if (flags&ONIGENC_CASE_ASCII_ONLY) {
9115 ret = rb_str_new(0, RSTRING_LEN(str));
9116 rb_str_ascii_casemap(str, ret, &flags, enc);
9117 }
9118 else {
9119 ret = rb_str_casemap(str, &flags, enc);
9120 }
9121 return ret;
9122}
9123
9124
9125/*
9126 * call-seq:
9127 * swapcase!(mapping) -> self or nil
9128 *
9129 * Like String#swapcase, except that:
9130 *
9131 * - Changes are made to +self+, not to copy of +self+.
9132 * - Returns +self+ if any changes are made, +nil+ otherwise.
9133 *
9134 * Related: see {Modifying}[rdoc-ref:String@Modifying].
9135 */
9136
9137static VALUE
9138rb_str_swapcase_bang(int argc, VALUE *argv, VALUE str)
9139{
9140 rb_encoding *enc;
9141 OnigCaseFoldType flags = ONIGENC_CASE_UPCASE | ONIGENC_CASE_DOWNCASE;
9142
9143 flags = check_case_options(argc, argv, flags);
9144 str_modify_keep_cr(str);
9145 enc = str_true_enc(str);
9146 if (flags&ONIGENC_CASE_ASCII_ONLY)
9147 rb_str_ascii_casemap(str, str, &flags, enc);
9148 else
9149 str_shared_replace(str, rb_str_casemap(str, &flags, enc));
9150
9151 if (ONIGENC_CASE_MODIFIED&flags) return str;
9152 return Qnil;
9153}
9154
9155
9156/*
9157 * call-seq:
9158 * swapcase(mapping = :ascii) -> new_string
9159 *
9160 * :include: doc/string/swapcase.rdoc
9161 *
9162 */
9163
9164static VALUE
9165rb_str_swapcase(int argc, VALUE *argv, VALUE str)
9166{
9167 rb_encoding *enc;
9168 OnigCaseFoldType flags = ONIGENC_CASE_UPCASE | ONIGENC_CASE_DOWNCASE;
9169 VALUE ret;
9170
9171 flags = check_case_options(argc, argv, flags);
9172 enc = str_true_enc(str);
9173 if (RSTRING_LEN(str) == 0 || !RSTRING_PTR(str)) return str_duplicate(rb_cString, str);
9174 if (flags&ONIGENC_CASE_ASCII_ONLY) {
9175 ret = rb_str_new(0, RSTRING_LEN(str));
9176 rb_str_ascii_casemap(str, ret, &flags, enc);
9177 }
9178 else {
9179 ret = rb_str_casemap(str, &flags, enc);
9180 }
9181 return ret;
9182}
9183
9184typedef unsigned char *USTR;
9185
9186struct tr {
9187 int gen;
9188 unsigned int now, max;
9189 const char *p, *pend;
9190};
9191
9192static unsigned int
9193trnext(struct tr *t, rb_encoding *enc)
9194{
9195 int n;
9196
9197 for (;;) {
9198 nextpart:
9199 if (!t->gen) {
9200 if (t->p == t->pend) return -1;
9201 if (rb_enc_ascget(t->p, t->pend, &n, enc) == '\\' && t->p + n < t->pend) {
9202 t->p += n;
9203 }
9204 t->now = rb_enc_codepoint_len(t->p, t->pend, &n, enc);
9205 t->p += n;
9206 if (rb_enc_ascget(t->p, t->pend, &n, enc) == '-' && t->p + n < t->pend) {
9207 t->p += n;
9208 if (t->p < t->pend) {
9209 unsigned int c = rb_enc_codepoint_len(t->p, t->pend, &n, enc);
9210 t->p += n;
9211 if (t->now > c) {
9212 if (t->now < 0x80 && c < 0x80) {
9213 rb_raise(rb_eArgError,
9214 "invalid range \"%c-%c\" in string transliteration",
9215 t->now, c);
9216 }
9217 else {
9218 rb_raise(rb_eArgError, "invalid range in string transliteration");
9219 }
9220 continue; /* not reached */
9221 }
9222 else if (t->now < c) {
9223 t->gen = 1;
9224 t->max = c;
9225 }
9226 }
9227 }
9228 return t->now;
9229 }
9230 else {
9231 while (ONIGENC_CODE_TO_MBCLEN(enc, ++t->now) <= 0) {
9232 if (t->now == t->max) {
9233 t->gen = 0;
9234 goto nextpart;
9235 }
9236 }
9237 if (t->now < t->max) {
9238 return t->now;
9239 }
9240 else {
9241 t->gen = 0;
9242 return t->max;
9243 }
9244 }
9245 }
9246}
9247
9248static VALUE rb_str_delete_bang(int,VALUE*,VALUE);
9249
9250static VALUE
9251tr_trans(VALUE str, VALUE src, VALUE repl, int sflag)
9252{
9253 const unsigned int errc = -1;
9254 unsigned int trans[256];
9255 rb_encoding *enc, *e1, *e2;
9256 struct tr trsrc, trrepl;
9257 int cflag = 0;
9258 unsigned int c, c0, last = 0;
9259 int modify = 0, i, l;
9260 unsigned char *s, *send;
9261 VALUE hash = 0;
9262 int singlebyte = single_byte_optimizable(str);
9263 int termlen;
9264 int cr;
9265
9266#define CHECK_IF_ASCII(c) \
9267 (void)((cr == ENC_CODERANGE_7BIT && !rb_isascii(c)) ? \
9268 (cr = ENC_CODERANGE_VALID) : 0)
9269
9270 StringValue(src);
9271 StringValue(repl);
9272 if (RSTRING_LEN(str) == 0 || !RSTRING_PTR(str)) return Qnil;
9273 if (RSTRING_LEN(repl) == 0) {
9274 return rb_str_delete_bang(1, &src, str);
9275 }
9276
9277 cr = ENC_CODERANGE(str);
9278 e1 = rb_enc_check(str, src);
9279 e2 = rb_enc_check(str, repl);
9280 if (e1 == e2) {
9281 enc = e1;
9282 }
9283 else {
9284 enc = rb_enc_check(src, repl);
9285 }
9286 trsrc.p = RSTRING_PTR(src); trsrc.pend = trsrc.p + RSTRING_LEN(src);
9287 if (RSTRING_LEN(src) > 1 &&
9288 rb_enc_ascget(trsrc.p, trsrc.pend, &l, enc) == '^' &&
9289 trsrc.p + l < trsrc.pend) {
9290 cflag = 1;
9291 trsrc.p += l;
9292 }
9293 trrepl.p = RSTRING_PTR(repl);
9294 trrepl.pend = trrepl.p + RSTRING_LEN(repl);
9295 trsrc.gen = trrepl.gen = 0;
9296 trsrc.now = trrepl.now = 0;
9297 trsrc.max = trrepl.max = 0;
9298
9299 if (cflag) {
9300 for (i=0; i<256; i++) {
9301 trans[i] = 1;
9302 }
9303 while ((c = trnext(&trsrc, enc)) != errc) {
9304 if (c < 256) {
9305 trans[c] = errc;
9306 }
9307 else {
9308 if (!hash) hash = rb_hash_new();
9309 rb_hash_aset(hash, UINT2NUM(c), Qtrue);
9310 }
9311 }
9312 while ((c = trnext(&trrepl, enc)) != errc)
9313 /* retrieve last replacer */;
9314 last = trrepl.now;
9315 for (i=0; i<256; i++) {
9316 if (trans[i] != errc) {
9317 trans[i] = last;
9318 }
9319 }
9320 }
9321 else {
9322 unsigned int r;
9323
9324 for (i=0; i<256; i++) {
9325 trans[i] = errc;
9326 }
9327 while ((c = trnext(&trsrc, enc)) != errc) {
9328 r = trnext(&trrepl, enc);
9329 if (r == errc) r = trrepl.now;
9330 if (c < 256) {
9331 trans[c] = r;
9332 if (rb_enc_codelen(r, enc) != 1) singlebyte = 0;
9333 }
9334 else {
9335 if (!hash) hash = rb_hash_new();
9336 rb_hash_aset(hash, UINT2NUM(c), UINT2NUM(r));
9337 }
9338 }
9339 }
9340
9341 if (cr == ENC_CODERANGE_VALID && rb_enc_asciicompat(e1))
9342 cr = ENC_CODERANGE_7BIT;
9343 str_modify_keep_cr(str);
9344 s = (unsigned char *)RSTRING_PTR(str); send = (unsigned char *)RSTRING_END(str);
9345 termlen = rb_enc_mbminlen(enc);
9346 if (sflag) {
9347 int clen, tlen;
9348 long offset, max = RSTRING_LEN(str);
9349 unsigned int save = -1;
9350 unsigned char *buf = ALLOC_N(unsigned char, max + termlen), *t = buf;
9351
9352 while (s < send) {
9353 int may_modify = 0;
9354
9355 int r = rb_enc_precise_mbclen((char *)s, (char *)send, e1);
9356 if (!MBCLEN_CHARFOUND_P(r)) {
9357 SIZED_FREE_N(buf, max + termlen);
9358 rb_raise(rb_eArgError, "invalid byte sequence in %s", rb_enc_name(e1));
9359 }
9360 clen = MBCLEN_CHARFOUND_LEN(r);
9361 c0 = c = rb_enc_mbc_to_codepoint((char *)s, (char *)send, e1);
9362
9363 tlen = enc == e1 ? clen : rb_enc_codelen(c, enc);
9364
9365 s += clen;
9366 if (c < 256) {
9367 c = trans[c];
9368 }
9369 else if (hash) {
9370 VALUE tmp = rb_hash_lookup(hash, UINT2NUM(c));
9371 if (NIL_P(tmp)) {
9372 if (cflag) c = last;
9373 else c = errc;
9374 }
9375 else if (cflag) c = errc;
9376 else c = NUM2INT(tmp);
9377 }
9378 else {
9379 c = errc;
9380 }
9381 if (c != (unsigned int)-1) {
9382 if (save == c) {
9383 CHECK_IF_ASCII(c);
9384 continue;
9385 }
9386 save = c;
9387 tlen = rb_enc_codelen(c, enc);
9388 modify = 1;
9389 }
9390 else {
9391 save = -1;
9392 c = c0;
9393 if (enc != e1) may_modify = 1;
9394 }
9395 if ((offset = t - buf) + tlen > max) {
9396 size_t MAYBE_UNUSED(old) = max + termlen;
9397 max = offset + tlen + (send - s);
9398 SIZED_REALLOC_N(buf, unsigned char, max + termlen, old);
9399 t = buf + offset;
9400 }
9401 rb_enc_mbcput(c, t, enc);
9402 if (may_modify && memcmp(s, t, tlen) != 0) {
9403 modify = 1;
9404 }
9405 CHECK_IF_ASCII(c);
9406 t += tlen;
9407 }
9408 if (!STR_EMBED_P(str)) {
9409 SIZED_FREE_N(STR_HEAP_PTR(str), STR_HEAP_SIZE(str));
9410 }
9411 TERM_FILL((char *)t, termlen);
9412 RSTRING(str)->as.heap.ptr = (char *)buf;
9413 STR_SET_LEN(str, t - buf);
9414 STR_SET_NOEMBED(str);
9415 RSTRING(str)->as.heap.aux.capa = max;
9416 }
9417 else if (rb_enc_mbmaxlen(enc) == 1 || (singlebyte && !hash)) {
9418 while (s < send) {
9419 c = (unsigned char)*s;
9420 if (trans[c] != errc) {
9421 if (!cflag) {
9422 c = trans[c];
9423 *s = c;
9424 modify = 1;
9425 }
9426 else {
9427 *s = last;
9428 modify = 1;
9429 }
9430 }
9431 CHECK_IF_ASCII(c);
9432 s++;
9433 }
9434 }
9435 else {
9436 int clen, tlen;
9437 long offset, max = (long)((send - s) * 1.2);
9438 unsigned char *buf = ALLOC_N(unsigned char, max + termlen), *t = buf;
9439
9440 while (s < send) {
9441 int may_modify = 0;
9442
9443 int r = rb_enc_precise_mbclen((char *)s, (char *)send, e1);
9444 if (!MBCLEN_CHARFOUND_P(r)) {
9445 SIZED_FREE_N(buf, max + termlen);
9446 rb_raise(rb_eArgError, "invalid byte sequence in %s", rb_enc_name(e1));
9447 }
9448 clen = MBCLEN_CHARFOUND_LEN(r);
9449 c0 = c = rb_enc_mbc_to_codepoint((char *)s, (char *)send, e1);
9450
9451 tlen = enc == e1 ? clen : rb_enc_codelen(c, enc);
9452
9453 if (c < 256) {
9454 c = trans[c];
9455 }
9456 else if (hash) {
9457 VALUE tmp = rb_hash_lookup(hash, UINT2NUM(c));
9458 if (NIL_P(tmp)) {
9459 if (cflag) c = last;
9460 else c = errc;
9461 }
9462 else if (cflag) c = errc;
9463 else c = NUM2INT(tmp);
9464 }
9465 else {
9466 c = cflag ? last : errc;
9467 }
9468 if (c != errc) {
9469 tlen = rb_enc_codelen(c, enc);
9470 modify = 1;
9471 }
9472 else {
9473 c = c0;
9474 if (enc != e1) may_modify = 1;
9475 }
9476 if ((offset = t - buf) + tlen > max) {
9477 size_t MAYBE_UNUSED(old) = max + termlen;
9478 max = offset + tlen + (long)((send - s) * 1.2);
9479 SIZED_REALLOC_N(buf, unsigned char, max + termlen, old);
9480 t = buf + offset;
9481 }
9482
9483 rb_enc_mbcput(c, t, enc);
9484 if (may_modify && memcmp(s, t, tlen) != 0) {
9485 modify = 1;
9486 }
9487 CHECK_IF_ASCII(c);
9488 s += clen;
9489 t += tlen;
9490 }
9491 if (!STR_EMBED_P(str)) {
9492 SIZED_FREE_N(STR_HEAP_PTR(str), STR_HEAP_SIZE(str));
9493 }
9494 TERM_FILL((char *)t, termlen);
9495 RSTRING(str)->as.heap.ptr = (char *)buf;
9496 STR_SET_LEN(str, t - buf);
9497 STR_SET_NOEMBED(str);
9498 RSTRING(str)->as.heap.aux.capa = max;
9499 }
9500
9501 if (modify) {
9502 if (cr != ENC_CODERANGE_BROKEN)
9503 ENC_CODERANGE_SET(str, cr);
9504 rb_enc_associate(str, enc);
9505 return str;
9506 }
9507 return Qnil;
9508}
9509
9511 unsigned char *buf;
9512 unsigned char *ptr;
9513 size_t capa;
9514 size_t initial_capa;
9515};
9516
9517static inline void
9518tr_buffer_init(struct tr_buffer *buffer, size_t initial_capa)
9519{
9520 if (initial_capa < 32) {
9521 initial_capa = 32;
9522 }
9523 *buffer = (struct tr_buffer){ .initial_capa = initial_capa };
9524}
9525
9526static inline void
9527tr_buffer_ensure_capa(struct tr_buffer *buffer, size_t extra_capa)
9528{
9529 size_t offset = buffer->ptr - buffer->buf;
9530 size_t required_capa = offset + extra_capa;
9531 if (UNLIKELY(buffer->capa < required_capa)) {
9532 size_t new_capa = buffer->capa ? buffer->capa : buffer->initial_capa;
9533 RUBY_ASSERT(new_capa >= 32); // Lower would cause infinite loop
9534 while (new_capa < required_capa) {
9535 new_capa = (size_t)(new_capa * 1.2);
9536 }
9537 SIZED_REALLOC_N(buffer->buf, unsigned char, new_capa, buffer->capa);
9538 buffer->ptr = buffer->buf + offset;
9539 buffer->capa = new_capa;
9540 }
9541}
9542
9543static inline void
9544tr_buffer_append(struct tr_buffer *buffer, const unsigned char *ptr, size_t len)
9545{
9546 if (len) {
9547 tr_buffer_ensure_capa(buffer, len);
9548 memcpy(buffer->ptr, ptr, len);
9549 buffer->ptr += len;
9550 }
9551}
9552
9553static inline void
9554tr_buffer_append_str(struct tr_buffer *buffer, VALUE str)
9555{
9556 tr_buffer_append(buffer, (unsigned char *)RSTRING_PTR(str), RSTRING_LEN(str));
9557}
9558
9559static inline void
9560tr_buffer_mbcput(struct tr_buffer *buffer, int codepoint, rb_encoding *enc)
9561{
9562 tr_buffer_ensure_capa(buffer, 4);
9563 buffer->ptr += rb_enc_mbcput(codepoint, buffer->ptr, enc);
9564}
9565
9566static inline void
9567tr_buffer_free(struct tr_buffer *buffer)
9568{
9569 if (buffer->buf) {
9570 SIZED_FREE_N(buffer->buf, buffer->capa);
9571 }
9572}
9573
9574struct tr_pair {
9575 VALUE search;
9576 VALUE replace;
9577};
9578
9580 struct tr_pair *pairs;
9581 size_t index;
9582 rb_encoding *enc;
9583 int cr;
9584};
9585
9586static int
9587tr_trans_pairs_coerce_i(st_data_t key, st_data_t value, st_data_t _args)
9588{
9589 struct tr_trans_pairs_coerce_args *args = (struct tr_trans_pairs_coerce_args *)_args;
9590 struct tr_pair *pair = &args->pairs[args->index];
9591 args->index++;
9592
9593 VALUE search = (VALUE)key;
9594 VALUE replace = (VALUE)value;
9595 StringValue(search);
9596 StringValue(replace);
9597
9598 if (RSTRING_LEN(search) != 1 && str_strlen(search, NULL) != 1) {
9599 rb_raise(rb_eArgError, "keys must be of size 1"); // TODO: better error message
9600 }
9601
9602 args->enc = rb_enc_check_multi_str(args->enc, &args->cr, search);
9603 args->enc = rb_enc_check_multi_str(args->enc, &args->cr, replace);
9604
9605 pair->search = search;
9606 pair->replace = replace;
9607 return ST_CONTINUE;
9608}
9609
9610#define TR_TRANS_PAIRS_SIMD_MAX_NEEDLES 16
9611
9613 const unsigned char *s;
9614 const unsigned char *send;
9615
9616#ifdef HAVE_SIMD
9617 unsigned char needles[TR_TRANS_PAIRS_SIMD_MAX_NEEDLES];
9618 unsigned int needles_count;
9619#ifdef HAVE_SIMD_NEON
9620 uint64_t matches_bitmap;
9621#endif
9622#ifdef HAVE_SIMD_SSE2
9623 int matches_bitmap;
9624#endif
9625#endif
9626
9627 VALUE trans_table[256];
9628};
9629
9630static inline VALUE
9631tr_trans_pairs_search_basic(struct tr_trans_pairs_search *search)
9632{
9633 while (search->s < search->send) {
9634 VALUE repl = search->trans_table[*search->s];
9635 if (UNLIKELY(repl)) {
9636 return repl;
9637 }
9638
9639 search->s++;
9640 }
9641
9642 return 0;
9643}
9644
9645#ifdef HAVE_SIMD_SSE2
9646static inline VALUE
9647tr_trans_pairs_next_match_sse2(struct tr_trans_pairs_search *search)
9648{
9649 RUBY_ASSERT(search->matches_bitmap > 0);
9650 size_t trailing_zeros = (size_t)ntz_int32(search->matches_bitmap);
9651
9652 RUBY_ASSERT(trailing_zeros < (sizeof(search->matches_bitmap) * CHAR_BIT));
9653 search->matches_bitmap >>= trailing_zeros;
9654 search->s += trailing_zeros;
9655
9656 RUBY_ASSERT(search->s <= search->send);
9657 return search->trans_table[*search->s];
9658}
9659
9660static inline VALUE
9661tr_trans_pairs_search_sse2(struct tr_trans_pairs_search *search)
9662{
9663 const unsigned int needles_count = search->needles_count;
9664 if (needles_count) {
9665 RBIMPL_ASSERT_OR_ASSUME(needles_count <= TR_TRANS_PAIRS_SIMD_MAX_NEEDLES);
9666
9667 if (search->matches_bitmap) {
9668 return tr_trans_pairs_next_match_sse2(search);
9669 }
9670
9671 if ((size_t)(search->send - search->s) >= sizeof(__m128i)) {
9672 unsigned int i;
9673 __m128i masks[TR_TRANS_PAIRS_SIMD_MAX_NEEDLES];
9674 for (i = 0; i < needles_count; i++) {
9675 masks[i] = _mm_set1_epi8(search->needles[i]);
9676 }
9677
9678 do {
9679 const __m128i bytes = _mm_loadu_si128((__m128i const *)search->s);
9680
9681 __m128i matches[TR_TRANS_PAIRS_SIMD_MAX_NEEDLES];
9682 for (i = 0; i < needles_count; i++) {
9683 matches[i] = _mm_cmpeq_epi8(bytes, masks[i]);
9684 }
9685
9686 for (i = 1; i < needles_count; i++) {
9687 matches[0] = _mm_or_si128(matches[0], matches[i]);
9688 }
9689
9690 const int bitmap = _mm_movemask_epi8(matches[0]);
9691
9692 if (bitmap) {
9693 search->matches_bitmap = bitmap;
9694 return tr_trans_pairs_next_match_sse2(search);
9695 }
9696 search->s += sizeof(__m128i);
9697 } while ((size_t)(search->send - search->s) >= sizeof(__m128i));
9698 }
9699 }
9700 return tr_trans_pairs_search_basic(search);
9701}
9702
9703#define tr_trans_pairs_search_impl tr_trans_pairs_search_sse2
9704#endif
9705
9706#ifdef HAVE_SIMD_NEON
9707static inline VALUE
9708tr_trans_pairs_next_match_neon(struct tr_trans_pairs_search *search)
9709{
9710 RUBY_ASSERT(search->matches_bitmap > 0);
9711 size_t trailing_zeros = (size_t)ntz_int64(search->matches_bitmap);
9712
9713 // uint64_t >>= 64 would be undefined behaviour
9714 RUBY_ASSERT(trailing_zeros < (sizeof(search->matches_bitmap) * CHAR_BIT));
9715 search->matches_bitmap >>= trailing_zeros;
9716 search->s += trailing_zeros / 4;
9717
9718 RUBY_ASSERT(search->s <= search->send);
9719 return search->trans_table[*search->s];
9720}
9721
9722static inline VALUE
9723tr_trans_pairs_search_neon(struct tr_trans_pairs_search *search)
9724{
9725 const unsigned int needles_count = search->needles_count;
9726 if (needles_count) {
9727 RBIMPL_ASSERT_OR_ASSUME(needles_count <= TR_TRANS_PAIRS_SIMD_MAX_NEEDLES);
9728
9729 if (search->matches_bitmap) {
9730 return tr_trans_pairs_next_match_neon(search);
9731 }
9732
9733 if ((size_t)(search->send - search->s) >= sizeof(uint8x16_t)) {
9734 unsigned int i;
9735 uint8x16_t masks[TR_TRANS_PAIRS_SIMD_MAX_NEEDLES];
9736 for (i = 0; i < needles_count; i++) {
9737 masks[i] = vdupq_n_u8(search->needles[i]);
9738 }
9739
9740 do {
9741 const uint8x16_t bytes = vld1q_u8(search->s);
9742
9743 uint8x16_t matches[TR_TRANS_PAIRS_SIMD_MAX_NEEDLES];
9744 for (i = 0; i < needles_count; i++) {
9745 matches[i] = vceqq_u8(bytes, masks[i]);
9746 }
9747
9748 for (i = 1; i < needles_count; i++) {
9749 matches[0] = vorrq_u8(matches[0], matches[i]);
9750 }
9751
9752 const uint8x8_t res = vshrn_n_u16(vreinterpretq_u16_u8(matches[0]), 4);
9753 const uint64_t bitmap = vget_lane_u64(vreinterpret_u64_u8(res), 0);
9754
9755 if (bitmap) {
9756 search->matches_bitmap = bitmap & 0x8888888888888888ull;
9757 return tr_trans_pairs_next_match_neon(search);
9758 }
9759 search->s += sizeof(uint8x16_t);
9760 } while ((size_t)(search->send - search->s) >= sizeof(uint8x16_t));
9761 }
9762 }
9763 return tr_trans_pairs_search_basic(search);
9764}
9765
9766#define tr_trans_pairs_search_impl tr_trans_pairs_search_neon
9767#endif
9768
9769#ifndef tr_trans_pairs_search_impl
9770#define tr_trans_pairs_search_impl tr_trans_pairs_search_basic
9771#endif
9772
9773static inline void
9774tr_trans_pairs_consume_match(struct tr_trans_pairs_search *search)
9775{
9776 search->s++;
9777#ifdef HAVE_SIMD
9778 search->matches_bitmap >>= 1;
9779#endif
9780}
9781
9782static VALUE
9783tr_trans_pairs(VALUE str, VALUE pairs_val)
9784{
9785 Check_Type(pairs_val, T_HASH);
9786 size_t pairs_count = RHASH_SIZE(pairs_val);
9787 mustnot_broken(str);
9788 rb_str_modify(str);
9789
9790 if (RSTRING_LEN(str) == 0 || !RSTRING_PTR(str) || pairs_count == 0) return Qnil;
9791
9792 VALUE pairs_handle;
9793 struct tr_pair *pairs = ALLOCV_N(struct tr_pair, pairs_handle, pairs_count);
9794
9795 int cr = rb_enc_str_coderange(str);
9796 rb_encoding *enc = rb_str_enc_get(str);
9797
9798 struct tr_trans_pairs_coerce_args coerce_args = {
9799 .pairs = pairs,
9800 .enc = enc,
9801 .cr = cr,
9802 };
9803 rb_hash_foreach(pairs_val, tr_trans_pairs_coerce_i, (VALUE)&coerce_args);
9804 rb_encoding *e1 = coerce_args.enc;
9805
9806 /* Keys could be deleted from pairs_val during rb_hash_foreach when coercing
9807 * the keys/values, so we need to update pairs_count to the number of pairs we
9808 * were actually able to extract from pairs_val. */
9809 pairs_count = coerce_args.index;
9810
9811 VALUE hash = 0;
9812
9813 const unsigned char *sstart = (unsigned char *)RSTRING_PTR(str);
9814 long str_len = RSTRING_LEN(str);
9815 int termlen = rb_enc_mbminlen(e1);
9816
9817 struct tr_buffer buffer;
9818 tr_buffer_init(&buffer, str_len);
9819 bool modify = false;
9820
9821 if (RB_LIKELY(rb_str_encindex_fastpath(rb_enc_to_index(e1)))) {
9822
9823 struct tr_trans_pairs_search search = {
9824 .s = sstart,
9825 .send = sstart + str_len,
9826 };
9827
9828 for (size_t index = 0; index < pairs_count; index++) {
9829 struct tr_pair *pair = &pairs[index];
9830
9831 char *ptr = RSTRING_PTR(pair->search);
9832 unsigned int codepoint = rb_enc_mbc_to_codepoint(ptr, RSTRING_END(pair->search), e1);
9833
9834 const unsigned char first_byte = (unsigned char)*ptr;
9835
9836#ifdef HAVE_SIMD
9837 if (pairs_count <= TR_TRANS_PAIRS_SIMD_MAX_NEEDLES) {
9838 search.needles[index] = first_byte;
9839 search.needles_count++;
9840 }
9841#endif
9842
9843 if (rb_enc_codelen(codepoint, e1) == 1) {
9844 search.trans_table[first_byte] = pair->replace;
9845 }
9846 else {
9847 search.trans_table[first_byte] = Qundef;
9848 if (!hash) {
9849 hash = rb_obj_hide(rb_hash_new_capa(pairs_count));
9850 }
9851 rb_hash_aset(hash, UINT2NUM(codepoint), pair->replace);
9852 }
9853 }
9854
9855 const unsigned char *checkpoint = search.s;
9856 VALUE repl;
9857 while ((repl = tr_trans_pairs_search_impl(&search))) {
9858 int clen = 1;
9859
9860 if (UNLIKELY(repl == Qundef)) {
9861 unsigned int c = rb_enc_mbc_to_codepoint((char *)search.s, (char *)search.send, e1);
9862 clen = rb_enc_codelen(c, e1);
9863 repl = rb_hash_lookup2(hash, UINT2NUM(c), 0);
9864 if (!repl) {
9865 tr_trans_pairs_consume_match(&search);
9866 continue;
9867 }
9868 }
9870
9871 modify = true;
9872
9873 if (checkpoint < search.s) {
9874 tr_buffer_append(&buffer, checkpoint, search.s - checkpoint);
9875 }
9876 tr_buffer_append_str(&buffer, repl);
9877 checkpoint = search.s + clen;
9878 tr_trans_pairs_consume_match(&search);
9879
9880 if (cr == ENC_CODERANGE_7BIT && rb_enc_str_coderange(repl) != ENC_CODERANGE_7BIT) {
9882 }
9883 }
9884
9885 if (modify && checkpoint < search.s) {
9886 tr_buffer_append(&buffer, checkpoint, search.s - checkpoint);
9887 }
9888 }
9889 else {
9890 const unsigned char *s = sstart;
9891 const unsigned char *send = sstart + str_len;
9892
9893 hash = rb_obj_hide(rb_hash_new_capa(pairs_count));
9894
9895 for (size_t index = 0; index < pairs_count; index++) {
9896 struct tr_pair *pair = &pairs[index];
9897
9898 unsigned int codepoint = rb_enc_mbc_to_codepoint(RSTRING_PTR(pair->search), RSTRING_END(pair->search), e1);
9899 rb_hash_aset(hash, UINT2NUM(codepoint), pair->replace);
9900 }
9901
9902 while (s < send) {
9903 bool may_modify = false;
9904
9905 int r = rb_enc_precise_mbclen((char *)s, (char *)send, e1);
9906 if (!MBCLEN_CHARFOUND_P(r)) {
9907 tr_buffer_free(&buffer);
9908 rb_raise(rb_eArgError, "invalid byte sequence in %s", rb_enc_name(e1));
9909 }
9910 int clen = MBCLEN_CHARFOUND_LEN(r);
9911 unsigned int c = rb_enc_mbc_to_codepoint((char *)s, (char *)send, e1);
9912 unsigned int c0 = c;
9913
9914 long tlen = enc == e1 ? clen : rb_enc_codelen(c, e1);
9915
9916 VALUE replacement = rb_hash_lookup(hash, UINT2NUM(c));
9917 if (NIL_P(replacement)) {
9918 tlen = enc == e1 ? clen : rb_enc_codelen(c, enc);
9919 c = c0;
9920 if (enc != e1) may_modify = true;
9921 }
9922 else {
9923 tlen = RSTRING_LEN(replacement);
9924 modify = true;
9925 }
9926
9927 if (NIL_P(replacement)) {
9928 tr_buffer_mbcput(&buffer, c, enc);
9929 }
9930 else {
9931 tr_buffer_append_str(&buffer, replacement);
9932 }
9933
9934 if (may_modify && memcmp(s, buffer.ptr - tlen, tlen) != 0) {
9935 modify = true;
9936 }
9937
9938 if (cr == ENC_CODERANGE_7BIT && !rb_isascii(c)) {
9940 }
9941
9942 s += clen;
9943 }
9944 }
9945
9946 if (!modify) {
9947 return Qnil;
9948 }
9949
9950 if (!STR_EMBED_P(str)) {
9951 SIZED_FREE_N(STR_HEAP_PTR(str), STR_HEAP_SIZE(str));
9952 }
9953 tr_buffer_ensure_capa(&buffer, termlen);
9954 TERM_FILL((char *)buffer.ptr, termlen);
9955 RSTRING(str)->as.heap.ptr = (char *)buffer.buf;
9956 STR_SET_LEN(str, buffer.ptr - buffer.buf);
9957 STR_SET_NOEMBED(str);
9958 RSTRING(str)->as.heap.aux.capa = buffer.capa - termlen;
9959
9960 RB_GC_GUARD(hash);
9961
9962 if (cr != ENC_CODERANGE_BROKEN)
9963 ENC_CODERANGE_SET(str, cr);
9964 rb_enc_associate(str, e1);
9965 return str;
9966}
9967
9968/*
9969 * call-seq:
9970 * tr!(selector, replacements) -> self or nil
9971 * tr!(pairs) -> self or nil
9972 *
9973 * Like String#tr, except:
9974 *
9975 * - Performs substitutions in +self+ (not in a copy of +self+).
9976 * - Returns +self+ if any modifications were made, +nil+ otherwise.
9977 *
9978 * Related: {Modifying}[rdoc-ref:String@Modifying].
9979 */
9980
9981static VALUE
9982rb_str_tr_bang(int argc, VALUE *argv, VALUE str)
9983{
9984 rb_check_arity(argc, 1, 2);
9985
9986 if (argc == 1) {
9987 VALUE pairs = argv[0];
9988 return tr_trans_pairs(str, pairs);
9989 }
9990
9991 VALUE src = argv[0], repl = argv[1];
9992 return tr_trans(str, src, repl, 0);
9993}
9994
9995
9996/*
9997 * call-seq:
9998 * tr(selector, replacements) -> new_string
9999 * tr(pairs) -> new_string
10000 *
10001 * Accepts either a +selector+ and a +replacements+ string,
10002 * or a single +pairs+ Hash.
10003 *
10004 * When a +pairs+ Hash is provided the keys, returns a copy of +self+ with
10005 * the keys of the hash replaced by the values.
10006 *
10007 * - They keys must be strings containing a single codepoints.
10008 * - The values can be of any length.
10009 *
10010 * Example:
10011 *
10012 * 'hello'.tr('e' => 'er', 'l' => '', 'o' => 'o !') #=> "hero !"
10013 *
10014 * When +selector+ and +replacements+are provided, returns a copy of +self+
10015 * with each character specified by string +selector+ translated to the
10016 * corresponding character in string +replacements+.
10017 * The correspondence is _positional_:
10018 *
10019 * - Each occurrence of the first character specified by +selector+
10020 * is translated to the first character in +replacements+.
10021 * - Each occurrence of the second character specified by +selector+
10022 * is translated to the second character in +replacements+.
10023 * - And so on.
10024 *
10025 * Example:
10026 *
10027 * 'hello'.tr('el', 'ip') #=> "hippo"
10028 *
10029 * If +replacements+ is shorter than +selector+,
10030 * it is implicitly padded with its own last character:
10031 *
10032 * 'hello'.tr('aeiou', '-') # => "h-ll-"
10033 * 'hello'.tr('aeiou', 'AA-') # => "hAll-"
10034 *
10035 * Arguments +selector+ and +replacements+ must be valid character selectors
10036 * (see {Character Selectors}[rdoc-ref:character_selectors.rdoc]),
10037 * and may use any of its valid forms, including negation, ranges, and escapes:
10038 *
10039 * 'hello'.tr('^aeiou', '-') # => "-e--o" # Negation.
10040 * 'ibm'.tr('b-z', 'a-z') # => "hal" # Range.
10041 * 'hel^lo'.tr('\^aeiou', '-') # => "h-l-l-" # Escaped leading caret.
10042 * 'i-b-m'.tr('b\-z', 'a-z') # => "ibabm" # Escaped embedded hyphen.
10043 * 'foo\\bar'.tr('ab\\', 'XYZ') # => "fooZYXr" # Escaped backslash.
10044 *
10045 * Related: see {Converting to New String}[rdoc-ref:String@Converting+to+New+String].
10046 */
10047
10048static VALUE
10049rb_str_tr(int argc, VALUE *argv, VALUE str)
10050{
10051 rb_check_arity(argc, 1, 2);
10052
10053 str = str_duplicate(rb_cString, str);
10054
10055 if (argc == 1) {
10056 VALUE pairs = argv[0];
10057 VALUE result = tr_trans_pairs(str, pairs);
10058 if (NIL_P(result)) result = str;
10059 return str;
10060 }
10061
10062 VALUE src = argv[0], repl = argv[1];
10063 tr_trans(str, src, repl, 0);
10064 return str;
10065}
10066
10067#define TR_TABLE_MAX (UCHAR_MAX+1)
10068#define TR_TABLE_SIZE (TR_TABLE_MAX+1)
10069static void
10070tr_setup_table(VALUE str, char stable[TR_TABLE_SIZE], int first,
10071 VALUE *tablep, VALUE *ctablep, rb_encoding *enc)
10072{
10073 const unsigned int errc = -1;
10074 char buf[TR_TABLE_MAX];
10075 struct tr tr;
10076 unsigned int c;
10077 VALUE table = 0, ptable = 0;
10078 int i, l, cflag = 0;
10079
10080 tr.p = RSTRING_PTR(str); tr.pend = tr.p + RSTRING_LEN(str);
10081 tr.gen = tr.now = tr.max = 0;
10082
10083 if (RSTRING_LEN(str) > 1 && rb_enc_ascget(tr.p, tr.pend, &l, enc) == '^') {
10084 cflag = 1;
10085 tr.p += l;
10086 }
10087 if (first) {
10088 for (i=0; i<TR_TABLE_MAX; i++) {
10089 stable[i] = 1;
10090 }
10091 stable[TR_TABLE_MAX] = cflag;
10092 }
10093 else if (stable[TR_TABLE_MAX] && !cflag) {
10094 stable[TR_TABLE_MAX] = 0;
10095 }
10096 for (i=0; i<TR_TABLE_MAX; i++) {
10097 buf[i] = cflag;
10098 }
10099
10100 while ((c = trnext(&tr, enc)) != errc) {
10101 if (c < TR_TABLE_MAX) {
10102 buf[(unsigned char)c] = !cflag;
10103 }
10104 else {
10105 VALUE key = UINT2NUM(c);
10106
10107 if (!table && (first || *tablep || stable[TR_TABLE_MAX])) {
10108 if (cflag) {
10109 ptable = *ctablep;
10110 table = ptable ? ptable : rb_hash_new();
10111 *ctablep = table;
10112 }
10113 else {
10114 table = rb_hash_new();
10115 ptable = *tablep;
10116 *tablep = table;
10117 }
10118 }
10119 if (table && (!ptable || (cflag ^ !NIL_P(rb_hash_aref(ptable, key))))) {
10120 rb_hash_aset(table, key, Qtrue);
10121 }
10122 }
10123 }
10124 for (i=0; i<TR_TABLE_MAX; i++) {
10125 stable[i] = stable[i] && buf[i];
10126 }
10127 if (!table && !cflag) {
10128 *tablep = 0;
10129 }
10130}
10131
10132
10133static int
10134tr_find(unsigned int c, const char table[TR_TABLE_SIZE], VALUE del, VALUE nodel)
10135{
10136 if (c < TR_TABLE_MAX) {
10137 return table[c] != 0;
10138 }
10139 else {
10140 VALUE v = UINT2NUM(c);
10141
10142 if (del) {
10143 if (!NIL_P(rb_hash_lookup(del, v)) &&
10144 (!nodel || NIL_P(rb_hash_lookup(nodel, v)))) {
10145 return TRUE;
10146 }
10147 }
10148 else if (nodel && !NIL_P(rb_hash_lookup(nodel, v))) {
10149 return FALSE;
10150 }
10151 return table[TR_TABLE_MAX] ? TRUE : FALSE;
10152 }
10153}
10154
10155/*
10156 * call-seq:
10157 * delete!(*selectors) -> self or nil
10158 *
10159 * Like String#delete, but modifies +self+ in place;
10160 * returns +self+ if any characters were deleted, +nil+ otherwise.
10161 *
10162 * Related: see {Modifying}[rdoc-ref:String@Modifying].
10163 */
10164
10165static VALUE
10166rb_str_delete_bang(int argc, VALUE *argv, VALUE str)
10167{
10168 char squeez[TR_TABLE_SIZE];
10169 rb_encoding *enc = 0;
10170 char *s, *send, *t;
10171 VALUE del = 0, nodel = 0;
10172 int modify = 0;
10173 int i, ascompat, cr;
10174
10175 if (RSTRING_LEN(str) == 0 || !RSTRING_PTR(str)) return Qnil;
10177 for (i=0; i<argc; i++) {
10178 VALUE s = argv[i];
10179
10180 StringValue(s);
10181 enc = rb_enc_check(str, s);
10182 tr_setup_table(s, squeez, i==0, &del, &nodel, enc);
10183 }
10184
10185 str_modify_keep_cr(str);
10186 ascompat = rb_enc_asciicompat(enc);
10187 s = t = RSTRING_PTR(str);
10188 send = RSTRING_END(str);
10189 cr = ascompat ? ENC_CODERANGE_7BIT : ENC_CODERANGE_VALID;
10190 while (s < send) {
10191 unsigned int c;
10192 int clen;
10193
10194 if (ascompat && (c = *(unsigned char*)s) < 0x80) {
10195 if (squeez[c]) {
10196 modify = 1;
10197 }
10198 else {
10199 if (t != s) *t = c;
10200 t++;
10201 }
10202 s++;
10203 }
10204 else {
10205 c = rb_enc_codepoint_len(s, send, &clen, enc);
10206
10207 if (tr_find(c, squeez, del, nodel)) {
10208 modify = 1;
10209 }
10210 else {
10211 if (t != s) rb_enc_mbcput(c, t, enc);
10212 t += clen;
10214 }
10215 s += clen;
10216 }
10217 }
10218 TERM_FILL(t, TERM_LEN(str));
10219 STR_SET_LEN(str, t - RSTRING_PTR(str));
10220 ENC_CODERANGE_SET(str, cr);
10221
10222 if (modify) return str;
10223 return Qnil;
10224}
10225
10226
10227/*
10228 * call-seq:
10229 * delete(*selectors) -> new_string
10230 *
10231 * :include: doc/string/delete.rdoc
10232 *
10233 */
10234
10235static VALUE
10236rb_str_delete(int argc, VALUE *argv, VALUE str)
10237{
10238 str = str_duplicate(rb_cString, str);
10239 rb_str_delete_bang(argc, argv, str);
10240 return str;
10241}
10242
10243
10244/*
10245 * call-seq:
10246 * squeeze!(*selectors) -> self or nil
10247 *
10248 * Like String#squeeze, except that:
10249 *
10250 * - Characters are squeezed in +self+ (not in a copy of +self+).
10251 * - Returns +self+ if any changes are made, +nil+ otherwise.
10252 *
10253 * Related: See {Modifying}[rdoc-ref:String@Modifying].
10254 */
10255
10256static VALUE
10257rb_str_squeeze_bang(int argc, VALUE *argv, VALUE str)
10258{
10259 char squeez[TR_TABLE_SIZE];
10260 rb_encoding *enc = 0;
10261 VALUE del = 0, nodel = 0;
10262 unsigned char *s, *send, *t;
10263 int i, modify = 0;
10264 int ascompat, singlebyte = single_byte_optimizable(str);
10265 unsigned int save;
10266
10267 if (argc == 0) {
10268 enc = STR_ENC_GET(str);
10269 }
10270 else {
10271 for (i=0; i<argc; i++) {
10272 VALUE s = argv[i];
10273
10274 StringValue(s);
10275 enc = rb_enc_check(str, s);
10276 if (singlebyte && !single_byte_optimizable(s))
10277 singlebyte = 0;
10278 tr_setup_table(s, squeez, i==0, &del, &nodel, enc);
10279 }
10280 }
10281
10282 str_modify_keep_cr(str);
10283 s = t = (unsigned char *)RSTRING_PTR(str);
10284 if (!s || RSTRING_LEN(str) == 0) return Qnil;
10285 send = (unsigned char *)RSTRING_END(str);
10286 save = -1;
10287 ascompat = rb_enc_asciicompat(enc);
10288
10289 if (singlebyte) {
10290 while (s < send) {
10291 unsigned int c = *s++;
10292 if (c != save || (argc > 0 && !squeez[c])) {
10293 *t++ = save = c;
10294 }
10295 }
10296 }
10297 else {
10298 while (s < send) {
10299 unsigned int c;
10300 int clen;
10301
10302 if (ascompat && (c = *s) < 0x80) {
10303 if (c != save || (argc > 0 && !squeez[c])) {
10304 *t++ = save = c;
10305 }
10306 s++;
10307 }
10308 else {
10309 c = rb_enc_codepoint_len((char *)s, (char *)send, &clen, enc);
10310
10311 if (c != save || (argc > 0 && !tr_find(c, squeez, del, nodel))) {
10312 if (t != s) rb_enc_mbcput(c, t, enc);
10313 save = c;
10314 t += clen;
10315 }
10316 s += clen;
10317 }
10318 }
10319 }
10320
10321 TERM_FILL((char *)t, TERM_LEN(str));
10322 if ((char *)t - RSTRING_PTR(str) != RSTRING_LEN(str)) {
10323 STR_SET_LEN(str, (char *)t - RSTRING_PTR(str));
10324 modify = 1;
10325 }
10326
10327 if (modify) return str;
10328 return Qnil;
10329}
10330
10331
10332/*
10333 * call-seq:
10334 * squeeze(*selectors) -> new_string
10335 *
10336 * :include: doc/string/squeeze.rdoc
10337 *
10338 */
10339
10340static VALUE
10341rb_str_squeeze(int argc, VALUE *argv, VALUE str)
10342{
10343 str = str_duplicate(rb_cString, str);
10344 rb_str_squeeze_bang(argc, argv, str);
10345 return str;
10346}
10347
10348
10349/*
10350 * call-seq:
10351 * tr_s!(selector, replacements) -> self or nil
10352 *
10353 * Like String#tr_s, except:
10354 *
10355 * - Modifies +self+ in place (not a copy of +self+).
10356 * - Returns +self+ if any changes were made, +nil+ otherwise.
10357 *
10358 * Related: {Modifying}[rdoc-ref:String@Modifying].
10359 */
10360
10361static VALUE
10362rb_str_tr_s_bang(VALUE str, VALUE src, VALUE repl)
10363{
10364 return tr_trans(str, src, repl, 1);
10365}
10366
10367
10368/*
10369 * call-seq:
10370 * tr_s(selector, replacements) -> new_string
10371 *
10372 * Like String#tr, except:
10373 *
10374 * - Also squeezes the modified portions of the translated string;
10375 * see String#squeeze.
10376 * - Returns the translated and squeezed string.
10377 *
10378 * Examples:
10379 *
10380 * 'hello'.tr_s('l', 'r') #=> "hero"
10381 * 'hello'.tr_s('el', '-') #=> "h-o"
10382 * 'hello'.tr_s('el', 'hx') #=> "hhxo"
10383 *
10384 * Related: see {Converting to New String}[rdoc-ref:String@Converting+to+New+String].
10385 *
10386 */
10387
10388static VALUE
10389rb_str_tr_s(VALUE str, VALUE src, VALUE repl)
10390{
10391 str = str_duplicate(rb_cString, str);
10392 tr_trans(str, src, repl, 1);
10393 return str;
10394}
10395
10396
10397/*
10398 * call-seq:
10399 * count(*selectors) -> integer
10400 *
10401 * :include: doc/string/count.rdoc
10402 */
10403
10404static VALUE
10405rb_str_count(int argc, VALUE *argv, VALUE str)
10406{
10407 char table[TR_TABLE_SIZE];
10408 rb_encoding *enc = 0;
10409 VALUE del = 0, nodel = 0, tstr;
10410 const char *s, *send;
10411 int i;
10412 int ascompat;
10413 size_t n = 0;
10414
10416
10417 tstr = argv[0];
10418 StringValue(tstr);
10419 enc = rb_enc_check(str, tstr);
10420 if (argc == 1) {
10421 const char *ptstr;
10422 if (RSTRING_LEN(tstr) == 1 && rb_enc_asciicompat(enc) &&
10423 (ptstr = RSTRING_PTR(tstr),
10424 ONIGENC_IS_ALLOWED_REVERSE_MATCH(enc, (const unsigned char *)ptstr, (const unsigned char *)ptstr+1)) &&
10425 !is_broken_string(str)) {
10426 int clen;
10427 unsigned char c = rb_enc_codepoint_len(ptstr, ptstr+1, &clen, enc);
10428
10429 s = RSTRING_PTR(str);
10430 if (!s || RSTRING_LEN(str) == 0) return INT2FIX(0);
10431 send = RSTRING_END(str);
10432 while (s < send) {
10433 if (*(unsigned char*)s++ == c) n++;
10434 }
10435 return SIZET2NUM(n);
10436 }
10437 }
10438
10439 tr_setup_table(tstr, table, TRUE, &del, &nodel, enc);
10440 for (i=1; i<argc; i++) {
10441 tstr = argv[i];
10442 StringValue(tstr);
10443 enc = rb_enc_check(str, tstr);
10444 tr_setup_table(tstr, table, FALSE, &del, &nodel, enc);
10445 }
10446
10447 s = RSTRING_PTR(str);
10448 if (!s || RSTRING_LEN(str) == 0) return INT2FIX(0);
10449 send = RSTRING_END(str);
10450 ascompat = rb_enc_asciicompat(enc);
10451 while (s < send) {
10452 unsigned int c;
10453
10454 if (ascompat && (c = *(unsigned char*)s) < 0x80) {
10455 if (table[c]) {
10456 n++;
10457 }
10458 s++;
10459 }
10460 else {
10461 int clen;
10462 c = rb_enc_codepoint_len(s, send, &clen, enc);
10463 if (tr_find(c, table, del, nodel)) {
10464 n++;
10465 }
10466 s += clen;
10467 }
10468 }
10469
10470 return SIZET2NUM(n);
10471}
10472
10473static VALUE
10474rb_fs_check(VALUE val)
10475{
10476 if (!NIL_P(val) && !RB_TYPE_P(val, T_STRING) && !RB_TYPE_P(val, T_REGEXP)) {
10477 val = rb_check_string_type(val);
10478 if (NIL_P(val)) return 0;
10479 }
10480 return val;
10481}
10482
10483static const char isspacetable[256] = {
10484 0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 0, 0,
10485 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10486 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10487 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10488 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10489 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10490 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10491 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10492 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10493 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10494 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10495 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10496 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10497 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10498 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10499 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0
10500};
10501
10502#define ascii_isspace(c) isspacetable[(unsigned char)(c)]
10503
10504static long
10505split_string(VALUE result, VALUE str, long beg, long len, long empty_count)
10506{
10507 if (empty_count >= 0 && len == 0) {
10508 return empty_count + 1;
10509 }
10510 if (empty_count > 0) {
10511 /* make different substrings */
10512 if (result) {
10513 do {
10514 rb_ary_push(result, str_new_empty_String(str));
10515 } while (--empty_count > 0);
10516 }
10517 else {
10518 do {
10519 rb_yield(str_new_empty_String(str));
10520 } while (--empty_count > 0);
10521 }
10522 }
10523 str = rb_str_subseq(str, beg, len);
10524 if (result) {
10525 rb_ary_push(result, str);
10526 }
10527 else {
10528 rb_yield(str);
10529 }
10530 return empty_count;
10531}
10532
10533typedef enum {
10534 SPLIT_TYPE_AWK, SPLIT_TYPE_STRING, SPLIT_TYPE_REGEXP, SPLIT_TYPE_CHARS
10535} split_type_t;
10536
10537static split_type_t
10538literal_split_pattern(VALUE spat, split_type_t default_type)
10539{
10540 rb_encoding *enc = STR_ENC_GET(spat);
10541 const char *ptr;
10542 long len;
10543 RSTRING_GETMEM(spat, ptr, len);
10544 if (len == 0) {
10545 /* Special case - split into chars */
10546 return SPLIT_TYPE_CHARS;
10547 }
10548 else if (rb_enc_asciicompat(enc)) {
10549 if (len == 1 && ptr[0] == ' ') {
10550 return SPLIT_TYPE_AWK;
10551 }
10552 }
10553 else {
10554 int l;
10555 if (rb_enc_ascget(ptr, ptr + len, &l, enc) == ' ' && len == l) {
10556 return SPLIT_TYPE_AWK;
10557 }
10558 }
10559 return default_type;
10560}
10561
10562/*
10563 * call-seq:
10564 * split(field_sep = $;, limit = 0) -> array_of_substrings
10565 * split(field_sep = $;, limit = 0) {|substring| ... } -> self
10566 *
10567 * :include: doc/string/split.rdoc
10568 *
10569 */
10570
10571static VALUE
10572rb_str_split_m(int argc, VALUE *argv, VALUE str)
10573{
10574 rb_encoding *enc;
10575 VALUE spat;
10576 VALUE limit;
10577 split_type_t split_type;
10578 long beg, end, i = 0, empty_count = -1;
10579 int lim = 0;
10580 VALUE result, tmp;
10581
10582 result = rb_block_given_p() ? Qfalse : Qnil;
10583 if (rb_scan_args(argc, argv, "02", &spat, &limit) == 2) {
10584 lim = NUM2INT(limit);
10585 if (lim <= 0) limit = Qnil;
10586 else if (lim == 1) {
10587 if (RSTRING_LEN(str) == 0)
10588 return result ? rb_ary_new2(0) : str;
10589 tmp = str_duplicate(rb_cString, str);
10590 if (!result) {
10591 rb_yield(tmp);
10592 return str;
10593 }
10594 return rb_ary_new3(1, tmp);
10595 }
10596 i = 1;
10597 }
10598 if (NIL_P(limit) && !lim) empty_count = 0;
10599
10600 enc = STR_ENC_GET(str);
10601 split_type = SPLIT_TYPE_REGEXP;
10602 if (!NIL_P(spat)) {
10603 spat = get_pat_quoted(spat, 0);
10604 }
10605 else if (NIL_P(spat = rb_fs)) {
10606 split_type = SPLIT_TYPE_AWK;
10607 }
10608 else if (!(spat = rb_fs_check(spat))) {
10609 rb_raise(rb_eTypeError, "value of $; must be String or Regexp");
10610 }
10611 else {
10612 rb_category_warn(RB_WARN_CATEGORY_DEPRECATED, "$; is set to non-nil value");
10613 }
10614 if (split_type != SPLIT_TYPE_AWK) {
10615 switch (BUILTIN_TYPE(spat)) {
10616 case T_REGEXP:
10617 rb_reg_options(spat); /* check if uninitialized */
10618 tmp = RREGEXP_SRC(spat);
10619 split_type = literal_split_pattern(tmp, SPLIT_TYPE_REGEXP);
10620 if (split_type == SPLIT_TYPE_AWK) {
10621 spat = tmp;
10622 split_type = SPLIT_TYPE_STRING;
10623 }
10624 break;
10625
10626 case T_STRING:
10627 mustnot_broken(spat);
10628 split_type = literal_split_pattern(spat, SPLIT_TYPE_STRING);
10629 break;
10630
10631 default:
10633 }
10634 }
10635
10636#define SPLIT_STR(beg, len) ( \
10637 empty_count = split_string(result, str, beg, len, empty_count), \
10638 str_mod_check(str, str_start, str_len))
10639
10640 beg = 0;
10641 const char *ptr = RSTRING_PTR(str);
10642 const char *const str_start = ptr;
10643 const long str_len = RSTRING_LEN(str);
10644 const char *const eptr = str_start + str_len;
10645 if (split_type == SPLIT_TYPE_AWK) {
10646 const char *bptr = ptr;
10647 int skip = 1;
10648 unsigned int c;
10649
10650 if (result) result = rb_ary_new();
10651 end = beg;
10652 if (is_ascii_string(str)) {
10653 while (ptr < eptr) {
10654 c = (unsigned char)*ptr++;
10655 if (skip) {
10656 if (ascii_isspace(c)) {
10657 beg = ptr - bptr;
10658 }
10659 else {
10660 end = ptr - bptr;
10661 skip = 0;
10662 if (!NIL_P(limit) && lim <= i) break;
10663 }
10664 }
10665 else if (ascii_isspace(c)) {
10666 SPLIT_STR(beg, end-beg);
10667 skip = 1;
10668 beg = ptr - bptr;
10669 if (!NIL_P(limit)) ++i;
10670 }
10671 else {
10672 end = ptr - bptr;
10673 }
10674 }
10675 }
10676 else {
10677 while (ptr < eptr) {
10678 int n;
10679
10680 c = rb_enc_codepoint_len(ptr, eptr, &n, enc);
10681 ptr += n;
10682 if (skip) {
10683 if (rb_isspace(c)) {
10684 beg = ptr - bptr;
10685 }
10686 else {
10687 end = ptr - bptr;
10688 skip = 0;
10689 if (!NIL_P(limit) && lim <= i) break;
10690 }
10691 }
10692 else if (rb_isspace(c)) {
10693 SPLIT_STR(beg, end-beg);
10694 skip = 1;
10695 beg = ptr - bptr;
10696 if (!NIL_P(limit)) ++i;
10697 }
10698 else {
10699 end = ptr - bptr;
10700 }
10701 }
10702 }
10703 }
10704 else if (split_type == SPLIT_TYPE_STRING) {
10705 const char *substr_start = ptr;
10706 const char *sptr = RSTRING_PTR(spat);
10707 long slen = RSTRING_LEN(spat);
10708
10709 if (result) result = rb_ary_new();
10710 mustnot_broken(str);
10711 enc = rb_enc_check(str, spat);
10712 while (ptr < eptr &&
10713 (end = rb_memsearch(sptr, slen, ptr, eptr - ptr, enc)) >= 0) {
10714 /* Check we are at the start of a char */
10715 const char *t = rb_enc_right_char_head(ptr, ptr + end, eptr, enc);
10716 if (t != ptr + end) {
10717 ptr = t;
10718 continue;
10719 }
10720 SPLIT_STR(substr_start - str_start, (ptr+end) - substr_start);
10721 str_mod_check(spat, sptr, slen);
10722 ptr += end + slen;
10723 substr_start = ptr;
10724 if (!NIL_P(limit) && lim <= ++i) break;
10725 }
10726 beg = ptr - str_start;
10727 }
10728 else if (split_type == SPLIT_TYPE_CHARS) {
10729 int n;
10730
10731 if (result) result = rb_ary_new_capa(RSTRING_LEN(str));
10732 mustnot_broken(str);
10733 enc = rb_enc_get(str);
10734 while (ptr < eptr &&
10735 (n = rb_enc_precise_mbclen(ptr, eptr, enc)) > 0) {
10736 SPLIT_STR(ptr - str_start, n);
10737 ptr += n;
10738 if (!NIL_P(limit) && lim <= ++i) break;
10739 }
10740 beg = ptr - str_start;
10741 }
10742 else {
10743 if (result) result = rb_ary_new();
10744 long len = RSTRING_LEN(str);
10745 long start = beg;
10746 int idx;
10747 int last_null = 0;
10748 VALUE match = 0;
10749
10750 for (; rb_reg_search(spat, str, start, 0) >= 0;
10751 (match ? (rb_match_unbusy(match), rb_backref_set(match)) : (void)0)) {
10752 match = rb_backref_get();
10753 if (!result) rb_match_busy(match);
10754 end = RMATCH_BEG(match, 0);
10755 if (start == end && RMATCH_BEG(match, 0) == RMATCH_END(match, 0)) {
10756 if (!ptr) {
10757 SPLIT_STR(0, 0);
10758 break;
10759 }
10760 else if (last_null == 1) {
10761 SPLIT_STR(beg, rb_enc_fast_mbclen(ptr+beg, eptr, enc));
10762 beg = start;
10763 }
10764 else {
10765 if (start == len)
10766 start++;
10767 else
10768 start += rb_enc_fast_mbclen(ptr+start,eptr,enc);
10769 last_null = 1;
10770 continue;
10771 }
10772 }
10773 else {
10774 SPLIT_STR(beg, end-beg);
10775 beg = start = RMATCH_END(match, 0);
10776 }
10777 last_null = 0;
10778
10779 for (idx = 1; idx < RMATCH_NREGS(match); idx++) {
10780 if (RMATCH_BEG(match, idx) == -1) continue;
10781 SPLIT_STR(RMATCH_BEG(match, idx), RMATCH_END(match, idx) - RMATCH_BEG(match, idx));
10782 }
10783 if (!NIL_P(limit) && lim <= ++i) break;
10784 }
10785 if (match) rb_match_unbusy(match);
10786 }
10787 if (RSTRING_LEN(str) > 0 && (!NIL_P(limit) || RSTRING_LEN(str) > beg || lim < 0)) {
10788 SPLIT_STR(beg, RSTRING_LEN(str)-beg);
10789 }
10790
10791 return result ? result : str;
10792}
10793
10794VALUE
10795rb_str_split(VALUE str, const char *sep0)
10796{
10797 VALUE sep;
10798
10799 StringValue(str);
10800 sep = rb_str_new_cstr(sep0);
10801 return rb_str_split_m(1, &sep, str);
10802}
10803
10804#define WANTARRAY(m, size) (!rb_block_given_p() ? rb_ary_new_capa(size) : 0)
10805
10806static inline int
10807enumerator_element(VALUE ary, VALUE e)
10808{
10809 if (ary) {
10810 rb_ary_push(ary, e);
10811 return 0;
10812 }
10813 else {
10814 rb_yield(e);
10815 return 1;
10816 }
10817}
10818
10819#define ENUM_ELEM(ary, e) enumerator_element(ary, e)
10820
10821static const char *
10822chomp_newline(const char *p, const char *e, rb_encoding *enc)
10823{
10824 const char *prev = rb_enc_prev_char(p, e, e, enc);
10825 if (rb_enc_is_newline(prev, e, enc)) {
10826 e = prev;
10827 prev = rb_enc_prev_char(p, e, e, enc);
10828 if (prev && rb_enc_ascget(prev, e, NULL, enc) == '\r')
10829 e = prev;
10830 }
10831 return e;
10832}
10833
10834static VALUE
10835get_rs(void)
10836{
10837 VALUE rs = rb_rs;
10838 if (!NIL_P(rs) &&
10839 (!RB_TYPE_P(rs, T_STRING) ||
10840 RSTRING_LEN(rs) != 1 ||
10841 RSTRING_PTR(rs)[0] != '\n')) {
10842 rb_category_warn(RB_WARN_CATEGORY_DEPRECATED, "$/ is set to non-default value");
10843 }
10844 return rs;
10845}
10846
10847#define rb_rs get_rs()
10848
10849static VALUE
10850rb_str_enumerate_lines(int argc, VALUE *argv, VALUE str, VALUE ary)
10851{
10852 rb_encoding *enc;
10853 VALUE line, rs, orig = str, opts = Qnil, chomp = Qfalse;
10854 const char *pend, *subptr, *subend, *rsptr, *hit, *adjusted;
10855 long pos, rslen;
10856 int rsnewline = 0;
10857
10858 if (rb_scan_args(argc, argv, "01:", &rs, &opts) == 0)
10859 rs = rb_rs;
10860 if (!NIL_P(opts)) {
10861 static ID keywords[1];
10862 if (!keywords[0]) {
10863 keywords[0] = rb_intern_const("chomp");
10864 }
10865 rb_get_kwargs(opts, keywords, 0, 1, &chomp);
10866 chomp = (!UNDEF_P(chomp) && RTEST(chomp));
10867 }
10868
10869 if (NIL_P(rs)) {
10870 if (!ENUM_ELEM(ary, str)) {
10871 return ary;
10872 }
10873 else {
10874 return orig;
10875 }
10876 }
10877
10878 if (!RSTRING_LEN(str)) goto end;
10879 str = rb_str_new_frozen(str);
10880 const char *const ptr = subptr = RSTRING_PTR(str);
10881 const long len = RSTRING_LEN(str);
10882 pend = RSTRING_END(str);
10883 StringValue(rs);
10884 rslen = RSTRING_LEN(rs);
10885
10886 if (rs == rb_default_rs)
10887 enc = rb_enc_get(str);
10888 else
10889 enc = rb_enc_check(str, rs);
10890
10891 if (rslen == 0) {
10892 /* paragraph mode */
10893 int n;
10894 const char *eol = NULL;
10895 subend = subptr;
10896 while (subend < pend) {
10897 long chomp_rslen = 0;
10898 do {
10899 if (rb_enc_ascget(subend, pend, &n, enc) != '\r')
10900 n = 0;
10901 rslen = n + rb_enc_mbclen(subend + n, pend, enc);
10902 if (rb_enc_is_newline(subend + n, pend, enc)) {
10903 if (eol == subend) break;
10904 subend += rslen;
10905 if (subptr) {
10906 eol = subend;
10907 chomp_rslen = -rslen;
10908 }
10909 }
10910 else {
10911 if (!subptr) subptr = subend;
10912 subend += rslen;
10913 }
10914 rslen = 0;
10915 } while (subend < pend);
10916 if (!subptr) break;
10917 if (rslen == 0) chomp_rslen = 0;
10918 line = rb_str_subseq(str, subptr - ptr,
10919 subend - subptr + (chomp ? chomp_rslen : rslen));
10920 if (ENUM_ELEM(ary, line)) {
10921 str_mod_check(str, ptr, len);
10922 }
10923 subptr = eol = NULL;
10924 }
10925 goto end;
10926 }
10927 else {
10928 rsptr = RSTRING_PTR(rs);
10929 if (RSTRING_LEN(rs) == rb_enc_mbminlen(enc) &&
10930 rb_enc_is_newline(rsptr, rsptr + RSTRING_LEN(rs), enc)) {
10931 rsnewline = 1;
10932 }
10933 }
10934
10935 if ((rs == rb_default_rs) && !rb_enc_asciicompat(enc)) {
10936 rs = rb_str_new(rsptr, rslen);
10937 rs = rb_str_encode(rs, rb_enc_from_encoding(enc), 0, Qnil);
10938 rsptr = RSTRING_PTR(rs);
10939 rslen = RSTRING_LEN(rs);
10940 }
10941
10942 while (subptr < pend) {
10943 pos = rb_memsearch(rsptr, rslen, subptr, pend - subptr, enc);
10944 if (pos < 0) break;
10945 hit = subptr + pos;
10946 adjusted = rb_enc_right_char_head(subptr, hit, pend, enc);
10947 if (hit != adjusted) {
10948 subptr = adjusted;
10949 continue;
10950 }
10951 subend = hit += rslen;
10952 if (chomp) {
10953 if (rsnewline) {
10954 subend = chomp_newline(subptr, subend, enc);
10955 }
10956 else {
10957 subend -= rslen;
10958 }
10959 }
10960 line = rb_str_subseq(str, subptr - ptr, subend - subptr);
10961 if (ENUM_ELEM(ary, line)) {
10962 str_mod_check(str, ptr, len);
10963 str_mod_check(rs, rsptr, rslen);
10964 }
10965 subptr = hit;
10966 }
10967
10968 if (subptr < pend) {
10969 if (chomp) {
10970 if (rsnewline) {
10971 pend = chomp_newline(subptr, pend, enc);
10972 }
10973 else if (pend - subptr >= rslen &&
10974 memcmp(pend - rslen, rsptr, rslen) == 0) {
10975 pend -= rslen;
10976 }
10977 }
10978 line = rb_str_subseq(str, subptr - ptr, pend - subptr);
10979 ENUM_ELEM(ary, line);
10980 RB_GC_GUARD(str);
10981 }
10982
10983 end:
10984 if (ary)
10985 return ary;
10986 else
10987 return orig;
10988}
10989
10990/*
10991 * call-seq:
10992 * each_line(record_separator = $/, chomp: false) {|substring| ... } -> self
10993 * each_line(record_separator = $/, chomp: false) -> enumerator
10994 *
10995 * :include: doc/string/each_line.rdoc
10996 *
10997 */
10998
10999static VALUE
11000rb_str_each_line(int argc, VALUE *argv, VALUE str)
11001{
11002 RETURN_SIZED_ENUMERATOR(str, argc, argv, 0);
11003 return rb_str_enumerate_lines(argc, argv, str, 0);
11004}
11005
11006/*
11007 * call-seq:
11008 * lines(record_separator = $/, chomp: false) -> array_of_strings
11009 *
11010 * Returns substrings ("lines") of +self+
11011 * according to the given arguments:
11012 *
11013 * s = <<~EOT
11014 * This is the first line.
11015 * This is line two.
11016 *
11017 * This is line four.
11018 * This is line five.
11019 * EOT
11020 *
11021 * With the default argument values:
11022 *
11023 * $/ # => "\n"
11024 * s.lines
11025 * # =>
11026 * ["This is the first line.\n",
11027 * "This is line two.\n",
11028 * "\n",
11029 * "This is line four.\n",
11030 * "This is line five.\n"]
11031 *
11032 * With a different +record_separator+:
11033 *
11034 * record_separator = ' is '
11035 * s.lines(record_separator)
11036 * # =>
11037 * ["This is ",
11038 * "the first line.\nThis is ",
11039 * "line two.\n\nThis is ",
11040 * "line four.\nThis is ",
11041 * "line five.\n"]
11042 *
11043 * With keyword argument +chomp+ as +true+,
11044 * removes the trailing newline from each line:
11045 *
11046 * s.lines(chomp: true)
11047 * # =>
11048 * ["This is the first line.",
11049 * "This is line two.",
11050 * "",
11051 * "This is line four.",
11052 * "This is line five."]
11053 *
11054 * Related: see {Converting to Non-String}[rdoc-ref:String@Converting+to+Non-String].
11055 */
11056
11057static VALUE
11058rb_str_lines(int argc, VALUE *argv, VALUE str)
11059{
11060 VALUE ary = WANTARRAY("lines", 0);
11061 return rb_str_enumerate_lines(argc, argv, str, ary);
11062}
11063
11064static VALUE
11065rb_str_each_byte_size(VALUE str, VALUE args, VALUE eobj)
11066{
11067 return LONG2FIX(RSTRING_LEN(str));
11068}
11069
11070static VALUE
11071rb_str_enumerate_bytes(VALUE str, VALUE ary)
11072{
11073 long i;
11074
11075 for (i=0; i<RSTRING_LEN(str); i++) {
11076 ENUM_ELEM(ary, INT2FIX((unsigned char)RSTRING_PTR(str)[i]));
11077 }
11078 if (ary)
11079 return ary;
11080 else
11081 return str;
11082}
11083
11084/*
11085 * call-seq:
11086 * each_byte {|byte| ... } -> self
11087 * each_byte -> enumerator
11088 *
11089 * :include: doc/string/each_byte.rdoc
11090 *
11091 */
11092
11093static VALUE
11094rb_str_each_byte(VALUE str)
11095{
11096 RETURN_SIZED_ENUMERATOR(str, 0, 0, rb_str_each_byte_size);
11097 return rb_str_enumerate_bytes(str, 0);
11098}
11099
11100/*
11101 * call-seq:
11102 * bytes -> array_of_bytes
11103 *
11104 * :include: doc/string/bytes.rdoc
11105 *
11106 */
11107
11108static VALUE
11109rb_str_bytes(VALUE str)
11110{
11111 VALUE ary = WANTARRAY("bytes", RSTRING_LEN(str));
11112 return rb_str_enumerate_bytes(str, ary);
11113}
11114
11115static VALUE
11116rb_str_each_char_size(VALUE str, VALUE args, VALUE eobj)
11117{
11118 return rb_str_length(str);
11119}
11120
11121static VALUE
11122rb_str_enumerate_chars(VALUE str, VALUE ary)
11123{
11124 VALUE orig = str;
11125 long i, len, n;
11126 const char *ptr;
11127 rb_encoding *enc;
11128
11129 str = rb_str_new_frozen(str);
11130 ptr = RSTRING_PTR(str);
11131 len = RSTRING_LEN(str);
11132 enc = rb_enc_get(str);
11133
11135 for (i = 0; i < len; i += n) {
11136 n = rb_enc_fast_mbclen(ptr + i, ptr + len, enc);
11137 ENUM_ELEM(ary, rb_str_subseq(str, i, n));
11138 }
11139 }
11140 else {
11141 for (i = 0; i < len; i += n) {
11142 n = rb_enc_mbclen(ptr + i, ptr + len, enc);
11143 ENUM_ELEM(ary, rb_str_subseq(str, i, n));
11144 }
11145 }
11146 RB_GC_GUARD(str);
11147 if (ary)
11148 return ary;
11149 else
11150 return orig;
11151}
11152
11153/*
11154 * call-seq:
11155 * each_char {|char| ... } -> self
11156 * each_char -> enumerator
11157 *
11158 * :include: doc/string/each_char.rdoc
11159 *
11160 */
11161
11162static VALUE
11163rb_str_each_char(VALUE str)
11164{
11165 RETURN_SIZED_ENUMERATOR(str, 0, 0, rb_str_each_char_size);
11166 return rb_str_enumerate_chars(str, 0);
11167}
11168
11169/*
11170 * call-seq:
11171 * chars -> array_of_characters
11172 *
11173 * :include: doc/string/chars.rdoc
11174 *
11175 */
11176
11177static VALUE
11178rb_str_chars(VALUE str)
11179{
11180 VALUE ary = WANTARRAY("chars", rb_str_strlen(str));
11181 return rb_str_enumerate_chars(str, ary);
11182}
11183
11184static VALUE
11185rb_str_enumerate_codepoints(VALUE str, VALUE ary)
11186{
11187 VALUE orig = str;
11188 int n;
11189 unsigned int c;
11190 const char *ptr, *end;
11191 rb_encoding *enc;
11192 int enc_asciicompat;
11193
11194 if (single_byte_optimizable(str))
11195 return rb_str_enumerate_bytes(str, ary);
11196
11197 str = rb_str_new_frozen(str);
11198 ptr = RSTRING_PTR(str);
11199 end = RSTRING_END(str);
11200 enc = STR_ENC_GET(str);
11201 enc_asciicompat = rb_enc_asciicompat(enc);
11202
11203 while (ptr < end) {
11204 /* Fast path: ASCII byte in an ASCII-compatible encoding is its own codepoint;
11205 * skip rb_enc_codepoint_len and return the byte directly.
11206 */
11207 n = 1;
11208 c = (enc_asciicompat && ISASCII(*ptr)) ?
11209 (unsigned char)*ptr : rb_enc_codepoint_len(ptr, end, &n, enc);
11210 ENUM_ELEM(ary, UINT2NUM(c));
11211 ptr += n;
11212 }
11213 RB_GC_GUARD(str);
11214 if (ary)
11215 return ary;
11216 else
11217 return orig;
11218}
11219
11220/*
11221 * call-seq:
11222 * each_codepoint {|codepoint| ... } -> self
11223 * each_codepoint -> enumerator
11224 *
11225 * :include: doc/string/each_codepoint.rdoc
11226 *
11227 */
11228
11229static VALUE
11230rb_str_each_codepoint(VALUE str)
11231{
11232 RETURN_SIZED_ENUMERATOR(str, 0, 0, rb_str_each_char_size);
11233 return rb_str_enumerate_codepoints(str, 0);
11234}
11235
11236/*
11237 * call-seq:
11238 * codepoints -> array_of_integers
11239 *
11240 * :include: doc/string/codepoints.rdoc
11241 *
11242 */
11243
11244static VALUE
11245rb_str_codepoints(VALUE str)
11246{
11247 VALUE ary = WANTARRAY("codepoints", rb_str_strlen(str));
11248 return rb_str_enumerate_codepoints(str, ary);
11249}
11250
11251static regex_t *
11252get_reg_grapheme_cluster(rb_encoding *enc)
11253{
11254 int encidx = rb_enc_to_index(enc);
11255
11256 const OnigUChar source_ascii[] = "\\X";
11257 const OnigUChar *source = source_ascii;
11258 size_t source_len = sizeof(source_ascii) - 1;
11259
11260 switch (encidx) {
11261#define CHARS_16BE(x) (OnigUChar)((x)>>8), (OnigUChar)(x)
11262#define CHARS_16LE(x) (OnigUChar)(x), (OnigUChar)((x)>>8)
11263#define CHARS_32BE(x) CHARS_16BE((x)>>16), CHARS_16BE(x)
11264#define CHARS_32LE(x) CHARS_16LE(x), CHARS_16LE((x)>>16)
11265#define CASE_UTF(e) \
11266 case ENCINDEX_UTF_##e: { \
11267 static const OnigUChar source_UTF_##e[] = {CHARS_##e('\\'), CHARS_##e('X')}; \
11268 source = source_UTF_##e; \
11269 source_len = sizeof(source_UTF_##e); \
11270 break; \
11271 }
11272 CASE_UTF(16BE); CASE_UTF(16LE); CASE_UTF(32BE); CASE_UTF(32LE);
11273#undef CASE_UTF
11274#undef CHARS_16BE
11275#undef CHARS_16LE
11276#undef CHARS_32BE
11277#undef CHARS_32LE
11278 }
11279
11280 regex_t *reg_grapheme_cluster;
11281 OnigErrorInfo einfo;
11282 int r = onig_new(&reg_grapheme_cluster, source, source + source_len,
11283 ONIG_OPTION_DEFAULT, enc, OnigDefaultSyntax, &einfo);
11284 if (r) {
11285 UChar message[ONIG_MAX_ERROR_MESSAGE_LEN];
11286 onig_error_code_to_str(message, r, &einfo);
11287 rb_fatal("cannot compile grapheme cluster regexp: %s", (char *)message);
11288 }
11289
11290 return reg_grapheme_cluster;
11291}
11292
11293static regex_t *
11294get_cached_reg_grapheme_cluster(rb_encoding *enc)
11295{
11296 int encidx = rb_enc_to_index(enc);
11297 static regex_t *reg_grapheme_cluster_utf8 = NULL;
11298
11299 if (encidx == rb_utf8_encindex()) {
11300 if (!reg_grapheme_cluster_utf8) {
11301 reg_grapheme_cluster_utf8 = get_reg_grapheme_cluster(enc);
11302 }
11303
11304 return reg_grapheme_cluster_utf8;
11305 }
11306
11307 return NULL;
11308}
11309
11310static VALUE
11311rb_str_each_grapheme_cluster_size(VALUE str, VALUE args, VALUE eobj)
11312{
11313 size_t grapheme_cluster_count = 0;
11314 rb_encoding *enc = get_encoding(str);
11315 const char *ptr, *end;
11316
11317 if (!rb_enc_unicode_p(enc)) {
11318 return rb_str_length(str);
11319 }
11320
11321 bool cached_reg_grapheme_cluster = true;
11322 regex_t *reg_grapheme_cluster = get_cached_reg_grapheme_cluster(enc);
11323 if (!reg_grapheme_cluster) {
11324 reg_grapheme_cluster = get_reg_grapheme_cluster(enc);
11325 cached_reg_grapheme_cluster = false;
11326 }
11327
11328 ptr = RSTRING_PTR(str);
11329 end = RSTRING_END(str);
11330
11331 while (ptr < end) {
11332 OnigPosition len = onig_match(reg_grapheme_cluster,
11333 (const OnigUChar *)ptr, (const OnigUChar *)end,
11334 (const OnigUChar *)ptr, NULL, 0);
11335 if (len <= 0) break;
11336 grapheme_cluster_count++;
11337 ptr += len;
11338 }
11339
11340 if (!cached_reg_grapheme_cluster) {
11341 onig_free(reg_grapheme_cluster);
11342 }
11343
11344 return SIZET2NUM(grapheme_cluster_count);
11345}
11346
11347static VALUE
11348rb_str_enumerate_grapheme_clusters(VALUE str, VALUE ary)
11349{
11350 VALUE orig = str;
11351 rb_encoding *enc = get_encoding(str);
11352 const char *ptr0, *ptr, *end;
11353
11354 if (!rb_enc_unicode_p(enc)) {
11355 return rb_str_enumerate_chars(str, ary);
11356 }
11357
11358 if (!ary) str = rb_str_new_frozen(str);
11359
11360 bool cached_reg_grapheme_cluster = true;
11361 regex_t *reg_grapheme_cluster = get_cached_reg_grapheme_cluster(enc);
11362 if (!reg_grapheme_cluster) {
11363 reg_grapheme_cluster = get_reg_grapheme_cluster(enc);
11364 cached_reg_grapheme_cluster = false;
11365 }
11366
11367 ptr0 = ptr = RSTRING_PTR(str);
11368 end = RSTRING_END(str);
11369
11370 while (ptr < end) {
11371 OnigPosition len = onig_match(reg_grapheme_cluster,
11372 (const OnigUChar *)ptr, (const OnigUChar *)end,
11373 (const OnigUChar *)ptr, NULL, 0);
11374 if (len <= 0) break;
11375 ENUM_ELEM(ary, rb_str_subseq(str, ptr-ptr0, len));
11376 ptr += len;
11377 }
11378
11379 if (!cached_reg_grapheme_cluster) {
11380 onig_free(reg_grapheme_cluster);
11381 }
11382
11383 RB_GC_GUARD(str);
11384 if (ary)
11385 return ary;
11386 else
11387 return orig;
11388}
11389
11390/*
11391 * call-seq:
11392 * each_grapheme_cluster {|grapheme_cluster| ... } -> self
11393 * each_grapheme_cluster -> enumerator
11394 *
11395 * :include: doc/string/each_grapheme_cluster.rdoc
11396 *
11397 */
11398
11399static VALUE
11400rb_str_each_grapheme_cluster(VALUE str)
11401{
11402 RETURN_SIZED_ENUMERATOR(str, 0, 0, rb_str_each_grapheme_cluster_size);
11403 return rb_str_enumerate_grapheme_clusters(str, 0);
11404}
11405
11406/*
11407 * call-seq:
11408 * grapheme_clusters -> array_of_grapheme_clusters
11409 *
11410 * :include: doc/string/grapheme_clusters.rdoc
11411 *
11412 */
11413
11414static VALUE
11415rb_str_grapheme_clusters(VALUE str)
11416{
11417 VALUE ary = WANTARRAY("grapheme_clusters", rb_str_strlen(str));
11418 return rb_str_enumerate_grapheme_clusters(str, ary);
11419}
11420
11421static long
11422chopped_length(VALUE str)
11423{
11424 rb_encoding *enc = STR_ENC_GET(str);
11425 const char *p, *p2, *beg, *end;
11426
11427 beg = RSTRING_PTR(str);
11428 end = beg + RSTRING_LEN(str);
11429 if (beg >= end) return 0;
11430 p = rb_enc_prev_char(beg, end, end, enc);
11431 if (!p) return 0;
11432 if (p > beg && rb_enc_ascget(p, end, 0, enc) == '\n') {
11433 p2 = rb_enc_prev_char(beg, p, end, enc);
11434 if (p2 && rb_enc_ascget(p2, end, 0, enc) == '\r') p = p2;
11435 }
11436 return p - beg;
11437}
11438
11439/*
11440 * call-seq:
11441 * chop! -> self or nil
11442 *
11443 * Like String#chop, except that:
11444 *
11445 * - Removes trailing characters from +self+ (not from a copy of +self+).
11446 * - Returns +self+ if any characters are removed, +nil+ otherwise.
11447 *
11448 * Related: see {Modifying}[rdoc-ref:String@Modifying].
11449 */
11450
11451static VALUE
11452rb_str_chop_bang(VALUE str)
11453{
11454 str_modify_keep_cr(str);
11455 if (RSTRING_LEN(str) > 0) {
11456 long len;
11457 len = chopped_length(str);
11458 STR_SET_LEN(str, len);
11459 TERM_FILL(&RSTRING_PTR(str)[len], TERM_LEN(str));
11460 if (ENC_CODERANGE(str) != ENC_CODERANGE_7BIT) {
11462 }
11463 return str;
11464 }
11465 return Qnil;
11466}
11467
11468
11469/*
11470 * call-seq:
11471 * chop -> new_string
11472 *
11473 * :include: doc/string/chop.rdoc
11474 *
11475 */
11476
11477static VALUE
11478rb_str_chop(VALUE str)
11479{
11480 return rb_str_subseq(str, 0, chopped_length(str));
11481}
11482
11483static long
11484smart_chomp(VALUE str, const char *e, const char *p)
11485{
11486 rb_encoding *enc = rb_enc_get(str);
11487 if (rb_enc_mbminlen(enc) > 1) {
11488 /* a receiver shorter than one character has nothing to chomp */
11489 if (e - p < rb_enc_mbminlen(enc)) return e - p;
11490 const char *pp = rb_enc_left_char_head(p, e-rb_enc_mbminlen(enc), e, enc);
11491 if (rb_enc_is_newline(pp, e, enc)) {
11492 e = pp;
11493 }
11494 pp = e - rb_enc_mbminlen(enc);
11495 if (pp >= p) {
11496 pp = rb_enc_left_char_head(p, pp, e, enc);
11497 if (rb_enc_ascget(pp, e, 0, enc) == '\r') {
11498 e = pp;
11499 }
11500 }
11501 }
11502 else {
11503 switch (*(e-1)) { /* not e[-1] to get rid of VC bug */
11504 case '\n':
11505 if (--e > p && *(e-1) == '\r') {
11506 --e;
11507 }
11508 break;
11509 case '\r':
11510 --e;
11511 break;
11512 }
11513 }
11514 return e - p;
11515}
11516
11517static long
11518chompped_length(VALUE str, VALUE rs)
11519{
11520 rb_encoding *enc;
11521 int newline;
11522 const char *pp, *e, *rsptr;
11523 long rslen;
11524 const char *const p = RSTRING_PTR(str);
11525 long len = RSTRING_LEN(str);
11526
11527 if (len == 0) return 0;
11528 e = p + len;
11529 if (rs == rb_default_rs) {
11530 return smart_chomp(str, e, p);
11531 }
11532
11533 enc = rb_enc_get(str);
11534 RSTRING_GETMEM(rs, rsptr, rslen);
11535 if (rslen == 0) {
11536 if (rb_enc_mbminlen(enc) > 1) {
11537 while (e - p >= rb_enc_mbminlen(enc)) {
11538 pp = rb_enc_left_char_head(p, e-rb_enc_mbminlen(enc), e, enc);
11539 if (!rb_enc_is_newline(pp, e, enc)) break;
11540 e = pp;
11541 pp -= rb_enc_mbminlen(enc);
11542 if (pp >= p) {
11543 pp = rb_enc_left_char_head(p, pp, e, enc);
11544 if (rb_enc_ascget(pp, e, 0, enc) == '\r') {
11545 e = pp;
11546 }
11547 }
11548 }
11549 }
11550 else {
11551 while (e > p && *(e-1) == '\n') {
11552 --e;
11553 if (e > p && *(e-1) == '\r')
11554 --e;
11555 }
11556 }
11557 return e - p;
11558 }
11559 if (rslen > len) return len;
11560
11561 enc = rb_enc_get(rs);
11562 newline = rsptr[rslen-1];
11563 if (rslen == rb_enc_mbminlen(enc)) {
11564 if (rslen == 1) {
11565 if (newline == '\n')
11566 return smart_chomp(str, e, p);
11567 }
11568 else {
11569 if (rb_enc_is_newline(rsptr, rsptr+rslen, enc))
11570 return smart_chomp(str, e, p);
11571 }
11572 }
11573
11574 enc = rb_enc_check(str, rs);
11575 if (is_broken_string(rs)) {
11576 return len;
11577 }
11578 pp = e - rslen;
11579 if (p[len-1] == newline &&
11580 (rslen <= 1 ||
11581 memcmp(rsptr, pp, rslen) == 0)) {
11582 if (at_char_boundary(p, pp, e, enc))
11583 return len - rslen;
11584 RB_GC_GUARD(rs);
11585 }
11586 return len;
11587}
11588
11594static VALUE
11595chomp_rs(int argc, const VALUE *argv)
11596{
11597 rb_check_arity(argc, 0, 1);
11598 if (argc > 0) {
11599 VALUE rs = argv[0];
11600 if (!NIL_P(rs)) StringValue(rs);
11601 return rs;
11602 }
11603 else {
11604 return rb_rs;
11605 }
11606}
11607
11608static VALUE
11609str_shrink(VALUE str, long len)
11610{
11611 str_modify_keep_cr(str);
11612 STR_SET_LEN(str, len);
11613 TERM_FILL(&RSTRING_PTR(str)[len], TERM_LEN(str));
11614 if (ENC_CODERANGE(str) != ENC_CODERANGE_7BIT) {
11616 }
11617 return str;
11618}
11619
11620VALUE
11621rb_str_chomp_string(VALUE str, VALUE rs)
11622{
11623 long olen = RSTRING_LEN(str);
11624 long len = chompped_length(str, rs);
11625 if (len >= olen) return Qnil;
11626 return str_shrink(str, len);
11627}
11628
11629/*
11630 * call-seq:
11631 * chomp!(line_sep = $/) -> self or nil
11632 *
11633 * Like String#chomp, except that:
11634 *
11635 * - Removes trailing characters from +self+ (not from a copy of +self+).
11636 * - Returns +self+ if any characters are removed, +nil+ otherwise.
11637 *
11638 * Related: see {Modifying}[rdoc-ref:String@Modifying].
11639 */
11640
11641static VALUE
11642rb_str_chomp_bang(int argc, VALUE *argv, VALUE str)
11643{
11644 VALUE rs;
11645 str_modifiable(str);
11646 if (RSTRING_LEN(str) == 0 && argc < 2) return Qnil;
11647 rs = chomp_rs(argc, argv);
11648 if (NIL_P(rs)) return Qnil;
11649 return rb_str_chomp_string(str, rs);
11650}
11651
11652
11653/*
11654 * call-seq:
11655 * chomp(line_sep = $/) -> new_string
11656 *
11657 * :include: doc/string/chomp.rdoc
11658 *
11659 */
11660
11661static VALUE
11662rb_str_chomp(int argc, VALUE *argv, VALUE str)
11663{
11664 VALUE rs = chomp_rs(argc, argv);
11665 if (NIL_P(rs)) return str_duplicate(rb_cString, str);
11666 return rb_str_subseq(str, 0, chompped_length(str, rs));
11667}
11668
11669static void
11670tr_setup_table_multi(char table[TR_TABLE_SIZE], VALUE *tablep, VALUE *ctablep,
11671 VALUE str, int num_selectors, VALUE *selectors)
11672{
11673 int i;
11674
11675 for (i=0; i<num_selectors; i++) {
11676 VALUE selector = selectors[i];
11677 rb_encoding *enc;
11678
11679 StringValue(selector);
11680 enc = rb_enc_check(str, selector);
11681 tr_setup_table(selector, table, i==0, tablep, ctablep, enc);
11682 }
11683}
11684
11685static long
11686lstrip_offset(VALUE str, const char *s, const char *e, rb_encoding *enc)
11687{
11688 const char *const start = s;
11689
11690 if (!s || s >= e) return 0;
11691
11692 /* remove spaces at head */
11693 if (single_byte_optimizable(str)) {
11694 while (s < e && (*s == '\0' || ascii_isspace(*s))) s++;
11695 }
11696 else {
11697 while (s < e) {
11698 int n;
11699 unsigned int cc = rb_enc_codepoint_len(s, e, &n, enc);
11700
11701 if (cc && !rb_isspace(cc)) break;
11702 s += n;
11703 }
11704 }
11705 return s - start;
11706}
11707
11708static long
11709lstrip_offset_table(VALUE str, const char *s, const char *e, rb_encoding *enc,
11710 char table[TR_TABLE_SIZE], VALUE del, VALUE nodel)
11711{
11712 const char *const start = s;
11713
11714 if (!s || s >= e) return 0;
11715
11716 /* remove leading characters in the table */
11717 while (s < e) {
11718 int n;
11719 unsigned int cc = rb_enc_codepoint_len(s, e, &n, enc);
11720
11721 if (!tr_find(cc, table, del, nodel)) break;
11722 s += n;
11723 }
11724 return s - start;
11725}
11726
11727/*
11728 * call-seq:
11729 * lstrip!(*selectors) -> self or nil
11730 *
11731 * Like String#lstrip, except that:
11732 *
11733 * - Performs stripping in +self+ (not in a copy of +self+).
11734 * - Returns +self+ if any characters are stripped, +nil+ otherwise.
11735 *
11736 * Related: see {Modifying}[rdoc-ref:String@Modifying].
11737 */
11738
11739static VALUE
11740rb_str_lstrip_bang(int argc, VALUE *argv, VALUE str)
11741{
11742 rb_encoding *enc;
11743 char *start;
11744 long olen, loffset;
11745
11746 str_modify_keep_cr(str);
11747 enc = STR_ENC_GET(str);
11748 RSTRING_GETMEM(str, start, olen);
11749 if (argc > 0) {
11750 char table[TR_TABLE_SIZE];
11751 VALUE del = 0, nodel = 0;
11752
11753 tr_setup_table_multi(table, &del, &nodel, str, argc, argv);
11754
11755 /* the selector conversion may have modified str */
11756 str_modify_keep_cr(str);
11757 enc = STR_ENC_GET(str);
11758 RSTRING_GETMEM(str, start, olen);
11759
11760 loffset = lstrip_offset_table(str, start, start+olen, enc, table, del, nodel);
11761 }
11762 else {
11763 loffset = lstrip_offset(str, start, start+olen, enc);
11764 }
11765
11766 if (loffset > 0) {
11767 long len = olen-loffset;
11768 memmove(start, start + loffset, len);
11769 STR_SET_LEN(str, len);
11770 TERM_FILL(start+len, rb_enc_mbminlen(enc));
11771 return str;
11772 }
11773 return Qnil;
11774}
11775
11776
11777/*
11778 * call-seq:
11779 * lstrip(*selectors) -> new_string
11780 *
11781 * Returns a copy of +self+ with leading whitespace removed;
11782 * see {Whitespace in Strings}[rdoc-ref:String@Whitespace+in+Strings]:
11783 *
11784 * whitespace = "\x00\t\n\v\f\r "
11785 * s = whitespace + 'abc' + whitespace
11786 * # => "\u0000\t\n\v\f\r abc\u0000\t\n\v\f\r "
11787 * s.lstrip
11788 * # => "abc\u0000\t\n\v\f\r "
11789 *
11790 * If +selectors+ are given, removes characters of +selectors+ from the beginning of +self+:
11791 *
11792 * s = "---abc+++"
11793 * s.lstrip("-") # => "abc+++"
11794 *
11795 * +selectors+ must be valid character selectors (see {Character Selectors}[rdoc-ref:character_selectors.rdoc]),
11796 * and may use any of its valid forms, including negation, ranges, and escapes:
11797 *
11798 * "01234abc56789".lstrip("0-9") # "abc56789"
11799 * "01234abc56789".lstrip("0-9", "^4-6") # "4abc56789"
11800 *
11801 * Related: see {Converting to New String}[rdoc-ref:String@Converting+to+New+String].
11802 */
11803
11804static VALUE
11805rb_str_lstrip(int argc, VALUE *argv, VALUE str)
11806{
11807 const char *start;
11808 long len, loffset;
11809
11810 RSTRING_GETMEM(str, start, len);
11811 if (argc > 0) {
11812 char table[TR_TABLE_SIZE];
11813 VALUE del = 0, nodel = 0;
11814
11815 tr_setup_table_multi(table, &del, &nodel, str, argc, argv);
11816
11817 /* the selector conversion may have modified str */
11818 RSTRING_GETMEM(str, start, len);
11819
11820 loffset = lstrip_offset_table(str, start, start+len, STR_ENC_GET(str), table, del, nodel);
11821 }
11822 else {
11823 loffset = lstrip_offset(str, start, start+len, STR_ENC_GET(str));
11824 }
11825 if (loffset <= 0) return str_duplicate(rb_cString, str);
11826 return rb_str_subseq(str, loffset, len - loffset);
11827}
11828
11829static long
11830rstrip_offset(VALUE str, const char *s, const char *e, rb_encoding *enc)
11831{
11832 const char *t;
11833
11834 rb_str_check_dummy_enc(enc);
11835 if (rb_enc_str_coderange(str) == ENC_CODERANGE_BROKEN) {
11836 rb_raise(rb_eEncCompatError, "invalid byte sequence in %s", rb_enc_name(enc));
11837 }
11838 if (!s || s >= e) return 0;
11839 t = e;
11840
11841 /* remove trailing spaces or '\0's */
11842 if (single_byte_optimizable(str)) {
11843 unsigned char c;
11844 while (s < t && ((c = *(t-1)) == '\0' || ascii_isspace(c))) t--;
11845 }
11846 else {
11847 const char *tp;
11848
11849 while ((tp = rb_enc_prev_char(s, t, e, enc)) != NULL) {
11850 unsigned int c = rb_enc_codepoint(tp, e, enc);
11851 if (c && !rb_isspace(c)) break;
11852 t = tp;
11853 }
11854 }
11855 return e - t;
11856}
11857
11858static long
11859rstrip_offset_table(VALUE str, const char *s, const char *e, rb_encoding *enc,
11860 char table[TR_TABLE_SIZE], VALUE del, VALUE nodel)
11861{
11862 const char *t, *tp;
11863
11864 rb_str_check_dummy_enc(enc);
11865 if (rb_enc_str_coderange(str) == ENC_CODERANGE_BROKEN) {
11866 rb_raise(rb_eEncCompatError, "invalid byte sequence in %s", rb_enc_name(enc));
11867 }
11868 if (!s || s >= e) return 0;
11869 t = e;
11870
11871 /* remove trailing characters in the table */
11872 while ((tp = rb_enc_prev_char(s, t, e, enc)) != NULL) {
11873 unsigned int c = rb_enc_codepoint(tp, e, enc);
11874 if (!tr_find(c, table, del, nodel)) break;
11875 t = tp;
11876 }
11877
11878 return e - t;
11879}
11880
11881/*
11882 * call-seq:
11883 * rstrip!(*selectors) -> self or nil
11884 *
11885 * Like String#rstrip, except that:
11886 *
11887 * - Performs stripping in +self+ (not in a copy of +self+).
11888 * - Returns +self+ if any characters are stripped, +nil+ otherwise.
11889 *
11890 * Related: see {Modifying}[rdoc-ref:String@Modifying].
11891 */
11892
11893static VALUE
11894rb_str_rstrip_bang(int argc, VALUE *argv, VALUE str)
11895{
11896 rb_encoding *enc;
11897 char *start;
11898 long olen, roffset;
11899
11900 str_modify_keep_cr(str);
11901 enc = STR_ENC_GET(str);
11902 RSTRING_GETMEM(str, start, olen);
11903 if (argc > 0) {
11904 char table[TR_TABLE_SIZE];
11905 VALUE del = 0, nodel = 0;
11906
11907 tr_setup_table_multi(table, &del, &nodel, str, argc, argv);
11908
11909 /* the selector conversion may have modified str */
11910 str_modify_keep_cr(str);
11911 enc = STR_ENC_GET(str);
11912 RSTRING_GETMEM(str, start, olen);
11913
11914 roffset = rstrip_offset_table(str, start, start+olen, enc, table, del, nodel);
11915 }
11916 else {
11917 roffset = rstrip_offset(str, start, start+olen, enc);
11918 }
11919 if (roffset > 0) {
11920 long len = olen - roffset;
11921
11922 STR_SET_LEN(str, len);
11923 TERM_FILL(start+len, rb_enc_mbminlen(enc));
11924 return str;
11925 }
11926 return Qnil;
11927}
11928
11929
11930/*
11931 * call-seq:
11932 * rstrip(*selectors) -> new_string
11933 *
11934 * Returns a copy of +self+ with trailing whitespace removed;
11935 * see {Whitespace in Strings}[rdoc-ref:String@Whitespace+in+Strings]:
11936 *
11937 * whitespace = "\x00\t\n\v\f\r "
11938 * s = whitespace + 'abc' + whitespace
11939 * s # => "\u0000\t\n\v\f\r abc\u0000\t\n\v\f\r "
11940 * s.rstrip # => "\u0000\t\n\v\f\r abc"
11941 *
11942 * If +selectors+ are given, removes characters of +selectors+ from the end of +self+:
11943 *
11944 * s = "---abc+++"
11945 * s.rstrip("+") # => "---abc"
11946 *
11947 * +selectors+ must be valid character selectors (see {Character Selectors}[rdoc-ref:character_selectors.rdoc]),
11948 * and may use any of its valid forms, including negation, ranges, and escapes:
11949 *
11950 * "01234abc56789".rstrip("0-9") # "01234abc"
11951 * "01234abc56789".rstrip("0-9", "^4-6") # "01234abc56"
11952 *
11953 * Related: see {Converting to New String}[rdoc-ref:String@Converting+to+New+String].
11954 */
11955
11956static VALUE
11957rb_str_rstrip(int argc, VALUE *argv, VALUE str)
11958{
11959 rb_encoding *enc;
11960 const char *start;
11961 long olen, roffset;
11962
11963 enc = STR_ENC_GET(str);
11964 RSTRING_GETMEM(str, start, olen);
11965 if (argc > 0) {
11966 char table[TR_TABLE_SIZE];
11967 VALUE del = 0, nodel = 0;
11968
11969 tr_setup_table_multi(table, &del, &nodel, str, argc, argv);
11970
11971 /* the selector conversion may have modified str */
11972 enc = STR_ENC_GET(str);
11973
11974 RSTRING_GETMEM(str, start, olen);
11975 roffset = rstrip_offset_table(str, start, start+olen, enc, table, del, nodel);
11976 }
11977 else {
11978 roffset = rstrip_offset(str, start, start+olen, enc);
11979 }
11980 if (roffset <= 0) return str_duplicate(rb_cString, str);
11981 return rb_str_subseq(str, 0, olen-roffset);
11982}
11983
11984
11985/*
11986 * call-seq:
11987 * strip!(*selectors) -> self or nil
11988 *
11989 * Like String#strip, except that:
11990 *
11991 * - Any modifications are made to +self+.
11992 * - Returns +self+ if any modification are made, +nil+ otherwise.
11993 *
11994 * Related: see {Modifying}[rdoc-ref:String@Modifying].
11995 */
11996
11997static VALUE
11998rb_str_strip_bang(int argc, VALUE *argv, VALUE str)
11999{
12000 char *start;
12001 long olen, loffset, roffset;
12002 rb_encoding *enc;
12003
12004 str_modify_keep_cr(str);
12005 enc = STR_ENC_GET(str);
12006 RSTRING_GETMEM(str, start, olen);
12007
12008 if (argc > 0) {
12009 char table[TR_TABLE_SIZE];
12010 VALUE del = 0, nodel = 0;
12011
12012 tr_setup_table_multi(table, &del, &nodel, str, argc, argv);
12013
12014 /* the selector conversion may have modified str */
12015 str_modify_keep_cr(str);
12016 enc = STR_ENC_GET(str);
12017 RSTRING_GETMEM(str, start, olen);
12018
12019 loffset = lstrip_offset_table(str, start, start+olen, enc, table, del, nodel);
12020 roffset = rstrip_offset_table(str, start+loffset, start+olen, enc, table, del, nodel);
12021 }
12022 else {
12023 loffset = lstrip_offset(str, start, start+olen, enc);
12024 roffset = rstrip_offset(str, start+loffset, start+olen, enc);
12025 }
12026
12027 if (loffset > 0 || roffset > 0) {
12028 long len = olen-roffset;
12029 if (loffset > 0) {
12030 len -= loffset;
12031 memmove(start, start + loffset, len);
12032 }
12033 STR_SET_LEN(str, len);
12034 TERM_FILL(start+len, rb_enc_mbminlen(enc));
12035 return str;
12036 }
12037 return Qnil;
12038}
12039
12040
12041/*
12042 * call-seq:
12043 * strip(*selectors) -> new_string
12044 *
12045 * Returns a copy of +self+ with leading and trailing whitespace removed;
12046 * see {Whitespace in Strings}[rdoc-ref:String@Whitespace+in+Strings]:
12047 *
12048 * whitespace = "\x00\t\n\v\f\r "
12049 * s = whitespace + 'abc' + whitespace
12050 * # => "\u0000\t\n\v\f\r abc\u0000\t\n\v\f\r "
12051 * s.strip # => "abc"
12052 *
12053 * If +selectors+ are given, removes characters of +selectors+ from both ends of +self+:
12054 *
12055 * s = "---abc+++"
12056 * s.strip("-+") # => "abc"
12057 * s.strip("+-") # => "abc"
12058 *
12059 * +selectors+ must be valid character selectors (see {Character Selectors}[rdoc-ref:character_selectors.rdoc]),
12060 * and may use any of its valid forms, including negation, ranges, and escapes:
12061 *
12062 * "01234abc56789".strip("0-9") # "abc"
12063 * "01234abc56789".strip("0-9", "^4-6") # "4abc56"
12064 *
12065 * Related: see {Converting to New String}[rdoc-ref:String@Converting+to+New+String].
12066 */
12067
12068static VALUE
12069rb_str_strip(int argc, VALUE *argv, VALUE str)
12070{
12071 const char *start;
12072 long olen, loffset, roffset;
12073 rb_encoding *enc = STR_ENC_GET(str);
12074
12075 RSTRING_GETMEM(str, start, olen);
12076
12077 if (argc > 0) {
12078 char table[TR_TABLE_SIZE];
12079 VALUE del = 0, nodel = 0;
12080
12081 tr_setup_table_multi(table, &del, &nodel, str, argc, argv);
12082
12083 /* the selector conversion may have modified str */
12084 enc = STR_ENC_GET(str);
12085 RSTRING_GETMEM(str, start, olen);
12086
12087 loffset = lstrip_offset_table(str, start, start+olen, enc, table, del, nodel);
12088 roffset = rstrip_offset_table(str, start+loffset, start+olen, enc, table, del, nodel);
12089 }
12090 else {
12091 loffset = lstrip_offset(str, start, start+olen, enc);
12092 roffset = rstrip_offset(str, start+loffset, start+olen, enc);
12093 }
12094
12095 if (loffset <= 0 && roffset <= 0) return str_duplicate(rb_cString, str);
12096 return rb_str_subseq(str, loffset, olen-loffset-roffset);
12097}
12098
12099static VALUE
12100scan_once(VALUE str, VALUE pat, long *start, int set_backref_str)
12101{
12102 VALUE result = Qnil;
12103 long end, pos = rb_pat_search(pat, str, *start, set_backref_str);
12104 if (pos >= 0) {
12105 VALUE match = Qnil;
12106 if (BUILTIN_TYPE(pat) == T_STRING) {
12107 end = pos + RSTRING_LEN(pat);
12108 }
12109 else {
12110 match = rb_backref_get();
12111 pos = RMATCH_BEG(match, 0);
12112 end = RMATCH_END(match, 0);
12113 }
12114
12115 if (pos == end) {
12116 rb_encoding *enc = STR_ENC_GET(str);
12117 /*
12118 * Always consume at least one character of the input string
12119 */
12120 if (RSTRING_LEN(str) > end)
12121 *start = end + rb_enc_fast_mbclen(RSTRING_PTR(str) + end,
12122 RSTRING_END(str), enc);
12123 else
12124 *start = end + 1;
12125 }
12126 else {
12127 *start = end;
12128 }
12129
12130 if (NIL_P(match) || RMATCH_NREGS(match) == 1) {
12131 result = rb_str_subseq(str, pos, end - pos);
12132 return result;
12133 }
12134 else {
12135 int num_regs = RMATCH_NREGS(match);
12136 result = rb_ary_new2(num_regs);
12137 for (int i = 1; i < num_regs; i++) {
12138 VALUE s = Qnil;
12139 if (RMATCH_BEG(match, i) >= 0) {
12140 s = rb_str_subseq(str, RMATCH_BEG(match, i), RMATCH_END(match, i) - RMATCH_BEG(match, i));
12141 }
12142
12143 rb_ary_push(result, s);
12144 }
12145 }
12146
12147 RB_GC_GUARD(match);
12148 }
12149
12150 return result;
12151}
12152
12153
12154/*
12155 * call-seq:
12156 * scan(pattern) -> array_of_results
12157 * scan(pattern) {|result| ... } -> self
12158 *
12159 * :include: doc/string/scan.rdoc
12160 *
12161 */
12162
12163static VALUE
12164rb_str_scan(VALUE str, VALUE pat)
12165{
12166 VALUE result;
12167 long start = 0;
12168 long last = -1, prev = 0;
12169 const char *p = RSTRING_PTR(str);
12170 long len = RSTRING_LEN(str);
12171
12172 pat = get_pat_quoted(pat, 1);
12173 mustnot_broken(str);
12174 if (!rb_block_given_p()) {
12175 VALUE ary = rb_ary_new();
12176
12177 while (!NIL_P(result = scan_once(str, pat, &start, 0))) {
12178 last = prev;
12179 prev = start;
12180 rb_ary_push(ary, result);
12181 }
12182 if (last >= 0) rb_pat_search(pat, str, last, 1);
12183 else rb_backref_set(Qnil);
12184 return ary;
12185 }
12186
12187 while (!NIL_P(result = scan_once(str, pat, &start, 1))) {
12188 last = prev;
12189 prev = start;
12190 rb_yield(result);
12191 str_mod_check(str, p, len);
12192 }
12193 if (last >= 0) rb_pat_search(pat, str, last, 1);
12194 return str;
12195}
12196
12197
12198/*
12199 * call-seq:
12200 * hex -> integer
12201 *
12202 * Interprets the leading substring of +self+ as hexadecimal, possibly signed;
12203 * returns its value as an integer.
12204 *
12205 * The leading substring is interpreted as hexadecimal when it begins with:
12206 *
12207 * - One or more character representing hexadecimal digits
12208 * (each in one of the ranges <tt>'0'..'9'</tt>, <tt>'a'..'f'</tt>, or <tt>'A'..'F'</tt>);
12209 * the string to be interpreted ends at the first character that does not represent a hexadecimal digit:
12210 *
12211 * 'f'.hex # => 15
12212 * '11'.hex # => 17
12213 * 'FFF'.hex # => 4095
12214 * 'fffg'.hex # => 4095
12215 * 'foo'.hex # => 15 # 'f' hexadecimal, 'oo' not.
12216 * 'bar'.hex # => 186 # 'ba' hexadecimal, 'r' not.
12217 * 'deadbeef'.hex # => 3735928559
12218 *
12219 * - <tt>'0x'</tt> or <tt>'0X'</tt>, followed by one or more hexadecimal digits:
12220 *
12221 * '0xfff'.hex # => 4095
12222 * '0xfffg'.hex # => 4095
12223 *
12224 * Any of the above may prefixed with <tt>'-'</tt>, which negates the interpreted value:
12225 *
12226 * '-fff'.hex # => -4095
12227 * '-0xFFF'.hex # => -4095
12228 *
12229 * For any substring not described above, returns zero:
12230 *
12231 * 'xxx'.hex # => 0
12232 * ''.hex # => 0
12233 *
12234 * Note that, unlike #oct, this method interprets only hexadecimal,
12235 * and not binary, octal, or decimal notations:
12236 *
12237 * '0b111'.hex # => 45329
12238 * '0o777'.hex # => 0
12239 * '0d999'.hex # => 55705
12240 *
12241 * Related: See {Converting to Non-String}[rdoc-ref:String@Converting+to+Non-String].
12242 */
12243
12244static VALUE
12245rb_str_hex(VALUE str)
12246{
12247 return rb_str_to_inum(str, 16, FALSE);
12248}
12249
12250
12251/*
12252 * call-seq:
12253 * oct -> integer
12254 *
12255 * Interprets the leading substring of +self+ as octal, binary, decimal, or hexadecimal, possibly signed;
12256 * returns their value as an integer.
12257 *
12258 * In brief:
12259 *
12260 * # Interpreted as octal.
12261 * '777'.oct # => 511
12262 * '777x'.oct # => 511
12263 * '0777'.oct # => 511
12264 * '0o777'.oct # => 511
12265 * '-777'.oct # => -511
12266 * # Not interpreted as octal.
12267 * '0b111'.oct # => 7 # Interpreted as binary.
12268 * '0d999'.oct # => 999 # Interpreted as decimal.
12269 * '0xfff'.oct # => 4095 # Interpreted as hexadecimal.
12270 *
12271 * The leading substring is interpreted as octal when it begins with:
12272 *
12273 * - One or more character representing octal digits
12274 * (each in the range <tt>'0'..'7'</tt>);
12275 * the string to be interpreted ends at the first character that does not represent an octal digit:
12276 *
12277 * '7'.oct @ => 7
12278 * '11'.oct # => 9
12279 * '777'.oct # => 511
12280 * '0777'.oct # => 511
12281 * '7778'.oct # => 511
12282 * '777x'.oct # => 511
12283 *
12284 * - <tt>'0o'</tt>, followed by one or more octal digits:
12285 *
12286 * '0o777'.oct # => 511
12287 * '0o7778'.oct # => 511
12288 *
12289 * The leading substring is _not_ interpreted as octal when it begins with:
12290 *
12291 * - <tt>'0b'</tt>, followed by one or more characters representing binary digits
12292 * (each in the range <tt>'0'..'1'</tt>);
12293 * the string to be interpreted ends at the first character that does not represent a binary digit.
12294 * the string is interpreted as binary digits (base 2):
12295 *
12296 * '0b111'.oct # => 7
12297 * '0b1112'.oct # => 7
12298 *
12299 * - <tt>'0d'</tt>, followed by one or more characters representing decimal digits
12300 * (each in the range <tt>'0'..'9'</tt>);
12301 * the string to be interpreted ends at the first character that does not represent a decimal digit.
12302 * the string is interpreted as decimal digits (base 10):
12303 *
12304 * '0d999'.oct # => 999
12305 * '0d999x'.oct # => 999
12306 *
12307 * - <tt>'0x'</tt>, followed by one or more characters representing hexadecimal digits
12308 * (each in one of the ranges <tt>'0'..'9'</tt>, <tt>'a'..'f'</tt>, or <tt>'A'..'F'</tt>);
12309 * the string to be interpreted ends at the first character that does not represent a hexadecimal digit.
12310 * the string is interpreted as hexadecimal digits (base 16):
12311 *
12312 * '0xfff'.oct # => 4095
12313 * '0xfffg'.oct # => 4095
12314 *
12315 * Any of the above may prefixed with <tt>'-'</tt>, which negates the interpreted value:
12316 *
12317 * '-777'.oct # => -511
12318 * '-0777'.oct # => -511
12319 * '-0b111'.oct # => -7
12320 * '-0xfff'.oct # => -4095
12321 *
12322 * For any substring not described above, returns zero:
12323 *
12324 * 'foo'.oct # => 0
12325 * ''.oct # => 0
12326 *
12327 * Related: see {Converting to Non-String}[rdoc-ref:String@Converting+to+Non-String].
12328 */
12329
12330static VALUE
12331rb_str_oct(VALUE str)
12332{
12333 return rb_str_to_inum(str, -8, FALSE);
12334}
12335
12336#ifndef HAVE_CRYPT_R
12337# include "ruby/thread_native.h"
12338# include "ruby/atomic.h"
12339
12340static struct {
12341 rb_nativethread_lock_t lock;
12342} crypt_mutex = {PTHREAD_MUTEX_INITIALIZER};
12343#endif
12344
12345/*
12346 * call-seq:
12347 * crypt(salt_str) -> new_string
12348 *
12349 * Returns the string generated by calling <code>crypt(3)</code>
12350 * standard library function with <code>str</code> and
12351 * <code>salt_str</code>, in this order, as its arguments. Please do
12352 * not use this method any longer. It is legacy; provided only for
12353 * backward compatibility with ruby scripts in earlier days. It is
12354 * bad to use in contemporary programs for several reasons:
12355 *
12356 * * Behaviour of C's <code>crypt(3)</code> depends on the OS it is
12357 * run. The generated string lacks data portability.
12358 *
12359 * * On some OSes such as Mac OS, <code>crypt(3)</code> never fails
12360 * (i.e. silently ends up in unexpected results).
12361 *
12362 * * On some OSes such as Mac OS, <code>crypt(3)</code> is not
12363 * thread safe.
12364 *
12365 * * So-called "traditional" usage of <code>crypt(3)</code> is very
12366 * very very weak. According to its manpage, Linux's traditional
12367 * <code>crypt(3)</code> output has only 2**56 variations; too
12368 * easy to brute force today. And this is the default behaviour.
12369 *
12370 * * In order to make things robust some OSes implement so-called
12371 * "modular" usage. To go through, you have to do a complex
12372 * build-up of the <code>salt_str</code> parameter, by hand.
12373 * Failure in generation of a proper salt string tends not to
12374 * yield any errors; typos in parameters are normally not
12375 * detectable.
12376 *
12377 * * For instance, in the following example, the second invocation
12378 * of String#crypt is wrong; it has a typo in "round=" (lacks
12379 * "s"). However the call does not fail and something unexpected
12380 * is generated.
12381 *
12382 * "foo".crypt("$5$rounds=1000$salt$") # OK, proper usage
12383 * "foo".crypt("$5$round=1000$salt$") # Typo not detected
12384 *
12385 * * Even in the "modular" mode, some hash functions are considered
12386 * archaic and no longer recommended at all; for instance module
12387 * <code>$1$</code> is officially abandoned by its author: see
12388 * http://phk.freebsd.dk/sagas/md5crypt_eol/ . For another
12389 * instance module <code>$3$</code> is considered completely
12390 * broken: see the manpage of FreeBSD.
12391 *
12392 * * On some OS such as Mac OS, there is no modular mode. Yet, as
12393 * written above, <code>crypt(3)</code> on Mac OS never fails.
12394 * This means even if you build up a proper salt string it
12395 * generates a traditional DES hash anyways, and there is no way
12396 * for you to be aware of.
12397 *
12398 * "foo".crypt("$5$rounds=1000$salt$") # => "$5fNPQMxC5j6."
12399 *
12400 * If for some reason you cannot migrate to other secure contemporary
12401 * password hashing algorithms, install the string-crypt gem and
12402 * <code>require 'string/crypt'</code> to continue using it.
12403 */
12404
12405static VALUE
12406rb_str_crypt(VALUE str, VALUE salt)
12407{
12408#ifdef HAVE_CRYPT_R
12409 VALUE databuf;
12410 struct crypt_data *data;
12411# define CRYPT_END() ALLOCV_END(databuf)
12412#else
12413 char *tmp_buf;
12414 extern char *crypt(const char *, const char *);
12415# define CRYPT_END() rb_nativethread_lock_unlock(&crypt_mutex.lock)
12416#endif
12417 VALUE result;
12418 const char *s, *saltp, *res;
12419#ifdef BROKEN_CRYPT
12420 char salt_8bit_clean[3];
12421#endif
12422
12423 StringValue(salt);
12424 mustnot_wchar(str);
12425 mustnot_wchar(salt);
12426 s = StringValueCStr(str);
12427 saltp = RSTRING_PTR(salt);
12428 if (RSTRING_LEN(salt) < 2 || !saltp[0] || !saltp[1]) {
12429 rb_raise(rb_eArgError, "salt too short (need >=2 bytes)");
12430 }
12431
12432#ifdef BROKEN_CRYPT
12433 if (!ISASCII((unsigned char)saltp[0]) || !ISASCII((unsigned char)saltp[1])) {
12434 salt_8bit_clean[0] = saltp[0] & 0x7f;
12435 salt_8bit_clean[1] = saltp[1] & 0x7f;
12436 salt_8bit_clean[2] = '\0';
12437 saltp = salt_8bit_clean;
12438 }
12439#endif
12440#ifdef HAVE_CRYPT_R
12441 data = ALLOCV(databuf, sizeof(struct crypt_data));
12442# ifdef HAVE_STRUCT_CRYPT_DATA_INITIALIZED
12443 data->initialized = 0;
12444# endif
12445 res = crypt_r(s, saltp, data);
12446#else
12447 rb_nativethread_lock_lock(&crypt_mutex.lock);
12448 res = crypt(s, saltp);
12449#endif
12450 if (!res) {
12451 int err = errno;
12452 CRYPT_END();
12453 rb_syserr_fail(err, "crypt");
12454 }
12455#ifdef HAVE_CRYPT_R
12456 result = rb_str_new_cstr(res);
12457 CRYPT_END();
12458#else
12459 // We need to copy this buffer because it's static and we need to unlock the mutex
12460 // before allocating a new object (the string to be returned). If we allocate while
12461 // holding the lock, we could run GC which fires the VM barrier and causes a deadlock
12462 // if other ractors are waiting on this lock.
12463 size_t res_size = strlen(res);
12464 tmp_buf = ALLOCA_N(char, res_size); // should be small enough to alloca
12465 memcpy(tmp_buf, res, res_size);
12466 CRYPT_END();
12467 result = rb_str_new(tmp_buf, res_size);
12468#endif
12469 return result;
12470}
12471
12472
12473/*
12474 * call-seq:
12475 * ord -> integer
12476 *
12477 * :include: doc/string/ord.rdoc
12478 *
12479 */
12480
12481static VALUE
12482rb_str_ord(VALUE s)
12483{
12484 unsigned int c;
12485
12486 c = rb_enc_codepoint(RSTRING_PTR(s), RSTRING_END(s), STR_ENC_GET(s));
12487 return UINT2NUM(c);
12488}
12489/*
12490 * call-seq:
12491 * sum(n = 16) -> integer
12492 *
12493 * :include: doc/string/sum.rdoc
12494 *
12495 */
12496
12497static VALUE
12498rb_str_sum(int argc, VALUE *argv, VALUE str)
12499{
12500 int bits = 16;
12501 char *ptr, *p, *pend;
12502 long len;
12503 VALUE sum = INT2FIX(0);
12504 unsigned long sum0 = 0;
12505
12506 if (rb_check_arity(argc, 0, 1) && (bits = NUM2INT(argv[0])) < 0) {
12507 bits = 0;
12508 }
12509 ptr = p = RSTRING_PTR(str);
12510 len = RSTRING_LEN(str);
12511 pend = p + len;
12512
12513 while (p < pend) {
12514 if (FIXNUM_MAX - UCHAR_MAX < sum0) {
12515 sum = rb_funcall(sum, '+', 1, LONG2FIX(sum0));
12516 str_mod_check(str, ptr, len);
12517 sum0 = 0;
12518 }
12519 sum0 += (unsigned char)*p;
12520 p++;
12521 }
12522
12523 if (bits == 0) {
12524 if (sum0) {
12525 sum = rb_funcall(sum, '+', 1, LONG2FIX(sum0));
12526 }
12527 }
12528 else {
12529 if (sum == INT2FIX(0)) {
12530 if (bits < (int)sizeof(long)*CHAR_BIT) {
12531 sum0 &= (((unsigned long)1)<<bits)-1;
12532 }
12533 sum = LONG2FIX(sum0);
12534 }
12535 else {
12536 VALUE mod;
12537
12538 if (sum0) {
12539 sum = rb_funcall(sum, '+', 1, LONG2FIX(sum0));
12540 }
12541
12542 mod = rb_funcall(INT2FIX(1), idLTLT, 1, INT2FIX(bits));
12543 mod = rb_funcall(mod, '-', 1, INT2FIX(1));
12544 sum = rb_funcall(sum, '&', 1, mod);
12545 }
12546 }
12547 return sum;
12548}
12549
12550static VALUE
12551rb_str_justify(int argc, VALUE *argv, VALUE str, char jflag)
12552{
12553 rb_encoding *enc;
12554 VALUE w;
12555 long width, len, flen = 1, fclen = 1;
12556 VALUE res;
12557 char *p;
12558 const char *f = " ";
12559 long n, size, llen, rlen, llen2 = 0, rlen2 = 0;
12560 VALUE pad;
12561 int singlebyte = 1, cr;
12562 int termlen;
12563
12564 rb_scan_args(argc, argv, "11", &w, &pad);
12565 enc = STR_ENC_GET(str);
12566 width = NUM2LONG(w);
12567 if (argc == 2) {
12568 StringValue(pad);
12569 enc = rb_enc_check(str, pad);
12570 f = RSTRING_PTR(pad);
12571 flen = RSTRING_LEN(pad);
12572 fclen = str_strlen(pad, enc); /* rb_enc_check */
12573 singlebyte = single_byte_optimizable(pad);
12574 if (flen == 0 || fclen == 0) {
12575 rb_raise(rb_eArgError, "zero width padding");
12576 }
12577 }
12578 termlen = rb_enc_mbminlen(enc);
12579 len = str_strlen(str, enc); /* rb_enc_check */
12580 if (width < 0 || len >= width) return str_duplicate(rb_cString, str);
12581 n = width - len;
12582 llen = (jflag == 'l') ? 0 : ((jflag == 'r') ? n : n/2);
12583 rlen = n - llen;
12584 cr = ENC_CODERANGE(str);
12585 if (flen > 1) {
12586 llen2 = str_offset(f, f + flen, llen % fclen, enc, singlebyte);
12587 rlen2 = str_offset(f, f + flen, rlen % fclen, enc, singlebyte);
12588 }
12589 size = RSTRING_LEN(str);
12590 if ((len = llen / fclen + rlen / fclen) >= LONG_MAX / flen ||
12591 (len *= flen) >= LONG_MAX - llen2 - rlen2 ||
12592 (len += llen2 + rlen2) >= LONG_MAX - size) {
12593 rb_raise(rb_eArgError, "argument too big");
12594 }
12595 len += size;
12596 res = str_enc_new(rb_cString, 0, len, enc);
12597 p = RSTRING_PTR(res);
12598 if (flen <= 1) {
12599 memset(p, *f, llen);
12600 p += llen;
12601 }
12602 else {
12603 while (llen >= fclen) {
12604 memcpy(p,f,flen);
12605 p += flen;
12606 llen -= fclen;
12607 }
12608 if (llen > 0) {
12609 memcpy(p, f, llen2);
12610 p += llen2;
12611 }
12612 }
12613 memcpy(p, RSTRING_PTR(str), size);
12614 p += size;
12615 if (flen <= 1) {
12616 memset(p, *f, rlen);
12617 p += rlen;
12618 }
12619 else {
12620 while (rlen >= fclen) {
12621 memcpy(p,f,flen);
12622 p += flen;
12623 rlen -= fclen;
12624 }
12625 if (rlen > 0) {
12626 memcpy(p, f, rlen2);
12627 p += rlen2;
12628 }
12629 }
12630 TERM_FILL(p, termlen);
12631 STR_SET_LEN(res, p-RSTRING_PTR(res));
12632
12633 if (argc == 2)
12634 cr = ENC_CODERANGE_AND(cr, ENC_CODERANGE(pad));
12635 if (cr != ENC_CODERANGE_BROKEN)
12636 ENC_CODERANGE_SET(res, cr);
12637
12638 RB_GC_GUARD(pad);
12639 return res;
12640}
12641
12642
12643/*
12644 * call-seq:
12645 * ljust(width, pad_string = ' ') -> new_string
12646 *
12647 * :include: doc/string/ljust.rdoc
12648 *
12649 */
12650
12651static VALUE
12652rb_str_ljust(int argc, VALUE *argv, VALUE str)
12653{
12654 return rb_str_justify(argc, argv, str, 'l');
12655}
12656
12657/*
12658 * call-seq:
12659 * rjust(width, pad_string = ' ') -> new_string
12660 *
12661 * :include: doc/string/rjust.rdoc
12662 *
12663 */
12664
12665static VALUE
12666rb_str_rjust(int argc, VALUE *argv, VALUE str)
12667{
12668 return rb_str_justify(argc, argv, str, 'r');
12669}
12670
12671
12672/*
12673 * call-seq:
12674 * center(size, pad_string = ' ') -> new_string
12675 *
12676 * :include: doc/string/center.rdoc
12677 *
12678 */
12679
12680static VALUE
12681rb_str_center(int argc, VALUE *argv, VALUE str)
12682{
12683 return rb_str_justify(argc, argv, str, 'c');
12684}
12685
12686/*
12687 * call-seq:
12688 * partition(pattern) -> [pre_match, first_match, post_match]
12689 *
12690 * :include: doc/string/partition.rdoc
12691 *
12692 */
12693
12694static VALUE
12695rb_str_partition(VALUE str, VALUE sep)
12696{
12697 long pos;
12698
12699 sep = get_pat_quoted(sep, 0);
12700 if (RB_TYPE_P(sep, T_REGEXP)) {
12701 if (rb_reg_search(sep, str, 0, 0) < 0) {
12702 goto failed;
12703 }
12704 VALUE match = rb_backref_get();
12705
12706 pos = RMATCH_BEG(match, 0);
12707 sep = rb_str_subseq(str, pos, RMATCH_END(match, 0) - pos);
12708 }
12709 else {
12710 pos = rb_str_index(str, sep, 0);
12711 if (pos < 0) goto failed;
12712 }
12713
12714 long rpos = pos + RSTRING_LEN(sep);
12715 if (rpos > RSTRING_LEN(str)) goto failed;
12716 return rb_ary_new3(3, rb_str_subseq(str, 0, pos),
12717 sep,
12718 rb_str_subseq(str, rpos, RSTRING_LEN(str)-rpos));
12719
12720 failed:
12721 return rb_ary_new3(3, str_duplicate(rb_cString, str), str_new_empty_String(str), str_new_empty_String(str));
12722}
12723
12724/*
12725 * call-seq:
12726 * rpartition(pattern) -> [pre_match, last_match, post_match]
12727 *
12728 * :include: doc/string/rpartition.rdoc
12729 *
12730 */
12731
12732static VALUE
12733rb_str_rpartition(VALUE str, VALUE sep)
12734{
12735 long pos;
12736
12737 sep = get_pat_quoted(sep, 0);
12738 if (RB_TYPE_P(sep, T_REGEXP)) {
12739 pos = RSTRING_LEN(str);
12740 if (rb_reg_search(sep, str, pos, 1) < 0) {
12741 goto failed;
12742 }
12743 VALUE match = rb_backref_get();
12744
12745 pos = RMATCH_BEG(match, 0);
12746 sep = rb_str_subseq(str, pos, RMATCH_END(match, 0) - pos);
12747 }
12748 else {
12749 /* str may have been modified by #to_str above */
12750 pos = rb_str_sublen(str, RSTRING_LEN(str));
12751 pos = rb_str_rindex(str, sep, pos);
12752 if (pos < 0) {
12753 goto failed;
12754 }
12755 }
12756
12757 long rpos = pos + RSTRING_LEN(sep);
12758 if (rpos > RSTRING_LEN(str)) goto failed;
12759 return rb_ary_new3(3, rb_str_subseq(str, 0, pos),
12760 sep,
12761 rb_str_subseq(str, rpos, RSTRING_LEN(str)-rpos));
12762 failed:
12763 return rb_ary_new3(3, str_new_empty_String(str), str_new_empty_String(str), str_duplicate(rb_cString, str));
12764}
12765
12766/*
12767 * call-seq:
12768 * start_with?(*patterns) -> true or false
12769 *
12770 * :include: doc/string/start_with_p.rdoc
12771 *
12772 */
12773
12774static VALUE
12775rb_str_start_with(int argc, VALUE *argv, VALUE str)
12776{
12777 int i;
12778
12779 for (i=0; i<argc; i++) {
12780 VALUE tmp = argv[i];
12781 if (RB_TYPE_P(tmp, T_REGEXP)) {
12782 if (rb_reg_start_with_p(tmp, str))
12783 return Qtrue;
12784 }
12785 else {
12786 const char *p, *s, *e;
12787 long slen, tlen;
12788 rb_encoding *enc;
12789
12790 StringValue(tmp);
12791 enc = rb_enc_check(str, tmp);
12792 if ((tlen = RSTRING_LEN(tmp)) == 0) return Qtrue;
12793 if ((slen = RSTRING_LEN(str)) < tlen) continue;
12794 p = RSTRING_PTR(str);
12795 e = p + slen;
12796 s = p + tlen;
12797 if (!at_char_right_boundary(p, s, e, enc))
12798 continue;
12799 if (memcmp(p, RSTRING_PTR(tmp), tlen) == 0)
12800 return Qtrue;
12801 }
12802 }
12803 return Qfalse;
12804}
12805
12806/*
12807 * call-seq:
12808 * end_with?(*strings) -> true or false
12809 *
12810 * :include: doc/string/end_with_p.rdoc
12811 *
12812 */
12813
12814static VALUE
12815rb_str_end_with(int argc, VALUE *argv, VALUE str)
12816{
12817 int i;
12818
12819 for (i=0; i<argc; i++) {
12820 VALUE tmp = argv[i];
12821 const char *p, *s, *e;
12822 long slen, tlen;
12823 rb_encoding *enc;
12824
12825 StringValue(tmp);
12826 enc = rb_enc_check(str, tmp);
12827 if ((tlen = RSTRING_LEN(tmp)) == 0) return Qtrue;
12828 if ((slen = RSTRING_LEN(str)) < tlen) continue;
12829 p = RSTRING_PTR(str);
12830 e = p + slen;
12831 s = e - tlen;
12832 if (!at_char_boundary(p, s, e, enc))
12833 continue;
12834 if (memcmp(s, RSTRING_PTR(tmp), tlen) == 0)
12835 return Qtrue;
12836 }
12837 return Qfalse;
12838}
12839
12849static long
12850deleted_prefix_length(VALUE str, VALUE prefix)
12851{
12852 const char *strptr, *prefixptr;
12853 long olen, prefixlen;
12854 rb_encoding *enc = rb_enc_get(str);
12855
12856 StringValue(prefix);
12857
12858 if (!is_broken_string(prefix) ||
12859 !rb_enc_asciicompat(enc) ||
12860 !rb_enc_asciicompat(rb_enc_get(prefix))) {
12861 enc = rb_enc_check(str, prefix);
12862 }
12863
12864 /* return 0 if not start with prefix */
12865 prefixlen = RSTRING_LEN(prefix);
12866 if (prefixlen <= 0) return 0;
12867 olen = RSTRING_LEN(str);
12868 if (olen < prefixlen) return 0;
12869 strptr = RSTRING_PTR(str);
12870 prefixptr = RSTRING_PTR(prefix);
12871 if (memcmp(strptr, prefixptr, prefixlen) != 0) return 0;
12872 if (is_broken_string(prefix)) {
12873 if (!is_broken_string(str)) {
12874 /* prefix in a valid string cannot be broken */
12875 return 0;
12876 }
12877 const char *strend = strptr + olen;
12878 const char *after_prefix = strptr + prefixlen;
12879 if (!at_char_right_boundary(strptr, after_prefix, strend, enc)) {
12880 /* prefix does not end at char-boundary */
12881 return 0;
12882 }
12883 }
12884 /* prefix part in `str` also should be valid. */
12885
12886 return prefixlen;
12887}
12888
12889/*
12890 * call-seq:
12891 * delete_prefix!(prefix) -> self or nil
12892 *
12893 * Like String#delete_prefix, except that +self+ is modified in place;
12894 * returns +self+ if the prefix is removed, +nil+ otherwise.
12895 *
12896 * Related: see {Modifying}[rdoc-ref:String@Modifying].
12897 */
12898
12899static VALUE
12900rb_str_delete_prefix_bang(VALUE str, VALUE prefix)
12901{
12902 long prefixlen;
12903 str_modify_keep_cr(str);
12904
12905 prefixlen = deleted_prefix_length(str, prefix);
12906 if (prefixlen <= 0) return Qnil;
12907
12908 return rb_str_drop_bytes(str, prefixlen);
12909}
12910
12911/*
12912 * call-seq:
12913 * delete_prefix(prefix) -> new_string
12914 *
12915 * :include: doc/string/delete_prefix.rdoc
12916 *
12917 */
12918
12919static VALUE
12920rb_str_delete_prefix(VALUE str, VALUE prefix)
12921{
12922 long prefixlen;
12923
12924 prefixlen = deleted_prefix_length(str, prefix);
12925 if (prefixlen <= 0) return str_duplicate(rb_cString, str);
12926
12927 return rb_str_subseq(str, prefixlen, RSTRING_LEN(str) - prefixlen);
12928}
12929
12939static long
12940deleted_suffix_length(VALUE str, VALUE suffix)
12941{
12942 const char *strptr, *suffixptr;
12943 long olen, suffixlen;
12944 rb_encoding *enc;
12945
12946 StringValue(suffix);
12947 if (is_broken_string(suffix)) return 0;
12948 enc = rb_enc_check(str, suffix);
12949
12950 /* return 0 if not start with suffix */
12951 suffixlen = RSTRING_LEN(suffix);
12952 if (suffixlen <= 0) return 0;
12953 olen = RSTRING_LEN(str);
12954 if (olen < suffixlen) return 0;
12955 strptr = RSTRING_PTR(str);
12956 suffixptr = RSTRING_PTR(suffix);
12957 const char *strend = strptr + olen;
12958 const char *before_suffix = strend - suffixlen;
12959 if (memcmp(before_suffix, suffixptr, suffixlen) != 0) return 0;
12960 if (!at_char_boundary(strptr, before_suffix, strend, enc)) return 0;
12961
12962 return suffixlen;
12963}
12964
12965/*
12966 * call-seq:
12967 * delete_suffix!(suffix) -> self or nil
12968 *
12969 * Like String#delete_suffix, except that +self+ is modified in place;
12970 * returns +self+ if the suffix is removed, +nil+ otherwise.
12971 *
12972 * Related: see {Modifying}[rdoc-ref:String@Modifying].
12973 */
12974
12975static VALUE
12976rb_str_delete_suffix_bang(VALUE str, VALUE suffix)
12977{
12978 long suffixlen;
12979 str_modifiable(str);
12980
12981 suffixlen = deleted_suffix_length(str, suffix);
12982 if (suffixlen <= 0) return Qnil;
12983
12984 return str_shrink(str, RSTRING_LEN(str) - suffixlen);
12985}
12986
12987/*
12988 * call-seq:
12989 * delete_suffix(suffix) -> new_string
12990 *
12991 * :include: doc/string/delete_suffix.rdoc
12992 *
12993 */
12994
12995static VALUE
12996rb_str_delete_suffix(VALUE str, VALUE suffix)
12997{
12998 long suffixlen;
12999
13000 suffixlen = deleted_suffix_length(str, suffix);
13001 if (suffixlen <= 0) return str_duplicate(rb_cString, str);
13002
13003 return rb_str_subseq(str, 0, RSTRING_LEN(str) - suffixlen);
13004}
13005
13006void
13007rb_str_setter(VALUE val, ID id, VALUE *var)
13008{
13009 if (!NIL_P(val) && !RB_TYPE_P(val, T_STRING)) {
13010 rb_raise(rb_eTypeError, "value of %"PRIsVALUE" must be String", rb_id2str(id));
13011 }
13012 *var = val;
13013}
13014
13015static void
13016nil_setter_warning(ID id)
13017{
13018 rb_warn_deprecated("non-nil '%"PRIsVALUE"'", NULL, rb_id2str(id));
13019}
13020
13021void
13022rb_deprecated_str_setter(VALUE val, ID id, VALUE *var)
13023{
13024 rb_str_setter(val, id, var);
13025 if (!NIL_P(*var)) {
13026 nil_setter_warning(id);
13027 }
13028}
13029
13030static void
13031rb_fs_setter(VALUE val, ID id, VALUE *var)
13032{
13033 val = rb_fs_check(val);
13034 if (!val) {
13035 rb_raise(rb_eTypeError,
13036 "value of %"PRIsVALUE" must be String or Regexp",
13037 rb_id2str(id));
13038 }
13039 if (!NIL_P(val)) {
13040 nil_setter_warning(id);
13041 }
13042 *var = val;
13043}
13044
13045
13046/*
13047 * call-seq:
13048 * force_encoding(encoding) -> self
13049 *
13050 * :include: doc/string/force_encoding.rdoc
13051 *
13052 */
13053
13054static VALUE
13055rb_str_force_encoding(VALUE str, VALUE enc)
13056{
13057 str_modifiable(str);
13058
13059 rb_encoding *encoding = rb_to_encoding(enc);
13060 int idx = rb_enc_to_index(encoding);
13061
13062 // If the encoding is unchanged, we do nothing.
13063 if (ENCODING_GET(str) == idx) {
13064 return str;
13065 }
13066
13067 rb_enc_associate_index(str, idx);
13068
13069 // If the coderange was 7bit and the new encoding is ASCII-compatible
13070 // we can keep the coderange.
13071 if (ENC_CODERANGE(str) == ENC_CODERANGE_7BIT && encoding && rb_enc_asciicompat(encoding)) {
13072 return str;
13073 }
13074
13076 return str;
13077}
13078
13079/*
13080 * call-seq:
13081 * b -> new_string
13082 *
13083 * :include: doc/string/b.rdoc
13084 *
13085 */
13086
13087static VALUE
13088rb_str_b(VALUE str)
13089{
13090 VALUE str2;
13091 if (STR_EMBED_P(str)) {
13092 str2 = str_alloc_embed(rb_cString, RSTRING_LEN(str) + TERM_LEN(str));
13093 }
13094 else {
13095 str2 = str_alloc_heap(rb_cString);
13096 }
13097 str_replace_shared_without_enc(str2, str);
13098
13099 if (rb_enc_asciicompat(STR_ENC_GET(str))) {
13100 // BINARY strings can never be broken; they're either 7-bit ASCII or VALID.
13101 // If we know the receiver's code range then we know the result's code range.
13102 int cr = ENC_CODERANGE(str);
13103 switch (cr) {
13104 case ENC_CODERANGE_7BIT:
13106 break;
13110 break;
13111 default:
13112 ENC_CODERANGE_CLEAR(str2);
13113 break;
13114 }
13115 }
13116
13117 return str2;
13118}
13119
13120/* Defined as a leaf builtin in string.rb, so this must never raise or call into Ruby. */
13121static VALUE
13122rb_str_valid_encoding_p(VALUE str)
13123{
13124 int cr = rb_enc_str_coderange(str);
13125
13126 return RBOOL(cr != ENC_CODERANGE_BROKEN);
13127}
13128
13129/* Defined as a leaf builtin in string.rb, so this must never raise or call into Ruby. */
13130static VALUE
13131rb_str_is_ascii_only_p(VALUE str)
13132{
13133 int cr = rb_enc_str_coderange(str);
13134
13135 return RBOOL(cr == ENC_CODERANGE_7BIT);
13136}
13137
13138VALUE
13140{
13141 static const char ellipsis[] = "...";
13142 const long ellipsislen = sizeof(ellipsis) - 1;
13143 rb_encoding *const enc = rb_enc_get(str);
13144 const long blen = RSTRING_LEN(str);
13145 const char *const p = RSTRING_PTR(str), *e = p + blen;
13146 VALUE estr, ret = 0;
13147
13148 if (len < 0) rb_raise(rb_eIndexError, "negative length %ld", len);
13149 if (len * rb_enc_mbminlen(enc) >= blen ||
13150 (e = rb_enc_nth(p, e, len, enc)) - p == blen) {
13151 ret = str;
13152 }
13153 else if (len <= ellipsislen ||
13154 !(e = rb_enc_step_back(p, e, e, len = ellipsislen, enc))) {
13155 if (rb_enc_asciicompat(enc)) {
13156 ret = rb_str_new(ellipsis, len);
13157 rb_enc_associate(ret, enc);
13158 }
13159 else {
13160 estr = rb_usascii_str_new(ellipsis, len);
13161 ret = rb_str_encode(estr, rb_enc_from_encoding(enc), 0, Qnil);
13162 }
13163 }
13164 else if (ret = rb_str_subseq(str, 0, e - p), rb_enc_asciicompat(enc)) {
13165 rb_str_cat(ret, ellipsis, ellipsislen);
13166 }
13167 else {
13168 estr = rb_str_encode(rb_usascii_str_new(ellipsis, ellipsislen),
13169 rb_enc_from_encoding(enc), 0, Qnil);
13170 rb_str_append(ret, estr);
13171 }
13172 return ret;
13173}
13174
13175static VALUE
13176str_compat_and_valid(VALUE str, rb_encoding *enc)
13177{
13178 int cr;
13179 str = StringValue(str);
13180 cr = rb_enc_str_coderange(str);
13181 if (cr == ENC_CODERANGE_BROKEN) {
13182 rb_raise(rb_eArgError, "replacement must be valid byte sequence '%+"PRIsVALUE"'", str);
13183 }
13184 else {
13185 rb_encoding *e = STR_ENC_GET(str);
13186 if (cr == ENC_CODERANGE_7BIT ? rb_enc_mbminlen(enc) != 1 : enc != e) {
13187 rb_raise(rb_eEncCompatError, "incompatible character encodings: %s and %s",
13188 rb_enc_inspect_name(enc), rb_enc_inspect_name(e));
13189 }
13190 }
13191 return str;
13192}
13193
13194static VALUE enc_str_scrub(rb_encoding *enc, VALUE str, VALUE repl, int cr);
13195
13196VALUE
13198{
13199 rb_encoding *enc = STR_ENC_GET(str);
13200 return enc_str_scrub(enc, str, repl, ENC_CODERANGE(str));
13201}
13202
13203VALUE
13204rb_enc_str_scrub(rb_encoding *enc, VALUE str, VALUE repl)
13205{
13206 int cr = ENC_CODERANGE_UNKNOWN;
13207 if (enc == STR_ENC_GET(str)) {
13208 /* cached coderange makes sense only when enc equals the
13209 * actual encoding of str */
13210 cr = ENC_CODERANGE(str);
13211 }
13212 return enc_str_scrub(enc, str, repl, cr);
13213}
13214
13215static VALUE
13216enc_str_scrub(rb_encoding *enc, VALUE str, VALUE repl, int cr)
13217{
13218 int encidx;
13219 VALUE buf = Qnil;
13220 const char *rep, *p, *e, *p1, *sp;
13221 long replen = -1;
13222 long slen;
13223
13224 if (rb_block_given_p()) {
13225 if (!NIL_P(repl))
13226 rb_raise(rb_eArgError, "both of block and replacement given");
13227 replen = 0;
13228 }
13229
13230 if (ENC_CODERANGE_CLEAN_P(cr))
13231 return Qnil;
13232
13233 if (!NIL_P(repl)) {
13234 repl = str_compat_and_valid(repl, enc);
13235 }
13236
13237 if (rb_enc_dummy_p(enc)) {
13238 return Qnil;
13239 }
13240 encidx = rb_enc_to_index(enc);
13241
13242#define DEFAULT_REPLACE_CHAR(str) do { \
13243 RBIMPL_ATTR_NONSTRING() static const char replace[sizeof(str)-1] = str; \
13244 rep = replace; replen = (int)sizeof(replace); \
13245 } while (0)
13246
13247 slen = RSTRING_LEN(str);
13248 p = RSTRING_PTR(str);
13249 e = RSTRING_END(str);
13250 p1 = p;
13251 sp = p;
13252
13253 if (rb_enc_asciicompat(enc)) {
13254 int rep7bit_p;
13255 if (!replen) {
13256 rep = NULL;
13257 rep7bit_p = FALSE;
13258 }
13259 else if (!NIL_P(repl)) {
13260 rep = RSTRING_PTR(repl);
13261 replen = RSTRING_LEN(repl);
13262 rep7bit_p = (ENC_CODERANGE(repl) == ENC_CODERANGE_7BIT);
13263 }
13264 else if (encidx == rb_utf8_encindex()) {
13265 DEFAULT_REPLACE_CHAR("\xEF\xBF\xBD");
13266 rep7bit_p = FALSE;
13267 }
13268 else {
13269 DEFAULT_REPLACE_CHAR("?");
13270 rep7bit_p = TRUE;
13271 }
13272 cr = ENC_CODERANGE_7BIT;
13273
13274 p = search_nonascii(p, e);
13275 if (!p) {
13276 p = e;
13277 }
13278 while (p < e) {
13279 int ret = rb_enc_precise_mbclen(p, e, enc);
13280 if (MBCLEN_NEEDMORE_P(ret)) {
13281 break;
13282 }
13283 else if (MBCLEN_CHARFOUND_P(ret)) {
13285 p += MBCLEN_CHARFOUND_LEN(ret);
13286 /* After a multibyte character, fast-skip the following ASCII run. */
13287 p = search_nonascii(p, e);
13288 if (!p) {
13289 p = e;
13290 break;
13291 }
13292 }
13293 else if (MBCLEN_INVALID_P(ret)) {
13294 /*
13295 * p1~p: valid ascii/multibyte chars
13296 * p ~e: invalid bytes + unknown bytes
13297 */
13298 long clen = rb_enc_mbmaxlen(enc);
13299 if (NIL_P(buf)) buf = rb_str_buf_new(RSTRING_LEN(str));
13300 if (p > p1) {
13301 rb_str_buf_cat(buf, p1, p - p1);
13302 }
13303
13304 if (e - p < clen) clen = e - p;
13305 if (clen <= 2) {
13306 clen = 1;
13307 }
13308 else {
13309 const char *q = p;
13310 clen--;
13311 for (; clen > 1; clen--) {
13312 ret = rb_enc_precise_mbclen(q, q + clen, enc);
13313 if (MBCLEN_NEEDMORE_P(ret)) break;
13314 if (MBCLEN_INVALID_P(ret)) continue;
13316 }
13317 }
13318 if (rep) {
13319 rb_str_buf_cat(buf, rep, replen);
13320 if (!rep7bit_p) cr = ENC_CODERANGE_VALID;
13321 }
13322 else {
13323 repl = rb_yield(rb_enc_str_new(p, clen, enc));
13324 str_mod_check(str, sp, slen);
13325 repl = str_compat_and_valid(repl, enc);
13326 rb_str_buf_cat(buf, RSTRING_PTR(repl), RSTRING_LEN(repl));
13329 }
13330 p += clen;
13331 p1 = p;
13332 p = search_nonascii(p, e);
13333 if (!p) {
13334 p = e;
13335 break;
13336 }
13337 }
13338 else {
13340 }
13341 }
13342 if (NIL_P(buf)) {
13343 if (p == e) {
13344 ENC_CODERANGE_SET(str, cr);
13345 return Qnil;
13346 }
13347 buf = rb_str_buf_new(RSTRING_LEN(str));
13348 }
13349 if (p1 < p) {
13350 rb_str_buf_cat(buf, p1, p - p1);
13351 }
13352 if (p < e) {
13353 if (rep) {
13354 rb_str_buf_cat(buf, rep, replen);
13355 if (!rep7bit_p) cr = ENC_CODERANGE_VALID;
13356 }
13357 else {
13358 repl = rb_yield(rb_enc_str_new(p, e-p, enc));
13359 str_mod_check(str, sp, slen);
13360 repl = str_compat_and_valid(repl, enc);
13361 rb_str_buf_cat(buf, RSTRING_PTR(repl), RSTRING_LEN(repl));
13364 }
13365 }
13366 }
13367 else {
13368 /* ASCII incompatible */
13369 long mbminlen = rb_enc_mbminlen(enc);
13370 if (!replen) {
13371 rep = NULL;
13372 }
13373 else if (!NIL_P(repl)) {
13374 rep = RSTRING_PTR(repl);
13375 replen = RSTRING_LEN(repl);
13376 }
13377 else if (encidx == ENCINDEX_UTF_16BE) {
13378 DEFAULT_REPLACE_CHAR("\xFF\xFD");
13379 }
13380 else if (encidx == ENCINDEX_UTF_16LE) {
13381 DEFAULT_REPLACE_CHAR("\xFD\xFF");
13382 }
13383 else if (encidx == ENCINDEX_UTF_32BE) {
13384 DEFAULT_REPLACE_CHAR("\x00\x00\xFF\xFD");
13385 }
13386 else if (encidx == ENCINDEX_UTF_32LE) {
13387 DEFAULT_REPLACE_CHAR("\xFD\xFF\x00\x00");
13388 }
13389 else {
13390 DEFAULT_REPLACE_CHAR("?");
13391 }
13392
13393 while (p < e) {
13394 int ret = rb_enc_precise_mbclen(p, e, enc);
13395 if (MBCLEN_NEEDMORE_P(ret)) {
13396 break;
13397 }
13398 else if (MBCLEN_CHARFOUND_P(ret)) {
13399 p += MBCLEN_CHARFOUND_LEN(ret);
13400 }
13401 else if (MBCLEN_INVALID_P(ret)) {
13402 const char *q = p;
13403 long clen = rb_enc_mbmaxlen(enc);
13404 if (NIL_P(buf)) buf = rb_str_buf_new(RSTRING_LEN(str));
13405 if (p > p1) rb_str_buf_cat(buf, p1, p - p1);
13406
13407 if (e - p < clen) clen = e - p;
13408 if (clen <= mbminlen * 2) {
13409 clen = mbminlen;
13410 }
13411 else {
13412 clen -= mbminlen;
13413 for (; clen > mbminlen; clen-=mbminlen) {
13414 ret = rb_enc_precise_mbclen(q, q + clen, enc);
13415 if (MBCLEN_NEEDMORE_P(ret)) break;
13416 if (MBCLEN_INVALID_P(ret)) continue;
13418 }
13419 }
13420 if (rep) {
13421 rb_str_buf_cat(buf, rep, replen);
13422 }
13423 else {
13424 repl = rb_yield(rb_enc_str_new(p, clen, enc));
13425 str_mod_check(str, sp, slen);
13426 repl = str_compat_and_valid(repl, enc);
13427 rb_str_buf_cat(buf, RSTRING_PTR(repl), RSTRING_LEN(repl));
13428 }
13429 p += clen;
13430 p1 = p;
13431 }
13432 else {
13434 }
13435 }
13436 if (NIL_P(buf)) {
13437 if (p == e) {
13439 return Qnil;
13440 }
13441 buf = rb_str_buf_new(RSTRING_LEN(str));
13442 }
13443 if (p1 < p) {
13444 rb_str_buf_cat(buf, p1, p - p1);
13445 }
13446 if (p < e) {
13447 if (rep) {
13448 rb_str_buf_cat(buf, rep, replen);
13449 }
13450 else {
13451 repl = rb_yield(rb_enc_str_new(p, e-p, enc));
13452 str_mod_check(str, sp, slen);
13453 repl = str_compat_and_valid(repl, enc);
13454 rb_str_buf_cat(buf, RSTRING_PTR(repl), RSTRING_LEN(repl));
13455 }
13456 }
13458 }
13459 ENCODING_CODERANGE_SET(buf, rb_enc_to_index(enc), cr);
13460 return buf;
13461}
13462
13463/*
13464 * call-seq:
13465 * scrub(replacement_string = default_replacement_string) -> new_string
13466 * scrub{|sequence| ... } -> new_string
13467 *
13468 * :include: doc/string/scrub.rdoc
13469 *
13470 */
13471static VALUE
13472str_scrub(int argc, VALUE *argv, VALUE str)
13473{
13474 VALUE repl = argc ? (rb_check_arity(argc, 0, 1), argv[0]) : Qnil;
13475 VALUE new = rb_str_scrub(str, repl);
13476 return NIL_P(new) ? str_duplicate(rb_cString, str): new;
13477}
13478
13479/*
13480 * call-seq:
13481 * scrub!(replacement_string = default_replacement_string) -> self
13482 * scrub!{|sequence| ... } -> self
13483 *
13484 * Like String#scrub, except that:
13485 *
13486 * - Any replacements are made in +self+.
13487 * - Returns +self+.
13488 *
13489 * Related: see {Modifying}[rdoc-ref:String@Modifying].
13490 *
13491 */
13492static VALUE
13493str_scrub_bang(int argc, VALUE *argv, VALUE str)
13494{
13495 VALUE repl = argc ? (rb_check_arity(argc, 0, 1), argv[0]) : Qnil;
13496 VALUE new = rb_str_scrub(str, repl);
13497 if (!NIL_P(new)) rb_str_replace(str, new);
13498 return str;
13499}
13500
13501static ID id_normalize;
13502static ID id_normalized_p;
13503static VALUE mUnicodeNormalize;
13504
13505static VALUE
13506unicode_normalize_common(int argc, VALUE *argv, VALUE str, ID id)
13507{
13508 static int UnicodeNormalizeRequired = 0;
13509 VALUE argv2[2];
13510
13511 if (!UnicodeNormalizeRequired) {
13512 rb_require("unicode_normalize/normalize.rb");
13513 UnicodeNormalizeRequired = 1;
13514 }
13515 argv2[0] = str;
13516 if (rb_check_arity(argc, 0, 1)) argv2[1] = argv[0];
13517 return rb_funcallv(mUnicodeNormalize, id, argc+1, argv2);
13518}
13519
13520/*
13521 * call-seq:
13522 * unicode_normalize(form = :nfc) -> string
13523 *
13524 * :include: doc/string/unicode_normalize.rdoc
13525 *
13526 */
13527static VALUE
13528rb_str_unicode_normalize(int argc, VALUE *argv, VALUE str)
13529{
13530 return unicode_normalize_common(argc, argv, str, id_normalize);
13531}
13532
13533/*
13534 * call-seq:
13535 * unicode_normalize!(form = :nfc) -> self
13536 *
13537 * Like String#unicode_normalize, except that the normalization
13538 * is performed on +self+ (not on a copy of +self+).
13539 *
13540 * Related: see {Modifying}[rdoc-ref:String@Modifying].
13541 *
13542 */
13543static VALUE
13544rb_str_unicode_normalize_bang(int argc, VALUE *argv, VALUE str)
13545{
13546 return rb_str_replace(str, unicode_normalize_common(argc, argv, str, id_normalize));
13547}
13548
13549/* call-seq:
13550 * unicode_normalized?(form = :nfc) -> true or false
13551 *
13552 * Returns whether +self+ is in the given +form+ of Unicode normalization;
13553 * see String#unicode_normalize.
13554 *
13555 * The +form+ must be one of +:nfc+, +:nfd+, +:nfkc+, or +:nfkd+.
13556 *
13557 * Examples:
13558 *
13559 * "a\u0300".unicode_normalized? # => false
13560 * "a\u0300".unicode_normalized?(:nfd) # => true
13561 * "\u00E0".unicode_normalized? # => true
13562 * "\u00E0".unicode_normalized?(:nfd) # => false
13563 *
13564 *
13565 * Raises an exception if +self+ is not in a Unicode encoding:
13566 *
13567 * s = "\xE0".force_encoding(Encoding::ISO_8859_1)
13568 * s.unicode_normalized? # Raises Encoding::CompatibilityError
13569 *
13570 * Related: see {Querying}[rdoc-ref:String@Querying].
13571 */
13572static VALUE
13573rb_str_unicode_normalized_p(int argc, VALUE *argv, VALUE str)
13574{
13575 return unicode_normalize_common(argc, argv, str, id_normalized_p);
13576}
13577
13578/**********************************************************************
13579 * Document-class: Symbol
13580 *
13581 * A +Symbol+ object represents a named identifier inside the Ruby interpreter.
13582 *
13583 * You can create a +Symbol+ object explicitly with:
13584 *
13585 * - A {symbol literal}[rdoc-ref:syntax/literals.rdoc@Symbol+Literals].
13586 *
13587 * The same +Symbol+ object will be
13588 * created for a given name or string for the duration of a program's
13589 * execution, regardless of the context or meaning of that name. Thus
13590 * if <code>Fred</code> is a constant in one context, a method in
13591 * another, and a class in a third, the +Symbol+ <code>:Fred</code>
13592 * will be the same object in all three contexts.
13593 *
13594 * module One
13595 * class Fred
13596 * end
13597 * $f1 = :Fred
13598 * end
13599 * module Two
13600 * Fred = 1
13601 * $f2 = :Fred
13602 * end
13603 * def Fred()
13604 * end
13605 * $f3 = :Fred
13606 * $f1.object_id #=> 2514190
13607 * $f2.object_id #=> 2514190
13608 * $f3.object_id #=> 2514190
13609 *
13610 * Constant, method, and variable names are returned as symbols:
13611 *
13612 * module One
13613 * Two = 2
13614 * def three; 3 end
13615 * @four = 4
13616 * @@five = 5
13617 * $six = 6
13618 * end
13619 * seven = 7
13620 *
13621 * One.constants
13622 * # => [:Two]
13623 * One.instance_methods(true)
13624 * # => [:three]
13625 * One.instance_variables
13626 * # => [:@four]
13627 * One.class_variables
13628 * # => [:@@five]
13629 * global_variables.grep(/six/)
13630 * # => [:$six]
13631 * local_variables
13632 * # => [:seven]
13633 *
13634 * A +Symbol+ object differs from a String object in that
13635 * a +Symbol+ object represents an identifier, while a String object
13636 * represents text or data.
13637 *
13638 * == What's Here
13639 *
13640 * First, what's elsewhere. Class +Symbol+:
13641 *
13642 * - Inherits from {class Object}[rdoc-ref:Object@Whats+Here].
13643 * - Includes {module Comparable}[rdoc-ref:Comparable@Whats+Here].
13644 *
13645 * Here, class +Symbol+ provides methods that are useful for:
13646 *
13647 * - {Querying}[rdoc-ref:Symbol@Methods+for+Querying]
13648 * - {Comparing}[rdoc-ref:Symbol@Methods+for+Comparing]
13649 * - {Converting}[rdoc-ref:Symbol@Methods+for+Converting]
13650 *
13651 * === Methods for Querying
13652 *
13653 * - ::all_symbols: Returns an array of the symbols currently in Ruby's symbol table.
13654 * - #=~: Returns the index of the first substring in symbol that matches a
13655 * given Regexp or other object; returns +nil+ if no match is found.
13656 * - #[], #slice : Returns a substring of symbol
13657 * determined by a given index, start/length, or range, or string.
13658 * - #empty?: Returns +true+ if +self.length+ is zero; +false+ otherwise.
13659 * - #encoding: Returns the Encoding object that represents the encoding
13660 * of symbol.
13661 * - #end_with?: Returns +true+ if symbol ends with
13662 * any of the given strings.
13663 * - #match: Returns a MatchData object if symbol
13664 * matches a given Regexp; +nil+ otherwise.
13665 * - #match?: Returns +true+ if symbol
13666 * matches a given Regexp; +false+ otherwise.
13667 * - #length, #size: Returns the number of characters in symbol.
13668 * - #start_with?: Returns +true+ if symbol starts with
13669 * any of the given strings.
13670 *
13671 * === Methods for Comparing
13672 *
13673 * - #<=>: Returns -1, 0, or 1 as a given symbol is smaller than, equal to,
13674 * or larger than symbol.
13675 * - #==, #===: Returns +true+ if a given symbol has the same content and
13676 * encoding.
13677 * - #casecmp: Ignoring case, returns -1, 0, or 1 as a given
13678 * symbol is smaller than, equal to, or larger than symbol.
13679 * - #casecmp?: Returns +true+ if symbol is equal to a given symbol
13680 * after Unicode case folding; +false+ otherwise.
13681 *
13682 * === Methods for Converting
13683 *
13684 * - #capitalize: Returns symbol with the first character upcased
13685 * and all other characters downcased.
13686 * - #downcase: Returns symbol with all characters downcased.
13687 * - #inspect: Returns the string representation of +self+ as a symbol literal.
13688 * - #name: Returns the frozen string corresponding to symbol.
13689 * - #succ, #next: Returns the symbol that is the successor to symbol.
13690 * - #swapcase: Returns symbol with all upcase characters downcased
13691 * and all downcase characters upcased.
13692 * - #to_proc: Returns a Proc object which responds to the method named by symbol.
13693 * - #to_s, #id2name: Returns the string corresponding to +self+.
13694 * - #to_sym, #intern: Returns +self+.
13695 * - #upcase: Returns symbol with all characters upcased.
13696 *
13697 */
13698
13699
13700/*
13701 * call-seq:
13702 * self == other -> true or false
13703 *
13704 * Returns whether +other+ is the same object as +self+.
13705 */
13706
13707#define sym_equal rb_obj_equal
13708
13709static int
13710sym_printable(const char *s, const char *send, rb_encoding *enc)
13711{
13712 while (s < send) {
13713 int n;
13714 int c = rb_enc_precise_mbclen(s, send, enc);
13715
13716 if (!MBCLEN_CHARFOUND_P(c)) return FALSE;
13717 n = MBCLEN_CHARFOUND_LEN(c);
13718 c = rb_enc_mbc_to_codepoint(s, send, enc);
13719 if (!rb_enc_isprint(c, enc)) return FALSE;
13720 s += n;
13721 }
13722 return TRUE;
13723}
13724
13725int
13726rb_str_symname_p(VALUE sym)
13727{
13728 rb_encoding *enc;
13729 const char *ptr;
13730 long len;
13731 rb_encoding *resenc = rb_default_internal_encoding();
13732
13733 if (resenc == NULL) resenc = rb_default_external_encoding();
13734 enc = STR_ENC_GET(sym);
13735 ptr = RSTRING_PTR(sym);
13736 len = RSTRING_LEN(sym);
13737 if ((resenc != enc && !rb_str_is_ascii_only_p(sym)) || len != (long)strlen(ptr) ||
13738 !rb_enc_symname2_p(ptr, len, enc) || !sym_printable(ptr, ptr + len, enc)) {
13739 return FALSE;
13740 }
13741 return TRUE;
13742}
13743
13744VALUE
13745rb_str_quote_unprintable(VALUE str)
13746{
13747 rb_encoding *enc;
13748 const char *ptr;
13749 long len;
13750 rb_encoding *resenc;
13751
13752 Check_Type(str, T_STRING);
13753 resenc = rb_default_internal_encoding();
13754 if (resenc == NULL) resenc = rb_default_external_encoding();
13755 enc = STR_ENC_GET(str);
13756 ptr = RSTRING_PTR(str);
13757 len = RSTRING_LEN(str);
13758 if ((resenc != enc && !rb_str_is_ascii_only_p(str)) ||
13759 !sym_printable(ptr, ptr + len, enc)) {
13760 return rb_str_escape(str);
13761 }
13762 return str;
13763}
13764
13765VALUE
13766rb_id_quote_unprintable(ID id)
13767{
13768 VALUE str = rb_id2str(id);
13769 if (!rb_str_symname_p(str)) {
13770 return rb_str_escape(str);
13771 }
13772 return str;
13773}
13774
13775/*
13776 * call-seq:
13777 * inspect -> string
13778 *
13779 * Returns a string representation of +self+ (including the leading colon):
13780 *
13781 * :foo.inspect # => ":foo"
13782 *
13783 * Related: Symbol#to_s, Symbol#name.
13784 *
13785 */
13786
13787static VALUE
13788sym_inspect(VALUE sym)
13789{
13790 VALUE str = rb_sym2str(sym);
13791 const char *ptr;
13792 long len;
13793 char *dest;
13794
13795 if (!rb_str_symname_p(str)) {
13796 str = rb_str_inspect(str);
13797 len = RSTRING_LEN(str);
13798 rb_str_resize(str, len + 1);
13799 dest = RSTRING_PTR(str);
13800 memmove(dest + 1, dest, len);
13801 }
13802 else {
13803 rb_encoding *enc = STR_ENC_GET(str);
13804 VALUE orig_str = str;
13805
13806 len = RSTRING_LEN(orig_str);
13807 str = rb_enc_str_new(0, len + 1, enc);
13808
13809 // Get data pointer after allocation
13810 ptr = RSTRING_PTR(orig_str);
13811 dest = RSTRING_PTR(str);
13812 memcpy(dest + 1, ptr, len);
13813
13814 RB_GC_GUARD(orig_str);
13815 }
13816 dest[0] = ':';
13817
13819
13820 return str;
13821}
13822
13823VALUE
13825{
13826 return rb_sym2str(sym);
13827}
13828
13829VALUE
13830rb_sym_proc_call(ID mid, int argc, const VALUE *argv, int kw_splat, VALUE passed_proc)
13831{
13832 VALUE obj;
13833
13834 if (argc < 1) {
13835 rb_raise(rb_eArgError, "no receiver given");
13836 }
13837 obj = argv[0];
13838 return rb_funcall_with_block_kw(obj, mid, argc - 1, argv + 1, passed_proc, kw_splat);
13839}
13840
13841/*
13842 * call-seq:
13843 * succ
13844 *
13845 * Equivalent to <tt>self.to_s.succ.to_sym</tt>:
13846 *
13847 * :foo.succ # => :fop
13848 *
13849 * Related: String#succ.
13850 */
13851
13852static VALUE
13853sym_succ(VALUE sym)
13854{
13855 return rb_str_intern(rb_str_succ(rb_sym2str(sym)));
13856}
13857
13858/*
13859 * call-seq:
13860 * self <=> other -> -1, 0, 1, or nil
13861 *
13862 * Compares +self+ and +other+, using String#<=>.
13863 *
13864 * Returns:
13865 *
13866 * - <tt>self.to_s <=> other.to_s</tt>, if +other+ is a symbol.
13867 * - +nil+, otherwise.
13868 *
13869 * Examples:
13870 *
13871 * :bar <=> :foo # => -1
13872 * :foo <=> :foo # => 0
13873 * :foo <=> :bar # => 1
13874 * :foo <=> 'bar' # => nil
13875 *
13876 * \Class \Symbol includes module Comparable,
13877 * each of whose methods uses Symbol#<=> for comparison.
13878 *
13879 * Related: String#<=>.
13880 */
13881
13882static VALUE
13883sym_cmp(VALUE sym, VALUE other)
13884{
13885 if (!SYMBOL_P(other)) {
13886 return Qnil;
13887 }
13888 return rb_str_cmp_m(rb_sym2str(sym), rb_sym2str(other));
13889}
13890
13891/*
13892 * call-seq:
13893 * casecmp(object) -> -1, 0, 1, or nil
13894 *
13895 * :include: doc/symbol/casecmp.rdoc
13896 *
13897 */
13898
13899static VALUE
13900sym_casecmp(VALUE sym, VALUE other)
13901{
13902 if (!SYMBOL_P(other)) {
13903 return Qnil;
13904 }
13905 return str_casecmp(rb_sym2str(sym), rb_sym2str(other));
13906}
13907
13908/*
13909 * call-seq:
13910 * casecmp?(object) -> true, false, or nil
13911 *
13912 * :include: doc/symbol/casecmp_p.rdoc
13913 *
13914 */
13915
13916static VALUE
13917sym_casecmp_p(VALUE sym, VALUE other)
13918{
13919 if (!SYMBOL_P(other)) {
13920 return Qnil;
13921 }
13922 return str_casecmp_p(rb_sym2str(sym), rb_sym2str(other));
13923}
13924
13925/*
13926 * call-seq:
13927 * self =~ other -> integer or nil
13928 *
13929 * Equivalent to <tt>self.to_s =~ other</tt>,
13930 * including possible updates to global variables;
13931 * see String#=~.
13932 *
13933 */
13934
13935static VALUE
13936sym_match(VALUE sym, VALUE other)
13937{
13938 return rb_str_match(rb_sym2str(sym), other);
13939}
13940
13941/*
13942 * call-seq:
13943 * match(pattern, offset = 0) -> matchdata or nil
13944 * match(pattern, offset = 0) {|matchdata| } -> object
13945 *
13946 * Equivalent to <tt>self.to_s.match</tt>,
13947 * including possible updates to global variables;
13948 * see String#match.
13949 *
13950 */
13951
13952static VALUE
13953sym_match_m(int argc, VALUE *argv, VALUE sym)
13954{
13955 return rb_str_match_m(argc, argv, rb_sym2str(sym));
13956}
13957
13958/*
13959 * call-seq:
13960 * match?(pattern, offset) -> true or false
13961 *
13962 * Equivalent to <tt>sym.to_s.match?</tt>;
13963 * see String#match.
13964 *
13965 */
13966
13967static VALUE
13968sym_match_m_p(int argc, VALUE *argv, VALUE sym)
13969{
13970 return rb_str_match_m_p(argc, argv, sym);
13971}
13972
13973/*
13974 * call-seq:
13975 * self[offset] -> string or nil
13976 * self[offset, size] -> string or nil
13977 * self[range] -> string or nil
13978 * self[regexp, capture = 0] -> string or nil
13979 * self[substring] -> string or nil
13980 *
13981 * Equivalent to <tt>symbol.to_s[]</tt>; see String#[].
13982 *
13983 */
13984
13985static VALUE
13986sym_aref(int argc, VALUE *argv, VALUE sym)
13987{
13988 return rb_str_aref_m(argc, argv, rb_sym2str(sym));
13989}
13990
13991/*
13992 * call-seq:
13993 * length -> integer
13994 *
13995 * Equivalent to <tt>self.to_s.length</tt>; see String#length.
13996 */
13997
13998static VALUE
13999sym_length(VALUE sym)
14000{
14001 return rb_str_length(rb_sym2str(sym));
14002}
14003
14004/*
14005 * call-seq:
14006 * upcase(mapping) -> symbol
14007 *
14008 * Equivalent to <tt>sym.to_s.upcase.to_sym</tt>.
14009 *
14010 * See String#upcase.
14011 *
14012 */
14013
14014static VALUE
14015sym_upcase(int argc, VALUE *argv, VALUE sym)
14016{
14017 return rb_str_intern(rb_str_upcase(argc, argv, rb_sym2str(sym)));
14018}
14019
14020/*
14021 * call-seq:
14022 * downcase(mapping) -> symbol
14023 *
14024 * Equivalent to <tt>sym.to_s.downcase.to_sym</tt>.
14025 *
14026 * See String#downcase.
14027 *
14028 * Related: Symbol#upcase.
14029 *
14030 */
14031
14032static VALUE
14033sym_downcase(int argc, VALUE *argv, VALUE sym)
14034{
14035 return rb_str_intern(rb_str_downcase(argc, argv, rb_sym2str(sym)));
14036}
14037
14038/*
14039 * call-seq:
14040 * capitalize(mapping) -> symbol
14041 *
14042 * Equivalent to <tt>sym.to_s.capitalize.to_sym</tt>.
14043 *
14044 * See String#capitalize.
14045 *
14046 */
14047
14048static VALUE
14049sym_capitalize(int argc, VALUE *argv, VALUE sym)
14050{
14051 return rb_str_intern(rb_str_capitalize(argc, argv, rb_sym2str(sym)));
14052}
14053
14054/*
14055 * call-seq:
14056 * swapcase(mapping) -> symbol
14057 *
14058 * Equivalent to <tt>sym.to_s.swapcase.to_sym</tt>.
14059 *
14060 * See String#swapcase.
14061 *
14062 */
14063
14064static VALUE
14065sym_swapcase(int argc, VALUE *argv, VALUE sym)
14066{
14067 return rb_str_intern(rb_str_swapcase(argc, argv, rb_sym2str(sym)));
14068}
14069
14070/*
14071 * call-seq:
14072 * start_with?(*string_or_regexp) -> true or false
14073 *
14074 * Equivalent to <tt>self.to_s.start_with?</tt>; see String#start_with?.
14075 *
14076 */
14077
14078static VALUE
14079sym_start_with(int argc, VALUE *argv, VALUE sym)
14080{
14081 return rb_str_start_with(argc, argv, rb_sym2str(sym));
14082}
14083
14084/*
14085 * call-seq:
14086 * end_with?(*strings) -> true or false
14087 *
14088 *
14089 * Equivalent to <tt>self.to_s.end_with?</tt>; see String#end_with?.
14090 *
14091 */
14092
14093static VALUE
14094sym_end_with(int argc, VALUE *argv, VALUE sym)
14095{
14096 return rb_str_end_with(argc, argv, rb_sym2str(sym));
14097}
14098
14099/*
14100 * call-seq:
14101 * encoding -> encoding
14102 *
14103 * Equivalent to <tt>self.to_s.encoding</tt>; see String#encoding.
14104 *
14105 */
14106
14107static VALUE
14108sym_encoding(VALUE sym)
14109{
14110 return rb_obj_encoding(rb_sym2str(sym));
14111}
14112
14113static VALUE
14114string_for_symbol(VALUE name)
14115{
14116 if (!RB_TYPE_P(name, T_STRING)) {
14117 VALUE tmp = rb_check_string_type(name);
14118 if (NIL_P(tmp)) {
14119 rb_raise(rb_eTypeError, "%+"PRIsVALUE" is not a symbol nor a string",
14120 name);
14121 }
14122 name = tmp;
14123 }
14124 return name;
14125}
14126
14127ID
14129{
14130 if (SYMBOL_P(name)) {
14131 return SYM2ID(name);
14132 }
14133 name = string_for_symbol(name);
14134 return rb_intern_str(name);
14135}
14136
14137VALUE
14139{
14140 if (SYMBOL_P(name)) {
14141 return name;
14142 }
14143 name = string_for_symbol(name);
14144 return rb_str_intern(name);
14145}
14146
14147/*
14148 * call-seq:
14149 * Symbol.all_symbols -> array_of_symbols
14150 *
14151 * Returns an array of all symbols currently in Ruby's symbol table:
14152 *
14153 * Symbol.all_symbols.size # => 9334
14154 * Symbol.all_symbols.take(3) # => [:!, :"\"", :"#"]
14155 *
14156 */
14157
14158static VALUE
14159sym_all_symbols(VALUE _)
14160{
14161 return rb_sym_all_symbols();
14162}
14163
14164VALUE
14165rb_str_to_interned_str(VALUE str)
14166{
14167 return rb_fstring(str);
14168}
14169
14170VALUE
14171rb_interned_str(const char *ptr, long len)
14172{
14173 struct RString fake_str = {RBASIC_INIT};
14174 int encidx = ENCINDEX_US_ASCII;
14175 int coderange = ENC_CODERANGE_7BIT;
14176 if (len > 0 && search_nonascii(ptr, ptr + len)) {
14177 encidx = ENCINDEX_ASCII_8BIT;
14178 coderange = ENC_CODERANGE_VALID;
14179 }
14180 VALUE str = setup_fake_str(&fake_str, ptr, len, encidx);
14181 ENC_CODERANGE_SET(str, coderange);
14182 return register_fstring(str, true, false);
14183}
14184
14185VALUE
14187{
14188 return rb_interned_str(ptr, strlen(ptr));
14189}
14190
14191VALUE
14192rb_enc_interned_str(const char *ptr, long len, rb_encoding *enc)
14193{
14194 if (enc != NULL && UNLIKELY(rb_enc_autoload_p(enc))) {
14195 rb_enc_autoload(enc);
14196 }
14197
14198 struct RString fake_str = {RBASIC_INIT};
14199 return register_fstring(rb_setup_fake_str(&fake_str, ptr, len, enc), true, false);
14200}
14201
14202VALUE
14203rb_enc_literal_str(const char *ptr, long len, rb_encoding *enc)
14204{
14205 if (enc != NULL && UNLIKELY(rb_enc_autoload_p(enc))) {
14206 rb_enc_autoload(enc);
14207 }
14208
14209 struct RString fake_str = {RBASIC_INIT};
14210 VALUE str = register_fstring(rb_setup_fake_str(&fake_str, ptr, len, enc), true, true);
14211 RUBY_ASSERT(RB_OBJ_SHAREABLE_P(str) && (rb_gc_verify_shareable(str), 1));
14212 return str;
14213}
14214
14215VALUE
14217{
14218 return rb_enc_interned_str(ptr, strlen(ptr), enc);
14219}
14220
14221#if USE_YJIT || USE_ZJIT
14222void
14223rb_jit_str_concat_codepoint(VALUE str, VALUE codepoint)
14224{
14225 if (RB_LIKELY(ENCODING_GET_INLINED(str) == rb_ascii8bit_encindex())) {
14226 ssize_t code = RB_NUM2SSIZE(codepoint);
14227
14228 if (RB_LIKELY(code >= 0 && code < 0xff)) {
14229 rb_str_buf_cat_byte(str, (char) code);
14230 return;
14231 }
14232 }
14233
14234 rb_str_concat(str, codepoint);
14235}
14236#endif
14237
14238static int
14239fstring_set_class_i(VALUE *str, void *data)
14240{
14241 RBASIC_SET_CLASS(*str, rb_cString);
14242
14243 return ST_CONTINUE;
14244}
14245
14246void
14247Init_String(void)
14248{
14249 rb_cString = rb_define_class("String", rb_cObject);
14250
14251 rb_concurrent_set_foreach_with_replace(fstring_table_obj, fstring_set_class_i, NULL);
14252
14254 rb_define_alloc_func(rb_cString, empty_str_alloc);
14255 rb_define_singleton_method(rb_cString, "new", rb_str_s_new, -1);
14256 rb_define_singleton_method(rb_cString, "try_convert", rb_str_s_try_convert, 1);
14257 rb_define_method(rb_cString, "initialize", rb_str_init, -1);
14259 rb_define_method(rb_cString, "initialize_copy", rb_str_replace, 1);
14260 rb_define_method(rb_cString, "<=>", rb_str_cmp_m, 1);
14263 rb_define_method(rb_cString, "eql?", rb_str_eql, 1);
14264 rb_define_method(rb_cString, "hash", rb_str_hash_m, 0);
14265 rb_define_method(rb_cString, "casecmp", rb_str_casecmp, 1);
14266 rb_define_method(rb_cString, "casecmp?", rb_str_casecmp_p, 1);
14269 rb_define_method(rb_cString, "%", rb_str_format_m, 1);
14270 rb_define_method(rb_cString, "[]", rb_str_aref_m, -1);
14271 rb_define_method(rb_cString, "[]=", rb_str_aset_m, -1);
14272 rb_define_method(rb_cString, "insert", rb_str_insert, 2);
14275 rb_define_method(rb_cString, "bytesize", rb_str_bytesize, 0);
14276 rb_define_method(rb_cString, "empty?", rb_str_empty, 0);
14277 rb_define_method(rb_cString, "=~", rb_str_match, 1);
14278 rb_define_method(rb_cString, "match", rb_str_match_m, -1);
14279 rb_define_method(rb_cString, "match?", rb_str_match_m_p, -1);
14281 rb_define_method(rb_cString, "succ!", rb_str_succ_bang, 0);
14283 rb_define_method(rb_cString, "next!", rb_str_succ_bang, 0);
14284 rb_define_method(rb_cString, "upto", rb_str_upto, -1);
14285 rb_define_method(rb_cString, "index", rb_str_index_m, -1);
14286 rb_define_method(rb_cString, "byteindex", rb_str_byteindex_m, -1);
14287 rb_define_method(rb_cString, "rindex", rb_str_rindex_m, -1);
14288 rb_define_method(rb_cString, "byterindex", rb_str_byterindex_m, -1);
14289 rb_define_method(rb_cString, "clear", rb_str_clear, 0);
14290 rb_define_method(rb_cString, "chr", rb_str_chr, 0);
14291 rb_define_method(rb_cString, "getbyte", rb_str_getbyte, 1);
14292 rb_define_method(rb_cString, "setbyte", rb_str_setbyte, 2);
14293 rb_define_method(rb_cString, "bit_get", rb_str_bit_get, -1);
14294 rb_define_method(rb_cString, "bit_set?", rb_str_bit_set_p, -1);
14295 rb_define_method(rb_cString, "bit_set", rb_str_bit_set, -1);
14296 rb_define_method(rb_cString, "bit_clear", rb_str_bit_clear, -1);
14297 rb_define_method(rb_cString, "bit_flip", rb_str_bit_flip, -1);
14298 rb_define_method(rb_cString, "bit_count", rb_str_bit_count, -1);
14299 rb_define_method(rb_cString, "bitwise_not", rb_str_bitwise_not, 0);
14300 rb_define_method(rb_cString, "bitwise_not!", rb_str_bitwise_not_bang, 0);
14301 rb_define_method(rb_cString, "bitwise_and", rb_str_bitwise_and, 1);
14302 rb_define_method(rb_cString, "bitwise_and!", rb_str_bitwise_and_bang, 1);
14303 rb_define_method(rb_cString, "bitwise_or", rb_str_bitwise_or, 1);
14304 rb_define_method(rb_cString, "bitwise_or!", rb_str_bitwise_or_bang, 1);
14305 rb_define_method(rb_cString, "bitwise_xor", rb_str_bitwise_xor, 1);
14306 rb_define_method(rb_cString, "bitwise_xor!", rb_str_bitwise_xor_bang, 1);
14307 rb_define_method(rb_cString, "byteslice", rb_str_byteslice, -1);
14308 rb_define_method(rb_cString, "bytesplice", rb_str_bytesplice, -1);
14309 rb_define_method(rb_cString, "scrub", str_scrub, -1);
14310 rb_define_method(rb_cString, "scrub!", str_scrub_bang, -1);
14312 rb_define_method(rb_cString, "+@", str_uplus, 0);
14313 rb_define_method(rb_cString, "-@", str_uminus, 0);
14314 rb_define_method(rb_cString, "dup", rb_str_dup_m, 0);
14315 rb_define_alias(rb_cString, "dedup", "-@");
14316
14317 rb_define_method(rb_cString, "to_i", rb_str_to_i, -1);
14318 rb_define_method(rb_cString, "to_f", rb_str_to_f, 0);
14319 rb_define_method(rb_cString, "to_s", rb_str_to_s, 0);
14320 rb_define_method(rb_cString, "to_str", rb_str_to_s, 0);
14323 rb_define_method(rb_cString, "undump", str_undump, 0);
14324
14325 sym_ascii = ID2SYM(rb_intern_const("ascii"));
14326 sym_turkic = ID2SYM(rb_intern_const("turkic"));
14327 sym_lithuanian = ID2SYM(rb_intern_const("lithuanian"));
14328 sym_fold = ID2SYM(rb_intern_const("fold"));
14329
14330 rb_define_method(rb_cString, "upcase", rb_str_upcase, -1);
14331 rb_define_method(rb_cString, "downcase", rb_str_downcase, -1);
14332 rb_define_method(rb_cString, "capitalize", rb_str_capitalize, -1);
14333 rb_define_method(rb_cString, "swapcase", rb_str_swapcase, -1);
14334
14335 rb_define_method(rb_cString, "upcase!", rb_str_upcase_bang, -1);
14336 rb_define_method(rb_cString, "downcase!", rb_str_downcase_bang, -1);
14337 rb_define_method(rb_cString, "capitalize!", rb_str_capitalize_bang, -1);
14338 rb_define_method(rb_cString, "swapcase!", rb_str_swapcase_bang, -1);
14339
14340 rb_define_method(rb_cString, "hex", rb_str_hex, 0);
14341 rb_define_method(rb_cString, "oct", rb_str_oct, 0);
14342 rb_define_method(rb_cString, "split", rb_str_split_m, -1);
14343 rb_define_method(rb_cString, "lines", rb_str_lines, -1);
14344 rb_define_method(rb_cString, "bytes", rb_str_bytes, 0);
14345 rb_define_method(rb_cString, "chars", rb_str_chars, 0);
14346 rb_define_method(rb_cString, "codepoints", rb_str_codepoints, 0);
14347 rb_define_method(rb_cString, "grapheme_clusters", rb_str_grapheme_clusters, 0);
14348 rb_define_method(rb_cString, "reverse", rb_str_reverse, 0);
14349 rb_define_method(rb_cString, "reverse!", rb_str_reverse_bang, 0);
14350 rb_define_method(rb_cString, "concat", rb_str_concat_multi, -1);
14351 rb_define_method(rb_cString, "append_as_bytes", rb_str_append_as_bytes, -1);
14353 rb_define_method(rb_cString, "prepend", rb_str_prepend_multi, -1);
14354 rb_define_method(rb_cString, "crypt", rb_str_crypt, 1);
14355 rb_define_method(rb_cString, "intern", rb_str_intern, 0); /* in symbol.c */
14356 rb_define_method(rb_cString, "to_sym", rb_str_intern, 0); /* in symbol.c */
14357 rb_define_method(rb_cString, "ord", rb_str_ord, 0);
14358
14359 rb_define_method(rb_cString, "include?", rb_str_include, 1);
14360 rb_define_method(rb_cString, "start_with?", rb_str_start_with, -1);
14361 rb_define_method(rb_cString, "end_with?", rb_str_end_with, -1);
14362
14363 rb_define_method(rb_cString, "scan", rb_str_scan, 1);
14364
14365 rb_define_method(rb_cString, "ljust", rb_str_ljust, -1);
14366 rb_define_method(rb_cString, "rjust", rb_str_rjust, -1);
14367 rb_define_method(rb_cString, "center", rb_str_center, -1);
14368
14369 rb_define_method(rb_cString, "sub", rb_str_sub, -1);
14370 rb_define_method(rb_cString, "gsub", rb_str_gsub, -1);
14371 rb_define_method(rb_cString, "chop", rb_str_chop, 0);
14372 rb_define_method(rb_cString, "chomp", rb_str_chomp, -1);
14373 rb_define_method(rb_cString, "strip", rb_str_strip, -1);
14374 rb_define_method(rb_cString, "lstrip", rb_str_lstrip, -1);
14375 rb_define_method(rb_cString, "rstrip", rb_str_rstrip, -1);
14376 rb_define_method(rb_cString, "delete_prefix", rb_str_delete_prefix, 1);
14377 rb_define_method(rb_cString, "delete_suffix", rb_str_delete_suffix, 1);
14378
14379 rb_define_method(rb_cString, "sub!", rb_str_sub_bang, -1);
14380 rb_define_method(rb_cString, "gsub!", rb_str_gsub_bang, -1);
14381 rb_define_method(rb_cString, "chop!", rb_str_chop_bang, 0);
14382 rb_define_method(rb_cString, "chomp!", rb_str_chomp_bang, -1);
14383 rb_define_method(rb_cString, "strip!", rb_str_strip_bang, -1);
14384 rb_define_method(rb_cString, "lstrip!", rb_str_lstrip_bang, -1);
14385 rb_define_method(rb_cString, "rstrip!", rb_str_rstrip_bang, -1);
14386 rb_define_method(rb_cString, "delete_prefix!", rb_str_delete_prefix_bang, 1);
14387 rb_define_method(rb_cString, "delete_suffix!", rb_str_delete_suffix_bang, 1);
14388
14389 rb_define_method(rb_cString, "tr", rb_str_tr, -1);
14390 rb_define_method(rb_cString, "tr_s", rb_str_tr_s, 2);
14391 rb_define_method(rb_cString, "delete", rb_str_delete, -1);
14392 rb_define_method(rb_cString, "squeeze", rb_str_squeeze, -1);
14393 rb_define_method(rb_cString, "count", rb_str_count, -1);
14394
14395 rb_define_method(rb_cString, "tr!", rb_str_tr_bang, -1);
14396 rb_define_method(rb_cString, "tr_s!", rb_str_tr_s_bang, 2);
14397 rb_define_method(rb_cString, "delete!", rb_str_delete_bang, -1);
14398 rb_define_method(rb_cString, "squeeze!", rb_str_squeeze_bang, -1);
14399
14400 rb_define_method(rb_cString, "each_line", rb_str_each_line, -1);
14401 rb_define_method(rb_cString, "each_byte", rb_str_each_byte, 0);
14402 rb_define_method(rb_cString, "each_char", rb_str_each_char, 0);
14403 rb_define_method(rb_cString, "each_codepoint", rb_str_each_codepoint, 0);
14404 rb_define_method(rb_cString, "each_grapheme_cluster", rb_str_each_grapheme_cluster, 0);
14405
14406 rb_define_method(rb_cString, "sum", rb_str_sum, -1);
14407
14408 rb_define_method(rb_cString, "slice", rb_str_aref_m, -1);
14409 rb_define_method(rb_cString, "slice!", rb_str_slice_bang, -1);
14410
14411 rb_define_method(rb_cString, "partition", rb_str_partition, 1);
14412 rb_define_method(rb_cString, "rpartition", rb_str_rpartition, 1);
14413
14414 rb_define_method(rb_cString, "encoding", rb_obj_encoding, 0); /* in encoding.c */
14415 rb_define_method(rb_cString, "force_encoding", rb_str_force_encoding, 1);
14416 rb_define_method(rb_cString, "b", rb_str_b, 0);
14417
14418 /* define UnicodeNormalize module here so that we don't have to look it up */
14419 mUnicodeNormalize = rb_define_module("UnicodeNormalize");
14420 id_normalize = rb_intern_const("normalize");
14421 id_normalized_p = rb_intern_const("normalized?");
14422
14423 rb_define_method(rb_cString, "unicode_normalize", rb_str_unicode_normalize, -1);
14424 rb_define_method(rb_cString, "unicode_normalize!", rb_str_unicode_normalize_bang, -1);
14425 rb_define_method(rb_cString, "unicode_normalized?", rb_str_unicode_normalized_p, -1);
14426
14427 rb_fs = Qnil;
14428 rb_define_hooked_variable("$;", &rb_fs, 0, rb_fs_setter);
14429 rb_define_hooked_variable("$-F", &rb_fs, 0, rb_fs_setter);
14430 rb_gc_register_address(&rb_fs);
14431
14432 rb_cSymbol = rb_define_class("Symbol", rb_cObject);
14436 rb_define_singleton_method(rb_cSymbol, "all_symbols", sym_all_symbols, 0);
14437
14438 rb_define_method(rb_cSymbol, "==", sym_equal, 1);
14439 rb_define_method(rb_cSymbol, "===", sym_equal, 1);
14440 rb_define_method(rb_cSymbol, "inspect", sym_inspect, 0);
14441 rb_define_method(rb_cSymbol, "to_proc", rb_sym_to_proc, 0); /* in proc.c */
14442 rb_define_method(rb_cSymbol, "succ", sym_succ, 0);
14443 rb_define_method(rb_cSymbol, "next", sym_succ, 0);
14444
14445 rb_define_method(rb_cSymbol, "<=>", sym_cmp, 1);
14446 rb_define_method(rb_cSymbol, "casecmp", sym_casecmp, 1);
14447 rb_define_method(rb_cSymbol, "casecmp?", sym_casecmp_p, 1);
14448 rb_define_method(rb_cSymbol, "=~", sym_match, 1);
14449
14450 rb_define_method(rb_cSymbol, "[]", sym_aref, -1);
14451 rb_define_method(rb_cSymbol, "slice", sym_aref, -1);
14452 rb_define_method(rb_cSymbol, "length", sym_length, 0);
14453 rb_define_method(rb_cSymbol, "size", sym_length, 0);
14454 rb_define_method(rb_cSymbol, "match", sym_match_m, -1);
14455 rb_define_method(rb_cSymbol, "match?", sym_match_m_p, -1);
14456
14457 rb_define_method(rb_cSymbol, "upcase", sym_upcase, -1);
14458 rb_define_method(rb_cSymbol, "downcase", sym_downcase, -1);
14459 rb_define_method(rb_cSymbol, "capitalize", sym_capitalize, -1);
14460 rb_define_method(rb_cSymbol, "swapcase", sym_swapcase, -1);
14461
14462 rb_define_method(rb_cSymbol, "start_with?", sym_start_with, -1);
14463 rb_define_method(rb_cSymbol, "end_with?", sym_end_with, -1);
14464
14465 rb_define_method(rb_cSymbol, "encoding", sym_encoding, 0);
14466}
14467
14468#include "string.rbinc"
#define RUBY_ASSERT_ALWAYS(expr,...)
A variant of RUBY_ASSERT that does not interface with RUBY_DEBUG.
Definition assert.h:199
#define RBIMPL_ASSERT_OR_ASSUME(...)
This is either RUBY_ASSERT or RBIMPL_ASSUME, depending on RUBY_DEBUG.
Definition assert.h:311
#define RUBY_ASSERT_BUILTIN_TYPE(obj, type)
A variant of RUBY_ASSERT that asserts when either RUBY_DEBUG or built-in type of obj is type.
Definition assert.h:291
#define RUBY_ASSERT(...)
Asserts that the given expression is truthy if and only if RUBY_DEBUG is truthy.
Definition assert.h:219
Atomic operations.
@ RUBY_ENC_CODERANGE_7BIT
The object holds 0 to 127 inclusive and nothing else.
Definition coderange.h:39
static enum ruby_coderange_type RB_ENC_CODERANGE_AND(enum ruby_coderange_type a, enum ruby_coderange_type b)
"Mix" two code ranges into one.
Definition coderange.h:162
static int rb_isspace(int c)
Our own locale-insensitive version of isspace(3).
Definition ctype.h:395
static int rb_isascii(int c)
Our own locale-insensitive version of isascii(3).
Definition ctype.h:209
#define rb_define_method(klass, mid, func, arity)
Defines klass#mid.
#define rb_define_singleton_method(klass, mid, func, arity)
Defines klass.mid.
static bool rb_enc_is_newline(const char *p, const char *e, rb_encoding *enc)
Queries if the passed pointer points to a newline character.
Definition ctype.h:43
static bool rb_enc_isprint(OnigCodePoint c, rb_encoding *enc)
Identical to rb_isprint(), except it additionally takes an encoding.
Definition ctype.h:180
static bool rb_enc_isctype(OnigCodePoint c, OnigCtype t, rb_encoding *enc)
Queries if the passed code point is of passed character type in the passed encoding.
Definition ctype.h:63
VALUE rb_enc_sprintf(rb_encoding *enc, const char *fmt,...)
Identical to rb_sprintf(), except it additionally takes an encoding.
Definition sprintf.c:1231
static VALUE RB_OBJ_FROZEN_RAW(VALUE obj)
This is an implementation detail of RB_OBJ_FROZEN().
Definition fl_type.h:699
static VALUE RB_FL_TEST_RAW(VALUE obj, VALUE flags)
This is an implementation detail of RB_FL_TEST().
Definition fl_type.h:407
void rb_include_module(VALUE klass, VALUE module)
Includes a module to a class.
Definition class.c:1769
void rb_define_alias(VALUE klass, const char *name1, const char *name2)
Defines an alias of a method.
Definition class.c:3094
void rb_undef_method(VALUE klass, const char *name)
Defines an undef of a method.
Definition class.c:2897
int rb_scan_args(int argc, const VALUE *argv, const char *fmt,...)
Retrieves argument from argc and argv to given VALUE references according to the format string.
Definition class.c:3384
int rb_block_given_p(void)
Determines if the current method is given a block.
Definition eval.c:1035
int rb_get_kwargs(VALUE keyword_hash, const ID *table, int required, int optional, VALUE *values)
Keyword argument deconstructor.
Definition class.c:3173
#define TYPE(_)
Old name of rb_type.
Definition value_type.h:108
#define ENCODING_SET_INLINED(obj, i)
Old name of RB_ENCODING_SET_INLINED.
Definition encoding.h:106
#define RB_INTEGER_TYPE_P
Old name of rb_integer_type_p.
Definition value_type.h:87
#define ENC_CODERANGE_7BIT
Old name of RUBY_ENC_CODERANGE_7BIT.
Definition coderange.h:180
#define ENC_CODERANGE_VALID
Old name of RUBY_ENC_CODERANGE_VALID.
Definition coderange.h:181
#define FL_UNSET_RAW
Old name of RB_FL_UNSET_RAW.
Definition fl_type.h:130
#define rb_str_buf_cat2
Old name of rb_usascii_str_new_cstr.
Definition string.h:1683
#define ALLOCV
Old name of RB_ALLOCV.
Definition memory.h:404
#define ISSPACE
Old name of rb_isspace.
Definition ctype.h:88
#define T_STRING
Old name of RUBY_T_STRING.
Definition value_type.h:78
#define ENC_CODERANGE_CLEAN_P(cr)
Old name of RB_ENC_CODERANGE_CLEAN_P.
Definition coderange.h:183
#define ENC_CODERANGE_AND(a, b)
Old name of RB_ENC_CODERANGE_AND.
Definition coderange.h:188
#define Qundef
Old name of RUBY_Qundef.
#define INT2FIX
Old name of RB_INT2FIX.
Definition long.h:48
#define OBJ_FROZEN
Old name of RB_OBJ_FROZEN.
Definition fl_type.h:133
#define rb_str_cat2
Old name of rb_str_cat_cstr.
Definition string.h:1684
#define UNREACHABLE
Old name of RBIMPL_UNREACHABLE.
Definition assume.h:28
#define ID2SYM
Old name of RB_ID2SYM.
Definition symbol.h:44
#define T_BIGNUM
Old name of RUBY_T_BIGNUM.
Definition value_type.h:57
#define OBJ_FREEZE
Old name of RB_OBJ_FREEZE.
Definition fl_type.h:131
#define T_FIXNUM
Old name of RUBY_T_FIXNUM.
Definition value_type.h:63
#define UNREACHABLE_RETURN
Old name of RBIMPL_UNREACHABLE_RETURN.
Definition assume.h:29
#define SYM2ID
Old name of RB_SYM2ID.
Definition symbol.h:45
#define ENC_CODERANGE(obj)
Old name of RB_ENC_CODERANGE.
Definition coderange.h:184
#define CLASS_OF
Old name of rb_class_of.
Definition globals.h:205
#define ENC_CODERANGE_UNKNOWN
Old name of RUBY_ENC_CODERANGE_UNKNOWN.
Definition coderange.h:179
#define SIZET2NUM
Old name of RB_SIZE2NUM.
Definition size_t.h:62
#define FIXABLE
Old name of RB_FIXABLE.
Definition fixnum.h:25
#define xmalloc
Old name of ruby_xmalloc.
Definition xmalloc.h:53
#define ENCODING_GET(obj)
Old name of RB_ENCODING_GET.
Definition encoding.h:109
#define LONG2FIX
Old name of RB_INT2FIX.
Definition long.h:49
#define ISDIGIT
Old name of rb_isdigit.
Definition ctype.h:93
#define ENC_CODERANGE_MASK
Old name of RUBY_ENC_CODERANGE_MASK.
Definition coderange.h:178
#define ZALLOC_N
Old name of RB_ZALLOC_N.
Definition memory.h:401
#define T_HASH
Old name of RUBY_T_HASH.
Definition value_type.h:65
#define ALLOC_N
Old name of RB_ALLOC_N.
Definition memory.h:399
#define MBCLEN_CHARFOUND_LEN(ret)
Old name of ONIGENC_MBCLEN_CHARFOUND_LEN.
Definition encoding.h:517
#define FL_TEST_RAW
Old name of RB_FL_TEST_RAW.
Definition fl_type.h:128
#define FL_SET
Old name of RB_FL_SET.
Definition fl_type.h:125
#define rb_ary_new3
Old name of rb_ary_new_from_args.
Definition array.h:658
#define ENCODING_INLINE_MAX
Old name of RUBY_ENCODING_INLINE_MAX.
Definition encoding.h:67
#define LONG2NUM
Old name of RB_LONG2NUM.
Definition long.h:50
#define FL_ANY_RAW
Old name of RB_FL_ANY_RAW.
Definition fl_type.h:122
#define ISALPHA
Old name of rb_isalpha.
Definition ctype.h:92
#define MBCLEN_INVALID_P(ret)
Old name of ONIGENC_MBCLEN_INVALID_P.
Definition encoding.h:518
#define ISASCII
Old name of rb_isascii.
Definition ctype.h:85
#define ULL2NUM
Old name of RB_ULL2NUM.
Definition long_long.h:31
#define TOLOWER
Old name of rb_tolower.
Definition ctype.h:101
#define Qtrue
Old name of RUBY_Qtrue.
#define ST2FIX
Old name of RB_ST2FIX.
Definition st_data_t.h:33
#define MBCLEN_NEEDMORE_P(ret)
Old name of ONIGENC_MBCLEN_NEEDMORE_P.
Definition encoding.h:519
#define FIXNUM_MAX
Old name of RUBY_FIXNUM_MAX.
Definition fixnum.h:26
#define NUM2INT
Old name of RB_NUM2INT.
Definition int.h:44
#define Qnil
Old name of RUBY_Qnil.
#define Qfalse
Old name of RUBY_Qfalse.
#define FIX2LONG
Old name of RB_FIX2LONG.
Definition long.h:46
#define ENC_CODERANGE_BROKEN
Old name of RUBY_ENC_CODERANGE_BROKEN.
Definition coderange.h:182
#define scan_hex(s, l, e)
Old name of ruby_scan_hex.
Definition util.h:108
#define NIL_P
Old name of RB_NIL_P.
#define ALLOCV_N
Old name of RB_ALLOCV_N.
Definition memory.h:405
#define MBCLEN_CHARFOUND_P(ret)
Old name of ONIGENC_MBCLEN_CHARFOUND_P.
Definition encoding.h:516
#define NUM2ULL
Old name of RB_NUM2ULL.
Definition long_long.h:35
#define DBL2NUM
Old name of rb_float_new.
Definition double.h:29
#define ISPRINT
Old name of rb_isprint.
Definition ctype.h:86
#define BUILTIN_TYPE
Old name of RB_BUILTIN_TYPE.
Definition value_type.h:85
#define ENCODING_SHIFT
Old name of RUBY_ENCODING_SHIFT.
Definition encoding.h:68
#define FL_TEST
Old name of RB_FL_TEST.
Definition fl_type.h:127
#define FL_FREEZE
Old name of RUBY_FL_FREEZE.
Definition fl_type.h:65
#define NUM2LONG
Old name of RB_NUM2LONG.
Definition long.h:51
#define ENCODING_GET_INLINED(obj)
Old name of RB_ENCODING_GET_INLINED.
Definition encoding.h:108
#define ENC_CODERANGE_CLEAR(obj)
Old name of RB_ENC_CODERANGE_CLEAR.
Definition coderange.h:187
#define FL_UNSET
Old name of RB_FL_UNSET.
Definition fl_type.h:129
#define UINT2NUM
Old name of RB_UINT2NUM.
Definition int.h:46
#define ENCODING_IS_ASCII8BIT(obj)
Old name of RB_ENCODING_IS_ASCII8BIT.
Definition encoding.h:110
#define FIXNUM_P
Old name of RB_FIXNUM_P.
#define CONST_ID
Old name of RUBY_CONST_ID.
Definition symbol.h:47
#define rb_ary_new2
Old name of rb_ary_new_capa.
Definition array.h:657
#define ENC_CODERANGE_SET(obj, cr)
Old name of RB_ENC_CODERANGE_SET.
Definition coderange.h:186
#define ENCODING_CODERANGE_SET(obj, encindex, cr)
Old name of RB_ENCODING_CODERANGE_SET.
Definition coderange.h:189
#define FL_SET_RAW
Old name of RB_FL_SET_RAW.
Definition fl_type.h:126
#define SYMBOL_P
Old name of RB_SYMBOL_P.
Definition value_type.h:88
#define OBJ_FROZEN_RAW
Old name of RB_OBJ_FROZEN_RAW.
Definition fl_type.h:134
#define T_REGEXP
Old name of RUBY_T_REGEXP.
Definition value_type.h:77
#define ENCODING_MASK
Old name of RUBY_ENCODING_MASK.
Definition encoding.h:69
void rb_category_warn(rb_warning_category_t category, const char *fmt,...)
Identical to rb_category_warning(), except it reports unless $VERBOSE is nil.
Definition error.c:478
void rb_exc_raise(VALUE mesg)
Raises an exception in the current thread.
Definition eval.c:678
void rb_syserr_fail(int e, const char *mesg)
Raises appropriate exception that represents a C errno.
Definition error.c:4084
VALUE rb_eRangeError
RangeError exception.
Definition error.c:1477
VALUE rb_eTypeError
TypeError exception.
Definition error.c:1473
VALUE rb_eEncCompatError
Encoding::CompatibilityError exception.
Definition error.c:1480
VALUE rb_eRuntimeError
RuntimeError exception.
Definition error.c:1471
VALUE rb_eIndexError
IndexError exception.
Definition error.c:1475
@ RB_WARN_CATEGORY_DEPRECATED
Warning is for deprecated features.
Definition error.h:48
VALUE rb_cObject
Object class.
Definition object.c:60
VALUE rb_any_to_s(VALUE obj)
Generates a textual representation of the given object.
Definition object.c:658
VALUE rb_obj_alloc(VALUE klass)
Allocates an instance of the given class.
Definition object.c:2252
VALUE rb_obj_hide(VALUE obj)
Make the object invisible from Ruby code.
Definition object.c:94
VALUE rb_class_new_instance_pass_kw(int argc, const VALUE *argv, VALUE klass)
Identical to rb_class_new_instance(), except it passes the passed keywords if any to the #initialize ...
Definition object.c:2270
VALUE rb_obj_frozen_p(VALUE obj)
Same as RB_OBJ_FROZEN(), but returns Qtrue/Qfalse instead of #bool.
Definition object.c:1316
double rb_str_to_dbl(VALUE str, int mode)
Identical to rb_cstr_to_dbl(), except it accepts a Ruby's string instead of C's.
Definition object.c:3642
VALUE rb_obj_class(VALUE obj)
Queries the class of an object.
Definition object.c:234
VALUE rb_obj_dup(VALUE obj)
Duplicates the given object.
Definition object.c:556
VALUE rb_cSymbol
Symbol class.
Definition string.c:86
VALUE rb_cRange
Range class.
Definition range.c:35
VALUE rb_equal(VALUE lhs, VALUE rhs)
This function is an optimised version of calling #==.
Definition object.c:140
VALUE rb_obj_is_kind_of(VALUE obj, VALUE klass)
Queries if the given object is an instance (of possibly descendants) of the given class.
Definition object.c:906
VALUE rb_obj_freeze(VALUE obj)
Same as RB_OBJ_FREEZE(), but returns the given object.
Definition object.c:1309
VALUE rb_mComparable
Comparable module.
Definition compar.c:19
VALUE rb_cString
String class.
Definition string.c:85
VALUE rb_to_int(VALUE val)
Identical to rb_check_to_int(), except it raises in case of conversion mismatch.
Definition object.c:3328
Encoding relates APIs.
static char * rb_enc_left_char_head(const char *s, const char *p, const char *e, rb_encoding *enc)
Queries the left boundary of a character.
Definition encoding.h:683
static char * rb_enc_right_char_head(const char *s, const char *p, const char *e, rb_encoding *enc)
Queries the right boundary of a character.
Definition encoding.h:704
static unsigned int rb_enc_codepoint(const char *p, const char *e, rb_encoding *enc)
Queries the code point of character pointed by the passed pointer.
Definition encoding.h:571
static int rb_enc_mbmaxlen(rb_encoding *enc)
Queries the maximum number of bytes that the passed encoding needs to represent a character.
Definition encoding.h:447
static int RB_ENCODING_GET_INLINED(VALUE obj)
Queries the encoding of the passed object.
Definition encoding.h:99
static int rb_enc_code_to_mbclen(int c, rb_encoding *enc)
Identical to rb_enc_codelen(), except it returns 0 for invalid code points.
Definition encoding.h:619
static char * rb_enc_step_back(const char *s, const char *p, const char *e, int n, rb_encoding *enc)
Scans the string backwards for n characters.
Definition encoding.h:726
VALUE rb_str_conv_enc(VALUE str, rb_encoding *from, rb_encoding *to)
Encoding conversion main routine.
Definition string.c:1379
VALUE rb_enc_str_new_static(const char *ptr, long len, rb_encoding *enc)
Identical to rb_enc_str_new(), except it takes a C string literal.
Definition string.c:1244
char * rb_enc_nth(const char *head, const char *tail, long nth, rb_encoding *enc)
Queries the n-th character.
Definition string.c:3116
VALUE rb_str_conv_enc_opts(VALUE str, rb_encoding *from, rb_encoding *to, int ecflags, VALUE ecopts)
Identical to rb_str_conv_enc(), except it additionally takes IO encoder options.
Definition string.c:1263
VALUE rb_enc_interned_str(const char *ptr, long len, rb_encoding *enc)
Identical to rb_enc_str_new(), except it returns a "f"string.
Definition string.c:14192
long rb_memsearch(const void *x, long m, const void *y, long n, rb_encoding *enc)
Looks for the passed string in the passed buffer.
Definition re.c:285
long rb_enc_strlen(const char *head, const char *tail, rb_encoding *enc)
Counts the number of characters of the passed string, according to the passed encoding.
Definition string.c:2398
VALUE rb_enc_str_buf_cat(VALUE str, const char *ptr, long len, rb_encoding *enc)
Identical to rb_str_cat(), except it additionally takes an encoding.
Definition string.c:3841
VALUE rb_enc_str_new_cstr(const char *ptr, rb_encoding *enc)
Identical to rb_enc_str_new(), except it assumes the passed pointer is a pointer to a C string.
Definition string.c:1175
VALUE rb_str_export_to_enc(VALUE obj, rb_encoding *enc)
Identical to rb_str_export(), except it additionally takes an encoding.
Definition string.c:1484
VALUE rb_external_str_new_with_enc(const char *ptr, long len, rb_encoding *enc)
Identical to rb_external_str_new(), except it additionally takes an encoding.
Definition string.c:1385
int rb_enc_str_asciionly_p(VALUE str)
Queries if the passed string is "ASCII only".
Definition string.c:988
VALUE rb_enc_interned_str_cstr(const char *ptr, rb_encoding *enc)
Identical to rb_enc_str_new_cstr(), except it returns a "f"string.
Definition string.c:14216
long rb_str_coderange_scan_restartable(const char *str, const char *end, rb_encoding *enc, int *cr)
Scans the passed string until it finds something odd.
Definition string.c:844
int rb_enc_symname2_p(const char *name, long len, rb_encoding *enc)
Identical to rb_enc_symname_p(), except it additionally takes the passed string's length.
Definition symbol.c:858
rb_econv_result_t rb_econv_convert(rb_econv_t *ec, const unsigned char **source_buffer_ptr, const unsigned char *source_buffer_end, unsigned char **destination_buffer_ptr, unsigned char *destination_buffer_end, int flags)
Converts a string from an encoding to another.
Definition transcode.c:1487
rb_econv_result_t
return value of rb_econv_convert()
Definition transcode.h:30
@ econv_finished
The conversion stopped after converting everything.
Definition transcode.h:57
@ econv_destination_buffer_full
The conversion stopped because there is no destination.
Definition transcode.h:46
rb_econv_t * rb_econv_open_opts(const char *source_encoding, const char *destination_encoding, int ecflags, VALUE ecopts)
Identical to rb_econv_open(), except it additionally takes a hash of optional strings.
Definition transcode.c:2730
VALUE rb_str_encode(VALUE str, VALUE to, int ecflags, VALUE ecopts)
Converts the contents of the passed string from its encoding to the passed one.
Definition transcode.c:2993
void rb_econv_close(rb_econv_t *ec)
Destructs a converter.
Definition transcode.c:1744
VALUE rb_funcall(VALUE recv, ID mid, int n,...)
Calls a method.
Definition vm_eval.c:1123
VALUE rb_funcallv(VALUE recv, ID mid, int argc, const VALUE *argv)
Identical to rb_funcall(), except it takes the method arguments as a C array.
Definition vm_eval.c:1081
VALUE rb_funcall_with_block_kw(VALUE recv, ID mid, int argc, const VALUE *argv, VALUE procval, int kw_splat)
Identical to rb_funcallv_with_block(), except you can specify how to handle the last element of the g...
Definition vm_eval.c:1210
VALUE rb_check_array_type(VALUE obj)
Try converting an object to its array representation using its to_ary method, if any.
VALUE rb_ary_new(void)
Allocates a new, empty array.
VALUE rb_ary_new_capa(long capa)
Identical to rb_ary_new(), except it additionally specifies how many rooms of objects it should alloc...
VALUE rb_ary_push(VALUE ary, VALUE elem)
Special case of rb_ary_cat() that it adds only one element.
VALUE rb_ary_freeze(VALUE obj)
Freeze an array, preventing further modifications.
#define RETURN_SIZED_ENUMERATOR(obj, argc, argv, size_fn)
This roughly resembles return enum_for(__callee__) unless block_given?.
Definition enumerator.h:208
#define RETURN_ENUMERATOR(obj, argc, argv)
Identical to RETURN_SIZED_ENUMERATOR(), except its size is unknown.
Definition enumerator.h:242
#define UNLIMITED_ARGUMENTS
This macro is used in conjunction with rb_check_arity().
Definition error.h:35
static int rb_check_arity(int argc, int min, int max)
Ensures that the passed integer is in the passed range.
Definition error.h:284
VALUE rb_fs
The field separator character for inputs, or the $;.
Definition string.c:723
VALUE rb_default_rs
This is the default value of rb_rs, i.e.
Definition io.c:209
VALUE rb_backref_get(void)
Queries the last match, or Regexp.last_match, or the $~.
Definition vm.c:2131
VALUE rb_sym_all_symbols(void)
Collects every single bits of symbols that have ever interned in the entire history of the current pr...
Definition symbol.c:1215
void rb_backref_set(VALUE md)
Updates $~.
Definition vm.c:2137
int rb_range_values(VALUE range, VALUE *begp, VALUE *endp, int *exclp)
Deconstructs a range into its components.
Definition range.c:1857
VALUE rb_range_beg_len(VALUE range, long *begp, long *lenp, long len, int err)
Deconstructs a numerical range.
Definition range.c:1945
int rb_reg_backref_number(VALUE match, VALUE backref)
Queries the index of the given named capture.
Definition re.c:1387
int rb_reg_options(VALUE re)
Queries the options of the passed regular expression.
Definition re.c:4476
VALUE rb_reg_match(VALUE re, VALUE str)
This is the match operator.
Definition re.c:3970
void rb_match_busy(VALUE md)
Asserts that the given MatchData is "occupied".
Definition re.c:1631
VALUE rb_reg_nth_match(int n, VALUE md)
Queries the nth captured substring.
Definition re.c:2071
void rb_str_free(VALUE str)
Destroys the given string for no reason.
Definition string.c:1797
VALUE rb_str_new_shared(VALUE str)
Identical to rb_str_new_cstr(), except it takes a Ruby's string instead of C's.
Definition string.c:1549
VALUE rb_str_plus(VALUE lhs, VALUE rhs)
Generates a new string, concatenating the former to the latter.
Definition string.c:2549
#define rb_utf8_str_new_cstr(str)
Identical to rb_str_new_cstr, except it generates a string of "UTF-8" encoding.
Definition string.h:1584
#define rb_hash_end(h)
Just another name of st_hash_end.
Definition string.h:946
#define rb_hash_uint32(h, i)
Just another name of st_hash_uint32.
Definition string.h:940
VALUE rb_str_append(VALUE dst, VALUE src)
Identical to rb_str_buf_append(), except it converts the right hand side before concatenating.
Definition string.c:3906
VALUE rb_filesystem_str_new(const char *ptr, long len)
Identical to rb_str_new(), except it generates a string of "filesystem" encoding.
Definition string.c:1460
VALUE rb_sym_to_s(VALUE sym)
This is an rb_sym2str() + rb_str_dup() combo.
Definition string.c:13824
VALUE rb_str_times(VALUE str, VALUE num)
Repetition of a string.
Definition string.c:2623
VALUE rb_external_str_new(const char *ptr, long len)
Identical to rb_str_new(), except it generates a string of "default external" encoding.
Definition string.c:1436
VALUE rb_str_tmp_new(long len)
Allocates a "temporary" string.
Definition string.c:1791
long rb_str_offset(VALUE str, long pos)
"Inverse" of rb_str_sublen().
Definition string.c:3144
VALUE rb_str_succ(VALUE orig)
Searches for the "successor" of a string.
Definition string.c:5450
int rb_str_hash_cmp(VALUE str1, VALUE str2)
Compares two strings.
Definition string.c:4269
VALUE rb_str_subseq(VALUE str, long beg, long len)
Identical to rb_str_substr(), except the numbers are interpreted as byte offsets instead of character...
Definition string.c:3259
VALUE rb_str_ellipsize(VALUE str, long len)
Shortens str and adds three dots, an ellipsis, if it is longer than len characters.
Definition string.c:13139
st_index_t rb_memhash(const void *ptr, long len)
This is a universal hash function.
Definition random.c:1720
#define rb_str_new(str, len)
Allocates an instance of rb_cString.
Definition string.h:1499
void rb_str_shared_replace(VALUE dst, VALUE src)
Replaces the contents of the former with the latter.
Definition string.c:1833
#define rb_str_buf_cat
Just another name of rb_str_cat.
Definition string.h:1682
VALUE rb_str_new_static(const char *ptr, long len)
Identical to rb_str_new(), except it takes a C string literal.
Definition string.c:1209
#define rb_usascii_str_new(str, len)
Identical to rb_str_new, except it generates a string of "US ASCII" encoding.
Definition string.h:1533
size_t rb_str_capacity(VALUE str)
Queries the capacity of the given string.
Definition string.c:1023
VALUE rb_str_new_frozen(VALUE str)
Creates a frozen copy of the string, if necessary.
Definition string.c:1555
VALUE rb_str_dup(VALUE str)
Duplicates a string.
Definition string.c:2031
st_index_t rb_str_hash(VALUE str)
Calculates a hash value of a string.
Definition string.c:4255
VALUE rb_str_cat(VALUE dst, const char *src, long srclen)
Destructively appends the passed contents to the string.
Definition string.c:3674
VALUE rb_str_locktmp(VALUE str)
Obtains a "temporary lock" of the string.
long rb_str_strlen(VALUE str)
Counts the number of characters (not bytes) that are stored inside of the given string.
Definition string.c:2485
VALUE rb_str_resurrect(VALUE str)
Like rb_str_dup(), but always create an instance of rb_cString regardless of the given object's class...
Definition string.c:2049
#define rb_str_buf_new_cstr(str)
Identical to rb_str_new_cstr, except done differently.
Definition string.h:1640
#define rb_usascii_str_new_cstr(str)
Identical to rb_str_new_cstr, except it generates a string of "US ASCII" encoding.
Definition string.h:1568
VALUE rb_str_replace(VALUE dst, VALUE src)
Replaces the contents of the former object with the stringised contents of the latter.
Definition string.c:6669
char * rb_str_subpos(VALUE str, long beg, long *len)
Identical to rb_str_substr(), except it returns a C's string instead of Ruby's.
Definition string.c:3267
rb_gvar_setter_t rb_str_setter
This is a rb_gvar_setter_t that refutes non-string assignments.
Definition string.h:1147
VALUE rb_interned_str_cstr(const char *ptr)
Identical to rb_interned_str(), except it assumes the passed pointer is a pointer to a C's string.
Definition string.c:14186
VALUE rb_filesystem_str_new_cstr(const char *ptr)
Identical to rb_filesystem_str_new(), except it assumes the passed pointer is a pointer to a C string...
Definition string.c:1466
#define rb_external_str_new_cstr(str)
Identical to rb_str_new_cstr, except it generates a string of "default external" encoding.
Definition string.h:1605
VALUE rb_str_buf_append(VALUE dst, VALUE src)
Identical to rb_str_cat_cstr(), except it takes Ruby's string instead of C's.
Definition string.c:3872
long rb_str_sublen(VALUE str, long pos)
Byte offset to character offset conversion.
Definition string.c:3191
VALUE rb_str_equal(VALUE str1, VALUE str2)
Equality of two strings.
Definition string.c:4376
void rb_str_set_len(VALUE str, long len)
Overwrites the length of the string.
Definition string.c:3493
VALUE rb_str_inspect(VALUE str)
Generates a "readable" version of the receiver.
Definition string.c:8145
void rb_must_asciicompat(VALUE obj)
Asserts that the given string's encoding is (Ruby's definition of) ASCII compatible.
Definition string.c:2855
VALUE rb_interned_str(const char *ptr, long len)
Identical to rb_str_new(), except it returns an infamous "f"string.
Definition string.c:14171
int rb_str_cmp(VALUE lhs, VALUE rhs)
Compares two strings, as in strcmp(3).
Definition string.c:4323
VALUE rb_str_concat(VALUE dst, VALUE src)
Identical to rb_str_append(), except it also accepts an integer as a codepoint.
Definition string.c:4143
int rb_str_comparable(VALUE str1, VALUE str2)
Checks if two strings are comparable each other or not.
Definition string.c:4298
#define rb_strlen_lit(str)
Length of a string literal.
Definition string.h:1693
VALUE rb_str_buf_cat_ascii(VALUE dst, const char *src)
Identical to rb_str_cat_cstr(), except it additionally assumes the source string be a NUL terminated ...
Definition string.c:3848
VALUE rb_str_freeze(VALUE str)
This is the implementation of String#freeze.
Definition string.c:3384
void rb_str_update(VALUE dst, long beg, long len, VALUE src)
Replaces some (or all) of the contents of the given string.
Definition string.c:5937
VALUE rb_str_scrub(VALUE str, VALUE repl)
"Cleanses" the string.
Definition string.c:13197
#define rb_locale_str_new_cstr(str)
Identical to rb_external_str_new_cstr, except it generates a string of "locale" encoding instead of "...
Definition string.h:1626
VALUE rb_str_new_with_class(VALUE obj, const char *ptr, long len)
Identical to rb_str_new(), except it takes the class of the allocating object.
Definition string.c:1747
#define rb_str_dup_frozen
Just another name of rb_str_new_frozen.
Definition string.h:632
VALUE rb_check_string_type(VALUE obj)
Try converting an object to its stringised representation using its to_str method,...
Definition string.c:3040
VALUE rb_str_substr(VALUE str, long beg, long len)
This is the implementation of two-argumented String#slice.
Definition string.c:3356
#define rb_str_cat_cstr(buf, str)
Identical to rb_str_cat(), except it assumes the passed pointer is a pointer to a C string.
Definition string.h:1657
VALUE rb_str_unlocktmp(VALUE str)
Releases a lock formerly obtained by rb_str_locktmp().
Definition string.c:3475
VALUE rb_utf8_str_new_static(const char *ptr, long len)
Identical to rb_str_new_static(), except it generates a string of "UTF-8" encoding instead of "binary...
Definition string.c:1238
#define rb_utf8_str_new(str, len)
Identical to rb_str_new, except it generates a string of "UTF-8" encoding.
Definition string.h:1550
void rb_str_modify_expand(VALUE str, long capa)
Identical to rb_str_modify(), except it additionally expands the capacity of the receiver.
Definition string.c:2809
VALUE rb_str_dump(VALUE str)
"Inverse" of rb_eval_string().
Definition string.c:8262
VALUE rb_locale_str_new(const char *ptr, long len)
Identical to rb_str_new(), except it generates a string of "locale" encoding.
Definition string.c:1448
VALUE rb_str_buf_new(long capa)
Allocates a "string buffer".
Definition string.c:1763
VALUE rb_str_length(VALUE)
Identical to rb_str_strlen(), except it returns the value in rb_cInteger.
Definition string.c:2499
#define rb_str_new_cstr(str)
Identical to rb_str_new, except it assumes the passed pointer is a pointer to a C string.
Definition string.h:1515
VALUE rb_str_drop_bytes(VALUE str, long len)
Shrinks the given string for the given number of bytes.
Definition string.c:5852
VALUE rb_str_split(VALUE str, const char *delim)
Divides the given string based on the given delimiter.
Definition string.c:10795
VALUE rb_usascii_str_new_static(const char *ptr, long len)
Identical to rb_str_new_static(), except it generates a string of "US ASCII" encoding instead of "bin...
Definition string.c:1232
VALUE rb_str_intern(VALUE str)
Identical to rb_to_symbol(), except it assumes the receiver being an instance of RString.
Definition symbol.c:1085
VALUE rb_obj_as_string(VALUE obj)
Try converting an object to its stringised representation using its to_s method, if any.
Definition string.c:1895
VALUE rb_ivar_set(VALUE obj, ID name, VALUE val)
Identical to rb_iv_set(), except it accepts the name as an ID instead of a C string.
Definition variable.c:2141
VALUE rb_ivar_defined(VALUE obj, ID name)
Queries if the instance variable is defined at the object.
Definition variable.c:2201
int rb_respond_to(VALUE obj, ID mid)
Queries if the object responds to the method.
Definition vm_method.c:3683
void rb_undef_alloc_func(VALUE klass)
Deletes the allocator function of a class.
Definition vm_method.c:1843
void rb_define_alloc_func(VALUE klass, rb_alloc_func_t func)
Sets the allocator function of a class.
static ID rb_intern_const(const char *str)
This is a "tiny optimisation" over rb_intern().
Definition symbol.h:285
VALUE rb_sym2str(VALUE symbol)
Obtain a frozen string representation of a symbol (not including the leading colon).
Definition symbol.c:1148
VALUE rb_to_symbol(VALUE name)
Identical to rb_intern_str(), except it generates a dynamic symbol if necessary.
Definition string.c:14138
ID rb_to_id(VALUE str)
Identical to rb_intern_str(), except it tries to convert the parameter object to an instance of rb_cS...
Definition string.c:14128
int capa
Designed capacity of the buffer.
Definition io.h:11
int off
Offset inside of ptr.
Definition io.h:5
int len
Length of the buffer.
Definition io.h:8
#define RB_OBJ_SET_SHAREABLE(obj)
Wrapper of rb_obj_set_shareable().
Definition ractor.h:290
#define RB_OBJ_SHAREABLE_P(obj)
Queries if the passed object has previously classified as shareable or not.
Definition ractor.h:255
long rb_reg_search(VALUE re, VALUE str, long pos, int dir)
Runs the passed regular expression over the passed string.
Definition re.c:2000
VALUE rb_reg_regcomp(VALUE str)
Creates a new instance of rb_cRegexp.
Definition re.c:3675
VALUE rb_str_format(int argc, const VALUE *argv, VALUE fmt)
Formats a string.
Definition sprintf.c:974
VALUE rb_yield(VALUE val)
Yields the block.
Definition vm_eval.c:1378
#define MEMCPY(p1, p2, type, n)
Handy macro to call memcpy.
Definition memory.h:372
#define ALLOCA_N(type, n)
Definition memory.h:292
#define MEMZERO(p, type, n)
Handy macro to erase a region of memory.
Definition memory.h:360
#define RB_GC_GUARD(v)
Prevents premature destruction of local objects.
Definition memory.h:167
void rb_define_hooked_variable(const char *q, VALUE *w, type *e, void_type *r)
Define a function-backended global variable.
VALUE type(ANYARGS)
ANYARGS-ed function type.
void rb_hash_foreach(VALUE q, int_type *w, VALUE e)
Iteration over the given hash.
VALUE rb_ensure(type *q, VALUE w, type *e, VALUE r)
An equivalent of ensure clause.
Defines RBIMPL_ATTR_NONSTRING.
static int RARRAY_LENINT(VALUE ary)
Identical to rb_array_len(), except it differs for the return type.
Definition rarray.h:280
#define RARRAY_CONST_PTR
Just another name of rb_array_const_ptr.
Definition rarray.h:51
static VALUE RBASIC_CLASS(VALUE obj)
Queries the class of an object.
Definition rbasic.h:166
#define RBASIC(obj)
Convenient casting macro.
Definition rbasic.h:40
#define RHASH_SIZE(h)
Queries the size of the hash.
Definition rhash.h:69
static VALUE RREGEXP_SRC(VALUE rexp)
Convenient getter function.
Definition rregexp.h:102
#define StringValue(v)
Ensures that the parameter object is a String.
Definition rstring.h:66
VALUE rb_str_export_locale(VALUE obj)
Identical to rb_str_export(), except it converts into the locale encoding instead.
Definition string.c:1478
char * rb_string_value_cstr(volatile VALUE *ptr)
Identical to rb_string_value_ptr(), except it additionally checks for the contents for viability as a...
Definition string.c:3011
static int RSTRING_LENINT(VALUE str)
Identical to RSTRING_LEN(), except it differs for the return type.
Definition rstring.h:438
static char * RSTRING_END(VALUE str)
Queries the end of the contents pointer of the string.
Definition rstring.h:409
#define RSTRING_GETMEM(str, ptrvar, lenvar)
Convenient macro to obtain the contents and length at once.
Definition rstring.h:450
VALUE rb_string_value(volatile VALUE *ptr)
Identical to rb_str_to_str(), except it fills the passed pointer with the converted object.
Definition string.c:2874
#define RSTRING(obj)
Convenient casting macro.
Definition rstring.h:41
VALUE rb_str_export(VALUE obj)
Identical to rb_str_to_str(), except it additionally converts the string into default external encodi...
Definition string.c:1472
char * rb_string_value_ptr(volatile VALUE *ptr)
Identical to rb_str_to_str(), except it returns the converted string's backend memory region.
Definition string.c:2887
VALUE rb_str_to_str(VALUE obj)
Identical to rb_check_string_type(), except it raises exceptions in case of conversion failures.
Definition string.c:1824
#define StringValueCStr(v)
Identical to StringValuePtr, except it additionally checks for the contents for viability as a C stri...
Definition rstring.h:89
#define DATA_PTR(obj)
Convenient casting macro for backward compatibility.
Definition rtypeddata.h:439
#define TypedData_Wrap_Struct(klass, data_type, sval)
Converts sval, a pointer to your struct, into a Ruby object.
Definition rtypeddata.h:557
VALUE rb_require(const char *feature)
Identical to rb_require_string(), except it takes C's string instead of Ruby's.
Definition load.c:1528
#define errno
Ractor-aware version of errno.
Definition ruby.h:388
#define RB_NUM2SSIZE
Converts an instance of rb_cInteger into C's ssize_t.
Definition size_t.h:49
#define RTEST
This is an old name of RB_TEST.
#define _(args)
This was a transition path from K&R to ANSI.
Definition stdarg.h:35
VALUE flags
Per-object flags.
Definition rbasic.h:81
Ruby's String.
Definition rstring.h:196
struct RBasic basic
Basic part, including flags and class.
Definition rstring.h:199
union RString::@60::@61::@63 aux
Auxiliary info.
long capa
Capacity of *ptr.
Definition rstring.h:232
long len
Length of the string, not including terminating NUL character.
Definition rstring.h:206
struct RString::@60::@61 heap
Strings that use separated memory region for contents use this pattern.
struct RString::@60::@62 embed
Embedded contents.
VALUE shared
Parent of the string.
Definition rstring.h:240
char * ptr
Pointer to the contents of the string.
Definition rstring.h:222
union RString::@60 as
String's specific fields.
This is the struct that holds necessary info for a struct.
Definition rtypeddata.h:242
Definition string.c:9186
void rb_nativethread_lock_lock(rb_nativethread_lock_t *lock)
Blocks until the current thread obtains a lock.
Definition thread.c:319
uintptr_t ID
Type that represents a Ruby identifier such as a variable name.
Definition value.h:52
uintptr_t VALUE
Type that represents a Ruby object.
Definition value.h:40
static enum ruby_value_type rb_type(VALUE obj)
Identical to RB_BUILTIN_TYPE(), except it can also accept special constants.
Definition value_type.h:225
static void Check_Type(VALUE v, enum ruby_value_type t)
Identical to RB_TYPE_P(), except it raises exceptions on predication failure.
Definition value_type.h:425
static bool RB_TYPE_P(VALUE obj, enum ruby_value_type t)
Queries if the given object is of given type.
Definition value_type.h:376
ruby_value_type
C-level type of an object.
Definition value_type.h:113