Ruby 4.1.0dev (2026-10-02 revision 6930c20b65d98b01d480b45788799c2878b4bcd0)
string.c (6930c20b65d98b01d480b45788799c2878b4bcd0)
1/**********************************************************************
2
3 string.c -
4
5 $Author$
6 created at: Mon Aug 9 17:12:58 JST 1993
7
8 Copyright (C) 1993-2007 Yukihiro Matsumoto
9 Copyright (C) 2000 Network Applied Communication Laboratory, Inc.
10 Copyright (C) 2000 Information-technology Promotion Agency, Japan
11
12**********************************************************************/
13
14#include "ruby/internal/config.h"
15
16#include <ctype.h>
17#include <errno.h>
18#include <math.h>
19
20#ifdef HAVE_UNISTD_H
21# include <unistd.h>
22#endif
23
24#include "debug_counter.h"
25#include "encindex.h"
26#include "id.h"
27#include "internal.h"
28#include "internal/array.h"
29#include "internal/bits.h"
30#include "internal/compar.h"
31#include "internal/compilers.h"
32#include "internal/concurrent_set.h"
33#include "internal/encoding.h"
34#include "internal/error.h"
35#include "internal/gc.h"
36#include "internal/hash.h"
37#include "internal/numeric.h"
38#include "internal/object.h"
39#include "internal/proc.h"
40#include "internal/re.h"
41#include "internal/sanitizers.h"
42#include "internal/simd.h"
43#include "internal/string.h"
44#include "internal/transcode.h"
45#include "probes.h"
46#include "ruby/encoding.h"
47#include "ruby/re.h"
48#include "ruby/thread.h"
49#include "ruby/util.h"
50#include "ruby/ractor.h"
51#include "ruby_assert.h"
52#include "shape.h"
53#include "vm_core.h"
54#include "vm_sync.h"
55#include "zjit.h"
57
58#if defined HAVE_CRYPT_R
59# if defined HAVE_CRYPT_H
60# include <crypt.h>
61# endif
62#elif !defined HAVE_CRYPT
63# include "missing/crypt.h"
64# define HAVE_CRYPT_R 1
65#endif
66
67#undef rb_str_new
68#undef rb_usascii_str_new
69#undef rb_utf8_str_new
70#undef rb_enc_str_new
71#undef rb_str_new_cstr
72#undef rb_usascii_str_new_cstr
73#undef rb_utf8_str_new_cstr
74#undef rb_enc_str_new_cstr
75#undef rb_external_str_new_cstr
76#undef rb_locale_str_new_cstr
77#undef rb_str_dup_frozen
78#undef rb_str_buf_new_cstr
79#undef rb_str_buf_cat
80#undef rb_str_buf_cat2
81#undef rb_str_cat2
82#undef rb_str_cat_cstr
83#undef rb_fstring_cstr
84
87
88/* Flags of RString
89 *
90 * 0: STR_SHARED (equal to ELTS_SHARED)
91 * The string is shared. The buffer this string points to is owned by
92 * another string (the shared root).
93 * 1: RSTRING_NOEMBED
94 * The string is not embedded. When a string is embedded, the contents
95 * follow the header. When a string is not embedded, the contents is
96 * on a separately allocated buffer.
97 * 2: STR_CHILLED (will be frozen in a future version)
98 * The string was allocated as a literal in a file without an explicit `frozen_string_literal` comment.
99 * It emits a deprecation warning when mutated for the first time.
100 * 4: STR_PRECOMPUTED_HASH
101 * The string is embedded and has its precomputed hashcode stored
102 * after the terminator.
103 * 5: STR_SHARED_ROOT
104 * Other strings may point to the contents of this string. When this
105 * flag is set, STR_SHARED must not be set.
106 * 6: STR_BORROWED
107 * When RSTRING_NOEMBED is set and klass is 0, this string is unsafe
108 * to be unshared by rb_str_tmp_frozen_release.
109 * 7: STR_TMPLOCK
110 * The pointer to the buffer is passed to a system call such as
111 * read(2). Any modification and realloc is prohibited.
112 * 8-9: ENC_CODERANGE
113 * Stores the coderange of the string.
114 * 10-16: ENCODING
115 * Stores the encoding of the string.
116 * 17: RSTRING_FSTR
117 * The string is a fstring. The string is deduplicated in the fstring
118 * table.
119 * 18: STR_NOFREE
120 * Do not free this string's buffer when the string is reclaimed
121 * by the garbage collector. Used for when the string buffer is a C
122 * string literal.
123 * 19: STR_FAKESTR
124 * The string is not allocated or managed by the garbage collector.
125 * Typically, the string object header (struct RString) is temporarily
126 * allocated on C stack.
127 */
128
129#define RUBY_MAX_CHAR_LEN 16
130#define STR_PRECOMPUTED_HASH FL_USER4
131#define STR_SHARED_ROOT FL_USER5
132#define STR_BORROWED FL_USER6
133#define STR_TMPLOCK FL_USER7
134#define STR_NOFREE FL_USER18
135
136#define STR_SET_NOEMBED(str) do {\
137 FL_SET((str), STR_NOEMBED);\
138 FL_UNSET((str), STR_SHARED | STR_SHARED_ROOT | STR_BORROWED);\
139} while (0)
140#define STR_SET_EMBED(str) FL_UNSET((str), STR_NOEMBED | STR_SHARED | STR_NOFREE)
141
142#define STR_SET_LEN(str, n) do { \
143 RSTRING(str)->len = (n); \
144} while (0)
145
146#define TERM_LEN(str) (rb_str_enc_fastpath(str) ? 1 : rb_enc_mbminlen(rb_enc_from_index(ENCODING_GET(str))))
147#define TERM_FILL(ptr, termlen) do {\
148 char *const term_fill_ptr = (ptr);\
149 const int term_fill_len = (termlen);\
150 *term_fill_ptr = '\0';\
151 if (UNLIKELY(term_fill_len > 1))\
152 memset(term_fill_ptr, 0, term_fill_len);\
153} while (0)
154
155#define RESIZE_CAPA(str,capacity) do {\
156 const int termlen = TERM_LEN(str);\
157 RESIZE_CAPA_TERM(str,capacity,termlen);\
158} while (0)
159#define RESIZE_CAPA_TERM(str,capacity,termlen) do {\
160 if (STR_EMBED_P(str)) {\
161 if (str_embed_capa(str) < capacity + termlen) {\
162 char *const tmp = ALLOC_N(char, (size_t)(capacity) + (termlen));\
163 const long tlen = RSTRING_LEN(str);\
164 memcpy(tmp, RSTRING_PTR(str), str_embed_capa(str));\
165 RSTRING(str)->as.heap.ptr = tmp;\
166 RSTRING(str)->len = tlen;\
167 STR_SET_NOEMBED(str);\
168 RSTRING(str)->as.heap.aux.capa = (capacity);\
169 }\
170 }\
171 else {\
172 RUBY_ASSERT(!FL_TEST((str), STR_SHARED)); \
173 SIZED_REALLOC_N(RSTRING(str)->as.heap.ptr, char, \
174 (size_t)(capacity) + (termlen), STR_HEAP_SIZE(str)); \
175 RSTRING(str)->as.heap.aux.capa = (capacity);\
176 }\
177} while (0)
178
179#define STR_SET_SHARED(str, shared_str) do { \
180 if (!FL_TEST(str, STR_FAKESTR)) { \
181 RUBY_ASSERT(RSTRING_PTR(shared_str) <= RSTRING_PTR(str)); \
182 RUBY_ASSERT(RSTRING_PTR(str) <= RSTRING_PTR(shared_str) + RSTRING_LEN(shared_str)); \
183 RB_OBJ_WRITE((str), &RSTRING(str)->as.heap.aux.shared, (shared_str)); \
184 FL_SET((str), STR_SHARED); \
185 rb_gc_register_pinning_obj(str); \
186 FL_SET((shared_str), STR_SHARED_ROOT); \
187 if (RBASIC_CLASS((shared_str)) == 0) /* for CoW-friendliness */ \
188 FL_SET_RAW((shared_str), STR_BORROWED); \
189 } \
190} while (0)
191
192#define STR_HEAP_PTR(str) (RSTRING(str)->as.heap.ptr)
193#define STR_HEAP_SIZE(str) ((size_t)RSTRING(str)->as.heap.aux.capa + TERM_LEN(str))
194/* TODO: include the terminator size in capa. */
195
196#define STR_ENC_GET(str) get_encoding(str)
197
198static inline bool
199zero_filled(const char *s, int n)
200{
201 for (; n > 0; --n) {
202 if (*s++) return false;
203 }
204 return true;
205}
206
207#if !defined SHARABLE_MIDDLE_SUBSTRING
208# define SHARABLE_MIDDLE_SUBSTRING 0
209#endif
210
211static inline bool
212SHARABLE_SUBSTRING_P(VALUE str, long beg, long len)
213{
214#if SHARABLE_MIDDLE_SUBSTRING
215 return true;
216#else
217 long end = beg + len;
218 long source_len = RSTRING_LEN(str);
219 return end == source_len || zero_filled(RSTRING_PTR(str) + end, TERM_LEN(str));
220#endif
221}
222
223static inline long
224str_embed_capa(VALUE str)
225{
226 return rb_obj_shape_slot_size(str) - offsetof(struct RString, as.embed.ary);
227}
228
229bool
230rb_str_reembeddable_p(VALUE str)
231{
232 return !FL_TEST(str, STR_NOFREE|STR_SHARED_ROOT|STR_SHARED);
233}
234
235/* True when other strings read this string's bytes out of its own slot, so the slot
236 * contents must stay valid for as long as the object does. */
237bool
238rb_str_embedded_shared_root_p(VALUE str)
239{
240 return STR_EMBED_P(str) && FL_TEST(str, STR_SHARED_ROOT);
241}
242
243static inline size_t
244rb_str_embed_size(long capa, long termlen)
245{
246 size_t size = offsetof(struct RString, as.embed.ary) + capa + termlen;
247 if (size < sizeof(struct RString)) size = sizeof(struct RString);
248 return size;
249}
250
251size_t
252rb_str_size_as_embedded(VALUE str)
253{
254 size_t real_size;
255 if (STR_EMBED_P(str)) {
256 size_t capa = RSTRING(str)->len;
257 if (FL_TEST_RAW(str, STR_PRECOMPUTED_HASH)) capa += sizeof(st_index_t);
258
259 real_size = rb_str_embed_size(capa, TERM_LEN(str));
260 }
261 /* if the string is not currently embedded, but it can be embedded, how
262 * much space would it require */
263 else if (rb_str_reembeddable_p(str)) {
264 size_t capa = RSTRING(str)->as.heap.aux.capa;
265 if (FL_TEST_RAW(str, STR_PRECOMPUTED_HASH)) capa += sizeof(st_index_t);
266
267 real_size = rb_str_embed_size(capa, TERM_LEN(str));
268 }
269 else {
270 real_size = sizeof(struct RString);
271 }
272
273 return real_size;
274}
275
276static inline bool
277STR_EMBEDDABLE_P(long len, long termlen)
278{
279 return rb_gc_size_allocatable_p(rb_str_embed_size(len, termlen));
280}
281
282/* Substrings and duplicated strings that need a slot larger than this are shared
283 * instead of copied. Larger slots hold fewer objects per page and trigger GC
284 * more often, which outweighs the copy they save; see [Feature #22186] for the
285 * benchmarks. */
286#define STR_COPY_MAX_EMBED_SIZE 256
287
288static VALUE str_replace_shared_without_enc(VALUE str2, VALUE str);
289static VALUE str_new_frozen(VALUE klass, VALUE orig);
290static VALUE str_new_frozen_buffer(VALUE klass, VALUE orig, int copy_encoding);
291static VALUE str_new_static(VALUE klass, const char *ptr, long len, int encindex);
292static VALUE str_new(VALUE klass, const char *ptr, long len);
293static void str_make_independent_expand(VALUE str, long len, long expand, const int termlen);
294static inline void str_modifiable(VALUE str);
295static VALUE rb_str_downcase(int argc, VALUE *argv, VALUE str);
296static inline VALUE str_alloc_embed(VALUE klass, size_t capa);
297
298static inline void
299str_make_independent(VALUE str)
300{
301 long len = RSTRING_LEN(str);
302 int termlen = TERM_LEN(str);
303 str_make_independent_expand((str), len, 0L, termlen);
304}
305
306static inline int str_dependent_p(VALUE str);
307
308void
309rb_str_make_independent(VALUE str)
310{
311 if (str_dependent_p(str)) {
312 str_make_independent(str);
313 }
314}
315
316void
317rb_str_make_embedded(VALUE str)
318{
319 RUBY_ASSERT(rb_str_reembeddable_p(str));
320 RUBY_ASSERT(!STR_EMBED_P(str));
321
322 int termlen = TERM_LEN(str);
323 char *buf = RSTRING(str)->as.heap.ptr;
324 long old_capa = RSTRING(str)->as.heap.aux.capa + termlen;
325 long len = RSTRING(str)->len;
326
327 STR_SET_EMBED(str);
328 STR_SET_LEN(str, len);
329
330 if (len > 0) {
331 memcpy(RSTRING_PTR(str), buf, len);
332 SIZED_FREE_N(buf, old_capa);
333 }
334
335 TERM_FILL(RSTRING(str)->as.embed.ary + len, termlen);
336}
337
338void
339rb_debug_rstring_null_ptr(const char *func)
340{
341 fprintf(stderr, "%s is returning NULL!! "
342 "SIGSEGV is highly expected to follow immediately.\n"
343 "If you could reproduce, attach your debugger here, "
344 "and look at the passed string.\n",
345 func);
346}
347
348/* symbols for [up|down|swap]case/capitalize options */
349static VALUE sym_ascii, sym_turkic, sym_lithuanian, sym_fold;
350
351static rb_encoding *
352get_encoding(VALUE str)
353{
354 return rb_enc_from_index(ENCODING_GET(str));
355}
356
357static void
358mustnot_broken(VALUE str)
359{
360 if (is_broken_string(str)) {
361 rb_raise(rb_eArgError, "invalid byte sequence in %s", rb_enc_name(STR_ENC_GET(str)));
362 }
363}
364
365static void
366mustnot_wchar(VALUE str)
367{
368 rb_encoding *enc = STR_ENC_GET(str);
369 if (rb_enc_mbminlen(enc) > 1) {
370 rb_raise(rb_eArgError, "wide char encoding: %s", rb_enc_name(enc));
371 }
372}
373
374static VALUE register_fstring(VALUE str, bool copy, bool force_precompute_hash);
375
376#if SIZEOF_LONG == SIZEOF_VOIDP
377#define PRECOMPUTED_FAKESTR_HASH 1
378#else
379#endif
380
381static inline bool
382BARE_STRING_P(VALUE str)
383{
384 return RBASIC_CLASS(str) == rb_cString && !rb_obj_shape_has_ivars(str);
385}
386
387static inline st_index_t
388str_do_hash(VALUE str)
389{
390 st_index_t h = rb_memhash((const void *)RSTRING_PTR(str), RSTRING_LEN(str));
391 int e = RSTRING_LEN(str) ? ENCODING_GET(str) : 0;
392 if (e && !is_ascii_string(str)) {
393 h = rb_hash_end(rb_hash_uint32(h, (uint32_t)e));
394 }
395 return h;
396}
397
398static VALUE
399str_store_precomputed_hash(VALUE str, st_index_t hash)
400{
401 RUBY_ASSERT(!FL_TEST_RAW(str, STR_PRECOMPUTED_HASH));
402 RUBY_ASSERT(STR_EMBED_P(str));
403
404#if RUBY_DEBUG
405 size_t used_bytes = (RSTRING_LEN(str) + TERM_LEN(str));
406 size_t free_bytes = str_embed_capa(str) - used_bytes;
407 RUBY_ASSERT(free_bytes >= sizeof(st_index_t));
408#endif
409
410 memcpy(RSTRING_END(str) + TERM_LEN(str), &hash, sizeof(hash));
411
412 FL_SET(str, STR_PRECOMPUTED_HASH);
413
414 return str;
415}
416
417VALUE
418rb_fstring(VALUE str)
419{
420 VALUE fstr;
421 int bare;
422
423 Check_Type(str, T_STRING);
424
425 if (FL_TEST(str, RSTRING_FSTR))
426 return str;
427
428 bare = BARE_STRING_P(str);
429 if (!bare) {
430 if (STR_EMBED_P(str)) {
431 OBJ_FREEZE(str);
432 return str;
433 }
434
435 if (FL_TEST_RAW(str, STR_SHARED_ROOT | STR_SHARED) == STR_SHARED_ROOT) {
437 return str;
438 }
439 }
440
441 if (!FL_TEST_RAW(str, FL_FREEZE | STR_NOFREE | STR_CHILLED))
442 rb_str_resize(str, RSTRING_LEN(str));
443
444 fstr = register_fstring(str, false, false);
445
446 if (!bare) {
447 str_replace_shared_without_enc(str, fstr);
448 OBJ_FREEZE(str);
449 return str;
450 }
451 return fstr;
452}
453
454static VALUE fstring_table_obj;
455
456static VALUE
457fstring_concurrent_set_hash(VALUE str)
458{
459#ifdef PRECOMPUTED_FAKESTR_HASH
460 st_index_t h;
461 if (FL_TEST_RAW(str, STR_FAKESTR)) {
462 // register_fstring precomputes the hash and stores it in capa for fake strings
463 h = (st_index_t)RSTRING(str)->as.heap.aux.capa;
464 }
465 else {
466 h = rb_str_hash(str);
467 }
468 // rb_str_hash doesn't include the encoding for ascii only strings, so
469 // we add it to avoid common collisions between `:sym.name` (ASCII) and `"sym"` (UTF-8)
470 return (VALUE)rb_hash_end(rb_hash_uint32(h, (uint32_t)ENCODING_GET_INLINED(str)));
471#else
472 return (VALUE)rb_str_hash(str);
473#endif
474}
475
476static bool
477fstring_concurrent_set_cmp(VALUE a, VALUE b)
478{
479 long alen, blen;
480 const char *aptr, *bptr;
481
484
485 RSTRING_GETMEM(a, aptr, alen);
486 RSTRING_GETMEM(b, bptr, blen);
487 return (alen == blen &&
488 ENCODING_GET(a) == ENCODING_GET(b) &&
489 memcmp(aptr, bptr, alen) == 0);
490}
491
493 bool copy;
494 bool force_precompute_hash;
495};
496
497static VALUE
498fstring_concurrent_set_create(VALUE str, void *data)
499{
500 struct fstr_create_arg *arg = data;
501
502 // Unless the string is empty or binary, its coderange has been precomputed.
503 int coderange = ENC_CODERANGE(str);
504
505 if (FL_TEST_RAW(str, STR_FAKESTR)) {
506 if (arg->copy) {
507 VALUE new_str;
508 long len = RSTRING_LEN(str);
509 long capa = len + sizeof(st_index_t);
510 int term_len = TERM_LEN(str);
511
512 if (arg->force_precompute_hash && STR_EMBEDDABLE_P(capa, term_len)) {
513 new_str = str_alloc_embed(rb_cString, capa + term_len);
514 memcpy(RSTRING_PTR(new_str), RSTRING_PTR(str), len);
515 STR_SET_LEN(new_str, RSTRING_LEN(str));
516 TERM_FILL(RSTRING_END(new_str), TERM_LEN(str));
517 rb_enc_copy(new_str, str);
518 str_store_precomputed_hash(new_str, str_do_hash(str));
519 }
520 else {
521 new_str = str_new(rb_cString, RSTRING(str)->as.heap.ptr, RSTRING(str)->len);
522 rb_enc_copy(new_str, str);
523#ifdef PRECOMPUTED_FAKESTR_HASH
524 if (rb_str_capacity(new_str) >= RSTRING_LEN(str) + term_len + sizeof(st_index_t)) {
525 str_store_precomputed_hash(new_str, (st_index_t)RSTRING(str)->as.heap.aux.capa);
526 }
527#endif
528 }
529 str = new_str;
530 }
531 else {
532 str = str_new_static(rb_cString, RSTRING(str)->as.heap.ptr,
533 RSTRING(str)->len,
534 ENCODING_GET(str));
535 }
536 OBJ_FREEZE(str);
537 }
538 else {
539 if (!OBJ_FROZEN(str) || CHILLED_STRING_P(str)) {
540 str = str_new_frozen(rb_cString, str);
541 }
542 if (STR_SHARED_P(str)) { /* str should not be shared */
543 /* shared substring */
544 str_make_independent(str);
546 }
547 if (!BARE_STRING_P(str)) {
548 str = str_new_frozen(rb_cString, str);
549 }
550 }
551
552 ENC_CODERANGE_SET(str, coderange);
553 RBASIC(str)->flags |= RSTRING_FSTR;
554 if (!RB_OBJ_SHAREABLE_P(str)) {
556 }
557 RUBY_ASSERT((rb_gc_verify_shareable(str), 1));
560 RUBY_ASSERT(!FL_TEST_RAW(str, STR_FAKESTR));
561 RUBY_ASSERT(!rb_obj_shape_has_ivars(str));
563 RUBY_ASSERT(!rb_objspace_garbage_object_p(str));
564
565 return str;
566}
567
568static const struct rb_concurrent_set_funcs fstring_concurrent_set_funcs = {
569 .hash = fstring_concurrent_set_hash,
570 .cmp = fstring_concurrent_set_cmp,
571 .create = fstring_concurrent_set_create,
572 .free = NULL,
573};
574
575void
576Init_fstring_table(void)
577{
578 fstring_table_obj = rb_concurrent_set_new(&fstring_concurrent_set_funcs, 8192);
579 rb_gc_register_address(&fstring_table_obj);
580}
581
582static VALUE
583register_fstring(VALUE str, bool copy, bool force_precompute_hash)
584{
585 struct fstr_create_arg args = {
586 .copy = copy,
587 .force_precompute_hash = force_precompute_hash
588 };
589
590#if SIZEOF_VOIDP == SIZEOF_LONG
591 if (FL_TEST_RAW(str, STR_FAKESTR)) {
592 // if the string hasn't been interned, we'll need the hash twice, so we
593 // compute it once and store it in capa
594 RSTRING(str)->as.heap.aux.capa = (long)str_do_hash(str);
595 }
596#endif
597
598 VALUE result = rb_concurrent_set_find_or_insert(&fstring_table_obj, str, &args);
599
600 RUBY_ASSERT(!rb_objspace_garbage_object_p(result));
602 RUBY_ASSERT(OBJ_FROZEN(result));
604 RUBY_ASSERT((rb_gc_verify_shareable(result), 1));
605 RUBY_ASSERT(!FL_TEST_RAW(result, STR_FAKESTR));
607
608 return result;
609}
610
611bool
612rb_obj_is_fstring_table(VALUE obj)
613{
614 ASSERT_vm_locking();
615
616 return obj == fstring_table_obj;
617}
618
619void
620rb_gc_free_fstring(VALUE obj)
621{
622 ASSERT_vm_locking_with_barrier();
623
624 RUBY_ASSERT(FL_TEST(obj, RSTRING_FSTR));
626 RUBY_ASSERT(!FL_TEST(obj, STR_SHARED));
627
628 rb_concurrent_set_delete_by_identity(fstring_table_obj, obj);
629
630 RB_DEBUG_COUNTER_INC(obj_str_fstr);
631
632 FL_UNSET(obj, RSTRING_FSTR);
633}
634
635void
636rb_fstring_foreach_with_replace(int (*callback)(VALUE *str, void *data), void *data)
637{
638 if (fstring_table_obj) {
639 rb_concurrent_set_foreach_with_replace(fstring_table_obj, callback, data);
640 }
641}
642
643static VALUE
644setup_fake_str(struct RString *fake_str, const char *name, long len, int encidx)
645{
646 fake_str->basic.flags = T_STRING|RSTRING_NOEMBED|STR_NOFREE|STR_FAKESTR;
647 RBASIC_SET_FULL_SHAPE_ID((VALUE)fake_str, ROOT_SHAPE_ID | SHAPE_ID_LAYOUT_OTHER);
648
649 if (!name) {
651 name = "";
652 }
653
654 ENCODING_SET_INLINED((VALUE)fake_str, encidx);
655
656 RBASIC_SET_CLASS_RAW((VALUE)fake_str, rb_cString);
657 fake_str->len = len;
658 fake_str->as.heap.ptr = (char *)name;
659 fake_str->as.heap.aux.capa = len;
660 return (VALUE)fake_str;
661}
662
663/*
664 * set up a fake string which refers a static string literal.
665 */
666VALUE
667rb_setup_fake_str(struct RString *fake_str, const char *name, long len, rb_encoding *enc)
668{
669 return setup_fake_str(fake_str, name, len, rb_enc_to_index(enc));
670}
671
672/*
673 * rb_fstring_new and rb_fstring_cstr family create or lookup a frozen
674 * shared string which refers a static string literal. `ptr` must
675 * point a constant string.
676 */
677VALUE
678rb_fstring_new(const char *ptr, long len)
679{
680 struct RString fake_str = {RBASIC_INIT};
681 return register_fstring(setup_fake_str(&fake_str, ptr, len, ENCINDEX_US_ASCII), false, false);
682}
683
684VALUE
685rb_fstring_enc_new(const char *ptr, long len, rb_encoding *enc)
686{
687 struct RString fake_str = {RBASIC_INIT};
688 return register_fstring(rb_setup_fake_str(&fake_str, ptr, len, enc), false, false);
689}
690
691VALUE
692rb_fstring_cstr(const char *ptr)
693{
694 return rb_fstring_new(ptr, strlen(ptr));
695}
696
697static inline bool
698single_byte_optimizable(VALUE str)
699{
700 int encindex = ENCODING_GET(str);
701 switch (encindex) {
702 case ENCINDEX_ASCII_8BIT:
703 case ENCINDEX_US_ASCII:
704 return true;
705 case ENCINDEX_UTF_8:
706 // For UTF-8 it's worth scanning the string coderange when unknown.
707 return rb_enc_str_coderange(str) == ENC_CODERANGE_7BIT;
708 }
709 /* Conservative. It may be ENC_CODERANGE_UNKNOWN. */
710 if (ENC_CODERANGE(str) == ENC_CODERANGE_7BIT) {
711 return true;
712 }
713
714 if (rb_enc_mbmaxlen(rb_enc_from_index(encindex)) == 1) {
715 return true;
716 }
717
718 /* Conservative. Possibly single byte.
719 * "\xa1" in Shift_JIS for example. */
720 return false;
721}
722
724
725static inline const char *
726search_nonascii(const char *p, const char *e)
727{
728 const char *s, *t;
729
730 if (p < e && !ISASCII(*p)) {
731 return p;
732 }
733
734#if defined(__STDC_VERSION__) && (__STDC_VERSION__ >= 199901L)
735# if SIZEOF_UINTPTR_T == 8
736# define NONASCII_MASK UINT64_C(0x8080808080808080)
737# elif SIZEOF_UINTPTR_T == 4
738# define NONASCII_MASK UINT32_C(0x80808080)
739# else
740# error "don't know what to do."
741# endif
742#else
743# if SIZEOF_UINTPTR_T == 8
744# define NONASCII_MASK ((uintptr_t)0x80808080UL << 32 | (uintptr_t)0x80808080UL)
745# elif SIZEOF_UINTPTR_T == 4
746# define NONASCII_MASK 0x80808080UL /* or...? */
747# else
748# error "don't know what to do."
749# endif
750#endif
751
752 if (UNALIGNED_WORD_ACCESS || e - p >= SIZEOF_VOIDP) {
753#if !UNALIGNED_WORD_ACCESS
754 if ((uintptr_t)p % SIZEOF_VOIDP) {
755 int l = SIZEOF_VOIDP - (uintptr_t)p % SIZEOF_VOIDP;
756 p += l;
757 switch (l) {
758 default: UNREACHABLE;
759#if SIZEOF_VOIDP > 4
760 case 7: if (p[-7]&0x80) return p-7;
761 case 6: if (p[-6]&0x80) return p-6;
762 case 5: if (p[-5]&0x80) return p-5;
763 case 4: if (p[-4]&0x80) return p-4;
764#endif
765 case 3: if (p[-3]&0x80) return p-3;
766 case 2: if (p[-2]&0x80) return p-2;
767 case 1: if (p[-1]&0x80) return p-1;
768 case 0: break;
769 }
770 }
771#endif
772#if defined(HAVE_BUILTIN___BUILTIN_ASSUME_ALIGNED) &&! UNALIGNED_WORD_ACCESS
773#define aligned_ptr(value) \
774 __builtin_assume_aligned((value), sizeof(uintptr_t))
775#else
776#define aligned_ptr(value) (value)
777#endif
778 s = aligned_ptr(p);
779 t = (e - (SIZEOF_VOIDP-1));
780#undef aligned_ptr
781 for (;s < t; s += sizeof(uintptr_t)) {
782 uintptr_t word;
783 memcpy(&word, s, sizeof(word));
784 if (word & NONASCII_MASK) {
785#ifdef WORDS_BIGENDIAN
786 return (const char *)s + (nlz_intptr(word&NONASCII_MASK)>>3);
787#else
788 return (const char *)s + (ntz_intptr(word&NONASCII_MASK)>>3);
789#endif
790 }
791 }
792 p = (const char *)s;
793 }
794
795 switch (e - p) {
796 default: UNREACHABLE;
797#if SIZEOF_VOIDP > 4
798 case 7: if (e[-7]&0x80) return e-7;
799 case 6: if (e[-6]&0x80) return e-6;
800 case 5: if (e[-5]&0x80) return e-5;
801 case 4: if (e[-4]&0x80) return e-4;
802#endif
803 case 3: if (e[-3]&0x80) return e-3;
804 case 2: if (e[-2]&0x80) return e-2;
805 case 1: if (e[-1]&0x80) return e-1;
806 case 0: return NULL;
807 }
808}
809
810static int
811coderange_scan(const char *p, long len, rb_encoding *enc)
812{
813 const char *e = p + len;
814
815 if (rb_enc_to_index(enc) == rb_ascii8bit_encindex()) {
816 /* enc is ASCII-8BIT. ASCII-8BIT string never be broken. */
817 p = search_nonascii(p, e);
819 }
820
821 if (rb_enc_asciicompat(enc)) {
822 p = search_nonascii(p, e);
823 if (!p) return ENC_CODERANGE_7BIT;
824 for (;;) {
825 int ret = rb_enc_precise_mbclen(p, e, enc);
827 p += MBCLEN_CHARFOUND_LEN(ret);
828 if (p == e) break;
829 p = search_nonascii(p, e);
830 if (!p) break;
831 }
832 }
833 else {
834 while (p < e) {
835 int ret = rb_enc_precise_mbclen(p, e, enc);
837 p += MBCLEN_CHARFOUND_LEN(ret);
838 }
839 }
840 return ENC_CODERANGE_VALID;
841}
842
843long
844rb_str_coderange_scan_restartable(const char *s, const char *e, rb_encoding *enc, int *cr)
845{
846 const char *p = s;
847
848 if (*cr == ENC_CODERANGE_BROKEN)
849 return e - s;
850
851 if (rb_enc_to_index(enc) == rb_ascii8bit_encindex()) {
852 /* enc is ASCII-8BIT. ASCII-8BIT string never be broken. */
853 if (*cr == ENC_CODERANGE_VALID) return e - s;
854 p = search_nonascii(p, e);
856 return e - s;
857 }
858 else if (rb_enc_asciicompat(enc)) {
859 p = search_nonascii(p, e);
860 if (!p) {
861 if (*cr != ENC_CODERANGE_VALID) *cr = ENC_CODERANGE_7BIT;
862 return e - s;
863 }
864 for (;;) {
865 int ret = rb_enc_precise_mbclen(p, e, enc);
866 if (!MBCLEN_CHARFOUND_P(ret)) {
868 return p - s;
869 }
870 p += MBCLEN_CHARFOUND_LEN(ret);
871 if (p == e) break;
872 p = search_nonascii(p, e);
873 if (!p) break;
874 }
875 }
876 else {
877 while (p < e) {
878 int ret = rb_enc_precise_mbclen(p, e, enc);
879 if (!MBCLEN_CHARFOUND_P(ret)) {
881 return p - s;
882 }
883 p += MBCLEN_CHARFOUND_LEN(ret);
884 }
885 }
887 return e - s;
888}
889
890static inline void
891str_enc_copy(VALUE str1, VALUE str2)
892{
893 rb_enc_set_index(str1, ENCODING_GET(str2));
894}
895
896/* Like str_enc_copy, but does not check frozen status of str1.
897 * You should use this only if you're certain that str1 is not frozen. */
898static inline void
899str_enc_copy_direct(VALUE str1, VALUE str2)
900{
901 int inlined_encoding = RB_ENCODING_GET_INLINED(str2);
902 if (inlined_encoding == ENCODING_INLINE_MAX) {
903 rb_enc_set_index(str1, rb_enc_get_index(str2));
904 }
905 else {
906 ENCODING_SET_INLINED(str1, inlined_encoding);
907 }
908}
909
910static void
911rb_enc_cr_str_copy_for_substr(VALUE dest, VALUE src)
912{
913 /* this function is designed for copying encoding and coderange
914 * from src to new string "dest" which is made from the part of src.
915 */
916 str_enc_copy(dest, src);
917 if (RSTRING_LEN(dest) == 0) {
918 if (!rb_enc_asciicompat(STR_ENC_GET(src)))
920 else
922 return;
923 }
924 switch (ENC_CODERANGE(src)) {
927 break;
929 if (!rb_enc_asciicompat(STR_ENC_GET(src)) ||
930 search_nonascii(RSTRING_PTR(dest), RSTRING_END(dest)))
932 else
934 break;
935 default:
936 break;
937 }
938}
939
940static void
941rb_enc_cr_str_exact_copy(VALUE dest, VALUE src)
942{
943 str_enc_copy(dest, src);
945}
946
947static int
948enc_coderange_scan(VALUE str, rb_encoding *enc)
949{
950 return coderange_scan(RSTRING_PTR(str), RSTRING_LEN(str), enc);
951}
952
953int
954rb_enc_str_coderange_scan(VALUE str, rb_encoding *enc)
955{
956 return enc_coderange_scan(str, enc);
957}
958
959int
960rbimpl_enc_str_coderange_scan(VALUE str)
961{
962 int cr = enc_coderange_scan(str, get_encoding(str));
963 ENC_CODERANGE_SET(str, cr);
964 return cr;
965}
966
967#undef rb_enc_str_coderange
968int
969rb_enc_str_coderange(VALUE str)
970{
971 int cr = ENC_CODERANGE(str);
972
973 if (cr == ENC_CODERANGE_UNKNOWN) {
974 cr = rbimpl_enc_str_coderange_scan(str);
975 }
976 return cr;
977}
978#define rb_enc_str_coderange rb_enc_str_coderange_inline
979
980static inline bool
981rb_enc_str_asciicompat(VALUE str)
982{
983 int encindex = ENCODING_GET_INLINED(str);
984 return rb_str_encindex_fastpath(encindex) || rb_enc_asciicompat(rb_enc_get_from_index(encindex));
985}
986
987int
989{
990 switch(ENC_CODERANGE(str)) {
992 return rb_enc_str_asciicompat(str) && is_ascii_string(str);
994 return true;
995 default:
996 return false;
997 }
998}
999
1000static inline void
1001str_mod_check(VALUE s, const char *p, long len)
1002{
1003 if (RSTRING_PTR(s) != p || RSTRING_LEN(s) != len){
1004 rb_raise(rb_eRuntimeError, "string modified");
1005 }
1006}
1007
1008static size_t
1009str_capacity(VALUE str, const int termlen)
1010{
1011 if (STR_EMBED_P(str)) {
1012 return str_embed_capa(str) - termlen;
1013 }
1014 else if (FL_ANY_RAW(str, STR_SHARED|STR_NOFREE)) {
1015 return RSTRING(str)->len;
1016 }
1017 else {
1018 return RSTRING(str)->as.heap.aux.capa;
1019 }
1020}
1021
1022size_t
1024{
1025 return str_capacity(str, TERM_LEN(str));
1026}
1027
1028static inline void
1029must_not_null(const char *ptr)
1030{
1031 if (!ptr) {
1032 rb_raise(rb_eArgError, "NULL pointer given");
1033 }
1034}
1035
1036static inline VALUE
1037str_alloc_embed(VALUE klass, size_t capa)
1038{
1039 size_t size = rb_str_embed_size(capa, 0);
1040 RUBY_ASSERT(size > 0);
1041 RUBY_ASSERT(rb_gc_size_allocatable_p(size));
1042
1043 NEWOBJ_OF(str, struct RString, klass, T_STRING, size);
1044
1045 str->len = 0;
1046 str->as.embed.ary[0] = 0;
1047
1048 return (VALUE)str;
1049}
1050
1051static inline VALUE
1052str_alloc_heap(VALUE klass)
1053{
1054 NEWOBJ_OF(str, struct RString, klass, T_STRING | STR_NOEMBED, sizeof(struct RString));
1055
1056 str->len = 0;
1057 str->as.heap.aux.capa = 0;
1058 str->as.heap.ptr = NULL;
1059
1060 return (VALUE)str;
1061}
1062
1063static inline VALUE
1064empty_str_alloc(VALUE klass)
1065{
1066 RUBY_DTRACE_CREATE_HOOK(STRING, 0);
1067 VALUE str = str_alloc_embed(klass, 0);
1068 memset(RSTRING(str)->as.embed.ary, 0, str_embed_capa(str));
1070 return str;
1071}
1072
1073static VALUE
1074str_enc_new(VALUE klass, const char *ptr, long len, rb_encoding *enc)
1075{
1076 VALUE str;
1077
1078 if (len < 0) {
1079 rb_raise(rb_eArgError, "negative string size (or size too big)");
1080 }
1081
1082 if (enc == NULL) {
1083 enc = rb_ascii8bit_encoding();
1084 }
1085
1086 RUBY_DTRACE_CREATE_HOOK(STRING, len);
1087
1088 int termlen = rb_enc_mbminlen(enc);
1089
1090 if (STR_EMBEDDABLE_P(len, termlen)) {
1091 str = str_alloc_embed(klass, len + termlen);
1092 if (len == 0) {
1093 ENC_CODERANGE_SET(str, rb_enc_asciicompat(enc) ? ENC_CODERANGE_7BIT : ENC_CODERANGE_VALID);
1094 }
1095 }
1096 else {
1097 str = str_alloc_heap(klass);
1098 RSTRING(str)->as.heap.aux.capa = len;
1099 /* :FIXME: @shyouhei guesses `len + termlen` is guaranteed to never
1100 * integer overflow. If we can STATIC_ASSERT that, the following
1101 * mul_add_mul can be reverted to a simple ALLOC_N. */
1102 RSTRING(str)->as.heap.ptr =
1103 rb_xmalloc_mul_add_mul(sizeof(char), len, sizeof(char), termlen);
1104 }
1105
1106 rb_enc_raw_set(str, enc);
1107
1108 if (ptr) {
1109 memcpy(RSTRING_PTR(str), ptr, len);
1110 }
1111 else {
1112 memset(RSTRING_PTR(str), 0, len);
1113 }
1114
1115 STR_SET_LEN(str, len);
1116 TERM_FILL(RSTRING_PTR(str) + len, termlen);
1117 return str;
1118}
1119
1120static VALUE
1121str_new(VALUE klass, const char *ptr, long len)
1122{
1123 return str_enc_new(klass, ptr, len, rb_ascii8bit_encoding());
1124}
1125
1126VALUE
1127rb_str_new(const char *ptr, long len)
1128{
1129 return str_new(rb_cString, ptr, len);
1130}
1131
1132VALUE
1133rb_usascii_str_new(const char *ptr, long len)
1134{
1135 return str_enc_new(rb_cString, ptr, len, rb_usascii_encoding());
1136}
1137
1138VALUE
1139rb_utf8_str_new(const char *ptr, long len)
1140{
1141 return str_enc_new(rb_cString, ptr, len, rb_utf8_encoding());
1142}
1143
1144VALUE
1145rb_enc_str_new(const char *ptr, long len, rb_encoding *enc)
1146{
1147 return str_enc_new(rb_cString, ptr, len, enc);
1148}
1149
1150VALUE
1152{
1153 must_not_null(ptr);
1154 /* rb_str_new_cstr() can take pointer from non-malloc-generated
1155 * memory regions, and that cannot be detected by the MSAN. Just
1156 * trust the programmer that the argument passed here is a sane C
1157 * string. */
1158 __msan_unpoison_string(ptr);
1159 return rb_str_new(ptr, strlen(ptr));
1160}
1161
1162VALUE
1164{
1165 return rb_enc_str_new_cstr(ptr, rb_usascii_encoding());
1166}
1167
1168VALUE
1170{
1171 return rb_enc_str_new_cstr(ptr, rb_utf8_encoding());
1172}
1173
1174VALUE
1176{
1177 must_not_null(ptr);
1178 if (rb_enc_mbminlen(enc) != 1) {
1179 rb_raise(rb_eArgError, "wchar encoding given");
1180 }
1181 return rb_enc_str_new(ptr, strlen(ptr), enc);
1182}
1183
1184static VALUE
1185str_new_static(VALUE klass, const char *ptr, long len, int encindex)
1186{
1187 VALUE str;
1188
1189 if (len < 0) {
1190 rb_raise(rb_eArgError, "negative string size (or size too big)");
1191 }
1192
1193 if (!ptr) {
1194 str = str_enc_new(klass, ptr, len, rb_enc_from_index(encindex));
1195 }
1196 else {
1197 RUBY_DTRACE_CREATE_HOOK(STRING, len);
1198 str = str_alloc_heap(klass);
1199 RSTRING(str)->len = len;
1200 RSTRING(str)->as.heap.ptr = (char *)ptr;
1201 RSTRING(str)->as.heap.aux.capa = len;
1202 RBASIC(str)->flags |= STR_NOFREE;
1203 rb_enc_associate_index(str, encindex);
1204 }
1205 return str;
1206}
1207
1208VALUE
1209rb_str_new_static(const char *ptr, long len)
1210{
1211 return str_new_static(rb_cString, ptr, len, 0);
1212}
1213
1214/* Take an xmalloc'd buffer as the String's body without copying it; the String owns it
1215 * from here and frees it like any other heap string. ptr must hold capa bytes plus the
1216 * terminator for encindex, which is what a Ractor courier's string node carries. */
1217VALUE
1218rb_str_new_owned(char *ptr, long len, long capa, int encindex)
1219{
1220 RUBY_DTRACE_CREATE_HOOK(STRING, len);
1221 VALUE str = str_alloc_heap(rb_cString);
1222 RSTRING(str)->len = len;
1223 RSTRING(str)->as.heap.ptr = ptr;
1224 /* Freed by size (STR_HEAP_SIZE = capa + terminator), so capa must describe the
1225 * allocation the caller made, not just the bytes in use. */
1226 RSTRING(str)->as.heap.aux.capa = capa;
1227 rb_enc_associate_index(str, encindex);
1228 return str;
1229}
1230
1231VALUE
1233{
1234 return str_new_static(rb_cString, ptr, len, ENCINDEX_US_ASCII);
1235}
1236
1237VALUE
1239{
1240 return str_new_static(rb_cString, ptr, len, ENCINDEX_UTF_8);
1241}
1242
1243VALUE
1245{
1246 return str_new_static(rb_cString, ptr, len, rb_enc_to_index(enc));
1247}
1248
1249static VALUE str_cat_conv_enc_opts(VALUE newstr, long ofs, const char *ptr, long len,
1250 rb_encoding *from, rb_encoding *to,
1251 int ecflags, VALUE ecopts);
1252
1253static inline bool
1254is_enc_ascii_string(VALUE str, rb_encoding *enc)
1255{
1256 int encidx = rb_enc_to_index(enc);
1257 if (rb_enc_get_index(str) == encidx)
1258 return is_ascii_string(str);
1259 return enc_coderange_scan(str, enc) == ENC_CODERANGE_7BIT;
1260}
1261
1262VALUE
1263rb_str_conv_enc_opts(VALUE str, rb_encoding *from, rb_encoding *to, int ecflags, VALUE ecopts)
1264{
1265 long len;
1266 const char *ptr;
1267 VALUE newstr;
1268
1269 if (!to) return str;
1270 if (!from) from = rb_enc_get(str);
1271 if (from == to) return str;
1272 if ((rb_enc_asciicompat(to) && is_enc_ascii_string(str, from)) ||
1273 rb_is_ascii8bit_enc(to)) {
1274 if (STR_ENC_GET(str) != to) {
1275 str = rb_str_dup(str);
1276 rb_enc_associate(str, to);
1277 }
1278 return str;
1279 }
1280
1281 RSTRING_GETMEM(str, ptr, len);
1282 newstr = str_cat_conv_enc_opts(rb_str_buf_new(len), 0, ptr, len,
1283 from, to, ecflags, ecopts);
1284 if (NIL_P(newstr)) {
1285 /* some error, return original */
1286 return str;
1287 }
1288 return newstr;
1289}
1290
1291VALUE
1292rb_str_cat_conv_enc_opts(VALUE newstr, long ofs, const char *ptr, long len,
1293 rb_encoding *from, int ecflags, VALUE ecopts)
1294{
1295 long olen;
1296
1297 olen = RSTRING_LEN(newstr);
1298 if (ofs < -olen || olen < ofs)
1299 rb_raise(rb_eIndexError, "index %ld out of string", ofs);
1300 if (ofs < 0) ofs += olen;
1301 if (!from) {
1302 STR_SET_LEN(newstr, ofs);
1303 return rb_str_cat(newstr, ptr, len);
1304 }
1305
1306 rb_str_modify(newstr);
1307 return str_cat_conv_enc_opts(newstr, ofs, ptr, len, from,
1308 rb_enc_get(newstr),
1309 ecflags, ecopts);
1310}
1311
1312VALUE
1313rb_str_initialize(VALUE str, const char *ptr, long len, rb_encoding *enc)
1314{
1315 STR_SET_LEN(str, 0);
1316 rb_enc_associate(str, enc);
1317 rb_str_cat(str, ptr, len);
1318 return str;
1319}
1320
1321static VALUE
1322str_cat_conv_enc_opts(VALUE newstr, long ofs, const char *ptr, long len,
1323 rb_encoding *from, rb_encoding *to,
1324 int ecflags, VALUE ecopts)
1325{
1326 rb_econv_t *ec;
1328 long olen;
1329 VALUE econv_wrapper;
1330 const unsigned char *start, *sp;
1331 unsigned char *dest, *dp;
1332 size_t converted_output = (size_t)ofs;
1333
1334 olen = rb_str_capacity(newstr);
1335
1336 econv_wrapper = rb_obj_alloc(rb_cEncodingConverter);
1337 RBASIC_CLEAR_CLASS(econv_wrapper);
1338 ec = rb_econv_open_opts(from->name, to->name, ecflags, ecopts);
1339 if (!ec) return Qnil;
1340 DATA_PTR(econv_wrapper) = ec;
1341
1342 sp = (unsigned char*)ptr;
1343 start = sp;
1344 while ((dest = (unsigned char*)RSTRING_PTR(newstr)),
1345 (dp = dest + converted_output),
1346 (ret = rb_econv_convert(ec, &sp, start + len, &dp, dest + olen, 0)),
1348 /* destination buffer short */
1349 size_t converted_input = sp - start;
1350 size_t rest = len - converted_input;
1351 converted_output = dp - dest;
1352 rb_str_set_len(newstr, converted_output);
1353 if (converted_input && converted_output &&
1354 rest < (LONG_MAX / converted_output)) {
1355 rest = (rest * converted_output) / converted_input;
1356 }
1357 else {
1358 rest = olen;
1359 }
1360 olen += rest < 2 ? 2 : rest;
1361 rb_str_resize(newstr, olen);
1362 }
1363 DATA_PTR(econv_wrapper) = 0;
1364 RB_GC_GUARD(econv_wrapper);
1365 rb_econv_close(ec);
1366 switch (ret) {
1367 case econv_finished:
1368 len = dp - (unsigned char*)RSTRING_PTR(newstr);
1369 rb_str_set_len(newstr, len);
1370 rb_enc_associate(newstr, to);
1371 return newstr;
1372
1373 default:
1374 return Qnil;
1375 }
1376}
1377
1378VALUE
1380{
1381 return rb_str_conv_enc_opts(str, from, to, 0, Qnil);
1382}
1383
1384VALUE
1386{
1387 rb_encoding *ienc;
1388 VALUE str;
1389 const int eidx = rb_enc_to_index(eenc);
1390
1391 if (!ptr) {
1392 return rb_enc_str_new(ptr, len, eenc);
1393 }
1394
1395 /* ASCII-8BIT case, no conversion */
1396 if ((eidx == rb_ascii8bit_encindex()) ||
1397 (eidx == rb_usascii_encindex() && search_nonascii(ptr, ptr + len))) {
1398 return rb_str_new(ptr, len);
1399 }
1400 /* no default_internal or same encoding, no conversion */
1401 ienc = rb_default_internal_encoding();
1402 if (!ienc || eenc == ienc) {
1403 return rb_enc_str_new(ptr, len, eenc);
1404 }
1405 /* ASCII compatible, and ASCII only string, no conversion in
1406 * default_internal */
1407 if ((eidx == rb_ascii8bit_encindex()) ||
1408 (eidx == rb_usascii_encindex()) ||
1409 (rb_enc_asciicompat(eenc) && !search_nonascii(ptr, ptr + len))) {
1410 return rb_enc_str_new(ptr, len, ienc);
1411 }
1412 /* convert from the given encoding to default_internal */
1413 str = rb_enc_str_new(NULL, 0, ienc);
1414 /* when the conversion failed for some reason, just ignore the
1415 * default_internal and result in the given encoding as-is. */
1416 if (NIL_P(rb_str_cat_conv_enc_opts(str, 0, ptr, len, eenc, 0, Qnil))) {
1417 rb_str_initialize(str, ptr, len, eenc);
1418 }
1419 return str;
1420}
1421
1422VALUE
1423rb_external_str_with_enc(VALUE str, rb_encoding *eenc)
1424{
1425 int eidx = rb_enc_to_index(eenc);
1426 if (eidx == rb_usascii_encindex() &&
1427 !is_ascii_string(str)) {
1428 rb_enc_associate_index(str, rb_ascii8bit_encindex());
1429 return str;
1430 }
1431 rb_enc_associate_index(str, eidx);
1432 return rb_str_conv_enc(str, eenc, rb_default_internal_encoding());
1433}
1434
1435VALUE
1436rb_external_str_new(const char *ptr, long len)
1437{
1438 return rb_external_str_new_with_enc(ptr, len, rb_default_external_encoding());
1439}
1440
1441VALUE
1443{
1444 return rb_external_str_new_with_enc(ptr, strlen(ptr), rb_default_external_encoding());
1445}
1446
1447VALUE
1448rb_locale_str_new(const char *ptr, long len)
1449{
1450 return rb_external_str_new_with_enc(ptr, len, rb_locale_encoding());
1451}
1452
1453VALUE
1455{
1456 return rb_external_str_new_with_enc(ptr, strlen(ptr), rb_locale_encoding());
1457}
1458
1459VALUE
1461{
1462 return rb_external_str_new_with_enc(ptr, len, rb_filesystem_encoding());
1463}
1464
1465VALUE
1467{
1468 return rb_external_str_new_with_enc(ptr, strlen(ptr), rb_filesystem_encoding());
1469}
1470
1471VALUE
1473{
1474 return rb_str_export_to_enc(str, rb_default_external_encoding());
1475}
1476
1477VALUE
1479{
1480 return rb_str_export_to_enc(str, rb_locale_encoding());
1481}
1482
1483VALUE
1485{
1486 return rb_str_conv_enc(str, STR_ENC_GET(str), enc);
1487}
1488
1489static VALUE
1490str_replace_shared_without_enc(VALUE str2, VALUE str)
1491{
1492 const int termlen = TERM_LEN(str);
1493 char *ptr;
1494 long len;
1495
1496 RSTRING_GETMEM(str, ptr, len);
1497 if (str_embed_capa(str2) >= len + termlen) {
1498 char *ptr2 = RSTRING(str2)->as.embed.ary;
1499 STR_SET_EMBED(str2);
1500 memcpy(ptr2, RSTRING_PTR(str), len);
1501 TERM_FILL(ptr2+len, termlen);
1502 }
1503 else {
1504 VALUE root;
1505 if (STR_SHARED_P(str)) {
1506 root = RSTRING(str)->as.heap.aux.shared;
1507 RSTRING_GETMEM(str, ptr, len);
1508 }
1509 else {
1510 root = rb_str_new_frozen(str);
1511 RSTRING_GETMEM(root, ptr, len);
1512 }
1513 RUBY_ASSERT(OBJ_FROZEN(root));
1514
1515 if (!STR_EMBED_P(str2) && !FL_TEST_RAW(str2, STR_SHARED|STR_NOFREE)) {
1516 if (FL_TEST_RAW(str2, STR_SHARED_ROOT)) {
1517 rb_fatal("about to free a possible shared root");
1518 }
1519 char *ptr2 = STR_HEAP_PTR(str2);
1520 if (ptr2 != ptr) {
1521 SIZED_FREE_N(ptr2, STR_HEAP_SIZE(str2));
1522 }
1523 }
1524 FL_SET(str2, STR_NOEMBED);
1525 RSTRING(str2)->as.heap.ptr = ptr;
1526 STR_SET_SHARED(str2, root);
1527 }
1528
1529 STR_SET_LEN(str2, len);
1530
1531 return str2;
1532}
1533
1534static VALUE
1535str_replace_shared(VALUE str2, VALUE str)
1536{
1537 str_replace_shared_without_enc(str2, str);
1538 rb_enc_cr_str_exact_copy(str2, str);
1539 return str2;
1540}
1541
1542static VALUE
1543str_new_shared(VALUE klass, VALUE str)
1544{
1545 return str_replace_shared(str_alloc_heap(klass), str);
1546}
1547
1548VALUE
1550{
1551 return str_new_shared(rb_obj_class(str), str);
1552}
1553
1554VALUE
1556{
1557 if (RB_FL_TEST_RAW(orig, FL_FREEZE | STR_CHILLED) == FL_FREEZE) return orig;
1558 return str_new_frozen(rb_obj_class(orig), orig);
1559}
1560
1561static VALUE
1562rb_str_new_frozen_String(VALUE orig)
1563{
1564 if (OBJ_FROZEN(orig) && rb_obj_class(orig) == rb_cString) return orig;
1565 return str_new_frozen(rb_cString, orig);
1566}
1567
1568
1569VALUE
1570rb_str_frozen_bare_string(VALUE orig)
1571{
1572 if (RB_LIKELY(BARE_STRING_P(orig) && OBJ_FROZEN_RAW(orig))) return orig;
1573 return str_new_frozen(rb_cString, orig);
1574}
1575
1576VALUE
1577rb_str_tmp_frozen_acquire(VALUE orig)
1578{
1579 if (OBJ_FROZEN_RAW(orig)) return orig;
1580 return str_new_frozen_buffer(0, orig, FALSE);
1581}
1582
1583VALUE
1585{
1586 if (OBJ_FROZEN_RAW(orig) && !STR_EMBED_P(orig)) return orig;
1587 if (STR_SHARED_P(orig) && !STR_EMBED_P(RSTRING(orig)->as.heap.aux.shared)) return rb_str_tmp_frozen_acquire(orig);
1588
1589 VALUE str = str_alloc_heap(0);
1590 OBJ_FREEZE(str);
1591 /* Always set the STR_SHARED_ROOT to ensure it does not get re-embedded. */
1592 FL_SET(str, STR_SHARED_ROOT);
1593
1594 size_t capa = str_capacity(orig, TERM_LEN(orig));
1595
1596 /* If the string is embedded then we want to create a copy that is heap
1597 * allocated. If the string is shared then the shared root must be
1598 * embedded, so we want to create a copy. If the string is a shared root
1599 * then it must be embedded, so we want to create a copy. */
1600 if (STR_EMBED_P(orig) || FL_TEST_RAW(orig, STR_SHARED | STR_SHARED_ROOT | RSTRING_FSTR)) {
1601 RSTRING(str)->as.heap.ptr = rb_xmalloc_mul_add_mul(sizeof(char), capa, sizeof(char), TERM_LEN(orig));
1602 memcpy(RSTRING(str)->as.heap.ptr, RSTRING_PTR(orig), capa);
1603 }
1604 else {
1605 /* orig must be heap allocated and not shared, so we can safely transfer
1606 * the pointer to str. */
1607 RSTRING(str)->as.heap.ptr = RSTRING(orig)->as.heap.ptr;
1608 RBASIC(str)->flags |= RBASIC(orig)->flags & STR_NOFREE;
1609 RBASIC(orig)->flags &= ~STR_NOFREE;
1610 STR_SET_SHARED(orig, str);
1611 /* str was just allocated here, so orig is its only child and it is
1612 * safe for rb_str_tmp_frozen_release to give the buffer back. */
1613 FL_UNSET_RAW(str, STR_BORROWED);
1614 if (RB_OBJ_SHAREABLE_P(orig)) {
1616 RUBY_ASSERT((rb_gc_verify_shareable(str), 1));
1617 }
1618 }
1619
1620 RSTRING(str)->len = RSTRING(orig)->len;
1621 RSTRING(str)->as.heap.aux.capa = capa + (TERM_LEN(orig) - TERM_LEN(str));
1622
1623 return str;
1624}
1625
1626void
1628{
1629 rb_str_tmp_frozen_release(orig, tmp);
1630}
1631
1632void
1633rb_str_tmp_frozen_release(VALUE orig, VALUE tmp)
1634{
1635 if (RBASIC_CLASS(tmp) != 0)
1636 return;
1637
1638 if (STR_EMBED_P(tmp)) {
1640 }
1641 else if (FL_TEST_RAW(orig, STR_SHARED | STR_TMPLOCK) == STR_SHARED &&
1642 !OBJ_FROZEN_RAW(orig)) {
1643 VALUE shared = RSTRING(orig)->as.heap.aux.shared;
1644
1645 if (shared == tmp && !FL_TEST_RAW(tmp, STR_BORROWED)) {
1646 RUBY_ASSERT(RSTRING(orig)->as.heap.ptr == RSTRING(tmp)->as.heap.ptr);
1647 RUBY_ASSERT(RSTRING_LEN(orig) == RSTRING_LEN(tmp));
1648
1649 /* Unshare orig since the root (tmp) only has this one child. */
1650 FL_UNSET_RAW(orig, STR_SHARED);
1651 RSTRING(orig)->as.heap.aux.capa = RSTRING(tmp)->as.heap.aux.capa + TERM_LEN(tmp) - TERM_LEN(orig);
1652 RBASIC(orig)->flags |= RBASIC(tmp)->flags & STR_NOFREE;
1654
1655 /* Make tmp embedded and empty so it is safe for sweeping. */
1656 STR_SET_EMBED(tmp);
1657 STR_SET_LEN(tmp, 0);
1658 }
1659 }
1660}
1661
1662static VALUE
1663str_new_frozen(VALUE klass, VALUE orig)
1664{
1665 return str_new_frozen_buffer(klass, orig, TRUE);
1666}
1667
1668/* Transfers ownership of orig's buffer to a new shared root string.
1669 * termlen is the terminator length of the returned string, which may differ
1670 * from orig's terminator length when the caller does not copy the encoding.
1671 * The capacity is stored without the terminator, so it must be adjusted for
1672 * the difference to keep the buffer size (capa + termlen) unchanged. */
1673static VALUE
1674heap_str_make_shared(VALUE klass, VALUE orig, int termlen)
1675{
1676 RUBY_ASSERT(!STR_EMBED_P(orig));
1677 RUBY_ASSERT(!STR_SHARED_P(orig));
1679
1680 VALUE str = str_alloc_heap(klass);
1681 STR_SET_LEN(str, RSTRING_LEN(orig));
1682 RSTRING(str)->as.heap.ptr = RSTRING_PTR(orig);
1683 RSTRING(str)->as.heap.aux.capa = RSTRING(orig)->as.heap.aux.capa + TERM_LEN(orig) - termlen;
1684 RBASIC(str)->flags |= RBASIC(orig)->flags & STR_NOFREE;
1685 RBASIC(orig)->flags &= ~STR_NOFREE;
1686 STR_SET_SHARED(orig, str);
1687 if (klass == 0)
1688 FL_UNSET_RAW(str, STR_BORROWED);
1689 return str;
1690}
1691
1692static VALUE
1693str_new_frozen_buffer(VALUE klass, VALUE orig, int copy_encoding)
1694{
1695 VALUE str;
1696
1697 long len = RSTRING_LEN(orig);
1698 rb_encoding *enc = copy_encoding ? STR_ENC_GET(orig) : rb_ascii8bit_encoding();
1699 int termlen = copy_encoding ? TERM_LEN(orig) : 1;
1700
1701 if (STR_EMBED_P(orig) || STR_EMBEDDABLE_P(len, termlen)) {
1702 str = str_enc_new(klass, RSTRING_PTR(orig), len, enc);
1703 RUBY_ASSERT(STR_EMBED_P(str));
1704 }
1705 else {
1706 if (FL_TEST_RAW(orig, STR_SHARED)) {
1707 VALUE shared = RSTRING(orig)->as.heap.aux.shared;
1708 long ofs = RSTRING(orig)->as.heap.ptr - RSTRING_PTR(shared);
1709 long rest = RSTRING_LEN(shared) - ofs - RSTRING_LEN(orig);
1710 RUBY_ASSERT(ofs >= 0);
1711 RUBY_ASSERT(rest >= 0);
1712 RUBY_ASSERT(ofs + rest <= RSTRING_LEN(shared));
1714
1715 if ((ofs > 0) || (rest > 0) ||
1716 (klass != RBASIC(shared)->klass) ||
1717 ENCODING_GET(shared) != ENCODING_GET(orig)) {
1718 str = str_new_shared(klass, shared);
1719 RUBY_ASSERT(!STR_EMBED_P(str));
1720 RSTRING(str)->as.heap.ptr += ofs;
1721 STR_SET_LEN(str, RSTRING_LEN(str) - (ofs + rest));
1722 }
1723 else {
1724 if (RBASIC_CLASS(shared) == 0)
1725 FL_SET_RAW(shared, STR_BORROWED);
1726 return shared;
1727 }
1728 }
1729 else if (STR_EMBEDDABLE_P(RSTRING_LEN(orig), TERM_LEN(orig))) {
1730 str = str_alloc_embed(klass, RSTRING_LEN(orig) + TERM_LEN(orig));
1731 STR_SET_EMBED(str);
1732 memcpy(RSTRING_PTR(str), RSTRING_PTR(orig), RSTRING_LEN(orig));
1733 STR_SET_LEN(str, RSTRING_LEN(orig));
1734 ENC_CODERANGE_SET(str, ENC_CODERANGE(orig));
1735 TERM_FILL(RSTRING_END(str), TERM_LEN(orig));
1736 }
1737 else {
1738 if (RB_OBJ_SHAREABLE_P(orig)) {
1739 str = str_new(klass, RSTRING_PTR(orig), RSTRING_LEN(orig));
1740 }
1741 else {
1742 str = heap_str_make_shared(klass, orig, termlen);
1743 }
1744 }
1745 }
1746
1747 if (copy_encoding) rb_enc_cr_str_exact_copy(str, orig);
1748 OBJ_FREEZE(str);
1749 return str;
1750}
1751
1752VALUE
1753rb_str_new_with_class(VALUE obj, const char *ptr, long len)
1754{
1755 return str_enc_new(rb_obj_class(obj), ptr, len, STR_ENC_GET(obj));
1756}
1757
1758static VALUE
1759str_new_empty_String(VALUE str)
1760{
1761 VALUE v = rb_str_new(0, 0);
1762 rb_enc_copy(v, str);
1763 return v;
1764}
1765
1766#define STR_BUF_MIN_SIZE 63
1767
1768VALUE
1770{
1771 if (STR_EMBEDDABLE_P(capa, 1)) {
1772 return str_alloc_embed(rb_cString, capa + 1);
1773 }
1774
1775 VALUE str = str_alloc_heap(rb_cString);
1776
1777 RSTRING(str)->as.heap.aux.capa = capa;
1778 RSTRING(str)->as.heap.ptr = ALLOC_N(char, (size_t)capa + 1);
1779 RSTRING(str)->as.heap.ptr[0] = '\0';
1780
1781 return str;
1782}
1783
1784VALUE
1786{
1787 VALUE str;
1788 long len = strlen(ptr);
1789
1790 str = rb_str_buf_new(len);
1791 rb_str_buf_cat(str, ptr, len);
1792
1793 return str;
1794}
1795
1796VALUE
1798{
1799 return str_new(0, 0, len);
1800}
1801
1802void
1804{
1805 if (STR_EMBED_P(str)) {
1806 RB_DEBUG_COUNTER_INC(obj_str_embed);
1807 }
1808 else if (FL_TEST(str, STR_SHARED | STR_NOFREE)) {
1809 (void)RB_DEBUG_COUNTER_INC_IF(obj_str_shared, FL_TEST(str, STR_SHARED));
1810 (void)RB_DEBUG_COUNTER_INC_IF(obj_str_shared, FL_TEST(str, STR_NOFREE));
1811 }
1812 else {
1813 RB_DEBUG_COUNTER_INC(obj_str_ptr);
1814 SIZED_FREE_N(STR_HEAP_PTR(str), STR_HEAP_SIZE(str));
1815 }
1816}
1817
1818size_t
1819rb_str_memsize(VALUE str)
1820{
1821 if (FL_TEST(str, STR_NOEMBED|STR_SHARED|STR_NOFREE) == STR_NOEMBED) {
1822 return STR_HEAP_SIZE(str);
1823 }
1824 else {
1825 return 0;
1826 }
1827}
1828
1829VALUE
1831{
1832 return rb_convert_type_with_id(str, T_STRING, "String", idTo_str);
1833}
1834
1835static inline void str_discard(VALUE str);
1836static void str_shared_replace(VALUE str, VALUE str2);
1837
1838void
1840{
1841 if (str != str2) str_shared_replace(str, str2);
1842}
1843
1844static void
1845str_shared_replace(VALUE str, VALUE str2)
1846{
1847 rb_encoding *enc;
1848 int cr;
1849 int termlen;
1850
1851 RUBY_ASSERT(str2 != str);
1852 enc = STR_ENC_GET(str2);
1853 cr = ENC_CODERANGE(str2);
1854 str_discard(str);
1855 termlen = rb_enc_mbminlen(enc);
1856
1857 STR_SET_LEN(str, RSTRING_LEN(str2));
1858
1859 if (str_embed_capa(str) >= RSTRING_LEN(str2) + termlen) {
1860 STR_SET_EMBED(str);
1861 memcpy(RSTRING_PTR(str), RSTRING_PTR(str2), (size_t)RSTRING_LEN(str2) + termlen);
1862 }
1863 else {
1864 if (STR_EMBED_P(str2)) {
1865 RUBY_ASSERT(!FL_TEST(str2, STR_SHARED));
1866 long len = RSTRING_LEN(str2);
1867 RUBY_ASSERT(len + termlen <= str_embed_capa(str2));
1868
1869 char *new_ptr = ALLOC_N(char, len + termlen);
1870 memcpy(new_ptr, RSTRING(str2)->as.embed.ary, len + termlen);
1871 RSTRING(str2)->as.heap.ptr = new_ptr;
1872 STR_SET_LEN(str2, len);
1873 RSTRING(str2)->as.heap.aux.capa = len;
1874 STR_SET_NOEMBED(str2);
1875 }
1876
1877 STR_SET_NOEMBED(str);
1878 FL_UNSET(str, STR_SHARED);
1879 RSTRING(str)->as.heap.ptr = RSTRING_PTR(str2);
1880
1881 if (FL_TEST(str2, STR_SHARED)) {
1882 VALUE shared = RSTRING(str2)->as.heap.aux.shared;
1883 STR_SET_SHARED(str, shared);
1884 }
1885 else {
1886 RSTRING(str)->as.heap.aux.capa = RSTRING(str2)->as.heap.aux.capa;
1887 }
1888
1889 /* abandon str2 */
1890 STR_SET_EMBED(str2);
1891 RSTRING_PTR(str2)[0] = 0;
1892 STR_SET_LEN(str2, 0);
1893 }
1894
1895 // We used str2's termlen above so we set enc raw
1896 // to avoid adjusting it based on str1's old enc.
1897 rb_enc_raw_set(str, enc);
1898 ENC_CODERANGE_SET(str, cr);
1899}
1900
1901VALUE
1903{
1904 VALUE str;
1905
1906 if (RB_TYPE_P(obj, T_STRING)) {
1907 return obj;
1908 }
1909 str = rb_funcall(obj, idTo_s, 0);
1910 return rb_obj_as_string_result(str, obj);
1911}
1912
1913VALUE
1914rb_obj_as_string_result(VALUE str, VALUE obj)
1915{
1916 if (!RB_TYPE_P(str, T_STRING))
1917 return rb_any_to_s(obj);
1918 return str;
1919}
1920
1921static VALUE
1922str_replace(VALUE str, VALUE str2)
1923{
1924 long len;
1925
1926 len = RSTRING_LEN(str2);
1927 if (STR_SHARED_P(str2)) {
1928 VALUE shared = RSTRING(str2)->as.heap.aux.shared;
1930 STR_SET_NOEMBED(str);
1931 STR_SET_LEN(str, len);
1932 RSTRING(str)->as.heap.ptr = RSTRING_PTR(str2);
1933 STR_SET_SHARED(str, shared);
1934 rb_enc_cr_str_exact_copy(str, str2);
1935 }
1936 else {
1937 str_replace_shared(str, str2);
1938 }
1939
1940 return str;
1941}
1942
1943static inline VALUE
1944ec_str_alloc_embed(struct rb_execution_context_struct *ec, VALUE klass, size_t capa)
1945{
1946 size_t size = rb_str_embed_size(capa, 0);
1947 RUBY_ASSERT(size > 0);
1948 RUBY_ASSERT(rb_gc_size_allocatable_p(size));
1949
1950 EC_NEWOBJ_OF(str, struct RString, klass, T_STRING, size, ec);
1951
1952 str->len = 0;
1953
1954 return (VALUE)str;
1955}
1956
1957static inline VALUE
1958ec_str_alloc_heap(struct rb_execution_context_struct *ec, VALUE klass)
1959{
1960 EC_NEWOBJ_OF(str, struct RString, klass, T_STRING | STR_NOEMBED, sizeof(struct RString), ec);
1961
1962 str->as.heap.aux.capa = 0;
1963 str->as.heap.ptr = NULL;
1964
1965 return (VALUE)str;
1966}
1967
1968static inline void
1969str_duplicate_setup_encoding(VALUE str, VALUE dup, VALUE flags)
1970{
1971 int encidx = 0;
1972 if ((flags & ENCODING_MASK) == (ENCODING_INLINE_MAX<<ENCODING_SHIFT)) {
1973 encidx = rb_enc_get_index(str);
1974 flags &= ~ENCODING_MASK;
1975 }
1976 FL_SET_RAW(dup, flags & ~FL_FREEZE);
1977 if (encidx) rb_enc_associate_index(dup, encidx);
1978}
1979
1980static const VALUE flag_mask = ENC_CODERANGE_MASK | ENCODING_MASK | FL_FREEZE;
1981
1982static inline void
1983str_duplicate_setup_embed(VALUE klass, VALUE str, VALUE dup)
1984{
1985 VALUE flags = FL_TEST_RAW(str, flag_mask);
1986 long len = RSTRING_LEN(str);
1987
1988 RUBY_ASSERT(STR_EMBED_P(dup));
1989 RUBY_ASSERT(str_embed_capa(dup) >= len + TERM_LEN(str));
1990 MEMCPY(RSTRING(dup)->as.embed.ary, RSTRING(str)->as.embed.ary, char, len + TERM_LEN(str));
1991 STR_SET_LEN(dup, RSTRING_LEN(str));
1992 str_duplicate_setup_encoding(str, dup, flags);
1993}
1994
1995static inline void
1996str_duplicate_setup_heap(VALUE klass, VALUE str, VALUE dup)
1997{
1998 VALUE flags = FL_TEST_RAW(str, flag_mask);
1999 VALUE root = str;
2000 if (FL_TEST_RAW(str, STR_SHARED)) {
2001 root = RSTRING(str)->as.heap.aux.shared;
2002 }
2003 else if (UNLIKELY(!OBJ_FROZEN_RAW(str))) {
2004 root = str = str_new_frozen(klass, str);
2005 flags = FL_TEST_RAW(str, flag_mask);
2006 }
2007 RUBY_ASSERT(!STR_SHARED_P(root));
2009
2010 RSTRING(dup)->as.heap.ptr = RSTRING_PTR(str);
2011 FL_SET_RAW(dup, RSTRING_NOEMBED);
2012 STR_SET_SHARED(dup, root);
2013 flags |= RSTRING_NOEMBED | STR_SHARED;
2014
2015 STR_SET_LEN(dup, RSTRING_LEN(str));
2016 str_duplicate_setup_encoding(str, dup, flags);
2017}
2018
2019static inline VALUE
2020str_duplicate(VALUE klass, VALUE str)
2021{
2022 VALUE dup;
2023 if (STR_EMBED_P(str) && rb_str_embed_size(RSTRING_LEN(str), 1) <= STR_COPY_MAX_EMBED_SIZE) {
2024 dup = str_alloc_embed(klass, RSTRING_LEN(str) + TERM_LEN(str));
2025
2026 str_duplicate_setup_embed(klass, str, dup);
2027 }
2028 else {
2029 dup = str_alloc_heap(klass);
2030
2031 str_duplicate_setup_heap(klass, str, dup);
2032 }
2033
2034 return dup;
2035}
2036
2037VALUE
2039{
2040 return str_duplicate(rb_obj_class(str), str);
2041}
2042
2043/* :nodoc: */
2044VALUE
2045rb_str_dup_m(VALUE str)
2046{
2047 if (LIKELY(BARE_STRING_P(str))) {
2048 return str_duplicate(rb_cString, str);
2049 }
2050 else {
2051 return rb_obj_dup(str);
2052 }
2053}
2054
2055VALUE
2057{
2058 RUBY_DTRACE_CREATE_HOOK(STRING, RSTRING_LEN(str));
2059 return str_duplicate(rb_cString, str);
2060}
2061
2062VALUE
2063rb_ec_str_resurrect(struct rb_execution_context_struct *ec, VALUE str, bool chilled)
2064{
2065 RUBY_DTRACE_CREATE_HOOK(STRING, RSTRING_LEN(str));
2066 VALUE new_str, klass = rb_cString;
2067
2068 if (!(chilled && RTEST(rb_ivar_defined(str, id_debug_created_info))) && STR_EMBED_P(str)) {
2069 new_str = ec_str_alloc_embed(ec, klass, RSTRING_LEN(str) + TERM_LEN(str));
2070 str_duplicate_setup_embed(klass, str, new_str);
2071 }
2072 else {
2073 new_str = ec_str_alloc_heap(ec, klass);
2074 str_duplicate_setup_heap(klass, str, new_str);
2075 }
2076 if (chilled) {
2077 FL_SET_RAW(new_str, STR_CHILLED);
2078 }
2079 return new_str;
2080}
2081
2082#if USE_ZJIT
2083bool
2084rb_zjit_str_resurrect_fastpath(VALUE str, bool chilled, size_t *size_out,
2085 VALUE *flags_out,
2086 long *len_out, size_t *byte_size_out)
2087{
2088 if (chilled && RTEST(rb_ivar_defined(str, id_debug_created_info))) return false;
2089
2090 if (!STR_EMBED_P(str)) return false;
2091
2092 long len = RSTRING_LEN(str);
2093 long termlen = TERM_LEN(str);
2094 size_t size = rb_str_embed_size(len + termlen, 0);
2095 if (!rb_gc_size_allocatable_p(size)) return false;
2096
2097 VALUE flags = FL_TEST_RAW(str, flag_mask);
2098
2099 if ((flags & ENCODING_MASK) == ((VALUE)ENCODING_INLINE_MAX << ENCODING_SHIFT)) {
2100 return false;
2101 }
2102
2103 flags &= ~FL_FREEZE;
2104 flags |= T_STRING;
2105 if (chilled) flags |= STR_CHILLED;
2106
2107 *size_out = size;
2108 *flags_out = flags;
2109 *len_out = len;
2110 *byte_size_out = (size_t)(len + termlen);
2111 return true;
2112}
2113#endif
2114
2115VALUE
2116rb_str_with_debug_created_info(VALUE str, VALUE path, int line)
2117{
2118 VALUE debug_info = rb_ary_new_from_args(2, path, INT2FIX(line));
2119 if (OBJ_FROZEN_RAW(str)) str = rb_str_dup(str);
2120 rb_ivar_set(str, id_debug_created_info, rb_ary_freeze(debug_info));
2121 FL_SET_RAW(str, STR_CHILLED);
2122 return rb_str_freeze(str);
2123}
2124
2125/*
2126 * The documentation block below uses an include (instead of inline text)
2127 * because the included text has non-ASCII characters (which are not allowed in a C file).
2128 */
2129
2130/*
2131 *
2132 * call-seq:
2133 * String.new(string = ''.encode(Encoding::ASCII_8BIT) , **options) -> new_string
2134 *
2135 * :include: doc/string/new.rdoc
2136 *
2137 */
2138
2139static VALUE
2140rb_str_init(int argc, VALUE *argv, VALUE str)
2141{
2142 static ID keyword_ids[2];
2143 VALUE orig, opt, venc, vcapa;
2144 VALUE kwargs[2];
2145 rb_encoding *enc = 0;
2146 int n;
2147
2148 if (!keyword_ids[0]) {
2149 keyword_ids[0] = rb_id_encoding();
2150 CONST_ID(keyword_ids[1], "capacity");
2151 }
2152
2153 n = rb_scan_args(argc, argv, "01:", &orig, &opt);
2154 if (!NIL_P(opt)) {
2155 rb_get_kwargs(opt, keyword_ids, 0, 2, kwargs);
2156 venc = kwargs[0];
2157 vcapa = kwargs[1];
2158 if (!UNDEF_P(venc) && !NIL_P(venc)) {
2159 enc = rb_to_encoding(venc);
2160 }
2161 if (!UNDEF_P(vcapa) && !NIL_P(vcapa)) {
2162 long capa = NUM2LONG(vcapa);
2163 long len = 0;
2164 int termlen = enc ? rb_enc_mbminlen(enc) : 1;
2165
2166 if (capa < STR_BUF_MIN_SIZE) {
2167 capa = STR_BUF_MIN_SIZE;
2168 }
2169 if (n == 1) {
2170 StringValue(orig);
2171 len = RSTRING_LEN(orig);
2172 if (capa < len) {
2173 capa = len;
2174 }
2175 if (orig == str) n = 0;
2176 }
2177 str_modifiable(str);
2178 if (STR_EMBED_P(str) || FL_TEST(str, STR_SHARED|STR_NOFREE)) {
2179 /* make noembed always */
2180 const size_t size = (size_t)capa + termlen;
2181 const char *const old_ptr = RSTRING_PTR(str);
2182 const size_t osize = RSTRING_LEN(str) + TERM_LEN(str);
2183 char *new_ptr = ALLOC_N(char, size);
2184 if (STR_EMBED_P(str)) RUBY_ASSERT((long)osize <= str_embed_capa(str));
2185 memcpy(new_ptr, old_ptr, osize < size ? osize : size);
2186 FL_UNSET_RAW(str, STR_SHARED|STR_NOFREE);
2187 RSTRING(str)->as.heap.ptr = new_ptr;
2188 }
2189 else if (STR_HEAP_SIZE(str) != (size_t)capa + termlen) {
2190 SIZED_REALLOC_N(RSTRING(str)->as.heap.ptr, char,
2191 (size_t)capa + termlen, STR_HEAP_SIZE(str));
2192 }
2193 STR_SET_LEN(str, len);
2194 TERM_FILL(&RSTRING(str)->as.heap.ptr[len], termlen);
2195 if (n == 1) {
2196 memcpy(RSTRING(str)->as.heap.ptr, RSTRING_PTR(orig), len);
2197 rb_enc_cr_str_exact_copy(str, orig);
2198 }
2199 FL_SET(str, STR_NOEMBED);
2200 RSTRING(str)->as.heap.aux.capa = capa;
2201 }
2202 else if (n == 1) {
2203 rb_str_replace(str, orig);
2204 }
2205 if (enc) {
2206 rb_enc_associate(str, enc);
2208 }
2209 }
2210 else if (n == 1) {
2211 rb_str_replace(str, orig);
2212 }
2213 return str;
2214}
2215
2216/* :nodoc: */
2217static VALUE
2218rb_str_s_new(int argc, VALUE *argv, VALUE klass)
2219{
2220 if (klass != rb_cString) {
2221 return rb_class_new_instance_pass_kw(argc, argv, klass);
2222 }
2223
2224 static ID keyword_ids[2];
2225 VALUE orig, opt, encoding = Qnil, capacity = Qnil;
2226 VALUE kwargs[2];
2227 rb_encoding *enc = NULL;
2228
2229 int n = rb_scan_args(argc, argv, "01:", &orig, &opt);
2230 if (NIL_P(opt)) {
2231 return rb_class_new_instance_pass_kw(argc, argv, klass);
2232 }
2233
2234 keyword_ids[0] = rb_id_encoding();
2235 CONST_ID(keyword_ids[1], "capacity");
2236 rb_get_kwargs(opt, keyword_ids, 0, 2, kwargs);
2237 encoding = kwargs[0];
2238 capacity = kwargs[1];
2239
2240 if (n == 1) {
2241 orig = StringValue(orig);
2242 }
2243 else {
2244 orig = Qnil;
2245 }
2246
2247 if (UNDEF_P(encoding)) {
2248 if (!NIL_P(orig)) {
2249 encoding = rb_obj_encoding(orig);
2250 }
2251 }
2252
2253 if (!UNDEF_P(encoding)) {
2254 enc = rb_to_encoding(encoding);
2255 }
2256
2257 // If capacity is nil, we're basically just duping `orig`.
2258 if (UNDEF_P(capacity)) {
2259 if (NIL_P(orig)) {
2260 VALUE empty_str = str_new(klass, "", 0);
2261 if (enc) {
2262 rb_enc_associate(empty_str, enc);
2263 }
2264 return empty_str;
2265 }
2266 VALUE copy = str_duplicate(klass, orig);
2267 rb_enc_associate(copy, enc);
2268 ENC_CODERANGE_CLEAR(copy);
2269 return copy;
2270 }
2271
2272 long capa = 0;
2273 capa = NUM2LONG(capacity);
2274 if (capa < 0) {
2275 capa = 0;
2276 }
2277
2278 if (!NIL_P(orig)) {
2279 long orig_capa = rb_str_capacity(orig);
2280 if (orig_capa > capa) {
2281 capa = orig_capa;
2282 }
2283 }
2284
2285 VALUE str = str_enc_new(klass, NULL, capa, enc);
2286 STR_SET_LEN(str, 0);
2287 TERM_FILL(RSTRING_PTR(str), enc ? rb_enc_mbmaxlen(enc) : 1);
2288
2289 if (!NIL_P(orig)) {
2290 rb_str_buf_append(str, orig);
2291 }
2292
2293 return str;
2294}
2295
2296#ifdef NONASCII_MASK
2297#define is_utf8_lead_byte(c) (((c)&0xC0) != 0x80)
2298
2299/*
2300 * UTF-8 leading bytes have either 0xxxxxxx or 11xxxxxx
2301 * bit representation. (see https://en.wikipedia.org/wiki/UTF-8)
2302 * Therefore, the following pseudocode can detect UTF-8 leading bytes.
2303 *
2304 * if (!(byte & 0x80))
2305 * byte |= 0x40; // turn on bit6
2306 * return ((byte>>6) & 1); // bit6 represent whether this byte is leading or not.
2307 *
2308 * This function calculates whether a byte is leading or not for all bytes
2309 * in the argument word by concurrently using the above logic, and then
2310 * adds up the number of leading bytes in the word.
2311 */
2312static inline uintptr_t
2313count_utf8_lead_bytes_with_word(const uintptr_t *s)
2314{
2315 uintptr_t d = *s;
2316
2317 /* Transform so that bit0 indicates whether we have a UTF-8 leading byte or not. */
2318 d = (d>>6) | (~d>>7);
2319 d &= NONASCII_MASK >> 7;
2320
2321 /* Gather all bytes. */
2322#if defined(HAVE_BUILTIN___BUILTIN_POPCOUNT) && defined(__POPCNT__)
2323 /* use only if it can use POPCNT */
2324 return rb_popcount_intptr(d);
2325#else
2326 d += (d>>8);
2327 d += (d>>16);
2328# if SIZEOF_VOIDP == 8
2329 d += (d>>32);
2330# endif
2331 return (d&0xF);
2332#endif
2333}
2334#endif
2335
2336static inline long
2337enc_strlen(const char *p, const char *e, rb_encoding *enc, int cr)
2338{
2339 long c;
2340 const char *q;
2341
2342 if (rb_enc_mbmaxlen(enc) == rb_enc_mbminlen(enc)) {
2343 long diff = (long)(e - p);
2344 return diff / rb_enc_mbminlen(enc) + !!(diff % rb_enc_mbminlen(enc));
2345 }
2346#ifdef NONASCII_MASK
2347 else if (cr == ENC_CODERANGE_VALID && enc == rb_utf8_encoding()) {
2348 uintptr_t len = 0;
2349 if ((int)sizeof(uintptr_t) * 2 < e - p) {
2350 const uintptr_t *s, *t;
2351 const uintptr_t lowbits = sizeof(uintptr_t) - 1;
2352 s = (const uintptr_t*)(~lowbits & ((uintptr_t)p + lowbits));
2353 t = (const uintptr_t*)(~lowbits & (uintptr_t)e);
2354 while (p < (const char *)s) {
2355 if (is_utf8_lead_byte(*p)) len++;
2356 p++;
2357 }
2358 while (s < t) {
2359 len += count_utf8_lead_bytes_with_word(s);
2360 s++;
2361 }
2362 p = (const char *)s;
2363 }
2364 while (p < e) {
2365 if (is_utf8_lead_byte(*p)) len++;
2366 p++;
2367 }
2368 return (long)len;
2369 }
2370#endif
2371 else if (rb_enc_asciicompat(enc)) {
2372 c = 0;
2373 if (ENC_CODERANGE_CLEAN_P(cr)) {
2374 while (p < e) {
2375 q = search_nonascii(p, e);
2376 if (!q)
2377 return c + (e - p);
2378 c += q - p;
2379 p = q;
2380 p += rb_enc_fast_mbclen(p, e, enc);
2381 c++;
2382 }
2383 }
2384 else {
2385 while (p < e) {
2386 q = search_nonascii(p, e);
2387 if (!q)
2388 return c + (e - p);
2389 c += q - p;
2390 p = q;
2391 p += rb_enc_mbclen(p, e, enc);
2392 c++;
2393 }
2394 }
2395 return c;
2396 }
2397
2398 for (c=0; p<e; c++) {
2399 p += rb_enc_mbclen(p, e, enc);
2400 }
2401 return c;
2402}
2403
2404long
2405rb_enc_strlen(const char *p, const char *e, rb_encoding *enc)
2406{
2407 return enc_strlen(p, e, enc, ENC_CODERANGE_UNKNOWN);
2408}
2409
2410/* To get strlen with cr
2411 * Note that given cr is not used.
2412 */
2413long
2414rb_enc_strlen_cr(const char *p, const char *e, rb_encoding *enc, int *cr)
2415{
2416 long c;
2417 const char *q;
2418 int ret;
2419
2420 *cr = 0;
2421 if (rb_enc_mbmaxlen(enc) == rb_enc_mbminlen(enc)) {
2422 long diff = (long)(e - p);
2423 return diff / rb_enc_mbminlen(enc) + !!(diff % rb_enc_mbminlen(enc));
2424 }
2425 else if (rb_enc_asciicompat(enc)) {
2426 c = 0;
2427 while (p < e) {
2428 q = search_nonascii(p, e);
2429 if (!q) {
2430 if (!*cr) *cr = ENC_CODERANGE_7BIT;
2431 return c + (e - p);
2432 }
2433 c += q - p;
2434 p = q;
2435 ret = rb_enc_precise_mbclen(p, e, enc);
2436 if (MBCLEN_CHARFOUND_P(ret)) {
2437 *cr |= ENC_CODERANGE_VALID;
2438 p += MBCLEN_CHARFOUND_LEN(ret);
2439 }
2440 else {
2442 p++;
2443 }
2444 c++;
2445 }
2446 if (!*cr) *cr = ENC_CODERANGE_7BIT;
2447 return c;
2448 }
2449
2450 for (c=0; p<e; c++) {
2451 ret = rb_enc_precise_mbclen(p, e, enc);
2452 if (MBCLEN_CHARFOUND_P(ret)) {
2453 *cr |= ENC_CODERANGE_VALID;
2454 p += MBCLEN_CHARFOUND_LEN(ret);
2455 }
2456 else {
2458 if (p + rb_enc_mbminlen(enc) <= e)
2459 p += rb_enc_mbminlen(enc);
2460 else
2461 p = e;
2462 }
2463 }
2464 if (!*cr) *cr = ENC_CODERANGE_7BIT;
2465 return c;
2466}
2467
2468/* enc must be str's enc or rb_enc_check(str, str2) */
2469static long
2470str_strlen(VALUE str, rb_encoding *enc)
2471{
2472 const char *p, *e;
2473 int cr;
2474
2475 if (single_byte_optimizable(str)) return RSTRING_LEN(str);
2476 if (!enc) enc = STR_ENC_GET(str);
2477 p = RSTRING_PTR(str);
2478 e = RSTRING_END(str);
2479 cr = ENC_CODERANGE(str);
2480
2481 if (cr == ENC_CODERANGE_UNKNOWN) {
2482 long n = rb_enc_strlen_cr(p, e, enc, &cr);
2483 if (cr) ENC_CODERANGE_SET(str, cr);
2484 return n;
2485 }
2486 else {
2487 return enc_strlen(p, e, enc, cr);
2488 }
2489}
2490
2491long
2493{
2494 return str_strlen(str, NULL);
2495}
2496
2497/*
2498 * call-seq:
2499 * length -> integer
2500 *
2501 * :include: doc/string/length.rdoc
2502 *
2503 */
2504
2505VALUE
2507{
2508 return LONG2NUM(str_strlen(str, NULL));
2509}
2510
2511/*
2512 * call-seq:
2513 * bytesize -> integer
2514 *
2515 * :include: doc/string/bytesize.rdoc
2516 *
2517 */
2518
2519VALUE
2520rb_str_bytesize(VALUE str)
2521{
2522 return LONG2NUM(RSTRING_LEN(str));
2523}
2524
2525/*
2526 * call-seq:
2527 * empty? -> true or false
2528 *
2529 * Returns whether the length of +self+ is zero:
2530 *
2531 * 'hello'.empty? # => false
2532 * ' '.empty? # => false
2533 * ''.empty? # => true
2534 *
2535 * Related: see {Querying}[rdoc-ref:String@Querying].
2536 */
2537
2538static VALUE
2539rb_str_empty(VALUE str)
2540{
2541 return RBOOL(RSTRING_LEN(str) == 0);
2542}
2543
2544/*
2545 * call-seq:
2546 * self + other_string -> new_string
2547 *
2548 * Returns a new string containing +other_string+ concatenated to +self+:
2549 *
2550 * 'Hello from ' + self.to_s # => "Hello from main"
2551 *
2552 * Related: see {Converting to New String}[rdoc-ref:String@Converting+to+New+String].
2553 */
2554
2555VALUE
2557{
2558 VALUE str3;
2559 rb_encoding *enc;
2560 const char *ptr1, *ptr2;
2561 char *ptr3;
2562 long len1, len2;
2563 int termlen;
2564
2565 StringValue(str2);
2566 enc = rb_enc_check_str(str1, str2);
2567 RSTRING_GETMEM(str1, ptr1, len1);
2568 RSTRING_GETMEM(str2, ptr2, len2);
2569 termlen = rb_enc_mbminlen(enc);
2570 if (len1 > LONG_MAX - len2) {
2571 rb_raise(rb_eArgError, "string size too big");
2572 }
2573 str3 = str_enc_new(rb_cString, 0, len1+len2, enc);
2574 ptr3 = RSTRING_PTR(str3);
2575 memcpy(ptr3, ptr1, len1);
2576 memcpy(ptr3+len1, ptr2, len2);
2577 TERM_FILL(&ptr3[len1+len2], termlen);
2578
2579 ENCODING_CODERANGE_SET(str3, rb_enc_to_index(enc),
2581 RB_GC_GUARD(str1);
2582 RB_GC_GUARD(str2);
2583 return str3;
2584}
2585
2586/* A variant of rb_str_plus that does not raise but return Qundef instead. */
2587VALUE
2588rb_str_opt_plus(VALUE str1, VALUE str2)
2589{
2592 long len1, len2;
2593 MAYBE_UNUSED(char) *ptr1, *ptr2;
2594 RSTRING_GETMEM(str1, ptr1, len1);
2595 RSTRING_GETMEM(str2, ptr2, len2);
2596 int enc1 = rb_enc_get_index(str1);
2597 int enc2 = rb_enc_get_index(str2);
2598
2599 if (enc1 < 0) {
2600 return Qundef;
2601 }
2602 else if (enc2 < 0) {
2603 return Qundef;
2604 }
2605 else if (enc1 != enc2) {
2606 return Qundef;
2607 }
2608 else if (len1 > LONG_MAX - len2) {
2609 return Qundef;
2610 }
2611 else {
2612 return rb_str_plus(str1, str2);
2613 }
2614
2615}
2616
2617/*
2618 * call-seq:
2619 * self * n -> new_string
2620 *
2621 * Returns a new string containing +n+ copies of +self+:
2622 *
2623 * 'Ho!' * 3 # => "Ho!Ho!Ho!"
2624 * 'No!' * 0 # => ""
2625 *
2626 * Related: see {Converting to New String}[rdoc-ref:String@Converting+to+New+String].
2627 */
2628
2629VALUE
2631{
2632 VALUE str2;
2633 long n, len;
2634 char *ptr2;
2635 int termlen;
2636
2637 if (times == INT2FIX(1)) {
2638 return str_duplicate(rb_cString, str);
2639 }
2640 if (times == INT2FIX(0)) {
2641 str2 = str_alloc_embed(rb_cString, 0);
2642 rb_enc_copy(str2, str);
2643 return str2;
2644 }
2645 len = NUM2LONG(times);
2646 if (len < 0) {
2647 rb_raise(rb_eArgError, "negative argument");
2648 }
2649 if (RSTRING_LEN(str) == 1 && RSTRING_PTR(str)[0] == 0) {
2650 if (STR_EMBEDDABLE_P(len, 1)) {
2651 str2 = str_alloc_embed(rb_cString, len + 1);
2652 memset(RSTRING_PTR(str2), 0, len + 1);
2653 }
2654 else {
2655 str2 = str_alloc_heap(rb_cString);
2656 RSTRING(str2)->as.heap.aux.capa = len;
2657 RSTRING(str2)->as.heap.ptr = ZALLOC_N(char, (size_t)len + 1);
2658 }
2659 STR_SET_LEN(str2, len);
2660 rb_enc_copy(str2, str);
2661 return str2;
2662 }
2663 if (len && LONG_MAX/len < RSTRING_LEN(str)) {
2664 rb_raise(rb_eArgError, "argument too big");
2665 }
2666
2667 len *= RSTRING_LEN(str);
2668 termlen = TERM_LEN(str);
2669 str2 = str_enc_new(rb_cString, 0, len, STR_ENC_GET(str));
2670 ptr2 = RSTRING_PTR(str2);
2671 if (len) {
2672 n = RSTRING_LEN(str);
2673 memcpy(ptr2, RSTRING_PTR(str), n);
2674 while (n <= len/2) {
2675 memcpy(ptr2 + n, ptr2, n);
2676 n *= 2;
2677 }
2678 memcpy(ptr2 + n, ptr2, len-n);
2679 }
2680 STR_SET_LEN(str2, len);
2681 TERM_FILL(&ptr2[len], termlen);
2682 rb_enc_cr_str_copy_for_substr(str2, str);
2683
2684 return str2;
2685}
2686
2687/*
2688 * call-seq:
2689 * self % object -> new_string
2690 *
2691 * Returns the result of formatting +object+ into the format specifications
2692 * contained in +self+
2693 * (see {Format Specifications}[rdoc-ref:language/format_specifications.rdoc]):
2694 *
2695 * '%05d' % 123 # => "00123"
2696 *
2697 * If +self+ contains multiple format specifications,
2698 * +object+ must be an array or hash containing the objects to be formatted:
2699 *
2700 * '%-5s: %016x' % [ 'ID', self.object_id ] # => "ID : 00002b054ec93168"
2701 * 'foo = %{foo}' % {foo: 'bar'} # => "foo = bar"
2702 * 'foo = %{foo}, baz = %{baz}' % {foo: 'bar', baz: 'bat'} # => "foo = bar, baz = bat"
2703 *
2704 * Related: see {Converting to New String}[rdoc-ref:String@Converting+to+New+String].
2705 */
2706
2707static VALUE
2708rb_str_format_m(VALUE str, VALUE arg)
2709{
2710 VALUE tmp = rb_check_array_type(arg);
2711
2712 if (!NIL_P(tmp)) {
2713 VALUE result = rb_str_format_ary(RARRAY_LENINT(tmp), RARRAY_CONST_PTR(tmp), str, tmp);
2714 RB_GC_GUARD(tmp);
2715 return result;
2716 }
2717 return rb_str_format(1, &arg, str);
2718}
2719
2720static inline void
2721rb_check_lockedtmp(VALUE str)
2722{
2723 if (FL_TEST(str, STR_TMPLOCK)) {
2724 rb_raise(rb_eRuntimeError, "can't modify string; temporarily locked");
2725 }
2726}
2727
2728// If none of these flags are set, we know we have an modifiable string.
2729// If any is set, we need to do more detailed checks.
2730#define STR_UNMODIFIABLE_MASK (FL_FREEZE | STR_TMPLOCK | STR_CHILLED)
2731static inline void
2732str_modifiable(VALUE str)
2733{
2734 RUBY_ASSERT(ruby_thread_has_gvl_p());
2735
2736 if (RB_UNLIKELY(FL_ANY_RAW(str, STR_UNMODIFIABLE_MASK))) {
2737 if (CHILLED_STRING_P(str)) {
2738 CHILLED_STRING_MUTATED(str);
2739 }
2740 rb_check_lockedtmp(str);
2741 rb_check_frozen(str);
2742 }
2743}
2744
2745static inline int
2746str_dependent_p(VALUE str)
2747{
2748 if (STR_EMBED_P(str) || !FL_TEST(str, STR_SHARED|STR_NOFREE)) {
2749 return FALSE;
2750 }
2751 else {
2752 return TRUE;
2753 }
2754}
2755
2756// If none of these flags are set, we know we have an independent string.
2757// If any is set, we need to do more detailed checks.
2758#define STR_DEPENDANT_MASK (STR_UNMODIFIABLE_MASK | STR_SHARED | STR_NOFREE)
2759static inline int
2760str_independent(VALUE str)
2761{
2762 RUBY_ASSERT(ruby_thread_has_gvl_p());
2763
2764 if (RB_UNLIKELY(FL_ANY_RAW(str, STR_DEPENDANT_MASK))) {
2765 str_modifiable(str);
2766 return !str_dependent_p(str);
2767 }
2768 return TRUE;
2769}
2770
2771static void
2772str_make_independent_expand(VALUE str, long len, long expand, const int termlen)
2773{
2774 RUBY_ASSERT(ruby_thread_has_gvl_p());
2775
2776 char *ptr;
2777 char *oldptr;
2778 long capa = len + expand;
2779
2780 if (len > capa) len = capa;
2781
2782 if (!STR_EMBED_P(str) && str_embed_capa(str) >= capa + termlen) {
2783 ptr = RSTRING(str)->as.heap.ptr;
2784 STR_SET_EMBED(str);
2785 memcpy(RSTRING(str)->as.embed.ary, ptr, len);
2786 TERM_FILL(RSTRING(str)->as.embed.ary + len, termlen);
2787 STR_SET_LEN(str, len);
2788 return;
2789 }
2790
2791 ptr = ALLOC_N(char, (size_t)capa + termlen);
2792 oldptr = RSTRING_PTR(str);
2793 if (oldptr) {
2794 memcpy(ptr, oldptr, len);
2795 }
2796 if (FL_TEST_RAW(str, STR_NOEMBED|STR_NOFREE|STR_SHARED) == STR_NOEMBED) {
2797 SIZED_FREE_N(oldptr, STR_HEAP_SIZE(str));
2798 }
2799 STR_SET_NOEMBED(str);
2800 FL_UNSET(str, STR_SHARED|STR_NOFREE);
2801 TERM_FILL(ptr + len, termlen);
2802 RSTRING(str)->as.heap.ptr = ptr;
2803 STR_SET_LEN(str, len);
2804 RSTRING(str)->as.heap.aux.capa = capa;
2805}
2806
2807void
2808rb_str_modify(VALUE str)
2809{
2810 if (!str_independent(str))
2811 str_make_independent(str);
2813}
2814
2815void
2817{
2818 RUBY_ASSERT(ruby_thread_has_gvl_p());
2819
2820 int termlen = TERM_LEN(str);
2821 long len = RSTRING_LEN(str);
2822
2823 if (expand < 0) {
2824 rb_raise(rb_eArgError, "negative expanding string size");
2825 }
2826 if (expand >= LONG_MAX - len) {
2827 rb_raise(rb_eArgError, "string size too big");
2828 }
2829
2830 if (!str_independent(str)) {
2831 str_make_independent_expand(str, len, expand, termlen);
2832 }
2833 else if (expand > 0) {
2834 RESIZE_CAPA_TERM(str, len + expand, termlen);
2835 }
2837}
2838
2839/* As rb_str_modify(), but don't clear coderange */
2840static void
2841str_modify_keep_cr(VALUE str)
2842{
2843 if (!str_independent(str))
2844 str_make_independent(str);
2846 /* Force re-scan later */
2848}
2849
2850static inline void
2851str_discard(VALUE str)
2852{
2853 str_modifiable(str);
2854 if (!STR_EMBED_P(str) && !FL_TEST(str, STR_SHARED|STR_NOFREE)) {
2855 SIZED_FREE_N(STR_HEAP_PTR(str), STR_HEAP_SIZE(str));
2856 RSTRING(str)->as.heap.ptr = 0;
2857 STR_SET_LEN(str, 0);
2858 }
2859}
2860
2861void
2863{
2864 int encindex = rb_enc_get_index(str);
2865
2866 if (RB_UNLIKELY(encindex == -1)) {
2867 rb_raise(rb_eTypeError, "not encoding capable object");
2868 }
2869
2870 if (RB_LIKELY(rb_str_encindex_fastpath(encindex))) {
2871 return;
2872 }
2873
2874 rb_encoding *enc = rb_enc_from_index(encindex);
2875 if (!rb_enc_asciicompat(enc)) {
2876 rb_raise(rb_eEncCompatError, "ASCII incompatible encoding: %s", rb_enc_name(enc));
2877 }
2878}
2879
2880VALUE
2882{
2883 RUBY_ASSERT(ruby_thread_has_gvl_p());
2884
2885 VALUE s = *ptr;
2886 if (!RB_TYPE_P(s, T_STRING)) {
2887 s = rb_str_to_str(s);
2888 *ptr = s;
2889 }
2890 return s;
2891}
2892
2893char *
2895{
2896 VALUE str = rb_string_value(ptr);
2897 return RSTRING_PTR(str);
2898}
2899
2900static const char *
2901str_null_char(const char *s, long len, const int minlen, rb_encoding *enc)
2902{
2903 const char *e = s + len;
2904
2905 for (; s + minlen <= e; s += rb_enc_mbclen(s, e, enc)) {
2906 if (zero_filled(s, minlen)) return s;
2907 }
2908 return 0;
2909}
2910
2911static char *
2912str_fill_term(VALUE str, char *s, long len, int termlen)
2913{
2914 /* This function assumes that (capa + termlen) bytes of memory
2915 * is allocated, like many other functions in this file.
2916 */
2917 if (str_dependent_p(str)) {
2918 if (!zero_filled(s + len, termlen))
2919 str_make_independent_expand(str, len, 0L, termlen);
2920 }
2921 else {
2922 TERM_FILL(s + len, termlen);
2923 return s;
2924 }
2925 return RSTRING_PTR(str);
2926}
2927
2928void
2929rb_str_change_terminator_length(VALUE str, const int oldtermlen, const int termlen)
2930{
2931 long capa = str_capacity(str, oldtermlen) + oldtermlen;
2932 long len = RSTRING_LEN(str);
2933
2934 RUBY_ASSERT(capa >= len);
2935 if (capa - len < termlen) {
2936 rb_check_lockedtmp(str);
2937 str_make_independent_expand(str, len, 0L, termlen);
2938 }
2939 else if (str_dependent_p(str)) {
2940 if (termlen > oldtermlen)
2941 str_make_independent_expand(str, len, 0L, termlen);
2942 }
2943 else {
2944 if (!STR_EMBED_P(str)) {
2945 /* modify capa instead of realloc */
2946 RUBY_ASSERT(!FL_TEST((str), STR_SHARED));
2947 RSTRING(str)->as.heap.aux.capa = capa - termlen;
2948 }
2949 if (termlen > oldtermlen) {
2950 TERM_FILL(RSTRING_PTR(str) + len, termlen);
2951 }
2952 }
2953
2954 return;
2955}
2956
2957static char *
2958str_null_check(VALUE str, int *w)
2959{
2960 char *s = RSTRING_PTR(str);
2961 long len = RSTRING_LEN(str);
2962 int minlen = 1;
2963
2964 if (RB_UNLIKELY(!rb_str_enc_fastpath(str))) {
2965 rb_encoding *enc = rb_str_enc_get(str);
2966 minlen = rb_enc_mbminlen(enc);
2967
2968 if (minlen > 1) {
2969 *w = 1;
2970 if (str_null_char(s, len, minlen, enc)) {
2971 return NULL;
2972 }
2973 return str_fill_term(str, s, len, minlen);
2974 }
2975 }
2976
2977 *w = 0;
2978 if (!s || memchr(s, 0, len)) {
2979 return NULL;
2980 }
2981 if (s[len]) {
2982 s = str_fill_term(str, s, len, minlen);
2983 }
2984 return s;
2985}
2986
2987static char *str_to_cstr(VALUE str);
2988
2989const char *
2990rb_str_null_check(VALUE str)
2991{
2993
2994 const char *s;
2995 long len;
2996 RSTRING_GETMEM(str, s, len);
2997
2998 if (RB_LIKELY(rb_str_enc_fastpath(str))) {
2999 if (!s || memchr(s, 0, len)) {
3000 rb_raise(rb_eArgError, "string contains null byte");
3001 }
3002 }
3003 else {
3004 str_to_cstr(str);
3005 }
3006
3007 return s;
3008}
3009
3010char *
3011rb_str_to_cstr(VALUE str)
3012{
3013 int w;
3014 return str_null_check(str, &w);
3015}
3016
3017char *
3019{
3020 VALUE str = rb_string_value(ptr);
3021 return str_to_cstr(str);
3022}
3023
3024static char *
3025str_to_cstr(VALUE str)
3026{
3027 int w;
3028 char *s = str_null_check(str, &w);
3029 if (!s) {
3030 if (w) {
3031 rb_raise(rb_eArgError, "string contains null char");
3032 }
3033 rb_raise(rb_eArgError, "string contains null byte");
3034 }
3035 return s;
3036}
3037
3038char *
3039rb_str_fill_terminator(VALUE str, const int newminlen)
3040{
3041 char *s = RSTRING_PTR(str);
3042 long len = RSTRING_LEN(str);
3043 return str_fill_term(str, s, len, newminlen);
3044}
3045
3046VALUE
3048{
3049 str = rb_check_convert_type_with_id(str, T_STRING, "String", idTo_str);
3050 return str;
3051}
3052
3053/*
3054 * call-seq:
3055 * String.try_convert(object) -> object, new_string, or nil
3056 *
3057 * Attempts to convert the given +object+ to a string.
3058 *
3059 * If +object+ is already a string, returns +object+, unmodified.
3060 *
3061 * Otherwise if +object+ responds to <tt>:to_str</tt>,
3062 * calls <tt>object.to_str</tt> and returns the result.
3063 *
3064 * Returns +nil+ if +object+ does not respond to <tt>:to_str</tt>.
3065 *
3066 * Raises an exception unless <tt>object.to_str</tt> returns a string.
3067 */
3068static VALUE
3069rb_str_s_try_convert(VALUE dummy, VALUE str)
3070{
3071 return rb_check_string_type(str);
3072}
3073
3074static char*
3075str_nth_len(const char *p, const char *e, long *nthp, rb_encoding *enc)
3076{
3077 long nth = *nthp;
3078 if (rb_enc_mbmaxlen(enc) == 1) {
3079 p += nth;
3080 }
3081 else if (rb_enc_mbmaxlen(enc) == rb_enc_mbminlen(enc)) {
3082 p += nth * rb_enc_mbmaxlen(enc);
3083 }
3084 else if (rb_enc_asciicompat(enc)) {
3085 const char *p2, *e2;
3086 int n;
3087
3088 while (p < e && 0 < nth) {
3089 e2 = p + nth;
3090 if (e < e2) {
3091 *nthp = nth;
3092 return (char *)e;
3093 }
3094 p2 = search_nonascii(p, e2);
3095 if (!p2) {
3096 nth -= e2 - p;
3097 *nthp = nth;
3098 return (char *)e2;
3099 }
3100 nth -= p2 - p;
3101 p = p2;
3102 n = rb_enc_mbclen(p, e, enc);
3103 p += n;
3104 nth--;
3105 }
3106 *nthp = nth;
3107 if (nth != 0) {
3108 return (char *)e;
3109 }
3110 return (char *)p;
3111 }
3112 else {
3113 while (p < e && nth--) {
3114 p += rb_enc_mbclen(p, e, enc);
3115 }
3116 }
3117 if (p > e) p = e;
3118 *nthp = nth;
3119 return (char*)p;
3120}
3121
3122char*
3123rb_enc_nth(const char *p, const char *e, long nth, rb_encoding *enc)
3124{
3125 return str_nth_len(p, e, &nth, enc);
3126}
3127
3128static char*
3129str_nth(const char *p, const char *e, long nth, rb_encoding *enc, int singlebyte)
3130{
3131 if (singlebyte)
3132 p += nth;
3133 else {
3134 p = str_nth_len(p, e, &nth, enc);
3135 }
3136 if (!p) return 0;
3137 if (p > e) p = e;
3138 return (char *)p;
3139}
3140
3141/* char offset to byte offset */
3142static long
3143str_offset(const char *p, const char *e, long nth, rb_encoding *enc, int singlebyte)
3144{
3145 const char *pp = str_nth(p, e, nth, enc, singlebyte);
3146 if (!pp) return e - p;
3147 return pp - p;
3148}
3149
3150long
3151rb_str_offset(VALUE str, long pos)
3152{
3153 return str_offset(RSTRING_PTR(str), RSTRING_END(str), pos,
3154 STR_ENC_GET(str), single_byte_optimizable(str));
3155}
3156
3157#ifdef NONASCII_MASK
3158static char *
3159str_utf8_nth(const char *p, const char *e, long *nthp)
3160{
3161 long nth = *nthp;
3162 if ((int)SIZEOF_VOIDP * 2 < e - p && (int)SIZEOF_VOIDP * 2 < nth) {
3163 const uintptr_t *s, *t;
3164 const uintptr_t lowbits = SIZEOF_VOIDP - 1;
3165 s = (const uintptr_t*)(~lowbits & ((uintptr_t)p + lowbits));
3166 t = (const uintptr_t*)(~lowbits & (uintptr_t)e);
3167 while (p < (const char *)s) {
3168 if (is_utf8_lead_byte(*p)) nth--;
3169 p++;
3170 }
3171 do {
3172 nth -= count_utf8_lead_bytes_with_word(s);
3173 s++;
3174 } while (s < t && (int)SIZEOF_VOIDP <= nth);
3175 p = (char *)s;
3176 }
3177 while (p < e) {
3178 if (is_utf8_lead_byte(*p)) {
3179 if (nth == 0) break;
3180 nth--;
3181 }
3182 p++;
3183 }
3184 *nthp = nth;
3185 return (char *)p;
3186}
3187
3188static long
3189str_utf8_offset(const char *p, const char *e, long nth)
3190{
3191 const char *pp = str_utf8_nth(p, e, &nth);
3192 return pp - p;
3193}
3194#endif
3195
3196/* byte offset to char offset */
3197long
3198rb_str_sublen(VALUE str, long pos)
3199{
3200 if (single_byte_optimizable(str) || pos < 0)
3201 return pos;
3202 else {
3203 const char *p = RSTRING_PTR(str);
3204 return enc_strlen(p, p + pos, STR_ENC_GET(str), ENC_CODERANGE(str));
3205 }
3206}
3207
3208static VALUE
3209str_subseq(VALUE str, long beg, long len)
3210{
3211 VALUE str2;
3212
3213 RUBY_ASSERT(beg >= 0);
3214 RUBY_ASSERT(len >= 0);
3215 RUBY_ASSERT(beg+len <= RSTRING_LEN(str));
3216
3217 const int termlen = TERM_LEN(str);
3218 if (!SHARABLE_SUBSTRING_P(str, beg, len)) {
3219 str2 = rb_enc_str_new(RSTRING_PTR(str) + beg, len, rb_str_enc_get(str));
3220 if (ENC_CODERANGE(str) == ENC_CODERANGE_7BIT) {
3222 }
3223 RB_GC_GUARD(str);
3224 return str2;
3225 }
3226
3227 /* Sharing allocates a shared root as well unless str can be one itself, so
3228 * a copy is worth a larger slot only when it saves that second object. */
3229 const bool root_available = STR_SHARED_P(str) ||
3230 RB_FL_TEST_RAW(str, FL_FREEZE | STR_CHILLED) == FL_FREEZE;
3231 const size_t max_embed_size = root_available ?
3232 rb_gc_size_slot_size(sizeof(struct RString)) : STR_COPY_MAX_EMBED_SIZE;
3233 const size_t embed_size = rb_str_embed_size(len, termlen);
3234
3235 if (embed_size <= max_embed_size && rb_gc_size_allocatable_p(embed_size)) {
3236 str2 = str_alloc_embed(rb_cString, len + termlen);
3237 char *ptr2 = RSTRING(str2)->as.embed.ary;
3238 memcpy(ptr2, RSTRING_PTR(str) + beg, len);
3239 TERM_FILL(ptr2 + len, termlen);
3240
3241 STR_SET_LEN(str2, len);
3242 if (ENC_CODERANGE(str) == ENC_CODERANGE_7BIT) {
3244 }
3245
3246 RB_GC_GUARD(str);
3247 }
3248 else {
3249 str2 = str_alloc_heap(rb_cString);
3250 str_replace_shared(str2, str);
3251 RUBY_ASSERT(!STR_EMBED_P(str2));
3252 if (ENC_CODERANGE(str) != ENC_CODERANGE_7BIT) {
3253 ENC_CODERANGE_CLEAR(str2);
3254 }
3255
3256 RSTRING(str2)->as.heap.ptr += beg;
3257 if (RSTRING_LEN(str2) > len) {
3258 STR_SET_LEN(str2, len);
3259 }
3260 }
3261
3262 return str2;
3263}
3264
3265VALUE
3266rb_str_subseq(VALUE str, long beg, long len)
3267{
3268 VALUE str2 = str_subseq(str, beg, len);
3269 rb_enc_cr_str_copy_for_substr(str2, str);
3270 return str2;
3271}
3272
3273char *
3274rb_str_subpos(VALUE str, long beg, long *lenp)
3275{
3276 long len = *lenp;
3277 long slen = -1L;
3278 const long blen = RSTRING_LEN(str);
3279 rb_encoding *enc = STR_ENC_GET(str);
3280 const char *p, *s = RSTRING_PTR(str), *e = s + blen;
3281
3282 if (len < 0) return 0;
3283 if (beg < 0 && -beg < 0) return 0;
3284 if (!blen) {
3285 len = 0;
3286 }
3287 if (single_byte_optimizable(str)) {
3288 if (beg > blen) return 0;
3289 if (beg < 0) {
3290 beg += blen;
3291 if (beg < 0) return 0;
3292 }
3293 if (len > blen - beg)
3294 len = blen - beg;
3295 if (len < 0) return 0;
3296 p = s + beg;
3297 goto end;
3298 }
3299 if (beg < 0) {
3300 if (len > -beg) len = -beg;
3301 if ((ENC_CODERANGE(str) == ENC_CODERANGE_VALID) &&
3302 (-beg * rb_enc_mbmaxlen(enc) < blen / 8)) {
3303 beg = -beg;
3304 while (beg-- > len && (e = rb_enc_prev_char(s, e, e, enc)) != 0);
3305 p = e;
3306 if (!p) return 0;
3307 while (len-- > 0 && (p = rb_enc_prev_char(s, p, e, enc)) != 0);
3308 if (!p) return 0;
3309 len = e - p;
3310 goto end;
3311 }
3312 else {
3313 slen = str_strlen(str, enc);
3314 beg += slen;
3315 if (beg < 0) return 0;
3316 p = s + beg;
3317 if (len == 0) goto end;
3318 }
3319 }
3320 else if (beg > 0 && beg > blen) {
3321 return 0;
3322 }
3323 if (len == 0) {
3324 if (beg > str_strlen(str, enc)) return 0; /* str's enc */
3325 p = s + beg;
3326 }
3327#ifdef NONASCII_MASK
3328 else if (ENC_CODERANGE(str) == ENC_CODERANGE_VALID &&
3329 enc == rb_utf8_encoding()) {
3330 p = str_utf8_nth(s, e, &beg);
3331 if (beg > 0) return 0;
3332 len = str_utf8_offset(p, e, len);
3333 }
3334#endif
3335 else if (rb_enc_mbmaxlen(enc) == rb_enc_mbminlen(enc)) {
3336 int char_sz = rb_enc_mbmaxlen(enc);
3337
3338 p = s + beg * char_sz;
3339 if (p > e) {
3340 return 0;
3341 }
3342 else if (len * char_sz > e - p)
3343 len = e - p;
3344 else
3345 len *= char_sz;
3346 }
3347 else if ((p = str_nth_len(s, e, &beg, enc)) == e) {
3348 if (beg > 0) return 0;
3349 len = 0;
3350 }
3351 else {
3352 len = str_offset(p, e, len, enc, 0);
3353 }
3354 end:
3355 *lenp = len;
3356 RB_GC_GUARD(str);
3357 return (char *)p;
3358}
3359
3360static VALUE str_substr(VALUE str, long beg, long len, int empty);
3361
3362VALUE
3363rb_str_substr(VALUE str, long beg, long len)
3364{
3365 return str_substr(str, beg, len, TRUE);
3366}
3367
3368VALUE
3369rb_str_substr_two_fixnums(VALUE str, VALUE beg, VALUE len, int empty)
3370{
3371 return str_substr(str, NUM2LONG(beg), NUM2LONG(len), empty);
3372}
3373
3374static VALUE
3375str_substr(VALUE str, long beg, long len, int empty)
3376{
3377 const char *p = rb_str_subpos(str, beg, &len);
3378
3379 if (!p) return Qnil;
3380 if (!len && !empty) return Qnil;
3381
3382 beg = p - RSTRING_PTR(str);
3383
3384 VALUE str2 = str_subseq(str, beg, len);
3385 rb_enc_cr_str_copy_for_substr(str2, str);
3386 return str2;
3387}
3388
3389/* :nodoc: */
3390VALUE
3392{
3393 if (CHILLED_STRING_P(str)) {
3394 FL_UNSET_RAW(str, STR_CHILLED);
3395 }
3396
3397 if (OBJ_FROZEN(str)) return str;
3398 rb_str_resize(str, RSTRING_LEN(str));
3399 return rb_obj_freeze(str);
3400}
3401
3402/*
3403 * call-seq:
3404 * +string -> new_string or self
3405 *
3406 * Returns +self+ if +self+ is not frozen and can be mutated
3407 * without warning issuance.
3408 *
3409 * Otherwise returns <tt>self.dup</tt>, which is not frozen.
3410 *
3411 * Related: see {Freezing/Unfreezing}[rdoc-ref:String@FreezingUnfreezing].
3412 */
3413static VALUE
3414str_uplus(VALUE str)
3415{
3416 if (OBJ_FROZEN(str) || CHILLED_STRING_P(str)) {
3417 return rb_str_dup(str);
3418 }
3419 else {
3420 return str;
3421 }
3422}
3423
3424/*
3425 * call-seq:
3426 * -self -> frozen_string
3427 *
3428 * Returns a frozen string equal to +self+.
3429 *
3430 * The returned string is +self+ if and only if all of the following are true:
3431 *
3432 * - +self+ is already frozen.
3433 * - +self+ is an instance of \String (rather than of a subclass of \String)
3434 * - +self+ has no instance variables set on it.
3435 *
3436 * Otherwise, the returned string is a frozen copy of +self+.
3437 *
3438 * Returning +self+, when possible, saves duplicating +self+;
3439 * see {Data deduplication}[https://en.wikipedia.org/wiki/Data_deduplication].
3440 *
3441 * It may also save duplicating other, already-existing, strings:
3442 *
3443 * s0 = 'foo'
3444 * s1 = 'foo'
3445 * s0.object_id == s1.object_id # => false
3446 * (-s0).object_id == (-s1).object_id # => true
3447 *
3448 * Note that method #-@ is convenient for defining a constant:
3449 *
3450 * FileName = -'config/database.yml'
3451 *
3452 * While its alias #dedup is better suited for chaining:
3453 *
3454 * 'foo'.dedup.gsub!('o')
3455 *
3456 * Related: see {Freezing/Unfreezing}[rdoc-ref:String@FreezingUnfreezing].
3457 */
3458static VALUE
3459str_uminus(VALUE str)
3460{
3461 if (!BARE_STRING_P(str) && !rb_obj_frozen_p(str)) {
3462 str = rb_str_dup(str);
3463 }
3464 return rb_fstring(str);
3465}
3466
3467RUBY_ALIAS_FUNCTION(rb_str_dup_frozen(VALUE str), rb_str_new_frozen, (str))
3468#define rb_str_dup_frozen rb_str_new_frozen
3469
3470VALUE
3472{
3473 rb_check_frozen(str);
3474 if (FL_TEST(str, STR_TMPLOCK)) {
3475 rb_raise(rb_eRuntimeError, "temporal locking already locked string");
3476 }
3477 FL_SET(str, STR_TMPLOCK);
3478 return str;
3479}
3480
3481VALUE
3483{
3484 rb_check_frozen(str);
3485 if (!FL_TEST(str, STR_TMPLOCK)) {
3486 rb_raise(rb_eRuntimeError, "temporal unlocking already unlocked string");
3487 }
3488 FL_UNSET(str, STR_TMPLOCK);
3489 return str;
3490}
3491
3492VALUE
3493rb_str_locktmp_ensure(VALUE str, VALUE (*func)(VALUE), VALUE arg)
3494{
3495 rb_str_locktmp(str);
3496 return rb_ensure(func, arg, rb_str_unlocktmp, str);
3497}
3498
3499void
3501{
3502 RUBY_ASSERT(ruby_thread_has_gvl_p());
3503
3504 long capa;
3505 const int termlen = TERM_LEN(str);
3506
3507 str_modifiable(str);
3508 if (STR_SHARED_P(str)) {
3509 rb_raise(rb_eRuntimeError, "can't set length of shared string");
3510 }
3511 if (len > (capa = (long)str_capacity(str, termlen)) || len < 0) {
3512 rb_bug("probable buffer overflow: %ld for %ld", len, capa);
3513 }
3514
3515 int cr = ENC_CODERANGE(str);
3516 if (len == 0) {
3517 /* Empty string does not contain non-ASCII */
3519 }
3520 else if (cr == ENC_CODERANGE_UNKNOWN) {
3521 /* Leave unknown. */
3522 }
3523 else if (len > RSTRING_LEN(str)) {
3524 if (ENC_CODERANGE_CLEAN_P(cr)) {
3525 /* Update the coderange regarding the extended part. */
3526 const char *const prev_end = RSTRING_END(str);
3527 const char *const new_end = RSTRING_PTR(str) + len;
3528 rb_encoding *enc = rb_enc_get(str);
3529 rb_str_coderange_scan_restartable(prev_end, new_end, enc, &cr);
3530 ENC_CODERANGE_SET(str, cr);
3531 }
3532 else if (cr == ENC_CODERANGE_BROKEN) {
3533 /* May be valid now, by appended part. */
3535 }
3536 }
3537 else if (len < RSTRING_LEN(str)) {
3538 if (cr != ENC_CODERANGE_7BIT) {
3539 /* ASCII-only string is keeping after truncated. Valid
3540 * and broken may be invalid or valid, leave unknown. */
3542 }
3543 }
3544
3545 STR_SET_LEN(str, len);
3546 TERM_FILL(&RSTRING_PTR(str)[len], termlen);
3547}
3548
3549VALUE
3550rb_str_resize(VALUE str, long len)
3551{
3552 if (len < 0) {
3553 rb_raise(rb_eArgError, "negative string size (or size too big)");
3554 }
3555
3556 int independent = str_independent(str);
3557 long slen = RSTRING_LEN(str);
3558 const int termlen = TERM_LEN(str);
3559
3560 if (slen > len || (termlen != 1 && slen < len)) {
3562 }
3563
3564 {
3565 long capa;
3566 if (STR_EMBED_P(str)) {
3567 if (len == slen) return str;
3568 if (str_embed_capa(str) >= len + termlen) {
3569 STR_SET_LEN(str, len);
3570 TERM_FILL(RSTRING(str)->as.embed.ary + len, termlen);
3571 return str;
3572 }
3573 str_make_independent_expand(str, slen, len - slen, termlen);
3574 }
3575 else if (str_embed_capa(str) >= len + termlen) {
3576 capa = RSTRING(str)->as.heap.aux.capa;
3577 char *ptr = STR_HEAP_PTR(str);
3578 STR_SET_EMBED(str);
3579 if (slen > len) slen = len;
3580 if (slen > 0) MEMCPY(RSTRING(str)->as.embed.ary, ptr, char, slen);
3581 TERM_FILL(RSTRING(str)->as.embed.ary + len, termlen);
3582 STR_SET_LEN(str, len);
3583 if (independent) {
3584 SIZED_FREE_N(ptr, capa + termlen);
3585 }
3586 return str;
3587 }
3588 else if (!independent) {
3589 if (len == slen) return str;
3590 str_make_independent_expand(str, slen, len - slen, termlen);
3591 }
3592 else if ((capa = RSTRING(str)->as.heap.aux.capa) < len ||
3593 (capa - len) > (len < 1024 ? len : 1024)) {
3594 SIZED_REALLOC_N(RSTRING(str)->as.heap.ptr, char,
3595 (size_t)len + termlen, STR_HEAP_SIZE(str));
3596 RSTRING(str)->as.heap.aux.capa = len;
3597 }
3598 else if (len == slen) return str;
3599 STR_SET_LEN(str, len);
3600 TERM_FILL(RSTRING(str)->as.heap.ptr + len, termlen); /* sentinel */
3601 }
3602 return str;
3603}
3604
3605static void
3606str_ensure_available_capa(VALUE str, long len)
3607{
3608 str_modify_keep_cr(str);
3609
3610 const int termlen = TERM_LEN(str);
3611 long olen = RSTRING_LEN(str);
3612
3613 if (RB_UNLIKELY(olen > LONG_MAX - len)) {
3614 rb_raise(rb_eArgError, "string sizes too big");
3615 }
3616
3617 long total = olen + len;
3618 long capa = str_capacity(str, termlen);
3619
3620 if (capa < total) {
3621 if (total >= LONG_MAX / 2) {
3622 capa = total;
3623 }
3624 while (total > capa) {
3625 capa = 2 * capa + termlen; /* == 2*(capa+termlen)-termlen */
3626 }
3627 RESIZE_CAPA_TERM(str, capa, termlen);
3628 }
3629}
3630
3631static VALUE
3632str_buf_cat4(VALUE str, const char *ptr, long len, bool keep_cr)
3633{
3634 if (keep_cr) {
3635 str_modify_keep_cr(str);
3636 }
3637 else {
3638 rb_str_modify(str);
3639 }
3640 if (len == 0) return 0;
3641
3642 long total, olen, off = -1;
3643 char *sptr;
3644 const int termlen = TERM_LEN(str);
3645
3646 RSTRING_GETMEM(str, sptr, olen);
3647 if (ptr >= sptr && ptr <= sptr + olen) {
3648 off = ptr - sptr;
3649 }
3650
3651 long capa = str_capacity(str, termlen);
3652
3653 if (olen > LONG_MAX - len) {
3654 rb_raise(rb_eArgError, "string sizes too big");
3655 }
3656 total = olen + len;
3657 if (capa < total) {
3658 if (total >= LONG_MAX / 2) {
3659 capa = total;
3660 }
3661 while (total > capa) {
3662 capa = 2 * capa + termlen; /* == 2*(capa+termlen)-termlen */
3663 }
3664 RESIZE_CAPA_TERM(str, capa, termlen);
3665 sptr = RSTRING_PTR(str);
3666 }
3667 if (off != -1) {
3668 ptr = sptr + off;
3669 }
3670 memcpy(sptr + olen, ptr, len);
3671 STR_SET_LEN(str, total);
3672 TERM_FILL(sptr + total, termlen); /* sentinel */
3673
3674 return str;
3675}
3676
3677#define str_buf_cat(str, ptr, len) str_buf_cat4((str), (ptr), len, false)
3678#define str_buf_cat2(str, ptr) str_buf_cat4((str), (ptr), rb_strlen_lit(ptr), false)
3679
3680VALUE
3681rb_str_cat(VALUE str, const char *ptr, long len)
3682{
3683 if (len == 0) return str;
3684 if (len < 0) {
3685 rb_raise(rb_eArgError, "negative string size (or size too big)");
3686 }
3687 return str_buf_cat(str, ptr, len);
3688}
3689
3690VALUE
3691rb_str_cat_cstr(VALUE str, const char *ptr)
3692{
3693 must_not_null(ptr);
3694 return rb_str_buf_cat(str, ptr, strlen(ptr));
3695}
3696
3697static void
3698rb_str_buf_cat_byte(VALUE str, unsigned char byte)
3699{
3700 RUBY_ASSERT(RB_ENCODING_GET_INLINED(str) == ENCINDEX_ASCII_8BIT || RB_ENCODING_GET_INLINED(str) == ENCINDEX_US_ASCII);
3701
3702 // We can't write directly to shared strings without impacting others, so we must make the string independent.
3703 if (UNLIKELY(!str_independent(str))) {
3704 str_make_independent(str);
3705 }
3706
3707 long string_length = -1;
3708 const int null_terminator_length = 1;
3709 char *sptr;
3710 RSTRING_GETMEM(str, sptr, string_length);
3711
3712 // Ensure the resulting string wouldn't be too long.
3713 if (UNLIKELY(string_length > LONG_MAX - 1)) {
3714 rb_raise(rb_eArgError, "string sizes too big");
3715 }
3716
3717 long string_capacity = str_capacity(str, null_terminator_length);
3718
3719 // Get the code range before any modifications since those might clear the code range.
3720 int cr = ENC_CODERANGE(str);
3721
3722 // Check if the string has spare string_capacity to write the new byte.
3723 if (LIKELY(string_capacity >= string_length + 1)) {
3724 // In fast path we can write the new byte and note the string's new length.
3725 sptr[string_length] = byte;
3726 STR_SET_LEN(str, string_length + 1);
3727 TERM_FILL(sptr + string_length + 1, null_terminator_length);
3728 }
3729 else {
3730 // If there's not enough string_capacity, make a call into the general string concatenation function.
3731 str_buf_cat(str, (char *)&byte, 1);
3732 }
3733
3734 // If the code range is already known, we can derive the resulting code range cheaply by looking at the byte we
3735 // just appended. If the code range is unknown, but the string was empty, then we can also derive the code range
3736 // by looking at the byte we just appended. Otherwise, we'd have to scan the bytes to determine the code range so
3737 // we leave it as unknown. It cannot be broken for binary strings so we don't need to handle that option.
3738 if (cr == ENC_CODERANGE_7BIT || string_length == 0) {
3739 if (ISASCII(byte)) {
3741 }
3742 else {
3744
3745 // Promote a US-ASCII string to ASCII-8BIT when a non-ASCII byte is appended.
3746 if (UNLIKELY(RB_ENCODING_GET_INLINED(str) == ENCINDEX_US_ASCII)) {
3747 rb_enc_associate_index(str, ENCINDEX_ASCII_8BIT);
3748 }
3749 }
3750 }
3751}
3752
3753RUBY_ALIAS_FUNCTION(rb_str_buf_cat(VALUE str, const char *ptr, long len), rb_str_cat, (str, ptr, len))
3754RUBY_ALIAS_FUNCTION(rb_str_buf_cat2(VALUE str, const char *ptr), rb_str_cat_cstr, (str, ptr))
3755RUBY_ALIAS_FUNCTION(rb_str_cat2(VALUE str, const char *ptr), rb_str_cat_cstr, (str, ptr))
3756
3757static VALUE
3758rb_enc_cr_str_buf_cat(VALUE str, const char *ptr, long len,
3759 int ptr_encindex, int ptr_cr, int *ptr_cr_ret)
3760{
3761 int str_encindex = ENCODING_GET(str);
3762 int res_encindex;
3763 int str_cr, res_cr;
3764 rb_encoding *str_enc, *ptr_enc;
3765
3766 str_cr = RSTRING_LEN(str) ? ENC_CODERANGE(str) : ENC_CODERANGE_7BIT;
3767
3768 if (str_encindex == ptr_encindex) {
3769 if (str_cr != ENC_CODERANGE_UNKNOWN && ptr_cr == ENC_CODERANGE_UNKNOWN) {
3770 ptr_cr = coderange_scan(ptr, len, rb_enc_from_index(ptr_encindex));
3771 }
3772 }
3773 else {
3774 str_enc = rb_enc_from_index(str_encindex);
3775 ptr_enc = rb_enc_from_index(ptr_encindex);
3776 if (!rb_enc_asciicompat(str_enc) || !rb_enc_asciicompat(ptr_enc)) {
3777 if (len == 0)
3778 return str;
3779 if (RSTRING_LEN(str) == 0) {
3780 rb_str_buf_cat(str, ptr, len);
3781 ENCODING_CODERANGE_SET(str, ptr_encindex, ptr_cr);
3782 rb_str_change_terminator_length(str, rb_enc_mbminlen(str_enc), rb_enc_mbminlen(ptr_enc));
3783 return str;
3784 }
3785 goto incompatible;
3786 }
3787 if (ptr_cr == ENC_CODERANGE_UNKNOWN) {
3788 ptr_cr = coderange_scan(ptr, len, ptr_enc);
3789 }
3790 if (str_cr == ENC_CODERANGE_UNKNOWN) {
3791 if (ENCODING_IS_ASCII8BIT(str) || ptr_cr != ENC_CODERANGE_7BIT) {
3792 str_cr = rb_enc_str_coderange(str);
3793 }
3794 }
3795 }
3796 if (ptr_cr_ret)
3797 *ptr_cr_ret = ptr_cr;
3798
3799 if (str_encindex != ptr_encindex &&
3800 str_cr != ENC_CODERANGE_7BIT &&
3801 ptr_cr != ENC_CODERANGE_7BIT) {
3802 str_enc = rb_enc_from_index(str_encindex);
3803 ptr_enc = rb_enc_from_index(ptr_encindex);
3804 goto incompatible;
3805 }
3806
3807 if (str_cr == ENC_CODERANGE_UNKNOWN) {
3808 res_encindex = str_encindex;
3809 res_cr = ENC_CODERANGE_UNKNOWN;
3810 }
3811 else if (str_cr == ENC_CODERANGE_7BIT) {
3812 if (ptr_cr == ENC_CODERANGE_7BIT) {
3813 res_encindex = str_encindex;
3814 res_cr = ENC_CODERANGE_7BIT;
3815 }
3816 else {
3817 res_encindex = ptr_encindex;
3818 res_cr = ptr_cr;
3819 }
3820 }
3821 else if (str_cr == ENC_CODERANGE_VALID) {
3822 res_encindex = str_encindex;
3823 if (ENC_CODERANGE_CLEAN_P(ptr_cr))
3824 res_cr = str_cr;
3825 else
3826 res_cr = ptr_cr;
3827 }
3828 else { /* str_cr == ENC_CODERANGE_BROKEN */
3829 res_encindex = str_encindex;
3830 res_cr = str_cr;
3831 if (0 < len) res_cr = ENC_CODERANGE_UNKNOWN;
3832 }
3833
3834 if (len < 0) {
3835 rb_raise(rb_eArgError, "negative string size (or size too big)");
3836 }
3837 str_buf_cat(str, ptr, len);
3838 ENCODING_CODERANGE_SET(str, res_encindex, res_cr);
3839 return str;
3840
3841 incompatible:
3842 rb_raise(rb_eEncCompatError, "incompatible character encodings: %s and %s",
3843 rb_enc_inspect_name(str_enc), rb_enc_inspect_name(ptr_enc));
3845}
3846
3847VALUE
3848rb_enc_str_buf_cat(VALUE str, const char *ptr, long len, rb_encoding *ptr_enc)
3849{
3850 return rb_enc_cr_str_buf_cat(str, ptr, len,
3851 rb_enc_to_index(ptr_enc), ENC_CODERANGE_UNKNOWN, NULL);
3852}
3853
3854VALUE
3856{
3857 /* ptr must reference NUL terminated ASCII string. */
3858 int encindex = ENCODING_GET(str);
3859 rb_encoding *enc = rb_enc_from_index(encindex);
3860 if (rb_enc_asciicompat(enc)) {
3861 return rb_enc_cr_str_buf_cat(str, ptr, strlen(ptr),
3862 encindex, ENC_CODERANGE_7BIT, 0);
3863 }
3864 else {
3865 char *buf = ALLOCA_N(char, rb_enc_mbmaxlen(enc));
3866 while (*ptr) {
3867 unsigned int c = (unsigned char)*ptr;
3868 int len = rb_enc_codelen(c, enc);
3869 rb_enc_mbcput(c, buf, enc);
3870 rb_enc_cr_str_buf_cat(str, buf, len,
3871 encindex, ENC_CODERANGE_VALID, 0);
3872 ptr++;
3873 }
3874 return str;
3875 }
3876}
3877
3878VALUE
3880{
3881 int str2_cr = rb_enc_str_coderange(str2);
3882
3883 if (rb_str_enc_fastpath(str)) {
3884 switch (str2_cr) {
3885 case ENC_CODERANGE_7BIT:
3886 // If RHS is 7bit we can do simple concatenation
3887 str_buf_cat4(str, RSTRING_PTR(str2), RSTRING_LEN(str2), true);
3888 RB_GC_GUARD(str2);
3889 return str;
3891 // If RHS is valid, we can do simple concatenation if encodings are the same
3892 if (ENCODING_GET_INLINED(str) == ENCODING_GET_INLINED(str2)) {
3893 str_buf_cat4(str, RSTRING_PTR(str2), RSTRING_LEN(str2), true);
3894 int str_cr = ENC_CODERANGE(str);
3895 if (UNLIKELY(str_cr != ENC_CODERANGE_VALID)) {
3896 ENC_CODERANGE_SET(str, RB_ENC_CODERANGE_AND(str_cr, str2_cr));
3897 }
3898 RB_GC_GUARD(str2);
3899 return str;
3900 }
3901 }
3902 }
3903
3904 rb_enc_cr_str_buf_cat(str, RSTRING_PTR(str2), RSTRING_LEN(str2),
3905 ENCODING_GET(str2), str2_cr, &str2_cr);
3906
3907 ENC_CODERANGE_SET(str2, str2_cr);
3908
3909 return str;
3910}
3911
3912VALUE
3914{
3915 StringValue(str2);
3916 return rb_str_buf_append(str, str2);
3917}
3918
3919VALUE
3920rb_str_concat_literals(size_t num, const VALUE *strary)
3921{
3922 VALUE str;
3923 size_t i, s = 0;
3924 unsigned long len = 1;
3925
3926 if (UNLIKELY(!num)) return rb_str_new(0, 0);
3927 if (UNLIKELY(num == 1)) return rb_str_resurrect(strary[0]);
3928
3929 for (i = 0; i < num; ++i) { len += RSTRING_LEN(strary[i]); }
3930 str = rb_str_buf_new(len);
3931 str_enc_copy_direct(str, strary[0]);
3932
3933 for (i = s; i < num; ++i) {
3934 const VALUE v = strary[i];
3935 int encidx = ENCODING_GET(v);
3936
3937 rb_str_buf_append(str, v);
3938 if (encidx != ENCINDEX_US_ASCII) {
3939 if (ENCODING_GET_INLINED(str) == ENCINDEX_US_ASCII)
3940 rb_enc_set_index(str, encidx);
3941 }
3942 }
3943 return str;
3944}
3945
3946/*
3947 * call-seq:
3948 * concat(*objects) -> string
3949 *
3950 * :include: doc/string/concat.rdoc
3951 */
3952static VALUE
3953rb_str_concat_multi(int argc, VALUE *argv, VALUE str)
3954{
3955 str_modifiable(str);
3956
3957 if (argc == 1) {
3958 return rb_str_concat(str, argv[0]);
3959 }
3960 else if (argc > 1) {
3961 int i;
3962 VALUE arg_str = rb_str_tmp_new(0);
3963 rb_enc_copy(arg_str, str);
3964 for (i = 0; i < argc; i++) {
3965 rb_str_concat(arg_str, argv[i]);
3966 }
3967 rb_str_buf_append(str, arg_str);
3968 }
3969
3970 return str;
3971}
3972
3973/*
3974 * call-seq:
3975 * append_as_bytes(*objects) -> self
3976 *
3977 * Concatenates each object in +objects+ into +self+; returns +self+;
3978 * performs no encoding validation or conversion:
3979 *
3980 * s = 'foo'
3981 * s.append_as_bytes(" \xE2\x82") # => "foo \xE2\x82"
3982 * s.valid_encoding? # => false
3983 * s.append_as_bytes("\xAC 12")
3984 * s.valid_encoding? # => true
3985 *
3986 * When a given object is an integer,
3987 * the value is considered an 8-bit byte;
3988 * if the integer occupies more than one byte (i.e,. is greater than 255),
3989 * appends only the low-order byte (similar to String#setbyte):
3990 *
3991 * s = ""
3992 * s.append_as_bytes(0, 257) # => "\u0000\u0001"
3993 * s.bytesize # => 2
3994 *
3995 * Related: see {Modifying}[rdoc-ref:String@Modifying].
3996 */
3997
3998VALUE
3999rb_str_append_as_bytes(int argc, VALUE *argv, VALUE str)
4000{
4001 long needed_capacity = 0;
4002 volatile VALUE t0;
4003 enum ruby_value_type *types = ALLOCV_N(enum ruby_value_type, t0, argc);
4004
4005 for (int index = 0; index < argc; index++) {
4006 VALUE obj = argv[index];
4007 enum ruby_value_type type = types[index] = rb_type(obj);
4008 switch (type) {
4009 case T_FIXNUM:
4010 case T_BIGNUM:
4011 needed_capacity++;
4012 break;
4013 case T_STRING:
4014 needed_capacity += RSTRING_LEN(obj);
4015 break;
4016 default:
4017 rb_raise(
4019 "wrong argument type %"PRIsVALUE" (expected String or Integer)",
4020 rb_obj_class(obj)
4021 );
4022 break;
4023 }
4024 }
4025
4026 str_ensure_available_capa(str, needed_capacity);
4027 char *sptr = RSTRING_END(str);
4028
4029 for (int index = 0; index < argc; index++) {
4030 VALUE obj = argv[index];
4031 enum ruby_value_type type = types[index];
4032 switch (type) {
4033 case T_FIXNUM:
4034 case T_BIGNUM: {
4035 argv[index] = obj = rb_int_and(obj, INT2FIX(0xff));
4036 char byte = (char)(NUM2INT(obj) & 0xFF);
4037 *sptr = byte;
4038 sptr++;
4039 break;
4040 }
4041 case T_STRING: {
4042 const char *ptr;
4043 long len;
4044 RSTRING_GETMEM(obj, ptr, len);
4045 memcpy(sptr, ptr, len);
4046 sptr += len;
4047 break;
4048 }
4049 default:
4050 rb_bug("append_as_bytes arguments should have been validated");
4051 }
4052 }
4053
4054 STR_SET_LEN(str, RSTRING_LEN(str) + needed_capacity);
4055 TERM_FILL(sptr, TERM_LEN(str)); /* sentinel */
4056
4057 int cr = ENC_CODERANGE(str);
4058 switch (cr) {
4059 case ENC_CODERANGE_7BIT: {
4060 for (int index = 0; index < argc; index++) {
4061 VALUE obj = argv[index];
4062 enum ruby_value_type type = types[index];
4063 switch (type) {
4064 case T_FIXNUM:
4065 case T_BIGNUM: {
4066 if (!ISASCII(NUM2INT(obj))) {
4067 goto clear_cr;
4068 }
4069 break;
4070 }
4071 case T_STRING: {
4072 if (ENC_CODERANGE(obj) != ENC_CODERANGE_7BIT) {
4073 goto clear_cr;
4074 }
4075 break;
4076 }
4077 default:
4078 rb_bug("append_as_bytes arguments should have been validated");
4079 }
4080 }
4081 break;
4082 }
4084 if (ENCODING_GET_INLINED(str) == ENCINDEX_ASCII_8BIT) {
4085 goto keep_cr;
4086 }
4087 else {
4088 goto clear_cr;
4089 }
4090 break;
4091 default:
4092 goto clear_cr;
4093 break;
4094 }
4095
4096 clear_cr:
4097 // If no fast path was hit, we clear the coderange.
4098 // append_as_bytes is predominantly meant to be used in
4099 // buffering situation, hence it's likely the coderange
4100 // will never be scanned, so it's not worth spending time
4101 // precomputing the coderange except for simple and common
4102 // situations.
4104 keep_cr:
4105 ALLOCV_END(t0);
4106 return str;
4107}
4108
4109/*
4110 * call-seq:
4111 * self << object -> self
4112 *
4113 * Appends a string representation of +object+ to +self+;
4114 * returns +self+.
4115 *
4116 * If +object+ is a string, appends it to +self+:
4117 *
4118 * s = 'foo'
4119 * s << 'bar' # => "foobar"
4120 * s # => "foobar"
4121 *
4122 * If +object+ is an integer,
4123 * its value is considered a codepoint;
4124 * converts the value to a character before concatenating:
4125 *
4126 * s = 'foo'
4127 * s << 33 # => "foo!"
4128 *
4129 * Additionally, if the codepoint is in range <tt>0..0xff</tt>
4130 * and the encoding of +self+ is Encoding::US_ASCII,
4131 * changes the encoding to Encoding::ASCII_8BIT:
4132 *
4133 * s = 'foo'.encode(Encoding::US_ASCII)
4134 * s.encoding # => #<Encoding:US-ASCII>
4135 * s << 0xff # => "foo\xFF"
4136 * s.encoding # => #<Encoding:BINARY (ASCII-8BIT)>
4137 *
4138 * Raises RangeError if that codepoint is not representable in the encoding of +self+:
4139 *
4140 * s = 'foo'
4141 * s.encoding # => <Encoding:UTF-8>
4142 * s << 0x00110000 # 1114112 out of char range (RangeError)
4143 * s = 'foo'.encode(Encoding::EUC_JP)
4144 * s << 0x00800080 # invalid codepoint 0x800080 in EUC-JP (RangeError)
4145 *
4146 * Related: see {Modifying}[rdoc-ref:String@Modifying].
4147 */
4148VALUE
4150{
4151 unsigned int code;
4152 rb_encoding *enc = STR_ENC_GET(str1);
4153 int encidx;
4154
4155 if (RB_INTEGER_TYPE_P(str2)) {
4156 if (rb_num_to_uint(str2, &code) == 0) {
4157 }
4158 else if (FIXNUM_P(str2)) {
4159 rb_raise(rb_eRangeError, "%ld out of char range", FIX2LONG(str2));
4160 }
4161 else {
4162 rb_raise(rb_eRangeError, "bignum out of char range");
4163 }
4164 }
4165 else {
4166 return rb_str_append(str1, str2);
4167 }
4168
4169 encidx = rb_ascii8bit_appendable_encoding_index(enc, code);
4170
4171 if (encidx >= 0) {
4172 rb_str_buf_cat_byte(str1, (unsigned char)code);
4173 }
4174 else {
4175 long pos = RSTRING_LEN(str1);
4176 int cr = ENC_CODERANGE(str1);
4177 int len;
4178 char *buf;
4179
4180 switch (len = rb_enc_codelen(code, enc)) {
4181 case ONIGERR_INVALID_CODE_POINT_VALUE:
4182 rb_raise(rb_eRangeError, "invalid codepoint 0x%X in %s", code, rb_enc_name(enc));
4183 break;
4184 case ONIGERR_TOO_BIG_WIDE_CHAR_VALUE:
4185 case 0:
4186 rb_raise(rb_eRangeError, "%u out of char range", code);
4187 break;
4188 }
4189 buf = ALLOCA_N(char, len + 1);
4190 rb_enc_mbcput(code, buf, enc);
4191 if (rb_enc_precise_mbclen(buf, buf + len + 1, enc) != len) {
4192 rb_raise(rb_eRangeError, "invalid codepoint 0x%X in %s", code, rb_enc_name(enc));
4193 }
4194 rb_str_resize(str1, pos+len);
4195 memcpy(RSTRING_PTR(str1) + pos, buf, len);
4196 if (cr == ENC_CODERANGE_7BIT && code > 127) {
4198 }
4199 else if (cr == ENC_CODERANGE_BROKEN) {
4201 }
4202 ENC_CODERANGE_SET(str1, cr);
4203 }
4204 return str1;
4205}
4206
4207int
4208rb_ascii8bit_appendable_encoding_index(rb_encoding *enc, unsigned int code)
4209{
4210 int encidx = rb_enc_to_index(enc);
4211
4212 if (encidx == ENCINDEX_ASCII_8BIT || encidx == ENCINDEX_US_ASCII) {
4213 /* US-ASCII automatically extended to ASCII-8BIT */
4214 if (code > 0xFF) {
4215 rb_raise(rb_eRangeError, "%u out of char range", code);
4216 }
4217 if (encidx == ENCINDEX_US_ASCII && code > 127) {
4218 return ENCINDEX_ASCII_8BIT;
4219 }
4220 return encidx;
4221 }
4222 else {
4223 return -1;
4224 }
4225}
4226
4227/*
4228 * call-seq:
4229 * prepend(*other_strings) -> new_string
4230 *
4231 * Prefixes to +self+ the concatenation of the given +other_strings+; returns +self+:
4232 *
4233 * 'baz'.prepend('foo', 'bar') # => "foobarbaz"
4234 *
4235 * Related: see {Modifying}[rdoc-ref:String@Modifying].
4236 *
4237 */
4238
4239static VALUE
4240rb_str_prepend_multi(int argc, VALUE *argv, VALUE str)
4241{
4242 str_modifiable(str);
4243
4244 if (argc == 1) {
4245 rb_str_update(str, 0L, 0L, argv[0]);
4246 }
4247 else if (argc > 1) {
4248 int i;
4249 VALUE arg_str = rb_str_tmp_new(0);
4250 rb_enc_copy(arg_str, str);
4251 for (i = 0; i < argc; i++) {
4252 rb_str_append(arg_str, argv[i]);
4253 }
4254 rb_str_update(str, 0L, 0L, arg_str);
4255 }
4256
4257 return str;
4258}
4259
4260st_index_t
4262{
4263 if (FL_TEST_RAW(str, STR_PRECOMPUTED_HASH)) {
4264 st_index_t precomputed_hash;
4265 memcpy(&precomputed_hash, RSTRING_END(str) + TERM_LEN(str), sizeof(precomputed_hash));
4266
4267 RUBY_ASSERT(precomputed_hash == str_do_hash(str));
4268 return precomputed_hash;
4269 }
4270
4271 return str_do_hash(str);
4272}
4273
4274int
4276{
4277 long len1, len2;
4278 const char *ptr1, *ptr2;
4279 RSTRING_GETMEM(str1, ptr1, len1);
4280 RSTRING_GETMEM(str2, ptr2, len2);
4281 return (len1 != len2 ||
4282 !rb_str_comparable(str1, str2) ||
4283 memcmp(ptr1, ptr2, len1) != 0);
4284}
4285
4286/*
4287 * call-seq:
4288 * hash -> integer
4289 *
4290 * :include: doc/string/hash.rdoc
4291 *
4292 */
4293
4294static VALUE
4295rb_str_hash_m(VALUE str)
4296{
4297 st_index_t hval = rb_str_hash(str);
4298 return ST2FIX(hval);
4299}
4300
4301#define lesser(a,b) (((a)>(b))?(b):(a))
4302
4303int
4305{
4306 int idx1, idx2;
4307 int rc1, rc2;
4308
4309 if (RSTRING_LEN(str1) == 0) return TRUE;
4310 if (RSTRING_LEN(str2) == 0) return TRUE;
4311 idx1 = ENCODING_GET(str1);
4312 idx2 = ENCODING_GET(str2);
4313 if (idx1 == idx2) return TRUE;
4314 rc1 = rb_enc_str_coderange(str1);
4315 rc2 = rb_enc_str_coderange(str2);
4316 if (rc1 == ENC_CODERANGE_7BIT) {
4317 if (rc2 == ENC_CODERANGE_7BIT) return TRUE;
4318 if (rb_enc_asciicompat(rb_enc_from_index(idx2)))
4319 return TRUE;
4320 }
4321 if (rc2 == ENC_CODERANGE_7BIT) {
4322 if (rb_enc_asciicompat(rb_enc_from_index(idx1)))
4323 return TRUE;
4324 }
4325 return FALSE;
4326}
4327
4328int
4330{
4331 long len1, len2;
4332 const char *ptr1, *ptr2;
4333 int retval;
4334
4335 if (str1 == str2) return 0;
4336 RSTRING_GETMEM(str1, ptr1, len1);
4337 RSTRING_GETMEM(str2, ptr2, len2);
4338 if (ptr1 == ptr2 || (retval = memcmp(ptr1, ptr2, lesser(len1, len2))) == 0) {
4339 if (len1 == len2) {
4340 if (!rb_str_comparable(str1, str2)) {
4341 if (ENCODING_GET(str1) > ENCODING_GET(str2))
4342 return 1;
4343 return -1;
4344 }
4345 return 0;
4346 }
4347 if (len1 > len2) return 1;
4348 return -1;
4349 }
4350 if (retval > 0) return 1;
4351 return -1;
4352}
4353
4354/*
4355 * call-seq:
4356 * self == other -> true or false
4357 *
4358 * Returns whether +other+ is equal to +self+.
4359 *
4360 * When +other+ is a string, returns whether +other+ has the same length and content as +self+:
4361 *
4362 * s = 'foo'
4363 * s == 'foo' # => true
4364 * s == 'food' # => false
4365 * s == 'FOO' # => false
4366 *
4367 * Returns +false+ if the two strings' encodings are not compatible:
4368 *
4369 * "\u{e4 f6 fc}".encode(Encoding::ISO_8859_1) == ("\u{c4 d6 dc}") # => false
4370 *
4371 * When +other+ is not a string:
4372 *
4373 * - If +other+ responds to method <tt>to_str</tt>,
4374 * <tt>other == self</tt> is called and its return value is returned.
4375 * - If +other+ does not respond to <tt>to_str</tt>,
4376 * +false+ is returned.
4377 *
4378 * Related: {Comparing}[rdoc-ref:String@Comparing].
4379 */
4380
4381VALUE
4383{
4384 if (str1 == str2) return Qtrue;
4385 if (!RB_TYPE_P(str2, T_STRING)) {
4386 if (!rb_respond_to(str2, idTo_str)) {
4387 return Qfalse;
4388 }
4389 return rb_equal(str2, str1);
4390 }
4391 return rb_str_eql_internal(str1, str2);
4392}
4393
4394/*
4395 * call-seq:
4396 * eql?(object) -> true or false
4397 *
4398 * :include: doc/string/eql_p.rdoc
4399 *
4400 */
4401
4402VALUE
4403rb_str_eql(VALUE str1, VALUE str2)
4404{
4405 if (str1 == str2) return Qtrue;
4406 if (!RB_TYPE_P(str2, T_STRING)) return Qfalse;
4407 return rb_str_eql_internal(str1, str2);
4408}
4409
4410/*
4411 * call-seq:
4412 * self <=> other -> -1, 0, 1, or nil
4413 *
4414 * Compares +self+ and +other+,
4415 * evaluating their _contents_, not their _lengths_.
4416 *
4417 * Returns:
4418 *
4419 * - +-1+, if +self+ is smaller.
4420 * - +0+, if the two are equal.
4421 * - +1+, if +self+ is larger.
4422 * - +nil+, if the two are incomparable.
4423 *
4424 * Examples:
4425 *
4426 * 'a' <=> 'b' # => -1
4427 * 'a' <=> 'ab' # => -1
4428 * 'a' <=> 'a' # => 0
4429 * 'b' <=> 'a' # => 1
4430 * 'ab' <=> 'a' # => 1
4431 * 'a' <=> :a # => nil
4432 *
4433 * \Class \String includes module Comparable,
4434 * each of whose methods uses String#<=> for comparison.
4435 *
4436 * Related: see {Comparing}[rdoc-ref:String@Comparing].
4437 */
4438
4439static VALUE
4440rb_str_cmp_m(VALUE str1, VALUE str2)
4441{
4442 int result;
4443 VALUE s = rb_check_string_type(str2);
4444 if (NIL_P(s)) {
4445 return rb_invcmp(str1, str2);
4446 }
4447 result = rb_str_cmp(str1, s);
4448 return INT2FIX(result);
4449}
4450
4451static VALUE str_casecmp(VALUE str1, VALUE str2);
4452static VALUE str_casecmp_p(VALUE str1, VALUE str2);
4453
4454/*
4455 * call-seq:
4456 * casecmp(other_string) -> -1, 0, 1, or nil
4457 *
4458 * Ignoring case, compares +self+ and +other_string+; returns:
4459 *
4460 * - -1 if <tt>self.downcase</tt> is smaller than <tt>other_string.downcase</tt>.
4461 * - 0 if the two are equal.
4462 * - 1 if <tt>self.downcase</tt> is larger than <tt>other_string.downcase</tt>.
4463 * - +nil+ if the two are incomparable.
4464 *
4465 * See {Case Mapping}[rdoc-ref:case_mapping.rdoc].
4466 *
4467 * Examples:
4468 *
4469 * 'foo'.casecmp('goo') # => -1
4470 * 'goo'.casecmp('foo') # => 1
4471 * 'foo'.casecmp('food') # => -1
4472 * 'food'.casecmp('foo') # => 1
4473 * 'FOO'.casecmp('foo') # => 0
4474 * 'foo'.casecmp('FOO') # => 0
4475 * 'foo'.casecmp(1) # => nil
4476 *
4477 * Related: see {Comparing}[rdoc-ref:String@Comparing].
4478 */
4479
4480VALUE
4481rb_str_casecmp(VALUE str1, VALUE str2)
4482{
4483 VALUE s = rb_check_string_type(str2);
4484 if (NIL_P(s)) {
4485 return Qnil;
4486 }
4487 return str_casecmp(str1, s);
4488}
4489
4490static VALUE
4491str_casecmp(VALUE str1, VALUE str2)
4492{
4493 long len;
4494 rb_encoding *enc;
4495 const char *p1, *p1end, *p2, *p2end;
4496
4497 enc = rb_enc_compatible(str1, str2);
4498 if (!enc) {
4499 return Qnil;
4500 }
4501
4502 p1 = RSTRING_PTR(str1); p1end = RSTRING_END(str1);
4503 p2 = RSTRING_PTR(str2); p2end = RSTRING_END(str2);
4504 if (single_byte_optimizable(str1) && single_byte_optimizable(str2)) {
4505 while (p1 < p1end && p2 < p2end) {
4506 if (*p1 != *p2) {
4507 unsigned int c1 = TOLOWER(*p1 & 0xff);
4508 unsigned int c2 = TOLOWER(*p2 & 0xff);
4509 if (c1 != c2)
4510 return INT2FIX(c1 < c2 ? -1 : 1);
4511 }
4512 p1++;
4513 p2++;
4514 }
4515 }
4516 else {
4517 while (p1 < p1end && p2 < p2end) {
4518 int l1, c1 = rb_enc_ascget(p1, p1end, &l1, enc);
4519 int l2, c2 = rb_enc_ascget(p2, p2end, &l2, enc);
4520
4521 if (0 <= c1 && 0 <= c2) {
4522 c1 = TOLOWER(c1);
4523 c2 = TOLOWER(c2);
4524 if (c1 != c2)
4525 return INT2FIX(c1 < c2 ? -1 : 1);
4526 }
4527 else {
4528 int r;
4529 l1 = rb_enc_mbclen(p1, p1end, enc);
4530 l2 = rb_enc_mbclen(p2, p2end, enc);
4531 len = l1 < l2 ? l1 : l2;
4532 r = memcmp(p1, p2, len);
4533 if (r != 0)
4534 return INT2FIX(r < 0 ? -1 : 1);
4535 if (l1 != l2)
4536 return INT2FIX(l1 < l2 ? -1 : 1);
4537 }
4538 p1 += l1;
4539 p2 += l2;
4540 }
4541 }
4542 if (p1 == p1end && p2 == p2end) return INT2FIX(0);
4543 if (p1 == p1end) return INT2FIX(-1);
4544 return INT2FIX(1);
4545}
4546
4547/*
4548 * call-seq:
4549 * casecmp?(other_string) -> true, false, or nil
4550 *
4551 * Returns +true+ if +self+ and +other_string+ are equal after
4552 * Unicode case folding, +false+ if unequal, +nil+ if incomparable.
4553 *
4554 * See {Case Mapping}[rdoc-ref:case_mapping.rdoc].
4555 *
4556 * Examples:
4557 *
4558 * 'foo'.casecmp?('goo') # => false
4559 * 'goo'.casecmp?('foo') # => false
4560 * 'foo'.casecmp?('food') # => false
4561 * 'food'.casecmp?('foo') # => false
4562 * 'FOO'.casecmp?('foo') # => true
4563 * 'foo'.casecmp?('FOO') # => true
4564 * 'foo'.casecmp?(1) # => nil
4565 *
4566 * Related: see {Comparing}[rdoc-ref:String@Comparing].
4567 */
4568
4569static VALUE
4570rb_str_casecmp_p(VALUE str1, VALUE str2)
4571{
4572 VALUE s = rb_check_string_type(str2);
4573 if (NIL_P(s)) {
4574 return Qnil;
4575 }
4576 return str_casecmp_p(str1, s);
4577}
4578
4579static VALUE
4580str_casecmp_p(VALUE str1, VALUE str2)
4581{
4582 rb_encoding *enc;
4583 VALUE folded_str1, folded_str2;
4584 VALUE fold_opt = sym_fold;
4585
4586 enc = rb_enc_compatible(str1, str2);
4587 if (!enc) {
4588 return Qnil;
4589 }
4590
4591 if (is_ascii_string(str1) && is_ascii_string(str2)) {
4592 if (RSTRING_LEN(str1) != RSTRING_LEN(str2)) return Qfalse;
4593 const char *p1 = RSTRING_PTR(str1), *p1end = RSTRING_END(str1);
4594 const char *p2 = RSTRING_PTR(str2);
4595 while (p1 < p1end) {
4596 if (*p1 != *p2 && TOLOWER((unsigned char)*p1) != TOLOWER((unsigned char)*p2)) {
4597 return Qfalse;
4598 }
4599 p1++;
4600 p2++;
4601 }
4602 return Qtrue;
4603 }
4604
4605 folded_str1 = rb_str_downcase(1, &fold_opt, str1);
4606 folded_str2 = rb_str_downcase(1, &fold_opt, str2);
4607
4608 return rb_str_eql(folded_str1, folded_str2);
4609}
4610
4611static long
4612strseq_core(const char *str_ptr, const char *str_ptr_end, long str_len,
4613 const char *sub_ptr, long sub_len, long offset, rb_encoding *enc)
4614{
4615 const char *search_start = str_ptr;
4616 long pos, search_len = str_len - offset;
4617
4618 for (;;) {
4619 const char *t;
4620 pos = rb_memsearch(sub_ptr, sub_len, search_start, search_len, enc);
4621 if (pos < 0) return pos;
4622 t = rb_enc_right_char_head(search_start, search_start+pos, str_ptr_end, enc);
4623 if (t == search_start + pos) break;
4624 search_len -= t - search_start;
4625 if (search_len <= 0) return -1;
4626 offset += t - search_start;
4627 search_start = t;
4628 }
4629 return pos + offset;
4630}
4631
4632/* found index in byte */
4633#define rb_str_index(str, sub, offset) rb_strseq_index(str, sub, offset, 0)
4634#define rb_str_byteindex(str, sub, offset) rb_strseq_index(str, sub, offset, 1)
4635
4636static long
4637rb_strseq_index(VALUE str, VALUE sub, long offset, int in_byte)
4638{
4639 const char *str_ptr, *str_ptr_end, *sub_ptr;
4640 long str_len, sub_len;
4641 rb_encoding *enc;
4642
4643 enc = rb_enc_check(str, sub);
4644 if (is_broken_string(sub)) return -1;
4645
4646 str_ptr = RSTRING_PTR(str);
4647 str_ptr_end = RSTRING_END(str);
4648 str_len = RSTRING_LEN(str);
4649 sub_ptr = RSTRING_PTR(sub);
4650 sub_len = RSTRING_LEN(sub);
4651
4652 if (str_len < sub_len) return -1;
4653
4654 if (offset != 0) {
4655 long str_len_char, sub_len_char;
4656 int single_byte = single_byte_optimizable(str);
4657 str_len_char = (in_byte || single_byte) ? str_len : str_strlen(str, enc);
4658 sub_len_char = in_byte ? sub_len : str_strlen(sub, enc);
4659 if (offset < 0) {
4660 offset += str_len_char;
4661 if (offset < 0) return -1;
4662 }
4663 if (str_len_char - offset < sub_len_char) return -1;
4664 if (!in_byte) offset = str_offset(str_ptr, str_ptr_end, offset, enc, single_byte);
4665 str_ptr += offset;
4666 }
4667 if (sub_len == 0) return offset;
4668
4669 /* need proceed one character at a time */
4670 return strseq_core(str_ptr, str_ptr_end, str_len, sub_ptr, sub_len, offset, enc);
4671}
4672
4673
4674/*
4675 * call-seq:
4676 * index(pattern, offset = 0) -> integer or nil
4677 *
4678 * :include: doc/string/index.rdoc
4679 *
4680 */
4681
4682static VALUE
4683rb_str_index_m(int argc, VALUE *argv, VALUE str)
4684{
4685 VALUE sub;
4686 VALUE initpos;
4687 rb_encoding *enc = STR_ENC_GET(str);
4688 long pos;
4689
4690 if (rb_scan_args(argc, argv, "11", &sub, &initpos) == 2) {
4691 long slen = str_strlen(str, enc); /* str's enc */
4692 pos = NUM2LONG(initpos);
4693 if (pos < 0 ? (pos += slen) < 0 : pos > slen) {
4694 if (RB_TYPE_P(sub, T_REGEXP)) {
4696 }
4697 return Qnil;
4698 }
4699 }
4700 else {
4701 pos = 0;
4702 }
4703
4704 if (RB_TYPE_P(sub, T_REGEXP)) {
4705 pos = str_offset(RSTRING_PTR(str), RSTRING_END(str), pos,
4706 enc, single_byte_optimizable(str));
4707
4708 if (rb_reg_search(sub, str, pos, 0) >= 0) {
4709 VALUE match = rb_backref_get();
4710 pos = rb_str_sublen(str, RMATCH_BEG(match, 0));
4711 return LONG2NUM(pos);
4712 }
4713 }
4714 else {
4715 StringValue(sub);
4716 pos = rb_str_index(str, sub, pos);
4717 if (pos >= 0) {
4718 pos = rb_str_sublen(str, pos);
4719 return LONG2NUM(pos);
4720 }
4721 }
4722 return Qnil;
4723}
4724
4725/* Ensure that the given pos is a valid character boundary.
4726 * Note that in this function, "character" means a code point
4727 * (Unicode scalar value), not a grapheme cluster.
4728 */
4729static void
4730str_ensure_byte_pos(VALUE str, long pos)
4731{
4732 if (!single_byte_optimizable(str)) {
4733 const char *s = RSTRING_PTR(str);
4734 const char *e = RSTRING_END(str);
4735 const char *p = s + pos;
4736 if (!at_char_boundary(s, p, e, rb_enc_get(str))) {
4737 rb_raise(rb_eIndexError,
4738 "offset %ld does not land on character boundary", pos);
4739 }
4740 }
4741}
4742
4743/*
4744 * call-seq:
4745 * byteindex(object, offset = 0) -> integer or nil
4746 *
4747 * Returns the 0-based integer index of a substring of +self+
4748 * specified by +object+ (a string or Regexp) and +offset+,
4749 * or +nil+ if there is no such substring;
4750 * the returned index is the count of _bytes_ (not characters).
4751 *
4752 * When +object+ is a string,
4753 * returns the index of the first found substring equal to +object+:
4754 *
4755 * s = 'foo' # => "foo"
4756 * s.size # => 3 # Three 1-byte characters.
4757 * s.bytesize # => 3 # Three bytes.
4758 * s.byteindex('f') # => 0
4759 * s.byteindex('o') # => 1
4760 * s.byteindex('oo') # => 1
4761 * s.byteindex('ooo') # => nil
4762 *
4763 * When +object+ is a Regexp,
4764 * returns the index of the first found substring matching +object+;
4765 * updates {Regexp-related global variables}[rdoc-ref:Regexp@Global+Variables]:
4766 *
4767 * s = 'foo'
4768 * s.byteindex(/f/) # => 0
4769 * $~ # => #<MatchData "f">
4770 * s.byteindex(/o/) # => 1
4771 * s.byteindex(/oo/) # => 1
4772 * s.byteindex(/ooo/) # => nil
4773 * $~ # => nil
4774 *
4775 * \Integer argument +offset+, if given, specifies the 0-based index
4776 * of the byte where searching is to begin.
4777 *
4778 * When +offset+ is non-negative,
4779 * searching begins at byte position +offset+:
4780 *
4781 * s = 'foo'
4782 * s.byteindex('o', 1) # => 1
4783 * s.byteindex('o', 2) # => 2
4784 * s.byteindex('o', 3) # => nil
4785 *
4786 * When +offset+ is negative, counts backward from the end of +self+:
4787 *
4788 * s = 'foo'
4789 * s.byteindex('o', -1) # => 2
4790 * s.byteindex('o', -2) # => 1
4791 * s.byteindex('o', -3) # => 1
4792 * s.byteindex('o', -4) # => nil
4793 *
4794 * Raises IndexError if the byte at +offset+ is not the first byte of a character:
4795 *
4796 * s = "\uFFFF\uFFFF" # => "\uFFFF\uFFFF"
4797 * s.size # => 2 # Two 3-byte characters.
4798 * s.bytesize # => 6 # Six bytes.
4799 * s.byteindex("\uFFFF") # => 0
4800 * s.byteindex("\uFFFF", 1) # Raises IndexError
4801 * s.byteindex("\uFFFF", 2) # Raises IndexError
4802 * s.byteindex("\uFFFF", 3) # => 3
4803 * s.byteindex("\uFFFF", 4) # Raises IndexError
4804 * s.byteindex("\uFFFF", 5) # Raises IndexError
4805 * s.byteindex("\uFFFF", 6) # => nil
4806 *
4807 * Related: see {Querying}[rdoc-ref:String@Querying].
4808 */
4809
4810static VALUE
4811rb_str_byteindex_m(int argc, VALUE *argv, VALUE str)
4812{
4813 VALUE sub;
4814 VALUE initpos;
4815 long pos;
4816
4817 if (rb_scan_args(argc, argv, "11", &sub, &initpos) == 2) {
4818 pos = NUM2LONG(initpos);
4819 long slen = RSTRING_LEN(str);
4820 if (pos < 0 ? (pos += slen) < 0 : pos > slen) {
4821 if (RB_TYPE_P(sub, T_REGEXP)) {
4823 }
4824 return Qnil;
4825 }
4826 }
4827 else {
4828 pos = 0;
4829 }
4830
4831 str_ensure_byte_pos(str, pos);
4832
4833 if (RB_TYPE_P(sub, T_REGEXP)) {
4834 if (rb_reg_search(sub, str, pos, 0) >= 0) {
4835 VALUE match = rb_backref_get();
4836 pos = RMATCH_BEG(match, 0);
4837 return LONG2NUM(pos);
4838 }
4839 }
4840 else {
4841 StringValue(sub);
4842 pos = rb_str_byteindex(str, sub, pos);
4843 if (pos >= 0) return LONG2NUM(pos);
4844 }
4845 return Qnil;
4846}
4847
4848static long
4849str_rindex(VALUE str, VALUE sub, const char *s, rb_encoding *enc)
4850{
4851 const char *hit, *adjusted, *sbeg, *e, *t;
4852 int c;
4853 long slen, searchlen;
4854
4855 sbeg = RSTRING_PTR(str);
4856 slen = RSTRING_LEN(sub);
4857 if (slen == 0) return s - sbeg;
4858 e = RSTRING_END(str);
4859 t = RSTRING_PTR(sub);
4860 c = *t & 0xff;
4861 searchlen = s - sbeg + 1;
4862
4863 if (s + slen <= e && memcmp(s, t, slen) == 0) {
4864 return s - sbeg;
4865 }
4866
4867 do {
4868 hit = memrchr(sbeg, c, searchlen);
4869 if (!hit) break;
4870 adjusted = rb_enc_left_char_head(sbeg, hit, e, enc);
4871 if (hit != adjusted) {
4872 searchlen = adjusted - sbeg;
4873 continue;
4874 }
4875 if (hit + slen <= e && memcmp(hit, t, slen) == 0)
4876 return hit - sbeg;
4877 searchlen = adjusted - sbeg;
4878 } while (searchlen > 0);
4879
4880 return -1;
4881}
4882
4883/* found index in byte */
4884static long
4885rb_str_rindex(VALUE str, VALUE sub, long pos)
4886{
4887 long len, slen;
4888 const char *sbeg, *s;
4889 rb_encoding *enc;
4890 int singlebyte;
4891
4892 enc = rb_enc_check(str, sub);
4893 if (is_broken_string(sub)) return -1;
4894 singlebyte = single_byte_optimizable(str);
4895 len = singlebyte ? RSTRING_LEN(str) : str_strlen(str, enc); /* rb_enc_check */
4896 slen = str_strlen(sub, enc); /* rb_enc_check */
4897
4898 /* substring longer than string */
4899 if (len < slen) return -1;
4900 /* character counts, so the byte tail can still be shorter than sub */
4901 if (len - pos < slen) pos = len - slen;
4902 if (len == 0) return pos;
4903
4904 sbeg = RSTRING_PTR(str);
4905
4906 if (pos == 0) {
4907 if (RSTRING_LEN(sub) <= RSTRING_LEN(str) &&
4908 memcmp(sbeg, RSTRING_PTR(sub), RSTRING_LEN(sub)) == 0) {
4909 return 0;
4910 }
4911 else {
4912 return -1;
4913 }
4914 }
4915
4916 s = str_nth(sbeg, RSTRING_END(str), pos, enc, singlebyte);
4917 return str_rindex(str, sub, s, enc);
4918}
4919
4920/*
4921 * call-seq:
4922 * rindex(pattern, offset = self.length) -> integer or nil
4923 *
4924 * :include:doc/string/rindex.rdoc
4925 *
4926 */
4927
4928static VALUE
4929rb_str_rindex_m(int argc, VALUE *argv, VALUE str)
4930{
4931 VALUE sub;
4932 VALUE initpos;
4933 rb_encoding *enc = STR_ENC_GET(str);
4934 long pos, len = str_strlen(str, enc); /* str's enc */
4935
4936 if (rb_scan_args(argc, argv, "11", &sub, &initpos) == 2) {
4937 pos = NUM2LONG(initpos);
4938 if (pos < 0 && (pos += len) < 0) {
4939 if (RB_TYPE_P(sub, T_REGEXP)) {
4941 }
4942 return Qnil;
4943 }
4944 if (pos > len) pos = len;
4945 }
4946 else {
4947 pos = len;
4948 }
4949
4950 if (RB_TYPE_P(sub, T_REGEXP)) {
4951 /* enc = rb_enc_check(str, sub); */
4952 pos = str_offset(RSTRING_PTR(str), RSTRING_END(str), pos,
4953 enc, single_byte_optimizable(str));
4954
4955 if (rb_reg_search(sub, str, pos, 1) >= 0) {
4956 VALUE match = rb_backref_get();
4957 pos = rb_str_sublen(str, RMATCH_BEG(match, 0));
4958 return LONG2NUM(pos);
4959 }
4960 }
4961 else {
4962 StringValue(sub);
4963 pos = rb_str_rindex(str, sub, pos);
4964 if (pos >= 0) {
4965 pos = rb_str_sublen(str, pos);
4966 return LONG2NUM(pos);
4967 }
4968 }
4969 return Qnil;
4970}
4971
4972static long
4973rb_str_byterindex(VALUE str, VALUE sub, long pos)
4974{
4975 long len, slen;
4976 const char *sbeg, *s;
4977 rb_encoding *enc;
4978
4979 enc = rb_enc_check(str, sub);
4980 if (is_broken_string(sub)) return -1;
4981 len = RSTRING_LEN(str);
4982 slen = RSTRING_LEN(sub);
4983
4984 /* substring longer than string */
4985 if (len < slen) return -1;
4986 if (len - pos < slen) pos = len - slen;
4987 if (len == 0) return pos;
4988
4989 sbeg = RSTRING_PTR(str);
4990
4991 if (pos == 0) {
4992 if (memcmp(sbeg, RSTRING_PTR(sub), RSTRING_LEN(sub)) == 0)
4993 return 0;
4994 else
4995 return -1;
4996 }
4997
4998 s = sbeg + pos;
4999 return str_rindex(str, sub, s, enc);
5000}
5001
5002/*
5003 * call-seq:
5004 * byterindex(object, offset = self.bytesize) -> integer or nil
5005 *
5006 * Returns the 0-based integer index of a substring of +self+
5007 * that is the _last_ match for the given +object+ (a string or Regexp) and +offset+,
5008 * or +nil+ if there is no such substring;
5009 * the returned index is the count of _bytes_ (not characters).
5010 *
5011 * When +object+ is a string,
5012 * returns the index of the _last_ found substring equal to +object+:
5013 *
5014 * s = 'foo' # => "foo"
5015 * s.size # => 3 # Three 1-byte characters.
5016 * s.bytesize # => 3 # Three bytes.
5017 * s.byterindex('f') # => 0
5018 * s.byterindex('o') # => 2
5019 * s.byterindex('oo') # => 1
5020 * s.byterindex('ooo') # => nil
5021 *
5022 * When +object+ is a Regexp,
5023 * returns the index of the last found substring matching +object+;
5024 * updates {Regexp-related global variables}[rdoc-ref:Regexp@Global+Variables]:
5025 *
5026 * s = 'foo'
5027 * s.byterindex(/f/) # => 0
5028 * $~ # => #<MatchData "f">
5029 * s.byterindex(/o/) # => 2
5030 * s.byterindex(/oo/) # => 1
5031 * s.byterindex(/ooo/) # => nil
5032 * $~ # => nil
5033 *
5034 * The last match means starting at the possible last position,
5035 * not the last of the longest matches:
5036 *
5037 * s = 'foo'
5038 * s.byterindex(/o+/) # => 2
5039 * $~ #=> #<MatchData "o">
5040 *
5041 * To get the last longest match, use a negative lookbehind:
5042 *
5043 * s = 'foo'
5044 * s.byterindex(/(?<!o)o+/) # => 1
5045 * $~ # => #<MatchData "oo">
5046 *
5047 * Or use method #byteindex with negative lookahead:
5048 *
5049 * s = 'foo'
5050 * s.byteindex(/o+(?!.*o)/) # => 1
5051 * $~ #=> #<MatchData "oo">
5052 *
5053 * \Integer argument +offset+, if given, specifies the 0-based index
5054 * of the byte where searching is to end.
5055 *
5056 * When +offset+ is non-negative,
5057 * searching ends at byte position +offset+:
5058 *
5059 * s = 'foo'
5060 * s.byterindex('o', 0) # => nil
5061 * s.byterindex('o', 1) # => 1
5062 * s.byterindex('o', 2) # => 2
5063 * s.byterindex('o', 3) # => 2
5064 *
5065 * When +offset+ is negative, counts backward from the end of +self+:
5066 *
5067 * s = 'foo'
5068 * s.byterindex('o', -1) # => 2
5069 * s.byterindex('o', -2) # => 1
5070 * s.byterindex('o', -3) # => nil
5071 *
5072 * Raises IndexError if the byte at +offset+ is not the first byte of a character:
5073 *
5074 * s = "\uFFFF\uFFFF" # => "\uFFFF\uFFFF"
5075 * s.size # => 2 # Two 3-byte characters.
5076 * s.bytesize # => 6 # Six bytes.
5077 * s.byterindex("\uFFFF") # => 3
5078 * s.byterindex("\uFFFF", 1) # Raises IndexError
5079 * s.byterindex("\uFFFF", 2) # Raises IndexError
5080 * s.byterindex("\uFFFF", 3) # => 3
5081 * s.byterindex("\uFFFF", 4) # Raises IndexError
5082 * s.byterindex("\uFFFF", 5) # Raises IndexError
5083 * s.byterindex("\uFFFF", 6) # => nil
5084 *
5085 * Related: see {Querying}[rdoc-ref:String@Querying].
5086 */
5087
5088static VALUE
5089rb_str_byterindex_m(int argc, VALUE *argv, VALUE str)
5090{
5091 VALUE sub;
5092 VALUE initpos;
5093 long pos;
5094
5095 if (rb_scan_args(argc, argv, "11", &sub, &initpos) == 2) {
5096 pos = NUM2LONG(initpos);
5097 long len = RSTRING_LEN(str);
5098 if (pos < 0 && (pos += len) < 0) {
5099 if (RB_TYPE_P(sub, T_REGEXP)) {
5101 }
5102 return Qnil;
5103 }
5104 if (pos > len) pos = len;
5105 }
5106 else {
5107 pos = RSTRING_LEN(str);
5108 }
5109
5110 str_ensure_byte_pos(str, pos);
5111
5112 if (RB_TYPE_P(sub, T_REGEXP)) {
5113 if (rb_reg_search(sub, str, pos, 1) >= 0) {
5114 VALUE match = rb_backref_get();
5115 pos = RMATCH_BEG(match, 0);
5116 return LONG2NUM(pos);
5117 }
5118 }
5119 else {
5120 StringValue(sub);
5121 pos = rb_str_byterindex(str, sub, pos);
5122 if (pos >= 0) return LONG2NUM(pos);
5123 }
5124 return Qnil;
5125}
5126
5127/*
5128 * call-seq:
5129 * self =~ other -> integer or nil
5130 *
5131 * When +other+ is a Regexp:
5132 *
5133 * - Returns the integer index (in characters) of the first match
5134 * for +self+ and +other+, or +nil+ if none;
5135 * - Updates {Regexp-related global variables}[rdoc-ref:Regexp@Global+Variables].
5136 *
5137 * Examples:
5138 *
5139 * 'foo' =~ /f/ # => 0
5140 * $~ # => #<MatchData "f">
5141 * 'foo' =~ /o/ # => 1
5142 * $~ # => #<MatchData "o">
5143 * 'foo' =~ /x/ # => nil
5144 * $~ # => nil
5145 *
5146 * Note that <tt>string =~ regexp</tt> is different from <tt>regexp =~ string</tt>
5147 * (see Regexp#=~):
5148 *
5149 * number = nil
5150 * 'no. 9' =~ /(?<number>\d+)/ # => 4
5151 * number # => nil # Not assigned.
5152 * /(?<number>\d+)/ =~ 'no. 9' # => 4
5153 * number # => "9" # Assigned.
5154 *
5155 * When +other+ is not a Regexp, returns the value
5156 * returned by <tt>other =~ self</tt>.
5157 *
5158 * Related: see {Querying}[rdoc-ref:String@Querying].
5159 */
5160
5161static VALUE
5162rb_str_match(VALUE x, VALUE y)
5163{
5164 switch (OBJ_BUILTIN_TYPE(y)) {
5165 case T_STRING:
5166 rb_raise(rb_eTypeError, "type mismatch: String given");
5167
5168 case T_REGEXP:
5169 return rb_reg_match(y, x);
5170
5171 default:
5172 return rb_funcall(y, idEqTilde, 1, x);
5173 }
5174}
5175
5176
5177static VALUE get_pat(VALUE);
5178
5179
5180/*
5181 * call-seq:
5182 * match(pattern, offset = 0) -> matchdata or nil
5183 * match(pattern, offset = 0) {|matchdata| ... } -> object
5184 *
5185 * Creates a MatchData object based on +self+ and the given arguments;
5186 * updates {Regexp Global Variables}[rdoc-ref:Regexp@Global+Variables].
5187 *
5188 * - Computes +regexp+ by converting +pattern+ (if not already a Regexp).
5189 *
5190 * regexp = Regexp.new(pattern)
5191 *
5192 * - Calls <tt>regexp.match</tt> with +self+ to compute +matchdata+.
5193 * If +offset+ is given, it is also passed (see Regexp#match).
5194 *
5195 * With no block given, returns the computed +matchdata+ or +nil+:
5196 *
5197 * 'foo'.match('f') # => #<MatchData "f">
5198 * 'foo'.match('o') # => #<MatchData "o">
5199 * 'foo'.match('x') # => nil
5200 * 'foo'.match('f', 1) # => nil
5201 * 'foo'.match('o', 1) # => #<MatchData "o">
5202 *
5203 * With a block given and computed +matchdata+ non-nil, calls the block with +matchdata+;
5204 * returns the block's return value:
5205 *
5206 * 'foo'.match(/o/) {|matchdata| matchdata } # => #<MatchData "o">
5207 *
5208 * With a block given and +nil+ +matchdata+, does not call the block:
5209 *
5210 * 'foo'.match(/x/) {|matchdata| fail 'Cannot happen' } # => nil
5211 *
5212 * Related: see {Querying}[rdoc-ref:String@Querying].
5213 */
5214
5215static VALUE
5216rb_str_match_m(int argc, VALUE *argv, VALUE str)
5217{
5218 VALUE re, result;
5219 if (argc < 1)
5220 rb_check_arity(argc, 1, 2);
5221 re = argv[0];
5222 argv[0] = str;
5223 result = rb_funcallv(get_pat(re), rb_intern("match"), argc, argv);
5224 if (!NIL_P(result) && rb_block_given_p()) {
5225 return rb_yield(result);
5226 }
5227 return result;
5228}
5229
5230/*
5231 * call-seq:
5232 * match?(pattern, offset = 0) -> true or false
5233 *
5234 * Returns whether a match is found for +self+ and the given arguments;
5235 * does not update {Regexp Global Variables}[rdoc-ref:Regexp@Global+Variables].
5236 *
5237 * Computes +regexp+ by converting +pattern+ (if not already a Regexp):
5238 *
5239 * regexp = Regexp.new(pattern)
5240 *
5241 * The search for +regexp+ in +self+ begins at the given character +offset+.
5242 * Returns +true+ if a match is found, +false+ otherwise:
5243 *
5244 * 'foo'.match?(/o/) # => true
5245 * 'foo'.match?('o') # => true
5246 * 'foo'.match?(/x/) # => false
5247 * 'foo'.match?('f', 1) # => false
5248 * 'foo'.match?('o', 1) # => true
5249 *
5250 * Related: see {Querying}[rdoc-ref:String@Querying].
5251 */
5252
5253static VALUE
5254rb_str_match_m_p(int argc, VALUE *argv, VALUE str)
5255{
5256 VALUE re;
5257 rb_check_arity(argc, 1, 2);
5258 re = get_pat(argv[0]);
5259 return rb_reg_match_p(re, str, argc > 1 ? NUM2LONG(argv[1]) : 0);
5260}
5261
5262enum neighbor_char {
5263 NEIGHBOR_NOT_CHAR,
5264 NEIGHBOR_FOUND,
5265 NEIGHBOR_WRAPPED
5266};
5267
5268static enum neighbor_char
5269enc_succ_char(char *p, long len, rb_encoding *enc)
5270{
5271 long i;
5272 int l;
5273
5274 if (rb_enc_mbminlen(enc) > 1) {
5275 /* wchar, trivial case */
5276 int r = rb_enc_precise_mbclen(p, p + len, enc), c;
5277 if (!MBCLEN_CHARFOUND_P(r)) {
5278 return NEIGHBOR_NOT_CHAR;
5279 }
5280 c = rb_enc_mbc_to_codepoint(p, p + len, enc) + 1;
5281 l = rb_enc_code_to_mbclen(c, enc);
5282 if (!l) return NEIGHBOR_NOT_CHAR;
5283 if (l != len) return NEIGHBOR_WRAPPED;
5284 rb_enc_mbcput(c, p, enc);
5285 r = rb_enc_precise_mbclen(p, p + len, enc);
5286 if (!MBCLEN_CHARFOUND_P(r)) {
5287 return NEIGHBOR_NOT_CHAR;
5288 }
5289 return NEIGHBOR_FOUND;
5290 }
5291 while (1) {
5292 for (i = len-1; 0 <= i && (unsigned char)p[i] == 0xff; i--)
5293 p[i] = '\0';
5294 if (i < 0)
5295 return NEIGHBOR_WRAPPED;
5296 ++((unsigned char*)p)[i];
5297 l = rb_enc_precise_mbclen(p, p+len, enc);
5298 if (MBCLEN_CHARFOUND_P(l)) {
5299 l = MBCLEN_CHARFOUND_LEN(l);
5300 if (l == len) {
5301 return NEIGHBOR_FOUND;
5302 }
5303 else {
5304 memset(p+l, 0xff, len-l);
5305 }
5306 }
5307 if (MBCLEN_INVALID_P(l) && i < len-1) {
5308 long len2;
5309 int l2;
5310 for (len2 = len-1; 0 < len2; len2--) {
5311 l2 = rb_enc_precise_mbclen(p, p+len2, enc);
5312 if (!MBCLEN_INVALID_P(l2))
5313 break;
5314 }
5315 memset(p+len2+1, 0xff, len-(len2+1));
5316 }
5317 }
5318}
5319
5320static enum neighbor_char
5321enc_pred_char(char *p, long len, rb_encoding *enc)
5322{
5323 long i;
5324 int l;
5325 if (rb_enc_mbminlen(enc) > 1) {
5326 /* wchar, trivial case */
5327 int r = rb_enc_precise_mbclen(p, p + len, enc), c;
5328 if (!MBCLEN_CHARFOUND_P(r)) {
5329 return NEIGHBOR_NOT_CHAR;
5330 }
5331 c = rb_enc_mbc_to_codepoint(p, p + len, enc);
5332 if (!c) return NEIGHBOR_NOT_CHAR;
5333 --c;
5334 l = rb_enc_code_to_mbclen(c, enc);
5335 if (!l) return NEIGHBOR_NOT_CHAR;
5336 if (l != len) return NEIGHBOR_WRAPPED;
5337 rb_enc_mbcput(c, p, enc);
5338 r = rb_enc_precise_mbclen(p, p + len, enc);
5339 if (!MBCLEN_CHARFOUND_P(r)) {
5340 return NEIGHBOR_NOT_CHAR;
5341 }
5342 return NEIGHBOR_FOUND;
5343 }
5344 while (1) {
5345 for (i = len-1; 0 <= i && (unsigned char)p[i] == 0; i--)
5346 p[i] = '\xff';
5347 if (i < 0)
5348 return NEIGHBOR_WRAPPED;
5349 --((unsigned char*)p)[i];
5350 l = rb_enc_precise_mbclen(p, p+len, enc);
5351 if (MBCLEN_CHARFOUND_P(l)) {
5352 l = MBCLEN_CHARFOUND_LEN(l);
5353 if (l == len) {
5354 return NEIGHBOR_FOUND;
5355 }
5356 else {
5357 memset(p+l, 0, len-l);
5358 }
5359 }
5360 if (MBCLEN_INVALID_P(l) && i < len-1) {
5361 long len2;
5362 int l2;
5363 for (len2 = len-1; 0 < len2; len2--) {
5364 l2 = rb_enc_precise_mbclen(p, p+len2, enc);
5365 if (!MBCLEN_INVALID_P(l2))
5366 break;
5367 }
5368 memset(p+len2+1, 0, len-(len2+1));
5369 }
5370 }
5371}
5372
5373/*
5374 overwrite +p+ by succeeding letter in +enc+ and returns
5375 NEIGHBOR_FOUND or NEIGHBOR_WRAPPED.
5376 When NEIGHBOR_WRAPPED, carried-out letter is stored into carry.
5377 assuming each ranges are successive, and mbclen
5378 never change in each ranges.
5379 NEIGHBOR_NOT_CHAR is returned if invalid character or the range has only one
5380 character.
5381 */
5382static enum neighbor_char
5383enc_succ_alnum_char(char *p, long len, rb_encoding *enc, char *carry)
5384{
5385 enum neighbor_char ret;
5386 unsigned int c;
5387 int ctype;
5388 int range;
5389 char save[ONIGENC_CODE_TO_MBC_MAXLEN];
5390
5391 /* skip 03A2, invalid char between GREEK CAPITAL LETTERS */
5392 int try;
5393 const int max_gaps = 1;
5394
5395 c = rb_enc_mbc_to_codepoint(p, p+len, enc);
5396 if (rb_enc_isctype(c, ONIGENC_CTYPE_DIGIT, enc))
5397 ctype = ONIGENC_CTYPE_DIGIT;
5398 else if (rb_enc_isctype(c, ONIGENC_CTYPE_ALPHA, enc))
5399 ctype = ONIGENC_CTYPE_ALPHA;
5400 else
5401 return NEIGHBOR_NOT_CHAR;
5402
5403 MEMCPY(save, p, char, len);
5404 for (try = 0; try <= max_gaps; ++try) {
5405 ret = enc_succ_char(p, len, enc);
5406 if (ret == NEIGHBOR_FOUND) {
5407 c = rb_enc_mbc_to_codepoint(p, p+len, enc);
5408 if (rb_enc_isctype(c, ctype, enc))
5409 return NEIGHBOR_FOUND;
5410 }
5411 }
5412 MEMCPY(p, save, char, len);
5413 range = 1;
5414 while (1) {
5415 MEMCPY(save, p, char, len);
5416 ret = enc_pred_char(p, len, enc);
5417 if (ret == NEIGHBOR_FOUND) {
5418 c = rb_enc_mbc_to_codepoint(p, p+len, enc);
5419 if (!rb_enc_isctype(c, ctype, enc)) {
5420 MEMCPY(p, save, char, len);
5421 break;
5422 }
5423 }
5424 else {
5425 MEMCPY(p, save, char, len);
5426 break;
5427 }
5428 range++;
5429 }
5430 if (range == 1) {
5431 return NEIGHBOR_NOT_CHAR;
5432 }
5433
5434 if (ctype != ONIGENC_CTYPE_DIGIT) {
5435 MEMCPY(carry, p, char, len);
5436 return NEIGHBOR_WRAPPED;
5437 }
5438
5439 MEMCPY(carry, p, char, len);
5440 enc_succ_char(carry, len, enc);
5441 return NEIGHBOR_WRAPPED;
5442}
5443
5444
5445static VALUE str_succ(VALUE str);
5446
5447/*
5448 * call-seq:
5449 * succ -> new_str
5450 *
5451 * :include: doc/string/succ.rdoc
5452 *
5453 */
5454
5455VALUE
5457{
5458 VALUE str;
5459 str = rb_str_new(RSTRING_PTR(orig), RSTRING_LEN(orig));
5460 rb_enc_cr_str_copy_for_substr(str, orig);
5461 return str_succ(str);
5462}
5463
5464static VALUE
5465str_succ(VALUE str)
5466{
5467 rb_encoding *enc;
5468 char *sbeg, *s, *e, *last_alnum = 0;
5469 int found_alnum = 0;
5470 long l, slen;
5471 char carry[ONIGENC_CODE_TO_MBC_MAXLEN] = "\1";
5472 long carry_pos = 0, carry_len = 1;
5473 enum neighbor_char neighbor = NEIGHBOR_FOUND;
5474
5475 slen = RSTRING_LEN(str);
5476 if (slen == 0) return str;
5477
5478 enc = STR_ENC_GET(str);
5479 sbeg = RSTRING_PTR(str);
5480 s = e = sbeg + slen;
5481
5482 while ((s = rb_enc_prev_char(sbeg, s, e, enc)) != 0) {
5483 if (neighbor == NEIGHBOR_NOT_CHAR && last_alnum) {
5484 if (ISALPHA(*last_alnum) ? ISDIGIT(*s) :
5485 ISDIGIT(*last_alnum) ? ISALPHA(*s) : 0) {
5486 break;
5487 }
5488 }
5489 l = rb_enc_precise_mbclen(s, e, enc);
5490 if (!ONIGENC_MBCLEN_CHARFOUND_P(l)) continue;
5491 l = ONIGENC_MBCLEN_CHARFOUND_LEN(l);
5492 neighbor = enc_succ_alnum_char(s, l, enc, carry);
5493 switch (neighbor) {
5494 case NEIGHBOR_NOT_CHAR:
5495 continue;
5496 case NEIGHBOR_FOUND:
5497 return str;
5498 case NEIGHBOR_WRAPPED:
5499 last_alnum = s;
5500 break;
5501 }
5502 found_alnum = 1;
5503 carry_pos = s - sbeg;
5504 carry_len = l;
5505 }
5506 if (!found_alnum) { /* str contains no alnum */
5507 s = e;
5508 while ((s = rb_enc_prev_char(sbeg, s, e, enc)) != 0) {
5509 enum neighbor_char neighbor;
5510 char tmp[ONIGENC_CODE_TO_MBC_MAXLEN];
5511 l = rb_enc_precise_mbclen(s, e, enc);
5512 if (!ONIGENC_MBCLEN_CHARFOUND_P(l)) continue;
5513 l = ONIGENC_MBCLEN_CHARFOUND_LEN(l);
5514 MEMCPY(tmp, s, char, l);
5515 neighbor = enc_succ_char(tmp, l, enc);
5516 switch (neighbor) {
5517 case NEIGHBOR_FOUND:
5518 MEMCPY(s, tmp, char, l);
5519 return str;
5520 break;
5521 case NEIGHBOR_WRAPPED:
5522 MEMCPY(s, tmp, char, l);
5523 break;
5524 case NEIGHBOR_NOT_CHAR:
5525 break;
5526 }
5527 if (rb_enc_precise_mbclen(s, s+l, enc) != l) {
5528 /* wrapped to \0...\0. search next valid char. */
5529 enc_succ_char(s, l, enc);
5530 }
5531 if (!rb_enc_asciicompat(enc)) {
5532 MEMCPY(carry, s, char, l);
5533 carry_len = l;
5534 }
5535 carry_pos = s - sbeg;
5536 }
5538 }
5539 RESIZE_CAPA(str, slen + carry_len);
5540 sbeg = RSTRING_PTR(str);
5541 s = sbeg + carry_pos;
5542 memmove(s + carry_len, s, slen - carry_pos);
5543 memmove(s, carry, carry_len);
5544 slen += carry_len;
5545 STR_SET_LEN(str, slen);
5546 TERM_FILL(&sbeg[slen], rb_enc_mbminlen(enc));
5547 rb_enc_str_coderange(str);
5548 return str;
5549}
5550
5551
5552/*
5553 * call-seq:
5554 * succ! -> self
5555 *
5556 * Like String#succ, but modifies +self+ in place; returns +self+.
5557 *
5558 * Related: see {Modifying}[rdoc-ref:String@Modifying].
5559 */
5560
5561static VALUE
5562rb_str_succ_bang(VALUE str)
5563{
5564 rb_str_modify(str);
5565 str_succ(str);
5566 return str;
5567}
5568
5569static int
5570all_digits_p(const char *s, long len)
5571{
5572 while (len-- > 0) {
5573 if (!ISDIGIT(*s)) return 0;
5574 s++;
5575 }
5576 return 1;
5577}
5578
5579static int
5580str_upto_i(VALUE str, VALUE arg)
5581{
5582 rb_yield(str);
5583 return 0;
5584}
5585
5586/*
5587 * call-seq:
5588 * upto(other_string, exclusive = false) {|string| ... } -> self
5589 * upto(other_string, exclusive = false) -> new_enumerator
5590 *
5591 * :include: doc/string/upto.rdoc
5592 *
5593 */
5594
5595static VALUE
5596rb_str_upto(int argc, VALUE *argv, VALUE beg)
5597{
5598 VALUE end, exclusive;
5599
5600 rb_scan_args(argc, argv, "11", &end, &exclusive);
5601 RETURN_ENUMERATOR(beg, argc, argv);
5602 return rb_str_upto_each(beg, end, RTEST(exclusive), str_upto_i, Qnil);
5603}
5604
5605VALUE
5606rb_str_upto_each(VALUE beg, VALUE end, int excl, int (*each)(VALUE, VALUE), VALUE arg)
5607{
5608 VALUE current, after_end;
5609 ID succ;
5610 int n, ascii;
5611 rb_encoding *enc;
5612
5613 CONST_ID(succ, "succ");
5614 StringValue(end);
5615 enc = rb_enc_check(beg, end);
5616 ascii = (is_ascii_string(beg) && is_ascii_string(end));
5617 /* single character */
5618 if (RSTRING_LEN(beg) == 1 && RSTRING_LEN(end) == 1 && ascii) {
5619 char c = RSTRING_PTR(beg)[0];
5620 char e = RSTRING_PTR(end)[0];
5621
5622 if (c > e || (excl && c == e)) return beg;
5623 for (;;) {
5624 VALUE str = rb_enc_str_new(&c, 1, enc);
5626 if ((*each)(str, arg)) break;
5627 if (!excl && c == e) break;
5628 c++;
5629 if (excl && c == e) break;
5630 }
5631 return beg;
5632 }
5633 /* both edges are all digits */
5634 if (ascii && ISDIGIT(RSTRING_PTR(beg)[0]) && ISDIGIT(RSTRING_PTR(end)[0]) &&
5635 all_digits_p(RSTRING_PTR(beg), RSTRING_LEN(beg)) &&
5636 all_digits_p(RSTRING_PTR(end), RSTRING_LEN(end))) {
5637 VALUE b, e;
5638 int width;
5639
5640 width = RSTRING_LENINT(beg);
5641 b = rb_str_to_inum(beg, 10, FALSE);
5642 e = rb_str_to_inum(end, 10, FALSE);
5643 if (FIXNUM_P(b) && FIXNUM_P(e)) {
5644 long bi = FIX2LONG(b);
5645 long ei = FIX2LONG(e);
5646 rb_encoding *usascii = rb_usascii_encoding();
5647
5648 while (bi <= ei) {
5649 if (excl && bi == ei) break;
5650 if ((*each)(rb_enc_sprintf(usascii, "%.*ld", width, bi), arg)) break;
5651 bi++;
5652 }
5653 }
5654 else {
5655 ID op = excl ? '<' : idLE;
5656 VALUE args[2], fmt = rb_fstring_lit("%.*d");
5657
5658 args[0] = INT2FIX(width);
5659 while (rb_funcall(b, op, 1, e)) {
5660 args[1] = b;
5661 if ((*each)(rb_str_format(numberof(args), args, fmt), arg)) break;
5662 b = rb_funcallv(b, succ, 0, 0);
5663 }
5664 }
5665 return beg;
5666 }
5667 /* normal case */
5668 n = rb_str_cmp(beg, end);
5669 if (n > 0 || (excl && n == 0)) return beg;
5670
5671 after_end = rb_funcallv(end, succ, 0, 0);
5672 current = str_duplicate(rb_cString, beg);
5673 while (!rb_str_equal(current, after_end)) {
5674 VALUE next = Qnil;
5675 if (excl || !rb_str_equal(current, end))
5676 next = rb_funcallv(current, succ, 0, 0);
5677 if ((*each)(current, arg)) break;
5678 if (NIL_P(next)) break;
5679 current = next;
5680 StringValue(current);
5681 if (excl && rb_str_equal(current, end)) break;
5682 if (RSTRING_LEN(current) > RSTRING_LEN(end) || RSTRING_LEN(current) == 0)
5683 break;
5684 }
5685
5686 return beg;
5687}
5688
5689VALUE
5690rb_str_upto_endless_each(VALUE beg, int (*each)(VALUE, VALUE), VALUE arg)
5691{
5692 VALUE current;
5693 ID succ;
5694
5695 CONST_ID(succ, "succ");
5696 /* both edges are all digits */
5697 if (is_ascii_string(beg) && ISDIGIT(RSTRING_PTR(beg)[0]) &&
5698 all_digits_p(RSTRING_PTR(beg), RSTRING_LEN(beg))) {
5699 VALUE b, args[2], fmt = rb_fstring_lit("%.*d");
5700 int width = RSTRING_LENINT(beg);
5701 b = rb_str_to_inum(beg, 10, FALSE);
5702 if (FIXNUM_P(b)) {
5703 long bi = FIX2LONG(b);
5704 rb_encoding *usascii = rb_usascii_encoding();
5705
5706 while (FIXABLE(bi)) {
5707 if ((*each)(rb_enc_sprintf(usascii, "%.*ld", width, bi), arg)) break;
5708 bi++;
5709 }
5710 b = LONG2NUM(bi);
5711 }
5712 args[0] = INT2FIX(width);
5713 while (1) {
5714 args[1] = b;
5715 if ((*each)(rb_str_format(numberof(args), args, fmt), arg)) break;
5716 b = rb_funcallv(b, succ, 0, 0);
5717 }
5718 }
5719 /* normal case */
5720 current = str_duplicate(rb_cString, beg);
5721 while (1) {
5722 VALUE next = rb_funcallv(current, succ, 0, 0);
5723 if ((*each)(current, arg)) break;
5724 current = next;
5725 StringValue(current);
5726 if (RSTRING_LEN(current) == 0)
5727 break;
5728 }
5729
5730 return beg;
5731}
5732
5733static int
5734include_range_i(VALUE str, VALUE arg)
5735{
5736 VALUE *argp = (VALUE *)arg;
5737 if (!rb_equal(str, *argp)) return 0;
5738 *argp = Qnil;
5739 return 1;
5740}
5741
5742VALUE
5743rb_str_include_range_p(VALUE beg, VALUE end, VALUE val, VALUE exclusive)
5744{
5745 beg = rb_str_new_frozen(beg);
5746 StringValue(end);
5747 end = rb_str_new_frozen(end);
5748 if (NIL_P(val)) return Qfalse;
5749 val = rb_check_string_type(val);
5750 if (NIL_P(val)) return Qfalse;
5751 if (rb_enc_asciicompat(STR_ENC_GET(beg)) &&
5752 rb_enc_asciicompat(STR_ENC_GET(end)) &&
5753 rb_enc_asciicompat(STR_ENC_GET(val))) {
5754 const char *bp = RSTRING_PTR(beg);
5755 const char *ep = RSTRING_PTR(end);
5756 const char *vp = RSTRING_PTR(val);
5757 if (RSTRING_LEN(beg) == 1 && RSTRING_LEN(end) == 1) {
5758 if (RSTRING_LEN(val) == 0 || RSTRING_LEN(val) > 1)
5759 return Qfalse;
5760 else {
5761 char b = *bp;
5762 char e = *ep;
5763 char v = *vp;
5764
5765 if (ISASCII(b) && ISASCII(e) && ISASCII(v)) {
5766 if (b <= v && v < e) return Qtrue;
5767 return RBOOL(!RTEST(exclusive) && v == e);
5768 }
5769 }
5770 }
5771#if 0
5772 /* both edges are all digits */
5773 if (ISDIGIT(*bp) && ISDIGIT(*ep) &&
5774 all_digits_p(bp, RSTRING_LEN(beg)) &&
5775 all_digits_p(ep, RSTRING_LEN(end))) {
5776 /* TODO */
5777 }
5778#endif
5779 }
5780 rb_str_upto_each(beg, end, RTEST(exclusive), include_range_i, (VALUE)&val);
5781
5782 return RBOOL(NIL_P(val));
5783}
5784
5785static VALUE
5786rb_str_subpat(VALUE str, VALUE re, VALUE backref)
5787{
5788 if (rb_reg_search(re, str, 0, 0) >= 0) {
5789 VALUE match = rb_backref_get();
5790 int nth = rb_reg_backref_number(match, backref);
5791 return rb_reg_nth_match(nth, match);
5792 }
5793 return Qnil;
5794}
5795
5796static VALUE
5797rb_str_aref(VALUE str, VALUE indx)
5798{
5799 long idx;
5800
5801 if (FIXNUM_P(indx)) {
5802 idx = FIX2LONG(indx);
5803 }
5804 else if (RB_TYPE_P(indx, T_REGEXP)) {
5805 return rb_str_subpat(str, indx, INT2FIX(0));
5806 }
5807 else if (RB_TYPE_P(indx, T_STRING)) {
5808 if (rb_str_index(str, indx, 0) != -1)
5809 return str_duplicate(rb_cString, indx);
5810 return Qnil;
5811 }
5812 else {
5813 /* check if indx is Range */
5814 long beg, len = str_strlen(str, NULL);
5815 switch (rb_range_beg_len(indx, &beg, &len, len, 0)) {
5816 case Qfalse:
5817 break;
5818 case Qnil:
5819 return Qnil;
5820 default:
5821 return rb_str_substr(str, beg, len);
5822 }
5823 idx = NUM2LONG(indx);
5824 }
5825
5826 return str_substr(str, idx, 1, FALSE);
5827}
5828
5829
5830/*
5831 * call-seq:
5832 * self[offset] -> new_string or nil
5833 * self[offset, size] -> new_string or nil
5834 * self[range] -> new_string or nil
5835 * self[regexp, capture = 0] -> new_string or nil
5836 * self[substring] -> new_string or nil
5837 *
5838 * :include: doc/string/aref.rdoc
5839 *
5840 */
5841
5842static VALUE
5843rb_str_aref_m(int argc, VALUE *argv, VALUE str)
5844{
5845 if (argc == 2) {
5846 if (RB_TYPE_P(argv[0], T_REGEXP)) {
5847 return rb_str_subpat(str, argv[0], argv[1]);
5848 }
5849 else {
5850 return rb_str_substr_two_fixnums(str, argv[0], argv[1], TRUE);
5851 }
5852 }
5853 rb_check_arity(argc, 1, 2);
5854 return rb_str_aref(str, argv[0]);
5855}
5856
5857VALUE
5859{
5860 char *ptr = RSTRING_PTR(str);
5861 long olen = RSTRING_LEN(str), nlen;
5862
5863 str_modifiable(str);
5864 if (len > olen) len = olen;
5865 nlen = olen - len;
5866 if (str_embed_capa(str) >= nlen + TERM_LEN(str)) {
5867 char *oldptr = ptr;
5868 size_t old_capa = RSTRING(str)->as.heap.aux.capa + TERM_LEN(str);
5869 int fl = (int)(RBASIC(str)->flags & (STR_NOEMBED|STR_SHARED|STR_NOFREE));
5870 STR_SET_EMBED(str);
5871 ptr = RSTRING(str)->as.embed.ary;
5872 memmove(ptr, oldptr + len, nlen);
5873 if (fl == STR_NOEMBED) {
5874 SIZED_FREE_N(oldptr, old_capa);
5875 }
5876 }
5877 else {
5878 if (!STR_SHARED_P(str)) {
5879 VALUE shared = heap_str_make_shared(rb_obj_class(str), str, TERM_LEN(str));
5880 rb_enc_cr_str_exact_copy(shared, str);
5882 }
5883 ptr = RSTRING(str)->as.heap.ptr += len;
5884 }
5885 STR_SET_LEN(str, nlen);
5886
5887 if (!SHARABLE_MIDDLE_SUBSTRING) {
5888 TERM_FILL(ptr + nlen, TERM_LEN(str));
5889 }
5891 return str;
5892}
5893
5894static void
5895rb_str_update_1(VALUE str, long beg, long len, VALUE val, long vbeg, long vlen)
5896{
5897 char *sptr;
5898 long slen;
5899 int cr;
5900
5901 if (beg == 0 && vlen == 0) {
5902 rb_str_drop_bytes(str, len);
5903 return;
5904 }
5905
5906 str_modify_keep_cr(str);
5907 RSTRING_GETMEM(str, sptr, slen);
5908 if (len < vlen) {
5909 /* expand string */
5910 RESIZE_CAPA(str, slen + vlen - len);
5911 sptr = RSTRING_PTR(str);
5912 }
5913
5915 cr = rb_enc_str_coderange(val);
5916 else
5918
5919 if (vlen != len) {
5920 memmove(sptr + beg + vlen,
5921 sptr + beg + len,
5922 slen - (beg + len));
5923 }
5924 if (vlen < beg && len < 0) {
5925 MEMZERO(sptr + slen, char, -len);
5926 }
5927 if (vlen > 0) {
5928 memmove(sptr + beg, RSTRING_PTR(val) + vbeg, vlen);
5929 }
5930 slen += vlen - len;
5931 STR_SET_LEN(str, slen);
5932 TERM_FILL(&sptr[slen], TERM_LEN(str));
5933 ENC_CODERANGE_SET(str, cr);
5934}
5935
5936static inline void
5937rb_str_update_0(VALUE str, long beg, long len, VALUE val)
5938{
5939 rb_str_update_1(str, beg, len, val, 0, RSTRING_LEN(val));
5940}
5941
5942void
5943rb_str_update(VALUE str, long beg, long len, VALUE val)
5944{
5945 long slen;
5946 char *p, *e;
5947 rb_encoding *enc;
5948 int singlebyte = single_byte_optimizable(str);
5949 int cr;
5950
5951 if (len < 0) rb_raise(rb_eIndexError, "negative length %ld", len);
5952
5953 StringValue(val);
5954 enc = rb_enc_check(str, val);
5955 slen = str_strlen(str, enc); /* rb_enc_check */
5956
5957 if ((slen < beg) || ((beg < 0) && (beg + slen < 0))) {
5958 rb_raise(rb_eIndexError, "index %ld out of string", beg);
5959 }
5960 if (beg < 0) {
5961 beg += slen;
5962 }
5963 RUBY_ASSERT(beg >= 0);
5964 RUBY_ASSERT(beg <= slen);
5965
5966 if (len > slen - beg) {
5967 len = slen - beg;
5968 }
5969 p = str_nth(RSTRING_PTR(str), RSTRING_END(str), beg, enc, singlebyte);
5970 if (!p) p = RSTRING_END(str);
5971 e = str_nth(p, RSTRING_END(str), len, enc, singlebyte);
5972 if (!e) e = RSTRING_END(str);
5973 /* error check */
5974 beg = p - RSTRING_PTR(str); /* physical position */
5975 len = e - p; /* physical length */
5976 rb_str_update_0(str, beg, len, val);
5977 rb_enc_associate(str, enc);
5979 if (cr != ENC_CODERANGE_BROKEN)
5980 ENC_CODERANGE_SET(str, cr);
5981}
5982
5983static void
5984rb_str_subpat_set(VALUE str, VALUE re, VALUE backref, VALUE val)
5985{
5986 int nth;
5987 VALUE match;
5988 long start, end, len;
5989 rb_encoding *enc;
5990
5991 if (rb_reg_search(re, str, 0, 0) < 0) {
5992 rb_raise(rb_eIndexError, "regexp not matched");
5993 }
5994 match = rb_backref_get();
5995 nth = rb_reg_backref_number(match, backref);
5996 int num_regs = RMATCH_NREGS(match);
5997 if ((nth >= num_regs) || ((nth < 0) && (-nth >= num_regs))) {
5998 rb_raise(rb_eIndexError, "index %d out of regexp", nth);
5999 }
6000 if (nth < 0) {
6001 nth += num_regs;
6002 }
6003
6004 start = RMATCH_BEG(match, nth);
6005 if (start == -1) {
6006 rb_raise(rb_eIndexError, "regexp group %d not matched", nth);
6007 }
6008 end = RMATCH_END(match, nth);
6009 len = end - start;
6010
6011 StringValue(val);
6012 if (start + len > RSTRING_LEN(str)) {
6013 rb_raise(rb_eRuntimeError, "string modified");
6014 }
6015
6016 enc = rb_enc_check_str(str, val);
6017 rb_str_update_0(str, start, len, val);
6018 rb_enc_associate(str, enc);
6019}
6020
6021static VALUE
6022rb_str_aset(VALUE str, VALUE indx, VALUE val)
6023{
6024 long idx, beg;
6025
6026 switch (TYPE(indx)) {
6027 case T_REGEXP:
6028 rb_str_subpat_set(str, indx, INT2FIX(0), val);
6029 return val;
6030
6031 case T_STRING:
6032 beg = rb_str_index(str, indx, 0);
6033 if (beg < 0) {
6034 rb_raise(rb_eIndexError, "string not matched");
6035 }
6036 beg = rb_str_sublen(str, beg);
6037 rb_str_update(str, beg, str_strlen(indx, NULL), val);
6038 return val;
6039
6040 default:
6041 /* check if indx is Range */
6042 {
6043 long beg, len;
6044 if (rb_range_beg_len(indx, &beg, &len, str_strlen(str, NULL), 2)) {
6045 rb_str_update(str, beg, len, val);
6046 return val;
6047 }
6048 }
6049 /* FALLTHROUGH */
6050
6051 case T_FIXNUM:
6052 idx = NUM2LONG(indx);
6053 rb_str_update(str, idx, 1, val);
6054 return val;
6055 }
6056}
6057
6058/*
6059 * call-seq:
6060 * self[index] = other_string -> new_string
6061 * self[start, length] = other_string -> new_string
6062 * self[range] = other_string -> new_string
6063 * self[regexp, capture = 0] = other_string -> new_string
6064 * self[substring] = other_string -> new_string
6065 *
6066 * :include: doc/string/aset.rdoc
6067 *
6068 */
6069
6070static VALUE
6071rb_str_aset_m(int argc, VALUE *argv, VALUE str)
6072{
6073 if (argc == 3) {
6074 if (RB_TYPE_P(argv[0], T_REGEXP)) {
6075 rb_str_subpat_set(str, argv[0], argv[1], argv[2]);
6076 }
6077 else {
6078 rb_str_update(str, NUM2LONG(argv[0]), NUM2LONG(argv[1]), argv[2]);
6079 }
6080 return argv[2];
6081 }
6082 rb_check_arity(argc, 2, 3);
6083 return rb_str_aset(str, argv[0], argv[1]);
6084}
6085
6086/*
6087 * call-seq:
6088 * insert(offset, other_string) -> self
6089 *
6090 * :include: doc/string/insert.rdoc
6091 *
6092 */
6093
6094static VALUE
6095rb_str_insert(VALUE str, VALUE idx, VALUE str2)
6096{
6097 long pos = NUM2LONG(idx);
6098
6099 if (pos == -1) {
6100 return rb_str_append(str, str2);
6101 }
6102 else if (pos < 0) {
6103 pos++;
6104 }
6105 rb_str_update(str, pos, 0, str2);
6106 return str;
6107}
6108
6109
6110/*
6111 * call-seq:
6112 * slice!(index) -> new_string or nil
6113 * slice!(start, length) -> new_string or nil
6114 * slice!(range) -> new_string or nil
6115 * slice!(regexp, capture = 0) -> new_string or nil
6116 * slice!(substring) -> new_string or nil
6117 *
6118 * Like String#[] (and its alias String#slice), except that:
6119 *
6120 * - Performs substitutions in +self+ (not in a copy of +self+).
6121 * - Returns the removed substring if any modifications were made, +nil+ otherwise.
6122 *
6123 * A few examples:
6124 *
6125 * s = 'hello'
6126 * s.slice!('e') # => "e"
6127 * s # => "hllo"
6128 * s.slice!('e') # => nil
6129 * s # => "hllo"
6130 *
6131 * Related: see {Modifying}[rdoc-ref:String@Modifying].
6132 */
6133
6134static VALUE
6135rb_str_slice_bang(int argc, VALUE *argv, VALUE str)
6136{
6137 VALUE result = Qnil;
6138 VALUE indx;
6139 long beg, len = 1;
6140 char *p;
6141
6142 rb_check_arity(argc, 1, 2);
6143 str_modify_keep_cr(str);
6144 indx = argv[0];
6145 if (RB_TYPE_P(indx, T_REGEXP)) {
6146 if (rb_reg_search(indx, str, 0, 0) < 0) return Qnil;
6147 VALUE match = rb_backref_get();
6148 int num_regs = RMATCH_NREGS(match);
6149 int nth = 0;
6150 if (argc > 1 && (nth = rb_reg_backref_number(match, argv[1])) < 0) {
6151 if ((nth += num_regs) <= 0) return Qnil;
6152 }
6153 else if (nth >= num_regs) return Qnil;
6154 beg = RMATCH_BEG(match, nth);
6155 len = RMATCH_END(match, nth) - beg;
6156 /* Converting the backref may have modified the string. */
6157 if (beg > RSTRING_LEN(str)) return Qnil;
6158 if (len > RSTRING_LEN(str) - beg) len = RSTRING_LEN(str) - beg;
6159 goto subseq;
6160 }
6161 else if (argc == 2) {
6162 beg = NUM2LONG(indx);
6163 len = NUM2LONG(argv[1]);
6164 goto num_index;
6165 }
6166 else if (FIXNUM_P(indx)) {
6167 beg = FIX2LONG(indx);
6168 if (!(p = rb_str_subpos(str, beg, &len))) return Qnil;
6169 if (!len) return Qnil;
6170 beg = p - RSTRING_PTR(str);
6171 goto subseq;
6172 }
6173 else if (RB_TYPE_P(indx, T_STRING)) {
6174 beg = rb_str_index(str, indx, 0);
6175 if (beg == -1) return Qnil;
6176 len = RSTRING_LEN(indx);
6177 result = str_duplicate(rb_cString, indx);
6178 goto squash;
6179 }
6180 else {
6181 switch (rb_range_beg_len(indx, &beg, &len, str_strlen(str, NULL), 0)) {
6182 case Qnil:
6183 return Qnil;
6184 case Qfalse:
6185 beg = NUM2LONG(indx);
6186 if (!(p = rb_str_subpos(str, beg, &len))) return Qnil;
6187 if (!len) return Qnil;
6188 beg = p - RSTRING_PTR(str);
6189 goto subseq;
6190 default:
6191 goto num_index;
6192 }
6193 }
6194
6195 num_index:
6196 if (!(p = rb_str_subpos(str, beg, &len))) return Qnil;
6197 beg = p - RSTRING_PTR(str);
6198
6199 subseq:
6200 result = rb_str_new(RSTRING_PTR(str)+beg, len);
6201 rb_enc_cr_str_copy_for_substr(result, str);
6202
6203 squash:
6204 if (len > 0) {
6205 if (beg == 0) {
6206 rb_str_drop_bytes(str, len);
6207 }
6208 else {
6209 char *sptr = RSTRING_PTR(str);
6210 long slen = RSTRING_LEN(str);
6211 if (beg + len > slen) /* pathological check */
6212 len = slen - beg;
6213 memmove(sptr + beg,
6214 sptr + beg + len,
6215 slen - (beg + len));
6216 slen -= len;
6217 STR_SET_LEN(str, slen);
6218 TERM_FILL(&sptr[slen], TERM_LEN(str));
6219 }
6220 }
6221 return result;
6222}
6223
6224static VALUE
6225get_pat(VALUE pat)
6226{
6227 VALUE val;
6228
6229 switch (OBJ_BUILTIN_TYPE(pat)) {
6230 case T_REGEXP:
6231 return pat;
6232
6233 case T_STRING:
6234 break;
6235
6236 default:
6237 val = rb_check_string_type(pat);
6238 if (NIL_P(val)) {
6239 Check_Type(pat, T_REGEXP);
6240 }
6241 pat = val;
6242 }
6243
6244 return rb_reg_regcomp(pat);
6245}
6246
6247static VALUE
6248get_pat_quoted(VALUE pat, int check)
6249{
6250 VALUE val;
6251
6252 switch (OBJ_BUILTIN_TYPE(pat)) {
6253 case T_REGEXP:
6254 return pat;
6255
6256 case T_STRING:
6257 break;
6258
6259 default:
6260 val = rb_check_string_type(pat);
6261 if (NIL_P(val)) {
6262 Check_Type(pat, T_REGEXP);
6263 }
6264 pat = val;
6265 }
6266 if (check && is_broken_string(pat)) {
6267 rb_exc_raise(rb_reg_check_preprocess(pat));
6268 }
6269 return pat;
6270}
6271
6272static long
6273rb_pat_search0(VALUE pat, VALUE str, long pos, int set_backref_str, VALUE *match)
6274{
6275 if (BUILTIN_TYPE(pat) == T_STRING) {
6276 pos = rb_str_byteindex(str, pat, pos);
6277 if (set_backref_str) {
6278 if (pos >= 0) {
6279 str = rb_str_new_frozen_String(str);
6280 VALUE match_data = rb_backref_set_string(str, pos, RSTRING_LEN(pat));
6281 if (match) {
6282 *match = match_data;
6283 }
6284 }
6285 else {
6287 }
6288 }
6289 return pos;
6290 }
6291 else {
6292 return rb_reg_search0(pat, str, pos, 0, set_backref_str, match);
6293 }
6294}
6295
6296static long
6297rb_pat_search(VALUE pat, VALUE str, long pos, int set_backref_str)
6298{
6299 return rb_pat_search0(pat, str, pos, set_backref_str, NULL);
6300}
6301
6302
6303/*
6304 * call-seq:
6305 * sub!(pattern, replacement) -> self or nil
6306 * sub!(pattern) {|match| ... } -> self or nil
6307 *
6308 * Like String#sub, except that:
6309 *
6310 * - Changes are made to +self+, not to copy of +self+.
6311 * - Returns +self+ if any changes are made, +nil+ otherwise.
6312 *
6313 * Related: see {Modifying}[rdoc-ref:String@Modifying].
6314 */
6315
6316static VALUE
6317rb_str_sub_bang(int argc, VALUE *argv, VALUE str)
6318{
6319 VALUE pat, repl, hash = Qnil;
6320 int iter = 0;
6321 long plen;
6322 int min_arity = rb_block_given_p() ? 1 : 2;
6323 long beg;
6324
6325 rb_check_arity(argc, min_arity, 2);
6326 if (argc == 1) {
6327 iter = 1;
6328 }
6329 else {
6330 repl = argv[1];
6331 if (!RB_TYPE_P(repl, T_STRING)) {
6332 hash = rb_check_hash_type(repl);
6333 if (NIL_P(hash)) {
6334 StringValue(repl);
6335 }
6336 }
6337 }
6338
6339 pat = get_pat_quoted(argv[0], 1);
6340
6341 str_modifiable(str);
6342 beg = rb_pat_search(pat, str, 0, 1);
6343 if (beg >= 0) {
6344 rb_encoding *enc;
6345 int cr = ENC_CODERANGE(str);
6346 long beg0, end0;
6347 VALUE match, match0 = Qnil;
6348 char *p, *rp;
6349 long len, rlen;
6350
6351 match = rb_backref_get();
6352 if (RB_TYPE_P(pat, T_STRING)) {
6353 beg0 = beg;
6354 end0 = beg0 + RSTRING_LEN(pat);
6355 match0 = pat;
6356 }
6357 else {
6358 beg0 = RMATCH_BEG(match, 0);
6359 end0 = RMATCH_END(match, 0);
6360 if (iter) match0 = rb_reg_nth_match(0, match);
6361 }
6362
6363 if (iter || !NIL_P(hash)) {
6364 p = RSTRING_PTR(str); len = RSTRING_LEN(str);
6365
6366 if (iter) {
6367 repl = rb_obj_as_string(rb_yield(match0));
6368 }
6369 else {
6370 repl = rb_hash_aref(hash, rb_str_subseq(str, beg0, end0 - beg0));
6371 repl = rb_obj_as_string(repl);
6372 }
6373 str_mod_check(str, p, len);
6374 rb_check_frozen(str);
6375 }
6376 else {
6377 repl = rb_reg_regsub_match(repl, str, match);
6378 }
6379
6380 enc = rb_enc_compatible(str, repl);
6381 if (!enc) {
6382 rb_encoding *str_enc = STR_ENC_GET(str);
6383 p = RSTRING_PTR(str); len = RSTRING_LEN(str);
6384 if (coderange_scan(p, beg0, str_enc) != ENC_CODERANGE_7BIT ||
6385 coderange_scan(p+end0, len-end0, str_enc) != ENC_CODERANGE_7BIT) {
6386 rb_raise(rb_eEncCompatError, "incompatible character encodings: %s and %s",
6387 rb_enc_inspect_name(str_enc),
6388 rb_enc_inspect_name(STR_ENC_GET(repl)));
6389 }
6390 enc = STR_ENC_GET(repl);
6391 }
6392 rb_str_modify(str);
6393 rb_enc_associate(str, enc);
6395 int cr2 = ENC_CODERANGE(repl);
6396 if (cr2 == ENC_CODERANGE_BROKEN ||
6397 (cr == ENC_CODERANGE_VALID && cr2 == ENC_CODERANGE_7BIT))
6399 else
6400 cr = cr2;
6401 }
6402 plen = end0 - beg0;
6403 rlen = RSTRING_LEN(repl);
6404 len = RSTRING_LEN(str);
6405 if (rlen > plen) {
6406 RESIZE_CAPA(str, len + rlen - plen);
6407 }
6408 p = RSTRING_PTR(str);
6409 if (rlen != plen) {
6410 memmove(p + beg0 + rlen, p + beg0 + plen, len - beg0 - plen);
6411 }
6412 rp = RSTRING_PTR(repl);
6413 memmove(p + beg0, rp, rlen);
6414 len += rlen - plen;
6415 STR_SET_LEN(str, len);
6416 TERM_FILL(&RSTRING_PTR(str)[len], TERM_LEN(str));
6417 ENC_CODERANGE_SET(str, cr);
6418
6419 RB_GC_GUARD(match);
6420
6421 return str;
6422 }
6423 return Qnil;
6424}
6425
6426
6427/*
6428 * call-seq:
6429 * sub(pattern, replacement) -> new_string
6430 * sub(pattern) {|match| ... } -> new_string
6431 *
6432 * :include: doc/string/sub.rdoc
6433 */
6434
6435static VALUE
6436rb_str_sub(int argc, VALUE *argv, VALUE str)
6437{
6438 str = str_duplicate(rb_cString, str);
6439 rb_str_sub_bang(argc, argv, str);
6440 return str;
6441}
6442
6443static VALUE
6444str_gsub(int argc, VALUE *argv, VALUE str, int bang)
6445{
6446 VALUE pat, val = Qnil, repl, match0 = Qnil, dest, hash = Qnil, match = Qnil;
6447 long beg, beg0, end0;
6448 long offset, blen, slen, len, last;
6449 enum {STR, ITER, FAST_MAP, MAP} mode = STR;
6450 char *sp, *cp;
6451 int need_backref_str = -1;
6452 rb_encoding *str_enc;
6453
6454 switch (argc) {
6455 case 1:
6456 RETURN_ENUMERATOR(str, argc, argv);
6457 mode = ITER;
6458 break;
6459 case 2:
6460 repl = argv[1];
6461 if (!RB_TYPE_P(repl, T_STRING)) {
6462 hash = rb_check_hash_type(repl);
6463 if (NIL_P(hash)) {
6464 StringValue(repl);
6465 }
6466 else if (rb_hash_default_unredefined(hash) && !FL_TEST_RAW(hash, RHASH_PROC_DEFAULT)) {
6467 mode = FAST_MAP;
6468 }
6469 else {
6470 mode = MAP;
6471 }
6472 }
6473 break;
6474 default:
6475 rb_error_arity(argc, 1, 2);
6476 }
6477
6478 pat = get_pat_quoted(argv[0], 1);
6479 beg = rb_pat_search0(pat, str, 0, need_backref_str, &match);
6480
6481 if (beg < 0) {
6482 if (bang) return Qnil; /* no match, no substitution */
6483 return str_duplicate(rb_cString, str);
6484 }
6485 if (bang) str_modify_keep_cr(str);
6486
6487 offset = 0;
6488 blen = RSTRING_LEN(str) + 30; /* len + margin */
6489 dest = rb_str_buf_new(blen);
6490 sp = RSTRING_PTR(str);
6491 slen = RSTRING_LEN(str);
6492 cp = sp;
6493 str_enc = STR_ENC_GET(str);
6494 rb_enc_associate(dest, str_enc);
6495 ENC_CODERANGE_SET(dest, rb_enc_asciicompat(str_enc) ? ENC_CODERANGE_7BIT : ENC_CODERANGE_VALID);
6496
6497 do {
6498 if (RB_TYPE_P(pat, T_STRING)) {
6499 beg0 = beg;
6500 end0 = beg0 + RSTRING_LEN(pat);
6501 match0 = pat;
6502 }
6503 else {
6504 beg0 = RMATCH_BEG(match, 0);
6505 end0 = RMATCH_END(match, 0);
6506 if (mode == ITER) match0 = rb_reg_nth_match(0, match);
6507 }
6508
6509 if (mode != STR) {
6510 if (mode == ITER) {
6511 val = rb_obj_as_string(rb_yield(match0));
6512 }
6513 else {
6514 struct RString fake_str = {RBASIC_INIT};
6515 VALUE key;
6516 if (mode == FAST_MAP) {
6517 // It is safe to use a fake_str here because we established that it won't escape,
6518 // as it's only used for `rb_hash_aref` and we checked the hash doesn't have a
6519 // default proc.
6520 key = setup_fake_str(&fake_str, sp + beg0, end0 - beg0, ENCODING_GET_INLINED(str));
6521 }
6522 else {
6523 key = rb_str_subseq(str, beg0, end0 - beg0);
6524 }
6525 val = rb_hash_aref(hash, key);
6526 val = rb_obj_as_string(val);
6527 }
6528 str_mod_check(str, sp, slen);
6529 if (val == dest) { /* paranoid check [ruby-dev:24827] */
6530 rb_raise(rb_eRuntimeError, "block should not cheat");
6531 }
6532 }
6533 else if (need_backref_str) {
6534 val = rb_reg_regsub_match(repl, str, match);
6535 if (need_backref_str < 0) {
6536 need_backref_str = val != repl;
6537 }
6538 }
6539 else {
6540 val = repl;
6541 }
6542
6543 len = beg0 - offset; /* copy pre-match substr */
6544 if (len) {
6545 rb_enc_str_buf_cat(dest, cp, len, str_enc);
6546 }
6547
6548 rb_str_buf_append(dest, val);
6549
6550 last = offset;
6551 offset = end0;
6552 if (beg0 == end0) {
6553 /*
6554 * Always consume at least one character of the input string
6555 * in order to prevent infinite loops.
6556 */
6557 if (RSTRING_LEN(str) <= end0) break;
6558 len = rb_enc_fast_mbclen(RSTRING_PTR(str)+end0, RSTRING_END(str), str_enc);
6559 rb_enc_str_buf_cat(dest, RSTRING_PTR(str)+end0, len, str_enc);
6560 offset = end0 + len;
6561 }
6562 cp = RSTRING_PTR(str) + offset;
6563 if (offset > RSTRING_LEN(str)) break;
6564
6565 // In FAST_MAP and STR mode the backref can't escape so we can re-use the MatchData safely.
6566 if (mode != FAST_MAP && mode != STR) {
6567 match = Qnil;
6568 }
6569 beg = rb_pat_search0(pat, str, offset, need_backref_str, &match);
6570
6571 RB_GC_GUARD(match);
6572 } while (beg >= 0);
6573
6574 if (RSTRING_LEN(str) > offset) {
6575 rb_enc_str_buf_cat(dest, cp, RSTRING_LEN(str) - offset, str_enc);
6576 }
6577 rb_pat_search0(pat, str, last, 1, &match);
6578 if (bang) {
6579 str_shared_replace(str, dest);
6580 }
6581 else {
6582 str = dest;
6583 }
6584
6585 return str;
6586}
6587
6588
6589/*
6590 * call-seq:
6591 * gsub!(pattern, replacement) -> self or nil
6592 * gsub!(pattern) {|match| ... } -> self or nil
6593 * gsub!(pattern) -> an_enumerator
6594 *
6595 * Like String#gsub, except that:
6596 *
6597 * - Performs substitutions in +self+ (not in a copy of +self+).
6598 * - Returns +self+ if any substitutions were performed, +nil+ otherwise.
6599 *
6600 * Related: see {Modifying}[rdoc-ref:String@Modifying].
6601 */
6602
6603static VALUE
6604rb_str_gsub_bang(int argc, VALUE *argv, VALUE str)
6605{
6606 str_modifiable(str);
6607 return str_gsub(argc, argv, str, 1);
6608}
6609
6610
6611/*
6612 * call-seq:
6613 * gsub(pattern, replacement) -> new_string
6614 * gsub(pattern) {|match| ... } -> new_string
6615 * gsub(pattern) -> enumerator
6616 *
6617 * Returns a copy of +self+ with zero or more substrings replaced.
6618 *
6619 * Argument +pattern+ may be a string or a Regexp;
6620 * argument +replacement+ may be a string or a Hash.
6621 * Varying types for the argument values makes this method very versatile.
6622 *
6623 * Below are some simple examples;
6624 * for many more examples, see {Substitution Methods}[rdoc-ref:String@Substitution+Methods].
6625 *
6626 * With arguments +pattern+ and string +replacement+ given,
6627 * replaces each matching substring with the given +replacement+ string:
6628 *
6629 * s = 'abracadabra'
6630 * s.gsub('ab', 'AB') # => "ABracadABra"
6631 * s.gsub(/[a-c]/, 'X') # => "XXrXXXdXXrX"
6632 *
6633 * With arguments +pattern+ and hash +replacement+ given,
6634 * replaces each matching substring with a value from the given +replacement+ hash,
6635 * or removes it:
6636 *
6637 * h = {'a' => 'A', 'b' => 'B', 'c' => 'C'}
6638 * s.gsub(/[a-c]/, h) # => "ABrACAdABrA" # 'a', 'b', 'c' replaced.
6639 * s.gsub(/[a-d]/, h) # => "ABrACAABrA" # 'd' removed.
6640 *
6641 * With argument +pattern+ and a block given,
6642 * calls the block with each matching substring;
6643 * replaces that substring with the block's return value:
6644 *
6645 * s.gsub(/[a-d]/) {|substring| substring.upcase }
6646 * # => "ABrACADABrA"
6647 *
6648 * With argument +pattern+ and no block given,
6649 * returns a new Enumerator.
6650 *
6651 * Related: see {Converting to New String}[rdoc-ref:String@Converting+to+New+String].
6652 */
6653
6654static VALUE
6655rb_str_gsub(int argc, VALUE *argv, VALUE str)
6656{
6657 return str_gsub(argc, argv, str, 0);
6658}
6659
6660
6661/*
6662 * call-seq:
6663 * replace(other_string) -> self
6664 *
6665 * Replaces the contents of +self+ with the contents of +other_string+;
6666 * returns +self+:
6667 *
6668 * s = 'foo' # => "foo"
6669 * s.replace('bar') # => "bar"
6670 *
6671 * Related: see {Modifying}[rdoc-ref:String@Modifying].
6672 */
6673
6674VALUE
6676{
6677 str_modifiable(str);
6678 if (str == str2) return str;
6679
6680 StringValue(str2);
6681 str_discard(str);
6682 return str_replace(str, str2);
6683}
6684
6685/*
6686 * call-seq:
6687 * clear -> self
6688 *
6689 * Removes the contents of +self+:
6690 *
6691 * s = 'foo'
6692 * s.clear # => ""
6693 * s # => ""
6694 *
6695 * Related: see {Modifying}[rdoc-ref:String@Modifying].
6696 */
6697
6698static VALUE
6699rb_str_clear(VALUE str)
6700{
6701 str_discard(str);
6702 STR_SET_EMBED(str);
6703 STR_SET_LEN(str, 0);
6704 RSTRING_PTR(str)[0] = 0;
6705 if (rb_enc_asciicompat(STR_ENC_GET(str)))
6707 else
6709 return str;
6710}
6711
6712/*
6713 * call-seq:
6714 * chr -> string
6715 *
6716 * :include: doc/string/chr.rdoc
6717 *
6718 */
6719
6720static VALUE
6721rb_str_chr(VALUE str)
6722{
6723 return rb_str_substr(str, 0, 1);
6724}
6725
6726/*
6727 * call-seq:
6728 * getbyte(index) -> integer or nil
6729 *
6730 * :include: doc/string/getbyte.rdoc
6731 *
6732 */
6733VALUE
6734rb_str_getbyte(VALUE str, VALUE index)
6735{
6736 long pos = NUM2LONG(index);
6737
6738 if (pos < 0)
6739 pos += RSTRING_LEN(str);
6740 if (pos < 0 || RSTRING_LEN(str) <= pos)
6741 return Qnil;
6742
6743 return INT2FIX((unsigned char)RSTRING_PTR(str)[pos]);
6744}
6745
6746/*
6747 * call-seq:
6748 * setbyte(index, integer) -> integer
6749 *
6750 * Sets the byte at zero-based offset +index+ to the value of the given +integer+;
6751 * returns +integer+:
6752 *
6753 * s = 'xyzzy'
6754 * s.setbyte(2, 129) # => 129
6755 * s # => "xy\x81zy"
6756 *
6757 * Related: see {Modifying}[rdoc-ref:String@Modifying].
6758 */
6759VALUE
6760rb_str_setbyte(VALUE str, VALUE index, VALUE value)
6761{
6762 long pos = NUM2LONG(index);
6763 char *ptr, *head, *left = 0;
6764 rb_encoding *enc;
6765 int cr = ENC_CODERANGE_UNKNOWN, width, nlen;
6766
6767 VALUE v = rb_to_int(value);
6768 VALUE w = rb_int_and(v, INT2FIX(0xff));
6769 char byte = (char)(NUM2INT(w) & 0xFF);
6770
6771 long len = RSTRING_LEN(str);
6772 if (pos < -len || len <= pos)
6773 rb_raise(rb_eIndexError, "index %ld out of string", pos);
6774 if (pos < 0)
6775 pos += len;
6776
6777 if (!str_independent(str))
6778 str_make_independent(str);
6779 enc = STR_ENC_GET(str);
6780 head = RSTRING_PTR(str);
6781 ptr = &head[pos];
6782 if (!STR_EMBED_P(str)) {
6783 cr = ENC_CODERANGE(str);
6784 switch (cr) {
6785 case ENC_CODERANGE_7BIT:
6786 left = ptr;
6787 *ptr = byte;
6788 if (ISASCII(byte)) goto end;
6789 nlen = rb_enc_precise_mbclen(left, head+len, enc);
6790 if (!MBCLEN_CHARFOUND_P(nlen))
6792 else
6794 goto end;
6796 left = rb_enc_left_char_head(head, ptr, head+len, enc);
6797 width = rb_enc_precise_mbclen(left, head+len, enc);
6798 *ptr = byte;
6799 nlen = rb_enc_precise_mbclen(left, head+len, enc);
6800 if (!MBCLEN_CHARFOUND_P(nlen))
6802 else if (MBCLEN_CHARFOUND_LEN(nlen) != width || ISASCII(byte))
6804 goto end;
6805 }
6806 }
6808 *ptr = byte;
6809
6810 end:
6811 return value;
6812}
6813
6814static inline bool
6815str_bit_offset_out_of_range(long byte_len, uint64_t bit_offset)
6816{
6817 /* Compare byte indexes to avoid overflowing byte_len * CHAR_BIT. */
6818 return bit_offset / CHAR_BIT >= (uint64_t)byte_len;
6819}
6820
6821/*
6822 * Keep both the full bit offset and its long representation. Most calls use a
6823 * Fixnum-sized offset and can stay on the original long fast path; only large
6824 * Bignum offsets need the uint64_t path below. This matters on platforms
6825 * where long is narrower than the address space, such as 32-bit and LLP64.
6826 */
6828 uint64_t value;
6829 long long_value;
6830 bool fits_long;
6831};
6832
6833static inline struct str_bit_offset
6834str_bit_offset_from_index(VALUE index)
6835{
6836 VALUE integer = rb_to_int(index);
6837 struct str_bit_offset offset;
6838
6839 /*
6840 * FIXNUM_P only decides whether the common long path is immediately usable.
6841 * This covers practically all offsets on LP64 platforms; Bignum offsets
6842 * are still accepted below when they fit in uint64_t, mainly for platforms
6843 * with 32-bit long where large strings can have Bignum bit offsets.
6844 */
6845 if (FIXNUM_P(integer)) {
6846 offset.long_value = FIX2LONG(integer);
6847 if (offset.long_value < 0) {
6848 rb_raise(rb_eIndexError, "bit index out of range");
6849 }
6850 offset.value = (uint64_t)offset.long_value;
6851 offset.fits_long = true;
6852 return offset;
6853 }
6854
6855 RUBY_ASSERT(RB_TYPE_P(integer, T_BIGNUM));
6856 if (rb_int_negative_p(integer)) {
6857 rb_raise(rb_eIndexError, "bit index out of range");
6858 }
6859 if (rb_cmpint(rb_int_cmp(integer, ULL2NUM(UINT64_MAX)), integer, ULL2NUM(UINT64_MAX)) > 0) {
6860 rb_raise(rb_eArgError, "bit index out of representable range");
6861 }
6862
6863 offset.value = (uint64_t)NUM2ULL(integer);
6864 if (offset.value <= (uint64_t)LONG_MAX) {
6865 offset.long_value = (long)offset.value;
6866 offset.fits_long = true;
6867 }
6868 else {
6869 offset.long_value = 0;
6870 offset.fits_long = false;
6871 }
6872 return offset;
6873}
6874
6875/*
6876 * Bit lengths share the offset's representable range.
6877 * A negative length is an ArgumentError rather than an IndexError.
6878 */
6879static uint64_t
6880str_bit_length_from_index(VALUE index)
6881{
6882 VALUE integer = rb_to_int(index);
6883
6884 if (FIXNUM_P(integer)) {
6885 long value = FIX2LONG(integer);
6886 if (value < 0) {
6887 rb_raise(rb_eArgError, "negative bit length");
6888 }
6889 return (uint64_t)value;
6890 }
6891
6892 RUBY_ASSERT(RB_TYPE_P(integer, T_BIGNUM));
6893 if (rb_int_negative_p(integer)) {
6894 rb_raise(rb_eArgError, "negative bit length");
6895 }
6896 if (rb_cmpint(rb_int_cmp(integer, ULL2NUM(UINT64_MAX)), integer, ULL2NUM(UINT64_MAX)) > 0) {
6897 rb_raise(rb_eArgError, "bit length out of representable range");
6898 }
6899 return (uint64_t)NUM2ULL(integer);
6900}
6901
6902static inline uint64_t
6903str_bit_size(long byte_len)
6904{
6905 /*
6906 * byte_len * CHAR_BIT overflows uint64_t only for byte_len >= 2**61 which cannot
6907 * be allocated. Saturate so that unreachable cases cannot wrap.
6908 */
6909 if ((uint64_t)byte_len > UINT64_MAX / CHAR_BIT) return UINT64_MAX;
6910 return (uint64_t)byte_len * CHAR_BIT;
6911}
6912
6914 uint64_t beg;
6915 uint64_t end_exclusive; /* meaningful only when end_open is false */
6916 bool end_open; /* a nil end: the region runs to the end of self */
6917};
6918
6919/*
6920 * Coerce a bit Range's endpoints to bit offsets. This may run arbitrary Ruby
6921 * (Integer#to_int on the endpoints), so it does NOT read the string's length:
6922 * The caller must resolve the length only after this returns, otherwise
6923 * to_int that reallocates self would leave a stale size.
6924 */
6925static void
6926str_bit_range_to_offsets(VALUE range, struct str_bit_range *out)
6927{
6928 VALUE beg_v, end_v;
6929 int excl;
6930
6931 /*
6932 * We don't use rb_range_beg_len: it counts negative endpoints from the end,
6933 * which is an IndexError for bit positions, and it is limited to long instead
6934 * of uint64_t.
6935 */
6936 rb_range_values(range, &beg_v, &end_v, &excl);
6937
6938 out->beg = NIL_P(beg_v) ? 0 : str_bit_offset_from_index(beg_v).value;
6939 if (NIL_P(end_v)) {
6940 out->end_open = true;
6941 out->end_exclusive = 0;
6942 }
6943 else {
6944 uint64_t end = str_bit_offset_from_index(end_v).value;
6945 out->end_open = false;
6946 /*
6947 * The saturation loses one position only for an inclusive end of
6948 * 2**64-1, which lies beyond any real string either way.
6949 */
6950 out->end_exclusive = (excl || end == UINT64_MAX) ? end : end + 1;
6951 }
6952}
6953
6954/*
6955 * Turn a coerced Range into (beg, len) against the now-current total bit size.
6956 * The length is deliberately not clamped to the bits available, so a reading
6957 * caller can clamp while a writing caller detects the overrun and raises.
6958 */
6959static bool
6960str_bit_range_resolve(const struct str_bit_range *range, uint64_t total_bits, uint64_t *begp, uint64_t *lenp)
6961{
6962 uint64_t beg = range->beg;
6963 if (beg > total_bits) return false;
6964
6965 uint64_t end_exclusive = range->end_open ? total_bits : range->end_exclusive;
6966 if (end_exclusive < beg) end_exclusive = beg;
6967
6968 *begp = beg;
6969 *lenp = end_exclusive - beg;
6970 return true;
6971}
6972
6973static bool
6974str_lsb_first_from_opts(VALUE opts)
6975{
6976 static ID keywords[1];
6977 VALUE vlsb_first;
6978
6979 if (!keywords[0]) {
6980 keywords[0] = rb_intern_const("lsb_first");
6981 }
6982
6983 rb_get_kwargs(opts, keywords, 0, 1, &vlsb_first);
6984 if (vlsb_first == Qundef || vlsb_first == Qtrue) {
6985 return true;
6986 }
6987 if (vlsb_first == Qfalse) {
6988 return false;
6989 }
6990 rb_raise(rb_eArgError, "lsb_first must be true or false");
6991 UNREACHABLE_RETURN(false);
6992}
6993
6994static bool
6995str_lsb_first(int argc, VALUE *argv, VALUE *index)
6996{
6997 VALUE opts;
6998
6999 rb_scan_args(argc, argv, "1:", index, &opts);
7000 return str_lsb_first_from_opts(opts);
7001}
7002
7003static inline uint64_t
7004str_logical_to_physical_bit64(uint64_t logical, bool lsb_first)
7005{
7006 return lsb_first ? logical : ((logical & ~(uint64_t)7) | (7 - (logical & 7)));
7007}
7008
7009static inline long
7010str_logical_to_physical_bit(long logical, bool lsb_first)
7011{
7012 return lsb_first ? logical : ((logical & ~7L) | (7 - (logical & 7L)));
7013}
7014
7016 long byte_index;
7017 unsigned int bit_offset;
7018};
7019
7020static inline struct str_bit_location
7021str_bit_location_from_offset(uint64_t logical, bool lsb_first)
7022{
7023 /*
7024 * When long is 32-bit, a bit offset for a large string can be a Bignum
7025 * while the byte index still fits in long, which is RSTRING_LEN's type.
7026 */
7027 uint64_t physical = str_logical_to_physical_bit64(logical, lsb_first);
7028 struct str_bit_location location;
7029 location.byte_index = (long)(physical / CHAR_BIT);
7030 location.bit_offset = (unsigned int)(physical % CHAR_BIT);
7031 return location;
7032}
7033
7034static inline int
7035str_get_bit(const char *ptr, long bit_index)
7036{
7037 return (((unsigned char)ptr[bit_index / CHAR_BIT]) >> (bit_index % CHAR_BIT)) & 1;
7038}
7039
7040static inline int
7041str_get_bit_location(const char *ptr, struct str_bit_location location)
7042{
7043 return (((unsigned char)ptr[location.byte_index]) >> location.bit_offset) & 1;
7044}
7045
7046static int
7047str_bit_get(int argc, VALUE *argv, VALUE str)
7048{
7049 VALUE index;
7050 bool lsb_first = str_lsb_first(argc, argv, &index);
7051 struct str_bit_offset offset = str_bit_offset_from_index(index);
7052
7053 if (str_bit_offset_out_of_range(RSTRING_LEN(str), offset.value)) {
7054 return -1;
7055 }
7056
7057 if (offset.fits_long) {
7058 return str_get_bit(RSTRING_PTR(str), str_logical_to_physical_bit(offset.long_value, lsb_first));
7059 }
7060 else {
7061 return str_get_bit_location(RSTRING_PTR(str), str_bit_location_from_offset(offset.value, lsb_first));
7062 }
7063}
7064
7065/*
7066 * call-seq:
7067 * bit_get(offset, lsb_first: true) -> 0, 1, or nil
7068 *
7069 * :include: doc/string/bit_get.rdoc
7070 *
7071 */
7072static VALUE
7073rb_str_bit_get(int argc, VALUE *argv, VALUE str)
7074{
7075 int bit = str_bit_get(argc, argv, str);
7076 return bit < 0 ? Qnil : INT2FIX(bit);
7077}
7078
7079/*
7080 * call-seq:
7081 * bit_set?(offset, lsb_first: true) -> true, false, or nil
7082 *
7083 * :include: doc/string/bit_set_p.rdoc
7084 *
7085 */
7086static VALUE
7087rb_str_bit_set_p(int argc, VALUE *argv, VALUE str)
7088{
7089 int bit = str_bit_get(argc, argv, str);
7090 return bit < 0 ? Qnil : RBOOL(bit);
7091}
7092
7093enum str_bit_mutation {
7094 STR_BIT_SET,
7095 STR_BIT_CLEAR,
7096 STR_BIT_FLIP
7097};
7098
7099/*
7100 * Mask for the logical in-byte positions lo..hi (0 <= lo <= hi <= 7) of one
7101 * byte. A contiguous logical run stays contiguous within a byte under both
7102 * numbering conventions; MSB-first only mirrors it.
7103 */
7104static inline unsigned char
7105str_bit_region_byte_mask(unsigned int lo, unsigned int hi, bool lsb_first)
7106{
7107 if (lsb_first) {
7108 return (unsigned char)((0xFFu >> (7 - hi)) & (0xFFu << lo));
7109 }
7110 else {
7111 return (unsigned char)((0xFFu >> lo) & (0xFFu << (7 - hi)));
7112 }
7113}
7114
7115static inline void
7116str_apply_bit_mask(unsigned char *byte, unsigned char mask, enum str_bit_mutation mutation)
7117{
7118 switch (mutation) {
7119 case STR_BIT_SET:
7120 *byte |= mask;
7121 break;
7122 case STR_BIT_CLEAR:
7123 *byte &= (unsigned char)~mask;
7124 break;
7125 case STR_BIT_FLIP:
7126 *byte ^= mask;
7127 break;
7128 }
7129}
7130
7131/* The caller has bounds-checked [beg, beg+len) and called rb_str_modify. */
7132static void
7133str_mutate_bit_region(unsigned char *ptr, uint64_t beg, uint64_t len, bool lsb_first, enum str_bit_mutation mutation)
7134{
7135 uint64_t first_bit = beg;
7136 uint64_t last_bit = beg + len - 1;
7137 long first_byte = (long)(first_bit / CHAR_BIT);
7138 long last_byte = (long)(last_bit / CHAR_BIT);
7139 unsigned int first_off = (unsigned int)(first_bit % CHAR_BIT);
7140 unsigned int last_off = (unsigned int)(last_bit % CHAR_BIT);
7141
7142 if (first_byte == last_byte) {
7143 str_apply_bit_mask(ptr + first_byte, str_bit_region_byte_mask(first_off, last_off, lsb_first), mutation);
7144 return;
7145 }
7146
7147 str_apply_bit_mask(ptr + first_byte, str_bit_region_byte_mask(first_off, 7, lsb_first), mutation);
7148 long middle_len = last_byte - first_byte - 1;
7149 if (middle_len > 0) {
7150 unsigned char *middle = ptr + first_byte + 1;
7151 switch (mutation) {
7152 case STR_BIT_SET:
7153 memset(middle, 0xFF, middle_len);
7154 break;
7155 case STR_BIT_CLEAR:
7156 memset(middle, 0, middle_len);
7157 break;
7158 case STR_BIT_FLIP:
7159 /*
7160 * Byte loop on purpose: the compiler auto-vectorizes it (verified on gcc 13.3
7161 * and clang 18.1 with x86_64), and being read-modify-write, the flip is memory-bound,
7162 * so a manual word-at-a-time XOR loop was measured to be no faster.
7163 */
7164 for (long i = 0; i < middle_len; i++) {
7165 middle[i] ^= 0xFF;
7166 }
7167 break;
7168 }
7169 }
7170 str_apply_bit_mask(ptr + last_byte, str_bit_region_byte_mask(0, last_off, lsb_first), mutation);
7171}
7172
7173static VALUE
7174str_mutate_single_bit(VALUE str, VALUE index, bool lsb_first, enum str_bit_mutation mutation)
7175{
7176 struct str_bit_offset offset = str_bit_offset_from_index(index);
7177 struct str_bit_location location;
7178 long bit_index;
7179 unsigned char *ptr;
7180 unsigned char mask;
7181
7182 rb_check_frozen(str);
7183
7184 if (str_bit_offset_out_of_range(RSTRING_LEN(str), offset.value)) {
7185 rb_raise(rb_eIndexError, "bit index out of range");
7186 }
7187
7188 rb_str_modify(str);
7189 ptr = (unsigned char *)RSTRING_PTR(str);
7190 if (offset.fits_long) {
7191 bit_index = str_logical_to_physical_bit(offset.long_value, lsb_first);
7192 mask = (unsigned char)(1u << (bit_index % CHAR_BIT));
7193 location.byte_index = bit_index / CHAR_BIT;
7194 }
7195 else {
7196 location = str_bit_location_from_offset(offset.value, lsb_first);
7197 mask = (unsigned char)(1u << location.bit_offset);
7198 }
7199
7200 str_apply_bit_mask(ptr + location.byte_index, mask, mutation);
7201 return str;
7202}
7203
7204static VALUE
7205str_mutate_bit(int argc, VALUE *argv, VALUE str, enum str_bit_mutation mutation)
7206{
7207 VALUE target, length_v, opts;
7208 uint64_t beg = 0, len = 0;
7209
7210 /* Count positional arguments so that an explicit nil is not mistaken for an omitted one. */
7211 int nargs = rb_scan_args(argc, argv, "11:", &target, &length_v, &opts);
7212 bool lsb_first = str_lsb_first_from_opts(opts);
7213
7214 bool is_range = rb_obj_is_kind_of(target, rb_cRange);
7215 if (nargs == 1 && !is_range) {
7216 return str_mutate_single_bit(str, target, lsb_first, mutation);
7217 }
7218
7219 struct str_bit_range range = {0};
7220 struct str_bit_offset offset;
7221 if (is_range) {
7222 if (nargs == 2) {
7223 rb_raise(rb_eArgError, "bit length not allowed with a Range");
7224 }
7225 str_bit_range_to_offsets(target, &range);
7226 }
7227 else {
7228 offset = str_bit_offset_from_index(target);
7229 len = str_bit_length_from_index(length_v);
7230 }
7231
7232 /* Even a zero-length write requires a mutable receiver. */
7233 rb_check_frozen(str);
7234
7235 /*
7236 * A region that begins past the end is out of range even when it is
7237 * empty, and one that runs past the end is not allowed to silently
7238 * shrink: both are errors for a mutation, unlike the clamping reads.
7239 * An empty region whose start is within 0..bitsize writes nothing.
7240 */
7241 uint64_t total_bits = str_bit_size(RSTRING_LEN(str));
7242 if (is_range) {
7243 if (!str_bit_range_resolve(&range, total_bits, &beg, &len) || len > total_bits - beg) {
7244 rb_raise(rb_eIndexError, "bit range out of range");
7245 }
7246 }
7247 else {
7248 beg = offset.value;
7249 if (beg > total_bits || len > total_bits - beg) {
7250 rb_raise(rb_eIndexError, "bit range out of range");
7251 }
7252 }
7253
7254 if (len == 0) return str;
7255
7256 rb_str_modify(str);
7257 str_mutate_bit_region((unsigned char *)RSTRING_PTR(str), beg, len, lsb_first, mutation);
7258 return str;
7259}
7260
7261/*
7262 * call-seq:
7263 * bit_set(offset, lsb_first: true) -> self
7264 * bit_set(offset, length, lsb_first: true) -> self
7265 * bit_set(range, lsb_first: true) -> self
7266 *
7267 * :include: doc/string/bit_set.rdoc
7268 *
7269 */
7270static VALUE
7271rb_str_bit_set(int argc, VALUE *argv, VALUE str)
7272{
7273 return str_mutate_bit(argc, argv, str, STR_BIT_SET);
7274}
7275
7276/*
7277 * call-seq:
7278 * bit_clear(offset, lsb_first: true) -> self
7279 * bit_clear(offset, length, lsb_first: true) -> self
7280 * bit_clear(range, lsb_first: true) -> self
7281 *
7282 * :include: doc/string/bit_clear.rdoc
7283 *
7284 */
7285static VALUE
7286rb_str_bit_clear(int argc, VALUE *argv, VALUE str)
7287{
7288 return str_mutate_bit(argc, argv, str, STR_BIT_CLEAR);
7289}
7290
7291/*
7292 * call-seq:
7293 * bit_flip(offset, lsb_first: true) -> self
7294 * bit_flip(offset, length, lsb_first: true) -> self
7295 * bit_flip(range, lsb_first: true) -> self
7296 *
7297 * :include: doc/string/bit_flip.rdoc
7298 *
7299 */
7300static VALUE
7301rb_str_bit_flip(int argc, VALUE *argv, VALUE str)
7302{
7303 return str_mutate_bit(argc, argv, str, STR_BIT_FLIP);
7304}
7305
7306static uint64_t
7307str_count_bits(const unsigned char *ptr, long len)
7308{
7309 uint64_t count = 0;
7310 long off = 0;
7311 long unrolled_end = len & ~31L;
7312 long aligned_end = len & ~7L;
7313
7314 // 32 bytes (256 bits) at a time
7315 for (; off < unrolled_end; off += 32) {
7316 uint64_t w0, w1, w2, w3;
7317 memcpy(&w0, ptr + off, 8);
7318 memcpy(&w1, ptr + off + 8, 8);
7319 memcpy(&w2, ptr + off + 16, 8);
7320 memcpy(&w3, ptr + off + 24, 8);
7321 count += rb_popcount64(w0);
7322 count += rb_popcount64(w1);
7323 count += rb_popcount64(w2);
7324 count += rb_popcount64(w3);
7325 }
7326
7327 // 8 bytes (64 bits) at a time
7328 for (; off < aligned_end; off += 8) {
7329 uint64_t word;
7330 memcpy(&word, ptr + off, 8);
7331 count += rb_popcount64(word);
7332 }
7333
7334 // remaining bytes
7335 if (off < len) {
7336 uint64_t word = 0;
7337 int shift = 0;
7338 for (; off < len; off++, shift += CHAR_BIT) {
7339 word |= (uint64_t)ptr[off] << shift;
7340 }
7341 count += rb_popcount64(word);
7342 }
7343
7344 return count;
7345}
7346
7347static uint64_t
7348str_count_bits_region(const unsigned char *ptr, uint64_t beg, uint64_t len, bool lsb_first)
7349{
7350 uint64_t first_bit = beg;
7351 uint64_t last_bit = beg + len - 1;
7352 long first_byte = (long)(first_bit / CHAR_BIT);
7353 long last_byte = (long)(last_bit / CHAR_BIT);
7354 unsigned int first_off = (unsigned int)(first_bit % CHAR_BIT);
7355 unsigned int last_off = (unsigned int)(last_bit % CHAR_BIT);
7356
7357 if (first_byte == last_byte) {
7358 return rb_popcount32((uint32_t)(ptr[first_byte] & str_bit_region_byte_mask(first_off, last_off, lsb_first)));
7359 }
7360
7361 uint64_t count = rb_popcount32((uint32_t)(ptr[first_byte] & str_bit_region_byte_mask(first_off, 7, lsb_first)));
7362 count += str_count_bits(ptr + first_byte + 1, last_byte - first_byte - 1);
7363 count += rb_popcount32((uint32_t)(ptr[last_byte] & str_bit_region_byte_mask(0, last_off, lsb_first)));
7364 return count;
7365}
7366
7367/*
7368 * call-seq:
7369 * bit_count -> integer
7370 * bit_count(offset, length, lsb_first: true) -> integer
7371 * bit_count(range, lsb_first: true) -> integer
7372 *
7373 * :include: doc/string/bit_count.rdoc
7374 *
7375 */
7376static VALUE
7377rb_str_bit_count(int argc, VALUE *argv, VALUE str)
7378{
7379 VALUE v0, v1, opts;
7380 uint64_t beg = 0, len = 0;
7381
7382 /* Count positional arguments so that an explicit nil is not mistaken for an omitted one. */
7383 int nargs = rb_scan_args(argc, argv, "02:", &v0, &v1, &opts);
7384 /*
7385 * A whole-string popcount is independent of bit numbering.
7386 * no-(offset|range)-argument form only validates lsb_first.
7387 */
7388 bool lsb_first = str_lsb_first_from_opts(opts);
7389
7390 if (nargs == 0) {
7391 return ULL2NUM(str_count_bits((const unsigned char *)RSTRING_PTR(str), RSTRING_LEN(str)));
7392 }
7393
7394 bool is_range = rb_obj_is_kind_of(v0, rb_cRange);
7395 struct str_bit_range range = {0};
7396 if (is_range) {
7397 if (nargs == 2) {
7398 rb_raise(rb_eArgError, "bit length not allowed with a Range");
7399 }
7400 str_bit_range_to_offsets(v0, &range);
7401 }
7402 else if (nargs == 1) {
7403 rb_raise(rb_eArgError, "no bit length given");
7404 }
7405 else {
7406 beg = str_bit_offset_from_index(v0).value;
7407 len = str_bit_length_from_index(v1);
7408 }
7409
7410 const unsigned char *ptr = (const unsigned char *)RSTRING_PTR(str);
7411 uint64_t total_bits = str_bit_size(RSTRING_LEN(str));
7412 if (is_range) {
7413 if (!str_bit_range_resolve(&range, total_bits, &beg, &len)) {
7414 return INT2FIX(0);
7415 }
7416 }
7417 else if (beg >= total_bits) {
7418 return INT2FIX(0);
7419 }
7420
7421 /* Reads clamp: only the part of the region that exists is counted. */
7422 if (len > total_bits - beg) len = total_bits - beg;
7423 if (len == 0) return INT2FIX(0);
7424 return ULL2NUM(str_count_bits_region(ptr, beg, len, lsb_first));
7425}
7426
7427static void
7428str_check_bitwise_length(VALUE str, VALUE other)
7429{
7430 if (RSTRING_LEN(str) != RSTRING_LEN(other)) {
7431 rb_raise(rb_eArgError, "operands must have the same length (%ld vs %ld)",
7432 RSTRING_LEN(str), RSTRING_LEN(other));
7433 }
7434}
7435
7436static VALUE
7437str_bitwise_result(VALUE str)
7438{
7439 long len = RSTRING_LEN(str);
7440 VALUE result = rb_str_buf_new(len);
7441 rb_str_resize(result, len);
7442 rb_enc_associate(result, rb_ascii8bit_encoding());
7443 ENC_CODERANGE_CLEAR(result);
7444 return result;
7445}
7446
7447#define STR_DEFINE_UNARY_BITWISE_KERNEL(name, expr_word, expr_byte) \
7448 static void \
7449 name(unsigned char *dst, const unsigned char *src, long len) \
7450 { \
7451 long off = 0; \
7452 long unrolled_end = len & ~31L; \
7453 long aligned_end = len & ~7L; \
7454 for (; off < unrolled_end; off += 32) { \
7455 uint64_t s0, s1, s2, s3; \
7456 memcpy(&s0, src + off, 8); \
7457 memcpy(&s1, src + off + 8, 8); \
7458 memcpy(&s2, src + off + 16, 8); \
7459 memcpy(&s3, src + off + 24, 8); \
7460 s0 = (expr_word(s0)); \
7461 s1 = (expr_word(s1)); \
7462 s2 = (expr_word(s2)); \
7463 s3 = (expr_word(s3)); \
7464 memcpy(dst + off, &s0, 8); \
7465 memcpy(dst + off + 8, &s1, 8); \
7466 memcpy(dst + off + 16, &s2, 8); \
7467 memcpy(dst + off + 24, &s3, 8); \
7468 } \
7469 for (; off < aligned_end; off += 8) { \
7470 uint64_t word; \
7471 memcpy(&word, src + off, 8); \
7472 word = (expr_word(word)); \
7473 memcpy(dst + off, &word, 8); \
7474 } \
7475 for (; off < len; off++) dst[off] = (expr_byte(src[off])); \
7476 }
7477
7478#define STR_DEFINE_BINARY_BITWISE_KERNEL(name, expr_word, expr_byte) \
7479 static void \
7480 name(unsigned char *dst, const unsigned char *lhs, \
7481 const unsigned char *rhs, long len) \
7482 { \
7483 long off = 0; \
7484 long unrolled_end = len & ~31L; \
7485 long aligned_end = len & ~7L; \
7486 for (; off < unrolled_end; off += 32) { \
7487 uint64_t l0, l1, l2, l3, r0, r1, r2, r3; \
7488 memcpy(&l0, lhs + off, 8); memcpy(&r0, rhs + off, 8); \
7489 memcpy(&l1, lhs + off + 8, 8); memcpy(&r1, rhs + off + 8, 8); \
7490 memcpy(&l2, lhs + off + 16, 8); memcpy(&r2, rhs + off + 16, 8); \
7491 memcpy(&l3, lhs + off + 24, 8); memcpy(&r3, rhs + off + 24, 8); \
7492 l0 = expr_word(l0, r0); \
7493 l1 = expr_word(l1, r1); \
7494 l2 = expr_word(l2, r2); \
7495 l3 = expr_word(l3, r3); \
7496 memcpy(dst + off, &l0, 8); \
7497 memcpy(dst + off + 8, &l1, 8); \
7498 memcpy(dst + off + 16, &l2, 8); \
7499 memcpy(dst + off + 24, &l3, 8); \
7500 } \
7501 for (; off < aligned_end; off += 8) { \
7502 uint64_t lhs_word, rhs_word; \
7503 memcpy(&lhs_word, lhs + off, 8); \
7504 memcpy(&rhs_word, rhs + off, 8); \
7505 lhs_word = expr_word(lhs_word, rhs_word); \
7506 memcpy(dst + off, &lhs_word, 8); \
7507 } \
7508 for (; off < len; off++) dst[off] = expr_byte(lhs[off], rhs[off]); \
7509 }
7510
7511#define STR_BITWISE_NOT_WORD(x) (~(x))
7512#define STR_BITWISE_NOT_BYTE(x) ((unsigned char)~(x))
7513#define STR_BITWISE_AND_WORD(x, y) ((x) & (y))
7514#define STR_BITWISE_AND_BYTE(x, y) ((unsigned char)((x) & (y)))
7515#define STR_BITWISE_OR_WORD(x, y) ((x) | (y))
7516#define STR_BITWISE_OR_BYTE(x, y) ((unsigned char)((x) | (y)))
7517#define STR_BITWISE_XOR_WORD(x, y) ((x) ^ (y))
7518#define STR_BITWISE_XOR_BYTE(x, y) ((unsigned char)((x) ^ (y)))
7519
7520STR_DEFINE_UNARY_BITWISE_KERNEL(str_bitwise_not, STR_BITWISE_NOT_WORD, STR_BITWISE_NOT_BYTE)
7521STR_DEFINE_BINARY_BITWISE_KERNEL(str_bitwise_and, STR_BITWISE_AND_WORD, STR_BITWISE_AND_BYTE)
7522STR_DEFINE_BINARY_BITWISE_KERNEL(str_bitwise_or, STR_BITWISE_OR_WORD, STR_BITWISE_OR_BYTE)
7523STR_DEFINE_BINARY_BITWISE_KERNEL(str_bitwise_xor, STR_BITWISE_XOR_WORD, STR_BITWISE_XOR_BYTE)
7524
7525/*
7526 * call-seq:
7527 * bitwise_not -> string
7528 *
7529 * :include: doc/string/bitwise_not.rdoc
7530 *
7531 */
7532static VALUE
7533rb_str_bitwise_not(VALUE str)
7534{
7535 long len = RSTRING_LEN(str);
7536 VALUE result = str_bitwise_result(str);
7537 str_bitwise_not((unsigned char *)RSTRING_PTR(result),
7538 (const unsigned char *)RSTRING_PTR(str), len);
7539 return result;
7540}
7541
7542/*
7543 * call-seq:
7544 * bitwise_not! -> self
7545 *
7546 * :include: doc/string/bitwise_not_bang.rdoc
7547 *
7548 */
7549static VALUE
7550rb_str_bitwise_not_bang(VALUE str)
7551{
7552 long len;
7553 unsigned char *ptr;
7554
7555 rb_str_modify(str);
7556 len = RSTRING_LEN(str);
7557 ptr = (unsigned char *)RSTRING_PTR(str);
7558 str_bitwise_not(ptr, ptr, len);
7559 return str;
7560}
7561
7562#define STR_DEFINE_BINARY_BITWISE_METHOD(name) \
7563 static VALUE \
7564 rb_str_bitwise_##name(VALUE str, VALUE other) \
7565 { \
7566 long len; \
7567 VALUE result; \
7568 StringValue(other); \
7569 str_check_bitwise_length(str, other); \
7570 len = RSTRING_LEN(str); \
7571 result = str_bitwise_result(str); \
7572 str_bitwise_##name((unsigned char *)RSTRING_PTR(result), \
7573 (const unsigned char *)RSTRING_PTR(str), \
7574 (const unsigned char *)RSTRING_PTR(other), len); \
7575 return result; \
7576 } \
7577 static VALUE \
7578 rb_str_bitwise_##name##_bang(VALUE str, VALUE other) \
7579 { \
7580 long len; \
7581 unsigned char *ptr; \
7582 StringValue(other); \
7583 str_check_bitwise_length(str, other); \
7584 rb_str_modify(str); \
7585 len = RSTRING_LEN(str); \
7586 ptr = (unsigned char *)RSTRING_PTR(str); \
7587 str_bitwise_##name(ptr, ptr, \
7588 (const unsigned char *)RSTRING_PTR(other), len); \
7589 return str; \
7590 }
7591
7592STR_DEFINE_BINARY_BITWISE_METHOD(and)
7593STR_DEFINE_BINARY_BITWISE_METHOD(or)
7594STR_DEFINE_BINARY_BITWISE_METHOD(xor)
7595
7596static VALUE
7597str_byte_substr(VALUE str, long beg, long len, int empty)
7598{
7599 long n = RSTRING_LEN(str);
7600
7601 if (beg > n || len < 0) return Qnil;
7602 if (beg < 0) {
7603 beg += n;
7604 if (beg < 0) return Qnil;
7605 }
7606 if (len > n - beg)
7607 len = n - beg;
7608 if (len <= 0) {
7609 if (!empty) return Qnil;
7610 len = 0;
7611 }
7612
7613 VALUE str2 = str_subseq(str, beg, len);
7614
7615 str_enc_copy_direct(str2, str);
7616
7617 if (RSTRING_LEN(str2) == 0) {
7618 if (!rb_enc_asciicompat(STR_ENC_GET(str)))
7620 else
7622 }
7623 else {
7624 switch (ENC_CODERANGE(str)) {
7625 case ENC_CODERANGE_7BIT:
7627 break;
7628 default:
7630 break;
7631 }
7632 }
7633
7634 return str2;
7635}
7636
7637VALUE
7638rb_str_byte_substr(VALUE str, VALUE beg, VALUE len)
7639{
7640 return str_byte_substr(str, NUM2LONG(beg), NUM2LONG(len), TRUE);
7641}
7642
7643static VALUE
7644str_byte_aref(VALUE str, VALUE indx)
7645{
7646 long idx;
7647 if (FIXNUM_P(indx)) {
7648 idx = FIX2LONG(indx);
7649 }
7650 else {
7651 /* check if indx is Range */
7652 long beg, len = RSTRING_LEN(str);
7653
7654 switch (rb_range_beg_len(indx, &beg, &len, len, 0)) {
7655 case Qfalse:
7656 break;
7657 case Qnil:
7658 return Qnil;
7659 default:
7660 return str_byte_substr(str, beg, len, TRUE);
7661 }
7662
7663 idx = NUM2LONG(indx);
7664 }
7665 return str_byte_substr(str, idx, 1, FALSE);
7666}
7667
7668/*
7669 * call-seq:
7670 * byteslice(offset, length = 1) -> string or nil
7671 * byteslice(range) -> string or nil
7672 *
7673 * :include: doc/string/byteslice.rdoc
7674 */
7675
7676static VALUE
7677rb_str_byteslice(int argc, VALUE *argv, VALUE str)
7678{
7679 if (argc == 2) {
7680 long beg = NUM2LONG(argv[0]);
7681 long len = NUM2LONG(argv[1]);
7682 return str_byte_substr(str, beg, len, TRUE);
7683 }
7684 rb_check_arity(argc, 1, 2);
7685 return str_byte_aref(str, argv[0]);
7686}
7687
7688static void
7689str_check_beg_len(VALUE str, long *beg, long *len)
7690{
7691 long end, slen = RSTRING_LEN(str);
7692
7693 if (*len < 0) rb_raise(rb_eIndexError, "negative length %ld", *len);
7694 if ((slen < *beg) || ((*beg < 0) && (*beg + slen < 0))) {
7695 rb_raise(rb_eIndexError, "index %ld out of string", *beg);
7696 }
7697 if (*beg < 0) {
7698 *beg += slen;
7699 }
7700 RUBY_ASSERT(*beg >= 0);
7701 RUBY_ASSERT(*beg <= slen);
7702
7703 if (*len > slen - *beg) {
7704 *len = slen - *beg;
7705 }
7706 end = *beg + *len;
7707 str_ensure_byte_pos(str, *beg);
7708 str_ensure_byte_pos(str, end);
7709}
7710
7711/*
7712 * call-seq:
7713 * bytesplice(offset, length, str) -> self
7714 * bytesplice(offset, length, str, str_offset, str_length) -> self
7715 * bytesplice(range, str) -> self
7716 * bytesplice(range, str, str_range) -> self
7717 *
7718 * :include: doc/string/bytesplice.rdoc
7719 */
7720
7721static VALUE
7722rb_str_bytesplice(int argc, VALUE *argv, VALUE str)
7723{
7724 long beg, len, vbeg, vlen;
7725 VALUE val;
7726 int cr;
7727
7728 rb_check_arity(argc, 2, 5);
7729 if (!(argc == 2 || argc == 3 || argc == 5)) {
7730 rb_raise(rb_eArgError, "wrong number of arguments (given %d, expected 2, 3, or 5)", argc);
7731 }
7732 if (argc == 2 || (argc == 3 && !RB_INTEGER_TYPE_P(argv[0]))) {
7733 if (!rb_range_beg_len(argv[0], &beg, &len, RSTRING_LEN(str), 2)) {
7734 rb_raise(rb_eTypeError, "wrong argument type %s (expected Range)",
7735 rb_builtin_class_name(argv[0]));
7736 }
7737 val = argv[1];
7738 StringValue(val);
7739 if (argc == 2) {
7740 /* bytesplice(range, str) */
7741 vbeg = 0;
7742 vlen = RSTRING_LEN(val);
7743 }
7744 else {
7745 /* bytesplice(range, str, str_range) */
7746 if (!rb_range_beg_len(argv[2], &vbeg, &vlen, RSTRING_LEN(val), 2)) {
7747 rb_raise(rb_eTypeError, "wrong argument type %s (expected Range)",
7748 rb_builtin_class_name(argv[2]));
7749 }
7750 }
7751 }
7752 else {
7753 beg = NUM2LONG(argv[0]);
7754 len = NUM2LONG(argv[1]);
7755 val = argv[2];
7756 StringValue(val);
7757 if (argc == 3) {
7758 /* bytesplice(index, length, str) */
7759 vbeg = 0;
7760 vlen = RSTRING_LEN(val);
7761 }
7762 else {
7763 /* bytesplice(index, length, str, str_index, str_length) */
7764 vbeg = NUM2LONG(argv[3]);
7765 vlen = NUM2LONG(argv[4]);
7766 }
7767 }
7768 str_check_beg_len(str, &beg, &len);
7769 str_check_beg_len(val, &vbeg, &vlen);
7770 str_modify_keep_cr(str);
7771
7772 if (RB_UNLIKELY(ENCODING_GET_INLINED(str) != ENCODING_GET_INLINED(val))) {
7773 rb_enc_associate(str, rb_enc_check(str, val));
7774 }
7775
7776 rb_str_update_1(str, beg, len, val, vbeg, vlen);
7778 if (cr != ENC_CODERANGE_BROKEN)
7779 ENC_CODERANGE_SET(str, cr);
7780 return str;
7781}
7782
7783/*
7784 * call-seq:
7785 * reverse -> new_string
7786 *
7787 * Returns a new string with the characters from +self+ in reverse order.
7788 *
7789 * 'drawer'.reverse # => "reward"
7790 * 'reviled'.reverse # => "deliver"
7791 * 'stressed'.reverse # => "desserts"
7792 * 'semordnilaps'.reverse # => "spalindromes"
7793 *
7794 * Related: see {Converting to New String}[rdoc-ref:String@Converting+to+New+String].
7795 */
7796
7797static VALUE
7798rb_str_reverse(VALUE str)
7799{
7800 rb_encoding *enc;
7801 VALUE rev;
7802 char *s, *e, *p;
7803 int cr;
7804
7805 if (RSTRING_LEN(str) <= 1) return str_duplicate(rb_cString, str);
7806 enc = STR_ENC_GET(str);
7807 rev = rb_str_new(0, RSTRING_LEN(str));
7808 s = RSTRING_PTR(str); e = RSTRING_END(str);
7809 p = RSTRING_END(rev);
7810 cr = ENC_CODERANGE(str);
7811
7812 if (RSTRING_LEN(str) > 1) {
7813 if (single_byte_optimizable(str)) {
7814 while (s < e) {
7815 *--p = *s++;
7816 }
7817 }
7818 else if (cr == ENC_CODERANGE_VALID) {
7819 while (s < e) {
7820 int clen = rb_enc_fast_mbclen(s, e, enc);
7821
7822 p -= clen;
7823 memcpy(p, s, clen);
7824 s += clen;
7825 }
7826 }
7827 else {
7828 cr = rb_enc_asciicompat(enc) ?
7830 while (s < e) {
7831 int clen = rb_enc_mbclen(s, e, enc);
7832
7833 if (clen > 1 || (*s & 0x80)) cr = ENC_CODERANGE_UNKNOWN;
7834 p -= clen;
7835 memcpy(p, s, clen);
7836 s += clen;
7837 }
7838 }
7839 }
7840 STR_SET_LEN(rev, RSTRING_LEN(str));
7841 str_enc_copy_direct(rev, str);
7842 ENC_CODERANGE_SET(rev, cr);
7843
7844 return rev;
7845}
7846
7847
7848/*
7849 * call-seq:
7850 * reverse! -> self
7851 *
7852 * Returns +self+ with its characters reversed:
7853 *
7854 * 'drawer'.reverse! # => "reward"
7855 * 'reviled'.reverse! # => "deliver"
7856 * 'stressed'.reverse! # => "desserts"
7857 * 'semordnilaps'.reverse! # => "spalindromes"
7858 *
7859 * Related: see {Modifying}[rdoc-ref:String@Modifying].
7860 */
7861
7862static VALUE
7863rb_str_reverse_bang(VALUE str)
7864{
7865 if (RSTRING_LEN(str) > 1) {
7866 if (single_byte_optimizable(str)) {
7867 char *s, *e, c;
7868
7869 str_modify_keep_cr(str);
7870 s = RSTRING_PTR(str);
7871 e = RSTRING_END(str) - 1;
7872 while (s < e) {
7873 c = *s;
7874 *s++ = *e;
7875 *e-- = c;
7876 }
7877 }
7878 else {
7879 str_shared_replace(str, rb_str_reverse(str));
7880 }
7881 }
7882 else {
7883 str_modify_keep_cr(str);
7884 }
7885 return str;
7886}
7887
7888
7889/*
7890 * call-seq:
7891 * include?(other_string) -> true or false
7892 *
7893 * Returns whether +self+ contains +other_string+:
7894 *
7895 * s = 'bar'
7896 * s.include?('ba') # => true
7897 * s.include?('ar') # => true
7898 * s.include?('bar') # => true
7899 * s.include?('a') # => true
7900 * s.include?('') # => true
7901 * s.include?('foo') # => false
7902 *
7903 * Related: see {Querying}[rdoc-ref:String@Querying].
7904 */
7905
7906VALUE
7907rb_str_include(VALUE str, VALUE arg)
7908{
7909 long i;
7910
7911 StringValue(arg);
7912 i = rb_str_index(str, arg, 0);
7913
7914 return RBOOL(i != -1);
7915}
7916
7917
7918/*
7919 * call-seq:
7920 * to_i(base = 10) -> integer
7921 *
7922 * Returns the result of interpreting leading characters in +self+
7923 * as an integer in the given +base+;
7924 * +base+ must be either +0+ or in range <tt>(2..36)</tt>:
7925 *
7926 * '123456'.to_i # => 123456
7927 * '123def'.to_i(16) # => 1195503
7928 *
7929 * With +base+ zero given, string +object+ may contain leading characters
7930 * to specify the actual base:
7931 *
7932 * '123def'.to_i(0) # => 123
7933 * '0123def'.to_i(0) # => 83
7934 * '0b123def'.to_i(0) # => 1
7935 * '0o123def'.to_i(0) # => 83
7936 * '0d123def'.to_i(0) # => 123
7937 * '0x123def'.to_i(0) # => 1195503
7938 *
7939 * Characters past a leading valid number (in the given +base+) are ignored:
7940 *
7941 * '12.345'.to_i # => 12
7942 * '12345'.to_i(2) # => 1
7943 *
7944 * Returns zero if there is no leading valid number:
7945 *
7946 * 'abcdef'.to_i # => 0
7947 * '2'.to_i(2) # => 0
7948 *
7949 * Related: see {Converting to Non-String}[rdoc-ref:String@Converting+to+Non-String].
7950 */
7951
7952static VALUE
7953rb_str_to_i(int argc, VALUE *argv, VALUE str)
7954{
7955 int base = 10;
7956
7957 if (rb_check_arity(argc, 0, 1) && (base = NUM2INT(argv[0])) < 0) {
7958 rb_raise(rb_eArgError, "invalid radix %d", base);
7959 }
7960 return rb_str_to_inum(str, base, FALSE);
7961}
7962
7963
7964/*
7965 * call-seq:
7966 * to_f -> float
7967 *
7968 * Returns the result of interpreting leading characters in +self+ as a Float:
7969 *
7970 * '3.14159'.to_f # => 3.14159
7971 * '1.234e-2'.to_f # => 0.01234
7972 *
7973 * Characters past a leading valid number are ignored:
7974 *
7975 * '3.14 (pi to two places)'.to_f # => 3.14
7976 *
7977 * Returns zero if there is no leading valid number:
7978 *
7979 * 'abcdef'.to_f # => 0.0
7980 *
7981 * See {Converting to Non-String}[rdoc-ref:String@Converting+to+Non-String].
7982 */
7983
7984static VALUE
7985rb_str_to_f(VALUE str)
7986{
7987 return DBL2NUM(rb_str_to_dbl(str, FALSE));
7988}
7989
7990
7991/*
7992 * call-seq:
7993 * to_s -> self or new_string
7994 *
7995 * Returns +self+ if +self+ is a +String+,
7996 * or +self+ converted to a +String+ if +self+ is a subclass of +String+.
7997 *
7998 * Related: see {Converting to New String}[rdoc-ref:String@Converting+to+New+String].
7999 */
8000
8001static VALUE
8002rb_str_to_s(VALUE str)
8003{
8004 if (rb_obj_class(str) != rb_cString) {
8005 return str_duplicate(rb_cString, str);
8006 }
8007 return str;
8008}
8009
8010#if 0
8011static void
8012str_cat_char(VALUE str, unsigned int c, rb_encoding *enc)
8013{
8014 char s[RUBY_MAX_CHAR_LEN];
8015 int n = rb_enc_codelen(c, enc);
8016
8017 rb_enc_mbcput(c, s, enc);
8018 rb_enc_str_buf_cat(str, s, n, enc);
8019}
8020#endif
8021
8022#define CHAR_ESC_LEN 13 /* sizeof(\x{ hex of 32bit unsigned int } \0) */
8023
8024int
8025rb_str_buf_cat_escaped_char(VALUE result, unsigned int c, int unicode_p)
8026{
8027 char buf[CHAR_ESC_LEN + 1];
8028 int l;
8029
8030#if SIZEOF_INT > 4
8031 c &= 0xffffffff;
8032#endif
8033 if (unicode_p) {
8034 if (c < 0x7F && ISPRINT(c)) {
8035 snprintf(buf, CHAR_ESC_LEN, "%c", c);
8036 }
8037 else if (c < 0x10000) {
8038 snprintf(buf, CHAR_ESC_LEN, "\\u%04X", c);
8039 }
8040 else {
8041 snprintf(buf, CHAR_ESC_LEN, "\\u{%X}", c);
8042 }
8043 }
8044 else {
8045 if (c < 0x100) {
8046 snprintf(buf, CHAR_ESC_LEN, "\\x%02X", c);
8047 }
8048 else {
8049 snprintf(buf, CHAR_ESC_LEN, "\\x{%X}", c);
8050 }
8051 }
8052 l = (int)strlen(buf); /* CHAR_ESC_LEN cannot exceed INT_MAX */
8053 rb_str_buf_cat(result, buf, l);
8054 return l;
8055}
8056
8057const char *
8058ruby_escaped_char(int c)
8059{
8060 switch (c) {
8061 case '\0': return "\\0";
8062 case '\n': return "\\n";
8063 case '\r': return "\\r";
8064 case '\t': return "\\t";
8065 case '\f': return "\\f";
8066 case '\013': return "\\v";
8067 case '\010': return "\\b";
8068 case '\007': return "\\a";
8069 case '\033': return "\\e";
8070 case '\x7f': return "\\c?";
8071 }
8072 return NULL;
8073}
8074
8075VALUE
8076rb_str_escape(VALUE str)
8077{
8078 int encidx = ENCODING_GET(str);
8079 rb_encoding *enc = rb_enc_from_index(encidx);
8080 const char *p = RSTRING_PTR(str);
8081 const char *pend = RSTRING_END(str);
8082 const char *prev = p;
8083 char buf[CHAR_ESC_LEN + 1];
8084 VALUE result = rb_str_buf_new(0);
8085 int unicode_p = rb_enc_unicode_p(enc);
8086 int asciicompat = rb_enc_asciicompat(enc);
8087
8088 while (p < pend) {
8089 unsigned int c;
8090 const char *cc;
8091 int n = rb_enc_precise_mbclen(p, pend, enc);
8092 if (!MBCLEN_CHARFOUND_P(n)) {
8093 if (p > prev) str_buf_cat(result, prev, p - prev);
8094 n = rb_enc_mbminlen(enc);
8095 if (pend < p + n)
8096 n = (int)(pend - p);
8097 while (n--) {
8098 snprintf(buf, CHAR_ESC_LEN, "\\x%02X", *p & 0377);
8099 str_buf_cat(result, buf, strlen(buf));
8100 prev = ++p;
8101 }
8102 continue;
8103 }
8104 n = MBCLEN_CHARFOUND_LEN(n);
8105 c = rb_enc_mbc_to_codepoint(p, pend, enc);
8106 p += n;
8107 cc = ruby_escaped_char(c);
8108 if (cc) {
8109 if (p - n > prev) str_buf_cat(result, prev, p - n - prev);
8110 str_buf_cat(result, cc, strlen(cc));
8111 prev = p;
8112 }
8113 else if (asciicompat && rb_enc_isascii(c, enc) && ISPRINT(c)) {
8114 }
8115 else {
8116 if (p - n > prev) str_buf_cat(result, prev, p - n - prev);
8117 rb_str_buf_cat_escaped_char(result, c, unicode_p);
8118 prev = p;
8119 }
8120 }
8121 if (p > prev) str_buf_cat(result, prev, p - prev);
8122 ENCODING_CODERANGE_SET(result, rb_usascii_encindex(), ENC_CODERANGE_7BIT);
8123
8124 return result;
8125}
8126
8127/* Lookup table for the inspect fast path. 1 marks bytes that need
8128 * no escaping. 0 marks bytes that need escape inspection: 0x00-0x1F
8129 * (control), 0x22 ("), 0x23 (#), 0x5C (\‍), 0x7F (DEL), 0x80-0xFF
8130 * (non-ASCII). */
8131static const bool inspect_no_escape[256] = {
8132 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, /* 0x00-0x0F */
8133 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, /* 0x10-0x1F */
8134 1, 1, 0, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, /* 0x20-0x2F */
8135 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, /* 0x30-0x3F */
8136 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, /* 0x40-0x4F */
8137 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 1, 1, 1, /* 0x50-0x5F */
8138 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, /* 0x60-0x6F */
8139 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, /* 0x70-0x7F */
8140};
8141
8142/*
8143 * call-seq:
8144 * inspect -> string
8145 *
8146 * :include: doc/string/inspect.rdoc
8147 *
8148 */
8149
8150VALUE
8152{
8153 int encidx = ENCODING_GET(str);
8154 rb_encoding *enc = rb_enc_from_index(encidx);
8155 const char *p, *pend, *prev;
8156 char buf[CHAR_ESC_LEN + 1];
8157 VALUE result = rb_str_buf_new(RSTRING_LEN(str) + 2); /* string content + surrounding quotes */
8158 rb_encoding *resenc = rb_default_internal_encoding();
8159 int unicode_p = rb_enc_unicode_p(enc);
8160 int asciicompat = rb_enc_asciicompat(enc);
8161 int cr = rb_enc_str_coderange(str);
8162
8163 if (resenc == NULL) resenc = rb_default_external_encoding();
8164 if (!rb_enc_asciicompat(resenc)) resenc = rb_usascii_encoding();
8165 rb_enc_associate(result, resenc);
8166 str_buf_cat2(result, "\"");
8167
8168 p = RSTRING_PTR(str); pend = RSTRING_END(str);
8169 prev = p;
8170 while (p < pend) {
8171 unsigned int c, cc;
8172 int n;
8173
8174 /* Fast path: bulk-skip runs of safe ASCII bytes via a lookup table.
8175 * Only well-formed strings (CR=7BIT for any encoding, or UTF-8 VALID)
8176 * are eligible. */
8177 if (cr == ENC_CODERANGE_7BIT ||
8178 (encidx == ENCINDEX_UTF_8 && cr == ENC_CODERANGE_VALID)) {
8179 while (p < pend && inspect_no_escape[(unsigned char)*p]) p++;
8180 if (p >= pend) break;
8181 }
8182
8183 n = rb_enc_precise_mbclen(p, pend, enc);
8184 if (!MBCLEN_CHARFOUND_P(n)) {
8185 if (p > prev) str_buf_cat(result, prev, p - prev);
8186 n = rb_enc_mbminlen(enc);
8187 if (pend < p + n)
8188 n = (int)(pend - p);
8189 while (n--) {
8190 snprintf(buf, CHAR_ESC_LEN, "\\x%02X", *p & 0377);
8191 str_buf_cat(result, buf, strlen(buf));
8192 prev = ++p;
8193 }
8194 continue;
8195 }
8196 n = MBCLEN_CHARFOUND_LEN(n);
8197 c = rb_enc_mbc_to_codepoint(p, pend, enc);
8198 p += n;
8199 if ((asciicompat || unicode_p) &&
8200 (c == '"'|| c == '\\' ||
8201 (c == '#' &&
8202 p < pend &&
8203 MBCLEN_CHARFOUND_P(rb_enc_precise_mbclen(p,pend,enc)) &&
8204 (cc = rb_enc_codepoint(p,pend,enc),
8205 (cc == '$' || cc == '@' || cc == '{'))))) {
8206 if (p - n > prev) str_buf_cat(result, prev, p - n - prev);
8207 str_buf_cat2(result, "\\");
8208 if (asciicompat || enc == resenc) {
8209 prev = p - n;
8210 continue;
8211 }
8212 }
8213 switch (c) {
8214 case '\n': cc = 'n'; break;
8215 case '\r': cc = 'r'; break;
8216 case '\t': cc = 't'; break;
8217 case '\f': cc = 'f'; break;
8218 case '\013': cc = 'v'; break;
8219 case '\010': cc = 'b'; break;
8220 case '\007': cc = 'a'; break;
8221 case 033: cc = 'e'; break;
8222 default: cc = 0; break;
8223 }
8224 if (cc) {
8225 if (p - n > prev) str_buf_cat(result, prev, p - n - prev);
8226 buf[0] = '\\';
8227 buf[1] = (char)cc;
8228 str_buf_cat(result, buf, 2);
8229 prev = p;
8230 continue;
8231 }
8232 /* The special casing of 0x85 (NEXT_LINE) here is because
8233 * Oniguruma historically treats it as printable, but it
8234 * doesn't match the print POSIX bracket class or character
8235 * property in regexps.
8236 *
8237 * See Ruby Bug #16842 for details:
8238 * https://bugs.ruby-lang.org/issues/16842
8239 */
8240 if ((enc == resenc && rb_enc_isprint(c, enc) && c != 0x85) ||
8241 (asciicompat && rb_enc_isascii(c, enc) && ISPRINT(c))) {
8242 continue;
8243 }
8244 else {
8245 if (p - n > prev) str_buf_cat(result, prev, p - n - prev);
8246 rb_str_buf_cat_escaped_char(result, c, unicode_p);
8247 prev = p;
8248 continue;
8249 }
8250 }
8251 if (p > prev) str_buf_cat(result, prev, p - prev);
8252 str_buf_cat2(result, "\"");
8253
8254 return result;
8255}
8256
8257#define IS_EVSTR(p,e) ((p) < (e) && (*(p) == '$' || *(p) == '@' || *(p) == '{'))
8258
8259/*
8260 * call-seq:
8261 * dump -> new_string
8262 *
8263 * :include: doc/string/dump.rdoc
8264 *
8265 */
8266
8267VALUE
8269{
8270 int encidx = rb_enc_get_index(str);
8271 rb_encoding *enc = rb_enc_from_index(encidx);
8272 long len;
8273 const char *p, *pend;
8274 char *q, *qend;
8275 VALUE result;
8276 int u8 = (encidx == rb_utf8_encindex());
8277 static const char nonascii_suffix[] = ".dup.force_encoding(\"%s\")";
8278
8279 len = 2; /* "" */
8280 if (!rb_enc_asciicompat(enc)) {
8281 len += strlen(nonascii_suffix) - rb_strlen_lit("%s");
8282 len += strlen(enc->name);
8283 }
8284
8285 p = RSTRING_PTR(str); pend = p + RSTRING_LEN(str);
8286 while (p < pend) {
8287 int clen;
8288 unsigned char c = *p++;
8289
8290 switch (c) {
8291 case '"': case '\\':
8292 case '\n': case '\r':
8293 case '\t': case '\f':
8294 case '\013': case '\010': case '\007': case '\033':
8295 clen = 2;
8296 break;
8297
8298 case '#':
8299 clen = IS_EVSTR(p, pend) ? 2 : 1;
8300 break;
8301
8302 default:
8303 if (ISPRINT(c)) {
8304 clen = 1;
8305 }
8306 else {
8307 if (u8 && c > 0x7F) { /* \u notation */
8308 int n = rb_enc_precise_mbclen(p-1, pend, enc);
8309 if (MBCLEN_CHARFOUND_P(n)) {
8310 unsigned int cc = rb_enc_mbc_to_codepoint(p-1, pend, enc);
8311 if (cc <= 0xFFFF)
8312 clen = 6; /* \uXXXX */
8313 else if (cc <= 0xFFFFF)
8314 clen = 9; /* \u{XXXXX} */
8315 else
8316 clen = 10; /* \u{XXXXXX} */
8317 p += MBCLEN_CHARFOUND_LEN(n)-1;
8318 break;
8319 }
8320 }
8321 clen = 4; /* \xNN */
8322 }
8323 break;
8324 }
8325
8326 if (clen > LONG_MAX - len) {
8327 rb_raise(rb_eRuntimeError, "string size too big");
8328 }
8329 len += clen;
8330 }
8331
8332 result = rb_str_new(0, len);
8333 p = RSTRING_PTR(str); pend = p + RSTRING_LEN(str);
8334 q = RSTRING_PTR(result); qend = q + len + 1;
8335
8336 *q++ = '"';
8337 while (p < pend) {
8338 unsigned char c = *p++;
8339
8340 if (c == '"' || c == '\\') {
8341 *q++ = '\\';
8342 *q++ = c;
8343 }
8344 else if (c == '#') {
8345 if (IS_EVSTR(p, pend)) *q++ = '\\';
8346 *q++ = '#';
8347 }
8348 else if (c == '\n') {
8349 *q++ = '\\';
8350 *q++ = 'n';
8351 }
8352 else if (c == '\r') {
8353 *q++ = '\\';
8354 *q++ = 'r';
8355 }
8356 else if (c == '\t') {
8357 *q++ = '\\';
8358 *q++ = 't';
8359 }
8360 else if (c == '\f') {
8361 *q++ = '\\';
8362 *q++ = 'f';
8363 }
8364 else if (c == '\013') {
8365 *q++ = '\\';
8366 *q++ = 'v';
8367 }
8368 else if (c == '\010') {
8369 *q++ = '\\';
8370 *q++ = 'b';
8371 }
8372 else if (c == '\007') {
8373 *q++ = '\\';
8374 *q++ = 'a';
8375 }
8376 else if (c == '\033') {
8377 *q++ = '\\';
8378 *q++ = 'e';
8379 }
8380 else if (ISPRINT(c)) {
8381 *q++ = c;
8382 }
8383 else {
8384 *q++ = '\\';
8385 if (u8) {
8386 int n = rb_enc_precise_mbclen(p-1, pend, enc) - 1;
8387 if (MBCLEN_CHARFOUND_P(n)) {
8388 int cc = rb_enc_mbc_to_codepoint(p-1, pend, enc);
8389 p += n;
8390 if (cc <= 0xFFFF)
8391 snprintf(q, qend-q, "u%04X", cc); /* \uXXXX */
8392 else
8393 snprintf(q, qend-q, "u{%X}", cc); /* \u{XXXXX} or \u{XXXXXX} */
8394 q += strlen(q);
8395 continue;
8396 }
8397 }
8398 snprintf(q, qend-q, "x%02X", c);
8399 q += 3;
8400 }
8401 }
8402 *q++ = '"';
8403 *q = '\0';
8404 if (!rb_enc_asciicompat(enc)) {
8405 snprintf(q, qend-q, nonascii_suffix, enc->name);
8406 encidx = rb_ascii8bit_encindex();
8407 }
8408 /* result from dump is ASCII */
8409 rb_enc_associate_index(result, encidx);
8411 return result;
8412}
8413
8414static int
8415unescape_ascii(unsigned int c)
8416{
8417 switch (c) {
8418 case 'n':
8419 return '\n';
8420 case 'r':
8421 return '\r';
8422 case 't':
8423 return '\t';
8424 case 'f':
8425 return '\f';
8426 case 'v':
8427 return '\13';
8428 case 'b':
8429 return '\010';
8430 case 'a':
8431 return '\007';
8432 case 'e':
8433 return 033;
8434 }
8436}
8437
8438static void
8439undump_after_backslash(VALUE undumped, const char **ss, const char *s_end, rb_encoding **penc, bool *utf8, bool *binary)
8440{
8441 const char *s = *ss;
8442 unsigned int c;
8443 int codelen;
8444 size_t hexlen;
8445 unsigned char buf[6];
8446 static rb_encoding *enc_utf8 = NULL;
8447
8448 switch (*s) {
8449 case '\\':
8450 case '"':
8451 case '#':
8452 rb_str_cat(undumped, s, 1); /* cat itself */
8453 s++;
8454 break;
8455 case 'n':
8456 case 'r':
8457 case 't':
8458 case 'f':
8459 case 'v':
8460 case 'b':
8461 case 'a':
8462 case 'e':
8463 *buf = unescape_ascii(*s);
8464 rb_str_cat(undumped, (char *)buf, 1);
8465 s++;
8466 break;
8467 case 'u':
8468 if (*binary) {
8469 rb_raise(rb_eRuntimeError, "hex escape and Unicode escape are mixed");
8470 }
8471 *utf8 = true;
8472 if (++s >= s_end) {
8473 rb_raise(rb_eRuntimeError, "invalid Unicode escape");
8474 }
8475 if (enc_utf8 == NULL) enc_utf8 = rb_utf8_encoding();
8476 if (*penc != enc_utf8) {
8477 *penc = enc_utf8;
8478 rb_enc_associate(undumped, enc_utf8);
8479 }
8480 if (*s == '{') { /* handle \u{...} form */
8481 s++;
8482 for (;;) {
8483 if (s >= s_end) {
8484 rb_raise(rb_eRuntimeError, "unterminated Unicode escape");
8485 }
8486 if (*s == '}') {
8487 s++;
8488 break;
8489 }
8490 if (ISSPACE(*s)) {
8491 s++;
8492 continue;
8493 }
8494 c = scan_hex(s, s_end-s, &hexlen);
8495 if (hexlen == 0 || hexlen > 6) {
8496 rb_raise(rb_eRuntimeError, "invalid Unicode escape");
8497 }
8498 if (c > 0x10ffff) {
8499 rb_raise(rb_eRuntimeError, "invalid Unicode codepoint (too large)");
8500 }
8501 if (0xd800 <= c && c <= 0xdfff) {
8502 rb_raise(rb_eRuntimeError, "invalid Unicode codepoint");
8503 }
8504 codelen = rb_enc_mbcput(c, (char *)buf, *penc);
8505 rb_str_cat(undumped, (char *)buf, codelen);
8506 s += hexlen;
8507 }
8508 }
8509 else { /* handle \uXXXX form */
8510 c = scan_hex(s, 4, &hexlen);
8511 if (hexlen != 4) {
8512 rb_raise(rb_eRuntimeError, "invalid Unicode escape");
8513 }
8514 if (0xd800 <= c && c <= 0xdfff) {
8515 rb_raise(rb_eRuntimeError, "invalid Unicode codepoint");
8516 }
8517 codelen = rb_enc_mbcput(c, (char *)buf, *penc);
8518 rb_str_cat(undumped, (char *)buf, codelen);
8519 s += hexlen;
8520 }
8521 break;
8522 case 'x':
8523 if (++s >= s_end) {
8524 rb_raise(rb_eRuntimeError, "invalid hex escape");
8525 }
8526 *buf = scan_hex(s, 2, &hexlen);
8527 if (hexlen != 2) {
8528 rb_raise(rb_eRuntimeError, "invalid hex escape");
8529 }
8530 if (!ISASCII(*buf)) {
8531 if (*utf8) {
8532 rb_raise(rb_eRuntimeError, "hex escape and Unicode escape are mixed");
8533 }
8534 *binary = true;
8535 }
8536 rb_str_cat(undumped, (char *)buf, 1);
8537 s += hexlen;
8538 break;
8539 default:
8540 rb_str_cat(undumped, s-1, 2);
8541 s++;
8542 }
8543
8544 *ss = s;
8545}
8546
8547static VALUE rb_str_is_ascii_only_p(VALUE str);
8548
8549/*
8550 * call-seq:
8551 * undump -> new_string
8552 *
8553 * Inverse of String#dump; returns a copy of +self+ with changes of the kinds made by String#dump "undone."
8554 *
8555 * Related: see {Converting to New String}[rdoc-ref:String@Converting+to+New+String].
8556 */
8557
8558static VALUE
8559str_undump(VALUE str)
8560{
8561 const char *s = RSTRING_PTR(str);
8562 const char *s_end = RSTRING_END(str);
8563 rb_encoding *enc = rb_enc_get(str);
8564 VALUE undumped = rb_enc_str_new(s, 0L, enc);
8565 bool utf8 = false;
8566 bool binary = false;
8567 int w;
8568
8570 if (rb_str_is_ascii_only_p(str) == Qfalse) {
8571 rb_raise(rb_eRuntimeError, "non-ASCII character detected");
8572 }
8573 if (!str_null_check(str, &w)) {
8574 rb_raise(rb_eRuntimeError, "string contains null byte");
8575 }
8576 if (RSTRING_LEN(str) < 2) goto invalid_format;
8577 if (*s != '"') goto invalid_format;
8578
8579 /* strip '"' at the start */
8580 s++;
8581
8582 for (;;) {
8583 if (s >= s_end) {
8584 rb_raise(rb_eRuntimeError, "unterminated dumped string");
8585 }
8586
8587 if (*s == '"') {
8588 /* epilogue */
8589 s++;
8590 if (s == s_end) {
8591 /* ascii compatible dumped string */
8592 break;
8593 }
8594 else {
8595 static const char force_encoding_suffix[] = ".force_encoding(\""; /* "\")" */
8596 static const char dup_suffix[] = ".dup";
8597 const char *encname;
8598 int encidx;
8599 ptrdiff_t size;
8600
8601 /* check separately for strings dumped by older versions */
8602 size = sizeof(dup_suffix) - 1;
8603 if (s_end - s > size && memcmp(s, dup_suffix, size) == 0) s += size;
8604
8605 size = sizeof(force_encoding_suffix) - 1;
8606 if (s_end - s <= size) goto invalid_format;
8607 if (memcmp(s, force_encoding_suffix, size) != 0) goto invalid_format;
8608 s += size;
8609
8610 if (utf8) {
8611 rb_raise(rb_eRuntimeError, "dumped string contained Unicode escape but used force_encoding");
8612 }
8613
8614 encname = s;
8615 s = memchr(s, '"', s_end-s);
8616 size = s - encname;
8617 if (!s) goto invalid_format;
8618 if (s_end - s != 2) goto invalid_format;
8619 if (s[0] != '"' || s[1] != ')') goto invalid_format;
8620
8621 encidx = rb_enc_find_index2(encname, (long)size);
8622 if (encidx < 0) {
8623 rb_raise(rb_eRuntimeError, "dumped string has unknown encoding name");
8624 }
8625 rb_enc_associate_index(undumped, encidx);
8626 }
8627 break;
8628 }
8629
8630 if (*s == '\\') {
8631 s++;
8632 if (s >= s_end) {
8633 rb_raise(rb_eRuntimeError, "invalid escape");
8634 }
8635 undump_after_backslash(undumped, &s, s_end, &enc, &utf8, &binary);
8636 }
8637 else {
8638 rb_str_cat(undumped, s++, 1);
8639 }
8640 }
8641
8642 RB_GC_GUARD(str);
8643
8644 return undumped;
8645invalid_format:
8646 rb_raise(rb_eRuntimeError, "invalid dumped string; not wrapped with '\"' nor '\"...\".force_encoding(\"...\")' form");
8647}
8648
8649static void
8650rb_str_check_dummy_enc(rb_encoding *enc)
8651{
8652 if (rb_enc_dummy_p(enc)) {
8653 rb_raise(rb_eEncCompatError, "incompatible encoding with this operation: %s",
8654 rb_enc_name(enc));
8655 }
8656}
8657
8658static rb_encoding *
8659str_true_enc(VALUE str)
8660{
8661 rb_encoding *enc = STR_ENC_GET(str);
8662 rb_str_check_dummy_enc(enc);
8663 return enc;
8664}
8665
8666static OnigCaseFoldType
8667check_case_options(int argc, VALUE *argv, OnigCaseFoldType flags)
8668{
8669 if (argc==0)
8670 return flags;
8671 if (argc>2)
8672 rb_raise(rb_eArgError, "too many options");
8673 if (argv[0]==sym_turkic) {
8674 flags |= ONIGENC_CASE_FOLD_TURKISH_AZERI;
8675 if (argc==2) {
8676 if (argv[1]==sym_lithuanian)
8677 flags |= ONIGENC_CASE_FOLD_LITHUANIAN;
8678 else
8679 rb_raise(rb_eArgError, "invalid second option");
8680 }
8681 }
8682 else if (argv[0]==sym_lithuanian) {
8683 flags |= ONIGENC_CASE_FOLD_LITHUANIAN;
8684 if (argc==2) {
8685 if (argv[1]==sym_turkic)
8686 flags |= ONIGENC_CASE_FOLD_TURKISH_AZERI;
8687 else
8688 rb_raise(rb_eArgError, "invalid second option");
8689 }
8690 }
8691 else if (argc>1)
8692 rb_raise(rb_eArgError, "too many options");
8693 else if (argv[0]==sym_ascii)
8694 flags |= ONIGENC_CASE_ASCII_ONLY;
8695 else if (argv[0]==sym_fold) {
8696 if ((flags & (ONIGENC_CASE_UPCASE|ONIGENC_CASE_DOWNCASE)) == ONIGENC_CASE_DOWNCASE)
8697 flags ^= ONIGENC_CASE_FOLD|ONIGENC_CASE_DOWNCASE;
8698 else
8699 rb_raise(rb_eArgError, "option :fold only allowed for downcasing");
8700 }
8701 else
8702 rb_raise(rb_eArgError, "invalid option");
8703 return flags;
8704}
8705
8706static inline bool
8707case_option_single_p(OnigCaseFoldType flags, rb_encoding *enc, VALUE str)
8708{
8709 if ((flags & ONIGENC_CASE_ASCII_ONLY) && (enc==rb_utf8_encoding() || rb_enc_mbmaxlen(enc) == 1))
8710 return true;
8711 return !(flags & ONIGENC_CASE_FOLD_TURKISH_AZERI) &&
8712 (ENC_CODERANGE(str) == ENC_CODERANGE_7BIT || rb_is_ascii8bit_enc(enc));
8713}
8714
8715/* 16 should be long enough to absorb any kind of single character length increase */
8716#define CASE_MAPPING_ADDITIONAL_LENGTH 20
8717#ifndef CASEMAP_DEBUG
8718# define CASEMAP_DEBUG 0
8719#endif
8720
8721struct mapping_buffer;
8722typedef struct mapping_buffer {
8723 size_t capa;
8724 size_t used;
8725 struct mapping_buffer *next;
8726 OnigUChar space[FLEX_ARY_LEN];
8728
8729static void
8730mapping_buffer_free(void *p)
8731{
8732 mapping_buffer *previous_buffer;
8733 mapping_buffer *current_buffer = p;
8734 while (current_buffer) {
8735 previous_buffer = current_buffer;
8736 current_buffer = current_buffer->next;
8737 ruby_xfree_sized(previous_buffer, offsetof(mapping_buffer, space) + previous_buffer->capa);
8738 }
8739}
8740
8741static const rb_data_type_t mapping_buffer_type = {
8742 "mapping_buffer",
8743 {0, mapping_buffer_free,},
8744 0, 0, RUBY_TYPED_THREAD_SAFE_FREE | RUBY_TYPED_WB_PROTECTED
8745};
8746
8747static VALUE
8748rb_str_casemap(VALUE source, OnigCaseFoldType *flags, rb_encoding *enc)
8749{
8750 VALUE target;
8751
8752 const OnigUChar *source_current, *source_end;
8753 int target_length = 0;
8754 VALUE buffer_anchor;
8755 mapping_buffer *current_buffer = 0;
8756 mapping_buffer **pre_buffer;
8757 size_t buffer_count = 0;
8758 int buffer_length_or_invalid;
8759
8760 if (RSTRING_LEN(source) == 0) return str_duplicate(rb_cString, source);
8761
8762 source_current = (OnigUChar*)RSTRING_PTR(source);
8763 source_end = (OnigUChar*)RSTRING_END(source);
8764
8765 buffer_anchor = TypedData_Wrap_Struct(0, &mapping_buffer_type, 0);
8766 pre_buffer = (mapping_buffer **)&DATA_PTR(buffer_anchor);
8767 while (source_current < source_end) {
8768 /* increase multiplier using buffer count to converge quickly */
8769 size_t capa = (size_t)(source_end-source_current)*++buffer_count + CASE_MAPPING_ADDITIONAL_LENGTH;
8770 if (CASEMAP_DEBUG) {
8771 fprintf(stderr, "Buffer allocation, capa is %"PRIuSIZE"\n", capa); /* for tuning */
8772 }
8773 current_buffer = xmalloc(offsetof(mapping_buffer, space) + capa);
8774 *pre_buffer = current_buffer;
8775 pre_buffer = &current_buffer->next;
8776 current_buffer->next = NULL;
8777 current_buffer->capa = capa;
8778 buffer_length_or_invalid = enc->case_map(flags,
8779 &source_current, source_end,
8780 current_buffer->space,
8781 current_buffer->space+current_buffer->capa,
8782 enc);
8783 if (buffer_length_or_invalid < 0) {
8784 current_buffer = DATA_PTR(buffer_anchor);
8785 DATA_PTR(buffer_anchor) = 0;
8786 mapping_buffer_free(current_buffer);
8787 rb_raise(rb_eArgError, "input string invalid");
8788 }
8789 target_length += current_buffer->used = buffer_length_or_invalid;
8790 }
8791 if (CASEMAP_DEBUG) {
8792 fprintf(stderr, "Buffer count is %"PRIuSIZE"\n", buffer_count); /* for tuning */
8793 }
8794
8795 if (buffer_count==1) {
8796 target = rb_str_new((const char*)current_buffer->space, target_length);
8797 }
8798 else {
8799 char *target_current;
8800
8801 target = rb_str_new(0, target_length);
8802 target_current = RSTRING_PTR(target);
8803 current_buffer = DATA_PTR(buffer_anchor);
8804 while (current_buffer) {
8805 memcpy(target_current, current_buffer->space, current_buffer->used);
8806 target_current += current_buffer->used;
8807 current_buffer = current_buffer->next;
8808 }
8809 }
8810 current_buffer = DATA_PTR(buffer_anchor);
8811 DATA_PTR(buffer_anchor) = 0;
8812 mapping_buffer_free(current_buffer);
8813
8814 RB_GC_GUARD(buffer_anchor);
8815
8816 /* TODO: check about string terminator character */
8817 str_enc_copy_direct(target, source);
8818 /*ENC_CODERANGE_SET(mapped, cr);*/
8819
8820 return target;
8821}
8822
8823static VALUE
8824rb_str_ascii_casemap(VALUE source, VALUE target, OnigCaseFoldType *flags, rb_encoding *enc)
8825{
8826 const OnigUChar *source_current, *source_end;
8827 OnigUChar *target_current, *target_end;
8828 long old_length = RSTRING_LEN(source);
8829 int length_or_invalid;
8830
8831 if (old_length == 0) return Qnil;
8832
8833 source_current = (OnigUChar*)RSTRING_PTR(source);
8834 source_end = (OnigUChar*)RSTRING_END(source);
8835 if (source == target) {
8836 target_current = (OnigUChar*)source_current;
8837 target_end = (OnigUChar*)source_end;
8838 }
8839 else {
8840 target_current = (OnigUChar*)RSTRING_PTR(target);
8841 target_end = (OnigUChar*)RSTRING_END(target);
8842 }
8843
8844 length_or_invalid = onigenc_ascii_only_case_map(flags,
8845 &source_current, source_end,
8846 target_current, target_end, enc);
8847 if (length_or_invalid < 0)
8848 rb_raise(rb_eArgError, "input string invalid");
8849 if (CASEMAP_DEBUG && length_or_invalid != old_length) {
8850 fprintf(stderr, "problem with rb_str_ascii_casemap"
8851 "; old_length=%ld, new_length=%d\n", old_length, length_or_invalid);
8852 rb_raise(rb_eArgError, "internal problem with rb_str_ascii_casemap"
8853 "; old_length=%ld, new_length=%d\n", old_length, length_or_invalid);
8854 }
8855
8856 str_enc_copy(target, source);
8857
8858 return target;
8859}
8860
8861static bool
8862upcase_single(VALUE str)
8863{
8864 char *s = RSTRING_PTR(str), *send = RSTRING_END(str);
8865 bool modified = false;
8866
8867 while (s < send) {
8868 unsigned int c = *(unsigned char*)s;
8869
8870 if ('a' <= c && c <= 'z') {
8871 *s = 'A' + (c - 'a');
8872 modified = true;
8873 }
8874 s++;
8875 }
8876 return modified;
8877}
8878
8879/*
8880 * call-seq:
8881 * upcase!(mapping) -> self or nil
8882 *
8883 * Like String#upcase, except that:
8884 *
8885 * - Changes character casings in +self+ (not in a copy of +self+).
8886 * - Returns +self+ if any changes are made, +nil+ otherwise.
8887 *
8888 * Related: See {Modifying}[rdoc-ref:String@Modifying].
8889 */
8890
8891static VALUE
8892rb_str_upcase_bang(int argc, VALUE *argv, VALUE str)
8893{
8894 rb_encoding *enc;
8895 OnigCaseFoldType flags = ONIGENC_CASE_UPCASE;
8896
8897 flags = check_case_options(argc, argv, flags);
8898 str_modify_keep_cr(str);
8899 enc = str_true_enc(str);
8900 if (case_option_single_p(flags, enc, str)) {
8901 if (upcase_single(str))
8902 flags |= ONIGENC_CASE_MODIFIED;
8903 }
8904 else if (flags&ONIGENC_CASE_ASCII_ONLY)
8905 rb_str_ascii_casemap(str, str, &flags, enc);
8906 else
8907 str_shared_replace(str, rb_str_casemap(str, &flags, enc));
8908
8909 if (ONIGENC_CASE_MODIFIED&flags) return str;
8910 return Qnil;
8911}
8912
8913
8914/*
8915 * call-seq:
8916 * upcase(mapping = :ascii) -> new_string
8917 *
8918 * :include: doc/string/upcase.rdoc
8919 */
8920
8921static VALUE
8922rb_str_upcase(int argc, VALUE *argv, VALUE str)
8923{
8924 rb_encoding *enc;
8925 OnigCaseFoldType flags = ONIGENC_CASE_UPCASE;
8926 VALUE ret;
8927
8928 flags = check_case_options(argc, argv, flags);
8929 enc = str_true_enc(str);
8930 if (case_option_single_p(flags, enc, str)) {
8931 ret = rb_str_new(RSTRING_PTR(str), RSTRING_LEN(str));
8932 str_enc_copy_direct(ret, str);
8933 upcase_single(ret);
8934 }
8935 else if (flags&ONIGENC_CASE_ASCII_ONLY) {
8936 ret = rb_str_new(0, RSTRING_LEN(str));
8937 rb_str_ascii_casemap(str, ret, &flags, enc);
8938 }
8939 else {
8940 ret = rb_str_casemap(str, &flags, enc);
8941 }
8942
8943 return ret;
8944}
8945
8946static bool
8947downcase_single(VALUE str)
8948{
8949 char *s = RSTRING_PTR(str), *send = RSTRING_END(str);
8950 bool modified = false;
8951
8952 while (s < send) {
8953 unsigned int c = *(unsigned char*)s;
8954
8955 if ('A' <= c && c <= 'Z') {
8956 *s = 'a' + (c - 'A');
8957 modified = true;
8958 }
8959 s++;
8960 }
8961
8962 return modified;
8963}
8964
8965/*
8966 * call-seq:
8967 * downcase!(mapping) -> self or nil
8968 *
8969 * Like String#downcase, except that:
8970 *
8971 * - Changes character casings in +self+ (not in a copy of +self+).
8972 * - Returns +self+ if any changes are made, +nil+ otherwise.
8973 *
8974 * Related: See {Modifying}[rdoc-ref:String@Modifying].
8975 */
8976
8977static VALUE
8978rb_str_downcase_bang(int argc, VALUE *argv, VALUE str)
8979{
8980 rb_encoding *enc;
8981 OnigCaseFoldType flags = ONIGENC_CASE_DOWNCASE;
8982
8983 flags = check_case_options(argc, argv, flags);
8984 str_modify_keep_cr(str);
8985 enc = str_true_enc(str);
8986 if (case_option_single_p(flags, enc, str)) {
8987 if (downcase_single(str))
8988 flags |= ONIGENC_CASE_MODIFIED;
8989 }
8990 else if (flags&ONIGENC_CASE_ASCII_ONLY)
8991 rb_str_ascii_casemap(str, str, &flags, enc);
8992 else
8993 str_shared_replace(str, rb_str_casemap(str, &flags, enc));
8994
8995 if (ONIGENC_CASE_MODIFIED&flags) return str;
8996 return Qnil;
8997}
8998
8999
9000/*
9001 * call-seq:
9002 * downcase(mapping = :ascii) -> new_string
9003 *
9004 * :include: doc/string/downcase.rdoc
9005 *
9006 */
9007
9008static VALUE
9009rb_str_downcase(int argc, VALUE *argv, VALUE str)
9010{
9011 rb_encoding *enc;
9012 OnigCaseFoldType flags = ONIGENC_CASE_DOWNCASE;
9013 VALUE ret;
9014
9015 flags = check_case_options(argc, argv, flags);
9016 enc = str_true_enc(str);
9017 if (case_option_single_p(flags, enc, str)) {
9018 ret = rb_str_new(RSTRING_PTR(str), RSTRING_LEN(str));
9019 str_enc_copy_direct(ret, str);
9020 downcase_single(ret);
9021 }
9022 else if (flags&ONIGENC_CASE_ASCII_ONLY) {
9023 ret = rb_str_new(0, RSTRING_LEN(str));
9024 rb_str_ascii_casemap(str, ret, &flags, enc);
9025 }
9026 else {
9027 ret = rb_str_casemap(str, &flags, enc);
9028 }
9029
9030 return ret;
9031}
9032
9033static bool
9034capitalize_single(VALUE str)
9035{
9036 char *s = RSTRING_PTR(str), *send = RSTRING_END(str);
9037 bool modified = false;
9038
9039 if (s < send) {
9040 unsigned int c = (unsigned char)*s;
9041
9042 if ('a' <= c && c <= 'z') {
9043 *s = 'A' + (c - 'a');
9044 modified = true;
9045 }
9046 s++;
9047 }
9048 while (s < send) {
9049 unsigned int c = (unsigned char)*s;
9050
9051 if ('A' <= c && c <= 'Z') {
9052 *s = 'a' + (c - 'A');
9053 modified = true;
9054 }
9055 s++;
9056 }
9057
9058 return modified;
9059}
9060
9061/*
9062 * call-seq:
9063 * capitalize!(mapping = :ascii) -> self or nil
9064 *
9065 * Like String#capitalize, except that:
9066 *
9067 * - Changes character casings in +self+ (not in a copy of +self+).
9068 * - Returns +self+ if any changes are made, +nil+ otherwise.
9069 *
9070 * Related: See {Modifying}[rdoc-ref:String@Modifying].
9071 */
9072
9073static VALUE
9074rb_str_capitalize_bang(int argc, VALUE *argv, VALUE str)
9075{
9076 rb_encoding *enc;
9077 OnigCaseFoldType flags = ONIGENC_CASE_UPCASE | ONIGENC_CASE_TITLECASE;
9078
9079 flags = check_case_options(argc, argv, flags);
9080 str_modify_keep_cr(str);
9081 enc = str_true_enc(str);
9082 if (RSTRING_LEN(str) == 0 || !RSTRING_PTR(str)) return Qnil;
9083 if (case_option_single_p(flags, enc, str)) {
9084 if (capitalize_single(str))
9085 flags |= ONIGENC_CASE_MODIFIED;
9086 }
9087 else if (flags&ONIGENC_CASE_ASCII_ONLY)
9088 rb_str_ascii_casemap(str, str, &flags, enc);
9089 else
9090 str_shared_replace(str, rb_str_casemap(str, &flags, enc));
9091
9092 if (ONIGENC_CASE_MODIFIED&flags) return str;
9093 return Qnil;
9094}
9095
9096
9097/*
9098 * call-seq:
9099 * capitalize(mapping = :ascii) -> new_string
9100 *
9101 * :include: doc/string/capitalize.rdoc
9102 *
9103 */
9104
9105static VALUE
9106rb_str_capitalize(int argc, VALUE *argv, VALUE str)
9107{
9108 rb_encoding *enc;
9109 OnigCaseFoldType flags = ONIGENC_CASE_UPCASE | ONIGENC_CASE_TITLECASE;
9110 VALUE ret;
9111
9112 flags = check_case_options(argc, argv, flags);
9113 enc = str_true_enc(str);
9114 if (RSTRING_LEN(str) == 0 || !RSTRING_PTR(str)) return str;
9115 if (case_option_single_p(flags, enc, str)) {
9116 ret = rb_str_new(RSTRING_PTR(str), RSTRING_LEN(str));
9117 str_enc_copy_direct(ret, str);
9118 capitalize_single(ret);
9119 }
9120 else if (flags&ONIGENC_CASE_ASCII_ONLY) {
9121 ret = rb_str_new(0, RSTRING_LEN(str));
9122 rb_str_ascii_casemap(str, ret, &flags, enc);
9123 }
9124 else {
9125 ret = rb_str_casemap(str, &flags, enc);
9126 }
9127 return ret;
9128}
9129
9130
9131/*
9132 * call-seq:
9133 * swapcase!(mapping) -> self or nil
9134 *
9135 * Like String#swapcase, except that:
9136 *
9137 * - Changes are made to +self+, not to copy of +self+.
9138 * - Returns +self+ if any changes are made, +nil+ otherwise.
9139 *
9140 * Related: see {Modifying}[rdoc-ref:String@Modifying].
9141 */
9142
9143static VALUE
9144rb_str_swapcase_bang(int argc, VALUE *argv, VALUE str)
9145{
9146 rb_encoding *enc;
9147 OnigCaseFoldType flags = ONIGENC_CASE_UPCASE | ONIGENC_CASE_DOWNCASE;
9148
9149 flags = check_case_options(argc, argv, flags);
9150 str_modify_keep_cr(str);
9151 enc = str_true_enc(str);
9152 if (flags&ONIGENC_CASE_ASCII_ONLY)
9153 rb_str_ascii_casemap(str, str, &flags, enc);
9154 else
9155 str_shared_replace(str, rb_str_casemap(str, &flags, enc));
9156
9157 if (ONIGENC_CASE_MODIFIED&flags) return str;
9158 return Qnil;
9159}
9160
9161
9162/*
9163 * call-seq:
9164 * swapcase(mapping = :ascii) -> new_string
9165 *
9166 * :include: doc/string/swapcase.rdoc
9167 *
9168 */
9169
9170static VALUE
9171rb_str_swapcase(int argc, VALUE *argv, VALUE str)
9172{
9173 rb_encoding *enc;
9174 OnigCaseFoldType flags = ONIGENC_CASE_UPCASE | ONIGENC_CASE_DOWNCASE;
9175 VALUE ret;
9176
9177 flags = check_case_options(argc, argv, flags);
9178 enc = str_true_enc(str);
9179 if (RSTRING_LEN(str) == 0 || !RSTRING_PTR(str)) return str_duplicate(rb_cString, str);
9180 if (flags&ONIGENC_CASE_ASCII_ONLY) {
9181 ret = rb_str_new(0, RSTRING_LEN(str));
9182 rb_str_ascii_casemap(str, ret, &flags, enc);
9183 }
9184 else {
9185 ret = rb_str_casemap(str, &flags, enc);
9186 }
9187 return ret;
9188}
9189
9190typedef unsigned char *USTR;
9191
9192struct tr {
9193 int gen;
9194 unsigned int now, max;
9195 const char *p, *pend;
9196};
9197
9198static unsigned int
9199trnext(struct tr *t, rb_encoding *enc)
9200{
9201 int n;
9202
9203 for (;;) {
9204 nextpart:
9205 if (!t->gen) {
9206 if (t->p == t->pend) return -1;
9207 if (rb_enc_ascget(t->p, t->pend, &n, enc) == '\\' && t->p + n < t->pend) {
9208 t->p += n;
9209 }
9210 t->now = rb_enc_codepoint_len(t->p, t->pend, &n, enc);
9211 t->p += n;
9212 if (rb_enc_ascget(t->p, t->pend, &n, enc) == '-' && t->p + n < t->pend) {
9213 t->p += n;
9214 if (t->p < t->pend) {
9215 unsigned int c = rb_enc_codepoint_len(t->p, t->pend, &n, enc);
9216 t->p += n;
9217 if (t->now > c) {
9218 if (t->now < 0x80 && c < 0x80) {
9219 rb_raise(rb_eArgError,
9220 "invalid range \"%c-%c\" in string transliteration",
9221 t->now, c);
9222 }
9223 else {
9224 rb_raise(rb_eArgError, "invalid range in string transliteration");
9225 }
9226 continue; /* not reached */
9227 }
9228 else if (t->now < c) {
9229 t->gen = 1;
9230 t->max = c;
9231 }
9232 }
9233 }
9234 return t->now;
9235 }
9236 else {
9237 while (ONIGENC_CODE_TO_MBCLEN(enc, ++t->now) <= 0) {
9238 if (t->now == t->max) {
9239 t->gen = 0;
9240 goto nextpart;
9241 }
9242 }
9243 if (t->now < t->max) {
9244 return t->now;
9245 }
9246 else {
9247 t->gen = 0;
9248 return t->max;
9249 }
9250 }
9251 }
9252}
9253
9254static VALUE rb_str_delete_bang(int,VALUE*,VALUE);
9255
9256static VALUE
9257tr_trans(VALUE str, VALUE src, VALUE repl, int sflag)
9258{
9259 const unsigned int errc = -1;
9260 unsigned int trans[256];
9261 rb_encoding *enc, *e1, *e2;
9262 struct tr trsrc, trrepl;
9263 int cflag = 0;
9264 unsigned int c, c0, last = 0;
9265 int modify = 0, i, l;
9266 unsigned char *s, *send;
9267 VALUE hash = 0;
9268 int singlebyte = single_byte_optimizable(str);
9269 int termlen;
9270 int cr;
9271
9272#define CHECK_IF_ASCII(c) \
9273 (void)((cr == ENC_CODERANGE_7BIT && !rb_isascii(c)) ? \
9274 (cr = ENC_CODERANGE_VALID) : 0)
9275
9276 StringValue(src);
9277 StringValue(repl);
9278 if (RSTRING_LEN(str) == 0 || !RSTRING_PTR(str)) return Qnil;
9279 if (RSTRING_LEN(repl) == 0) {
9280 return rb_str_delete_bang(1, &src, str);
9281 }
9282
9283 cr = ENC_CODERANGE(str);
9284 e1 = rb_enc_check(str, src);
9285 e2 = rb_enc_check(str, repl);
9286 if (e1 == e2) {
9287 enc = e1;
9288 }
9289 else {
9290 enc = rb_enc_check(src, repl);
9291 }
9292 trsrc.p = RSTRING_PTR(src); trsrc.pend = trsrc.p + RSTRING_LEN(src);
9293 if (RSTRING_LEN(src) > 1 &&
9294 rb_enc_ascget(trsrc.p, trsrc.pend, &l, enc) == '^' &&
9295 trsrc.p + l < trsrc.pend) {
9296 cflag = 1;
9297 trsrc.p += l;
9298 }
9299 trrepl.p = RSTRING_PTR(repl);
9300 trrepl.pend = trrepl.p + RSTRING_LEN(repl);
9301 trsrc.gen = trrepl.gen = 0;
9302 trsrc.now = trrepl.now = 0;
9303 trsrc.max = trrepl.max = 0;
9304
9305 if (cflag) {
9306 for (i=0; i<256; i++) {
9307 trans[i] = 1;
9308 }
9309 while ((c = trnext(&trsrc, enc)) != errc) {
9310 if (c < 256) {
9311 trans[c] = errc;
9312 }
9313 else {
9314 if (!hash) hash = rb_hash_new();
9315 rb_hash_aset(hash, UINT2NUM(c), Qtrue);
9316 }
9317 }
9318 while ((c = trnext(&trrepl, enc)) != errc)
9319 /* retrieve last replacer */;
9320 last = trrepl.now;
9321 for (i=0; i<256; i++) {
9322 if (trans[i] != errc) {
9323 trans[i] = last;
9324 }
9325 }
9326 }
9327 else {
9328 unsigned int r;
9329
9330 for (i=0; i<256; i++) {
9331 trans[i] = errc;
9332 }
9333 while ((c = trnext(&trsrc, enc)) != errc) {
9334 r = trnext(&trrepl, enc);
9335 if (r == errc) r = trrepl.now;
9336 if (c < 256) {
9337 trans[c] = r;
9338 if (rb_enc_codelen(r, enc) != 1) singlebyte = 0;
9339 }
9340 else {
9341 if (!hash) hash = rb_hash_new();
9342 rb_hash_aset(hash, UINT2NUM(c), UINT2NUM(r));
9343 }
9344 }
9345 }
9346
9347 if (cr == ENC_CODERANGE_VALID && rb_enc_asciicompat(e1))
9348 cr = ENC_CODERANGE_7BIT;
9349 str_modify_keep_cr(str);
9350 s = (unsigned char *)RSTRING_PTR(str); send = (unsigned char *)RSTRING_END(str);
9351 termlen = rb_enc_mbminlen(enc);
9352 if (sflag) {
9353 int clen, tlen;
9354 long offset, max = RSTRING_LEN(str);
9355 unsigned int save = -1;
9356 unsigned char *buf = ALLOC_N(unsigned char, max + termlen), *t = buf;
9357
9358 while (s < send) {
9359 int may_modify = 0;
9360
9361 int r = rb_enc_precise_mbclen((char *)s, (char *)send, e1);
9362 if (!MBCLEN_CHARFOUND_P(r)) {
9363 SIZED_FREE_N(buf, max + termlen);
9364 rb_raise(rb_eArgError, "invalid byte sequence in %s", rb_enc_name(e1));
9365 }
9366 clen = MBCLEN_CHARFOUND_LEN(r);
9367 c0 = c = rb_enc_mbc_to_codepoint((char *)s, (char *)send, e1);
9368
9369 tlen = enc == e1 ? clen : rb_enc_codelen(c, enc);
9370
9371 s += clen;
9372 if (c < 256) {
9373 c = trans[c];
9374 }
9375 else if (hash) {
9376 VALUE tmp = rb_hash_lookup(hash, UINT2NUM(c));
9377 if (NIL_P(tmp)) {
9378 if (cflag) c = last;
9379 else c = errc;
9380 }
9381 else if (cflag) c = errc;
9382 else c = NUM2INT(tmp);
9383 }
9384 else {
9385 c = errc;
9386 }
9387 if (c != (unsigned int)-1) {
9388 if (save == c) {
9389 CHECK_IF_ASCII(c);
9390 continue;
9391 }
9392 save = c;
9393 tlen = rb_enc_codelen(c, enc);
9394 modify = 1;
9395 }
9396 else {
9397 save = -1;
9398 c = c0;
9399 if (enc != e1) may_modify = 1;
9400 }
9401 if ((offset = t - buf) + tlen > max) {
9402 size_t MAYBE_UNUSED(old) = max + termlen;
9403 max = offset + tlen + (send - s);
9404 SIZED_REALLOC_N(buf, unsigned char, max + termlen, old);
9405 t = buf + offset;
9406 }
9407 rb_enc_mbcput(c, t, enc);
9408 if (may_modify && memcmp(s, t, tlen) != 0) {
9409 modify = 1;
9410 }
9411 CHECK_IF_ASCII(c);
9412 t += tlen;
9413 }
9414 if (!STR_EMBED_P(str)) {
9415 SIZED_FREE_N(STR_HEAP_PTR(str), STR_HEAP_SIZE(str));
9416 }
9417 TERM_FILL((char *)t, termlen);
9418 RSTRING(str)->as.heap.ptr = (char *)buf;
9419 STR_SET_LEN(str, t - buf);
9420 STR_SET_NOEMBED(str);
9421 RSTRING(str)->as.heap.aux.capa = max;
9422 }
9423 else if (rb_enc_mbmaxlen(enc) == 1 || (singlebyte && !hash)) {
9424 while (s < send) {
9425 c = (unsigned char)*s;
9426 if (trans[c] != errc) {
9427 if (!cflag) {
9428 c = trans[c];
9429 *s = c;
9430 modify = 1;
9431 }
9432 else {
9433 *s = last;
9434 modify = 1;
9435 }
9436 }
9437 CHECK_IF_ASCII(c);
9438 s++;
9439 }
9440 }
9441 else {
9442 int clen, tlen;
9443 long offset, max = (long)((send - s) * 1.2);
9444 unsigned char *buf = ALLOC_N(unsigned char, max + termlen), *t = buf;
9445
9446 while (s < send) {
9447 int may_modify = 0;
9448
9449 int r = rb_enc_precise_mbclen((char *)s, (char *)send, e1);
9450 if (!MBCLEN_CHARFOUND_P(r)) {
9451 SIZED_FREE_N(buf, max + termlen);
9452 rb_raise(rb_eArgError, "invalid byte sequence in %s", rb_enc_name(e1));
9453 }
9454 clen = MBCLEN_CHARFOUND_LEN(r);
9455 c0 = c = rb_enc_mbc_to_codepoint((char *)s, (char *)send, e1);
9456
9457 tlen = enc == e1 ? clen : rb_enc_codelen(c, enc);
9458
9459 if (c < 256) {
9460 c = trans[c];
9461 }
9462 else if (hash) {
9463 VALUE tmp = rb_hash_lookup(hash, UINT2NUM(c));
9464 if (NIL_P(tmp)) {
9465 if (cflag) c = last;
9466 else c = errc;
9467 }
9468 else if (cflag) c = errc;
9469 else c = NUM2INT(tmp);
9470 }
9471 else {
9472 c = cflag ? last : errc;
9473 }
9474 if (c != errc) {
9475 tlen = rb_enc_codelen(c, enc);
9476 modify = 1;
9477 }
9478 else {
9479 c = c0;
9480 if (enc != e1) may_modify = 1;
9481 }
9482 if ((offset = t - buf) + tlen > max) {
9483 size_t MAYBE_UNUSED(old) = max + termlen;
9484 max = offset + tlen + (long)((send - s) * 1.2);
9485 SIZED_REALLOC_N(buf, unsigned char, max + termlen, old);
9486 t = buf + offset;
9487 }
9488
9489 rb_enc_mbcput(c, t, enc);
9490 if (may_modify && memcmp(s, t, tlen) != 0) {
9491 modify = 1;
9492 }
9493 CHECK_IF_ASCII(c);
9494 s += clen;
9495 t += tlen;
9496 }
9497 if (!STR_EMBED_P(str)) {
9498 SIZED_FREE_N(STR_HEAP_PTR(str), STR_HEAP_SIZE(str));
9499 }
9500 TERM_FILL((char *)t, termlen);
9501 RSTRING(str)->as.heap.ptr = (char *)buf;
9502 STR_SET_LEN(str, t - buf);
9503 STR_SET_NOEMBED(str);
9504 RSTRING(str)->as.heap.aux.capa = max;
9505 }
9506
9507 if (modify) {
9508 if (cr != ENC_CODERANGE_BROKEN)
9509 ENC_CODERANGE_SET(str, cr);
9510 rb_enc_associate(str, enc);
9511 return str;
9512 }
9513 return Qnil;
9514}
9515
9517 unsigned char *buf;
9518 unsigned char *ptr;
9519 size_t capa;
9520 size_t initial_capa;
9521};
9522
9523static inline void
9524tr_buffer_init(struct tr_buffer *buffer, size_t initial_capa)
9525{
9526 if (initial_capa < 32) {
9527 initial_capa = 32;
9528 }
9529 *buffer = (struct tr_buffer){ .initial_capa = initial_capa };
9530}
9531
9532static inline void
9533tr_buffer_ensure_capa(struct tr_buffer *buffer, size_t extra_capa)
9534{
9535 size_t offset = buffer->ptr - buffer->buf;
9536 size_t required_capa = offset + extra_capa;
9537 if (UNLIKELY(buffer->capa < required_capa)) {
9538 size_t new_capa = buffer->capa ? buffer->capa : buffer->initial_capa;
9539 RUBY_ASSERT(new_capa >= 32); // Lower would cause infinite loop
9540 while (new_capa < required_capa) {
9541 new_capa = (size_t)(new_capa * 1.2);
9542 }
9543 SIZED_REALLOC_N(buffer->buf, unsigned char, new_capa, buffer->capa);
9544 buffer->ptr = buffer->buf + offset;
9545 buffer->capa = new_capa;
9546 }
9547}
9548
9549static inline void
9550tr_buffer_append(struct tr_buffer *buffer, const unsigned char *ptr, size_t len)
9551{
9552 if (len) {
9553 tr_buffer_ensure_capa(buffer, len);
9554 memcpy(buffer->ptr, ptr, len);
9555 buffer->ptr += len;
9556 }
9557}
9558
9559static inline void
9560tr_buffer_append_str(struct tr_buffer *buffer, VALUE str)
9561{
9562 tr_buffer_append(buffer, (unsigned char *)RSTRING_PTR(str), RSTRING_LEN(str));
9563}
9564
9565static inline void
9566tr_buffer_mbcput(struct tr_buffer *buffer, int codepoint, rb_encoding *enc)
9567{
9568 tr_buffer_ensure_capa(buffer, 4);
9569 buffer->ptr += rb_enc_mbcput(codepoint, buffer->ptr, enc);
9570}
9571
9572static inline void
9573tr_buffer_free(struct tr_buffer *buffer)
9574{
9575 if (buffer->buf) {
9576 SIZED_FREE_N(buffer->buf, buffer->capa);
9577 }
9578}
9579
9580struct tr_pair {
9581 VALUE search;
9582 VALUE replace;
9583};
9584
9586 struct tr_pair *pairs;
9587 size_t index;
9588 rb_encoding *enc;
9589 int cr;
9590};
9591
9592static int
9593tr_trans_pairs_coerce_i(st_data_t key, st_data_t value, st_data_t _args)
9594{
9595 struct tr_trans_pairs_coerce_args *args = (struct tr_trans_pairs_coerce_args *)_args;
9596 struct tr_pair *pair = &args->pairs[args->index];
9597 args->index++;
9598
9599 VALUE search = (VALUE)key;
9600 VALUE replace = (VALUE)value;
9601 StringValue(search);
9602 StringValue(replace);
9603
9604 if (RSTRING_LEN(search) != 1 && str_strlen(search, NULL) != 1) {
9605 rb_raise(rb_eArgError, "keys must be of size 1"); // TODO: better error message
9606 }
9607
9608 args->enc = rb_enc_check_multi_str(args->enc, &args->cr, search);
9609 args->enc = rb_enc_check_multi_str(args->enc, &args->cr, replace);
9610
9611 pair->search = search;
9612 pair->replace = replace;
9613 return ST_CONTINUE;
9614}
9615
9616#define TR_TRANS_PAIRS_SIMD_MAX_NEEDLES 16
9617
9619 const unsigned char *s;
9620 const unsigned char *send;
9621
9622#ifdef HAVE_SIMD
9623 unsigned char needles[TR_TRANS_PAIRS_SIMD_MAX_NEEDLES];
9624 unsigned int needles_count;
9625#ifdef HAVE_SIMD_NEON
9626 uint64_t matches_bitmap;
9627#endif
9628#ifdef HAVE_SIMD_SSE2
9629 int matches_bitmap;
9630#endif
9631#endif
9632
9633 VALUE trans_table[256];
9634};
9635
9636static inline VALUE
9637tr_trans_pairs_search_basic(struct tr_trans_pairs_search *search)
9638{
9639 while (search->s < search->send) {
9640 VALUE repl = search->trans_table[*search->s];
9641 if (UNLIKELY(repl)) {
9642 return repl;
9643 }
9644
9645 search->s++;
9646 }
9647
9648 return 0;
9649}
9650
9651#ifdef HAVE_SIMD_SSE2
9652static inline VALUE
9653tr_trans_pairs_next_match_sse2(struct tr_trans_pairs_search *search)
9654{
9655 RUBY_ASSERT(search->matches_bitmap > 0);
9656 size_t trailing_zeros = (size_t)ntz_int32(search->matches_bitmap);
9657
9658 RUBY_ASSERT(trailing_zeros < (sizeof(search->matches_bitmap) * CHAR_BIT));
9659 search->matches_bitmap >>= trailing_zeros;
9660 search->s += trailing_zeros;
9661
9662 RUBY_ASSERT(search->s <= search->send);
9663 return search->trans_table[*search->s];
9664}
9665
9666static inline VALUE
9667tr_trans_pairs_search_sse2(struct tr_trans_pairs_search *search)
9668{
9669 const unsigned int needles_count = search->needles_count;
9670 if (needles_count) {
9671 RBIMPL_ASSERT_OR_ASSUME(needles_count <= TR_TRANS_PAIRS_SIMD_MAX_NEEDLES);
9672
9673 if (search->matches_bitmap) {
9674 return tr_trans_pairs_next_match_sse2(search);
9675 }
9676
9677 if ((size_t)(search->send - search->s) >= sizeof(__m128i)) {
9678 unsigned int i;
9679 __m128i masks[TR_TRANS_PAIRS_SIMD_MAX_NEEDLES];
9680 for (i = 0; i < needles_count; i++) {
9681 masks[i] = _mm_set1_epi8(search->needles[i]);
9682 }
9683
9684 do {
9685 const __m128i bytes = _mm_loadu_si128((__m128i const *)search->s);
9686
9687 __m128i matches[TR_TRANS_PAIRS_SIMD_MAX_NEEDLES];
9688 for (i = 0; i < needles_count; i++) {
9689 matches[i] = _mm_cmpeq_epi8(bytes, masks[i]);
9690 }
9691
9692 for (i = 1; i < needles_count; i++) {
9693 matches[0] = _mm_or_si128(matches[0], matches[i]);
9694 }
9695
9696 const int bitmap = _mm_movemask_epi8(matches[0]);
9697
9698 if (bitmap) {
9699 search->matches_bitmap = bitmap;
9700 return tr_trans_pairs_next_match_sse2(search);
9701 }
9702 search->s += sizeof(__m128i);
9703 } while ((size_t)(search->send - search->s) >= sizeof(__m128i));
9704 }
9705 }
9706 return tr_trans_pairs_search_basic(search);
9707}
9708
9709#define tr_trans_pairs_search_impl tr_trans_pairs_search_sse2
9710#endif
9711
9712#ifdef HAVE_SIMD_NEON
9713static inline VALUE
9714tr_trans_pairs_next_match_neon(struct tr_trans_pairs_search *search)
9715{
9716 RUBY_ASSERT(search->matches_bitmap > 0);
9717 size_t trailing_zeros = (size_t)ntz_int64(search->matches_bitmap);
9718
9719 // uint64_t >>= 64 would be undefined behaviour
9720 RUBY_ASSERT(trailing_zeros < (sizeof(search->matches_bitmap) * CHAR_BIT));
9721 search->matches_bitmap >>= trailing_zeros;
9722 search->s += trailing_zeros / 4;
9723
9724 RUBY_ASSERT(search->s <= search->send);
9725 return search->trans_table[*search->s];
9726}
9727
9728static inline VALUE
9729tr_trans_pairs_search_neon(struct tr_trans_pairs_search *search)
9730{
9731 const unsigned int needles_count = search->needles_count;
9732 if (needles_count) {
9733 RBIMPL_ASSERT_OR_ASSUME(needles_count <= TR_TRANS_PAIRS_SIMD_MAX_NEEDLES);
9734
9735 if (search->matches_bitmap) {
9736 return tr_trans_pairs_next_match_neon(search);
9737 }
9738
9739 if ((size_t)(search->send - search->s) >= sizeof(uint8x16_t)) {
9740 unsigned int i;
9741 uint8x16_t masks[TR_TRANS_PAIRS_SIMD_MAX_NEEDLES];
9742 for (i = 0; i < needles_count; i++) {
9743 masks[i] = vdupq_n_u8(search->needles[i]);
9744 }
9745
9746 do {
9747 const uint8x16_t bytes = vld1q_u8(search->s);
9748
9749 uint8x16_t matches[TR_TRANS_PAIRS_SIMD_MAX_NEEDLES];
9750 for (i = 0; i < needles_count; i++) {
9751 matches[i] = vceqq_u8(bytes, masks[i]);
9752 }
9753
9754 for (i = 1; i < needles_count; i++) {
9755 matches[0] = vorrq_u8(matches[0], matches[i]);
9756 }
9757
9758 const uint8x8_t res = vshrn_n_u16(vreinterpretq_u16_u8(matches[0]), 4);
9759 const uint64_t bitmap = vget_lane_u64(vreinterpret_u64_u8(res), 0);
9760
9761 if (bitmap) {
9762 search->matches_bitmap = bitmap & 0x8888888888888888ull;
9763 return tr_trans_pairs_next_match_neon(search);
9764 }
9765 search->s += sizeof(uint8x16_t);
9766 } while ((size_t)(search->send - search->s) >= sizeof(uint8x16_t));
9767 }
9768 }
9769 return tr_trans_pairs_search_basic(search);
9770}
9771
9772#define tr_trans_pairs_search_impl tr_trans_pairs_search_neon
9773#endif
9774
9775#ifndef tr_trans_pairs_search_impl
9776#define tr_trans_pairs_search_impl tr_trans_pairs_search_basic
9777#endif
9778
9779static inline void
9780tr_trans_pairs_consume_match(struct tr_trans_pairs_search *search)
9781{
9782 search->s++;
9783#ifdef HAVE_SIMD
9784 search->matches_bitmap >>= 1;
9785#endif
9786}
9787
9788static VALUE
9789tr_trans_pairs(VALUE str, VALUE pairs_val)
9790{
9791 Check_Type(pairs_val, T_HASH);
9792 size_t pairs_count = RHASH_SIZE(pairs_val);
9793 mustnot_broken(str);
9794 rb_str_modify(str);
9795
9796 if (RSTRING_LEN(str) == 0 || !RSTRING_PTR(str) || pairs_count == 0) return Qnil;
9797
9798 VALUE pairs_handle;
9799 struct tr_pair *pairs = ALLOCV_N(struct tr_pair, pairs_handle, pairs_count);
9800
9801 int cr = rb_enc_str_coderange(str);
9802 rb_encoding *enc = rb_str_enc_get(str);
9803
9804 struct tr_trans_pairs_coerce_args coerce_args = {
9805 .pairs = pairs,
9806 .enc = enc,
9807 .cr = cr,
9808 };
9809 rb_hash_foreach(pairs_val, tr_trans_pairs_coerce_i, (VALUE)&coerce_args);
9810 rb_encoding *e1 = coerce_args.enc;
9811
9812 /* Keys could be deleted from pairs_val during rb_hash_foreach when coercing
9813 * the keys/values, so we need to update pairs_count to the number of pairs we
9814 * were actually able to extract from pairs_val. */
9815 pairs_count = coerce_args.index;
9816
9817 VALUE hash = 0;
9818
9819 const unsigned char *sstart = (unsigned char *)RSTRING_PTR(str);
9820 long str_len = RSTRING_LEN(str);
9821 int termlen = rb_enc_mbminlen(e1);
9822
9823 struct tr_buffer buffer;
9824 tr_buffer_init(&buffer, str_len);
9825 bool modify = false;
9826
9827 if (RB_LIKELY(rb_str_encindex_fastpath(rb_enc_to_index(e1)))) {
9828
9829 struct tr_trans_pairs_search search = {
9830 .s = sstart,
9831 .send = sstart + str_len,
9832 };
9833
9834 for (size_t index = 0; index < pairs_count; index++) {
9835 struct tr_pair *pair = &pairs[index];
9836
9837 char *ptr = RSTRING_PTR(pair->search);
9838 unsigned int codepoint = rb_enc_mbc_to_codepoint(ptr, RSTRING_END(pair->search), e1);
9839
9840 const unsigned char first_byte = (unsigned char)*ptr;
9841
9842#ifdef HAVE_SIMD
9843 if (pairs_count <= TR_TRANS_PAIRS_SIMD_MAX_NEEDLES) {
9844 search.needles[index] = first_byte;
9845 search.needles_count++;
9846 }
9847#endif
9848
9849 if (rb_enc_codelen(codepoint, e1) == 1) {
9850 search.trans_table[first_byte] = pair->replace;
9851 }
9852 else {
9853 search.trans_table[first_byte] = Qundef;
9854 if (!hash) {
9855 hash = rb_obj_hide(rb_hash_new_capa(pairs_count));
9856 }
9857 rb_hash_aset(hash, UINT2NUM(codepoint), pair->replace);
9858 }
9859 }
9860
9861 const unsigned char *checkpoint = search.s;
9862 VALUE repl;
9863 while ((repl = tr_trans_pairs_search_impl(&search))) {
9864 int clen = 1;
9865
9866 if (UNLIKELY(repl == Qundef)) {
9867 unsigned int c = rb_enc_mbc_to_codepoint((char *)search.s, (char *)search.send, e1);
9868 clen = rb_enc_codelen(c, e1);
9869 repl = rb_hash_lookup2(hash, UINT2NUM(c), 0);
9870 if (!repl) {
9871 tr_trans_pairs_consume_match(&search);
9872 continue;
9873 }
9874 }
9876
9877 modify = true;
9878
9879 if (checkpoint < search.s) {
9880 tr_buffer_append(&buffer, checkpoint, search.s - checkpoint);
9881 }
9882 tr_buffer_append_str(&buffer, repl);
9883 checkpoint = search.s + clen;
9884 tr_trans_pairs_consume_match(&search);
9885
9886 if (cr == ENC_CODERANGE_7BIT && rb_enc_str_coderange(repl) != ENC_CODERANGE_7BIT) {
9888 }
9889 }
9890
9891 if (modify && checkpoint < search.s) {
9892 tr_buffer_append(&buffer, checkpoint, search.s - checkpoint);
9893 }
9894 }
9895 else {
9896 const unsigned char *s = sstart;
9897 const unsigned char *send = sstart + str_len;
9898
9899 hash = rb_obj_hide(rb_hash_new_capa(pairs_count));
9900
9901 for (size_t index = 0; index < pairs_count; index++) {
9902 struct tr_pair *pair = &pairs[index];
9903
9904 unsigned int codepoint = rb_enc_mbc_to_codepoint(RSTRING_PTR(pair->search), RSTRING_END(pair->search), e1);
9905 rb_hash_aset(hash, UINT2NUM(codepoint), pair->replace);
9906 }
9907
9908 while (s < send) {
9909 bool may_modify = false;
9910
9911 int r = rb_enc_precise_mbclen((char *)s, (char *)send, e1);
9912 if (!MBCLEN_CHARFOUND_P(r)) {
9913 tr_buffer_free(&buffer);
9914 rb_raise(rb_eArgError, "invalid byte sequence in %s", rb_enc_name(e1));
9915 }
9916 int clen = MBCLEN_CHARFOUND_LEN(r);
9917 unsigned int c = rb_enc_mbc_to_codepoint((char *)s, (char *)send, e1);
9918 unsigned int c0 = c;
9919
9920 long tlen = enc == e1 ? clen : rb_enc_codelen(c, e1);
9921
9922 VALUE replacement = rb_hash_lookup(hash, UINT2NUM(c));
9923 if (NIL_P(replacement)) {
9924 tlen = enc == e1 ? clen : rb_enc_codelen(c, enc);
9925 c = c0;
9926 if (enc != e1) may_modify = true;
9927 }
9928 else {
9929 tlen = RSTRING_LEN(replacement);
9930 modify = true;
9931 }
9932
9933 if (NIL_P(replacement)) {
9934 tr_buffer_mbcput(&buffer, c, enc);
9935 }
9936 else {
9937 tr_buffer_append_str(&buffer, replacement);
9938 }
9939
9940 if (may_modify && memcmp(s, buffer.ptr - tlen, tlen) != 0) {
9941 modify = true;
9942 }
9943
9944 if (cr == ENC_CODERANGE_7BIT && !rb_isascii(c)) {
9946 }
9947
9948 s += clen;
9949 }
9950 }
9951
9952 ALLOCV_END(pairs_handle);
9953
9954 if (!modify) {
9955 return Qnil;
9956 }
9957
9958 if (!STR_EMBED_P(str)) {
9959 SIZED_FREE_N(STR_HEAP_PTR(str), STR_HEAP_SIZE(str));
9960 }
9961 tr_buffer_ensure_capa(&buffer, termlen);
9962 TERM_FILL((char *)buffer.ptr, termlen);
9963 RSTRING(str)->as.heap.ptr = (char *)buffer.buf;
9964 STR_SET_LEN(str, buffer.ptr - buffer.buf);
9965 STR_SET_NOEMBED(str);
9966 RSTRING(str)->as.heap.aux.capa = buffer.capa - termlen;
9967
9968 RB_GC_GUARD(hash);
9969
9970 if (cr != ENC_CODERANGE_BROKEN)
9971 ENC_CODERANGE_SET(str, cr);
9972 rb_enc_associate(str, e1);
9973 return str;
9974}
9975
9976/*
9977 * call-seq:
9978 * tr!(selector, replacements) -> self or nil
9979 * tr!(pairs) -> self or nil
9980 *
9981 * Like String#tr, except:
9982 *
9983 * - Performs substitutions in +self+ (not in a copy of +self+).
9984 * - Returns +self+ if any modifications were made, +nil+ otherwise.
9985 *
9986 * Related: {Modifying}[rdoc-ref:String@Modifying].
9987 */
9988
9989static VALUE
9990rb_str_tr_bang(int argc, VALUE *argv, VALUE str)
9991{
9992 rb_check_arity(argc, 1, 2);
9993
9994 if (argc == 1) {
9995 VALUE pairs = argv[0];
9996 return tr_trans_pairs(str, pairs);
9997 }
9998
9999 VALUE src = argv[0], repl = argv[1];
10000 return tr_trans(str, src, repl, 0);
10001}
10002
10003
10004/*
10005 * call-seq:
10006 * tr(selector, replacements) -> new_string
10007 * tr(pairs) -> new_string
10008 *
10009 * Accepts either a +selector+ and a +replacements+ string,
10010 * or a single +pairs+ Hash.
10011 *
10012 * When a +pairs+ Hash is provided the keys, returns a copy of +self+ with
10013 * the keys of the hash replaced by the values.
10014 *
10015 * - They keys must be strings containing a single codepoints.
10016 * - The values can be of any length.
10017 *
10018 * Example:
10019 *
10020 * 'hello'.tr('e' => 'er', 'l' => '', 'o' => 'o !') #=> "hero !"
10021 *
10022 * When +selector+ and +replacements+are provided, returns a copy of +self+
10023 * with each character specified by string +selector+ translated to the
10024 * corresponding character in string +replacements+.
10025 * The correspondence is _positional_:
10026 *
10027 * - Each occurrence of the first character specified by +selector+
10028 * is translated to the first character in +replacements+.
10029 * - Each occurrence of the second character specified by +selector+
10030 * is translated to the second character in +replacements+.
10031 * - And so on.
10032 *
10033 * Example:
10034 *
10035 * 'hello'.tr('el', 'ip') #=> "hippo"
10036 *
10037 * If +replacements+ is shorter than +selector+,
10038 * it is implicitly padded with its own last character:
10039 *
10040 * 'hello'.tr('aeiou', '-') # => "h-ll-"
10041 * 'hello'.tr('aeiou', 'AA-') # => "hAll-"
10042 *
10043 * Arguments +selector+ and +replacements+ must be valid character selectors
10044 * (see {Character Selectors}[rdoc-ref:character_selectors.rdoc]),
10045 * and may use any of its valid forms, including negation, ranges, and escapes:
10046 *
10047 * 'hello'.tr('^aeiou', '-') # => "-e--o" # Negation.
10048 * 'ibm'.tr('b-z', 'a-z') # => "hal" # Range.
10049 * 'hel^lo'.tr('\^aeiou', '-') # => "h-l-l-" # Escaped leading caret.
10050 * 'i-b-m'.tr('b\-z', 'a-z') # => "ibabm" # Escaped embedded hyphen.
10051 * 'foo\\bar'.tr('ab\\', 'XYZ') # => "fooZYXr" # Escaped backslash.
10052 *
10053 * Related: see {Converting to New String}[rdoc-ref:String@Converting+to+New+String].
10054 */
10055
10056static VALUE
10057rb_str_tr(int argc, VALUE *argv, VALUE str)
10058{
10059 rb_check_arity(argc, 1, 2);
10060
10061 str = str_duplicate(rb_cString, str);
10062
10063 if (argc == 1) {
10064 VALUE pairs = argv[0];
10065 VALUE result = tr_trans_pairs(str, pairs);
10066 if (NIL_P(result)) result = str;
10067 return str;
10068 }
10069
10070 VALUE src = argv[0], repl = argv[1];
10071 tr_trans(str, src, repl, 0);
10072 return str;
10073}
10074
10075#define TR_TABLE_MAX (UCHAR_MAX+1)
10076#define TR_TABLE_SIZE (TR_TABLE_MAX+1)
10077static void
10078tr_setup_table(VALUE str, char stable[TR_TABLE_SIZE], int first,
10079 VALUE *tablep, VALUE *ctablep, rb_encoding *enc)
10080{
10081 const unsigned int errc = -1;
10082 char buf[TR_TABLE_MAX];
10083 struct tr tr;
10084 unsigned int c;
10085 VALUE table = 0, ptable = 0;
10086 int i, l, cflag = 0;
10087
10088 tr.p = RSTRING_PTR(str); tr.pend = tr.p + RSTRING_LEN(str);
10089 tr.gen = tr.now = tr.max = 0;
10090
10091 if (RSTRING_LEN(str) > 1 && rb_enc_ascget(tr.p, tr.pend, &l, enc) == '^') {
10092 cflag = 1;
10093 tr.p += l;
10094 }
10095 if (first) {
10096 for (i=0; i<TR_TABLE_MAX; i++) {
10097 stable[i] = 1;
10098 }
10099 stable[TR_TABLE_MAX] = cflag;
10100 }
10101 else if (stable[TR_TABLE_MAX] && !cflag) {
10102 stable[TR_TABLE_MAX] = 0;
10103 }
10104 for (i=0; i<TR_TABLE_MAX; i++) {
10105 buf[i] = cflag;
10106 }
10107
10108 while ((c = trnext(&tr, enc)) != errc) {
10109 if (c < TR_TABLE_MAX) {
10110 buf[(unsigned char)c] = !cflag;
10111 }
10112 else {
10113 VALUE key = UINT2NUM(c);
10114
10115 if (!table && (first || *tablep || stable[TR_TABLE_MAX])) {
10116 if (cflag) {
10117 ptable = *ctablep;
10118 table = ptable ? ptable : rb_hash_new();
10119 *ctablep = table;
10120 }
10121 else {
10122 table = rb_hash_new();
10123 ptable = *tablep;
10124 *tablep = table;
10125 }
10126 }
10127 if (table && (!ptable || (cflag ^ !NIL_P(rb_hash_aref(ptable, key))))) {
10128 rb_hash_aset(table, key, Qtrue);
10129 }
10130 }
10131 }
10132 for (i=0; i<TR_TABLE_MAX; i++) {
10133 stable[i] = stable[i] && buf[i];
10134 }
10135 if (!table && !cflag) {
10136 *tablep = 0;
10137 }
10138}
10139
10140
10141static int
10142tr_find(unsigned int c, const char table[TR_TABLE_SIZE], VALUE del, VALUE nodel)
10143{
10144 if (c < TR_TABLE_MAX) {
10145 return table[c] != 0;
10146 }
10147 else {
10148 VALUE v = UINT2NUM(c);
10149
10150 if (del) {
10151 if (!NIL_P(rb_hash_lookup(del, v)) &&
10152 (!nodel || NIL_P(rb_hash_lookup(nodel, v)))) {
10153 return TRUE;
10154 }
10155 }
10156 else if (nodel && !NIL_P(rb_hash_lookup(nodel, v))) {
10157 return FALSE;
10158 }
10159 return table[TR_TABLE_MAX] ? TRUE : FALSE;
10160 }
10161}
10162
10163/*
10164 * call-seq:
10165 * delete!(*selectors) -> self or nil
10166 *
10167 * Like String#delete, but modifies +self+ in place;
10168 * returns +self+ if any characters were deleted, +nil+ otherwise.
10169 *
10170 * Related: see {Modifying}[rdoc-ref:String@Modifying].
10171 */
10172
10173static VALUE
10174rb_str_delete_bang(int argc, VALUE *argv, VALUE str)
10175{
10176 char squeez[TR_TABLE_SIZE];
10177 rb_encoding *enc = 0;
10178 char *s, *send, *t;
10179 VALUE del = 0, nodel = 0;
10180 int modify = 0;
10181 int i, ascompat, cr;
10182
10183 if (RSTRING_LEN(str) == 0 || !RSTRING_PTR(str)) return Qnil;
10185 for (i=0; i<argc; i++) {
10186 VALUE s = argv[i];
10187
10188 StringValue(s);
10189 enc = rb_enc_check(str, s);
10190 tr_setup_table(s, squeez, i==0, &del, &nodel, enc);
10191 }
10192
10193 str_modify_keep_cr(str);
10194 ascompat = rb_enc_asciicompat(enc);
10195 s = t = RSTRING_PTR(str);
10196 send = RSTRING_END(str);
10197 cr = ascompat ? ENC_CODERANGE_7BIT : ENC_CODERANGE_VALID;
10198 while (s < send) {
10199 unsigned int c;
10200 int clen;
10201
10202 if (ascompat && (c = *(unsigned char*)s) < 0x80) {
10203 if (squeez[c]) {
10204 modify = 1;
10205 }
10206 else {
10207 if (t != s) *t = c;
10208 t++;
10209 }
10210 s++;
10211 }
10212 else {
10213 c = rb_enc_codepoint_len(s, send, &clen, enc);
10214
10215 if (tr_find(c, squeez, del, nodel)) {
10216 modify = 1;
10217 }
10218 else {
10219 if (t != s) rb_enc_mbcput(c, t, enc);
10220 t += clen;
10222 }
10223 s += clen;
10224 }
10225 }
10226 TERM_FILL(t, TERM_LEN(str));
10227 STR_SET_LEN(str, t - RSTRING_PTR(str));
10228 ENC_CODERANGE_SET(str, cr);
10229
10230 if (modify) return str;
10231 return Qnil;
10232}
10233
10234
10235/*
10236 * call-seq:
10237 * delete(*selectors) -> new_string
10238 *
10239 * :include: doc/string/delete.rdoc
10240 *
10241 */
10242
10243static VALUE
10244rb_str_delete(int argc, VALUE *argv, VALUE str)
10245{
10246 str = str_duplicate(rb_cString, str);
10247 rb_str_delete_bang(argc, argv, str);
10248 return str;
10249}
10250
10251
10252/*
10253 * call-seq:
10254 * squeeze!(*selectors) -> self or nil
10255 *
10256 * Like String#squeeze, except that:
10257 *
10258 * - Characters are squeezed in +self+ (not in a copy of +self+).
10259 * - Returns +self+ if any changes are made, +nil+ otherwise.
10260 *
10261 * Related: See {Modifying}[rdoc-ref:String@Modifying].
10262 */
10263
10264static VALUE
10265rb_str_squeeze_bang(int argc, VALUE *argv, VALUE str)
10266{
10267 char squeez[TR_TABLE_SIZE];
10268 rb_encoding *enc = 0;
10269 VALUE del = 0, nodel = 0;
10270 unsigned char *s, *send, *t;
10271 int i, modify = 0;
10272 int ascompat, singlebyte = single_byte_optimizable(str);
10273 unsigned int save;
10274
10275 if (argc == 0) {
10276 enc = STR_ENC_GET(str);
10277 }
10278 else {
10279 for (i=0; i<argc; i++) {
10280 VALUE s = argv[i];
10281
10282 StringValue(s);
10283 enc = rb_enc_check(str, s);
10284 if (singlebyte && !single_byte_optimizable(s))
10285 singlebyte = 0;
10286 tr_setup_table(s, squeez, i==0, &del, &nodel, enc);
10287 }
10288 }
10289
10290 str_modify_keep_cr(str);
10291 s = t = (unsigned char *)RSTRING_PTR(str);
10292 if (!s || RSTRING_LEN(str) == 0) return Qnil;
10293 send = (unsigned char *)RSTRING_END(str);
10294 save = -1;
10295 ascompat = rb_enc_asciicompat(enc);
10296
10297 if (singlebyte) {
10298 while (s < send) {
10299 unsigned int c = *s++;
10300 if (c != save || (argc > 0 && !squeez[c])) {
10301 *t++ = save = c;
10302 }
10303 }
10304 }
10305 else {
10306 while (s < send) {
10307 unsigned int c;
10308 int clen;
10309
10310 if (ascompat && (c = *s) < 0x80) {
10311 if (c != save || (argc > 0 && !squeez[c])) {
10312 *t++ = save = c;
10313 }
10314 s++;
10315 }
10316 else {
10317 c = rb_enc_codepoint_len((char *)s, (char *)send, &clen, enc);
10318
10319 if (c != save || (argc > 0 && !tr_find(c, squeez, del, nodel))) {
10320 if (t != s) rb_enc_mbcput(c, t, enc);
10321 save = c;
10322 t += clen;
10323 }
10324 s += clen;
10325 }
10326 }
10327 }
10328
10329 TERM_FILL((char *)t, TERM_LEN(str));
10330 if ((char *)t - RSTRING_PTR(str) != RSTRING_LEN(str)) {
10331 STR_SET_LEN(str, (char *)t - RSTRING_PTR(str));
10332 modify = 1;
10333 }
10334
10335 if (modify) return str;
10336 return Qnil;
10337}
10338
10339
10340/*
10341 * call-seq:
10342 * squeeze(*selectors) -> new_string
10343 *
10344 * :include: doc/string/squeeze.rdoc
10345 *
10346 */
10347
10348static VALUE
10349rb_str_squeeze(int argc, VALUE *argv, VALUE str)
10350{
10351 str = str_duplicate(rb_cString, str);
10352 rb_str_squeeze_bang(argc, argv, str);
10353 return str;
10354}
10355
10356
10357/*
10358 * call-seq:
10359 * tr_s!(selector, replacements) -> self or nil
10360 *
10361 * Like String#tr_s, except:
10362 *
10363 * - Modifies +self+ in place (not a copy of +self+).
10364 * - Returns +self+ if any changes were made, +nil+ otherwise.
10365 *
10366 * Related: {Modifying}[rdoc-ref:String@Modifying].
10367 */
10368
10369static VALUE
10370rb_str_tr_s_bang(VALUE str, VALUE src, VALUE repl)
10371{
10372 return tr_trans(str, src, repl, 1);
10373}
10374
10375
10376/*
10377 * call-seq:
10378 * tr_s(selector, replacements) -> new_string
10379 *
10380 * Like String#tr, except:
10381 *
10382 * - Also squeezes the modified portions of the translated string;
10383 * see String#squeeze.
10384 * - Returns the translated and squeezed string.
10385 *
10386 * Examples:
10387 *
10388 * 'hello'.tr_s('l', 'r') #=> "hero"
10389 * 'hello'.tr_s('el', '-') #=> "h-o"
10390 * 'hello'.tr_s('el', 'hx') #=> "hhxo"
10391 *
10392 * Related: see {Converting to New String}[rdoc-ref:String@Converting+to+New+String].
10393 *
10394 */
10395
10396static VALUE
10397rb_str_tr_s(VALUE str, VALUE src, VALUE repl)
10398{
10399 str = str_duplicate(rb_cString, str);
10400 tr_trans(str, src, repl, 1);
10401 return str;
10402}
10403
10404
10405/*
10406 * call-seq:
10407 * count(*selectors) -> integer
10408 *
10409 * :include: doc/string/count.rdoc
10410 */
10411
10412static VALUE
10413rb_str_count(int argc, VALUE *argv, VALUE str)
10414{
10415 char table[TR_TABLE_SIZE];
10416 rb_encoding *enc = 0;
10417 VALUE del = 0, nodel = 0, tstr;
10418 const char *s, *send;
10419 int i;
10420 int ascompat;
10421 size_t n = 0;
10422
10424
10425 tstr = argv[0];
10426 StringValue(tstr);
10427 enc = rb_enc_check(str, tstr);
10428 if (argc == 1) {
10429 const char *ptstr;
10430 if (RSTRING_LEN(tstr) == 1 && rb_enc_asciicompat(enc) &&
10431 (ptstr = RSTRING_PTR(tstr),
10432 ONIGENC_IS_ALLOWED_REVERSE_MATCH(enc, (const unsigned char *)ptstr, (const unsigned char *)ptstr+1)) &&
10433 !is_broken_string(str)) {
10434 int clen;
10435 unsigned char c = rb_enc_codepoint_len(ptstr, ptstr+1, &clen, enc);
10436
10437 s = RSTRING_PTR(str);
10438 if (!s || RSTRING_LEN(str) == 0) return INT2FIX(0);
10439 send = RSTRING_END(str);
10440 while (s < send) {
10441 if (*(unsigned char*)s++ == c) n++;
10442 }
10443 return SIZET2NUM(n);
10444 }
10445 }
10446
10447 tr_setup_table(tstr, table, TRUE, &del, &nodel, enc);
10448 for (i=1; i<argc; i++) {
10449 tstr = argv[i];
10450 StringValue(tstr);
10451 enc = rb_enc_check(str, tstr);
10452 tr_setup_table(tstr, table, FALSE, &del, &nodel, enc);
10453 }
10454
10455 s = RSTRING_PTR(str);
10456 if (!s || RSTRING_LEN(str) == 0) return INT2FIX(0);
10457 send = RSTRING_END(str);
10458 ascompat = rb_enc_asciicompat(enc);
10459 while (s < send) {
10460 unsigned int c;
10461
10462 if (ascompat && (c = *(unsigned char*)s) < 0x80) {
10463 if (table[c]) {
10464 n++;
10465 }
10466 s++;
10467 }
10468 else {
10469 int clen;
10470 c = rb_enc_codepoint_len(s, send, &clen, enc);
10471 if (tr_find(c, table, del, nodel)) {
10472 n++;
10473 }
10474 s += clen;
10475 }
10476 }
10477
10478 return SIZET2NUM(n);
10479}
10480
10481static VALUE
10482rb_fs_check(VALUE val)
10483{
10484 if (!NIL_P(val) && !RB_TYPE_P(val, T_STRING) && !RB_TYPE_P(val, T_REGEXP)) {
10485 val = rb_check_string_type(val);
10486 if (NIL_P(val)) return 0;
10487 }
10488 return val;
10489}
10490
10491static const char isspacetable[256] = {
10492 0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 0, 0,
10493 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10494 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10495 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10496 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10497 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10498 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10499 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10500 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10501 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10502 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10503 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10504 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10505 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10506 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10507 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0
10508};
10509
10510#define ascii_isspace(c) isspacetable[(unsigned char)(c)]
10511
10512static long
10513split_string(VALUE result, VALUE str, long beg, long len, long empty_count)
10514{
10515 if (empty_count >= 0 && len == 0) {
10516 return empty_count + 1;
10517 }
10518 if (empty_count > 0) {
10519 /* make different substrings */
10520 if (result) {
10521 do {
10522 rb_ary_push(result, str_new_empty_String(str));
10523 } while (--empty_count > 0);
10524 }
10525 else {
10526 do {
10527 rb_yield(str_new_empty_String(str));
10528 } while (--empty_count > 0);
10529 }
10530 }
10531 str = rb_str_subseq(str, beg, len);
10532 if (result) {
10533 rb_ary_push(result, str);
10534 }
10535 else {
10536 rb_yield(str);
10537 }
10538 return empty_count;
10539}
10540
10541typedef enum {
10542 SPLIT_TYPE_AWK, SPLIT_TYPE_STRING, SPLIT_TYPE_REGEXP, SPLIT_TYPE_CHARS
10543} split_type_t;
10544
10545static split_type_t
10546literal_split_pattern(VALUE spat, split_type_t default_type)
10547{
10548 rb_encoding *enc = STR_ENC_GET(spat);
10549 const char *ptr;
10550 long len;
10551 RSTRING_GETMEM(spat, ptr, len);
10552 if (len == 0) {
10553 /* Special case - split into chars */
10554 return SPLIT_TYPE_CHARS;
10555 }
10556 else if (rb_enc_asciicompat(enc)) {
10557 if (len == 1 && ptr[0] == ' ') {
10558 return SPLIT_TYPE_AWK;
10559 }
10560 }
10561 else {
10562 int l;
10563 if (rb_enc_ascget(ptr, ptr + len, &l, enc) == ' ' && len == l) {
10564 return SPLIT_TYPE_AWK;
10565 }
10566 }
10567 return default_type;
10568}
10569
10570/*
10571 * call-seq:
10572 * split(field_sep = $;, limit = 0) -> array_of_substrings
10573 * split(field_sep = $;, limit = 0) {|substring| ... } -> self
10574 *
10575 * :include: doc/string/split.rdoc
10576 *
10577 */
10578
10579static VALUE
10580rb_str_split_m(int argc, VALUE *argv, VALUE str)
10581{
10582 rb_encoding *enc;
10583 VALUE spat;
10584 VALUE limit;
10585 split_type_t split_type;
10586 long beg, end, i = 0, empty_count = -1;
10587 int lim = 0;
10588 VALUE result, tmp;
10589
10590 result = rb_block_given_p() ? Qfalse : Qnil;
10591 if (rb_scan_args(argc, argv, "02", &spat, &limit) == 2) {
10592 lim = NUM2INT(limit);
10593 if (lim <= 0) limit = Qnil;
10594 else if (lim == 1) {
10595 if (RSTRING_LEN(str) == 0)
10596 return result ? rb_ary_new2(0) : str;
10597 tmp = str_duplicate(rb_cString, str);
10598 if (!result) {
10599 rb_yield(tmp);
10600 return str;
10601 }
10602 return rb_ary_new3(1, tmp);
10603 }
10604 i = 1;
10605 }
10606 if (NIL_P(limit) && !lim) empty_count = 0;
10607
10608 enc = STR_ENC_GET(str);
10609 split_type = SPLIT_TYPE_REGEXP;
10610 if (!NIL_P(spat)) {
10611 spat = get_pat_quoted(spat, 0);
10612 }
10613 else if (NIL_P(spat = rb_fs)) {
10614 split_type = SPLIT_TYPE_AWK;
10615 }
10616 else if (!(spat = rb_fs_check(spat))) {
10617 rb_raise(rb_eTypeError, "value of $; must be String or Regexp");
10618 }
10619 else {
10620 rb_category_warn(RB_WARN_CATEGORY_DEPRECATED, "$; is set to non-nil value");
10621 }
10622 if (split_type != SPLIT_TYPE_AWK) {
10623 switch (BUILTIN_TYPE(spat)) {
10624 case T_REGEXP:
10625 rb_reg_options(spat); /* check if uninitialized */
10626 tmp = RREGEXP_SRC(spat);
10627 split_type = literal_split_pattern(tmp, SPLIT_TYPE_REGEXP);
10628 if (split_type == SPLIT_TYPE_AWK) {
10629 spat = tmp;
10630 split_type = SPLIT_TYPE_STRING;
10631 }
10632 break;
10633
10634 case T_STRING:
10635 mustnot_broken(spat);
10636 split_type = literal_split_pattern(spat, SPLIT_TYPE_STRING);
10637 break;
10638
10639 default:
10641 }
10642 }
10643
10644#define SPLIT_STR(beg, len) ( \
10645 empty_count = split_string(result, str, beg, len, empty_count), \
10646 str_mod_check(str, str_start, str_len))
10647
10648 beg = 0;
10649 const char *ptr = RSTRING_PTR(str);
10650 const char *const str_start = ptr;
10651 const long str_len = RSTRING_LEN(str);
10652 const char *const eptr = str_start + str_len;
10653 if (split_type == SPLIT_TYPE_AWK) {
10654 const char *bptr = ptr;
10655 int skip = 1;
10656 unsigned int c;
10657
10658 if (result) result = rb_ary_new();
10659 end = beg;
10660 if (is_ascii_string(str)) {
10661 while (ptr < eptr) {
10662 c = (unsigned char)*ptr++;
10663 if (skip) {
10664 if (ascii_isspace(c)) {
10665 beg = ptr - bptr;
10666 }
10667 else {
10668 end = ptr - bptr;
10669 skip = 0;
10670 if (!NIL_P(limit) && lim <= i) break;
10671 }
10672 }
10673 else if (ascii_isspace(c)) {
10674 SPLIT_STR(beg, end-beg);
10675 skip = 1;
10676 beg = ptr - bptr;
10677 if (!NIL_P(limit)) ++i;
10678 }
10679 else {
10680 end = ptr - bptr;
10681 }
10682 }
10683 }
10684 else {
10685 while (ptr < eptr) {
10686 int n;
10687
10688 c = rb_enc_codepoint_len(ptr, eptr, &n, enc);
10689 ptr += n;
10690 if (skip) {
10691 if (rb_isspace(c)) {
10692 beg = ptr - bptr;
10693 }
10694 else {
10695 end = ptr - bptr;
10696 skip = 0;
10697 if (!NIL_P(limit) && lim <= i) break;
10698 }
10699 }
10700 else if (rb_isspace(c)) {
10701 SPLIT_STR(beg, end-beg);
10702 skip = 1;
10703 beg = ptr - bptr;
10704 if (!NIL_P(limit)) ++i;
10705 }
10706 else {
10707 end = ptr - bptr;
10708 }
10709 }
10710 }
10711 }
10712 else if (split_type == SPLIT_TYPE_STRING) {
10713 const char *substr_start = ptr;
10714 const char *sptr = RSTRING_PTR(spat);
10715 long slen = RSTRING_LEN(spat);
10716
10717 if (result) result = rb_ary_new();
10718 mustnot_broken(str);
10719 enc = rb_enc_check(str, spat);
10720 while (ptr < eptr &&
10721 (end = rb_memsearch(sptr, slen, ptr, eptr - ptr, enc)) >= 0) {
10722 /* Check we are at the start of a char */
10723 const char *t = rb_enc_right_char_head(ptr, ptr + end, eptr, enc);
10724 if (t != ptr + end) {
10725 ptr = t;
10726 continue;
10727 }
10728 SPLIT_STR(substr_start - str_start, (ptr+end) - substr_start);
10729 str_mod_check(spat, sptr, slen);
10730 ptr += end + slen;
10731 substr_start = ptr;
10732 if (!NIL_P(limit) && lim <= ++i) break;
10733 }
10734 beg = ptr - str_start;
10735 }
10736 else if (split_type == SPLIT_TYPE_CHARS) {
10737 int n;
10738
10739 if (result) result = rb_ary_new_capa(RSTRING_LEN(str));
10740 mustnot_broken(str);
10741 enc = rb_enc_get(str);
10742 while (ptr < eptr &&
10743 (n = rb_enc_precise_mbclen(ptr, eptr, enc)) > 0) {
10744 SPLIT_STR(ptr - str_start, n);
10745 ptr += n;
10746 if (!NIL_P(limit) && lim <= ++i) break;
10747 }
10748 beg = ptr - str_start;
10749 }
10750 else {
10751 if (result) result = rb_ary_new();
10752 long len = RSTRING_LEN(str);
10753 long start = beg;
10754 int idx;
10755 int last_null = 0;
10756 VALUE match = 0;
10757
10758 for (; rb_reg_search(spat, str, start, 0) >= 0;
10759 (match ? (rb_match_unbusy(match), rb_backref_set(match)) : (void)0)) {
10760 match = rb_backref_get();
10761 if (!result) rb_match_busy(match);
10762 end = RMATCH_BEG(match, 0);
10763 if (start == end && RMATCH_BEG(match, 0) == RMATCH_END(match, 0)) {
10764 if (!ptr) {
10765 SPLIT_STR(0, 0);
10766 break;
10767 }
10768 else if (last_null == 1) {
10769 SPLIT_STR(beg, rb_enc_fast_mbclen(ptr+beg, eptr, enc));
10770 beg = start;
10771 }
10772 else {
10773 if (start == len)
10774 start++;
10775 else
10776 start += rb_enc_fast_mbclen(ptr+start,eptr,enc);
10777 last_null = 1;
10778 continue;
10779 }
10780 }
10781 else {
10782 SPLIT_STR(beg, end-beg);
10783 beg = start = RMATCH_END(match, 0);
10784 }
10785 last_null = 0;
10786
10787 for (idx = 1; idx < RMATCH_NREGS(match); idx++) {
10788 if (RMATCH_BEG(match, idx) == -1) continue;
10789 SPLIT_STR(RMATCH_BEG(match, idx), RMATCH_END(match, idx) - RMATCH_BEG(match, idx));
10790 }
10791 if (!NIL_P(limit) && lim <= ++i) break;
10792 }
10793 if (match) rb_match_unbusy(match);
10794 }
10795 if (RSTRING_LEN(str) > 0 && (!NIL_P(limit) || RSTRING_LEN(str) > beg || lim < 0)) {
10796 SPLIT_STR(beg, RSTRING_LEN(str)-beg);
10797 }
10798
10799 return result ? result : str;
10800}
10801
10802VALUE
10803rb_str_split(VALUE str, const char *sep0)
10804{
10805 VALUE sep;
10806
10807 StringValue(str);
10808 sep = rb_str_new_cstr(sep0);
10809 return rb_str_split_m(1, &sep, str);
10810}
10811
10812#define WANTARRAY(m, size) (!rb_block_given_p() ? rb_ary_new_capa(size) : 0)
10813
10814static inline int
10815enumerator_element(VALUE ary, VALUE e)
10816{
10817 if (ary) {
10818 rb_ary_push(ary, e);
10819 return 0;
10820 }
10821 else {
10822 rb_yield(e);
10823 return 1;
10824 }
10825}
10826
10827#define ENUM_ELEM(ary, e) enumerator_element(ary, e)
10828
10829static const char *
10830chomp_newline(const char *p, const char *e, rb_encoding *enc)
10831{
10832 const char *prev = rb_enc_prev_char(p, e, e, enc);
10833 if (rb_enc_is_newline(prev, e, enc)) {
10834 e = prev;
10835 prev = rb_enc_prev_char(p, e, e, enc);
10836 if (prev && rb_enc_ascget(prev, e, NULL, enc) == '\r')
10837 e = prev;
10838 }
10839 return e;
10840}
10841
10842static VALUE
10843get_rs(void)
10844{
10845 VALUE rs = rb_rs;
10846 if (!NIL_P(rs) &&
10847 (!RB_TYPE_P(rs, T_STRING) ||
10848 RSTRING_LEN(rs) != 1 ||
10849 RSTRING_PTR(rs)[0] != '\n')) {
10850 rb_category_warn(RB_WARN_CATEGORY_DEPRECATED, "$/ is set to non-default value");
10851 }
10852 return rs;
10853}
10854
10855#define rb_rs get_rs()
10856
10857static VALUE
10858rb_str_enumerate_lines(int argc, VALUE *argv, VALUE str, VALUE ary)
10859{
10860 rb_encoding *enc;
10861 VALUE line, rs, orig = str, opts = Qnil, chomp = Qfalse;
10862 const char *pend, *subptr, *subend, *rsptr, *hit, *adjusted;
10863 long pos, rslen;
10864 int rsnewline = 0;
10865
10866 if (rb_scan_args(argc, argv, "01:", &rs, &opts) == 0)
10867 rs = rb_rs;
10868 if (!NIL_P(opts)) {
10869 static ID keywords[1];
10870 if (!keywords[0]) {
10871 keywords[0] = rb_intern_const("chomp");
10872 }
10873 rb_get_kwargs(opts, keywords, 0, 1, &chomp);
10874 chomp = (!UNDEF_P(chomp) && RTEST(chomp));
10875 }
10876
10877 if (NIL_P(rs)) {
10878 if (!ENUM_ELEM(ary, str)) {
10879 return ary;
10880 }
10881 else {
10882 return orig;
10883 }
10884 }
10885
10886 if (!RSTRING_LEN(str)) goto end;
10887 str = rb_str_new_frozen(str);
10888 const char *const ptr = subptr = RSTRING_PTR(str);
10889 const long len = RSTRING_LEN(str);
10890 pend = RSTRING_END(str);
10891 StringValue(rs);
10892 rslen = RSTRING_LEN(rs);
10893
10894 if (rs == rb_default_rs)
10895 enc = rb_enc_get(str);
10896 else
10897 enc = rb_enc_check(str, rs);
10898
10899 if (rslen == 0) {
10900 /* paragraph mode */
10901 int n;
10902 const char *eol = NULL;
10903 subend = subptr;
10904 while (subend < pend) {
10905 long chomp_rslen = 0;
10906 do {
10907 if (rb_enc_ascget(subend, pend, &n, enc) != '\r')
10908 n = 0;
10909 rslen = n + rb_enc_mbclen(subend + n, pend, enc);
10910 if (rb_enc_is_newline(subend + n, pend, enc)) {
10911 if (eol == subend) break;
10912 subend += rslen;
10913 if (subptr) {
10914 eol = subend;
10915 chomp_rslen = -rslen;
10916 }
10917 }
10918 else {
10919 if (!subptr) subptr = subend;
10920 subend += rslen;
10921 }
10922 rslen = 0;
10923 } while (subend < pend);
10924 if (!subptr) break;
10925 if (rslen == 0) chomp_rslen = 0;
10926 line = rb_str_subseq(str, subptr - ptr,
10927 subend - subptr + (chomp ? chomp_rslen : rslen));
10928 if (ENUM_ELEM(ary, line)) {
10929 str_mod_check(str, ptr, len);
10930 }
10931 subptr = eol = NULL;
10932 }
10933 goto end;
10934 }
10935 else {
10936 rsptr = RSTRING_PTR(rs);
10937 if (RSTRING_LEN(rs) == rb_enc_mbminlen(enc) &&
10938 rb_enc_is_newline(rsptr, rsptr + RSTRING_LEN(rs), enc)) {
10939 rsnewline = 1;
10940 }
10941 }
10942
10943 if ((rs == rb_default_rs) && !rb_enc_asciicompat(enc)) {
10944 rs = rb_str_new(rsptr, rslen);
10945 rs = rb_str_encode(rs, rb_enc_from_encoding(enc), 0, Qnil);
10946 rsptr = RSTRING_PTR(rs);
10947 rslen = RSTRING_LEN(rs);
10948 }
10949
10950 while (subptr < pend) {
10951 pos = rb_memsearch(rsptr, rslen, subptr, pend - subptr, enc);
10952 if (pos < 0) break;
10953 hit = subptr + pos;
10954 adjusted = rb_enc_right_char_head(subptr, hit, pend, enc);
10955 if (hit != adjusted) {
10956 subptr = adjusted;
10957 continue;
10958 }
10959 subend = hit += rslen;
10960 if (chomp) {
10961 if (rsnewline) {
10962 subend = chomp_newline(subptr, subend, enc);
10963 }
10964 else {
10965 subend -= rslen;
10966 }
10967 }
10968 line = rb_str_subseq(str, subptr - ptr, subend - subptr);
10969 if (ENUM_ELEM(ary, line)) {
10970 str_mod_check(str, ptr, len);
10971 str_mod_check(rs, rsptr, rslen);
10972 }
10973 subptr = hit;
10974 }
10975
10976 if (subptr < pend) {
10977 if (chomp) {
10978 if (rsnewline) {
10979 pend = chomp_newline(subptr, pend, enc);
10980 }
10981 else if (pend - subptr >= rslen &&
10982 memcmp(pend - rslen, rsptr, rslen) == 0) {
10983 pend -= rslen;
10984 }
10985 }
10986 line = rb_str_subseq(str, subptr - ptr, pend - subptr);
10987 ENUM_ELEM(ary, line);
10988 RB_GC_GUARD(str);
10989 }
10990
10991 end:
10992 if (ary)
10993 return ary;
10994 else
10995 return orig;
10996}
10997
10998/*
10999 * call-seq:
11000 * each_line(record_separator = $/, chomp: false) {|substring| ... } -> self
11001 * each_line(record_separator = $/, chomp: false) -> enumerator
11002 *
11003 * :include: doc/string/each_line.rdoc
11004 *
11005 */
11006
11007static VALUE
11008rb_str_each_line(int argc, VALUE *argv, VALUE str)
11009{
11010 RETURN_SIZED_ENUMERATOR(str, argc, argv, 0);
11011 return rb_str_enumerate_lines(argc, argv, str, 0);
11012}
11013
11014/*
11015 * call-seq:
11016 * lines(record_separator = $/, chomp: false) -> array_of_strings
11017 *
11018 * Returns substrings ("lines") of +self+
11019 * according to the given arguments:
11020 *
11021 * s = <<~EOT
11022 * This is the first line.
11023 * This is line two.
11024 *
11025 * This is line four.
11026 * This is line five.
11027 * EOT
11028 *
11029 * With the default argument values:
11030 *
11031 * $/ # => "\n"
11032 * s.lines
11033 * # =>
11034 * ["This is the first line.\n",
11035 * "This is line two.\n",
11036 * "\n",
11037 * "This is line four.\n",
11038 * "This is line five.\n"]
11039 *
11040 * With a different +record_separator+:
11041 *
11042 * record_separator = ' is '
11043 * s.lines(record_separator)
11044 * # =>
11045 * ["This is ",
11046 * "the first line.\nThis is ",
11047 * "line two.\n\nThis is ",
11048 * "line four.\nThis is ",
11049 * "line five.\n"]
11050 *
11051 * With keyword argument +chomp+ as +true+,
11052 * removes the trailing newline from each line:
11053 *
11054 * s.lines(chomp: true)
11055 * # =>
11056 * ["This is the first line.",
11057 * "This is line two.",
11058 * "",
11059 * "This is line four.",
11060 * "This is line five."]
11061 *
11062 * Related: see {Converting to Non-String}[rdoc-ref:String@Converting+to+Non-String].
11063 */
11064
11065static VALUE
11066rb_str_lines(int argc, VALUE *argv, VALUE str)
11067{
11068 VALUE ary = WANTARRAY("lines", 0);
11069 return rb_str_enumerate_lines(argc, argv, str, ary);
11070}
11071
11072static VALUE
11073rb_str_each_byte_size(VALUE str, VALUE args, VALUE eobj)
11074{
11075 return LONG2FIX(RSTRING_LEN(str));
11076}
11077
11078static VALUE
11079rb_str_enumerate_bytes(VALUE str, VALUE ary)
11080{
11081 long i;
11082
11083 for (i=0; i<RSTRING_LEN(str); i++) {
11084 ENUM_ELEM(ary, INT2FIX((unsigned char)RSTRING_PTR(str)[i]));
11085 }
11086 if (ary)
11087 return ary;
11088 else
11089 return str;
11090}
11091
11092/*
11093 * call-seq:
11094 * each_byte {|byte| ... } -> self
11095 * each_byte -> enumerator
11096 *
11097 * :include: doc/string/each_byte.rdoc
11098 *
11099 */
11100
11101static VALUE
11102rb_str_each_byte(VALUE str)
11103{
11104 RETURN_SIZED_ENUMERATOR(str, 0, 0, rb_str_each_byte_size);
11105 return rb_str_enumerate_bytes(str, 0);
11106}
11107
11108/*
11109 * call-seq:
11110 * bytes -> array_of_bytes
11111 *
11112 * :include: doc/string/bytes.rdoc
11113 *
11114 */
11115
11116static VALUE
11117rb_str_bytes(VALUE str)
11118{
11119 VALUE ary = WANTARRAY("bytes", RSTRING_LEN(str));
11120 return rb_str_enumerate_bytes(str, ary);
11121}
11122
11123static VALUE
11124rb_str_each_char_size(VALUE str, VALUE args, VALUE eobj)
11125{
11126 return rb_str_length(str);
11127}
11128
11129static VALUE
11130rb_str_enumerate_chars(VALUE str, VALUE ary)
11131{
11132 VALUE orig = str;
11133 long i, len, n;
11134 const char *ptr;
11135 rb_encoding *enc;
11136
11137 str = rb_str_new_frozen(str);
11138 ptr = RSTRING_PTR(str);
11139 len = RSTRING_LEN(str);
11140 enc = rb_enc_get(str);
11141
11143 for (i = 0; i < len; i += n) {
11144 n = rb_enc_fast_mbclen(ptr + i, ptr + len, enc);
11145 ENUM_ELEM(ary, rb_str_subseq(str, i, n));
11146 }
11147 }
11148 else {
11149 for (i = 0; i < len; i += n) {
11150 n = rb_enc_mbclen(ptr + i, ptr + len, enc);
11151 ENUM_ELEM(ary, rb_str_subseq(str, i, n));
11152 }
11153 }
11154 RB_GC_GUARD(str);
11155 if (ary)
11156 return ary;
11157 else
11158 return orig;
11159}
11160
11161/*
11162 * call-seq:
11163 * each_char {|char| ... } -> self
11164 * each_char -> enumerator
11165 *
11166 * :include: doc/string/each_char.rdoc
11167 *
11168 */
11169
11170static VALUE
11171rb_str_each_char(VALUE str)
11172{
11173 RETURN_SIZED_ENUMERATOR(str, 0, 0, rb_str_each_char_size);
11174 return rb_str_enumerate_chars(str, 0);
11175}
11176
11177/*
11178 * call-seq:
11179 * chars -> array_of_characters
11180 *
11181 * :include: doc/string/chars.rdoc
11182 *
11183 */
11184
11185static VALUE
11186rb_str_chars(VALUE str)
11187{
11188 VALUE ary = WANTARRAY("chars", rb_str_strlen(str));
11189 return rb_str_enumerate_chars(str, ary);
11190}
11191
11192static VALUE
11193rb_str_enumerate_codepoints(VALUE str, VALUE ary)
11194{
11195 VALUE orig = str;
11196 int n;
11197 unsigned int c;
11198 const char *ptr, *end;
11199 rb_encoding *enc;
11200 int enc_asciicompat;
11201
11202 if (single_byte_optimizable(str))
11203 return rb_str_enumerate_bytes(str, ary);
11204
11205 str = rb_str_new_frozen(str);
11206 ptr = RSTRING_PTR(str);
11207 end = RSTRING_END(str);
11208 enc = STR_ENC_GET(str);
11209 enc_asciicompat = rb_enc_asciicompat(enc);
11210
11211 while (ptr < end) {
11212 /* Fast path: ASCII byte in an ASCII-compatible encoding is its own codepoint;
11213 * skip rb_enc_codepoint_len and return the byte directly.
11214 */
11215 n = 1;
11216 c = (enc_asciicompat && ISASCII(*ptr)) ?
11217 (unsigned char)*ptr : rb_enc_codepoint_len(ptr, end, &n, enc);
11218 ENUM_ELEM(ary, UINT2NUM(c));
11219 ptr += n;
11220 }
11221 RB_GC_GUARD(str);
11222 if (ary)
11223 return ary;
11224 else
11225 return orig;
11226}
11227
11228/*
11229 * call-seq:
11230 * each_codepoint {|codepoint| ... } -> self
11231 * each_codepoint -> enumerator
11232 *
11233 * :include: doc/string/each_codepoint.rdoc
11234 *
11235 */
11236
11237static VALUE
11238rb_str_each_codepoint(VALUE str)
11239{
11240 RETURN_SIZED_ENUMERATOR(str, 0, 0, rb_str_each_char_size);
11241 return rb_str_enumerate_codepoints(str, 0);
11242}
11243
11244/*
11245 * call-seq:
11246 * codepoints -> array_of_integers
11247 *
11248 * :include: doc/string/codepoints.rdoc
11249 *
11250 */
11251
11252static VALUE
11253rb_str_codepoints(VALUE str)
11254{
11255 VALUE ary = WANTARRAY("codepoints", rb_str_strlen(str));
11256 return rb_str_enumerate_codepoints(str, ary);
11257}
11258
11259static regex_t *
11260get_reg_grapheme_cluster(rb_encoding *enc)
11261{
11262 int encidx = rb_enc_to_index(enc);
11263
11264 const OnigUChar source_ascii[] = "\\X";
11265 const OnigUChar *source = source_ascii;
11266 size_t source_len = sizeof(source_ascii) - 1;
11267
11268 switch (encidx) {
11269#define CHARS_16BE(x) (OnigUChar)((x)>>8), (OnigUChar)(x)
11270#define CHARS_16LE(x) (OnigUChar)(x), (OnigUChar)((x)>>8)
11271#define CHARS_32BE(x) CHARS_16BE((x)>>16), CHARS_16BE(x)
11272#define CHARS_32LE(x) CHARS_16LE(x), CHARS_16LE((x)>>16)
11273#define CASE_UTF(e) \
11274 case ENCINDEX_UTF_##e: { \
11275 static const OnigUChar source_UTF_##e[] = {CHARS_##e('\\'), CHARS_##e('X')}; \
11276 source = source_UTF_##e; \
11277 source_len = sizeof(source_UTF_##e); \
11278 break; \
11279 }
11280 CASE_UTF(16BE); CASE_UTF(16LE); CASE_UTF(32BE); CASE_UTF(32LE);
11281#undef CASE_UTF
11282#undef CHARS_16BE
11283#undef CHARS_16LE
11284#undef CHARS_32BE
11285#undef CHARS_32LE
11286 }
11287
11288 regex_t *reg_grapheme_cluster;
11289 OnigErrorInfo einfo;
11290 int r = onig_new(&reg_grapheme_cluster, source, source + source_len,
11291 ONIG_OPTION_DEFAULT, enc, OnigDefaultSyntax, &einfo);
11292 if (r) {
11293 UChar message[ONIG_MAX_ERROR_MESSAGE_LEN];
11294 onig_error_code_to_str(message, r, &einfo);
11295 rb_fatal("cannot compile grapheme cluster regexp: %s", (char *)message);
11296 }
11297
11298 return reg_grapheme_cluster;
11299}
11300
11301static regex_t *
11302get_cached_reg_grapheme_cluster(rb_encoding *enc)
11303{
11304 int encidx = rb_enc_to_index(enc);
11305 static regex_t *reg_grapheme_cluster_utf8 = NULL;
11306
11307 if (encidx == rb_utf8_encindex()) {
11308 if (!reg_grapheme_cluster_utf8) {
11309 reg_grapheme_cluster_utf8 = get_reg_grapheme_cluster(enc);
11310 }
11311
11312 return reg_grapheme_cluster_utf8;
11313 }
11314
11315 return NULL;
11316}
11317
11318static VALUE
11319rb_str_each_grapheme_cluster_size(VALUE str, VALUE args, VALUE eobj)
11320{
11321 size_t grapheme_cluster_count = 0;
11322 rb_encoding *enc = get_encoding(str);
11323 const char *ptr, *end;
11324
11325 if (!rb_enc_unicode_p(enc)) {
11326 return rb_str_length(str);
11327 }
11328
11329 bool cached_reg_grapheme_cluster = true;
11330 regex_t *reg_grapheme_cluster = get_cached_reg_grapheme_cluster(enc);
11331 if (!reg_grapheme_cluster) {
11332 reg_grapheme_cluster = get_reg_grapheme_cluster(enc);
11333 cached_reg_grapheme_cluster = false;
11334 }
11335
11336 ptr = RSTRING_PTR(str);
11337 end = RSTRING_END(str);
11338
11339 while (ptr < end) {
11340 OnigPosition len = onig_match(reg_grapheme_cluster,
11341 (const OnigUChar *)ptr, (const OnigUChar *)end,
11342 (const OnigUChar *)ptr, NULL, 0);
11343 if (len <= 0) break;
11344 grapheme_cluster_count++;
11345 ptr += len;
11346 }
11347
11348 if (!cached_reg_grapheme_cluster) {
11349 onig_free(reg_grapheme_cluster);
11350 }
11351
11352 return SIZET2NUM(grapheme_cluster_count);
11353}
11354
11355static VALUE
11356rb_str_enumerate_grapheme_clusters(VALUE str, VALUE ary)
11357{
11358 VALUE orig = str;
11359 rb_encoding *enc = get_encoding(str);
11360 const char *ptr0, *ptr, *end;
11361
11362 if (!rb_enc_unicode_p(enc)) {
11363 return rb_str_enumerate_chars(str, ary);
11364 }
11365
11366 if (!ary) str = rb_str_new_frozen(str);
11367
11368 bool cached_reg_grapheme_cluster = true;
11369 regex_t *reg_grapheme_cluster = get_cached_reg_grapheme_cluster(enc);
11370 if (!reg_grapheme_cluster) {
11371 reg_grapheme_cluster = get_reg_grapheme_cluster(enc);
11372 cached_reg_grapheme_cluster = false;
11373 }
11374
11375 ptr0 = ptr = RSTRING_PTR(str);
11376 end = RSTRING_END(str);
11377
11378 while (ptr < end) {
11379 OnigPosition len = onig_match(reg_grapheme_cluster,
11380 (const OnigUChar *)ptr, (const OnigUChar *)end,
11381 (const OnigUChar *)ptr, NULL, 0);
11382 if (len <= 0) break;
11383 ENUM_ELEM(ary, rb_str_subseq(str, ptr-ptr0, len));
11384 ptr += len;
11385 }
11386
11387 if (!cached_reg_grapheme_cluster) {
11388 onig_free(reg_grapheme_cluster);
11389 }
11390
11391 RB_GC_GUARD(str);
11392 if (ary)
11393 return ary;
11394 else
11395 return orig;
11396}
11397
11398/*
11399 * call-seq:
11400 * each_grapheme_cluster {|grapheme_cluster| ... } -> self
11401 * each_grapheme_cluster -> enumerator
11402 *
11403 * :include: doc/string/each_grapheme_cluster.rdoc
11404 *
11405 */
11406
11407static VALUE
11408rb_str_each_grapheme_cluster(VALUE str)
11409{
11410 RETURN_SIZED_ENUMERATOR(str, 0, 0, rb_str_each_grapheme_cluster_size);
11411 return rb_str_enumerate_grapheme_clusters(str, 0);
11412}
11413
11414/*
11415 * call-seq:
11416 * grapheme_clusters -> array_of_grapheme_clusters
11417 *
11418 * :include: doc/string/grapheme_clusters.rdoc
11419 *
11420 */
11421
11422static VALUE
11423rb_str_grapheme_clusters(VALUE str)
11424{
11425 VALUE ary = WANTARRAY("grapheme_clusters", rb_str_strlen(str));
11426 return rb_str_enumerate_grapheme_clusters(str, ary);
11427}
11428
11429static long
11430chopped_length(VALUE str)
11431{
11432 rb_encoding *enc = STR_ENC_GET(str);
11433 const char *p, *p2, *beg, *end;
11434
11435 beg = RSTRING_PTR(str);
11436 end = beg + RSTRING_LEN(str);
11437 if (beg >= end) return 0;
11438 p = rb_enc_prev_char(beg, end, end, enc);
11439 if (!p) return 0;
11440 if (p > beg && rb_enc_ascget(p, end, 0, enc) == '\n') {
11441 p2 = rb_enc_prev_char(beg, p, end, enc);
11442 if (p2 && rb_enc_ascget(p2, end, 0, enc) == '\r') p = p2;
11443 }
11444 return p - beg;
11445}
11446
11447/*
11448 * call-seq:
11449 * chop! -> self or nil
11450 *
11451 * Like String#chop, except that:
11452 *
11453 * - Removes trailing characters from +self+ (not from a copy of +self+).
11454 * - Returns +self+ if any characters are removed, +nil+ otherwise.
11455 *
11456 * Related: see {Modifying}[rdoc-ref:String@Modifying].
11457 */
11458
11459static VALUE
11460rb_str_chop_bang(VALUE str)
11461{
11462 str_modify_keep_cr(str);
11463 if (RSTRING_LEN(str) > 0) {
11464 long len;
11465 len = chopped_length(str);
11466 STR_SET_LEN(str, len);
11467 TERM_FILL(&RSTRING_PTR(str)[len], TERM_LEN(str));
11468 if (ENC_CODERANGE(str) != ENC_CODERANGE_7BIT) {
11470 }
11471 return str;
11472 }
11473 return Qnil;
11474}
11475
11476
11477/*
11478 * call-seq:
11479 * chop -> new_string
11480 *
11481 * :include: doc/string/chop.rdoc
11482 *
11483 */
11484
11485static VALUE
11486rb_str_chop(VALUE str)
11487{
11488 return rb_str_subseq(str, 0, chopped_length(str));
11489}
11490
11491static long
11492smart_chomp(VALUE str, const char *e, const char *p)
11493{
11494 rb_encoding *enc = rb_enc_get(str);
11495 if (rb_enc_mbminlen(enc) > 1) {
11496 /* a receiver shorter than one character has nothing to chomp */
11497 if (e - p < rb_enc_mbminlen(enc)) return e - p;
11498 const char *pp = rb_enc_left_char_head(p, e-rb_enc_mbminlen(enc), e, enc);
11499 if (rb_enc_is_newline(pp, e, enc)) {
11500 e = pp;
11501 }
11502 pp = e - rb_enc_mbminlen(enc);
11503 if (pp >= p) {
11504 pp = rb_enc_left_char_head(p, pp, e, enc);
11505 if (rb_enc_ascget(pp, e, 0, enc) == '\r') {
11506 e = pp;
11507 }
11508 }
11509 }
11510 else {
11511 switch (*(e-1)) { /* not e[-1] to get rid of VC bug */
11512 case '\n':
11513 if (--e > p && *(e-1) == '\r') {
11514 --e;
11515 }
11516 break;
11517 case '\r':
11518 --e;
11519 break;
11520 }
11521 }
11522 return e - p;
11523}
11524
11525static long
11526chompped_length(VALUE str, VALUE rs)
11527{
11528 rb_encoding *enc;
11529 int newline;
11530 const char *pp, *e, *rsptr;
11531 long rslen;
11532 const char *const p = RSTRING_PTR(str);
11533 long len = RSTRING_LEN(str);
11534
11535 if (len == 0) return 0;
11536 e = p + len;
11537 if (rs == rb_default_rs) {
11538 return smart_chomp(str, e, p);
11539 }
11540
11541 enc = rb_enc_get(str);
11542 RSTRING_GETMEM(rs, rsptr, rslen);
11543 if (rslen == 0) {
11544 if (rb_enc_mbminlen(enc) > 1) {
11545 while (e - p >= rb_enc_mbminlen(enc)) {
11546 pp = rb_enc_left_char_head(p, e-rb_enc_mbminlen(enc), e, enc);
11547 if (!rb_enc_is_newline(pp, e, enc)) break;
11548 e = pp;
11549 pp -= rb_enc_mbminlen(enc);
11550 if (pp >= p) {
11551 pp = rb_enc_left_char_head(p, pp, e, enc);
11552 if (rb_enc_ascget(pp, e, 0, enc) == '\r') {
11553 e = pp;
11554 }
11555 }
11556 }
11557 }
11558 else {
11559 while (e > p && *(e-1) == '\n') {
11560 --e;
11561 if (e > p && *(e-1) == '\r')
11562 --e;
11563 }
11564 }
11565 return e - p;
11566 }
11567 if (rslen > len) return len;
11568
11569 enc = rb_enc_get(rs);
11570 newline = rsptr[rslen-1];
11571 if (rslen == rb_enc_mbminlen(enc)) {
11572 if (rslen == 1) {
11573 if (newline == '\n')
11574 return smart_chomp(str, e, p);
11575 }
11576 else {
11577 if (rb_enc_is_newline(rsptr, rsptr+rslen, enc))
11578 return smart_chomp(str, e, p);
11579 }
11580 }
11581
11582 enc = rb_enc_check(str, rs);
11583 if (is_broken_string(rs)) {
11584 return len;
11585 }
11586 pp = e - rslen;
11587 if (p[len-1] == newline &&
11588 (rslen <= 1 ||
11589 memcmp(rsptr, pp, rslen) == 0)) {
11590 if (at_char_boundary(p, pp, e, enc))
11591 return len - rslen;
11592 RB_GC_GUARD(rs);
11593 }
11594 return len;
11595}
11596
11602static VALUE
11603chomp_rs(int argc, const VALUE *argv)
11604{
11605 rb_check_arity(argc, 0, 1);
11606 if (argc > 0) {
11607 VALUE rs = argv[0];
11608 if (!NIL_P(rs)) StringValue(rs);
11609 return rs;
11610 }
11611 else {
11612 return rb_rs;
11613 }
11614}
11615
11616static VALUE
11617str_shrink(VALUE str, long len)
11618{
11619 str_modify_keep_cr(str);
11620 STR_SET_LEN(str, len);
11621 TERM_FILL(&RSTRING_PTR(str)[len], TERM_LEN(str));
11622 if (ENC_CODERANGE(str) != ENC_CODERANGE_7BIT) {
11624 }
11625 return str;
11626}
11627
11628VALUE
11629rb_str_chomp_string(VALUE str, VALUE rs)
11630{
11631 long olen = RSTRING_LEN(str);
11632 long len = chompped_length(str, rs);
11633 if (len >= olen) return Qnil;
11634 return str_shrink(str, len);
11635}
11636
11637/*
11638 * call-seq:
11639 * chomp!(line_sep = $/) -> self or nil
11640 *
11641 * Like String#chomp, except that:
11642 *
11643 * - Removes trailing characters from +self+ (not from a copy of +self+).
11644 * - Returns +self+ if any characters are removed, +nil+ otherwise.
11645 *
11646 * Related: see {Modifying}[rdoc-ref:String@Modifying].
11647 */
11648
11649static VALUE
11650rb_str_chomp_bang(int argc, VALUE *argv, VALUE str)
11651{
11652 VALUE rs;
11653 str_modifiable(str);
11654 if (RSTRING_LEN(str) == 0 && argc < 2) return Qnil;
11655 rs = chomp_rs(argc, argv);
11656 if (NIL_P(rs)) return Qnil;
11657 return rb_str_chomp_string(str, rs);
11658}
11659
11660
11661/*
11662 * call-seq:
11663 * chomp(line_sep = $/) -> new_string
11664 *
11665 * :include: doc/string/chomp.rdoc
11666 *
11667 */
11668
11669static VALUE
11670rb_str_chomp(int argc, VALUE *argv, VALUE str)
11671{
11672 VALUE rs = chomp_rs(argc, argv);
11673 if (NIL_P(rs)) return str_duplicate(rb_cString, str);
11674 return rb_str_subseq(str, 0, chompped_length(str, rs));
11675}
11676
11677static void
11678tr_setup_table_multi(char table[TR_TABLE_SIZE], VALUE *tablep, VALUE *ctablep,
11679 VALUE str, int num_selectors, VALUE *selectors)
11680{
11681 int i;
11682
11683 for (i=0; i<num_selectors; i++) {
11684 VALUE selector = selectors[i];
11685 rb_encoding *enc;
11686
11687 StringValue(selector);
11688 enc = rb_enc_check(str, selector);
11689 tr_setup_table(selector, table, i==0, tablep, ctablep, enc);
11690 }
11691}
11692
11693static long
11694lstrip_offset(VALUE str, const char *s, const char *e, rb_encoding *enc)
11695{
11696 const char *const start = s;
11697
11698 if (!s || s >= e) return 0;
11699
11700 /* remove spaces at head */
11701 if (single_byte_optimizable(str)) {
11702 while (s < e && (*s == '\0' || ascii_isspace(*s))) s++;
11703 }
11704 else {
11705 while (s < e) {
11706 int n;
11707 unsigned int cc = rb_enc_codepoint_len(s, e, &n, enc);
11708
11709 if (cc && !rb_isspace(cc)) break;
11710 s += n;
11711 }
11712 }
11713 return s - start;
11714}
11715
11716static long
11717lstrip_offset_table(VALUE str, const char *s, const char *e, rb_encoding *enc,
11718 char table[TR_TABLE_SIZE], VALUE del, VALUE nodel)
11719{
11720 const char *const start = s;
11721
11722 if (!s || s >= e) return 0;
11723
11724 /* remove leading characters in the table */
11725 while (s < e) {
11726 int n;
11727 unsigned int cc = rb_enc_codepoint_len(s, e, &n, enc);
11728
11729 if (!tr_find(cc, table, del, nodel)) break;
11730 s += n;
11731 }
11732 return s - start;
11733}
11734
11735/*
11736 * call-seq:
11737 * lstrip!(*selectors) -> self or nil
11738 *
11739 * Like String#lstrip, except that:
11740 *
11741 * - Performs stripping in +self+ (not in a copy of +self+).
11742 * - Returns +self+ if any characters are stripped, +nil+ otherwise.
11743 *
11744 * Related: see {Modifying}[rdoc-ref:String@Modifying].
11745 */
11746
11747static VALUE
11748rb_str_lstrip_bang(int argc, VALUE *argv, VALUE str)
11749{
11750 rb_encoding *enc;
11751 char *start;
11752 long olen, loffset;
11753
11754 str_modify_keep_cr(str);
11755 enc = STR_ENC_GET(str);
11756 RSTRING_GETMEM(str, start, olen);
11757 if (argc > 0) {
11758 char table[TR_TABLE_SIZE];
11759 VALUE del = 0, nodel = 0;
11760
11761 tr_setup_table_multi(table, &del, &nodel, str, argc, argv);
11762
11763 /* the selector conversion may have modified str */
11764 str_modify_keep_cr(str);
11765 enc = STR_ENC_GET(str);
11766 RSTRING_GETMEM(str, start, olen);
11767
11768 loffset = lstrip_offset_table(str, start, start+olen, enc, table, del, nodel);
11769 }
11770 else {
11771 loffset = lstrip_offset(str, start, start+olen, enc);
11772 }
11773
11774 if (loffset > 0) {
11775 long len = olen-loffset;
11776 memmove(start, start + loffset, len);
11777 STR_SET_LEN(str, len);
11778 TERM_FILL(start+len, rb_enc_mbminlen(enc));
11779 return str;
11780 }
11781 return Qnil;
11782}
11783
11784
11785/*
11786 * call-seq:
11787 * lstrip(*selectors) -> new_string
11788 *
11789 * Returns a copy of +self+ with leading whitespace removed;
11790 * see {Whitespace in Strings}[rdoc-ref:String@Whitespace+in+Strings]:
11791 *
11792 * whitespace = "\x00\t\n\v\f\r "
11793 * s = whitespace + 'abc' + whitespace
11794 * # => "\u0000\t\n\v\f\r abc\u0000\t\n\v\f\r "
11795 * s.lstrip
11796 * # => "abc\u0000\t\n\v\f\r "
11797 *
11798 * If +selectors+ are given, removes characters of +selectors+ from the beginning of +self+:
11799 *
11800 * s = "---abc+++"
11801 * s.lstrip("-") # => "abc+++"
11802 *
11803 * +selectors+ must be valid character selectors (see {Character Selectors}[rdoc-ref:character_selectors.rdoc]),
11804 * and may use any of its valid forms, including negation, ranges, and escapes:
11805 *
11806 * "01234abc56789".lstrip("0-9") # "abc56789"
11807 * "01234abc56789".lstrip("0-9", "^4-6") # "4abc56789"
11808 *
11809 * Related: see {Converting to New String}[rdoc-ref:String@Converting+to+New+String].
11810 */
11811
11812static VALUE
11813rb_str_lstrip(int argc, VALUE *argv, VALUE str)
11814{
11815 const char *start;
11816 long len, loffset;
11817
11818 RSTRING_GETMEM(str, start, len);
11819 if (argc > 0) {
11820 char table[TR_TABLE_SIZE];
11821 VALUE del = 0, nodel = 0;
11822
11823 tr_setup_table_multi(table, &del, &nodel, str, argc, argv);
11824
11825 /* the selector conversion may have modified str */
11826 RSTRING_GETMEM(str, start, len);
11827
11828 loffset = lstrip_offset_table(str, start, start+len, STR_ENC_GET(str), table, del, nodel);
11829 }
11830 else {
11831 loffset = lstrip_offset(str, start, start+len, STR_ENC_GET(str));
11832 }
11833 if (loffset <= 0) return str_duplicate(rb_cString, str);
11834 return rb_str_subseq(str, loffset, len - loffset);
11835}
11836
11837static long
11838rstrip_offset(VALUE str, const char *s, const char *e, rb_encoding *enc)
11839{
11840 const char *t;
11841
11842 rb_str_check_dummy_enc(enc);
11843 if (rb_enc_str_coderange(str) == ENC_CODERANGE_BROKEN) {
11844 rb_raise(rb_eEncCompatError, "invalid byte sequence in %s", rb_enc_name(enc));
11845 }
11846 if (!s || s >= e) return 0;
11847 t = e;
11848
11849 /* remove trailing spaces or '\0's */
11850 if (single_byte_optimizable(str)) {
11851 unsigned char c;
11852 while (s < t && ((c = *(t-1)) == '\0' || ascii_isspace(c))) t--;
11853 }
11854 else {
11855 const char *tp;
11856
11857 while ((tp = rb_enc_prev_char(s, t, e, enc)) != NULL) {
11858 unsigned int c = rb_enc_codepoint(tp, e, enc);
11859 if (c && !rb_isspace(c)) break;
11860 t = tp;
11861 }
11862 }
11863 return e - t;
11864}
11865
11866static long
11867rstrip_offset_table(VALUE str, const char *s, const char *e, rb_encoding *enc,
11868 char table[TR_TABLE_SIZE], VALUE del, VALUE nodel)
11869{
11870 const char *t, *tp;
11871
11872 rb_str_check_dummy_enc(enc);
11873 if (rb_enc_str_coderange(str) == ENC_CODERANGE_BROKEN) {
11874 rb_raise(rb_eEncCompatError, "invalid byte sequence in %s", rb_enc_name(enc));
11875 }
11876 if (!s || s >= e) return 0;
11877 t = e;
11878
11879 /* remove trailing characters in the table */
11880 while ((tp = rb_enc_prev_char(s, t, e, enc)) != NULL) {
11881 unsigned int c = rb_enc_codepoint(tp, e, enc);
11882 if (!tr_find(c, table, del, nodel)) break;
11883 t = tp;
11884 }
11885
11886 return e - t;
11887}
11888
11889/*
11890 * call-seq:
11891 * rstrip!(*selectors) -> self or nil
11892 *
11893 * Like String#rstrip, except that:
11894 *
11895 * - Performs stripping in +self+ (not in a copy of +self+).
11896 * - Returns +self+ if any characters are stripped, +nil+ otherwise.
11897 *
11898 * Related: see {Modifying}[rdoc-ref:String@Modifying].
11899 */
11900
11901static VALUE
11902rb_str_rstrip_bang(int argc, VALUE *argv, VALUE str)
11903{
11904 rb_encoding *enc;
11905 char *start;
11906 long olen, roffset;
11907
11908 str_modify_keep_cr(str);
11909 enc = STR_ENC_GET(str);
11910 RSTRING_GETMEM(str, start, olen);
11911 if (argc > 0) {
11912 char table[TR_TABLE_SIZE];
11913 VALUE del = 0, nodel = 0;
11914
11915 tr_setup_table_multi(table, &del, &nodel, str, argc, argv);
11916
11917 /* the selector conversion may have modified str */
11918 str_modify_keep_cr(str);
11919 enc = STR_ENC_GET(str);
11920 RSTRING_GETMEM(str, start, olen);
11921
11922 roffset = rstrip_offset_table(str, start, start+olen, enc, table, del, nodel);
11923 }
11924 else {
11925 roffset = rstrip_offset(str, start, start+olen, enc);
11926 }
11927 if (roffset > 0) {
11928 long len = olen - roffset;
11929
11930 STR_SET_LEN(str, len);
11931 TERM_FILL(start+len, rb_enc_mbminlen(enc));
11932 return str;
11933 }
11934 return Qnil;
11935}
11936
11937
11938/*
11939 * call-seq:
11940 * rstrip(*selectors) -> new_string
11941 *
11942 * Returns a copy of +self+ with trailing whitespace removed;
11943 * see {Whitespace in Strings}[rdoc-ref:String@Whitespace+in+Strings]:
11944 *
11945 * whitespace = "\x00\t\n\v\f\r "
11946 * s = whitespace + 'abc' + whitespace
11947 * s # => "\u0000\t\n\v\f\r abc\u0000\t\n\v\f\r "
11948 * s.rstrip # => "\u0000\t\n\v\f\r abc"
11949 *
11950 * If +selectors+ are given, removes characters of +selectors+ from the end of +self+:
11951 *
11952 * s = "---abc+++"
11953 * s.rstrip("+") # => "---abc"
11954 *
11955 * +selectors+ must be valid character selectors (see {Character Selectors}[rdoc-ref:character_selectors.rdoc]),
11956 * and may use any of its valid forms, including negation, ranges, and escapes:
11957 *
11958 * "01234abc56789".rstrip("0-9") # "01234abc"
11959 * "01234abc56789".rstrip("0-9", "^4-6") # "01234abc56"
11960 *
11961 * Related: see {Converting to New String}[rdoc-ref:String@Converting+to+New+String].
11962 */
11963
11964static VALUE
11965rb_str_rstrip(int argc, VALUE *argv, VALUE str)
11966{
11967 rb_encoding *enc;
11968 const char *start;
11969 long olen, roffset;
11970
11971 enc = STR_ENC_GET(str);
11972 RSTRING_GETMEM(str, start, olen);
11973 if (argc > 0) {
11974 char table[TR_TABLE_SIZE];
11975 VALUE del = 0, nodel = 0;
11976
11977 tr_setup_table_multi(table, &del, &nodel, str, argc, argv);
11978
11979 /* the selector conversion may have modified str */
11980 enc = STR_ENC_GET(str);
11981
11982 RSTRING_GETMEM(str, start, olen);
11983 roffset = rstrip_offset_table(str, start, start+olen, enc, table, del, nodel);
11984 }
11985 else {
11986 roffset = rstrip_offset(str, start, start+olen, enc);
11987 }
11988 if (roffset <= 0) return str_duplicate(rb_cString, str);
11989 return rb_str_subseq(str, 0, olen-roffset);
11990}
11991
11992
11993/*
11994 * call-seq:
11995 * strip!(*selectors) -> self or nil
11996 *
11997 * Like String#strip, except that:
11998 *
11999 * - Any modifications are made to +self+.
12000 * - Returns +self+ if any modification are made, +nil+ otherwise.
12001 *
12002 * Related: see {Modifying}[rdoc-ref:String@Modifying].
12003 */
12004
12005static VALUE
12006rb_str_strip_bang(int argc, VALUE *argv, VALUE str)
12007{
12008 char *start;
12009 long olen, loffset, roffset;
12010 rb_encoding *enc;
12011
12012 str_modify_keep_cr(str);
12013 enc = STR_ENC_GET(str);
12014 RSTRING_GETMEM(str, start, olen);
12015
12016 if (argc > 0) {
12017 char table[TR_TABLE_SIZE];
12018 VALUE del = 0, nodel = 0;
12019
12020 tr_setup_table_multi(table, &del, &nodel, str, argc, argv);
12021
12022 /* the selector conversion may have modified str */
12023 str_modify_keep_cr(str);
12024 enc = STR_ENC_GET(str);
12025 RSTRING_GETMEM(str, start, olen);
12026
12027 loffset = lstrip_offset_table(str, start, start+olen, enc, table, del, nodel);
12028 roffset = rstrip_offset_table(str, start+loffset, start+olen, enc, table, del, nodel);
12029 }
12030 else {
12031 loffset = lstrip_offset(str, start, start+olen, enc);
12032 roffset = rstrip_offset(str, start+loffset, start+olen, enc);
12033 }
12034
12035 if (loffset > 0 || roffset > 0) {
12036 long len = olen-roffset;
12037 if (loffset > 0) {
12038 len -= loffset;
12039 memmove(start, start + loffset, len);
12040 }
12041 STR_SET_LEN(str, len);
12042 TERM_FILL(start+len, rb_enc_mbminlen(enc));
12043 return str;
12044 }
12045 return Qnil;
12046}
12047
12048
12049/*
12050 * call-seq:
12051 * strip(*selectors) -> new_string
12052 *
12053 * Returns a copy of +self+ with leading and trailing whitespace removed;
12054 * see {Whitespace in Strings}[rdoc-ref:String@Whitespace+in+Strings]:
12055 *
12056 * whitespace = "\x00\t\n\v\f\r "
12057 * s = whitespace + 'abc' + whitespace
12058 * # => "\u0000\t\n\v\f\r abc\u0000\t\n\v\f\r "
12059 * s.strip # => "abc"
12060 *
12061 * If +selectors+ are given, removes characters of +selectors+ from both ends of +self+:
12062 *
12063 * s = "---abc+++"
12064 * s.strip("-+") # => "abc"
12065 * s.strip("+-") # => "abc"
12066 *
12067 * +selectors+ must be valid character selectors (see {Character Selectors}[rdoc-ref:character_selectors.rdoc]),
12068 * and may use any of its valid forms, including negation, ranges, and escapes:
12069 *
12070 * "01234abc56789".strip("0-9") # "abc"
12071 * "01234abc56789".strip("0-9", "^4-6") # "4abc56"
12072 *
12073 * Related: see {Converting to New String}[rdoc-ref:String@Converting+to+New+String].
12074 */
12075
12076static VALUE
12077rb_str_strip(int argc, VALUE *argv, VALUE str)
12078{
12079 const char *start;
12080 long olen, loffset, roffset;
12081 rb_encoding *enc = STR_ENC_GET(str);
12082
12083 RSTRING_GETMEM(str, start, olen);
12084
12085 if (argc > 0) {
12086 char table[TR_TABLE_SIZE];
12087 VALUE del = 0, nodel = 0;
12088
12089 tr_setup_table_multi(table, &del, &nodel, str, argc, argv);
12090
12091 /* the selector conversion may have modified str */
12092 enc = STR_ENC_GET(str);
12093 RSTRING_GETMEM(str, start, olen);
12094
12095 loffset = lstrip_offset_table(str, start, start+olen, enc, table, del, nodel);
12096 roffset = rstrip_offset_table(str, start+loffset, start+olen, enc, table, del, nodel);
12097 }
12098 else {
12099 loffset = lstrip_offset(str, start, start+olen, enc);
12100 roffset = rstrip_offset(str, start+loffset, start+olen, enc);
12101 }
12102
12103 if (loffset <= 0 && roffset <= 0) return str_duplicate(rb_cString, str);
12104 return rb_str_subseq(str, loffset, olen-loffset-roffset);
12105}
12106
12107static VALUE
12108scan_once(VALUE str, VALUE pat, long *start, int set_backref_str)
12109{
12110 VALUE result = Qnil;
12111 long end, pos = rb_pat_search(pat, str, *start, set_backref_str);
12112 if (pos >= 0) {
12113 VALUE match = Qnil;
12114 if (BUILTIN_TYPE(pat) == T_STRING) {
12115 end = pos + RSTRING_LEN(pat);
12116 }
12117 else {
12118 match = rb_backref_get();
12119 pos = RMATCH_BEG(match, 0);
12120 end = RMATCH_END(match, 0);
12121 }
12122
12123 if (pos == end) {
12124 rb_encoding *enc = STR_ENC_GET(str);
12125 /*
12126 * Always consume at least one character of the input string
12127 */
12128 if (RSTRING_LEN(str) > end)
12129 *start = end + rb_enc_fast_mbclen(RSTRING_PTR(str) + end,
12130 RSTRING_END(str), enc);
12131 else
12132 *start = end + 1;
12133 }
12134 else {
12135 *start = end;
12136 }
12137
12138 if (NIL_P(match) || RMATCH_NREGS(match) == 1) {
12139 result = rb_str_subseq(str, pos, end - pos);
12140 return result;
12141 }
12142 else {
12143 int num_regs = RMATCH_NREGS(match);
12144 result = rb_ary_new2(num_regs);
12145 for (int i = 1; i < num_regs; i++) {
12146 VALUE s = Qnil;
12147 if (RMATCH_BEG(match, i) >= 0) {
12148 s = rb_str_subseq(str, RMATCH_BEG(match, i), RMATCH_END(match, i) - RMATCH_BEG(match, i));
12149 }
12150
12151 rb_ary_push(result, s);
12152 }
12153 }
12154
12155 RB_GC_GUARD(match);
12156 }
12157
12158 return result;
12159}
12160
12161
12162/*
12163 * call-seq:
12164 * scan(pattern) -> array_of_results
12165 * scan(pattern) {|result| ... } -> self
12166 *
12167 * :include: doc/string/scan.rdoc
12168 *
12169 */
12170
12171static VALUE
12172rb_str_scan(VALUE str, VALUE pat)
12173{
12174 VALUE result;
12175 long start = 0;
12176 long last = -1, prev = 0;
12177 const char *p = RSTRING_PTR(str);
12178 long len = RSTRING_LEN(str);
12179
12180 pat = get_pat_quoted(pat, 1);
12181 mustnot_broken(str);
12182 if (!rb_block_given_p()) {
12183 VALUE ary = rb_ary_new();
12184
12185 while (!NIL_P(result = scan_once(str, pat, &start, 0))) {
12186 last = prev;
12187 prev = start;
12188 rb_ary_push(ary, result);
12189 }
12190 if (last >= 0) rb_pat_search(pat, str, last, 1);
12191 else rb_backref_set(Qnil);
12192 return ary;
12193 }
12194
12195 while (!NIL_P(result = scan_once(str, pat, &start, 1))) {
12196 last = prev;
12197 prev = start;
12198 rb_yield(result);
12199 str_mod_check(str, p, len);
12200 }
12201 if (last >= 0) rb_pat_search(pat, str, last, 1);
12202 return str;
12203}
12204
12205
12206/*
12207 * call-seq:
12208 * hex -> integer
12209 *
12210 * Interprets the leading substring of +self+ as hexadecimal, possibly signed;
12211 * returns its value as an integer.
12212 *
12213 * The leading substring is interpreted as hexadecimal when it begins with:
12214 *
12215 * - One or more character representing hexadecimal digits
12216 * (each in one of the ranges <tt>'0'..'9'</tt>, <tt>'a'..'f'</tt>, or <tt>'A'..'F'</tt>);
12217 * the string to be interpreted ends at the first character that does not represent a hexadecimal digit:
12218 *
12219 * 'f'.hex # => 15
12220 * '11'.hex # => 17
12221 * 'FFF'.hex # => 4095
12222 * 'fffg'.hex # => 4095
12223 * 'foo'.hex # => 15 # 'f' hexadecimal, 'oo' not.
12224 * 'bar'.hex # => 186 # 'ba' hexadecimal, 'r' not.
12225 * 'deadbeef'.hex # => 3735928559
12226 *
12227 * - <tt>'0x'</tt> or <tt>'0X'</tt>, followed by one or more hexadecimal digits:
12228 *
12229 * '0xfff'.hex # => 4095
12230 * '0xfffg'.hex # => 4095
12231 *
12232 * Any of the above may prefixed with <tt>'-'</tt>, which negates the interpreted value:
12233 *
12234 * '-fff'.hex # => -4095
12235 * '-0xFFF'.hex # => -4095
12236 *
12237 * For any substring not described above, returns zero:
12238 *
12239 * 'xxx'.hex # => 0
12240 * ''.hex # => 0
12241 *
12242 * Note that, unlike #oct, this method interprets only hexadecimal,
12243 * and not binary, octal, or decimal notations:
12244 *
12245 * '0b111'.hex # => 45329
12246 * '0o777'.hex # => 0
12247 * '0d999'.hex # => 55705
12248 *
12249 * Related: See {Converting to Non-String}[rdoc-ref:String@Converting+to+Non-String].
12250 */
12251
12252static VALUE
12253rb_str_hex(VALUE str)
12254{
12255 return rb_str_to_inum(str, 16, FALSE);
12256}
12257
12258
12259/*
12260 * call-seq:
12261 * oct -> integer
12262 *
12263 * Interprets the leading substring of +self+ as octal, binary, decimal, or hexadecimal, possibly signed;
12264 * returns their value as an integer.
12265 *
12266 * In brief:
12267 *
12268 * # Interpreted as octal.
12269 * '777'.oct # => 511
12270 * '777x'.oct # => 511
12271 * '0777'.oct # => 511
12272 * '0o777'.oct # => 511
12273 * '-777'.oct # => -511
12274 * # Not interpreted as octal.
12275 * '0b111'.oct # => 7 # Interpreted as binary.
12276 * '0d999'.oct # => 999 # Interpreted as decimal.
12277 * '0xfff'.oct # => 4095 # Interpreted as hexadecimal.
12278 *
12279 * The leading substring is interpreted as octal when it begins with:
12280 *
12281 * - One or more character representing octal digits
12282 * (each in the range <tt>'0'..'7'</tt>);
12283 * the string to be interpreted ends at the first character that does not represent an octal digit:
12284 *
12285 * '7'.oct @ => 7
12286 * '11'.oct # => 9
12287 * '777'.oct # => 511
12288 * '0777'.oct # => 511
12289 * '7778'.oct # => 511
12290 * '777x'.oct # => 511
12291 *
12292 * - <tt>'0o'</tt>, followed by one or more octal digits:
12293 *
12294 * '0o777'.oct # => 511
12295 * '0o7778'.oct # => 511
12296 *
12297 * The leading substring is _not_ interpreted as octal when it begins with:
12298 *
12299 * - <tt>'0b'</tt>, followed by one or more characters representing binary digits
12300 * (each in the range <tt>'0'..'1'</tt>);
12301 * the string to be interpreted ends at the first character that does not represent a binary digit.
12302 * the string is interpreted as binary digits (base 2):
12303 *
12304 * '0b111'.oct # => 7
12305 * '0b1112'.oct # => 7
12306 *
12307 * - <tt>'0d'</tt>, followed by one or more characters representing decimal digits
12308 * (each in the range <tt>'0'..'9'</tt>);
12309 * the string to be interpreted ends at the first character that does not represent a decimal digit.
12310 * the string is interpreted as decimal digits (base 10):
12311 *
12312 * '0d999'.oct # => 999
12313 * '0d999x'.oct # => 999
12314 *
12315 * - <tt>'0x'</tt>, followed by one or more characters representing hexadecimal digits
12316 * (each in one of the ranges <tt>'0'..'9'</tt>, <tt>'a'..'f'</tt>, or <tt>'A'..'F'</tt>);
12317 * the string to be interpreted ends at the first character that does not represent a hexadecimal digit.
12318 * the string is interpreted as hexadecimal digits (base 16):
12319 *
12320 * '0xfff'.oct # => 4095
12321 * '0xfffg'.oct # => 4095
12322 *
12323 * Any of the above may prefixed with <tt>'-'</tt>, which negates the interpreted value:
12324 *
12325 * '-777'.oct # => -511
12326 * '-0777'.oct # => -511
12327 * '-0b111'.oct # => -7
12328 * '-0xfff'.oct # => -4095
12329 *
12330 * For any substring not described above, returns zero:
12331 *
12332 * 'foo'.oct # => 0
12333 * ''.oct # => 0
12334 *
12335 * Related: see {Converting to Non-String}[rdoc-ref:String@Converting+to+Non-String].
12336 */
12337
12338static VALUE
12339rb_str_oct(VALUE str)
12340{
12341 return rb_str_to_inum(str, -8, FALSE);
12342}
12343
12344#ifndef HAVE_CRYPT_R
12345# include "ruby/thread_native.h"
12346# include "ruby/atomic.h"
12347
12348static struct {
12349 rb_nativethread_lock_t lock;
12350} crypt_mutex = {PTHREAD_MUTEX_INITIALIZER};
12351#endif
12352
12353/*
12354 * call-seq:
12355 * crypt(salt_str) -> new_string
12356 *
12357 * Returns the string generated by calling <code>crypt(3)</code>
12358 * standard library function with <code>str</code> and
12359 * <code>salt_str</code>, in this order, as its arguments. Please do
12360 * not use this method any longer. It is legacy; provided only for
12361 * backward compatibility with ruby scripts in earlier days. It is
12362 * bad to use in contemporary programs for several reasons:
12363 *
12364 * * Behaviour of C's <code>crypt(3)</code> depends on the OS it is
12365 * run. The generated string lacks data portability.
12366 *
12367 * * On some OSes such as Mac OS, <code>crypt(3)</code> never fails
12368 * (i.e. silently ends up in unexpected results).
12369 *
12370 * * On some OSes such as Mac OS, <code>crypt(3)</code> is not
12371 * thread safe.
12372 *
12373 * * So-called "traditional" usage of <code>crypt(3)</code> is very
12374 * very very weak. According to its manpage, Linux's traditional
12375 * <code>crypt(3)</code> output has only 2**56 variations; too
12376 * easy to brute force today. And this is the default behaviour.
12377 *
12378 * * In order to make things robust some OSes implement so-called
12379 * "modular" usage. To go through, you have to do a complex
12380 * build-up of the <code>salt_str</code> parameter, by hand.
12381 * Failure in generation of a proper salt string tends not to
12382 * yield any errors; typos in parameters are normally not
12383 * detectable.
12384 *
12385 * * For instance, in the following example, the second invocation
12386 * of String#crypt is wrong; it has a typo in "round=" (lacks
12387 * "s"). However the call does not fail and something unexpected
12388 * is generated.
12389 *
12390 * "foo".crypt("$5$rounds=1000$salt$") # OK, proper usage
12391 * "foo".crypt("$5$round=1000$salt$") # Typo not detected
12392 *
12393 * * Even in the "modular" mode, some hash functions are considered
12394 * archaic and no longer recommended at all; for instance module
12395 * <code>$1$</code> is officially abandoned by its author: see
12396 * http://phk.freebsd.dk/sagas/md5crypt_eol/ . For another
12397 * instance module <code>$3$</code> is considered completely
12398 * broken: see the manpage of FreeBSD.
12399 *
12400 * * On some OS such as Mac OS, there is no modular mode. Yet, as
12401 * written above, <code>crypt(3)</code> on Mac OS never fails.
12402 * This means even if you build up a proper salt string it
12403 * generates a traditional DES hash anyways, and there is no way
12404 * for you to be aware of.
12405 *
12406 * "foo".crypt("$5$rounds=1000$salt$") # => "$5fNPQMxC5j6."
12407 *
12408 * If for some reason you cannot migrate to other secure contemporary
12409 * password hashing algorithms, install the string-crypt gem and
12410 * <code>require 'string/crypt'</code> to continue using it.
12411 */
12412
12413static VALUE
12414rb_str_crypt(VALUE str, VALUE salt)
12415{
12416#ifdef HAVE_CRYPT_R
12417 VALUE databuf;
12418 struct crypt_data *data;
12419# define CRYPT_END() ALLOCV_END(databuf)
12420#else
12421 char *tmp_buf;
12422 extern char *crypt(const char *, const char *);
12423# define CRYPT_END() rb_nativethread_lock_unlock(&crypt_mutex.lock)
12424#endif
12425 VALUE result;
12426 const char *s, *saltp, *res;
12427#ifdef BROKEN_CRYPT
12428 char salt_8bit_clean[3];
12429#endif
12430
12431 StringValue(salt);
12432 mustnot_wchar(str);
12433 mustnot_wchar(salt);
12434 s = StringValueCStr(str);
12435 saltp = RSTRING_PTR(salt);
12436 if (RSTRING_LEN(salt) < 2 || !saltp[0] || !saltp[1]) {
12437 rb_raise(rb_eArgError, "salt too short (need >=2 bytes)");
12438 }
12439
12440#ifdef BROKEN_CRYPT
12441 if (!ISASCII((unsigned char)saltp[0]) || !ISASCII((unsigned char)saltp[1])) {
12442 salt_8bit_clean[0] = saltp[0] & 0x7f;
12443 salt_8bit_clean[1] = saltp[1] & 0x7f;
12444 salt_8bit_clean[2] = '\0';
12445 saltp = salt_8bit_clean;
12446 }
12447#endif
12448#ifdef HAVE_CRYPT_R
12449 data = ALLOCV(databuf, sizeof(struct crypt_data));
12450# ifdef HAVE_STRUCT_CRYPT_DATA_INITIALIZED
12451 data->initialized = 0;
12452# endif
12453 res = crypt_r(s, saltp, data);
12454#else
12455 rb_nativethread_lock_lock(&crypt_mutex.lock);
12456 res = crypt(s, saltp);
12457#endif
12458 if (!res) {
12459 int err = errno;
12460 CRYPT_END();
12461 rb_syserr_fail(err, "crypt");
12462 }
12463#ifdef HAVE_CRYPT_R
12464 result = rb_str_new_cstr(res);
12465 CRYPT_END();
12466#else
12467 // We need to copy this buffer because it's static and we need to unlock the mutex
12468 // before allocating a new object (the string to be returned). If we allocate while
12469 // holding the lock, we could run GC which fires the VM barrier and causes a deadlock
12470 // if other ractors are waiting on this lock.
12471 size_t res_size = strlen(res);
12472 tmp_buf = ALLOCA_N(char, res_size); // should be small enough to alloca
12473 memcpy(tmp_buf, res, res_size);
12474 CRYPT_END();
12475 result = rb_str_new(tmp_buf, res_size);
12476#endif
12477 return result;
12478}
12479
12480
12481/*
12482 * call-seq:
12483 * ord -> integer
12484 *
12485 * :include: doc/string/ord.rdoc
12486 *
12487 */
12488
12489static VALUE
12490rb_str_ord(VALUE s)
12491{
12492 unsigned int c;
12493
12494 c = rb_enc_codepoint(RSTRING_PTR(s), RSTRING_END(s), STR_ENC_GET(s));
12495 return UINT2NUM(c);
12496}
12497/*
12498 * call-seq:
12499 * sum(n = 16) -> integer
12500 *
12501 * :include: doc/string/sum.rdoc
12502 *
12503 */
12504
12505static VALUE
12506rb_str_sum(int argc, VALUE *argv, VALUE str)
12507{
12508 int bits = 16;
12509 char *ptr, *p, *pend;
12510 long len;
12511 VALUE sum = INT2FIX(0);
12512 unsigned long sum0 = 0;
12513
12514 if (rb_check_arity(argc, 0, 1) && (bits = NUM2INT(argv[0])) < 0) {
12515 bits = 0;
12516 }
12517 ptr = p = RSTRING_PTR(str);
12518 len = RSTRING_LEN(str);
12519 pend = p + len;
12520
12521 while (p < pend) {
12522 if (FIXNUM_MAX - UCHAR_MAX < sum0) {
12523 sum = rb_funcall(sum, '+', 1, LONG2FIX(sum0));
12524 str_mod_check(str, ptr, len);
12525 sum0 = 0;
12526 }
12527 sum0 += (unsigned char)*p;
12528 p++;
12529 }
12530
12531 if (bits == 0) {
12532 if (sum0) {
12533 sum = rb_funcall(sum, '+', 1, LONG2FIX(sum0));
12534 }
12535 }
12536 else {
12537 if (sum == INT2FIX(0)) {
12538 if (bits < (int)sizeof(long)*CHAR_BIT) {
12539 sum0 &= (((unsigned long)1)<<bits)-1;
12540 }
12541 sum = LONG2FIX(sum0);
12542 }
12543 else {
12544 VALUE mod;
12545
12546 if (sum0) {
12547 sum = rb_funcall(sum, '+', 1, LONG2FIX(sum0));
12548 }
12549
12550 mod = rb_funcall(INT2FIX(1), idLTLT, 1, INT2FIX(bits));
12551 mod = rb_funcall(mod, '-', 1, INT2FIX(1));
12552 sum = rb_funcall(sum, '&', 1, mod);
12553 }
12554 }
12555 return sum;
12556}
12557
12558static VALUE
12559rb_str_justify(int argc, VALUE *argv, VALUE str, char jflag)
12560{
12561 rb_encoding *enc;
12562 VALUE w;
12563 long width, len, flen = 1, fclen = 1;
12564 VALUE res;
12565 char *p;
12566 const char *f = " ";
12567 long n, size, llen, rlen, llen2 = 0, rlen2 = 0;
12568 VALUE pad;
12569 int singlebyte = 1, cr;
12570 int termlen;
12571
12572 rb_scan_args(argc, argv, "11", &w, &pad);
12573 enc = STR_ENC_GET(str);
12574 width = NUM2LONG(w);
12575 if (argc == 2) {
12576 StringValue(pad);
12577 enc = rb_enc_check(str, pad);
12578 f = RSTRING_PTR(pad);
12579 flen = RSTRING_LEN(pad);
12580 fclen = str_strlen(pad, enc); /* rb_enc_check */
12581 singlebyte = single_byte_optimizable(pad);
12582 if (flen == 0 || fclen == 0) {
12583 rb_raise(rb_eArgError, "zero width padding");
12584 }
12585 }
12586 termlen = rb_enc_mbminlen(enc);
12587 len = str_strlen(str, enc); /* rb_enc_check */
12588 if (width < 0 || len >= width) return str_duplicate(rb_cString, str);
12589 n = width - len;
12590 llen = (jflag == 'l') ? 0 : ((jflag == 'r') ? n : n/2);
12591 rlen = n - llen;
12592 cr = ENC_CODERANGE(str);
12593 if (flen > 1) {
12594 llen2 = str_offset(f, f + flen, llen % fclen, enc, singlebyte);
12595 rlen2 = str_offset(f, f + flen, rlen % fclen, enc, singlebyte);
12596 }
12597 size = RSTRING_LEN(str);
12598 if ((len = llen / fclen + rlen / fclen) >= LONG_MAX / flen ||
12599 (len *= flen) >= LONG_MAX - llen2 - rlen2 ||
12600 (len += llen2 + rlen2) >= LONG_MAX - size) {
12601 rb_raise(rb_eArgError, "argument too big");
12602 }
12603 len += size;
12604 res = str_enc_new(rb_cString, 0, len, enc);
12605 p = RSTRING_PTR(res);
12606 if (flen <= 1) {
12607 memset(p, *f, llen);
12608 p += llen;
12609 }
12610 else {
12611 while (llen >= fclen) {
12612 memcpy(p,f,flen);
12613 p += flen;
12614 llen -= fclen;
12615 }
12616 if (llen > 0) {
12617 memcpy(p, f, llen2);
12618 p += llen2;
12619 }
12620 }
12621 memcpy(p, RSTRING_PTR(str), size);
12622 p += size;
12623 if (flen <= 1) {
12624 memset(p, *f, rlen);
12625 p += rlen;
12626 }
12627 else {
12628 while (rlen >= fclen) {
12629 memcpy(p,f,flen);
12630 p += flen;
12631 rlen -= fclen;
12632 }
12633 if (rlen > 0) {
12634 memcpy(p, f, rlen2);
12635 p += rlen2;
12636 }
12637 }
12638 TERM_FILL(p, termlen);
12639 STR_SET_LEN(res, p-RSTRING_PTR(res));
12640
12641 if (argc == 2)
12642 cr = ENC_CODERANGE_AND(cr, ENC_CODERANGE(pad));
12643 if (cr != ENC_CODERANGE_BROKEN)
12644 ENC_CODERANGE_SET(res, cr);
12645
12646 RB_GC_GUARD(pad);
12647 return res;
12648}
12649
12650
12651/*
12652 * call-seq:
12653 * ljust(width, pad_string = ' ') -> new_string
12654 *
12655 * :include: doc/string/ljust.rdoc
12656 *
12657 */
12658
12659static VALUE
12660rb_str_ljust(int argc, VALUE *argv, VALUE str)
12661{
12662 return rb_str_justify(argc, argv, str, 'l');
12663}
12664
12665/*
12666 * call-seq:
12667 * rjust(width, pad_string = ' ') -> new_string
12668 *
12669 * :include: doc/string/rjust.rdoc
12670 *
12671 */
12672
12673static VALUE
12674rb_str_rjust(int argc, VALUE *argv, VALUE str)
12675{
12676 return rb_str_justify(argc, argv, str, 'r');
12677}
12678
12679
12680/*
12681 * call-seq:
12682 * center(size, pad_string = ' ') -> new_string
12683 *
12684 * :include: doc/string/center.rdoc
12685 *
12686 */
12687
12688static VALUE
12689rb_str_center(int argc, VALUE *argv, VALUE str)
12690{
12691 return rb_str_justify(argc, argv, str, 'c');
12692}
12693
12694/*
12695 * call-seq:
12696 * partition(pattern) -> [pre_match, first_match, post_match]
12697 *
12698 * :include: doc/string/partition.rdoc
12699 *
12700 */
12701
12702static VALUE
12703rb_str_partition(VALUE str, VALUE sep)
12704{
12705 long pos;
12706
12707 sep = get_pat_quoted(sep, 0);
12708 if (RB_TYPE_P(sep, T_REGEXP)) {
12709 if (rb_reg_search(sep, str, 0, 0) < 0) {
12710 goto failed;
12711 }
12712 VALUE match = rb_backref_get();
12713
12714 pos = RMATCH_BEG(match, 0);
12715 sep = rb_str_subseq(str, pos, RMATCH_END(match, 0) - pos);
12716 }
12717 else {
12718 pos = rb_str_index(str, sep, 0);
12719 if (pos < 0) goto failed;
12720 }
12721
12722 long rpos = pos + RSTRING_LEN(sep);
12723 if (rpos > RSTRING_LEN(str)) goto failed;
12724 return rb_ary_new3(3, rb_str_subseq(str, 0, pos),
12725 sep,
12726 rb_str_subseq(str, rpos, RSTRING_LEN(str)-rpos));
12727
12728 failed:
12729 return rb_ary_new3(3, str_duplicate(rb_cString, str), str_new_empty_String(str), str_new_empty_String(str));
12730}
12731
12732/*
12733 * call-seq:
12734 * rpartition(pattern) -> [pre_match, last_match, post_match]
12735 *
12736 * :include: doc/string/rpartition.rdoc
12737 *
12738 */
12739
12740static VALUE
12741rb_str_rpartition(VALUE str, VALUE sep)
12742{
12743 long pos;
12744
12745 sep = get_pat_quoted(sep, 0);
12746 if (RB_TYPE_P(sep, T_REGEXP)) {
12747 pos = RSTRING_LEN(str);
12748 if (rb_reg_search(sep, str, pos, 1) < 0) {
12749 goto failed;
12750 }
12751 VALUE match = rb_backref_get();
12752
12753 pos = RMATCH_BEG(match, 0);
12754 sep = rb_str_subseq(str, pos, RMATCH_END(match, 0) - pos);
12755 }
12756 else {
12757 /* str may have been modified by #to_str above */
12758 pos = rb_str_sublen(str, RSTRING_LEN(str));
12759 pos = rb_str_rindex(str, sep, pos);
12760 if (pos < 0) {
12761 goto failed;
12762 }
12763 }
12764
12765 long rpos = pos + RSTRING_LEN(sep);
12766 if (rpos > RSTRING_LEN(str)) goto failed;
12767 return rb_ary_new3(3, rb_str_subseq(str, 0, pos),
12768 sep,
12769 rb_str_subseq(str, rpos, RSTRING_LEN(str)-rpos));
12770 failed:
12771 return rb_ary_new3(3, str_new_empty_String(str), str_new_empty_String(str), str_duplicate(rb_cString, str));
12772}
12773
12774/*
12775 * call-seq:
12776 * start_with?(*patterns) -> true or false
12777 *
12778 * :include: doc/string/start_with_p.rdoc
12779 *
12780 */
12781
12782static VALUE
12783rb_str_start_with(int argc, VALUE *argv, VALUE str)
12784{
12785 int i;
12786
12787 for (i=0; i<argc; i++) {
12788 VALUE tmp = argv[i];
12789 if (RB_TYPE_P(tmp, T_REGEXP)) {
12790 if (rb_reg_start_with_p(tmp, str))
12791 return Qtrue;
12792 }
12793 else {
12794 const char *p, *s, *e;
12795 long slen, tlen;
12796 rb_encoding *enc;
12797
12798 StringValue(tmp);
12799 enc = rb_enc_check(str, tmp);
12800 if ((tlen = RSTRING_LEN(tmp)) == 0) return Qtrue;
12801 if ((slen = RSTRING_LEN(str)) < tlen) continue;
12802 p = RSTRING_PTR(str);
12803 e = p + slen;
12804 s = p + tlen;
12805 if (!at_char_right_boundary(p, s, e, enc))
12806 continue;
12807 if (memcmp(p, RSTRING_PTR(tmp), tlen) == 0)
12808 return Qtrue;
12809 }
12810 }
12811 return Qfalse;
12812}
12813
12814/*
12815 * call-seq:
12816 * end_with?(*strings) -> true or false
12817 *
12818 * :include: doc/string/end_with_p.rdoc
12819 *
12820 */
12821
12822static VALUE
12823rb_str_end_with(int argc, VALUE *argv, VALUE str)
12824{
12825 int i;
12826
12827 for (i=0; i<argc; i++) {
12828 VALUE tmp = argv[i];
12829 const char *p, *s, *e;
12830 long slen, tlen;
12831 rb_encoding *enc;
12832
12833 StringValue(tmp);
12834 enc = rb_enc_check(str, tmp);
12835 if ((tlen = RSTRING_LEN(tmp)) == 0) return Qtrue;
12836 if ((slen = RSTRING_LEN(str)) < tlen) continue;
12837 p = RSTRING_PTR(str);
12838 e = p + slen;
12839 s = e - tlen;
12840 if (!at_char_boundary(p, s, e, enc))
12841 continue;
12842 if (memcmp(s, RSTRING_PTR(tmp), tlen) == 0)
12843 return Qtrue;
12844 }
12845 return Qfalse;
12846}
12847
12857static long
12858deleted_prefix_length(VALUE str, VALUE prefix)
12859{
12860 const char *strptr, *prefixptr;
12861 long olen, prefixlen;
12862 rb_encoding *enc = rb_enc_get(str);
12863
12864 StringValue(prefix);
12865
12866 if (!is_broken_string(prefix) ||
12867 !rb_enc_asciicompat(enc) ||
12868 !rb_enc_asciicompat(rb_enc_get(prefix))) {
12869 enc = rb_enc_check(str, prefix);
12870 }
12871
12872 /* return 0 if not start with prefix */
12873 prefixlen = RSTRING_LEN(prefix);
12874 if (prefixlen <= 0) return 0;
12875 olen = RSTRING_LEN(str);
12876 if (olen < prefixlen) return 0;
12877 strptr = RSTRING_PTR(str);
12878 prefixptr = RSTRING_PTR(prefix);
12879 if (memcmp(strptr, prefixptr, prefixlen) != 0) return 0;
12880 if (is_broken_string(prefix)) {
12881 if (!is_broken_string(str)) {
12882 /* prefix in a valid string cannot be broken */
12883 return 0;
12884 }
12885 const char *strend = strptr + olen;
12886 const char *after_prefix = strptr + prefixlen;
12887 if (!at_char_right_boundary(strptr, after_prefix, strend, enc)) {
12888 /* prefix does not end at char-boundary */
12889 return 0;
12890 }
12891 }
12892 /* prefix part in `str` also should be valid. */
12893
12894 return prefixlen;
12895}
12896
12897/*
12898 * call-seq:
12899 * delete_prefix!(prefix) -> self or nil
12900 *
12901 * Like String#delete_prefix, except that +self+ is modified in place;
12902 * returns +self+ if the prefix is removed, +nil+ otherwise.
12903 *
12904 * Related: see {Modifying}[rdoc-ref:String@Modifying].
12905 */
12906
12907static VALUE
12908rb_str_delete_prefix_bang(VALUE str, VALUE prefix)
12909{
12910 long prefixlen;
12911 str_modify_keep_cr(str);
12912
12913 prefixlen = deleted_prefix_length(str, prefix);
12914 if (prefixlen <= 0) return Qnil;
12915
12916 return rb_str_drop_bytes(str, prefixlen);
12917}
12918
12919/*
12920 * call-seq:
12921 * delete_prefix(prefix) -> new_string
12922 *
12923 * :include: doc/string/delete_prefix.rdoc
12924 *
12925 */
12926
12927static VALUE
12928rb_str_delete_prefix(VALUE str, VALUE prefix)
12929{
12930 long prefixlen;
12931
12932 prefixlen = deleted_prefix_length(str, prefix);
12933 if (prefixlen <= 0) return str_duplicate(rb_cString, str);
12934
12935 return rb_str_subseq(str, prefixlen, RSTRING_LEN(str) - prefixlen);
12936}
12937
12947static long
12948deleted_suffix_length(VALUE str, VALUE suffix)
12949{
12950 const char *strptr, *suffixptr;
12951 long olen, suffixlen;
12952 rb_encoding *enc;
12953
12954 StringValue(suffix);
12955 if (is_broken_string(suffix)) return 0;
12956 enc = rb_enc_check(str, suffix);
12957
12958 /* return 0 if not start with suffix */
12959 suffixlen = RSTRING_LEN(suffix);
12960 if (suffixlen <= 0) return 0;
12961 olen = RSTRING_LEN(str);
12962 if (olen < suffixlen) return 0;
12963 strptr = RSTRING_PTR(str);
12964 suffixptr = RSTRING_PTR(suffix);
12965 const char *strend = strptr + olen;
12966 const char *before_suffix = strend - suffixlen;
12967 if (memcmp(before_suffix, suffixptr, suffixlen) != 0) return 0;
12968 if (!at_char_boundary(strptr, before_suffix, strend, enc)) return 0;
12969
12970 return suffixlen;
12971}
12972
12973/*
12974 * call-seq:
12975 * delete_suffix!(suffix) -> self or nil
12976 *
12977 * Like String#delete_suffix, except that +self+ is modified in place;
12978 * returns +self+ if the suffix is removed, +nil+ otherwise.
12979 *
12980 * Related: see {Modifying}[rdoc-ref:String@Modifying].
12981 */
12982
12983static VALUE
12984rb_str_delete_suffix_bang(VALUE str, VALUE suffix)
12985{
12986 long suffixlen;
12987 str_modifiable(str);
12988
12989 suffixlen = deleted_suffix_length(str, suffix);
12990 if (suffixlen <= 0) return Qnil;
12991
12992 return str_shrink(str, RSTRING_LEN(str) - suffixlen);
12993}
12994
12995/*
12996 * call-seq:
12997 * delete_suffix(suffix) -> new_string
12998 *
12999 * :include: doc/string/delete_suffix.rdoc
13000 *
13001 */
13002
13003static VALUE
13004rb_str_delete_suffix(VALUE str, VALUE suffix)
13005{
13006 long suffixlen;
13007
13008 suffixlen = deleted_suffix_length(str, suffix);
13009 if (suffixlen <= 0) return str_duplicate(rb_cString, str);
13010
13011 return rb_str_subseq(str, 0, RSTRING_LEN(str) - suffixlen);
13012}
13013
13014void
13015rb_str_setter(VALUE val, ID id, VALUE *var)
13016{
13017 if (!NIL_P(val) && !RB_TYPE_P(val, T_STRING)) {
13018 rb_raise(rb_eTypeError, "value of %"PRIsVALUE" must be String", rb_id2str(id));
13019 }
13020 *var = val;
13021}
13022
13023static void
13024nil_setter_warning(ID id)
13025{
13026 rb_warn_deprecated("non-nil '%"PRIsVALUE"'", NULL, rb_id2str(id));
13027}
13028
13029void
13030rb_deprecated_str_setter(VALUE val, ID id, VALUE *var)
13031{
13032 rb_str_setter(val, id, var);
13033 if (!NIL_P(*var)) {
13034 nil_setter_warning(id);
13035 }
13036}
13037
13038static void
13039rb_fs_setter(VALUE val, ID id, VALUE *var)
13040{
13041 val = rb_fs_check(val);
13042 if (!val) {
13043 rb_raise(rb_eTypeError,
13044 "value of %"PRIsVALUE" must be String or Regexp",
13045 rb_id2str(id));
13046 }
13047 if (!NIL_P(val)) {
13048 nil_setter_warning(id);
13049 }
13050 *var = val;
13051}
13052
13053
13054/*
13055 * call-seq:
13056 * force_encoding(encoding) -> self
13057 *
13058 * :include: doc/string/force_encoding.rdoc
13059 *
13060 */
13061
13062static VALUE
13063rb_str_force_encoding(VALUE str, VALUE enc)
13064{
13065 str_modifiable(str);
13066
13067 rb_encoding *encoding = rb_to_encoding(enc);
13068 int idx = rb_enc_to_index(encoding);
13069
13070 // If the encoding is unchanged, we do nothing.
13071 if (ENCODING_GET(str) == idx) {
13072 return str;
13073 }
13074
13075 rb_enc_associate_index(str, idx);
13076
13077 // If the coderange was 7bit and the new encoding is ASCII-compatible
13078 // we can keep the coderange.
13079 if (ENC_CODERANGE(str) == ENC_CODERANGE_7BIT && encoding && rb_enc_asciicompat(encoding)) {
13080 return str;
13081 }
13082
13084 return str;
13085}
13086
13087/*
13088 * call-seq:
13089 * b -> new_string
13090 *
13091 * :include: doc/string/b.rdoc
13092 *
13093 */
13094
13095static VALUE
13096rb_str_b(VALUE str)
13097{
13098 VALUE str2;
13099 if (STR_EMBED_P(str)) {
13100 str2 = str_alloc_embed(rb_cString, RSTRING_LEN(str) + TERM_LEN(str));
13101 }
13102 else {
13103 str2 = str_alloc_heap(rb_cString);
13104 }
13105 str_replace_shared_without_enc(str2, str);
13106
13107 if (rb_enc_asciicompat(STR_ENC_GET(str))) {
13108 // BINARY strings can never be broken; they're either 7-bit ASCII or VALID.
13109 // If we know the receiver's code range then we know the result's code range.
13110 int cr = ENC_CODERANGE(str);
13111 switch (cr) {
13112 case ENC_CODERANGE_7BIT:
13114 break;
13118 break;
13119 default:
13120 ENC_CODERANGE_CLEAR(str2);
13121 break;
13122 }
13123 }
13124
13125 return str2;
13126}
13127
13128/* Defined as a leaf builtin in string.rb, so this must never raise or call into Ruby. */
13129static VALUE
13130rb_str_valid_encoding_p(VALUE str)
13131{
13132 int cr = rb_enc_str_coderange(str);
13133
13134 return RBOOL(cr != ENC_CODERANGE_BROKEN);
13135}
13136
13137/* Defined as a leaf builtin in string.rb, so this must never raise or call into Ruby. */
13138static VALUE
13139rb_str_is_ascii_only_p(VALUE str)
13140{
13141 int cr = rb_enc_str_coderange(str);
13142
13143 return RBOOL(cr == ENC_CODERANGE_7BIT);
13144}
13145
13146VALUE
13148{
13149 static const char ellipsis[] = "...";
13150 const long ellipsislen = sizeof(ellipsis) - 1;
13151 rb_encoding *const enc = rb_enc_get(str);
13152 const long blen = RSTRING_LEN(str);
13153 const char *const p = RSTRING_PTR(str), *e = p + blen;
13154 VALUE estr, ret = 0;
13155
13156 if (len < 0) rb_raise(rb_eIndexError, "negative length %ld", len);
13157 if (len * rb_enc_mbminlen(enc) >= blen ||
13158 (e = rb_enc_nth(p, e, len, enc)) - p == blen) {
13159 ret = str;
13160 }
13161 else if (len <= ellipsislen ||
13162 !(e = rb_enc_step_back(p, e, e, len = ellipsislen, enc))) {
13163 if (rb_enc_asciicompat(enc)) {
13164 ret = rb_str_new(ellipsis, len);
13165 rb_enc_associate(ret, enc);
13166 }
13167 else {
13168 estr = rb_usascii_str_new(ellipsis, len);
13169 ret = rb_str_encode(estr, rb_enc_from_encoding(enc), 0, Qnil);
13170 }
13171 }
13172 else if (ret = rb_str_subseq(str, 0, e - p), rb_enc_asciicompat(enc)) {
13173 rb_str_cat(ret, ellipsis, ellipsislen);
13174 }
13175 else {
13176 estr = rb_str_encode(rb_usascii_str_new(ellipsis, ellipsislen),
13177 rb_enc_from_encoding(enc), 0, Qnil);
13178 rb_str_append(ret, estr);
13179 }
13180 return ret;
13181}
13182
13183static VALUE
13184str_compat_and_valid(VALUE str, rb_encoding *enc)
13185{
13186 int cr;
13187 str = StringValue(str);
13188 cr = rb_enc_str_coderange(str);
13189 if (cr == ENC_CODERANGE_BROKEN) {
13190 rb_raise(rb_eArgError, "replacement must be valid byte sequence '%+"PRIsVALUE"'", str);
13191 }
13192 else {
13193 rb_encoding *e = STR_ENC_GET(str);
13194 if (cr == ENC_CODERANGE_7BIT ? rb_enc_mbminlen(enc) != 1 : enc != e) {
13195 rb_raise(rb_eEncCompatError, "incompatible character encodings: %s and %s",
13196 rb_enc_inspect_name(enc), rb_enc_inspect_name(e));
13197 }
13198 }
13199 return str;
13200}
13201
13202static VALUE enc_str_scrub(rb_encoding *enc, VALUE str, VALUE repl, int cr);
13203
13204VALUE
13206{
13207 rb_encoding *enc = STR_ENC_GET(str);
13208 return enc_str_scrub(enc, str, repl, ENC_CODERANGE(str));
13209}
13210
13211VALUE
13212rb_enc_str_scrub(rb_encoding *enc, VALUE str, VALUE repl)
13213{
13214 int cr = ENC_CODERANGE_UNKNOWN;
13215 if (enc == STR_ENC_GET(str)) {
13216 /* cached coderange makes sense only when enc equals the
13217 * actual encoding of str */
13218 cr = ENC_CODERANGE(str);
13219 }
13220 return enc_str_scrub(enc, str, repl, cr);
13221}
13222
13223static VALUE
13224enc_str_scrub(rb_encoding *enc, VALUE str, VALUE repl, int cr)
13225{
13226 int encidx;
13227 VALUE buf = Qnil;
13228 const char *rep, *p, *e, *p1, *sp;
13229 long replen = -1;
13230 long slen;
13231
13232 if (rb_block_given_p()) {
13233 if (!NIL_P(repl))
13234 rb_raise(rb_eArgError, "both of block and replacement given");
13235 replen = 0;
13236 }
13237
13238 if (ENC_CODERANGE_CLEAN_P(cr))
13239 return Qnil;
13240
13241 if (!NIL_P(repl)) {
13242 repl = str_compat_and_valid(repl, enc);
13243 }
13244
13245 if (rb_enc_dummy_p(enc)) {
13246 return Qnil;
13247 }
13248 encidx = rb_enc_to_index(enc);
13249
13250#define DEFAULT_REPLACE_CHAR(str) do { \
13251 RBIMPL_ATTR_NONSTRING() static const char replace[sizeof(str)-1] = str; \
13252 rep = replace; replen = (int)sizeof(replace); \
13253 } while (0)
13254
13255 slen = RSTRING_LEN(str);
13256 p = RSTRING_PTR(str);
13257 e = RSTRING_END(str);
13258 p1 = p;
13259 sp = p;
13260
13261 if (rb_enc_asciicompat(enc)) {
13262 int rep7bit_p;
13263 if (!replen) {
13264 rep = NULL;
13265 rep7bit_p = FALSE;
13266 }
13267 else if (!NIL_P(repl)) {
13268 rep = RSTRING_PTR(repl);
13269 replen = RSTRING_LEN(repl);
13270 rep7bit_p = (ENC_CODERANGE(repl) == ENC_CODERANGE_7BIT);
13271 }
13272 else if (encidx == rb_utf8_encindex()) {
13273 DEFAULT_REPLACE_CHAR("\xEF\xBF\xBD");
13274 rep7bit_p = FALSE;
13275 }
13276 else {
13277 DEFAULT_REPLACE_CHAR("?");
13278 rep7bit_p = TRUE;
13279 }
13280 cr = ENC_CODERANGE_7BIT;
13281
13282 p = search_nonascii(p, e);
13283 if (!p) {
13284 p = e;
13285 }
13286 while (p < e) {
13287 int ret = rb_enc_precise_mbclen(p, e, enc);
13288 if (MBCLEN_NEEDMORE_P(ret)) {
13289 break;
13290 }
13291 else if (MBCLEN_CHARFOUND_P(ret)) {
13293 p += MBCLEN_CHARFOUND_LEN(ret);
13294 /* After a multibyte character, fast-skip the following ASCII run. */
13295 p = search_nonascii(p, e);
13296 if (!p) {
13297 p = e;
13298 break;
13299 }
13300 }
13301 else if (MBCLEN_INVALID_P(ret)) {
13302 /*
13303 * p1~p: valid ascii/multibyte chars
13304 * p ~e: invalid bytes + unknown bytes
13305 */
13306 long clen = rb_enc_mbmaxlen(enc);
13307 if (NIL_P(buf)) buf = rb_str_buf_new(RSTRING_LEN(str));
13308 if (p > p1) {
13309 rb_str_buf_cat(buf, p1, p - p1);
13310 }
13311
13312 if (e - p < clen) clen = e - p;
13313 if (clen <= 2) {
13314 clen = 1;
13315 }
13316 else {
13317 const char *q = p;
13318 clen--;
13319 for (; clen > 1; clen--) {
13320 ret = rb_enc_precise_mbclen(q, q + clen, enc);
13321 if (MBCLEN_NEEDMORE_P(ret)) break;
13322 if (MBCLEN_INVALID_P(ret)) continue;
13324 }
13325 }
13326 if (rep) {
13327 rb_str_buf_cat(buf, rep, replen);
13328 if (!rep7bit_p) cr = ENC_CODERANGE_VALID;
13329 }
13330 else {
13331 repl = rb_yield(rb_enc_str_new(p, clen, enc));
13332 str_mod_check(str, sp, slen);
13333 repl = str_compat_and_valid(repl, enc);
13334 rb_str_buf_cat(buf, RSTRING_PTR(repl), RSTRING_LEN(repl));
13337 }
13338 p += clen;
13339 p1 = p;
13340 p = search_nonascii(p, e);
13341 if (!p) {
13342 p = e;
13343 break;
13344 }
13345 }
13346 else {
13348 }
13349 }
13350 if (NIL_P(buf)) {
13351 if (p == e) {
13352 ENC_CODERANGE_SET(str, cr);
13353 return Qnil;
13354 }
13355 buf = rb_str_buf_new(RSTRING_LEN(str));
13356 }
13357 if (p1 < p) {
13358 rb_str_buf_cat(buf, p1, p - p1);
13359 }
13360 if (p < e) {
13361 if (rep) {
13362 rb_str_buf_cat(buf, rep, replen);
13363 if (!rep7bit_p) cr = ENC_CODERANGE_VALID;
13364 }
13365 else {
13366 repl = rb_yield(rb_enc_str_new(p, e-p, enc));
13367 str_mod_check(str, sp, slen);
13368 repl = str_compat_and_valid(repl, enc);
13369 rb_str_buf_cat(buf, RSTRING_PTR(repl), RSTRING_LEN(repl));
13372 }
13373 }
13374 }
13375 else {
13376 /* ASCII incompatible */
13377 long mbminlen = rb_enc_mbminlen(enc);
13378 if (!replen) {
13379 rep = NULL;
13380 }
13381 else if (!NIL_P(repl)) {
13382 rep = RSTRING_PTR(repl);
13383 replen = RSTRING_LEN(repl);
13384 }
13385 else if (encidx == ENCINDEX_UTF_16BE) {
13386 DEFAULT_REPLACE_CHAR("\xFF\xFD");
13387 }
13388 else if (encidx == ENCINDEX_UTF_16LE) {
13389 DEFAULT_REPLACE_CHAR("\xFD\xFF");
13390 }
13391 else if (encidx == ENCINDEX_UTF_32BE) {
13392 DEFAULT_REPLACE_CHAR("\x00\x00\xFF\xFD");
13393 }
13394 else if (encidx == ENCINDEX_UTF_32LE) {
13395 DEFAULT_REPLACE_CHAR("\xFD\xFF\x00\x00");
13396 }
13397 else {
13398 DEFAULT_REPLACE_CHAR("?");
13399 }
13400
13401 while (p < e) {
13402 int ret = rb_enc_precise_mbclen(p, e, enc);
13403 if (MBCLEN_NEEDMORE_P(ret)) {
13404 break;
13405 }
13406 else if (MBCLEN_CHARFOUND_P(ret)) {
13407 p += MBCLEN_CHARFOUND_LEN(ret);
13408 }
13409 else if (MBCLEN_INVALID_P(ret)) {
13410 const char *q = p;
13411 long clen = rb_enc_mbmaxlen(enc);
13412 if (NIL_P(buf)) buf = rb_str_buf_new(RSTRING_LEN(str));
13413 if (p > p1) rb_str_buf_cat(buf, p1, p - p1);
13414
13415 if (e - p < clen) clen = e - p;
13416 if (clen <= mbminlen * 2) {
13417 clen = mbminlen;
13418 }
13419 else {
13420 clen -= mbminlen;
13421 for (; clen > mbminlen; clen-=mbminlen) {
13422 ret = rb_enc_precise_mbclen(q, q + clen, enc);
13423 if (MBCLEN_NEEDMORE_P(ret)) break;
13424 if (MBCLEN_INVALID_P(ret)) continue;
13426 }
13427 }
13428 if (rep) {
13429 rb_str_buf_cat(buf, rep, replen);
13430 }
13431 else {
13432 repl = rb_yield(rb_enc_str_new(p, clen, enc));
13433 str_mod_check(str, sp, slen);
13434 repl = str_compat_and_valid(repl, enc);
13435 rb_str_buf_cat(buf, RSTRING_PTR(repl), RSTRING_LEN(repl));
13436 }
13437 p += clen;
13438 p1 = p;
13439 }
13440 else {
13442 }
13443 }
13444 if (NIL_P(buf)) {
13445 if (p == e) {
13447 return Qnil;
13448 }
13449 buf = rb_str_buf_new(RSTRING_LEN(str));
13450 }
13451 if (p1 < p) {
13452 rb_str_buf_cat(buf, p1, p - p1);
13453 }
13454 if (p < e) {
13455 if (rep) {
13456 rb_str_buf_cat(buf, rep, replen);
13457 }
13458 else {
13459 repl = rb_yield(rb_enc_str_new(p, e-p, enc));
13460 str_mod_check(str, sp, slen);
13461 repl = str_compat_and_valid(repl, enc);
13462 rb_str_buf_cat(buf, RSTRING_PTR(repl), RSTRING_LEN(repl));
13463 }
13464 }
13466 }
13467 ENCODING_CODERANGE_SET(buf, rb_enc_to_index(enc), cr);
13468 return buf;
13469}
13470
13471/*
13472 * call-seq:
13473 * scrub(replacement_string = default_replacement_string) -> new_string
13474 * scrub{|sequence| ... } -> new_string
13475 *
13476 * :include: doc/string/scrub.rdoc
13477 *
13478 */
13479static VALUE
13480str_scrub(int argc, VALUE *argv, VALUE str)
13481{
13482 VALUE repl = argc ? (rb_check_arity(argc, 0, 1), argv[0]) : Qnil;
13483 VALUE new = rb_str_scrub(str, repl);
13484 return NIL_P(new) ? str_duplicate(rb_cString, str): new;
13485}
13486
13487/*
13488 * call-seq:
13489 * scrub!(replacement_string = default_replacement_string) -> self
13490 * scrub!{|sequence| ... } -> self
13491 *
13492 * Like String#scrub, except that:
13493 *
13494 * - Any replacements are made in +self+.
13495 * - Returns +self+.
13496 *
13497 * Related: see {Modifying}[rdoc-ref:String@Modifying].
13498 *
13499 */
13500static VALUE
13501str_scrub_bang(int argc, VALUE *argv, VALUE str)
13502{
13503 VALUE repl = argc ? (rb_check_arity(argc, 0, 1), argv[0]) : Qnil;
13504 VALUE new = rb_str_scrub(str, repl);
13505 if (!NIL_P(new)) rb_str_replace(str, new);
13506 return str;
13507}
13508
13509static ID id_normalize;
13510static ID id_normalized_p;
13511static VALUE mUnicodeNormalize;
13512
13513static VALUE
13514unicode_normalize_common(int argc, VALUE *argv, VALUE str, ID id)
13515{
13516 static int UnicodeNormalizeRequired = 0;
13517 VALUE argv2[2];
13518
13519 if (!UnicodeNormalizeRequired) {
13520 rb_require("unicode_normalize/normalize.rb");
13521 UnicodeNormalizeRequired = 1;
13522 }
13523 argv2[0] = str;
13524 if (rb_check_arity(argc, 0, 1)) argv2[1] = argv[0];
13525 return rb_funcallv(mUnicodeNormalize, id, argc+1, argv2);
13526}
13527
13528/*
13529 * call-seq:
13530 * unicode_normalize(form = :nfc) -> string
13531 *
13532 * :include: doc/string/unicode_normalize.rdoc
13533 *
13534 */
13535static VALUE
13536rb_str_unicode_normalize(int argc, VALUE *argv, VALUE str)
13537{
13538 return unicode_normalize_common(argc, argv, str, id_normalize);
13539}
13540
13541/*
13542 * call-seq:
13543 * unicode_normalize!(form = :nfc) -> self
13544 *
13545 * Like String#unicode_normalize, except that the normalization
13546 * is performed on +self+ (not on a copy of +self+).
13547 *
13548 * Related: see {Modifying}[rdoc-ref:String@Modifying].
13549 *
13550 */
13551static VALUE
13552rb_str_unicode_normalize_bang(int argc, VALUE *argv, VALUE str)
13553{
13554 return rb_str_replace(str, unicode_normalize_common(argc, argv, str, id_normalize));
13555}
13556
13557/* call-seq:
13558 * unicode_normalized?(form = :nfc) -> true or false
13559 *
13560 * Returns whether +self+ is in the given +form+ of Unicode normalization;
13561 * see String#unicode_normalize.
13562 *
13563 * The +form+ must be one of +:nfc+, +:nfd+, +:nfkc+, or +:nfkd+.
13564 *
13565 * Examples:
13566 *
13567 * "a\u0300".unicode_normalized? # => false
13568 * "a\u0300".unicode_normalized?(:nfd) # => true
13569 * "\u00E0".unicode_normalized? # => true
13570 * "\u00E0".unicode_normalized?(:nfd) # => false
13571 *
13572 *
13573 * Raises an exception if +self+ is not in a Unicode encoding:
13574 *
13575 * s = "\xE0".force_encoding(Encoding::ISO_8859_1)
13576 * s.unicode_normalized? # Raises Encoding::CompatibilityError
13577 *
13578 * Related: see {Querying}[rdoc-ref:String@Querying].
13579 */
13580static VALUE
13581rb_str_unicode_normalized_p(int argc, VALUE *argv, VALUE str)
13582{
13583 return unicode_normalize_common(argc, argv, str, id_normalized_p);
13584}
13585
13586/**********************************************************************
13587 * Document-class: Symbol
13588 *
13589 * A +Symbol+ object represents a named identifier inside the Ruby interpreter.
13590 *
13591 * You can create a +Symbol+ object explicitly with:
13592 *
13593 * - A {symbol literal}[rdoc-ref:syntax/literals.rdoc@Symbol+Literals].
13594 *
13595 * The same +Symbol+ object will be
13596 * created for a given name or string for the duration of a program's
13597 * execution, regardless of the context or meaning of that name. Thus
13598 * if <code>Fred</code> is a constant in one context, a method in
13599 * another, and a class in a third, the +Symbol+ <code>:Fred</code>
13600 * will be the same object in all three contexts.
13601 *
13602 * module One
13603 * class Fred
13604 * end
13605 * $f1 = :Fred
13606 * end
13607 * module Two
13608 * Fred = 1
13609 * $f2 = :Fred
13610 * end
13611 * def Fred()
13612 * end
13613 * $f3 = :Fred
13614 * $f1.object_id #=> 2514190
13615 * $f2.object_id #=> 2514190
13616 * $f3.object_id #=> 2514190
13617 *
13618 * Constant, method, and variable names are returned as symbols:
13619 *
13620 * module One
13621 * Two = 2
13622 * def three; 3 end
13623 * @four = 4
13624 * @@five = 5
13625 * $six = 6
13626 * end
13627 * seven = 7
13628 *
13629 * One.constants
13630 * # => [:Two]
13631 * One.instance_methods(true)
13632 * # => [:three]
13633 * One.instance_variables
13634 * # => [:@four]
13635 * One.class_variables
13636 * # => [:@@five]
13637 * global_variables.grep(/six/)
13638 * # => [:$six]
13639 * local_variables
13640 * # => [:seven]
13641 *
13642 * A +Symbol+ object differs from a String object in that
13643 * a +Symbol+ object represents an identifier, while a String object
13644 * represents text or data.
13645 *
13646 * == What's Here
13647 *
13648 * First, what's elsewhere. Class +Symbol+:
13649 *
13650 * - Inherits from {class Object}[rdoc-ref:Object@Whats+Here].
13651 * - Includes {module Comparable}[rdoc-ref:Comparable@Whats+Here].
13652 *
13653 * Here, class +Symbol+ provides methods that are useful for:
13654 *
13655 * - {Querying}[rdoc-ref:Symbol@Methods+for+Querying]
13656 * - {Comparing}[rdoc-ref:Symbol@Methods+for+Comparing]
13657 * - {Converting}[rdoc-ref:Symbol@Methods+for+Converting]
13658 *
13659 * === Methods for Querying
13660 *
13661 * - ::all_symbols: Returns an array of the symbols currently in Ruby's symbol table.
13662 * - #=~: Returns the index of the first substring in symbol that matches a
13663 * given Regexp or other object; returns +nil+ if no match is found.
13664 * - #[], #slice : Returns a substring of symbol
13665 * determined by a given index, start/length, or range, or string.
13666 * - #empty?: Returns +true+ if +self.length+ is zero; +false+ otherwise.
13667 * - #encoding: Returns the Encoding object that represents the encoding
13668 * of symbol.
13669 * - #end_with?: Returns +true+ if symbol ends with
13670 * any of the given strings.
13671 * - #match: Returns a MatchData object if symbol
13672 * matches a given Regexp; +nil+ otherwise.
13673 * - #match?: Returns +true+ if symbol
13674 * matches a given Regexp; +false+ otherwise.
13675 * - #length, #size: Returns the number of characters in symbol.
13676 * - #start_with?: Returns +true+ if symbol starts with
13677 * any of the given strings.
13678 *
13679 * === Methods for Comparing
13680 *
13681 * - #<=>: Returns -1, 0, or 1 as a given symbol is smaller than, equal to,
13682 * or larger than symbol.
13683 * - #==, #===: Returns +true+ if a given symbol has the same content and
13684 * encoding.
13685 * - #casecmp: Ignoring case, returns -1, 0, or 1 as a given
13686 * symbol is smaller than, equal to, or larger than symbol.
13687 * - #casecmp?: Returns +true+ if symbol is equal to a given symbol
13688 * after Unicode case folding; +false+ otherwise.
13689 *
13690 * === Methods for Converting
13691 *
13692 * - #capitalize: Returns symbol with the first character upcased
13693 * and all other characters downcased.
13694 * - #downcase: Returns symbol with all characters downcased.
13695 * - #inspect: Returns the string representation of +self+ as a symbol literal.
13696 * - #name: Returns the frozen string corresponding to symbol.
13697 * - #succ, #next: Returns the symbol that is the successor to symbol.
13698 * - #swapcase: Returns symbol with all upcase characters downcased
13699 * and all downcase characters upcased.
13700 * - #to_proc: Returns a Proc object which responds to the method named by symbol.
13701 * - #to_s, #id2name: Returns the string corresponding to +self+.
13702 * - #to_sym, #intern: Returns +self+.
13703 * - #upcase: Returns symbol with all characters upcased.
13704 *
13705 */
13706
13707
13708/*
13709 * call-seq:
13710 * self == other -> true or false
13711 *
13712 * Returns whether +other+ is the same object as +self+.
13713 */
13714
13715#define sym_equal rb_obj_equal
13716
13717static int
13718sym_printable(const char *s, const char *send, rb_encoding *enc)
13719{
13720 while (s < send) {
13721 int n;
13722 int c = rb_enc_precise_mbclen(s, send, enc);
13723
13724 if (!MBCLEN_CHARFOUND_P(c)) return FALSE;
13725 n = MBCLEN_CHARFOUND_LEN(c);
13726 c = rb_enc_mbc_to_codepoint(s, send, enc);
13727 if (!rb_enc_isprint(c, enc)) return FALSE;
13728 s += n;
13729 }
13730 return TRUE;
13731}
13732
13733int
13734rb_str_symname_p(VALUE sym)
13735{
13736 rb_encoding *enc;
13737 const char *ptr;
13738 long len;
13739 rb_encoding *resenc = rb_default_internal_encoding();
13740
13741 if (resenc == NULL) resenc = rb_default_external_encoding();
13742 enc = STR_ENC_GET(sym);
13743 ptr = RSTRING_PTR(sym);
13744 len = RSTRING_LEN(sym);
13745 if ((resenc != enc && !rb_str_is_ascii_only_p(sym)) || len != (long)strlen(ptr) ||
13746 !rb_enc_symname2_p(ptr, len, enc) || !sym_printable(ptr, ptr + len, enc)) {
13747 return FALSE;
13748 }
13749 return TRUE;
13750}
13751
13752VALUE
13753rb_str_quote_unprintable(VALUE str)
13754{
13755 rb_encoding *enc;
13756 const char *ptr;
13757 long len;
13758 rb_encoding *resenc;
13759
13760 Check_Type(str, T_STRING);
13761 resenc = rb_default_internal_encoding();
13762 if (resenc == NULL) resenc = rb_default_external_encoding();
13763 enc = STR_ENC_GET(str);
13764 ptr = RSTRING_PTR(str);
13765 len = RSTRING_LEN(str);
13766 if ((resenc != enc && !rb_str_is_ascii_only_p(str)) ||
13767 !sym_printable(ptr, ptr + len, enc)) {
13768 return rb_str_escape(str);
13769 }
13770 return str;
13771}
13772
13773VALUE
13774rb_id_quote_unprintable(ID id)
13775{
13776 VALUE str = rb_id2str(id);
13777 if (!rb_str_symname_p(str)) {
13778 return rb_str_escape(str);
13779 }
13780 return str;
13781}
13782
13783/*
13784 * call-seq:
13785 * inspect -> string
13786 *
13787 * Returns a string representation of +self+ (including the leading colon):
13788 *
13789 * :foo.inspect # => ":foo"
13790 *
13791 * Related: Symbol#to_s, Symbol#name.
13792 *
13793 */
13794
13795static VALUE
13796sym_inspect(VALUE sym)
13797{
13798 VALUE str = rb_sym2str(sym);
13799 const char *ptr;
13800 long len;
13801 char *dest;
13802
13803 if (!rb_str_symname_p(str)) {
13804 str = rb_str_inspect(str);
13805 len = RSTRING_LEN(str);
13806 rb_str_resize(str, len + 1);
13807 dest = RSTRING_PTR(str);
13808 memmove(dest + 1, dest, len);
13809 }
13810 else {
13811 rb_encoding *enc = STR_ENC_GET(str);
13812 VALUE orig_str = str;
13813
13814 len = RSTRING_LEN(orig_str);
13815 str = rb_enc_str_new(0, len + 1, enc);
13816
13817 // Get data pointer after allocation
13818 ptr = RSTRING_PTR(orig_str);
13819 dest = RSTRING_PTR(str);
13820 memcpy(dest + 1, ptr, len);
13821
13822 RB_GC_GUARD(orig_str);
13823 }
13824 dest[0] = ':';
13825
13827
13828 return str;
13829}
13830
13831VALUE
13833{
13834 return rb_sym2str(sym);
13835}
13836
13837VALUE
13838rb_sym_proc_call(ID mid, int argc, const VALUE *argv, int kw_splat, VALUE passed_proc)
13839{
13840 VALUE obj;
13841
13842 if (argc < 1) {
13843 rb_raise(rb_eArgError, "no receiver given");
13844 }
13845 obj = argv[0];
13846 return rb_funcall_with_block_kw(obj, mid, argc - 1, argv + 1, passed_proc, kw_splat);
13847}
13848
13849/*
13850 * call-seq:
13851 * succ
13852 *
13853 * Equivalent to <tt>self.to_s.succ.to_sym</tt>:
13854 *
13855 * :foo.succ # => :fop
13856 *
13857 * Related: String#succ.
13858 */
13859
13860static VALUE
13861sym_succ(VALUE sym)
13862{
13863 return rb_str_intern(rb_str_succ(rb_sym2str(sym)));
13864}
13865
13866/*
13867 * call-seq:
13868 * self <=> other -> -1, 0, 1, or nil
13869 *
13870 * Compares +self+ and +other+, using String#<=>.
13871 *
13872 * Returns:
13873 *
13874 * - <tt>self.to_s <=> other.to_s</tt>, if +other+ is a symbol.
13875 * - +nil+, otherwise.
13876 *
13877 * Examples:
13878 *
13879 * :bar <=> :foo # => -1
13880 * :foo <=> :foo # => 0
13881 * :foo <=> :bar # => 1
13882 * :foo <=> 'bar' # => nil
13883 *
13884 * \Class \Symbol includes module Comparable,
13885 * each of whose methods uses Symbol#<=> for comparison.
13886 *
13887 * Related: String#<=>.
13888 */
13889
13890static VALUE
13891sym_cmp(VALUE sym, VALUE other)
13892{
13893 if (!SYMBOL_P(other)) {
13894 return Qnil;
13895 }
13896 return rb_str_cmp_m(rb_sym2str(sym), rb_sym2str(other));
13897}
13898
13899/*
13900 * call-seq:
13901 * casecmp(object) -> -1, 0, 1, or nil
13902 *
13903 * :include: doc/symbol/casecmp.rdoc
13904 *
13905 */
13906
13907static VALUE
13908sym_casecmp(VALUE sym, VALUE other)
13909{
13910 if (!SYMBOL_P(other)) {
13911 return Qnil;
13912 }
13913 return str_casecmp(rb_sym2str(sym), rb_sym2str(other));
13914}
13915
13916/*
13917 * call-seq:
13918 * casecmp?(object) -> true, false, or nil
13919 *
13920 * :include: doc/symbol/casecmp_p.rdoc
13921 *
13922 */
13923
13924static VALUE
13925sym_casecmp_p(VALUE sym, VALUE other)
13926{
13927 if (!SYMBOL_P(other)) {
13928 return Qnil;
13929 }
13930 return str_casecmp_p(rb_sym2str(sym), rb_sym2str(other));
13931}
13932
13933/*
13934 * call-seq:
13935 * self =~ other -> integer or nil
13936 *
13937 * Equivalent to <tt>self.to_s =~ other</tt>,
13938 * including possible updates to global variables;
13939 * see String#=~.
13940 *
13941 */
13942
13943static VALUE
13944sym_match(VALUE sym, VALUE other)
13945{
13946 return rb_str_match(rb_sym2str(sym), other);
13947}
13948
13949/*
13950 * call-seq:
13951 * match(pattern, offset = 0) -> matchdata or nil
13952 * match(pattern, offset = 0) {|matchdata| } -> object
13953 *
13954 * Equivalent to <tt>self.to_s.match</tt>,
13955 * including possible updates to global variables;
13956 * see String#match.
13957 *
13958 */
13959
13960static VALUE
13961sym_match_m(int argc, VALUE *argv, VALUE sym)
13962{
13963 return rb_str_match_m(argc, argv, rb_sym2str(sym));
13964}
13965
13966/*
13967 * call-seq:
13968 * match?(pattern, offset) -> true or false
13969 *
13970 * Equivalent to <tt>sym.to_s.match?</tt>;
13971 * see String#match.
13972 *
13973 */
13974
13975static VALUE
13976sym_match_m_p(int argc, VALUE *argv, VALUE sym)
13977{
13978 return rb_str_match_m_p(argc, argv, sym);
13979}
13980
13981/*
13982 * call-seq:
13983 * self[offset] -> string or nil
13984 * self[offset, size] -> string or nil
13985 * self[range] -> string or nil
13986 * self[regexp, capture = 0] -> string or nil
13987 * self[substring] -> string or nil
13988 *
13989 * Equivalent to <tt>symbol.to_s[]</tt>; see String#[].
13990 *
13991 */
13992
13993static VALUE
13994sym_aref(int argc, VALUE *argv, VALUE sym)
13995{
13996 return rb_str_aref_m(argc, argv, rb_sym2str(sym));
13997}
13998
13999/*
14000 * call-seq:
14001 * length -> integer
14002 *
14003 * Equivalent to <tt>self.to_s.length</tt>; see String#length.
14004 */
14005
14006static VALUE
14007sym_length(VALUE sym)
14008{
14009 return rb_str_length(rb_sym2str(sym));
14010}
14011
14012/*
14013 * call-seq:
14014 * upcase(mapping) -> symbol
14015 *
14016 * Equivalent to <tt>sym.to_s.upcase.to_sym</tt>.
14017 *
14018 * See String#upcase.
14019 *
14020 */
14021
14022static VALUE
14023sym_upcase(int argc, VALUE *argv, VALUE sym)
14024{
14025 return rb_str_intern(rb_str_upcase(argc, argv, rb_sym2str(sym)));
14026}
14027
14028/*
14029 * call-seq:
14030 * downcase(mapping) -> symbol
14031 *
14032 * Equivalent to <tt>sym.to_s.downcase.to_sym</tt>.
14033 *
14034 * See String#downcase.
14035 *
14036 * Related: Symbol#upcase.
14037 *
14038 */
14039
14040static VALUE
14041sym_downcase(int argc, VALUE *argv, VALUE sym)
14042{
14043 return rb_str_intern(rb_str_downcase(argc, argv, rb_sym2str(sym)));
14044}
14045
14046/*
14047 * call-seq:
14048 * capitalize(mapping) -> symbol
14049 *
14050 * Equivalent to <tt>sym.to_s.capitalize.to_sym</tt>.
14051 *
14052 * See String#capitalize.
14053 *
14054 */
14055
14056static VALUE
14057sym_capitalize(int argc, VALUE *argv, VALUE sym)
14058{
14059 return rb_str_intern(rb_str_capitalize(argc, argv, rb_sym2str(sym)));
14060}
14061
14062/*
14063 * call-seq:
14064 * swapcase(mapping) -> symbol
14065 *
14066 * Equivalent to <tt>sym.to_s.swapcase.to_sym</tt>.
14067 *
14068 * See String#swapcase.
14069 *
14070 */
14071
14072static VALUE
14073sym_swapcase(int argc, VALUE *argv, VALUE sym)
14074{
14075 return rb_str_intern(rb_str_swapcase(argc, argv, rb_sym2str(sym)));
14076}
14077
14078/*
14079 * call-seq:
14080 * start_with?(*string_or_regexp) -> true or false
14081 *
14082 * Equivalent to <tt>self.to_s.start_with?</tt>; see String#start_with?.
14083 *
14084 */
14085
14086static VALUE
14087sym_start_with(int argc, VALUE *argv, VALUE sym)
14088{
14089 return rb_str_start_with(argc, argv, rb_sym2str(sym));
14090}
14091
14092/*
14093 * call-seq:
14094 * end_with?(*strings) -> true or false
14095 *
14096 *
14097 * Equivalent to <tt>self.to_s.end_with?</tt>; see String#end_with?.
14098 *
14099 */
14100
14101static VALUE
14102sym_end_with(int argc, VALUE *argv, VALUE sym)
14103{
14104 return rb_str_end_with(argc, argv, rb_sym2str(sym));
14105}
14106
14107/*
14108 * call-seq:
14109 * encoding -> encoding
14110 *
14111 * Equivalent to <tt>self.to_s.encoding</tt>; see String#encoding.
14112 *
14113 */
14114
14115static VALUE
14116sym_encoding(VALUE sym)
14117{
14118 return rb_obj_encoding(rb_sym2str(sym));
14119}
14120
14121static VALUE
14122string_for_symbol(VALUE name)
14123{
14124 if (!RB_TYPE_P(name, T_STRING)) {
14125 VALUE tmp = rb_check_string_type(name);
14126 if (NIL_P(tmp)) {
14127 rb_raise(rb_eTypeError, "%+"PRIsVALUE" is not a symbol nor a string",
14128 name);
14129 }
14130 name = tmp;
14131 }
14132 return name;
14133}
14134
14135ID
14137{
14138 if (SYMBOL_P(name)) {
14139 return SYM2ID(name);
14140 }
14141 name = string_for_symbol(name);
14142 return rb_intern_str(name);
14143}
14144
14145VALUE
14147{
14148 if (SYMBOL_P(name)) {
14149 return name;
14150 }
14151 name = string_for_symbol(name);
14152 return rb_str_intern(name);
14153}
14154
14155/*
14156 * call-seq:
14157 * Symbol.all_symbols -> array_of_symbols
14158 *
14159 * Returns an array of all symbols currently in Ruby's symbol table:
14160 *
14161 * Symbol.all_symbols.size # => 9334
14162 * Symbol.all_symbols.take(3) # => [:!, :"\"", :"#"]
14163 *
14164 */
14165
14166static VALUE
14167sym_all_symbols(VALUE _)
14168{
14169 return rb_sym_all_symbols();
14170}
14171
14172VALUE
14173rb_str_to_interned_str(VALUE str)
14174{
14175 return rb_fstring(str);
14176}
14177
14178VALUE
14179rb_interned_str(const char *ptr, long len)
14180{
14181 struct RString fake_str = {RBASIC_INIT};
14182 int encidx = ENCINDEX_US_ASCII;
14183 int coderange = ENC_CODERANGE_7BIT;
14184 if (len > 0 && search_nonascii(ptr, ptr + len)) {
14185 encidx = ENCINDEX_ASCII_8BIT;
14186 coderange = ENC_CODERANGE_VALID;
14187 }
14188 VALUE str = setup_fake_str(&fake_str, ptr, len, encidx);
14189 ENC_CODERANGE_SET(str, coderange);
14190 return register_fstring(str, true, false);
14191}
14192
14193VALUE
14195{
14196 return rb_interned_str(ptr, strlen(ptr));
14197}
14198
14199VALUE
14200rb_enc_interned_str(const char *ptr, long len, rb_encoding *enc)
14201{
14202 if (enc != NULL && UNLIKELY(rb_enc_autoload_p(enc))) {
14203 rb_enc_autoload(enc);
14204 }
14205
14206 struct RString fake_str = {RBASIC_INIT};
14207 return register_fstring(rb_setup_fake_str(&fake_str, ptr, len, enc), true, false);
14208}
14209
14210VALUE
14211rb_enc_literal_str(const char *ptr, long len, rb_encoding *enc)
14212{
14213 if (enc != NULL && UNLIKELY(rb_enc_autoload_p(enc))) {
14214 rb_enc_autoload(enc);
14215 }
14216
14217 struct RString fake_str = {RBASIC_INIT};
14218 VALUE str = register_fstring(rb_setup_fake_str(&fake_str, ptr, len, enc), true, true);
14219 RUBY_ASSERT(RB_OBJ_SHAREABLE_P(str) && (rb_gc_verify_shareable(str), 1));
14220 return str;
14221}
14222
14223VALUE
14225{
14226 return rb_enc_interned_str(ptr, strlen(ptr), enc);
14227}
14228
14229#if USE_YJIT || USE_ZJIT
14230void
14231rb_jit_str_concat_codepoint(VALUE str, VALUE codepoint)
14232{
14233 if (RB_LIKELY(ENCODING_GET_INLINED(str) == rb_ascii8bit_encindex())) {
14234 ssize_t code = RB_NUM2SSIZE(codepoint);
14235
14236 if (RB_LIKELY(code >= 0 && code < 0xff)) {
14237 rb_str_buf_cat_byte(str, (char) code);
14238 return;
14239 }
14240 }
14241
14242 rb_str_concat(str, codepoint);
14243}
14244#endif
14245
14246static int
14247fstring_set_class_i(VALUE *str, void *data)
14248{
14249 RBASIC_SET_CLASS(*str, rb_cString);
14250
14251 return ST_CONTINUE;
14252}
14253
14254void
14255Init_String(void)
14256{
14257 rb_cString = rb_define_class("String", rb_cObject);
14258
14259 rb_concurrent_set_foreach_with_replace(fstring_table_obj, fstring_set_class_i, NULL);
14260
14262 rb_define_alloc_func(rb_cString, empty_str_alloc);
14263 rb_define_singleton_method(rb_cString, "new", rb_str_s_new, -1);
14264 rb_define_singleton_method(rb_cString, "try_convert", rb_str_s_try_convert, 1);
14265 rb_define_method(rb_cString, "initialize", rb_str_init, -1);
14267 rb_define_method(rb_cString, "initialize_copy", rb_str_replace, 1);
14268 rb_define_method(rb_cString, "<=>", rb_str_cmp_m, 1);
14271 rb_define_method(rb_cString, "eql?", rb_str_eql, 1);
14272 rb_define_method(rb_cString, "hash", rb_str_hash_m, 0);
14273 rb_define_method(rb_cString, "casecmp", rb_str_casecmp, 1);
14274 rb_define_method(rb_cString, "casecmp?", rb_str_casecmp_p, 1);
14277 rb_define_method(rb_cString, "%", rb_str_format_m, 1);
14278 rb_define_method(rb_cString, "[]", rb_str_aref_m, -1);
14279 rb_define_method(rb_cString, "[]=", rb_str_aset_m, -1);
14280 rb_define_method(rb_cString, "insert", rb_str_insert, 2);
14283 rb_define_method(rb_cString, "bytesize", rb_str_bytesize, 0);
14284 rb_define_method(rb_cString, "empty?", rb_str_empty, 0);
14285 rb_define_method(rb_cString, "=~", rb_str_match, 1);
14286 rb_define_method(rb_cString, "match", rb_str_match_m, -1);
14287 rb_define_method(rb_cString, "match?", rb_str_match_m_p, -1);
14289 rb_define_method(rb_cString, "succ!", rb_str_succ_bang, 0);
14291 rb_define_method(rb_cString, "next!", rb_str_succ_bang, 0);
14292 rb_define_method(rb_cString, "upto", rb_str_upto, -1);
14293 rb_define_method(rb_cString, "index", rb_str_index_m, -1);
14294 rb_define_method(rb_cString, "byteindex", rb_str_byteindex_m, -1);
14295 rb_define_method(rb_cString, "rindex", rb_str_rindex_m, -1);
14296 rb_define_method(rb_cString, "byterindex", rb_str_byterindex_m, -1);
14297 rb_define_method(rb_cString, "clear", rb_str_clear, 0);
14298 rb_define_method(rb_cString, "chr", rb_str_chr, 0);
14299 rb_define_method(rb_cString, "getbyte", rb_str_getbyte, 1);
14300 rb_define_method(rb_cString, "setbyte", rb_str_setbyte, 2);
14301 rb_define_method(rb_cString, "bit_get", rb_str_bit_get, -1);
14302 rb_define_method(rb_cString, "bit_set?", rb_str_bit_set_p, -1);
14303 rb_define_method(rb_cString, "bit_set", rb_str_bit_set, -1);
14304 rb_define_method(rb_cString, "bit_clear", rb_str_bit_clear, -1);
14305 rb_define_method(rb_cString, "bit_flip", rb_str_bit_flip, -1);
14306 rb_define_method(rb_cString, "bit_count", rb_str_bit_count, -1);
14307 rb_define_method(rb_cString, "bitwise_not", rb_str_bitwise_not, 0);
14308 rb_define_method(rb_cString, "bitwise_not!", rb_str_bitwise_not_bang, 0);
14309 rb_define_method(rb_cString, "bitwise_and", rb_str_bitwise_and, 1);
14310 rb_define_method(rb_cString, "bitwise_and!", rb_str_bitwise_and_bang, 1);
14311 rb_define_method(rb_cString, "bitwise_or", rb_str_bitwise_or, 1);
14312 rb_define_method(rb_cString, "bitwise_or!", rb_str_bitwise_or_bang, 1);
14313 rb_define_method(rb_cString, "bitwise_xor", rb_str_bitwise_xor, 1);
14314 rb_define_method(rb_cString, "bitwise_xor!", rb_str_bitwise_xor_bang, 1);
14315 rb_define_method(rb_cString, "byteslice", rb_str_byteslice, -1);
14316 rb_define_method(rb_cString, "bytesplice", rb_str_bytesplice, -1);
14317 rb_define_method(rb_cString, "scrub", str_scrub, -1);
14318 rb_define_method(rb_cString, "scrub!", str_scrub_bang, -1);
14320 rb_define_method(rb_cString, "+@", str_uplus, 0);
14321 rb_define_method(rb_cString, "-@", str_uminus, 0);
14322 rb_define_method(rb_cString, "dup", rb_str_dup_m, 0);
14323 rb_define_alias(rb_cString, "dedup", "-@");
14324
14325 rb_define_method(rb_cString, "to_i", rb_str_to_i, -1);
14326 rb_define_method(rb_cString, "to_f", rb_str_to_f, 0);
14327 rb_define_method(rb_cString, "to_s", rb_str_to_s, 0);
14328 rb_define_method(rb_cString, "to_str", rb_str_to_s, 0);
14331 rb_define_method(rb_cString, "undump", str_undump, 0);
14332
14333 sym_ascii = ID2SYM(rb_intern_const("ascii"));
14334 sym_turkic = ID2SYM(rb_intern_const("turkic"));
14335 sym_lithuanian = ID2SYM(rb_intern_const("lithuanian"));
14336 sym_fold = ID2SYM(rb_intern_const("fold"));
14337
14338 rb_define_method(rb_cString, "upcase", rb_str_upcase, -1);
14339 rb_define_method(rb_cString, "downcase", rb_str_downcase, -1);
14340 rb_define_method(rb_cString, "capitalize", rb_str_capitalize, -1);
14341 rb_define_method(rb_cString, "swapcase", rb_str_swapcase, -1);
14342
14343 rb_define_method(rb_cString, "upcase!", rb_str_upcase_bang, -1);
14344 rb_define_method(rb_cString, "downcase!", rb_str_downcase_bang, -1);
14345 rb_define_method(rb_cString, "capitalize!", rb_str_capitalize_bang, -1);
14346 rb_define_method(rb_cString, "swapcase!", rb_str_swapcase_bang, -1);
14347
14348 rb_define_method(rb_cString, "hex", rb_str_hex, 0);
14349 rb_define_method(rb_cString, "oct", rb_str_oct, 0);
14350 rb_define_method(rb_cString, "split", rb_str_split_m, -1);
14351 rb_define_method(rb_cString, "lines", rb_str_lines, -1);
14352 rb_define_method(rb_cString, "bytes", rb_str_bytes, 0);
14353 rb_define_method(rb_cString, "chars", rb_str_chars, 0);
14354 rb_define_method(rb_cString, "codepoints", rb_str_codepoints, 0);
14355 rb_define_method(rb_cString, "grapheme_clusters", rb_str_grapheme_clusters, 0);
14356 rb_define_method(rb_cString, "reverse", rb_str_reverse, 0);
14357 rb_define_method(rb_cString, "reverse!", rb_str_reverse_bang, 0);
14358 rb_define_method(rb_cString, "concat", rb_str_concat_multi, -1);
14359 rb_define_method(rb_cString, "append_as_bytes", rb_str_append_as_bytes, -1);
14361 rb_define_method(rb_cString, "prepend", rb_str_prepend_multi, -1);
14362 rb_define_method(rb_cString, "crypt", rb_str_crypt, 1);
14363 rb_define_method(rb_cString, "intern", rb_str_intern, 0); /* in symbol.c */
14364 rb_define_method(rb_cString, "to_sym", rb_str_intern, 0); /* in symbol.c */
14365 rb_define_method(rb_cString, "ord", rb_str_ord, 0);
14366
14367 rb_define_method(rb_cString, "include?", rb_str_include, 1);
14368 rb_define_method(rb_cString, "start_with?", rb_str_start_with, -1);
14369 rb_define_method(rb_cString, "end_with?", rb_str_end_with, -1);
14370
14371 rb_define_method(rb_cString, "scan", rb_str_scan, 1);
14372
14373 rb_define_method(rb_cString, "ljust", rb_str_ljust, -1);
14374 rb_define_method(rb_cString, "rjust", rb_str_rjust, -1);
14375 rb_define_method(rb_cString, "center", rb_str_center, -1);
14376
14377 rb_define_method(rb_cString, "sub", rb_str_sub, -1);
14378 rb_define_method(rb_cString, "gsub", rb_str_gsub, -1);
14379 rb_define_method(rb_cString, "chop", rb_str_chop, 0);
14380 rb_define_method(rb_cString, "chomp", rb_str_chomp, -1);
14381 rb_define_method(rb_cString, "strip", rb_str_strip, -1);
14382 rb_define_method(rb_cString, "lstrip", rb_str_lstrip, -1);
14383 rb_define_method(rb_cString, "rstrip", rb_str_rstrip, -1);
14384 rb_define_method(rb_cString, "delete_prefix", rb_str_delete_prefix, 1);
14385 rb_define_method(rb_cString, "delete_suffix", rb_str_delete_suffix, 1);
14386
14387 rb_define_method(rb_cString, "sub!", rb_str_sub_bang, -1);
14388 rb_define_method(rb_cString, "gsub!", rb_str_gsub_bang, -1);
14389 rb_define_method(rb_cString, "chop!", rb_str_chop_bang, 0);
14390 rb_define_method(rb_cString, "chomp!", rb_str_chomp_bang, -1);
14391 rb_define_method(rb_cString, "strip!", rb_str_strip_bang, -1);
14392 rb_define_method(rb_cString, "lstrip!", rb_str_lstrip_bang, -1);
14393 rb_define_method(rb_cString, "rstrip!", rb_str_rstrip_bang, -1);
14394 rb_define_method(rb_cString, "delete_prefix!", rb_str_delete_prefix_bang, 1);
14395 rb_define_method(rb_cString, "delete_suffix!", rb_str_delete_suffix_bang, 1);
14396
14397 rb_define_method(rb_cString, "tr", rb_str_tr, -1);
14398 rb_define_method(rb_cString, "tr_s", rb_str_tr_s, 2);
14399 rb_define_method(rb_cString, "delete", rb_str_delete, -1);
14400 rb_define_method(rb_cString, "squeeze", rb_str_squeeze, -1);
14401 rb_define_method(rb_cString, "count", rb_str_count, -1);
14402
14403 rb_define_method(rb_cString, "tr!", rb_str_tr_bang, -1);
14404 rb_define_method(rb_cString, "tr_s!", rb_str_tr_s_bang, 2);
14405 rb_define_method(rb_cString, "delete!", rb_str_delete_bang, -1);
14406 rb_define_method(rb_cString, "squeeze!", rb_str_squeeze_bang, -1);
14407
14408 rb_define_method(rb_cString, "each_line", rb_str_each_line, -1);
14409 rb_define_method(rb_cString, "each_byte", rb_str_each_byte, 0);
14410 rb_define_method(rb_cString, "each_char", rb_str_each_char, 0);
14411 rb_define_method(rb_cString, "each_codepoint", rb_str_each_codepoint, 0);
14412 rb_define_method(rb_cString, "each_grapheme_cluster", rb_str_each_grapheme_cluster, 0);
14413
14414 rb_define_method(rb_cString, "sum", rb_str_sum, -1);
14415
14416 rb_define_method(rb_cString, "slice", rb_str_aref_m, -1);
14417 rb_define_method(rb_cString, "slice!", rb_str_slice_bang, -1);
14418
14419 rb_define_method(rb_cString, "partition", rb_str_partition, 1);
14420 rb_define_method(rb_cString, "rpartition", rb_str_rpartition, 1);
14421
14422 rb_define_method(rb_cString, "encoding", rb_obj_encoding, 0); /* in encoding.c */
14423 rb_define_method(rb_cString, "force_encoding", rb_str_force_encoding, 1);
14424 rb_define_method(rb_cString, "b", rb_str_b, 0);
14425
14426 /* define UnicodeNormalize module here so that we don't have to look it up */
14427 mUnicodeNormalize = rb_define_module("UnicodeNormalize");
14428 id_normalize = rb_intern_const("normalize");
14429 id_normalized_p = rb_intern_const("normalized?");
14430
14431 rb_define_method(rb_cString, "unicode_normalize", rb_str_unicode_normalize, -1);
14432 rb_define_method(rb_cString, "unicode_normalize!", rb_str_unicode_normalize_bang, -1);
14433 rb_define_method(rb_cString, "unicode_normalized?", rb_str_unicode_normalized_p, -1);
14434
14435 rb_fs = Qnil;
14436 rb_define_hooked_variable("$;", &rb_fs, 0, rb_fs_setter);
14437 rb_define_hooked_variable("$-F", &rb_fs, 0, rb_fs_setter);
14438 rb_gc_register_address(&rb_fs);
14439
14440 rb_cSymbol = rb_define_class("Symbol", rb_cObject);
14444 rb_define_singleton_method(rb_cSymbol, "all_symbols", sym_all_symbols, 0);
14445
14446 rb_define_method(rb_cSymbol, "==", sym_equal, 1);
14447 rb_define_method(rb_cSymbol, "===", sym_equal, 1);
14448 rb_define_method(rb_cSymbol, "inspect", sym_inspect, 0);
14449 rb_define_method(rb_cSymbol, "to_proc", rb_sym_to_proc, 0); /* in proc.c */
14450 rb_define_method(rb_cSymbol, "succ", sym_succ, 0);
14451 rb_define_method(rb_cSymbol, "next", sym_succ, 0);
14452
14453 rb_define_method(rb_cSymbol, "<=>", sym_cmp, 1);
14454 rb_define_method(rb_cSymbol, "casecmp", sym_casecmp, 1);
14455 rb_define_method(rb_cSymbol, "casecmp?", sym_casecmp_p, 1);
14456 rb_define_method(rb_cSymbol, "=~", sym_match, 1);
14457
14458 rb_define_method(rb_cSymbol, "[]", sym_aref, -1);
14459 rb_define_method(rb_cSymbol, "slice", sym_aref, -1);
14460 rb_define_method(rb_cSymbol, "length", sym_length, 0);
14461 rb_define_method(rb_cSymbol, "size", sym_length, 0);
14462 rb_define_method(rb_cSymbol, "match", sym_match_m, -1);
14463 rb_define_method(rb_cSymbol, "match?", sym_match_m_p, -1);
14464
14465 rb_define_method(rb_cSymbol, "upcase", sym_upcase, -1);
14466 rb_define_method(rb_cSymbol, "downcase", sym_downcase, -1);
14467 rb_define_method(rb_cSymbol, "capitalize", sym_capitalize, -1);
14468 rb_define_method(rb_cSymbol, "swapcase", sym_swapcase, -1);
14469
14470 rb_define_method(rb_cSymbol, "start_with?", sym_start_with, -1);
14471 rb_define_method(rb_cSymbol, "end_with?", sym_end_with, -1);
14472
14473 rb_define_method(rb_cSymbol, "encoding", sym_encoding, 0);
14474}
14475
14476#include "string.rbinc"
#define RUBY_ASSERT_ALWAYS(expr,...)
A variant of RUBY_ASSERT that does not interface with RUBY_DEBUG.
Definition assert.h:199
#define RBIMPL_ASSERT_OR_ASSUME(...)
This is either RUBY_ASSERT or RBIMPL_ASSUME, depending on RUBY_DEBUG.
Definition assert.h:311
#define RUBY_ASSERT_BUILTIN_TYPE(obj, type)
A variant of RUBY_ASSERT that asserts when either RUBY_DEBUG or built-in type of obj is type.
Definition assert.h:291
#define RUBY_ASSERT(...)
Asserts that the given expression is truthy if and only if RUBY_DEBUG is truthy.
Definition assert.h:219
Atomic operations.
@ RUBY_ENC_CODERANGE_7BIT
The object holds 0 to 127 inclusive and nothing else.
Definition coderange.h:39
static enum ruby_coderange_type RB_ENC_CODERANGE_AND(enum ruby_coderange_type a, enum ruby_coderange_type b)
"Mix" two code ranges into one.
Definition coderange.h:162
static int rb_isspace(int c)
Our own locale-insensitive version of isspace(3).
Definition ctype.h:395
static int rb_isascii(int c)
Our own locale-insensitive version of isascii(3).
Definition ctype.h:209
#define rb_define_method(klass, mid, func, arity)
Defines klass#mid.
#define rb_define_singleton_method(klass, mid, func, arity)
Defines klass.mid.
static bool rb_enc_is_newline(const char *p, const char *e, rb_encoding *enc)
Queries if the passed pointer points to a newline character.
Definition ctype.h:43
static bool rb_enc_isprint(OnigCodePoint c, rb_encoding *enc)
Identical to rb_isprint(), except it additionally takes an encoding.
Definition ctype.h:180
static bool rb_enc_isctype(OnigCodePoint c, OnigCtype t, rb_encoding *enc)
Queries if the passed code point is of passed character type in the passed encoding.
Definition ctype.h:63
VALUE rb_enc_sprintf(rb_encoding *enc, const char *fmt,...)
Identical to rb_sprintf(), except it additionally takes an encoding.
Definition sprintf.c:1231
static VALUE RB_OBJ_FROZEN_RAW(VALUE obj)
This is an implementation detail of RB_OBJ_FROZEN().
Definition fl_type.h:699
static VALUE RB_FL_TEST_RAW(VALUE obj, VALUE flags)
This is an implementation detail of RB_FL_TEST().
Definition fl_type.h:407
void rb_include_module(VALUE klass, VALUE module)
Includes a module to a class.
Definition class.c:1769
void rb_define_alias(VALUE klass, const char *name1, const char *name2)
Defines an alias of a method.
Definition class.c:3094
void rb_undef_method(VALUE klass, const char *name)
Defines an undef of a method.
Definition class.c:2897
int rb_scan_args(int argc, const VALUE *argv, const char *fmt,...)
Retrieves argument from argc and argv to given VALUE references according to the format string.
Definition class.c:3384
int rb_block_given_p(void)
Determines if the current method is given a block.
Definition eval.c:1035
int rb_get_kwargs(VALUE keyword_hash, const ID *table, int required, int optional, VALUE *values)
Keyword argument deconstructor.
Definition class.c:3173
#define TYPE(_)
Old name of rb_type.
Definition value_type.h:108
#define ENCODING_SET_INLINED(obj, i)
Old name of RB_ENCODING_SET_INLINED.
Definition encoding.h:106
#define RB_INTEGER_TYPE_P
Old name of rb_integer_type_p.
Definition value_type.h:87
#define ENC_CODERANGE_7BIT
Old name of RUBY_ENC_CODERANGE_7BIT.
Definition coderange.h:180
#define ENC_CODERANGE_VALID
Old name of RUBY_ENC_CODERANGE_VALID.
Definition coderange.h:181
#define FL_UNSET_RAW
Old name of RB_FL_UNSET_RAW.
Definition fl_type.h:130
#define rb_str_buf_cat2
Old name of rb_usascii_str_new_cstr.
Definition string.h:1707
#define ALLOCV
Old name of RB_ALLOCV.
Definition memory.h:404
#define ISSPACE
Old name of rb_isspace.
Definition ctype.h:88
#define T_STRING
Old name of RUBY_T_STRING.
Definition value_type.h:78
#define ENC_CODERANGE_CLEAN_P(cr)
Old name of RB_ENC_CODERANGE_CLEAN_P.
Definition coderange.h:183
#define ENC_CODERANGE_AND(a, b)
Old name of RB_ENC_CODERANGE_AND.
Definition coderange.h:188
#define Qundef
Old name of RUBY_Qundef.
#define INT2FIX
Old name of RB_INT2FIX.
Definition long.h:48
#define OBJ_FROZEN
Old name of RB_OBJ_FROZEN.
Definition fl_type.h:133
#define rb_str_cat2
Old name of rb_str_cat_cstr.
Definition string.h:1708
#define UNREACHABLE
Old name of RBIMPL_UNREACHABLE.
Definition assume.h:28
#define ID2SYM
Old name of RB_ID2SYM.
Definition symbol.h:44
#define T_BIGNUM
Old name of RUBY_T_BIGNUM.
Definition value_type.h:57
#define OBJ_FREEZE
Old name of RB_OBJ_FREEZE.
Definition fl_type.h:131
#define T_FIXNUM
Old name of RUBY_T_FIXNUM.
Definition value_type.h:63
#define UNREACHABLE_RETURN
Old name of RBIMPL_UNREACHABLE_RETURN.
Definition assume.h:29
#define SYM2ID
Old name of RB_SYM2ID.
Definition symbol.h:45
#define ENC_CODERANGE(obj)
Old name of RB_ENC_CODERANGE.
Definition coderange.h:184
#define CLASS_OF
Old name of rb_class_of.
Definition globals.h:205
#define ENC_CODERANGE_UNKNOWN
Old name of RUBY_ENC_CODERANGE_UNKNOWN.
Definition coderange.h:179
#define SIZET2NUM
Old name of RB_SIZE2NUM.
Definition size_t.h:62
#define FIXABLE
Old name of RB_FIXABLE.
Definition fixnum.h:25
#define xmalloc
Old name of ruby_xmalloc.
Definition xmalloc.h:53
#define ENCODING_GET(obj)
Old name of RB_ENCODING_GET.
Definition encoding.h:109
#define LONG2FIX
Old name of RB_INT2FIX.
Definition long.h:49
#define ISDIGIT
Old name of rb_isdigit.
Definition ctype.h:93
#define ENC_CODERANGE_MASK
Old name of RUBY_ENC_CODERANGE_MASK.
Definition coderange.h:178
#define ZALLOC_N
Old name of RB_ZALLOC_N.
Definition memory.h:401
#define T_HASH
Old name of RUBY_T_HASH.
Definition value_type.h:65
#define ALLOC_N
Old name of RB_ALLOC_N.
Definition memory.h:399
#define MBCLEN_CHARFOUND_LEN(ret)
Old name of ONIGENC_MBCLEN_CHARFOUND_LEN.
Definition encoding.h:517
#define FL_TEST_RAW
Old name of RB_FL_TEST_RAW.
Definition fl_type.h:128
#define FL_SET
Old name of RB_FL_SET.
Definition fl_type.h:125
#define rb_ary_new3
Old name of rb_ary_new_from_args.
Definition array.h:658
#define ENCODING_INLINE_MAX
Old name of RUBY_ENCODING_INLINE_MAX.
Definition encoding.h:67
#define LONG2NUM
Old name of RB_LONG2NUM.
Definition long.h:50
#define FL_ANY_RAW
Old name of RB_FL_ANY_RAW.
Definition fl_type.h:122
#define ISALPHA
Old name of rb_isalpha.
Definition ctype.h:92
#define MBCLEN_INVALID_P(ret)
Old name of ONIGENC_MBCLEN_INVALID_P.
Definition encoding.h:518
#define ISASCII
Old name of rb_isascii.
Definition ctype.h:85
#define ULL2NUM
Old name of RB_ULL2NUM.
Definition long_long.h:31
#define TOLOWER
Old name of rb_tolower.
Definition ctype.h:101
#define Qtrue
Old name of RUBY_Qtrue.
#define ST2FIX
Old name of RB_ST2FIX.
Definition st_data_t.h:33
#define MBCLEN_NEEDMORE_P(ret)
Old name of ONIGENC_MBCLEN_NEEDMORE_P.
Definition encoding.h:519
#define FIXNUM_MAX
Old name of RUBY_FIXNUM_MAX.
Definition fixnum.h:26
#define NUM2INT
Old name of RB_NUM2INT.
Definition int.h:44
#define Qnil
Old name of RUBY_Qnil.
#define Qfalse
Old name of RUBY_Qfalse.
#define FIX2LONG
Old name of RB_FIX2LONG.
Definition long.h:46
#define ENC_CODERANGE_BROKEN
Old name of RUBY_ENC_CODERANGE_BROKEN.
Definition coderange.h:182
#define scan_hex(s, l, e)
Old name of ruby_scan_hex.
Definition util.h:108
#define NIL_P
Old name of RB_NIL_P.
#define ALLOCV_N
Old name of RB_ALLOCV_N.
Definition memory.h:405
#define MBCLEN_CHARFOUND_P(ret)
Old name of ONIGENC_MBCLEN_CHARFOUND_P.
Definition encoding.h:516
#define NUM2ULL
Old name of RB_NUM2ULL.
Definition long_long.h:35
#define DBL2NUM
Old name of rb_float_new.
Definition double.h:29
#define ISPRINT
Old name of rb_isprint.
Definition ctype.h:86
#define BUILTIN_TYPE
Old name of RB_BUILTIN_TYPE.
Definition value_type.h:85
#define ENCODING_SHIFT
Old name of RUBY_ENCODING_SHIFT.
Definition encoding.h:68
#define FL_TEST
Old name of RB_FL_TEST.
Definition fl_type.h:127
#define FL_FREEZE
Old name of RUBY_FL_FREEZE.
Definition fl_type.h:65
#define NUM2LONG
Old name of RB_NUM2LONG.
Definition long.h:51
#define ENCODING_GET_INLINED(obj)
Old name of RB_ENCODING_GET_INLINED.
Definition encoding.h:108
#define ENC_CODERANGE_CLEAR(obj)
Old name of RB_ENC_CODERANGE_CLEAR.
Definition coderange.h:187
#define FL_UNSET
Old name of RB_FL_UNSET.
Definition fl_type.h:129
#define UINT2NUM
Old name of RB_UINT2NUM.
Definition int.h:46
#define ENCODING_IS_ASCII8BIT(obj)
Old name of RB_ENCODING_IS_ASCII8BIT.
Definition encoding.h:110
#define FIXNUM_P
Old name of RB_FIXNUM_P.
#define CONST_ID
Old name of RUBY_CONST_ID.
Definition symbol.h:47
#define rb_ary_new2
Old name of rb_ary_new_capa.
Definition array.h:657
#define ENC_CODERANGE_SET(obj, cr)
Old name of RB_ENC_CODERANGE_SET.
Definition coderange.h:186
#define ENCODING_CODERANGE_SET(obj, encindex, cr)
Old name of RB_ENCODING_CODERANGE_SET.
Definition coderange.h:189
#define FL_SET_RAW
Old name of RB_FL_SET_RAW.
Definition fl_type.h:126
#define ALLOCV_END
Old name of RB_ALLOCV_END.
Definition memory.h:406
#define SYMBOL_P
Old name of RB_SYMBOL_P.
Definition value_type.h:88
#define OBJ_FROZEN_RAW
Old name of RB_OBJ_FROZEN_RAW.
Definition fl_type.h:134
#define T_REGEXP
Old name of RUBY_T_REGEXP.
Definition value_type.h:77
#define ENCODING_MASK
Old name of RUBY_ENCODING_MASK.
Definition encoding.h:69
void rb_category_warn(rb_warning_category_t category, const char *fmt,...)
Identical to rb_category_warning(), except it reports unless $VERBOSE is nil.
Definition error.c:478
void rb_exc_raise(VALUE mesg)
Raises an exception in the current thread.
Definition eval.c:678
void rb_syserr_fail(int e, const char *mesg)
Raises appropriate exception that represents a C errno.
Definition error.c:4084
VALUE rb_eRangeError
RangeError exception.
Definition error.c:1477
VALUE rb_eTypeError
TypeError exception.
Definition error.c:1473
VALUE rb_eEncCompatError
Encoding::CompatibilityError exception.
Definition error.c:1480
VALUE rb_eRuntimeError
RuntimeError exception.
Definition error.c:1471
VALUE rb_eIndexError
IndexError exception.
Definition error.c:1475
@ RB_WARN_CATEGORY_DEPRECATED
Warning is for deprecated features.
Definition error.h:48
VALUE rb_cObject
Object class.
Definition object.c:60
VALUE rb_any_to_s(VALUE obj)
Generates a textual representation of the given object.
Definition object.c:658
VALUE rb_obj_alloc(VALUE klass)
Allocates an instance of the given class.
Definition object.c:2252
VALUE rb_obj_hide(VALUE obj)
Make the object invisible from Ruby code.
Definition object.c:94
VALUE rb_class_new_instance_pass_kw(int argc, const VALUE *argv, VALUE klass)
Identical to rb_class_new_instance(), except it passes the passed keywords if any to the #initialize ...
Definition object.c:2270
VALUE rb_obj_frozen_p(VALUE obj)
Same as RB_OBJ_FROZEN(), but returns Qtrue/Qfalse instead of #bool.
Definition object.c:1316
double rb_str_to_dbl(VALUE str, int mode)
Identical to rb_cstr_to_dbl(), except it accepts a Ruby's string instead of C's.
Definition object.c:3642
VALUE rb_obj_class(VALUE obj)
Queries the class of an object.
Definition object.c:234
VALUE rb_obj_dup(VALUE obj)
Duplicates the given object.
Definition object.c:556
VALUE rb_cSymbol
Symbol class.
Definition string.c:86
VALUE rb_cRange
Range class.
Definition range.c:35
VALUE rb_equal(VALUE lhs, VALUE rhs)
This function is an optimised version of calling #==.
Definition object.c:140
VALUE rb_obj_is_kind_of(VALUE obj, VALUE klass)
Queries if the given object is an instance (of possibly descendants) of the given class.
Definition object.c:906
VALUE rb_obj_freeze(VALUE obj)
Same as RB_OBJ_FREEZE(), but returns the given object.
Definition object.c:1309
VALUE rb_mComparable
Comparable module.
Definition compar.c:19
VALUE rb_cString
String class.
Definition string.c:85
VALUE rb_to_int(VALUE val)
Identical to rb_check_to_int(), except it raises in case of conversion mismatch.
Definition object.c:3328
Encoding relates APIs.
static char * rb_enc_left_char_head(const char *s, const char *p, const char *e, rb_encoding *enc)
Queries the left boundary of a character.
Definition encoding.h:683
static char * rb_enc_right_char_head(const char *s, const char *p, const char *e, rb_encoding *enc)
Queries the right boundary of a character.
Definition encoding.h:704
static unsigned int rb_enc_codepoint(const char *p, const char *e, rb_encoding *enc)
Queries the code point of character pointed by the passed pointer.
Definition encoding.h:571
static int rb_enc_mbmaxlen(rb_encoding *enc)
Queries the maximum number of bytes that the passed encoding needs to represent a character.
Definition encoding.h:447
static int RB_ENCODING_GET_INLINED(VALUE obj)
Queries the encoding of the passed object.
Definition encoding.h:99
static int rb_enc_code_to_mbclen(int c, rb_encoding *enc)
Identical to rb_enc_codelen(), except it returns 0 for invalid code points.
Definition encoding.h:619
static char * rb_enc_step_back(const char *s, const char *p, const char *e, int n, rb_encoding *enc)
Scans the string backwards for n characters.
Definition encoding.h:726
VALUE rb_str_conv_enc(VALUE str, rb_encoding *from, rb_encoding *to)
Encoding conversion main routine.
Definition string.c:1379
VALUE rb_enc_str_new_static(const char *ptr, long len, rb_encoding *enc)
Identical to rb_enc_str_new(), except it takes a C string literal.
Definition string.c:1244
char * rb_enc_nth(const char *head, const char *tail, long nth, rb_encoding *enc)
Queries the n-th character.
Definition string.c:3123
VALUE rb_str_conv_enc_opts(VALUE str, rb_encoding *from, rb_encoding *to, int ecflags, VALUE ecopts)
Identical to rb_str_conv_enc(), except it additionally takes IO encoder options.
Definition string.c:1263
VALUE rb_enc_interned_str(const char *ptr, long len, rb_encoding *enc)
Identical to rb_enc_str_new(), except it returns a "f"string.
Definition string.c:14200
long rb_memsearch(const void *x, long m, const void *y, long n, rb_encoding *enc)
Looks for the passed string in the passed buffer.
Definition re.c:285
long rb_enc_strlen(const char *head, const char *tail, rb_encoding *enc)
Counts the number of characters of the passed string, according to the passed encoding.
Definition string.c:2405
VALUE rb_enc_str_buf_cat(VALUE str, const char *ptr, long len, rb_encoding *enc)
Identical to rb_str_cat(), except it additionally takes an encoding.
Definition string.c:3848
VALUE rb_enc_str_new_cstr(const char *ptr, rb_encoding *enc)
Identical to rb_enc_str_new(), except it assumes the passed pointer is a pointer to a C string.
Definition string.c:1175
VALUE rb_str_export_to_enc(VALUE obj, rb_encoding *enc)
Identical to rb_str_export(), except it additionally takes an encoding.
Definition string.c:1484
VALUE rb_external_str_new_with_enc(const char *ptr, long len, rb_encoding *enc)
Identical to rb_external_str_new(), except it additionally takes an encoding.
Definition string.c:1385
int rb_enc_str_asciionly_p(VALUE str)
Queries if the passed string is "ASCII only".
Definition string.c:988
VALUE rb_enc_interned_str_cstr(const char *ptr, rb_encoding *enc)
Identical to rb_enc_str_new_cstr(), except it returns a "f"string.
Definition string.c:14224
long rb_str_coderange_scan_restartable(const char *str, const char *end, rb_encoding *enc, int *cr)
Scans the passed string until it finds something odd.
Definition string.c:844
int rb_enc_symname2_p(const char *name, long len, rb_encoding *enc)
Identical to rb_enc_symname_p(), except it additionally takes the passed string's length.
Definition symbol.c:858
rb_econv_result_t rb_econv_convert(rb_econv_t *ec, const unsigned char **source_buffer_ptr, const unsigned char *source_buffer_end, unsigned char **destination_buffer_ptr, unsigned char *destination_buffer_end, int flags)
Converts a string from an encoding to another.
Definition transcode.c:1487
rb_econv_result_t
return value of rb_econv_convert()
Definition transcode.h:30
@ econv_finished
The conversion stopped after converting everything.
Definition transcode.h:57
@ econv_destination_buffer_full
The conversion stopped because there is no destination.
Definition transcode.h:46
rb_econv_t * rb_econv_open_opts(const char *source_encoding, const char *destination_encoding, int ecflags, VALUE ecopts)
Identical to rb_econv_open(), except it additionally takes a hash of optional strings.
Definition transcode.c:2730
VALUE rb_str_encode(VALUE str, VALUE to, int ecflags, VALUE ecopts)
Converts the contents of the passed string from its encoding to the passed one.
Definition transcode.c:2993
void rb_econv_close(rb_econv_t *ec)
Destructs a converter.
Definition transcode.c:1744
VALUE rb_funcall(VALUE recv, ID mid, int n,...)
Calls a method.
Definition vm_eval.c:1123
VALUE rb_funcallv(VALUE recv, ID mid, int argc, const VALUE *argv)
Identical to rb_funcall(), except it takes the method arguments as a C array.
Definition vm_eval.c:1081
VALUE rb_funcall_with_block_kw(VALUE recv, ID mid, int argc, const VALUE *argv, VALUE procval, int kw_splat)
Identical to rb_funcallv_with_block(), except you can specify how to handle the last element of the g...
Definition vm_eval.c:1210
VALUE rb_check_array_type(VALUE obj)
Try converting an object to its array representation using its to_ary method, if any.
VALUE rb_ary_new(void)
Allocates a new, empty array.
VALUE rb_ary_new_capa(long capa)
Identical to rb_ary_new(), except it additionally specifies how many rooms of objects it should alloc...
VALUE rb_ary_push(VALUE ary, VALUE elem)
Special case of rb_ary_cat() that it adds only one element.
VALUE rb_ary_freeze(VALUE obj)
Freeze an array, preventing further modifications.
#define RETURN_SIZED_ENUMERATOR(obj, argc, argv, size_fn)
This roughly resembles return enum_for(__callee__) unless block_given?.
Definition enumerator.h:208
#define RETURN_ENUMERATOR(obj, argc, argv)
Identical to RETURN_SIZED_ENUMERATOR(), except its size is unknown.
Definition enumerator.h:242
#define UNLIMITED_ARGUMENTS
This macro is used in conjunction with rb_check_arity().
Definition error.h:35
static int rb_check_arity(int argc, int min, int max)
Ensures that the passed integer is in the passed range.
Definition error.h:284
VALUE rb_fs
The field separator character for inputs, or the $;.
Definition string.c:723
VALUE rb_default_rs
This is the default value of rb_rs, i.e.
Definition io.c:209
VALUE rb_backref_get(void)
Queries the last match, or Regexp.last_match, or the $~.
Definition vm.c:2131
VALUE rb_sym_all_symbols(void)
Collects every single bits of symbols that have ever interned in the entire history of the current pr...
Definition symbol.c:1215
void rb_backref_set(VALUE md)
Updates $~.
Definition vm.c:2137
int rb_range_values(VALUE range, VALUE *begp, VALUE *endp, int *exclp)
Deconstructs a range into its components.
Definition range.c:1857
VALUE rb_range_beg_len(VALUE range, long *begp, long *lenp, long len, int err)
Deconstructs a numerical range.
Definition range.c:1945
int rb_reg_backref_number(VALUE match, VALUE backref)
Queries the index of the given named capture.
Definition re.c:1387
int rb_reg_options(VALUE re)
Queries the options of the passed regular expression.
Definition re.c:4476
VALUE rb_reg_match(VALUE re, VALUE str)
This is the match operator.
Definition re.c:3970
void rb_match_busy(VALUE md)
Asserts that the given MatchData is "occupied".
Definition re.c:1631
VALUE rb_reg_nth_match(int n, VALUE md)
Queries the nth captured substring.
Definition re.c:2071
void rb_str_free(VALUE str)
Destroys the given string for no reason.
Definition string.c:1803
VALUE rb_str_new_shared(VALUE str)
Identical to rb_str_new_cstr(), except it takes a Ruby's string instead of C's.
Definition string.c:1549
VALUE rb_str_plus(VALUE lhs, VALUE rhs)
Generates a new string, concatenating the former to the latter.
Definition string.c:2556
#define rb_utf8_str_new_cstr(str)
Identical to rb_str_new_cstr, except it generates a string of "UTF-8" encoding.
Definition string.h:1608
#define rb_hash_end(h)
Just another name of st_hash_end.
Definition string.h:970
#define rb_hash_uint32(h, i)
Just another name of st_hash_uint32.
Definition string.h:964
VALUE rb_str_append(VALUE dst, VALUE src)
Identical to rb_str_buf_append(), except it converts the right hand side before concatenating.
Definition string.c:3913
VALUE rb_filesystem_str_new(const char *ptr, long len)
Identical to rb_str_new(), except it generates a string of "filesystem" encoding.
Definition string.c:1460
VALUE rb_sym_to_s(VALUE sym)
This is an rb_sym2str() + rb_str_dup() combo.
Definition string.c:13832
VALUE rb_str_times(VALUE str, VALUE num)
Repetition of a string.
Definition string.c:2630
VALUE rb_external_str_new(const char *ptr, long len)
Identical to rb_str_new(), except it generates a string of "default external" encoding.
Definition string.c:1436
VALUE rb_str_tmp_new(long len)
Allocates a "temporary" string.
Definition string.c:1797
long rb_str_offset(VALUE str, long pos)
"Inverse" of rb_str_sublen().
Definition string.c:3151
VALUE rb_str_succ(VALUE orig)
Searches for the "successor" of a string.
Definition string.c:5456
int rb_str_hash_cmp(VALUE str1, VALUE str2)
Compares two strings.
Definition string.c:4275
VALUE rb_str_subseq(VALUE str, long beg, long len)
Identical to rb_str_substr(), except the numbers are interpreted as byte offsets instead of character...
Definition string.c:3266
VALUE rb_str_ellipsize(VALUE str, long len)
Shortens str and adds three dots, an ellipsis, if it is longer than len characters.
Definition string.c:13147
st_index_t rb_memhash(const void *ptr, long len)
This is a universal hash function.
Definition random.c:1720
#define rb_str_new(str, len)
Allocates an instance of rb_cString.
Definition string.h:1523
void rb_str_shared_replace(VALUE dst, VALUE src)
Replaces the contents of the former with the latter.
Definition string.c:1839
#define rb_str_buf_cat
Just another name of rb_str_cat.
Definition string.h:1706
VALUE rb_str_new_static(const char *ptr, long len)
Identical to rb_str_new(), except it takes a C string literal.
Definition string.c:1209
#define rb_usascii_str_new(str, len)
Identical to rb_str_new, except it generates a string of "US ASCII" encoding.
Definition string.h:1557
size_t rb_str_capacity(VALUE str)
Queries the capacity of the given string.
Definition string.c:1023
VALUE rb_str_new_frozen(VALUE str)
Creates a frozen copy of the string, if necessary.
Definition string.c:1555
VALUE rb_str_dup(VALUE str)
Duplicates a string.
Definition string.c:2038
st_index_t rb_str_hash(VALUE str)
Calculates a hash value of a string.
Definition string.c:4261
VALUE rb_str_cat(VALUE dst, const char *src, long srclen)
Destructively appends the passed contents to the string.
Definition string.c:3681
VALUE rb_str_locktmp(VALUE str)
Obtains a "temporary lock" of the string.
long rb_str_strlen(VALUE str)
Counts the number of characters (not bytes) that are stored inside of the given string.
Definition string.c:2492
VALUE rb_str_resurrect(VALUE str)
Like rb_str_dup(), but always create an instance of rb_cString regardless of the given object's class...
Definition string.c:2056
#define rb_str_buf_new_cstr(str)
Identical to rb_str_new_cstr, except done differently.
Definition string.h:1664
#define rb_usascii_str_new_cstr(str)
Identical to rb_str_new_cstr, except it generates a string of "US ASCII" encoding.
Definition string.h:1592
VALUE rb_str_replace(VALUE dst, VALUE src)
Replaces the contents of the former object with the stringised contents of the latter.
Definition string.c:6675
VALUE rb_str_no_gvl_safe_acquire(VALUE orig)
Creates a frozen copy of orig that guarantees the RSTRING_PTR is safe to use in operations that relea...
Definition string.c:1584
char * rb_str_subpos(VALUE str, long beg, long *len)
Identical to rb_str_substr(), except it returns a C's string instead of Ruby's.
Definition string.c:3274
rb_gvar_setter_t rb_str_setter
This is a rb_gvar_setter_t that refutes non-string assignments.
Definition string.h:1171
VALUE rb_interned_str_cstr(const char *ptr)
Identical to rb_interned_str(), except it assumes the passed pointer is a pointer to a C's string.
Definition string.c:14194
VALUE rb_filesystem_str_new_cstr(const char *ptr)
Identical to rb_filesystem_str_new(), except it assumes the passed pointer is a pointer to a C string...
Definition string.c:1466
#define rb_external_str_new_cstr(str)
Identical to rb_str_new_cstr, except it generates a string of "default external" encoding.
Definition string.h:1629
VALUE rb_str_buf_append(VALUE dst, VALUE src)
Identical to rb_str_cat_cstr(), except it takes Ruby's string instead of C's.
Definition string.c:3879
long rb_str_sublen(VALUE str, long pos)
Byte offset to character offset conversion.
Definition string.c:3198
VALUE rb_str_equal(VALUE str1, VALUE str2)
Equality of two strings.
Definition string.c:4382
void rb_str_no_gvl_safe_release(VALUE orig, VALUE tmp)
Releases a string created from rb_str_no_gvl_safe_acquire.
Definition string.c:1627
void rb_str_set_len(VALUE str, long len)
Overwrites the length of the string.
Definition string.c:3500
VALUE rb_str_inspect(VALUE str)
Generates a "readable" version of the receiver.
Definition string.c:8151
void rb_must_asciicompat(VALUE obj)
Asserts that the given string's encoding is (Ruby's definition of) ASCII compatible.
Definition string.c:2862
VALUE rb_interned_str(const char *ptr, long len)
Identical to rb_str_new(), except it returns an infamous "f"string.
Definition string.c:14179
int rb_str_cmp(VALUE lhs, VALUE rhs)
Compares two strings, as in strcmp(3).
Definition string.c:4329
VALUE rb_str_concat(VALUE dst, VALUE src)
Identical to rb_str_append(), except it also accepts an integer as a codepoint.
Definition string.c:4149
int rb_str_comparable(VALUE str1, VALUE str2)
Checks if two strings are comparable each other or not.
Definition string.c:4304
#define rb_strlen_lit(str)
Length of a string literal.
Definition string.h:1717
VALUE rb_str_buf_cat_ascii(VALUE dst, const char *src)
Identical to rb_str_cat_cstr(), except it additionally assumes the source string be a NUL terminated ...
Definition string.c:3855
VALUE rb_str_freeze(VALUE str)
This is the implementation of String#freeze.
Definition string.c:3391
void rb_str_update(VALUE dst, long beg, long len, VALUE src)
Replaces some (or all) of the contents of the given string.
Definition string.c:5943
VALUE rb_str_scrub(VALUE str, VALUE repl)
"Cleanses" the string.
Definition string.c:13205
#define rb_locale_str_new_cstr(str)
Identical to rb_external_str_new_cstr, except it generates a string of "locale" encoding instead of "...
Definition string.h:1650
VALUE rb_str_new_with_class(VALUE obj, const char *ptr, long len)
Identical to rb_str_new(), except it takes the class of the allocating object.
Definition string.c:1753
#define rb_str_dup_frozen
Just another name of rb_str_new_frozen.
Definition string.h:656
VALUE rb_check_string_type(VALUE obj)
Try converting an object to its stringised representation using its to_str method,...
Definition string.c:3047
VALUE rb_str_substr(VALUE str, long beg, long len)
This is the implementation of two-argumented String#slice.
Definition string.c:3363
#define rb_str_cat_cstr(buf, str)
Identical to rb_str_cat(), except it assumes the passed pointer is a pointer to a C string.
Definition string.h:1681
VALUE rb_str_unlocktmp(VALUE str)
Releases a lock formerly obtained by rb_str_locktmp().
Definition string.c:3482
VALUE rb_utf8_str_new_static(const char *ptr, long len)
Identical to rb_str_new_static(), except it generates a string of "UTF-8" encoding instead of "binary...
Definition string.c:1238
#define rb_utf8_str_new(str, len)
Identical to rb_str_new, except it generates a string of "UTF-8" encoding.
Definition string.h:1574
void rb_str_modify_expand(VALUE str, long capa)
Identical to rb_str_modify(), except it additionally expands the capacity of the receiver.
Definition string.c:2816
VALUE rb_str_dump(VALUE str)
"Inverse" of rb_eval_string().
Definition string.c:8268
VALUE rb_locale_str_new(const char *ptr, long len)
Identical to rb_str_new(), except it generates a string of "locale" encoding.
Definition string.c:1448
VALUE rb_str_buf_new(long capa)
Allocates a "string buffer".
Definition string.c:1769
VALUE rb_str_length(VALUE)
Identical to rb_str_strlen(), except it returns the value in rb_cInteger.
Definition string.c:2506
#define rb_str_new_cstr(str)
Identical to rb_str_new, except it assumes the passed pointer is a pointer to a C string.
Definition string.h:1539
VALUE rb_str_drop_bytes(VALUE str, long len)
Shrinks the given string for the given number of bytes.
Definition string.c:5858
VALUE rb_str_split(VALUE str, const char *delim)
Divides the given string based on the given delimiter.
Definition string.c:10803
VALUE rb_usascii_str_new_static(const char *ptr, long len)
Identical to rb_str_new_static(), except it generates a string of "US ASCII" encoding instead of "bin...
Definition string.c:1232
VALUE rb_str_intern(VALUE str)
Identical to rb_to_symbol(), except it assumes the receiver being an instance of RString.
Definition symbol.c:1085
VALUE rb_obj_as_string(VALUE obj)
Try converting an object to its stringised representation using its to_s method, if any.
Definition string.c:1902
VALUE rb_ivar_set(VALUE obj, ID name, VALUE val)
Identical to rb_iv_set(), except it accepts the name as an ID instead of a C string.
Definition variable.c:2141
VALUE rb_ivar_defined(VALUE obj, ID name)
Queries if the instance variable is defined at the object.
Definition variable.c:2201
int rb_respond_to(VALUE obj, ID mid)
Queries if the object responds to the method.
Definition vm_method.c:3693
void rb_undef_alloc_func(VALUE klass)
Deletes the allocator function of a class.
Definition vm_method.c:1846
void rb_define_alloc_func(VALUE klass, rb_alloc_func_t func)
Sets the allocator function of a class.
static ID rb_intern_const(const char *str)
This is a "tiny optimisation" over rb_intern().
Definition symbol.h:285
VALUE rb_sym2str(VALUE symbol)
Obtain a frozen string representation of a symbol (not including the leading colon).
Definition symbol.c:1148
VALUE rb_to_symbol(VALUE name)
Identical to rb_intern_str(), except it generates a dynamic symbol if necessary.
Definition string.c:14146
ID rb_to_id(VALUE str)
Identical to rb_intern_str(), except it tries to convert the parameter object to an instance of rb_cS...
Definition string.c:14136
int capa
Designed capacity of the buffer.
Definition io.h:11
int off
Offset inside of ptr.
Definition io.h:5
int len
Length of the buffer.
Definition io.h:8
#define RB_OBJ_SET_SHAREABLE(obj)
Wrapper of rb_obj_set_shareable().
Definition ractor.h:290
#define RB_OBJ_SHAREABLE_P(obj)
Queries if the passed object has previously classified as shareable or not.
Definition ractor.h:255
long rb_reg_search(VALUE re, VALUE str, long pos, int dir)
Runs the passed regular expression over the passed string.
Definition re.c:2000
VALUE rb_reg_regcomp(VALUE str)
Creates a new instance of rb_cRegexp.
Definition re.c:3675
VALUE rb_str_format(int argc, const VALUE *argv, VALUE fmt)
Formats a string.
Definition sprintf.c:974
VALUE rb_yield(VALUE val)
Yields the block.
Definition vm_eval.c:1378
#define MEMCPY(p1, p2, type, n)
Handy macro to call memcpy.
Definition memory.h:372
#define ALLOCA_N(type, n)
Definition memory.h:292
#define MEMZERO(p, type, n)
Handy macro to erase a region of memory.
Definition memory.h:360
#define RB_GC_GUARD(v)
Prevents premature destruction of local objects.
Definition memory.h:167
void rb_define_hooked_variable(const char *q, VALUE *w, type *e, void_type *r)
Define a function-backended global variable.
VALUE type(ANYARGS)
ANYARGS-ed function type.
void rb_hash_foreach(VALUE q, int_type *w, VALUE e)
Iteration over the given hash.
VALUE rb_ensure(type *q, VALUE w, type *e, VALUE r)
An equivalent of ensure clause.
Defines RBIMPL_ATTR_NONSTRING.
static int RARRAY_LENINT(VALUE ary)
Identical to rb_array_len(), except it differs for the return type.
Definition rarray.h:280
#define RARRAY_CONST_PTR
Just another name of rb_array_const_ptr.
Definition rarray.h:51
static VALUE RBASIC_CLASS(VALUE obj)
Queries the class of an object.
Definition rbasic.h:166
#define RBASIC(obj)
Convenient casting macro.
Definition rbasic.h:40
#define RHASH_SIZE(h)
Queries the size of the hash.
Definition rhash.h:57
static VALUE RREGEXP_SRC(VALUE rexp)
Convenient getter function.
Definition rregexp.h:102
#define StringValue(v)
Ensures that the parameter object is a String.
Definition rstring.h:66
VALUE rb_str_export_locale(VALUE obj)
Identical to rb_str_export(), except it converts into the locale encoding instead.
Definition string.c:1478
char * rb_string_value_cstr(volatile VALUE *ptr)
Identical to rb_string_value_ptr(), except it additionally checks for the contents for viability as a...
Definition string.c:3018
static int RSTRING_LENINT(VALUE str)
Identical to RSTRING_LEN(), except it differs for the return type.
Definition rstring.h:438
static char * RSTRING_END(VALUE str)
Queries the end of the contents pointer of the string.
Definition rstring.h:409
#define RSTRING_GETMEM(str, ptrvar, lenvar)
Convenient macro to obtain the contents and length at once.
Definition rstring.h:450
VALUE rb_string_value(volatile VALUE *ptr)
Identical to rb_str_to_str(), except it fills the passed pointer with the converted object.
Definition string.c:2881
#define RSTRING(obj)
Convenient casting macro.
Definition rstring.h:41
VALUE rb_str_export(VALUE obj)
Identical to rb_str_to_str(), except it additionally converts the string into default external encodi...
Definition string.c:1472
char * rb_string_value_ptr(volatile VALUE *ptr)
Identical to rb_str_to_str(), except it returns the converted string's backend memory region.
Definition string.c:2894
VALUE rb_str_to_str(VALUE obj)
Identical to rb_check_string_type(), except it raises exceptions in case of conversion failures.
Definition string.c:1830
#define StringValueCStr(v)
Identical to StringValuePtr, except it additionally checks for the contents for viability as a C stri...
Definition rstring.h:89
#define DATA_PTR(obj)
Convenient casting macro for backward compatibility.
Definition rtypeddata.h:439
#define TypedData_Wrap_Struct(klass, data_type, sval)
Converts sval, a pointer to your struct, into a Ruby object.
Definition rtypeddata.h:557
VALUE rb_require(const char *feature)
Identical to rb_require_string(), except it takes C's string instead of Ruby's.
Definition load.c:1528
#define errno
Ractor-aware version of errno.
Definition ruby.h:388
#define RB_NUM2SSIZE
Converts an instance of rb_cInteger into C's ssize_t.
Definition size_t.h:49
#define RTEST
This is an old name of RB_TEST.
#define _(args)
This was a transition path from K&R to ANSI.
Definition stdarg.h:35
VALUE flags
Per-object flags.
Definition rbasic.h:81
Ruby's String.
Definition rstring.h:196
struct RBasic basic
Basic part, including flags and class.
Definition rstring.h:199
union RString::@60::@61::@63 aux
Auxiliary info.
long capa
Capacity of *ptr.
Definition rstring.h:232
long len
Length of the string, not including terminating NUL character.
Definition rstring.h:206
struct RString::@60::@61 heap
Strings that use separated memory region for contents use this pattern.
struct RString::@60::@62 embed
Embedded contents.
VALUE shared
Parent of the string.
Definition rstring.h:240
char * ptr
Pointer to the contents of the string.
Definition rstring.h:222
union RString::@60 as
String's specific fields.
This is the struct that holds necessary info for a struct.
Definition rtypeddata.h:242
Definition string.c:9192
void rb_nativethread_lock_lock(rb_nativethread_lock_t *lock)
Blocks until the current thread obtains a lock.
Definition thread.c:319
uintptr_t ID
Type that represents a Ruby identifier such as a variable name.
Definition value.h:52
uintptr_t VALUE
Type that represents a Ruby object.
Definition value.h:40
static enum ruby_value_type rb_type(VALUE obj)
Identical to RB_BUILTIN_TYPE(), except it can also accept special constants.
Definition value_type.h:225
static void Check_Type(VALUE v, enum ruby_value_type t)
Identical to RB_TYPE_P(), except it raises exceptions on predication failure.
Definition value_type.h:425
static bool RB_TYPE_P(VALUE obj, enum ruby_value_type t)
Queries if the given object is of given type.
Definition value_type.h:376
ruby_value_type
C-level type of an object.
Definition value_type.h:113