Ruby 4.1.0dev (2026-10-02 revision 3b8479580399341b79b91b1953f66dab339c92a6)
string.c (3b8479580399341b79b91b1953f66dab339c92a6)
1/**********************************************************************
2
3 string.c -
4
5 $Author$
6 created at: Mon Aug 9 17:12:58 JST 1993
7
8 Copyright (C) 1993-2007 Yukihiro Matsumoto
9 Copyright (C) 2000 Network Applied Communication Laboratory, Inc.
10 Copyright (C) 2000 Information-technology Promotion Agency, Japan
11
12**********************************************************************/
13
14#include "ruby/internal/config.h"
15
16#include <ctype.h>
17#include <errno.h>
18#include <math.h>
19
20#ifdef HAVE_UNISTD_H
21# include <unistd.h>
22#endif
23
24#include "debug_counter.h"
25#include "encindex.h"
26#include "id.h"
27#include "internal.h"
28#include "internal/array.h"
29#include "internal/bits.h"
30#include "internal/compar.h"
31#include "internal/compilers.h"
32#include "internal/concurrent_set.h"
33#include "internal/encoding.h"
34#include "internal/error.h"
35#include "internal/gc.h"
36#include "internal/hash.h"
37#include "internal/numeric.h"
38#include "internal/object.h"
39#include "internal/proc.h"
40#include "internal/re.h"
41#include "internal/sanitizers.h"
42#include "internal/simd.h"
43#include "internal/string.h"
44#include "internal/transcode.h"
45#include "probes.h"
46#include "ruby/encoding.h"
47#include "ruby/re.h"
48#include "ruby/thread.h"
49#include "ruby/util.h"
50#include "ruby/ractor.h"
51#include "ruby_assert.h"
52#include "shape.h"
53#include "vm_core.h"
54#include "vm_sync.h"
55#include "zjit.h"
57
58#if defined HAVE_CRYPT_R
59# if defined HAVE_CRYPT_H
60# include <crypt.h>
61# endif
62#elif !defined HAVE_CRYPT
63# include "missing/crypt.h"
64# define HAVE_CRYPT_R 1
65#endif
66
67#undef rb_str_new
68#undef rb_usascii_str_new
69#undef rb_utf8_str_new
70#undef rb_enc_str_new
71#undef rb_str_new_cstr
72#undef rb_usascii_str_new_cstr
73#undef rb_utf8_str_new_cstr
74#undef rb_enc_str_new_cstr
75#undef rb_external_str_new_cstr
76#undef rb_locale_str_new_cstr
77#undef rb_str_dup_frozen
78#undef rb_str_buf_new_cstr
79#undef rb_str_buf_cat
80#undef rb_str_buf_cat2
81#undef rb_str_cat2
82#undef rb_str_cat_cstr
83#undef rb_fstring_cstr
84
87
88/* Flags of RString
89 *
90 * 0: STR_SHARED (equal to ELTS_SHARED)
91 * The string is shared. The buffer this string points to is owned by
92 * another string (the shared root).
93 * 1: RSTRING_NOEMBED
94 * The string is not embedded. When a string is embedded, the contents
95 * follow the header. When a string is not embedded, the contents is
96 * on a separately allocated buffer.
97 * 2: STR_CHILLED (will be frozen in a future version)
98 * The string was allocated as a literal in a file without an explicit `frozen_string_literal` comment.
99 * It emits a deprecation warning when mutated for the first time.
100 * 4: STR_PRECOMPUTED_HASH
101 * The string is embedded and has its precomputed hashcode stored
102 * after the terminator.
103 * 5: STR_SHARED_ROOT
104 * Other strings may point to the contents of this string. When this
105 * flag is set, STR_SHARED must not be set.
106 * 6: STR_BORROWED
107 * When RSTRING_NOEMBED is set and klass is 0, this string is unsafe
108 * to be unshared by rb_str_tmp_frozen_release.
109 * 7: STR_TMPLOCK
110 * The pointer to the buffer is passed to a system call such as
111 * read(2). Any modification and realloc is prohibited.
112 * 8-9: ENC_CODERANGE
113 * Stores the coderange of the string.
114 * 10-16: ENCODING
115 * Stores the encoding of the string.
116 * 17: RSTRING_FSTR
117 * The string is a fstring. The string is deduplicated in the fstring
118 * table.
119 * 18: STR_NOFREE
120 * Do not free this string's buffer when the string is reclaimed
121 * by the garbage collector. Used for when the string buffer is a C
122 * string literal.
123 * 19: STR_FAKESTR
124 * The string is not allocated or managed by the garbage collector.
125 * Typically, the string object header (struct RString) is temporarily
126 * allocated on C stack.
127 */
128
129#define RUBY_MAX_CHAR_LEN 16
130#define STR_PRECOMPUTED_HASH FL_USER4
131#define STR_SHARED_ROOT FL_USER5
132#define STR_BORROWED FL_USER6
133#define STR_TMPLOCK FL_USER7
134#define STR_NOFREE FL_USER18
135
136#define STR_SET_NOEMBED(str) do {\
137 FL_SET((str), STR_NOEMBED);\
138 FL_UNSET((str), STR_SHARED | STR_SHARED_ROOT | STR_BORROWED);\
139} while (0)
140#define STR_SET_EMBED(str) FL_UNSET((str), STR_NOEMBED | STR_SHARED | STR_NOFREE)
141
142#define STR_SET_LEN(str, n) do { \
143 RSTRING(str)->len = (n); \
144} while (0)
145
146#define TERM_LEN(str) (rb_str_enc_fastpath(str) ? 1 : rb_enc_mbminlen(rb_enc_from_index(ENCODING_GET(str))))
147#define TERM_FILL(ptr, termlen) do {\
148 char *const term_fill_ptr = (ptr);\
149 const int term_fill_len = (termlen);\
150 *term_fill_ptr = '\0';\
151 if (UNLIKELY(term_fill_len > 1))\
152 memset(term_fill_ptr, 0, term_fill_len);\
153} while (0)
154
155#define RESIZE_CAPA(str,capacity) do {\
156 const int termlen = TERM_LEN(str);\
157 RESIZE_CAPA_TERM(str,capacity,termlen);\
158} while (0)
159#define RESIZE_CAPA_TERM(str,capacity,termlen) do {\
160 if (STR_EMBED_P(str)) {\
161 if (str_embed_capa(str) < capacity + termlen) {\
162 char *const tmp = ALLOC_N(char, (size_t)(capacity) + (termlen));\
163 const long tlen = RSTRING_LEN(str);\
164 memcpy(tmp, RSTRING_PTR(str), str_embed_capa(str));\
165 RSTRING(str)->as.heap.ptr = tmp;\
166 RSTRING(str)->len = tlen;\
167 STR_SET_NOEMBED(str);\
168 RSTRING(str)->as.heap.aux.capa = (capacity);\
169 }\
170 }\
171 else {\
172 RUBY_ASSERT(!FL_TEST((str), STR_SHARED)); \
173 SIZED_REALLOC_N(RSTRING(str)->as.heap.ptr, char, \
174 (size_t)(capacity) + (termlen), STR_HEAP_SIZE(str)); \
175 RSTRING(str)->as.heap.aux.capa = (capacity);\
176 }\
177} while (0)
178
179#define STR_SET_SHARED(str, shared_str) do { \
180 if (!FL_TEST(str, STR_FAKESTR)) { \
181 RUBY_ASSERT(RSTRING_PTR(shared_str) <= RSTRING_PTR(str)); \
182 RUBY_ASSERT(RSTRING_PTR(str) <= RSTRING_PTR(shared_str) + RSTRING_LEN(shared_str)); \
183 RB_OBJ_WRITE((str), &RSTRING(str)->as.heap.aux.shared, (shared_str)); \
184 FL_SET((str), STR_SHARED); \
185 rb_gc_register_pinning_obj(str); \
186 FL_SET((shared_str), STR_SHARED_ROOT); \
187 if (RBASIC_CLASS((shared_str)) == 0) /* for CoW-friendliness */ \
188 FL_SET_RAW((shared_str), STR_BORROWED); \
189 } \
190} while (0)
191
192#define STR_HEAP_PTR(str) (RSTRING(str)->as.heap.ptr)
193#define STR_HEAP_SIZE(str) ((size_t)RSTRING(str)->as.heap.aux.capa + TERM_LEN(str))
194/* TODO: include the terminator size in capa. */
195
196#define STR_ENC_GET(str) get_encoding(str)
197
198static inline bool
199zero_filled(const char *s, int n)
200{
201 for (; n > 0; --n) {
202 if (*s++) return false;
203 }
204 return true;
205}
206
207#if !defined SHARABLE_MIDDLE_SUBSTRING
208# define SHARABLE_MIDDLE_SUBSTRING 0
209#endif
210
211static inline bool
212SHARABLE_SUBSTRING_P(VALUE str, long beg, long len)
213{
214#if SHARABLE_MIDDLE_SUBSTRING
215 return true;
216#else
217 long end = beg + len;
218 long source_len = RSTRING_LEN(str);
219 return end == source_len || zero_filled(RSTRING_PTR(str) + end, TERM_LEN(str));
220#endif
221}
222
223static inline long
224str_embed_capa(VALUE str)
225{
226 return rb_obj_shape_slot_size(str) - offsetof(struct RString, as.embed.ary);
227}
228
229bool
230rb_str_reembeddable_p(VALUE str)
231{
232 return !FL_TEST(str, STR_NOFREE|STR_SHARED_ROOT|STR_SHARED);
233}
234
235/* True when other strings read this string's bytes out of its own slot, so the slot
236 * contents must stay valid for as long as the object does. */
237bool
238rb_str_embedded_shared_root_p(VALUE str)
239{
240 return STR_EMBED_P(str) && FL_TEST(str, STR_SHARED_ROOT);
241}
242
243static inline size_t
244rb_str_embed_size(long capa, long termlen)
245{
246 size_t size = offsetof(struct RString, as.embed.ary) + capa + termlen;
247 if (size < sizeof(struct RString)) size = sizeof(struct RString);
248 return size;
249}
250
251size_t
252rb_str_size_as_embedded(VALUE str)
253{
254 size_t real_size;
255 if (STR_EMBED_P(str)) {
256 size_t capa = RSTRING(str)->len;
257 if (FL_TEST_RAW(str, STR_PRECOMPUTED_HASH)) capa += sizeof(st_index_t);
258
259 real_size = rb_str_embed_size(capa, TERM_LEN(str));
260 }
261 /* if the string is not currently embedded, but it can be embedded, how
262 * much space would it require */
263 else if (rb_str_reembeddable_p(str)) {
264 size_t capa = RSTRING(str)->as.heap.aux.capa;
265 if (FL_TEST_RAW(str, STR_PRECOMPUTED_HASH)) capa += sizeof(st_index_t);
266
267 real_size = rb_str_embed_size(capa, TERM_LEN(str));
268 }
269 else {
270 real_size = sizeof(struct RString);
271 }
272
273 return real_size;
274}
275
276static inline bool
277STR_EMBEDDABLE_P(long len, long termlen)
278{
279 return rb_gc_size_allocatable_p(rb_str_embed_size(len, termlen));
280}
281
282/* Substrings and duplicated strings that need a slot larger than this are shared
283 * instead of copied. Larger slots hold fewer objects per page and trigger GC
284 * more often, which outweighs the copy they save; see [Feature #22186] for the
285 * benchmarks. */
286#define STR_COPY_MAX_EMBED_SIZE 256
287
288static VALUE str_replace_shared_without_enc(VALUE str2, VALUE str);
289static VALUE str_new_frozen(VALUE klass, VALUE orig);
290static VALUE str_new_frozen_buffer(VALUE klass, VALUE orig, int copy_encoding);
291static VALUE str_new_static(VALUE klass, const char *ptr, long len, int encindex);
292static VALUE str_new(VALUE klass, const char *ptr, long len);
293static void str_make_independent_expand(VALUE str, long len, long expand, const int termlen);
294static inline void str_modifiable(VALUE str);
295static VALUE rb_str_downcase(int argc, VALUE *argv, VALUE str);
296static inline VALUE str_alloc_embed(VALUE klass, size_t capa);
297
298static inline void
299str_make_independent(VALUE str)
300{
301 long len = RSTRING_LEN(str);
302 int termlen = TERM_LEN(str);
303 str_make_independent_expand((str), len, 0L, termlen);
304}
305
306static inline int str_dependent_p(VALUE str);
307
308void
309rb_str_make_independent(VALUE str)
310{
311 if (str_dependent_p(str)) {
312 str_make_independent(str);
313 }
314}
315
316void
317rb_str_make_embedded(VALUE str)
318{
319 RUBY_ASSERT(rb_str_reembeddable_p(str));
320 RUBY_ASSERT(!STR_EMBED_P(str));
321
322 int termlen = TERM_LEN(str);
323 char *buf = RSTRING(str)->as.heap.ptr;
324 long old_capa = RSTRING(str)->as.heap.aux.capa + termlen;
325 long len = RSTRING(str)->len;
326
327 STR_SET_EMBED(str);
328 STR_SET_LEN(str, len);
329
330 if (len > 0) {
331 memcpy(RSTRING_PTR(str), buf, len);
332 SIZED_FREE_N(buf, old_capa);
333 }
334
335 TERM_FILL(RSTRING(str)->as.embed.ary + len, termlen);
336}
337
338void
339rb_debug_rstring_null_ptr(const char *func)
340{
341 fprintf(stderr, "%s is returning NULL!! "
342 "SIGSEGV is highly expected to follow immediately.\n"
343 "If you could reproduce, attach your debugger here, "
344 "and look at the passed string.\n",
345 func);
346}
347
348/* symbols for [up|down|swap]case/capitalize options */
349static VALUE sym_ascii, sym_turkic, sym_lithuanian, sym_fold;
350
351static rb_encoding *
352get_encoding(VALUE str)
353{
354 return rb_enc_from_index(ENCODING_GET(str));
355}
356
357static void
358mustnot_broken(VALUE str)
359{
360 if (is_broken_string(str)) {
361 rb_raise(rb_eArgError, "invalid byte sequence in %s", rb_enc_name(STR_ENC_GET(str)));
362 }
363}
364
365static void
366mustnot_wchar(VALUE str)
367{
368 rb_encoding *enc = STR_ENC_GET(str);
369 if (rb_enc_mbminlen(enc) > 1) {
370 rb_raise(rb_eArgError, "wide char encoding: %s", rb_enc_name(enc));
371 }
372}
373
374static VALUE register_fstring(VALUE str, bool copy, bool force_precompute_hash);
375
376#if SIZEOF_LONG == SIZEOF_VOIDP
377#define PRECOMPUTED_FAKESTR_HASH 1
378#else
379#endif
380
381static inline bool
382BARE_STRING_P(VALUE str)
383{
384 return RBASIC_CLASS(str) == rb_cString && !rb_obj_shape_has_ivars(str);
385}
386
387static inline st_index_t
388str_do_hash(VALUE str)
389{
390 st_index_t h = rb_memhash((const void *)RSTRING_PTR(str), RSTRING_LEN(str));
391 int e = RSTRING_LEN(str) ? ENCODING_GET(str) : 0;
392 if (e && !is_ascii_string(str)) {
393 h = rb_hash_end(rb_hash_uint32(h, (uint32_t)e));
394 }
395 return h;
396}
397
398static VALUE
399str_store_precomputed_hash(VALUE str, st_index_t hash)
400{
401 RUBY_ASSERT(!FL_TEST_RAW(str, STR_PRECOMPUTED_HASH));
402 RUBY_ASSERT(STR_EMBED_P(str));
403
404#if RUBY_DEBUG
405 size_t used_bytes = (RSTRING_LEN(str) + TERM_LEN(str));
406 size_t free_bytes = str_embed_capa(str) - used_bytes;
407 RUBY_ASSERT(free_bytes >= sizeof(st_index_t));
408#endif
409
410 memcpy(RSTRING_END(str) + TERM_LEN(str), &hash, sizeof(hash));
411
412 FL_SET(str, STR_PRECOMPUTED_HASH);
413
414 return str;
415}
416
417VALUE
418rb_fstring(VALUE str)
419{
420 VALUE fstr;
421 int bare;
422
423 Check_Type(str, T_STRING);
424
425 if (FL_TEST(str, RSTRING_FSTR))
426 return str;
427
428 bare = BARE_STRING_P(str);
429 if (!bare) {
430 if (STR_EMBED_P(str)) {
431 OBJ_FREEZE(str);
432 return str;
433 }
434
435 if (FL_TEST_RAW(str, STR_SHARED_ROOT | STR_SHARED) == STR_SHARED_ROOT) {
437 return str;
438 }
439 }
440
441 if (!FL_TEST_RAW(str, FL_FREEZE | STR_NOFREE | STR_CHILLED))
442 rb_str_resize(str, RSTRING_LEN(str));
443
444 fstr = register_fstring(str, false, false);
445
446 if (!bare) {
447 str_replace_shared_without_enc(str, fstr);
448 OBJ_FREEZE(str);
449 return str;
450 }
451 return fstr;
452}
453
454static VALUE fstring_table_obj;
455
456static VALUE
457fstring_concurrent_set_hash(VALUE str)
458{
459#ifdef PRECOMPUTED_FAKESTR_HASH
460 st_index_t h;
461 if (FL_TEST_RAW(str, STR_FAKESTR)) {
462 // register_fstring precomputes the hash and stores it in capa for fake strings
463 h = (st_index_t)RSTRING(str)->as.heap.aux.capa;
464 }
465 else {
466 h = rb_str_hash(str);
467 }
468 // rb_str_hash doesn't include the encoding for ascii only strings, so
469 // we add it to avoid common collisions between `:sym.name` (ASCII) and `"sym"` (UTF-8)
470 return (VALUE)rb_hash_end(rb_hash_uint32(h, (uint32_t)ENCODING_GET_INLINED(str)));
471#else
472 return (VALUE)rb_str_hash(str);
473#endif
474}
475
476static bool
477fstring_concurrent_set_cmp(VALUE a, VALUE b)
478{
479 long alen, blen;
480 const char *aptr, *bptr;
481
484
485 RSTRING_GETMEM(a, aptr, alen);
486 RSTRING_GETMEM(b, bptr, blen);
487 return (alen == blen &&
488 ENCODING_GET(a) == ENCODING_GET(b) &&
489 memcmp(aptr, bptr, alen) == 0);
490}
491
493 bool copy;
494 bool force_precompute_hash;
495};
496
497static VALUE
498fstring_concurrent_set_create(VALUE str, void *data)
499{
500 struct fstr_create_arg *arg = data;
501
502 // Unless the string is empty or binary, its coderange has been precomputed.
503 int coderange = ENC_CODERANGE(str);
504
505 if (FL_TEST_RAW(str, STR_FAKESTR)) {
506 if (arg->copy) {
507 VALUE new_str;
508 long len = RSTRING_LEN(str);
509 long capa = len + sizeof(st_index_t);
510 int term_len = TERM_LEN(str);
511
512 if (arg->force_precompute_hash && STR_EMBEDDABLE_P(capa, term_len)) {
513 new_str = str_alloc_embed(rb_cString, capa + term_len);
514 memcpy(RSTRING_PTR(new_str), RSTRING_PTR(str), len);
515 STR_SET_LEN(new_str, RSTRING_LEN(str));
516 TERM_FILL(RSTRING_END(new_str), TERM_LEN(str));
517 rb_enc_copy(new_str, str);
518 str_store_precomputed_hash(new_str, str_do_hash(str));
519 }
520 else {
521 new_str = str_new(rb_cString, RSTRING(str)->as.heap.ptr, RSTRING(str)->len);
522 rb_enc_copy(new_str, str);
523#ifdef PRECOMPUTED_FAKESTR_HASH
524 if (rb_str_capacity(new_str) >= RSTRING_LEN(str) + term_len + sizeof(st_index_t)) {
525 str_store_precomputed_hash(new_str, (st_index_t)RSTRING(str)->as.heap.aux.capa);
526 }
527#endif
528 }
529 str = new_str;
530 }
531 else {
532 str = str_new_static(rb_cString, RSTRING(str)->as.heap.ptr,
533 RSTRING(str)->len,
534 ENCODING_GET(str));
535 }
536 OBJ_FREEZE(str);
537 }
538 else {
539 if (!OBJ_FROZEN(str) || CHILLED_STRING_P(str)) {
540 str = str_new_frozen(rb_cString, str);
541 }
542 if (STR_SHARED_P(str)) { /* str should not be shared */
543 /* shared substring */
544 str_make_independent(str);
546 }
547 if (!BARE_STRING_P(str)) {
548 str = str_new_frozen(rb_cString, str);
549 }
550 }
551
552 ENC_CODERANGE_SET(str, coderange);
553 RBASIC(str)->flags |= RSTRING_FSTR;
554 if (!RB_OBJ_SHAREABLE_P(str)) {
556 }
557 RUBY_ASSERT((rb_gc_verify_shareable(str), 1));
560 RUBY_ASSERT(!FL_TEST_RAW(str, STR_FAKESTR));
561 RUBY_ASSERT(!rb_obj_shape_has_ivars(str));
563 RUBY_ASSERT(!rb_objspace_garbage_object_p(str));
564
565 return str;
566}
567
568static const struct rb_concurrent_set_funcs fstring_concurrent_set_funcs = {
569 .hash = fstring_concurrent_set_hash,
570 .cmp = fstring_concurrent_set_cmp,
571 .create = fstring_concurrent_set_create,
572 .free = NULL,
573};
574
575void
576Init_fstring_table(void)
577{
578 fstring_table_obj = rb_concurrent_set_new(&fstring_concurrent_set_funcs, 8192);
579 rb_gc_register_address(&fstring_table_obj);
580}
581
582static VALUE
583register_fstring(VALUE str, bool copy, bool force_precompute_hash)
584{
585 struct fstr_create_arg args = {
586 .copy = copy,
587 .force_precompute_hash = force_precompute_hash
588 };
589
590#if SIZEOF_VOIDP == SIZEOF_LONG
591 if (FL_TEST_RAW(str, STR_FAKESTR)) {
592 // if the string hasn't been interned, we'll need the hash twice, so we
593 // compute it once and store it in capa
594 RSTRING(str)->as.heap.aux.capa = (long)str_do_hash(str);
595 }
596#endif
597
598 VALUE result = rb_concurrent_set_find_or_insert(&fstring_table_obj, str, &args);
599
600 RUBY_ASSERT(!rb_objspace_garbage_object_p(result));
602 RUBY_ASSERT(OBJ_FROZEN(result));
604 RUBY_ASSERT((rb_gc_verify_shareable(result), 1));
605 RUBY_ASSERT(!FL_TEST_RAW(result, STR_FAKESTR));
607
608 return result;
609}
610
611bool
612rb_obj_is_fstring_table(VALUE obj)
613{
614 ASSERT_vm_locking();
615
616 return obj == fstring_table_obj;
617}
618
619void
620rb_gc_free_fstring(VALUE obj)
621{
622 ASSERT_vm_locking_with_barrier();
623
624 RUBY_ASSERT(FL_TEST(obj, RSTRING_FSTR));
626 RUBY_ASSERT(!FL_TEST(obj, STR_SHARED));
627
628 rb_concurrent_set_delete_by_identity(fstring_table_obj, obj);
629
630 RB_DEBUG_COUNTER_INC(obj_str_fstr);
631
632 FL_UNSET(obj, RSTRING_FSTR);
633}
634
635void
636rb_fstring_foreach_with_replace(int (*callback)(VALUE *str, void *data), void *data)
637{
638 if (fstring_table_obj) {
639 rb_concurrent_set_foreach_with_replace(fstring_table_obj, callback, data);
640 }
641}
642
643static VALUE
644setup_fake_str(struct RString *fake_str, const char *name, long len, int encidx)
645{
646 fake_str->basic.flags = T_STRING|RSTRING_NOEMBED|STR_NOFREE|STR_FAKESTR;
647 RBASIC_SET_FULL_SHAPE_ID((VALUE)fake_str, ROOT_SHAPE_ID | SHAPE_ID_LAYOUT_OTHER);
648
649 if (!name) {
651 name = "";
652 }
653
654 ENCODING_SET_INLINED((VALUE)fake_str, encidx);
655
656 RBASIC_SET_CLASS_RAW((VALUE)fake_str, rb_cString);
657 fake_str->len = len;
658 fake_str->as.heap.ptr = (char *)name;
659 fake_str->as.heap.aux.capa = len;
660 return (VALUE)fake_str;
661}
662
663/*
664 * set up a fake string which refers a static string literal.
665 */
666VALUE
667rb_setup_fake_str(struct RString *fake_str, const char *name, long len, rb_encoding *enc)
668{
669 return setup_fake_str(fake_str, name, len, rb_enc_to_index(enc));
670}
671
672/*
673 * rb_fstring_new and rb_fstring_cstr family create or lookup a frozen
674 * shared string which refers a static string literal. `ptr` must
675 * point a constant string.
676 */
677VALUE
678rb_fstring_new(const char *ptr, long len)
679{
680 struct RString fake_str = {RBASIC_INIT};
681 return register_fstring(setup_fake_str(&fake_str, ptr, len, ENCINDEX_US_ASCII), false, false);
682}
683
684VALUE
685rb_fstring_enc_new(const char *ptr, long len, rb_encoding *enc)
686{
687 struct RString fake_str = {RBASIC_INIT};
688 return register_fstring(rb_setup_fake_str(&fake_str, ptr, len, enc), false, false);
689}
690
691VALUE
692rb_fstring_cstr(const char *ptr)
693{
694 return rb_fstring_new(ptr, strlen(ptr));
695}
696
697static inline bool
698single_byte_optimizable(VALUE str)
699{
700 int encindex = ENCODING_GET(str);
701 switch (encindex) {
702 case ENCINDEX_ASCII_8BIT:
703 case ENCINDEX_US_ASCII:
704 return true;
705 case ENCINDEX_UTF_8:
706 // For UTF-8 it's worth scanning the string coderange when unknown.
707 return rb_enc_str_coderange(str) == ENC_CODERANGE_7BIT;
708 }
709 /* Conservative. It may be ENC_CODERANGE_UNKNOWN. */
710 if (ENC_CODERANGE(str) == ENC_CODERANGE_7BIT) {
711 return true;
712 }
713
714 if (rb_enc_mbmaxlen(rb_enc_from_index(encindex)) == 1) {
715 return true;
716 }
717
718 /* Conservative. Possibly single byte.
719 * "\xa1" in Shift_JIS for example. */
720 return false;
721}
722
724
725static inline const char *
726search_nonascii(const char *p, const char *e)
727{
728 const char *s, *t;
729
730 if (p < e && !ISASCII(*p)) {
731 return p;
732 }
733
734#if defined(__STDC_VERSION__) && (__STDC_VERSION__ >= 199901L)
735# if SIZEOF_UINTPTR_T == 8
736# define NONASCII_MASK UINT64_C(0x8080808080808080)
737# elif SIZEOF_UINTPTR_T == 4
738# define NONASCII_MASK UINT32_C(0x80808080)
739# else
740# error "don't know what to do."
741# endif
742#else
743# if SIZEOF_UINTPTR_T == 8
744# define NONASCII_MASK ((uintptr_t)0x80808080UL << 32 | (uintptr_t)0x80808080UL)
745# elif SIZEOF_UINTPTR_T == 4
746# define NONASCII_MASK 0x80808080UL /* or...? */
747# else
748# error "don't know what to do."
749# endif
750#endif
751
752 if (UNALIGNED_WORD_ACCESS || e - p >= SIZEOF_VOIDP) {
753#if !UNALIGNED_WORD_ACCESS
754 if ((uintptr_t)p % SIZEOF_VOIDP) {
755 int l = SIZEOF_VOIDP - (uintptr_t)p % SIZEOF_VOIDP;
756 p += l;
757 switch (l) {
758 default: UNREACHABLE;
759#if SIZEOF_VOIDP > 4
760 case 7: if (p[-7]&0x80) return p-7;
761 case 6: if (p[-6]&0x80) return p-6;
762 case 5: if (p[-5]&0x80) return p-5;
763 case 4: if (p[-4]&0x80) return p-4;
764#endif
765 case 3: if (p[-3]&0x80) return p-3;
766 case 2: if (p[-2]&0x80) return p-2;
767 case 1: if (p[-1]&0x80) return p-1;
768 case 0: break;
769 }
770 }
771#endif
772#if defined(HAVE_BUILTIN___BUILTIN_ASSUME_ALIGNED) &&! UNALIGNED_WORD_ACCESS
773#define aligned_ptr(value) \
774 __builtin_assume_aligned((value), sizeof(uintptr_t))
775#else
776#define aligned_ptr(value) (value)
777#endif
778 s = aligned_ptr(p);
779 t = (e - (SIZEOF_VOIDP-1));
780#undef aligned_ptr
781 for (;s < t; s += sizeof(uintptr_t)) {
782 uintptr_t word;
783 memcpy(&word, s, sizeof(word));
784 if (word & NONASCII_MASK) {
785#ifdef WORDS_BIGENDIAN
786 return (const char *)s + (nlz_intptr(word&NONASCII_MASK)>>3);
787#else
788 return (const char *)s + (ntz_intptr(word&NONASCII_MASK)>>3);
789#endif
790 }
791 }
792 p = (const char *)s;
793 }
794
795 switch (e - p) {
796 default: UNREACHABLE;
797#if SIZEOF_VOIDP > 4
798 case 7: if (e[-7]&0x80) return e-7;
799 case 6: if (e[-6]&0x80) return e-6;
800 case 5: if (e[-5]&0x80) return e-5;
801 case 4: if (e[-4]&0x80) return e-4;
802#endif
803 case 3: if (e[-3]&0x80) return e-3;
804 case 2: if (e[-2]&0x80) return e-2;
805 case 1: if (e[-1]&0x80) return e-1;
806 case 0: return NULL;
807 }
808}
809
810static int
811coderange_scan(const char *p, long len, rb_encoding *enc)
812{
813 const char *e = p + len;
814
815 if (rb_enc_to_index(enc) == rb_ascii8bit_encindex()) {
816 /* enc is ASCII-8BIT. ASCII-8BIT string never be broken. */
817 p = search_nonascii(p, e);
819 }
820
821 if (rb_enc_asciicompat(enc)) {
822 p = search_nonascii(p, e);
823 if (!p) return ENC_CODERANGE_7BIT;
824 for (;;) {
825 int ret = rb_enc_precise_mbclen(p, e, enc);
827 p += MBCLEN_CHARFOUND_LEN(ret);
828 if (p == e) break;
829 p = search_nonascii(p, e);
830 if (!p) break;
831 }
832 }
833 else {
834 while (p < e) {
835 int ret = rb_enc_precise_mbclen(p, e, enc);
837 p += MBCLEN_CHARFOUND_LEN(ret);
838 }
839 }
840 return ENC_CODERANGE_VALID;
841}
842
843long
844rb_str_coderange_scan_restartable(const char *s, const char *e, rb_encoding *enc, int *cr)
845{
846 const char *p = s;
847
848 if (*cr == ENC_CODERANGE_BROKEN)
849 return e - s;
850
851 if (rb_enc_to_index(enc) == rb_ascii8bit_encindex()) {
852 /* enc is ASCII-8BIT. ASCII-8BIT string never be broken. */
853 if (*cr == ENC_CODERANGE_VALID) return e - s;
854 p = search_nonascii(p, e);
856 return e - s;
857 }
858 else if (rb_enc_asciicompat(enc)) {
859 p = search_nonascii(p, e);
860 if (!p) {
861 if (*cr != ENC_CODERANGE_VALID) *cr = ENC_CODERANGE_7BIT;
862 return e - s;
863 }
864 for (;;) {
865 int ret = rb_enc_precise_mbclen(p, e, enc);
866 if (!MBCLEN_CHARFOUND_P(ret)) {
868 return p - s;
869 }
870 p += MBCLEN_CHARFOUND_LEN(ret);
871 if (p == e) break;
872 p = search_nonascii(p, e);
873 if (!p) break;
874 }
875 }
876 else {
877 while (p < e) {
878 int ret = rb_enc_precise_mbclen(p, e, enc);
879 if (!MBCLEN_CHARFOUND_P(ret)) {
881 return p - s;
882 }
883 p += MBCLEN_CHARFOUND_LEN(ret);
884 }
885 }
887 return e - s;
888}
889
890static inline void
891str_enc_copy(VALUE str1, VALUE str2)
892{
893 rb_enc_set_index(str1, ENCODING_GET(str2));
894}
895
896/* Like str_enc_copy, but does not check frozen status of str1.
897 * You should use this only if you're certain that str1 is not frozen. */
898static inline void
899str_enc_copy_direct(VALUE str1, VALUE str2)
900{
901 int inlined_encoding = RB_ENCODING_GET_INLINED(str2);
902 if (inlined_encoding == ENCODING_INLINE_MAX) {
903 rb_enc_set_index(str1, rb_enc_get_index(str2));
904 }
905 else {
906 ENCODING_SET_INLINED(str1, inlined_encoding);
907 }
908}
909
910static void
911rb_enc_cr_str_copy_for_substr(VALUE dest, VALUE src)
912{
913 /* this function is designed for copying encoding and coderange
914 * from src to new string "dest" which is made from the part of src.
915 */
916 str_enc_copy(dest, src);
917 if (RSTRING_LEN(dest) == 0) {
918 if (!rb_enc_asciicompat(STR_ENC_GET(src)))
920 else
922 return;
923 }
924 switch (ENC_CODERANGE(src)) {
927 break;
929 if (!rb_enc_asciicompat(STR_ENC_GET(src)) ||
930 search_nonascii(RSTRING_PTR(dest), RSTRING_END(dest)))
932 else
934 break;
935 default:
936 break;
937 }
938}
939
940static void
941rb_enc_cr_str_exact_copy(VALUE dest, VALUE src)
942{
943 str_enc_copy(dest, src);
945}
946
947static int
948enc_coderange_scan(VALUE str, rb_encoding *enc)
949{
950 return coderange_scan(RSTRING_PTR(str), RSTRING_LEN(str), enc);
951}
952
953int
954rb_enc_str_coderange_scan(VALUE str, rb_encoding *enc)
955{
956 return enc_coderange_scan(str, enc);
957}
958
959int
960rbimpl_enc_str_coderange_scan(VALUE str)
961{
962 int cr = enc_coderange_scan(str, get_encoding(str));
963 ENC_CODERANGE_SET(str, cr);
964 return cr;
965}
966
967#undef rb_enc_str_coderange
968int
969rb_enc_str_coderange(VALUE str)
970{
971 int cr = ENC_CODERANGE(str);
972
973 if (cr == ENC_CODERANGE_UNKNOWN) {
974 cr = rbimpl_enc_str_coderange_scan(str);
975 }
976 return cr;
977}
978#define rb_enc_str_coderange rb_enc_str_coderange_inline
979
980static inline bool
981rb_enc_str_asciicompat(VALUE str)
982{
983 int encindex = ENCODING_GET_INLINED(str);
984 return rb_str_encindex_fastpath(encindex) || rb_enc_asciicompat(rb_enc_get_from_index(encindex));
985}
986
987int
989{
990 switch(ENC_CODERANGE(str)) {
992 return rb_enc_str_asciicompat(str) && is_ascii_string(str);
994 return true;
995 default:
996 return false;
997 }
998}
999
1000static inline void
1001str_mod_check(VALUE s, const char *p, long len)
1002{
1003 if (RSTRING_PTR(s) != p || RSTRING_LEN(s) != len){
1004 rb_raise(rb_eRuntimeError, "string modified");
1005 }
1006}
1007
1008static size_t
1009str_capacity(VALUE str, const int termlen)
1010{
1011 if (STR_EMBED_P(str)) {
1012 return str_embed_capa(str) - termlen;
1013 }
1014 else if (FL_ANY_RAW(str, STR_SHARED|STR_NOFREE)) {
1015 return RSTRING(str)->len;
1016 }
1017 else {
1018 return RSTRING(str)->as.heap.aux.capa;
1019 }
1020}
1021
1022size_t
1024{
1025 return str_capacity(str, TERM_LEN(str));
1026}
1027
1028static inline void
1029must_not_null(const char *ptr)
1030{
1031 if (!ptr) {
1032 rb_raise(rb_eArgError, "NULL pointer given");
1033 }
1034}
1035
1036static inline VALUE
1037str_alloc_embed(VALUE klass, size_t capa)
1038{
1039 size_t size = rb_str_embed_size(capa, 0);
1040 RUBY_ASSERT(size > 0);
1041 RUBY_ASSERT(rb_gc_size_allocatable_p(size));
1042
1043 NEWOBJ_OF(str, struct RString, klass, T_STRING, size);
1044
1045 str->len = 0;
1046 str->as.embed.ary[0] = 0;
1047
1048 return (VALUE)str;
1049}
1050
1051static inline VALUE
1052str_alloc_heap(VALUE klass)
1053{
1054 NEWOBJ_OF(str, struct RString, klass, T_STRING | STR_NOEMBED, sizeof(struct RString));
1055
1056 str->len = 0;
1057 str->as.heap.aux.capa = 0;
1058 str->as.heap.ptr = NULL;
1059
1060 return (VALUE)str;
1061}
1062
1063static inline VALUE
1064empty_str_alloc(VALUE klass)
1065{
1066 RUBY_DTRACE_CREATE_HOOK(STRING, 0);
1067 VALUE str = str_alloc_embed(klass, 0);
1068 memset(RSTRING(str)->as.embed.ary, 0, str_embed_capa(str));
1070 return str;
1071}
1072
1073static VALUE
1074str_enc_new(VALUE klass, const char *ptr, long len, rb_encoding *enc)
1075{
1076 VALUE str;
1077
1078 if (len < 0) {
1079 rb_raise(rb_eArgError, "negative string size (or size too big)");
1080 }
1081
1082 if (enc == NULL) {
1083 enc = rb_ascii8bit_encoding();
1084 }
1085
1086 RUBY_DTRACE_CREATE_HOOK(STRING, len);
1087
1088 int termlen = rb_enc_mbminlen(enc);
1089
1090 if (STR_EMBEDDABLE_P(len, termlen)) {
1091 str = str_alloc_embed(klass, len + termlen);
1092 if (len == 0) {
1093 ENC_CODERANGE_SET(str, rb_enc_asciicompat(enc) ? ENC_CODERANGE_7BIT : ENC_CODERANGE_VALID);
1094 }
1095 }
1096 else {
1097 str = str_alloc_heap(klass);
1098 RSTRING(str)->as.heap.aux.capa = len;
1099 /* :FIXME: @shyouhei guesses `len + termlen` is guaranteed to never
1100 * integer overflow. If we can STATIC_ASSERT that, the following
1101 * mul_add_mul can be reverted to a simple ALLOC_N. */
1102 RSTRING(str)->as.heap.ptr =
1103 rb_xmalloc_mul_add_mul(sizeof(char), len, sizeof(char), termlen);
1104 }
1105
1106 rb_enc_raw_set(str, enc);
1107
1108 if (ptr) {
1109 memcpy(RSTRING_PTR(str), ptr, len);
1110 }
1111 else {
1112 memset(RSTRING_PTR(str), 0, len);
1113 }
1114
1115 STR_SET_LEN(str, len);
1116 TERM_FILL(RSTRING_PTR(str) + len, termlen);
1117 return str;
1118}
1119
1120static VALUE
1121str_new(VALUE klass, const char *ptr, long len)
1122{
1123 return str_enc_new(klass, ptr, len, rb_ascii8bit_encoding());
1124}
1125
1126VALUE
1127rb_str_new(const char *ptr, long len)
1128{
1129 return str_new(rb_cString, ptr, len);
1130}
1131
1132VALUE
1133rb_usascii_str_new(const char *ptr, long len)
1134{
1135 return str_enc_new(rb_cString, ptr, len, rb_usascii_encoding());
1136}
1137
1138VALUE
1139rb_utf8_str_new(const char *ptr, long len)
1140{
1141 return str_enc_new(rb_cString, ptr, len, rb_utf8_encoding());
1142}
1143
1144VALUE
1145rb_enc_str_new(const char *ptr, long len, rb_encoding *enc)
1146{
1147 return str_enc_new(rb_cString, ptr, len, enc);
1148}
1149
1150VALUE
1152{
1153 must_not_null(ptr);
1154 /* rb_str_new_cstr() can take pointer from non-malloc-generated
1155 * memory regions, and that cannot be detected by the MSAN. Just
1156 * trust the programmer that the argument passed here is a sane C
1157 * string. */
1158 __msan_unpoison_string(ptr);
1159 return rb_str_new(ptr, strlen(ptr));
1160}
1161
1162VALUE
1164{
1165 return rb_enc_str_new_cstr(ptr, rb_usascii_encoding());
1166}
1167
1168VALUE
1170{
1171 return rb_enc_str_new_cstr(ptr, rb_utf8_encoding());
1172}
1173
1174VALUE
1176{
1177 must_not_null(ptr);
1178 if (rb_enc_mbminlen(enc) != 1) {
1179 rb_raise(rb_eArgError, "wchar encoding given");
1180 }
1181 return rb_enc_str_new(ptr, strlen(ptr), enc);
1182}
1183
1184static VALUE
1185str_new_static(VALUE klass, const char *ptr, long len, int encindex)
1186{
1187 VALUE str;
1188
1189 if (len < 0) {
1190 rb_raise(rb_eArgError, "negative string size (or size too big)");
1191 }
1192
1193 if (!ptr) {
1194 str = str_enc_new(klass, ptr, len, rb_enc_from_index(encindex));
1195 }
1196 else {
1197 RUBY_DTRACE_CREATE_HOOK(STRING, len);
1198 str = str_alloc_heap(klass);
1199 RSTRING(str)->len = len;
1200 RSTRING(str)->as.heap.ptr = (char *)ptr;
1201 RSTRING(str)->as.heap.aux.capa = len;
1202 RBASIC(str)->flags |= STR_NOFREE;
1203 rb_enc_associate_index(str, encindex);
1204 }
1205 return str;
1206}
1207
1208VALUE
1209rb_str_new_static(const char *ptr, long len)
1210{
1211 return str_new_static(rb_cString, ptr, len, 0);
1212}
1213
1214/* Take an xmalloc'd buffer as the String's body without copying it; the String owns it
1215 * from here and frees it like any other heap string. ptr must hold capa bytes plus the
1216 * terminator for encindex, which is what a Ractor courier's string node carries. */
1217VALUE
1218rb_str_new_owned(char *ptr, long len, long capa, int encindex)
1219{
1220 RUBY_DTRACE_CREATE_HOOK(STRING, len);
1221 VALUE str = str_alloc_heap(rb_cString);
1222 RSTRING(str)->len = len;
1223 RSTRING(str)->as.heap.ptr = ptr;
1224 /* Freed by size (STR_HEAP_SIZE = capa + terminator), so capa must describe the
1225 * allocation the caller made, not just the bytes in use. */
1226 RSTRING(str)->as.heap.aux.capa = capa;
1227 rb_enc_associate_index(str, encindex);
1228 return str;
1229}
1230
1231VALUE
1233{
1234 return str_new_static(rb_cString, ptr, len, ENCINDEX_US_ASCII);
1235}
1236
1237VALUE
1239{
1240 return str_new_static(rb_cString, ptr, len, ENCINDEX_UTF_8);
1241}
1242
1243VALUE
1245{
1246 return str_new_static(rb_cString, ptr, len, rb_enc_to_index(enc));
1247}
1248
1249static VALUE str_cat_conv_enc_opts(VALUE newstr, long ofs, const char *ptr, long len,
1250 rb_encoding *from, rb_encoding *to,
1251 int ecflags, VALUE ecopts);
1252
1253static inline bool
1254is_enc_ascii_string(VALUE str, rb_encoding *enc)
1255{
1256 int encidx = rb_enc_to_index(enc);
1257 if (rb_enc_get_index(str) == encidx)
1258 return is_ascii_string(str);
1259 return enc_coderange_scan(str, enc) == ENC_CODERANGE_7BIT;
1260}
1261
1262VALUE
1263rb_str_conv_enc_opts(VALUE str, rb_encoding *from, rb_encoding *to, int ecflags, VALUE ecopts)
1264{
1265 long len;
1266 const char *ptr;
1267 VALUE newstr;
1268
1269 if (!to) return str;
1270 if (!from) from = rb_enc_get(str);
1271 if (from == to) return str;
1272 if ((rb_enc_asciicompat(to) && is_enc_ascii_string(str, from)) ||
1273 rb_is_ascii8bit_enc(to)) {
1274 if (STR_ENC_GET(str) != to) {
1275 str = rb_str_dup(str);
1276 rb_enc_associate(str, to);
1277 }
1278 return str;
1279 }
1280
1281 RSTRING_GETMEM(str, ptr, len);
1282 newstr = str_cat_conv_enc_opts(rb_str_buf_new(len), 0, ptr, len,
1283 from, to, ecflags, ecopts);
1284 if (NIL_P(newstr)) {
1285 /* some error, return original */
1286 return str;
1287 }
1288 return newstr;
1289}
1290
1291VALUE
1292rb_str_cat_conv_enc_opts(VALUE newstr, long ofs, const char *ptr, long len,
1293 rb_encoding *from, int ecflags, VALUE ecopts)
1294{
1295 long olen;
1296
1297 olen = RSTRING_LEN(newstr);
1298 if (ofs < -olen || olen < ofs)
1299 rb_raise(rb_eIndexError, "index %ld out of string", ofs);
1300 if (ofs < 0) ofs += olen;
1301 if (!from) {
1302 STR_SET_LEN(newstr, ofs);
1303 return rb_str_cat(newstr, ptr, len);
1304 }
1305
1306 rb_str_modify(newstr);
1307 return str_cat_conv_enc_opts(newstr, ofs, ptr, len, from,
1308 rb_enc_get(newstr),
1309 ecflags, ecopts);
1310}
1311
1312VALUE
1313rb_str_initialize(VALUE str, const char *ptr, long len, rb_encoding *enc)
1314{
1315 STR_SET_LEN(str, 0);
1316 rb_enc_associate(str, enc);
1317 rb_str_cat(str, ptr, len);
1318 return str;
1319}
1320
1321static VALUE
1322str_cat_conv_enc_opts(VALUE newstr, long ofs, const char *ptr, long len,
1323 rb_encoding *from, rb_encoding *to,
1324 int ecflags, VALUE ecopts)
1325{
1326 rb_econv_t *ec;
1328 long olen;
1329 VALUE econv_wrapper;
1330 const unsigned char *start, *sp;
1331 unsigned char *dest, *dp;
1332 size_t converted_output = (size_t)ofs;
1333
1334 olen = rb_str_capacity(newstr);
1335
1336 econv_wrapper = rb_obj_alloc(rb_cEncodingConverter);
1337 RBASIC_CLEAR_CLASS(econv_wrapper);
1338 ec = rb_econv_open_opts(from->name, to->name, ecflags, ecopts);
1339 if (!ec) return Qnil;
1340 DATA_PTR(econv_wrapper) = ec;
1341
1342 sp = (unsigned char*)ptr;
1343 start = sp;
1344 while ((dest = (unsigned char*)RSTRING_PTR(newstr)),
1345 (dp = dest + converted_output),
1346 (ret = rb_econv_convert(ec, &sp, start + len, &dp, dest + olen, 0)),
1348 /* destination buffer short */
1349 size_t converted_input = sp - start;
1350 size_t rest = len - converted_input;
1351 converted_output = dp - dest;
1352 rb_str_set_len(newstr, converted_output);
1353 if (converted_input && converted_output &&
1354 rest < (LONG_MAX / converted_output)) {
1355 rest = (rest * converted_output) / converted_input;
1356 }
1357 else {
1358 rest = olen;
1359 }
1360 olen += rest < 2 ? 2 : rest;
1361 rb_str_resize(newstr, olen);
1362 }
1363 DATA_PTR(econv_wrapper) = 0;
1364 RB_GC_GUARD(econv_wrapper);
1365 rb_econv_close(ec);
1366 switch (ret) {
1367 case econv_finished:
1368 len = dp - (unsigned char*)RSTRING_PTR(newstr);
1369 rb_str_set_len(newstr, len);
1370 rb_enc_associate(newstr, to);
1371 return newstr;
1372
1373 default:
1374 return Qnil;
1375 }
1376}
1377
1378VALUE
1380{
1381 return rb_str_conv_enc_opts(str, from, to, 0, Qnil);
1382}
1383
1384VALUE
1386{
1387 rb_encoding *ienc;
1388 VALUE str;
1389 const int eidx = rb_enc_to_index(eenc);
1390
1391 if (!ptr) {
1392 return rb_enc_str_new(ptr, len, eenc);
1393 }
1394
1395 /* ASCII-8BIT case, no conversion */
1396 if ((eidx == rb_ascii8bit_encindex()) ||
1397 (eidx == rb_usascii_encindex() && search_nonascii(ptr, ptr + len))) {
1398 return rb_str_new(ptr, len);
1399 }
1400 /* no default_internal or same encoding, no conversion */
1401 ienc = rb_default_internal_encoding();
1402 if (!ienc || eenc == ienc) {
1403 return rb_enc_str_new(ptr, len, eenc);
1404 }
1405 /* ASCII compatible, and ASCII only string, no conversion in
1406 * default_internal */
1407 if ((eidx == rb_ascii8bit_encindex()) ||
1408 (eidx == rb_usascii_encindex()) ||
1409 (rb_enc_asciicompat(eenc) && !search_nonascii(ptr, ptr + len))) {
1410 return rb_enc_str_new(ptr, len, ienc);
1411 }
1412 /* convert from the given encoding to default_internal */
1413 str = rb_enc_str_new(NULL, 0, ienc);
1414 /* when the conversion failed for some reason, just ignore the
1415 * default_internal and result in the given encoding as-is. */
1416 if (NIL_P(rb_str_cat_conv_enc_opts(str, 0, ptr, len, eenc, 0, Qnil))) {
1417 rb_str_initialize(str, ptr, len, eenc);
1418 }
1419 return str;
1420}
1421
1422VALUE
1423rb_external_str_with_enc(VALUE str, rb_encoding *eenc)
1424{
1425 int eidx = rb_enc_to_index(eenc);
1426 if (eidx == rb_usascii_encindex() &&
1427 !is_ascii_string(str)) {
1428 rb_enc_associate_index(str, rb_ascii8bit_encindex());
1429 return str;
1430 }
1431 rb_enc_associate_index(str, eidx);
1432 return rb_str_conv_enc(str, eenc, rb_default_internal_encoding());
1433}
1434
1435VALUE
1436rb_external_str_new(const char *ptr, long len)
1437{
1438 return rb_external_str_new_with_enc(ptr, len, rb_default_external_encoding());
1439}
1440
1441VALUE
1443{
1444 return rb_external_str_new_with_enc(ptr, strlen(ptr), rb_default_external_encoding());
1445}
1446
1447VALUE
1448rb_locale_str_new(const char *ptr, long len)
1449{
1450 return rb_external_str_new_with_enc(ptr, len, rb_locale_encoding());
1451}
1452
1453VALUE
1455{
1456 return rb_external_str_new_with_enc(ptr, strlen(ptr), rb_locale_encoding());
1457}
1458
1459VALUE
1461{
1462 return rb_external_str_new_with_enc(ptr, len, rb_filesystem_encoding());
1463}
1464
1465VALUE
1467{
1468 return rb_external_str_new_with_enc(ptr, strlen(ptr), rb_filesystem_encoding());
1469}
1470
1471VALUE
1473{
1474 return rb_str_export_to_enc(str, rb_default_external_encoding());
1475}
1476
1477VALUE
1479{
1480 return rb_str_export_to_enc(str, rb_locale_encoding());
1481}
1482
1483VALUE
1485{
1486 return rb_str_conv_enc(str, STR_ENC_GET(str), enc);
1487}
1488
1489static VALUE
1490str_replace_shared_without_enc(VALUE str2, VALUE str)
1491{
1492 const int termlen = TERM_LEN(str);
1493 char *ptr;
1494 long len;
1495
1496 RSTRING_GETMEM(str, ptr, len);
1497 if (str_embed_capa(str2) >= len + termlen) {
1498 char *ptr2 = RSTRING(str2)->as.embed.ary;
1499 STR_SET_EMBED(str2);
1500 memcpy(ptr2, RSTRING_PTR(str), len);
1501 TERM_FILL(ptr2+len, termlen);
1502 }
1503 else {
1504 VALUE root;
1505 if (STR_SHARED_P(str)) {
1506 root = RSTRING(str)->as.heap.aux.shared;
1507 RSTRING_GETMEM(str, ptr, len);
1508 }
1509 else {
1510 root = rb_str_new_frozen(str);
1511 RSTRING_GETMEM(root, ptr, len);
1512 }
1513 RUBY_ASSERT(OBJ_FROZEN(root));
1514
1515 if (!STR_EMBED_P(str2) && !FL_TEST_RAW(str2, STR_SHARED|STR_NOFREE)) {
1516 if (FL_TEST_RAW(str2, STR_SHARED_ROOT)) {
1517 rb_fatal("about to free a possible shared root");
1518 }
1519 char *ptr2 = STR_HEAP_PTR(str2);
1520 if (ptr2 != ptr) {
1521 SIZED_FREE_N(ptr2, STR_HEAP_SIZE(str2));
1522 }
1523 }
1524 FL_SET(str2, STR_NOEMBED);
1525 RSTRING(str2)->as.heap.ptr = ptr;
1526 STR_SET_SHARED(str2, root);
1527 }
1528
1529 STR_SET_LEN(str2, len);
1530
1531 return str2;
1532}
1533
1534static VALUE
1535str_replace_shared(VALUE str2, VALUE str)
1536{
1537 str_replace_shared_without_enc(str2, str);
1538 rb_enc_cr_str_exact_copy(str2, str);
1539 return str2;
1540}
1541
1542static VALUE
1543str_new_shared(VALUE klass, VALUE str)
1544{
1545 return str_replace_shared(str_alloc_heap(klass), str);
1546}
1547
1548VALUE
1550{
1551 return str_new_shared(rb_obj_class(str), str);
1552}
1553
1554VALUE
1556{
1557 if (RB_FL_TEST_RAW(orig, FL_FREEZE | STR_CHILLED) == FL_FREEZE) return orig;
1558 return str_new_frozen(rb_obj_class(orig), orig);
1559}
1560
1561static VALUE
1562rb_str_new_frozen_String(VALUE orig)
1563{
1564 if (OBJ_FROZEN(orig) && rb_obj_class(orig) == rb_cString) return orig;
1565 return str_new_frozen(rb_cString, orig);
1566}
1567
1568
1569VALUE
1570rb_str_frozen_bare_string(VALUE orig)
1571{
1572 if (RB_LIKELY(BARE_STRING_P(orig) && OBJ_FROZEN_RAW(orig))) return orig;
1573 return str_new_frozen(rb_cString, orig);
1574}
1575
1576VALUE
1577rb_str_tmp_frozen_acquire(VALUE orig)
1578{
1579 if (OBJ_FROZEN_RAW(orig)) return orig;
1580 return str_new_frozen_buffer(0, orig, FALSE);
1581}
1582
1583VALUE
1585{
1586 if (OBJ_FROZEN_RAW(orig) && !STR_EMBED_P(orig)) return orig;
1587 if (STR_SHARED_P(orig) && !STR_EMBED_P(RSTRING(orig)->as.heap.aux.shared)) return rb_str_tmp_frozen_acquire(orig);
1588
1589 VALUE str = str_alloc_heap(0);
1590 OBJ_FREEZE(str);
1591 /* Always set the STR_SHARED_ROOT to ensure it does not get re-embedded. */
1592 FL_SET(str, STR_SHARED_ROOT);
1593
1594 size_t capa = str_capacity(orig, TERM_LEN(orig));
1595
1596 /* If the string is embedded then we want to create a copy that is heap
1597 * allocated. If the string is shared then the shared root must be
1598 * embedded, so we want to create a copy. If the string is a shared root
1599 * then it must be embedded, so we want to create a copy. */
1600 if (STR_EMBED_P(orig) || FL_TEST_RAW(orig, STR_SHARED | STR_SHARED_ROOT | RSTRING_FSTR)) {
1601 RSTRING(str)->as.heap.ptr = rb_xmalloc_mul_add_mul(sizeof(char), capa, sizeof(char), TERM_LEN(orig));
1602 memcpy(RSTRING(str)->as.heap.ptr, RSTRING_PTR(orig), capa);
1603 }
1604 else {
1605 /* orig must be heap allocated and not shared, so we can safely transfer
1606 * the pointer to str. */
1607 RSTRING(str)->as.heap.ptr = RSTRING(orig)->as.heap.ptr;
1608 RBASIC(str)->flags |= RBASIC(orig)->flags & STR_NOFREE;
1609 RBASIC(orig)->flags &= ~STR_NOFREE;
1610 STR_SET_SHARED(orig, str);
1611 /* str was just allocated here, so orig is its only child and it is
1612 * safe for rb_str_tmp_frozen_release to give the buffer back. */
1613 FL_UNSET_RAW(str, STR_BORROWED);
1614 if (RB_OBJ_SHAREABLE_P(orig)) {
1616 RUBY_ASSERT((rb_gc_verify_shareable(str), 1));
1617 }
1618 }
1619
1620 RSTRING(str)->len = RSTRING(orig)->len;
1621 RSTRING(str)->as.heap.aux.capa = capa + (TERM_LEN(orig) - TERM_LEN(str));
1622
1623 return str;
1624}
1625
1626void
1628{
1629 rb_str_tmp_frozen_release(orig, tmp);
1630}
1631
1632void
1633rb_str_tmp_frozen_release(VALUE orig, VALUE tmp)
1634{
1635 if (RBASIC_CLASS(tmp) != 0)
1636 return;
1637
1638 if (STR_EMBED_P(tmp)) {
1640 }
1641 else if (FL_TEST_RAW(orig, STR_SHARED | STR_TMPLOCK) == STR_SHARED &&
1642 !OBJ_FROZEN_RAW(orig)) {
1643 VALUE shared = RSTRING(orig)->as.heap.aux.shared;
1644
1645 if (shared == tmp && !FL_TEST_RAW(tmp, STR_BORROWED)) {
1646 RUBY_ASSERT(RSTRING(orig)->as.heap.ptr == RSTRING(tmp)->as.heap.ptr);
1647 RUBY_ASSERT(RSTRING_LEN(orig) == RSTRING_LEN(tmp));
1648
1649 /* Unshare orig since the root (tmp) only has this one child. */
1650 FL_UNSET_RAW(orig, STR_SHARED);
1651 RSTRING(orig)->as.heap.aux.capa = RSTRING(tmp)->as.heap.aux.capa + TERM_LEN(tmp) - TERM_LEN(orig);
1652 RBASIC(orig)->flags |= RBASIC(tmp)->flags & STR_NOFREE;
1654
1655 /* Make tmp embedded and empty so it is safe for sweeping. */
1656 STR_SET_EMBED(tmp);
1657 STR_SET_LEN(tmp, 0);
1658 }
1659 }
1660}
1661
1662static VALUE
1663str_new_frozen(VALUE klass, VALUE orig)
1664{
1665 return str_new_frozen_buffer(klass, orig, TRUE);
1666}
1667
1668/* Transfers ownership of orig's buffer to a new shared root string.
1669 * termlen is the terminator length of the returned string, which may differ
1670 * from orig's terminator length when the caller does not copy the encoding.
1671 * The capacity is stored without the terminator, so it must be adjusted for
1672 * the difference to keep the buffer size (capa + termlen) unchanged. */
1673static VALUE
1674heap_str_make_shared(VALUE klass, VALUE orig, int termlen)
1675{
1676 RUBY_ASSERT(!STR_EMBED_P(orig));
1677 RUBY_ASSERT(!STR_SHARED_P(orig));
1679
1680 VALUE str = str_alloc_heap(klass);
1681 STR_SET_LEN(str, RSTRING_LEN(orig));
1682 RSTRING(str)->as.heap.ptr = RSTRING_PTR(orig);
1683 RSTRING(str)->as.heap.aux.capa = RSTRING(orig)->as.heap.aux.capa + TERM_LEN(orig) - termlen;
1684 RBASIC(str)->flags |= RBASIC(orig)->flags & STR_NOFREE;
1685 RBASIC(orig)->flags &= ~STR_NOFREE;
1686 STR_SET_SHARED(orig, str);
1687 if (klass == 0)
1688 FL_UNSET_RAW(str, STR_BORROWED);
1689 return str;
1690}
1691
1692static VALUE
1693str_new_frozen_buffer(VALUE klass, VALUE orig, int copy_encoding)
1694{
1695 VALUE str;
1696
1697 long len = RSTRING_LEN(orig);
1698 rb_encoding *enc = copy_encoding ? STR_ENC_GET(orig) : rb_ascii8bit_encoding();
1699 int termlen = copy_encoding ? TERM_LEN(orig) : 1;
1700
1701 if (STR_EMBED_P(orig) || STR_EMBEDDABLE_P(len, termlen)) {
1702 str = str_enc_new(klass, RSTRING_PTR(orig), len, enc);
1703 RUBY_ASSERT(STR_EMBED_P(str));
1704 }
1705 else {
1706 if (FL_TEST_RAW(orig, STR_SHARED)) {
1707 VALUE shared = RSTRING(orig)->as.heap.aux.shared;
1708 long ofs = RSTRING(orig)->as.heap.ptr - RSTRING_PTR(shared);
1709 long rest = RSTRING_LEN(shared) - ofs - RSTRING_LEN(orig);
1710 RUBY_ASSERT(ofs >= 0);
1711 RUBY_ASSERT(rest >= 0);
1712 RUBY_ASSERT(ofs + rest <= RSTRING_LEN(shared));
1714
1715 if ((ofs > 0) || (rest > 0) ||
1716 (klass != RBASIC(shared)->klass) ||
1717 ENCODING_GET(shared) != ENCODING_GET(orig)) {
1718 str = str_new_shared(klass, shared);
1719 RUBY_ASSERT(!STR_EMBED_P(str));
1720 RSTRING(str)->as.heap.ptr += ofs;
1721 STR_SET_LEN(str, RSTRING_LEN(str) - (ofs + rest));
1722 }
1723 else {
1724 if (RBASIC_CLASS(shared) == 0)
1725 FL_SET_RAW(shared, STR_BORROWED);
1726 return shared;
1727 }
1728 }
1729 else if (STR_EMBEDDABLE_P(RSTRING_LEN(orig), TERM_LEN(orig))) {
1730 str = str_alloc_embed(klass, RSTRING_LEN(orig) + TERM_LEN(orig));
1731 STR_SET_EMBED(str);
1732 memcpy(RSTRING_PTR(str), RSTRING_PTR(orig), RSTRING_LEN(orig));
1733 STR_SET_LEN(str, RSTRING_LEN(orig));
1734 ENC_CODERANGE_SET(str, ENC_CODERANGE(orig));
1735 TERM_FILL(RSTRING_END(str), TERM_LEN(orig));
1736 }
1737 else {
1738 if (RB_OBJ_SHAREABLE_P(orig)) {
1739 str = str_new(klass, RSTRING_PTR(orig), RSTRING_LEN(orig));
1740 }
1741 else {
1742 str = heap_str_make_shared(klass, orig, termlen);
1743 }
1744 }
1745 }
1746
1747 if (copy_encoding) rb_enc_cr_str_exact_copy(str, orig);
1748 OBJ_FREEZE(str);
1749 return str;
1750}
1751
1752VALUE
1753rb_str_new_with_class(VALUE obj, const char *ptr, long len)
1754{
1755 return str_enc_new(rb_obj_class(obj), ptr, len, STR_ENC_GET(obj));
1756}
1757
1758static VALUE
1759str_new_empty_String(VALUE str)
1760{
1761 VALUE v = rb_str_new(0, 0);
1762 rb_enc_copy(v, str);
1763 return v;
1764}
1765
1766#define STR_BUF_MIN_SIZE 63
1767
1768VALUE
1770{
1771 if (STR_EMBEDDABLE_P(capa, 1)) {
1772 return str_alloc_embed(rb_cString, capa + 1);
1773 }
1774
1775 VALUE str = str_alloc_heap(rb_cString);
1776
1777 RSTRING(str)->as.heap.aux.capa = capa;
1778 RSTRING(str)->as.heap.ptr = ALLOC_N(char, (size_t)capa + 1);
1779 RSTRING(str)->as.heap.ptr[0] = '\0';
1780
1781 return str;
1782}
1783
1784VALUE
1786{
1787 VALUE str;
1788 long len = strlen(ptr);
1789
1790 str = rb_str_buf_new(len);
1791 rb_str_buf_cat(str, ptr, len);
1792
1793 return str;
1794}
1795
1796VALUE
1798{
1799 return str_new(0, 0, len);
1800}
1801
1802void
1804{
1805 if (STR_EMBED_P(str)) {
1806 RB_DEBUG_COUNTER_INC(obj_str_embed);
1807 }
1808 else if (FL_TEST(str, STR_SHARED | STR_NOFREE)) {
1809 (void)RB_DEBUG_COUNTER_INC_IF(obj_str_shared, FL_TEST(str, STR_SHARED));
1810 (void)RB_DEBUG_COUNTER_INC_IF(obj_str_shared, FL_TEST(str, STR_NOFREE));
1811 }
1812 else {
1813 RB_DEBUG_COUNTER_INC(obj_str_ptr);
1814 SIZED_FREE_N(STR_HEAP_PTR(str), STR_HEAP_SIZE(str));
1815 }
1816}
1817
1818size_t
1819rb_str_memsize(VALUE str)
1820{
1821 if (FL_TEST(str, STR_NOEMBED|STR_SHARED|STR_NOFREE) == STR_NOEMBED) {
1822 return STR_HEAP_SIZE(str);
1823 }
1824 else {
1825 return 0;
1826 }
1827}
1828
1829VALUE
1831{
1832 return rb_convert_type_with_id(str, T_STRING, "String", idTo_str);
1833}
1834
1835static inline void str_discard(VALUE str);
1836static void str_shared_replace(VALUE str, VALUE str2);
1837
1838void
1840{
1841 if (str != str2) str_shared_replace(str, str2);
1842}
1843
1844static void
1845str_shared_replace(VALUE str, VALUE str2)
1846{
1847 rb_encoding *enc;
1848 int cr;
1849 int termlen;
1850
1851 RUBY_ASSERT(str2 != str);
1852 enc = STR_ENC_GET(str2);
1853 cr = ENC_CODERANGE(str2);
1854 str_discard(str);
1855 termlen = rb_enc_mbminlen(enc);
1856
1857 STR_SET_LEN(str, RSTRING_LEN(str2));
1858
1859 if (str_embed_capa(str) >= RSTRING_LEN(str2) + termlen) {
1860 STR_SET_EMBED(str);
1861 memcpy(RSTRING_PTR(str), RSTRING_PTR(str2), (size_t)RSTRING_LEN(str2) + termlen);
1862 }
1863 else {
1864 if (STR_EMBED_P(str2)) {
1865 RUBY_ASSERT(!FL_TEST(str2, STR_SHARED));
1866 long len = RSTRING_LEN(str2);
1867 RUBY_ASSERT(len + termlen <= str_embed_capa(str2));
1868
1869 char *new_ptr = ALLOC_N(char, len + termlen);
1870 memcpy(new_ptr, RSTRING(str2)->as.embed.ary, len + termlen);
1871 RSTRING(str2)->as.heap.ptr = new_ptr;
1872 STR_SET_LEN(str2, len);
1873 RSTRING(str2)->as.heap.aux.capa = len;
1874 STR_SET_NOEMBED(str2);
1875 }
1876
1877 STR_SET_NOEMBED(str);
1878 FL_UNSET(str, STR_SHARED);
1879 RSTRING(str)->as.heap.ptr = RSTRING_PTR(str2);
1880
1881 if (FL_TEST(str2, STR_SHARED)) {
1882 VALUE shared = RSTRING(str2)->as.heap.aux.shared;
1883 STR_SET_SHARED(str, shared);
1884 }
1885 else {
1886 RSTRING(str)->as.heap.aux.capa = RSTRING(str2)->as.heap.aux.capa;
1887 }
1888
1889 /* abandon str2 */
1890 STR_SET_EMBED(str2);
1891 RSTRING_PTR(str2)[0] = 0;
1892 STR_SET_LEN(str2, 0);
1893 }
1894
1895 // We used str2's termlen above so we set enc raw
1896 // to avoid adjusting it based on str1's old enc.
1897 rb_enc_raw_set(str, enc);
1898 ENC_CODERANGE_SET(str, cr);
1899}
1900
1901VALUE
1903{
1904 VALUE str;
1905
1906 if (RB_TYPE_P(obj, T_STRING)) {
1907 return obj;
1908 }
1909 str = rb_funcall(obj, idTo_s, 0);
1910 return rb_obj_as_string_result(str, obj);
1911}
1912
1913VALUE
1914rb_obj_as_string_result(VALUE str, VALUE obj)
1915{
1916 if (!RB_TYPE_P(str, T_STRING))
1917 return rb_any_to_s(obj);
1918 return str;
1919}
1920
1921static VALUE
1922str_replace(VALUE str, VALUE str2)
1923{
1924 long len;
1925
1926 len = RSTRING_LEN(str2);
1927 if (STR_SHARED_P(str2)) {
1928 VALUE shared = RSTRING(str2)->as.heap.aux.shared;
1930 STR_SET_NOEMBED(str);
1931 STR_SET_LEN(str, len);
1932 RSTRING(str)->as.heap.ptr = RSTRING_PTR(str2);
1933 STR_SET_SHARED(str, shared);
1934 rb_enc_cr_str_exact_copy(str, str2);
1935 }
1936 else {
1937 str_replace_shared(str, str2);
1938 }
1939
1940 return str;
1941}
1942
1943static inline VALUE
1944ec_str_alloc_embed(struct rb_execution_context_struct *ec, VALUE klass, size_t capa)
1945{
1946 size_t size = rb_str_embed_size(capa, 0);
1947 RUBY_ASSERT(size > 0);
1948 RUBY_ASSERT(rb_gc_size_allocatable_p(size));
1949
1950 EC_NEWOBJ_OF(str, struct RString, klass, T_STRING, size, ec);
1951
1952 str->len = 0;
1953
1954 return (VALUE)str;
1955}
1956
1957static inline VALUE
1958ec_str_alloc_heap(struct rb_execution_context_struct *ec, VALUE klass)
1959{
1960 EC_NEWOBJ_OF(str, struct RString, klass, T_STRING | STR_NOEMBED, sizeof(struct RString), ec);
1961
1962 str->as.heap.aux.capa = 0;
1963 str->as.heap.ptr = NULL;
1964
1965 return (VALUE)str;
1966}
1967
1968static inline void
1969str_duplicate_setup_encoding(VALUE str, VALUE dup, VALUE flags)
1970{
1971 int encidx = 0;
1972 if ((flags & ENCODING_MASK) == (ENCODING_INLINE_MAX<<ENCODING_SHIFT)) {
1973 encidx = rb_enc_get_index(str);
1974 flags &= ~ENCODING_MASK;
1975 }
1976 FL_SET_RAW(dup, flags & ~FL_FREEZE);
1977 if (encidx) rb_enc_associate_index(dup, encidx);
1978}
1979
1980static const VALUE flag_mask = ENC_CODERANGE_MASK | ENCODING_MASK | FL_FREEZE;
1981
1982static inline void
1983str_duplicate_setup_embed(VALUE klass, VALUE str, VALUE dup)
1984{
1985 VALUE flags = FL_TEST_RAW(str, flag_mask);
1986 long len = RSTRING_LEN(str);
1987
1988 RUBY_ASSERT(STR_EMBED_P(dup));
1989 RUBY_ASSERT(str_embed_capa(dup) >= len + TERM_LEN(str));
1990 MEMCPY(RSTRING(dup)->as.embed.ary, RSTRING(str)->as.embed.ary, char, len + TERM_LEN(str));
1991 STR_SET_LEN(dup, RSTRING_LEN(str));
1992 str_duplicate_setup_encoding(str, dup, flags);
1993}
1994
1995static inline void
1996str_duplicate_setup_heap(VALUE klass, VALUE str, VALUE dup)
1997{
1998 VALUE flags = FL_TEST_RAW(str, flag_mask);
1999 VALUE root = str;
2000 if (FL_TEST_RAW(str, STR_SHARED)) {
2001 root = RSTRING(str)->as.heap.aux.shared;
2002 }
2003 else if (UNLIKELY(!OBJ_FROZEN_RAW(str))) {
2004 root = str = str_new_frozen(klass, str);
2005 flags = FL_TEST_RAW(str, flag_mask);
2006 }
2007 RUBY_ASSERT(!STR_SHARED_P(root));
2009
2010 RSTRING(dup)->as.heap.ptr = RSTRING_PTR(str);
2011 FL_SET_RAW(dup, RSTRING_NOEMBED);
2012 STR_SET_SHARED(dup, root);
2013 flags |= RSTRING_NOEMBED | STR_SHARED;
2014
2015 STR_SET_LEN(dup, RSTRING_LEN(str));
2016 str_duplicate_setup_encoding(str, dup, flags);
2017}
2018
2019static inline VALUE
2020str_duplicate(VALUE klass, VALUE str)
2021{
2022 VALUE dup;
2023 if (STR_EMBED_P(str) && rb_str_embed_size(RSTRING_LEN(str), 1) <= STR_COPY_MAX_EMBED_SIZE) {
2024 dup = str_alloc_embed(klass, RSTRING_LEN(str) + TERM_LEN(str));
2025
2026 str_duplicate_setup_embed(klass, str, dup);
2027 }
2028 else {
2029 dup = str_alloc_heap(klass);
2030
2031 str_duplicate_setup_heap(klass, str, dup);
2032 }
2033
2034 return dup;
2035}
2036
2037VALUE
2039{
2040 return str_duplicate(rb_obj_class(str), str);
2041}
2042
2043/* :nodoc: */
2044VALUE
2045rb_str_dup_m(VALUE str)
2046{
2047 if (LIKELY(BARE_STRING_P(str))) {
2048 return str_duplicate(rb_cString, str);
2049 }
2050 else {
2051 return rb_obj_dup(str);
2052 }
2053}
2054
2055VALUE
2057{
2058 RUBY_DTRACE_CREATE_HOOK(STRING, RSTRING_LEN(str));
2059 return str_duplicate(rb_cString, str);
2060}
2061
2062VALUE
2063rb_ec_str_resurrect(struct rb_execution_context_struct *ec, VALUE str, bool chilled)
2064{
2065 RUBY_DTRACE_CREATE_HOOK(STRING, RSTRING_LEN(str));
2066 VALUE new_str, klass = rb_cString;
2067
2068 if (!(chilled && RTEST(rb_ivar_defined(str, id_debug_created_info))) && STR_EMBED_P(str)) {
2069 new_str = ec_str_alloc_embed(ec, klass, RSTRING_LEN(str) + TERM_LEN(str));
2070 str_duplicate_setup_embed(klass, str, new_str);
2071 }
2072 else {
2073 new_str = ec_str_alloc_heap(ec, klass);
2074 str_duplicate_setup_heap(klass, str, new_str);
2075 }
2076 if (chilled) {
2077 FL_SET_RAW(new_str, STR_CHILLED);
2078 }
2079 return new_str;
2080}
2081
2082#if USE_ZJIT
2083bool
2084rb_zjit_str_resurrect_fastpath(VALUE str, bool chilled, size_t *size_out,
2085 VALUE *flags_out,
2086 long *len_out, size_t *byte_size_out)
2087{
2088 if (chilled && RTEST(rb_ivar_defined(str, id_debug_created_info))) return false;
2089
2090 if (!STR_EMBED_P(str)) return false;
2091
2092 long len = RSTRING_LEN(str);
2093 long termlen = TERM_LEN(str);
2094 size_t size = rb_str_embed_size(len + termlen, 0);
2095 if (!rb_gc_size_allocatable_p(size)) return false;
2096
2097 VALUE flags = FL_TEST_RAW(str, flag_mask);
2098
2099 if ((flags & ENCODING_MASK) == ((VALUE)ENCODING_INLINE_MAX << ENCODING_SHIFT)) {
2100 return false;
2101 }
2102
2103 flags &= ~FL_FREEZE;
2104 flags |= T_STRING;
2105 if (chilled) flags |= STR_CHILLED;
2106
2107 *size_out = size;
2108 *flags_out = flags;
2109 *len_out = len;
2110 *byte_size_out = (size_t)(len + termlen);
2111 return true;
2112}
2113#endif
2114
2115VALUE
2116rb_str_with_debug_created_info(VALUE str, VALUE path, int line)
2117{
2118 VALUE debug_info = rb_ary_new_from_args(2, path, INT2FIX(line));
2119 if (OBJ_FROZEN_RAW(str)) str = rb_str_dup(str);
2120 rb_ivar_set(str, id_debug_created_info, rb_ary_freeze(debug_info));
2121 FL_SET_RAW(str, STR_CHILLED);
2122 return rb_str_freeze(str);
2123}
2124
2125/*
2126 * The documentation block below uses an include (instead of inline text)
2127 * because the included text has non-ASCII characters (which are not allowed in a C file).
2128 */
2129
2130/*
2131 *
2132 * call-seq:
2133 * String.new(string = ''.encode(Encoding::ASCII_8BIT) , **options) -> new_string
2134 *
2135 * :include: doc/string/new.rdoc
2136 *
2137 */
2138
2139static VALUE
2140rb_str_init(int argc, VALUE *argv, VALUE str)
2141{
2142 static ID keyword_ids[2];
2143 VALUE orig, opt, venc, vcapa;
2144 VALUE kwargs[2];
2145 rb_encoding *enc = 0;
2146 int n;
2147
2148 if (!keyword_ids[0]) {
2149 keyword_ids[0] = rb_id_encoding();
2150 CONST_ID(keyword_ids[1], "capacity");
2151 }
2152
2153 n = rb_scan_args(argc, argv, "01:", &orig, &opt);
2154 if (!NIL_P(opt)) {
2155 rb_get_kwargs(opt, keyword_ids, 0, 2, kwargs);
2156 venc = kwargs[0];
2157 vcapa = kwargs[1];
2158 if (!UNDEF_P(venc) && !NIL_P(venc)) {
2159 enc = rb_to_encoding(venc);
2160 }
2161 if (!UNDEF_P(vcapa) && !NIL_P(vcapa)) {
2162 long capa = NUM2LONG(vcapa);
2163 long len = 0;
2164 int termlen = enc ? rb_enc_mbminlen(enc) : 1;
2165
2166 if (capa < STR_BUF_MIN_SIZE) {
2167 capa = STR_BUF_MIN_SIZE;
2168 }
2169 if (n == 1) {
2170 StringValue(orig);
2171 len = RSTRING_LEN(orig);
2172 if (capa < len) {
2173 capa = len;
2174 }
2175 if (orig == str) n = 0;
2176 }
2177 str_modifiable(str);
2178 if (STR_EMBED_P(str) || FL_TEST(str, STR_SHARED|STR_NOFREE)) {
2179 /* make noembed always */
2180 const size_t size = (size_t)capa + termlen;
2181 const char *const old_ptr = RSTRING_PTR(str);
2182 const size_t osize = RSTRING_LEN(str) + TERM_LEN(str);
2183 char *new_ptr = ALLOC_N(char, size);
2184 if (STR_EMBED_P(str)) RUBY_ASSERT((long)osize <= str_embed_capa(str));
2185 memcpy(new_ptr, old_ptr, osize < size ? osize : size);
2186 FL_UNSET_RAW(str, STR_SHARED|STR_NOFREE);
2187 RSTRING(str)->as.heap.ptr = new_ptr;
2188 }
2189 else if (STR_HEAP_SIZE(str) != (size_t)capa + termlen) {
2190 SIZED_REALLOC_N(RSTRING(str)->as.heap.ptr, char,
2191 (size_t)capa + termlen, STR_HEAP_SIZE(str));
2192 }
2193 STR_SET_LEN(str, len);
2194 TERM_FILL(&RSTRING(str)->as.heap.ptr[len], termlen);
2195 if (n == 1) {
2196 memcpy(RSTRING(str)->as.heap.ptr, RSTRING_PTR(orig), len);
2197 rb_enc_cr_str_exact_copy(str, orig);
2198 }
2199 FL_SET(str, STR_NOEMBED);
2200 RSTRING(str)->as.heap.aux.capa = capa;
2201 }
2202 else if (n == 1) {
2203 rb_str_replace(str, orig);
2204 }
2205 if (enc) {
2206 rb_enc_associate(str, enc);
2208 }
2209 }
2210 else if (n == 1) {
2211 rb_str_replace(str, orig);
2212 }
2213 return str;
2214}
2215
2216/* :nodoc: */
2217static VALUE
2218rb_str_s_new(int argc, VALUE *argv, VALUE klass)
2219{
2220 if (klass != rb_cString) {
2221 return rb_class_new_instance_pass_kw(argc, argv, klass);
2222 }
2223
2224 static ID keyword_ids[2];
2225 VALUE orig, opt, encoding = Qnil, capacity = Qnil;
2226 VALUE kwargs[2];
2227 rb_encoding *enc = NULL;
2228
2229 int n = rb_scan_args(argc, argv, "01:", &orig, &opt);
2230 if (NIL_P(opt)) {
2231 return rb_class_new_instance_pass_kw(argc, argv, klass);
2232 }
2233
2234 keyword_ids[0] = rb_id_encoding();
2235 CONST_ID(keyword_ids[1], "capacity");
2236 rb_get_kwargs(opt, keyword_ids, 0, 2, kwargs);
2237 encoding = kwargs[0];
2238 capacity = kwargs[1];
2239
2240 if (n == 1) {
2241 orig = StringValue(orig);
2242 }
2243 else {
2244 orig = Qnil;
2245 }
2246
2247 if (UNDEF_P(encoding)) {
2248 if (!NIL_P(orig)) {
2249 encoding = rb_obj_encoding(orig);
2250 }
2251 }
2252
2253 if (!UNDEF_P(encoding)) {
2254 enc = rb_to_encoding(encoding);
2255 }
2256
2257 // If capacity is nil, we're basically just duping `orig`.
2258 if (UNDEF_P(capacity)) {
2259 if (NIL_P(orig)) {
2260 VALUE empty_str = str_new(klass, "", 0);
2261 if (enc) {
2262 rb_enc_associate(empty_str, enc);
2263 }
2264 return empty_str;
2265 }
2266 VALUE copy = str_duplicate(klass, orig);
2267 rb_enc_associate(copy, enc);
2268 ENC_CODERANGE_CLEAR(copy);
2269 return copy;
2270 }
2271
2272 long capa = 0;
2273 capa = NUM2LONG(capacity);
2274 if (capa < 0) {
2275 capa = 0;
2276 }
2277
2278 if (!NIL_P(orig)) {
2279 long orig_capa = rb_str_capacity(orig);
2280 if (orig_capa > capa) {
2281 capa = orig_capa;
2282 }
2283 }
2284
2285 VALUE str = str_enc_new(klass, NULL, capa, enc);
2286 STR_SET_LEN(str, 0);
2287 TERM_FILL(RSTRING_PTR(str), enc ? rb_enc_mbmaxlen(enc) : 1);
2288
2289 if (!NIL_P(orig)) {
2290 rb_str_buf_append(str, orig);
2291 }
2292
2293 return str;
2294}
2295
2296#ifdef NONASCII_MASK
2297#define is_utf8_lead_byte(c) (((c)&0xC0) != 0x80)
2298
2299/*
2300 * UTF-8 leading bytes have either 0xxxxxxx or 11xxxxxx
2301 * bit representation. (see https://en.wikipedia.org/wiki/UTF-8)
2302 * Therefore, the following pseudocode can detect UTF-8 leading bytes.
2303 *
2304 * if (!(byte & 0x80))
2305 * byte |= 0x40; // turn on bit6
2306 * return ((byte>>6) & 1); // bit6 represent whether this byte is leading or not.
2307 *
2308 * This function calculates whether a byte is leading or not for all bytes
2309 * in the argument word by concurrently using the above logic, and then
2310 * adds up the number of leading bytes in the word.
2311 */
2312static inline uintptr_t
2313count_utf8_lead_bytes_with_word(const uintptr_t *s)
2314{
2315 uintptr_t d = *s;
2316
2317 /* Transform so that bit0 indicates whether we have a UTF-8 leading byte or not. */
2318 d = (d>>6) | (~d>>7);
2319 d &= NONASCII_MASK >> 7;
2320
2321 /* Gather all bytes. */
2322#if defined(HAVE_BUILTIN___BUILTIN_POPCOUNT) && defined(__POPCNT__)
2323 /* use only if it can use POPCNT */
2324 return rb_popcount_intptr(d);
2325#else
2326 d += (d>>8);
2327 d += (d>>16);
2328# if SIZEOF_VOIDP == 8
2329 d += (d>>32);
2330# endif
2331 return (d&0xF);
2332#endif
2333}
2334#endif
2335
2336static inline long
2337enc_strlen(const char *p, const char *e, rb_encoding *enc, int cr)
2338{
2339 long c;
2340 const char *q;
2341
2342 if (rb_enc_mbmaxlen(enc) == rb_enc_mbminlen(enc)) {
2343 long diff = (long)(e - p);
2344 return diff / rb_enc_mbminlen(enc) + !!(diff % rb_enc_mbminlen(enc));
2345 }
2346#ifdef NONASCII_MASK
2347 else if (cr == ENC_CODERANGE_VALID && enc == rb_utf8_encoding()) {
2348 uintptr_t len = 0;
2349 if ((int)sizeof(uintptr_t) * 2 < e - p) {
2350 const uintptr_t *s, *t;
2351 const uintptr_t lowbits = sizeof(uintptr_t) - 1;
2352 s = (const uintptr_t*)(~lowbits & ((uintptr_t)p + lowbits));
2353 t = (const uintptr_t*)(~lowbits & (uintptr_t)e);
2354 while (p < (const char *)s) {
2355 if (is_utf8_lead_byte(*p)) len++;
2356 p++;
2357 }
2358 while (s < t) {
2359 len += count_utf8_lead_bytes_with_word(s);
2360 s++;
2361 }
2362 p = (const char *)s;
2363 }
2364 while (p < e) {
2365 if (is_utf8_lead_byte(*p)) len++;
2366 p++;
2367 }
2368 return (long)len;
2369 }
2370#endif
2371 else if (rb_enc_asciicompat(enc)) {
2372 c = 0;
2373 if (ENC_CODERANGE_CLEAN_P(cr)) {
2374 while (p < e) {
2375 q = search_nonascii(p, e);
2376 if (!q)
2377 return c + (e - p);
2378 c += q - p;
2379 p = q;
2380 p += rb_enc_fast_mbclen(p, e, enc);
2381 c++;
2382 }
2383 }
2384 else {
2385 while (p < e) {
2386 q = search_nonascii(p, e);
2387 if (!q)
2388 return c + (e - p);
2389 c += q - p;
2390 p = q;
2391 p += rb_enc_mbclen(p, e, enc);
2392 c++;
2393 }
2394 }
2395 return c;
2396 }
2397
2398 for (c=0; p<e; c++) {
2399 p += rb_enc_mbclen(p, e, enc);
2400 }
2401 return c;
2402}
2403
2404long
2405rb_enc_strlen(const char *p, const char *e, rb_encoding *enc)
2406{
2407 return enc_strlen(p, e, enc, ENC_CODERANGE_UNKNOWN);
2408}
2409
2410/* To get strlen with cr
2411 * Note that given cr is not used.
2412 */
2413long
2414rb_enc_strlen_cr(const char *p, const char *e, rb_encoding *enc, int *cr)
2415{
2416 long c;
2417 const char *q;
2418 int ret;
2419
2420 *cr = 0;
2421 if (rb_enc_mbmaxlen(enc) == rb_enc_mbminlen(enc)) {
2422 long diff = (long)(e - p);
2423 return diff / rb_enc_mbminlen(enc) + !!(diff % rb_enc_mbminlen(enc));
2424 }
2425 else if (rb_enc_asciicompat(enc)) {
2426 c = 0;
2427 while (p < e) {
2428 q = search_nonascii(p, e);
2429 if (!q) {
2430 if (!*cr) *cr = ENC_CODERANGE_7BIT;
2431 return c + (e - p);
2432 }
2433 c += q - p;
2434 p = q;
2435 ret = rb_enc_precise_mbclen(p, e, enc);
2436 if (MBCLEN_CHARFOUND_P(ret)) {
2437 *cr |= ENC_CODERANGE_VALID;
2438 p += MBCLEN_CHARFOUND_LEN(ret);
2439 }
2440 else {
2442 p++;
2443 }
2444 c++;
2445 }
2446 if (!*cr) *cr = ENC_CODERANGE_7BIT;
2447 return c;
2448 }
2449
2450 for (c=0; p<e; c++) {
2451 ret = rb_enc_precise_mbclen(p, e, enc);
2452 if (MBCLEN_CHARFOUND_P(ret)) {
2453 *cr |= ENC_CODERANGE_VALID;
2454 p += MBCLEN_CHARFOUND_LEN(ret);
2455 }
2456 else {
2458 if (p + rb_enc_mbminlen(enc) <= e)
2459 p += rb_enc_mbminlen(enc);
2460 else
2461 p = e;
2462 }
2463 }
2464 if (!*cr) *cr = ENC_CODERANGE_7BIT;
2465 return c;
2466}
2467
2468/* enc must be str's enc or rb_enc_check(str, str2) */
2469static long
2470str_strlen(VALUE str, rb_encoding *enc)
2471{
2472 const char *p, *e;
2473 int cr;
2474
2475 if (single_byte_optimizable(str)) return RSTRING_LEN(str);
2476 if (!enc) enc = STR_ENC_GET(str);
2477 p = RSTRING_PTR(str);
2478 e = RSTRING_END(str);
2479 cr = ENC_CODERANGE(str);
2480
2481 if (cr == ENC_CODERANGE_UNKNOWN) {
2482 long n = rb_enc_strlen_cr(p, e, enc, &cr);
2483 if (cr) ENC_CODERANGE_SET(str, cr);
2484 return n;
2485 }
2486 else {
2487 return enc_strlen(p, e, enc, cr);
2488 }
2489}
2490
2491long
2493{
2494 return str_strlen(str, NULL);
2495}
2496
2497/*
2498 * call-seq:
2499 * length -> integer
2500 *
2501 * :include: doc/string/length.rdoc
2502 *
2503 */
2504
2505VALUE
2507{
2508 return LONG2NUM(str_strlen(str, NULL));
2509}
2510
2511/*
2512 * call-seq:
2513 * bytesize -> integer
2514 *
2515 * :include: doc/string/bytesize.rdoc
2516 *
2517 */
2518
2519VALUE
2520rb_str_bytesize(VALUE str)
2521{
2522 return LONG2NUM(RSTRING_LEN(str));
2523}
2524
2525/*
2526 * call-seq:
2527 * empty? -> true or false
2528 *
2529 * Returns whether the length of +self+ is zero:
2530 *
2531 * 'hello'.empty? # => false
2532 * ' '.empty? # => false
2533 * ''.empty? # => true
2534 *
2535 * Related: see {Querying}[rdoc-ref:String@Querying].
2536 */
2537
2538static VALUE
2539rb_str_empty(VALUE str)
2540{
2541 return RBOOL(RSTRING_LEN(str) == 0);
2542}
2543
2544/*
2545 * call-seq:
2546 * self + other_string -> new_string
2547 *
2548 * Returns a new string containing +other_string+ concatenated to +self+:
2549 *
2550 * 'Hello from ' + self.to_s # => "Hello from main"
2551 *
2552 * Related: see {Converting to New String}[rdoc-ref:String@Converting+to+New+String].
2553 */
2554
2555VALUE
2557{
2558 VALUE str3;
2559 rb_encoding *enc;
2560 const char *ptr1, *ptr2;
2561 char *ptr3;
2562 long len1, len2;
2563 int termlen;
2564
2565 StringValue(str2);
2566 enc = rb_enc_check_str(str1, str2);
2567 RSTRING_GETMEM(str1, ptr1, len1);
2568 RSTRING_GETMEM(str2, ptr2, len2);
2569 termlen = rb_enc_mbminlen(enc);
2570 if (len1 > LONG_MAX - len2) {
2571 rb_raise(rb_eArgError, "string size too big");
2572 }
2573 str3 = str_enc_new(rb_cString, 0, len1+len2, enc);
2574 ptr3 = RSTRING_PTR(str3);
2575 memcpy(ptr3, ptr1, len1);
2576 memcpy(ptr3+len1, ptr2, len2);
2577 TERM_FILL(&ptr3[len1+len2], termlen);
2578
2579 ENCODING_CODERANGE_SET(str3, rb_enc_to_index(enc),
2581 RB_GC_GUARD(str1);
2582 RB_GC_GUARD(str2);
2583 return str3;
2584}
2585
2586/* A variant of rb_str_plus that does not raise but return Qundef instead. */
2587VALUE
2588rb_str_opt_plus(VALUE str1, VALUE str2)
2589{
2592 long len1, len2;
2593 MAYBE_UNUSED(char) *ptr1, *ptr2;
2594 RSTRING_GETMEM(str1, ptr1, len1);
2595 RSTRING_GETMEM(str2, ptr2, len2);
2596 int enc1 = rb_enc_get_index(str1);
2597 int enc2 = rb_enc_get_index(str2);
2598
2599 if (enc1 < 0) {
2600 return Qundef;
2601 }
2602 else if (enc2 < 0) {
2603 return Qundef;
2604 }
2605 else if (enc1 != enc2) {
2606 return Qundef;
2607 }
2608 else if (len1 > LONG_MAX - len2) {
2609 return Qundef;
2610 }
2611 else {
2612 return rb_str_plus(str1, str2);
2613 }
2614
2615}
2616
2617/*
2618 * call-seq:
2619 * self * n -> new_string
2620 *
2621 * Returns a new string containing +n+ copies of +self+:
2622 *
2623 * 'Ho!' * 3 # => "Ho!Ho!Ho!"
2624 * 'No!' * 0 # => ""
2625 *
2626 * Related: see {Converting to New String}[rdoc-ref:String@Converting+to+New+String].
2627 */
2628
2629VALUE
2631{
2632 VALUE str2;
2633 long n, len;
2634 char *ptr2;
2635 int termlen;
2636
2637 if (times == INT2FIX(1)) {
2638 return str_duplicate(rb_cString, str);
2639 }
2640 if (times == INT2FIX(0)) {
2641 str2 = str_alloc_embed(rb_cString, 0);
2642 rb_enc_copy(str2, str);
2643 return str2;
2644 }
2645 len = NUM2LONG(times);
2646 if (len < 0) {
2647 rb_raise(rb_eArgError, "negative argument");
2648 }
2649 if (RSTRING_LEN(str) == 1 && RSTRING_PTR(str)[0] == 0) {
2650 if (STR_EMBEDDABLE_P(len, 1)) {
2651 str2 = str_alloc_embed(rb_cString, len + 1);
2652 memset(RSTRING_PTR(str2), 0, len + 1);
2653 }
2654 else {
2655 str2 = str_alloc_heap(rb_cString);
2656 RSTRING(str2)->as.heap.aux.capa = len;
2657 RSTRING(str2)->as.heap.ptr = ZALLOC_N(char, (size_t)len + 1);
2658 }
2659 STR_SET_LEN(str2, len);
2660 rb_enc_copy(str2, str);
2661 return str2;
2662 }
2663 if (len && LONG_MAX/len < RSTRING_LEN(str)) {
2664 rb_raise(rb_eArgError, "argument too big");
2665 }
2666
2667 len *= RSTRING_LEN(str);
2668 termlen = TERM_LEN(str);
2669 str2 = str_enc_new(rb_cString, 0, len, STR_ENC_GET(str));
2670 ptr2 = RSTRING_PTR(str2);
2671 if (len) {
2672 n = RSTRING_LEN(str);
2673 memcpy(ptr2, RSTRING_PTR(str), n);
2674 while (n <= len/2) {
2675 memcpy(ptr2 + n, ptr2, n);
2676 n *= 2;
2677 }
2678 memcpy(ptr2 + n, ptr2, len-n);
2679 }
2680 STR_SET_LEN(str2, len);
2681 TERM_FILL(&ptr2[len], termlen);
2682 rb_enc_cr_str_copy_for_substr(str2, str);
2683
2684 return str2;
2685}
2686
2687/*
2688 * call-seq:
2689 * self % object -> new_string
2690 *
2691 * Returns the result of formatting +object+ into the format specifications
2692 * contained in +self+
2693 * (see {Format Specifications}[rdoc-ref:language/format_specifications.rdoc]):
2694 *
2695 * '%05d' % 123 # => "00123"
2696 *
2697 * If +self+ contains multiple format specifications,
2698 * +object+ must be an array or hash containing the objects to be formatted:
2699 *
2700 * '%-5s: %016x' % [ 'ID', self.object_id ] # => "ID : 00002b054ec93168"
2701 * 'foo = %{foo}' % {foo: 'bar'} # => "foo = bar"
2702 * 'foo = %{foo}, baz = %{baz}' % {foo: 'bar', baz: 'bat'} # => "foo = bar, baz = bat"
2703 *
2704 * Related: see {Converting to New String}[rdoc-ref:String@Converting+to+New+String].
2705 */
2706
2707static VALUE
2708rb_str_format_m(VALUE str, VALUE arg)
2709{
2710 VALUE tmp = rb_check_array_type(arg);
2711
2712 if (!NIL_P(tmp)) {
2713 VALUE result = rb_str_format_ary(RARRAY_LENINT(tmp), RARRAY_CONST_PTR(tmp), str, tmp);
2714 RB_GC_GUARD(tmp);
2715 return result;
2716 }
2717 return rb_str_format(1, &arg, str);
2718}
2719
2720static inline void
2721rb_check_lockedtmp(VALUE str)
2722{
2723 if (FL_TEST(str, STR_TMPLOCK)) {
2724 rb_raise(rb_eRuntimeError, "can't modify string; temporarily locked");
2725 }
2726}
2727
2728// If none of these flags are set, we know we have an modifiable string.
2729// If any is set, we need to do more detailed checks.
2730#define STR_UNMODIFIABLE_MASK (FL_FREEZE | STR_TMPLOCK | STR_CHILLED)
2731static inline void
2732str_modifiable(VALUE str)
2733{
2734 RUBY_ASSERT(ruby_thread_has_gvl_p());
2735
2736 if (RB_UNLIKELY(FL_ANY_RAW(str, STR_UNMODIFIABLE_MASK))) {
2737 if (CHILLED_STRING_P(str)) {
2738 CHILLED_STRING_MUTATED(str);
2739 }
2740 rb_check_lockedtmp(str);
2741 rb_check_frozen(str);
2742 }
2743}
2744
2745static inline int
2746str_dependent_p(VALUE str)
2747{
2748 if (STR_EMBED_P(str) || !FL_TEST(str, STR_SHARED|STR_NOFREE)) {
2749 return FALSE;
2750 }
2751 else {
2752 return TRUE;
2753 }
2754}
2755
2756// If none of these flags are set, we know we have an independent string.
2757// If any is set, we need to do more detailed checks.
2758#define STR_DEPENDANT_MASK (STR_UNMODIFIABLE_MASK | STR_SHARED | STR_NOFREE)
2759static inline int
2760str_independent(VALUE str)
2761{
2762 RUBY_ASSERT(ruby_thread_has_gvl_p());
2763
2764 if (RB_UNLIKELY(FL_ANY_RAW(str, STR_DEPENDANT_MASK))) {
2765 str_modifiable(str);
2766 return !str_dependent_p(str);
2767 }
2768 return TRUE;
2769}
2770
2771static void
2772str_make_independent_expand(VALUE str, long len, long expand, const int termlen)
2773{
2774 RUBY_ASSERT(ruby_thread_has_gvl_p());
2775
2776 char *ptr;
2777 char *oldptr;
2778 long capa = len + expand;
2779
2780 if (len > capa) len = capa;
2781
2782 if (!STR_EMBED_P(str) && str_embed_capa(str) >= capa + termlen) {
2783 ptr = RSTRING(str)->as.heap.ptr;
2784 STR_SET_EMBED(str);
2785 memcpy(RSTRING(str)->as.embed.ary, ptr, len);
2786 TERM_FILL(RSTRING(str)->as.embed.ary + len, termlen);
2787 STR_SET_LEN(str, len);
2788 return;
2789 }
2790
2791 ptr = ALLOC_N(char, (size_t)capa + termlen);
2792 oldptr = RSTRING_PTR(str);
2793 if (oldptr) {
2794 memcpy(ptr, oldptr, len);
2795 }
2796 if (FL_TEST_RAW(str, STR_NOEMBED|STR_NOFREE|STR_SHARED) == STR_NOEMBED) {
2797 SIZED_FREE_N(oldptr, STR_HEAP_SIZE(str));
2798 }
2799 STR_SET_NOEMBED(str);
2800 FL_UNSET(str, STR_SHARED|STR_NOFREE);
2801 TERM_FILL(ptr + len, termlen);
2802 RSTRING(str)->as.heap.ptr = ptr;
2803 STR_SET_LEN(str, len);
2804 RSTRING(str)->as.heap.aux.capa = capa;
2805}
2806
2807void
2808rb_str_modify(VALUE str)
2809{
2810 if (!str_independent(str))
2811 str_make_independent(str);
2813}
2814
2815void
2817{
2818 RUBY_ASSERT(ruby_thread_has_gvl_p());
2819
2820 int termlen = TERM_LEN(str);
2821 long len = RSTRING_LEN(str);
2822
2823 if (expand < 0) {
2824 rb_raise(rb_eArgError, "negative expanding string size");
2825 }
2826 if (expand >= LONG_MAX - len) {
2827 rb_raise(rb_eArgError, "string size too big");
2828 }
2829
2830 if (!str_independent(str)) {
2831 str_make_independent_expand(str, len, expand, termlen);
2832 }
2833 else if (expand > 0) {
2834 RESIZE_CAPA_TERM(str, len + expand, termlen);
2835 }
2837}
2838
2839/* As rb_str_modify(), but don't clear coderange */
2840static void
2841str_modify_keep_cr(VALUE str)
2842{
2843 if (!str_independent(str))
2844 str_make_independent(str);
2846 /* Force re-scan later */
2848}
2849
2850static inline void
2851str_discard(VALUE str)
2852{
2853 str_modifiable(str);
2854 if (!STR_EMBED_P(str) && !FL_TEST(str, STR_SHARED|STR_NOFREE)) {
2855 SIZED_FREE_N(STR_HEAP_PTR(str), STR_HEAP_SIZE(str));
2856 RSTRING(str)->as.heap.ptr = 0;
2857 STR_SET_LEN(str, 0);
2858 }
2859}
2860
2861void
2863{
2864 int encindex = rb_enc_get_index(str);
2865
2866 if (RB_UNLIKELY(encindex == -1)) {
2867 rb_raise(rb_eTypeError, "not encoding capable object");
2868 }
2869
2870 if (RB_LIKELY(rb_str_encindex_fastpath(encindex))) {
2871 return;
2872 }
2873
2874 rb_encoding *enc = rb_enc_from_index(encindex);
2875 if (!rb_enc_asciicompat(enc)) {
2876 rb_raise(rb_eEncCompatError, "ASCII incompatible encoding: %s", rb_enc_name(enc));
2877 }
2878}
2879
2880VALUE
2882{
2883 RUBY_ASSERT(ruby_thread_has_gvl_p());
2884
2885 VALUE s = *ptr;
2886 if (!RB_TYPE_P(s, T_STRING)) {
2887 s = rb_str_to_str(s);
2888 *ptr = s;
2889 }
2890 return s;
2891}
2892
2893char *
2895{
2896 VALUE str = rb_string_value(ptr);
2897 return RSTRING_PTR(str);
2898}
2899
2900static const char *
2901str_null_char(const char *s, long len, const int minlen, rb_encoding *enc)
2902{
2903 const char *e = s + len;
2904
2905 for (; s + minlen <= e; s += rb_enc_mbclen(s, e, enc)) {
2906 if (zero_filled(s, minlen)) return s;
2907 }
2908 return 0;
2909}
2910
2911static char *
2912str_fill_term(VALUE str, char *s, long len, int termlen)
2913{
2914 /* This function assumes that (capa + termlen) bytes of memory
2915 * is allocated, like many other functions in this file.
2916 */
2917 if (str_dependent_p(str)) {
2918 if (!zero_filled(s + len, termlen))
2919 str_make_independent_expand(str, len, 0L, termlen);
2920 }
2921 else {
2922 TERM_FILL(s + len, termlen);
2923 return s;
2924 }
2925 return RSTRING_PTR(str);
2926}
2927
2928void
2929rb_str_change_terminator_length(VALUE str, const int oldtermlen, const int termlen)
2930{
2931 long capa = str_capacity(str, oldtermlen) + oldtermlen;
2932 long len = RSTRING_LEN(str);
2933
2934 RUBY_ASSERT(capa >= len);
2935 if (capa - len < termlen) {
2936 rb_check_lockedtmp(str);
2937 str_make_independent_expand(str, len, 0L, termlen);
2938 }
2939 else if (str_dependent_p(str)) {
2940 if (termlen > oldtermlen)
2941 str_make_independent_expand(str, len, 0L, termlen);
2942 }
2943 else {
2944 if (!STR_EMBED_P(str)) {
2945 /* modify capa instead of realloc */
2946 RUBY_ASSERT(!FL_TEST((str), STR_SHARED));
2947 RSTRING(str)->as.heap.aux.capa = capa - termlen;
2948 }
2949 if (termlen > oldtermlen) {
2950 TERM_FILL(RSTRING_PTR(str) + len, termlen);
2951 }
2952 }
2953
2954 return;
2955}
2956
2957static char *
2958str_null_check(VALUE str, int *w)
2959{
2960 char *s = RSTRING_PTR(str);
2961 long len = RSTRING_LEN(str);
2962 int minlen = 1;
2963
2964 if (RB_UNLIKELY(!rb_str_enc_fastpath(str))) {
2965 rb_encoding *enc = rb_str_enc_get(str);
2966 minlen = rb_enc_mbminlen(enc);
2967
2968 if (minlen > 1) {
2969 *w = 1;
2970 if (str_null_char(s, len, minlen, enc)) {
2971 return NULL;
2972 }
2973 return str_fill_term(str, s, len, minlen);
2974 }
2975 }
2976
2977 *w = 0;
2978 if (!s || memchr(s, 0, len)) {
2979 return NULL;
2980 }
2981 if (s[len]) {
2982 s = str_fill_term(str, s, len, minlen);
2983 }
2984 return s;
2985}
2986
2987static char *str_to_cstr(VALUE str);
2988
2989const char *
2990rb_str_null_check(VALUE str)
2991{
2993
2994 const char *s;
2995 long len;
2996 RSTRING_GETMEM(str, s, len);
2997
2998 if (RB_LIKELY(rb_str_enc_fastpath(str))) {
2999 if (!s || memchr(s, 0, len)) {
3000 rb_raise(rb_eArgError, "string contains null byte");
3001 }
3002 }
3003 else {
3004 str_to_cstr(str);
3005 }
3006
3007 return s;
3008}
3009
3010char *
3011rb_str_to_cstr(VALUE str)
3012{
3013 int w;
3014 return str_null_check(str, &w);
3015}
3016
3017char *
3019{
3020 VALUE str = rb_string_value(ptr);
3021 return str_to_cstr(str);
3022}
3023
3024static char *
3025str_to_cstr(VALUE str)
3026{
3027 int w;
3028 char *s = str_null_check(str, &w);
3029 if (!s) {
3030 if (w) {
3031 rb_raise(rb_eArgError, "string contains null char");
3032 }
3033 rb_raise(rb_eArgError, "string contains null byte");
3034 }
3035 return s;
3036}
3037
3038char *
3039rb_str_fill_terminator(VALUE str, const int newminlen)
3040{
3041 char *s = RSTRING_PTR(str);
3042 long len = RSTRING_LEN(str);
3043 return str_fill_term(str, s, len, newminlen);
3044}
3045
3046VALUE
3048{
3049 str = rb_check_convert_type_with_id(str, T_STRING, "String", idTo_str);
3050 return str;
3051}
3052
3053/*
3054 * call-seq:
3055 * String.try_convert(object) -> object, new_string, or nil
3056 *
3057 * Attempts to convert the given +object+ to a string.
3058 *
3059 * If +object+ is already a string, returns +object+, unmodified.
3060 *
3061 * Otherwise if +object+ responds to <tt>:to_str</tt>,
3062 * calls <tt>object.to_str</tt> and returns the result.
3063 *
3064 * Returns +nil+ if +object+ does not respond to <tt>:to_str</tt>.
3065 *
3066 * Raises an exception unless <tt>object.to_str</tt> returns a string.
3067 */
3068static VALUE
3069rb_str_s_try_convert(VALUE dummy, VALUE str)
3070{
3071 return rb_check_string_type(str);
3072}
3073
3074static char*
3075str_nth_len(const char *p, const char *e, long *nthp, rb_encoding *enc)
3076{
3077 long nth = *nthp;
3078 if (rb_enc_mbmaxlen(enc) == 1) {
3079 p += nth;
3080 }
3081 else if (rb_enc_mbmaxlen(enc) == rb_enc_mbminlen(enc)) {
3082 p += nth * rb_enc_mbmaxlen(enc);
3083 }
3084 else if (rb_enc_asciicompat(enc)) {
3085 const char *p2, *e2;
3086 int n;
3087
3088 while (p < e && 0 < nth) {
3089 e2 = p + nth;
3090 if (e < e2) {
3091 *nthp = nth;
3092 return (char *)e;
3093 }
3094 p2 = search_nonascii(p, e2);
3095 if (!p2) {
3096 nth -= e2 - p;
3097 *nthp = nth;
3098 return (char *)e2;
3099 }
3100 nth -= p2 - p;
3101 p = p2;
3102 n = rb_enc_mbclen(p, e, enc);
3103 p += n;
3104 nth--;
3105 }
3106 *nthp = nth;
3107 if (nth != 0) {
3108 return (char *)e;
3109 }
3110 return (char *)p;
3111 }
3112 else {
3113 while (p < e && nth--) {
3114 p += rb_enc_mbclen(p, e, enc);
3115 }
3116 }
3117 if (p > e) p = e;
3118 *nthp = nth;
3119 return (char*)p;
3120}
3121
3122char*
3123rb_enc_nth(const char *p, const char *e, long nth, rb_encoding *enc)
3124{
3125 return str_nth_len(p, e, &nth, enc);
3126}
3127
3128static char*
3129str_nth(const char *p, const char *e, long nth, rb_encoding *enc, int singlebyte)
3130{
3131 if (singlebyte)
3132 p += nth;
3133 else {
3134 p = str_nth_len(p, e, &nth, enc);
3135 }
3136 if (!p) return 0;
3137 if (p > e) p = e;
3138 return (char *)p;
3139}
3140
3141/* char offset to byte offset */
3142static long
3143str_offset(const char *p, const char *e, long nth, rb_encoding *enc, int singlebyte)
3144{
3145 const char *pp = str_nth(p, e, nth, enc, singlebyte);
3146 if (!pp) return e - p;
3147 return pp - p;
3148}
3149
3150long
3151rb_str_offset(VALUE str, long pos)
3152{
3153 return str_offset(RSTRING_PTR(str), RSTRING_END(str), pos,
3154 STR_ENC_GET(str), single_byte_optimizable(str));
3155}
3156
3157#ifdef NONASCII_MASK
3158static char *
3159str_utf8_nth(const char *p, const char *e, long *nthp)
3160{
3161 long nth = *nthp;
3162 if ((int)SIZEOF_VOIDP * 2 < e - p && (int)SIZEOF_VOIDP * 2 < nth) {
3163 const uintptr_t *s, *t;
3164 const uintptr_t lowbits = SIZEOF_VOIDP - 1;
3165 s = (const uintptr_t*)(~lowbits & ((uintptr_t)p + lowbits));
3166 t = (const uintptr_t*)(~lowbits & (uintptr_t)e);
3167 while (p < (const char *)s) {
3168 if (is_utf8_lead_byte(*p)) nth--;
3169 p++;
3170 }
3171 do {
3172 nth -= count_utf8_lead_bytes_with_word(s);
3173 s++;
3174 } while (s < t && (int)SIZEOF_VOIDP <= nth);
3175 p = (char *)s;
3176 }
3177 while (p < e) {
3178 if (is_utf8_lead_byte(*p)) {
3179 if (nth == 0) break;
3180 nth--;
3181 }
3182 p++;
3183 }
3184 *nthp = nth;
3185 return (char *)p;
3186}
3187
3188static long
3189str_utf8_offset(const char *p, const char *e, long nth)
3190{
3191 const char *pp = str_utf8_nth(p, e, &nth);
3192 return pp - p;
3193}
3194#endif
3195
3196/* byte offset to char offset */
3197long
3198rb_str_sublen(VALUE str, long pos)
3199{
3200 if (single_byte_optimizable(str) || pos < 0)
3201 return pos;
3202 else {
3203 const char *p = RSTRING_PTR(str);
3204 return enc_strlen(p, p + pos, STR_ENC_GET(str), ENC_CODERANGE(str));
3205 }
3206}
3207
3208static VALUE
3209str_subseq(VALUE str, long beg, long len)
3210{
3211 VALUE str2;
3212
3213 RUBY_ASSERT(beg >= 0);
3214 RUBY_ASSERT(len >= 0);
3215 RUBY_ASSERT(beg+len <= RSTRING_LEN(str));
3216
3217 const int termlen = TERM_LEN(str);
3218 if (!SHARABLE_SUBSTRING_P(str, beg, len)) {
3219 str2 = rb_enc_str_new(RSTRING_PTR(str) + beg, len, rb_str_enc_get(str));
3220 if (ENC_CODERANGE(str) == ENC_CODERANGE_7BIT) {
3222 }
3223 RB_GC_GUARD(str);
3224 return str2;
3225 }
3226
3227 /* Sharing allocates a shared root as well unless str can be one itself, so
3228 * a copy is worth a larger slot only when it saves that second object. */
3229 const bool root_available = STR_SHARED_P(str) ||
3230 RB_FL_TEST_RAW(str, FL_FREEZE | STR_CHILLED) == FL_FREEZE;
3231 const size_t max_embed_size = root_available ?
3232 rb_gc_size_slot_size(sizeof(struct RString)) : STR_COPY_MAX_EMBED_SIZE;
3233 const size_t embed_size = rb_str_embed_size(len, termlen);
3234
3235 if (embed_size <= max_embed_size && rb_gc_size_allocatable_p(embed_size)) {
3236 str2 = str_alloc_embed(rb_cString, len + termlen);
3237 char *ptr2 = RSTRING(str2)->as.embed.ary;
3238 memcpy(ptr2, RSTRING_PTR(str) + beg, len);
3239 TERM_FILL(ptr2 + len, termlen);
3240
3241 STR_SET_LEN(str2, len);
3242 if (ENC_CODERANGE(str) == ENC_CODERANGE_7BIT) {
3244 }
3245
3246 RB_GC_GUARD(str);
3247 }
3248 else {
3249 str2 = str_alloc_heap(rb_cString);
3250 str_replace_shared(str2, str);
3251 RUBY_ASSERT(!STR_EMBED_P(str2));
3252 if (ENC_CODERANGE(str) != ENC_CODERANGE_7BIT) {
3253 ENC_CODERANGE_CLEAR(str2);
3254 }
3255
3256 RSTRING(str2)->as.heap.ptr += beg;
3257 if (RSTRING_LEN(str2) > len) {
3258 STR_SET_LEN(str2, len);
3259 }
3260 }
3261
3262 return str2;
3263}
3264
3265VALUE
3266rb_str_subseq(VALUE str, long beg, long len)
3267{
3268 VALUE str2 = str_subseq(str, beg, len);
3269 rb_enc_cr_str_copy_for_substr(str2, str);
3270 return str2;
3271}
3272
3273char *
3274rb_str_subpos(VALUE str, long beg, long *lenp)
3275{
3276 long len = *lenp;
3277 long slen = -1L;
3278 const long blen = RSTRING_LEN(str);
3279 rb_encoding *enc = STR_ENC_GET(str);
3280 const char *p, *s = RSTRING_PTR(str), *e = s + blen;
3281
3282 if (len < 0) return 0;
3283 if (beg < 0 && -beg < 0) return 0;
3284 if (!blen) {
3285 len = 0;
3286 }
3287 if (single_byte_optimizable(str)) {
3288 if (beg > blen) return 0;
3289 if (beg < 0) {
3290 beg += blen;
3291 if (beg < 0) return 0;
3292 }
3293 if (len > blen - beg)
3294 len = blen - beg;
3295 if (len < 0) return 0;
3296 p = s + beg;
3297 goto end;
3298 }
3299 if (beg < 0) {
3300 if (len > -beg) len = -beg;
3301 if ((ENC_CODERANGE(str) == ENC_CODERANGE_VALID) &&
3302 (-beg * rb_enc_mbmaxlen(enc) < blen / 8)) {
3303 beg = -beg;
3304 while (beg-- > len && (e = rb_enc_prev_char(s, e, e, enc)) != 0);
3305 p = e;
3306 if (!p) return 0;
3307 while (len-- > 0 && (p = rb_enc_prev_char(s, p, e, enc)) != 0);
3308 if (!p) return 0;
3309 len = e - p;
3310 goto end;
3311 }
3312 else {
3313 slen = str_strlen(str, enc);
3314 beg += slen;
3315 if (beg < 0) return 0;
3316 p = s + beg;
3317 if (len == 0) goto end;
3318 }
3319 }
3320 else if (beg > 0 && beg > blen) {
3321 return 0;
3322 }
3323 if (len == 0) {
3324 if (beg > str_strlen(str, enc)) return 0; /* str's enc */
3325 p = s + beg;
3326 }
3327#ifdef NONASCII_MASK
3328 else if (ENC_CODERANGE(str) == ENC_CODERANGE_VALID &&
3329 enc == rb_utf8_encoding()) {
3330 p = str_utf8_nth(s, e, &beg);
3331 if (beg > 0) return 0;
3332 len = str_utf8_offset(p, e, len);
3333 }
3334#endif
3335 else if (rb_enc_mbmaxlen(enc) == rb_enc_mbminlen(enc)) {
3336 int char_sz = rb_enc_mbmaxlen(enc);
3337
3338 p = s + beg * char_sz;
3339 if (p > e) {
3340 return 0;
3341 }
3342 else if (len * char_sz > e - p)
3343 len = e - p;
3344 else
3345 len *= char_sz;
3346 }
3347 else if ((p = str_nth_len(s, e, &beg, enc)) == e) {
3348 if (beg > 0) return 0;
3349 len = 0;
3350 }
3351 else {
3352 len = str_offset(p, e, len, enc, 0);
3353 }
3354 end:
3355 *lenp = len;
3356 RB_GC_GUARD(str);
3357 return (char *)p;
3358}
3359
3360static VALUE str_substr(VALUE str, long beg, long len, int empty);
3361
3362VALUE
3363rb_str_substr(VALUE str, long beg, long len)
3364{
3365 return str_substr(str, beg, len, TRUE);
3366}
3367
3368VALUE
3369rb_str_substr_two_fixnums(VALUE str, VALUE beg, VALUE len, int empty)
3370{
3371 return str_substr(str, NUM2LONG(beg), NUM2LONG(len), empty);
3372}
3373
3374static VALUE
3375str_substr(VALUE str, long beg, long len, int empty)
3376{
3377 const char *p = rb_str_subpos(str, beg, &len);
3378
3379 if (!p) return Qnil;
3380 if (!len && !empty) return Qnil;
3381
3382 beg = p - RSTRING_PTR(str);
3383
3384 VALUE str2 = str_subseq(str, beg, len);
3385 rb_enc_cr_str_copy_for_substr(str2, str);
3386 return str2;
3387}
3388
3389/* :nodoc: */
3390VALUE
3392{
3393 if (CHILLED_STRING_P(str)) {
3394 FL_UNSET_RAW(str, STR_CHILLED);
3395 }
3396
3397 if (OBJ_FROZEN(str)) return str;
3398 rb_str_resize(str, RSTRING_LEN(str));
3399 return rb_obj_freeze(str);
3400}
3401
3402/*
3403 * call-seq:
3404 * +string -> new_string or self
3405 *
3406 * Returns +self+ if +self+ is not frozen and can be mutated
3407 * without warning issuance.
3408 *
3409 * Otherwise returns <tt>self.dup</tt>, which is not frozen.
3410 *
3411 * Related: see {Freezing/Unfreezing}[rdoc-ref:String@FreezingUnfreezing].
3412 */
3413static VALUE
3414str_uplus(VALUE str)
3415{
3416 if (OBJ_FROZEN(str) || CHILLED_STRING_P(str)) {
3417 return rb_str_dup(str);
3418 }
3419 else {
3420 return str;
3421 }
3422}
3423
3424/*
3425 * call-seq:
3426 * -self -> frozen_string
3427 *
3428 * Returns a frozen string equal to +self+.
3429 *
3430 * The returned string is +self+ if and only if all of the following are true:
3431 *
3432 * - +self+ is already frozen.
3433 * - +self+ is an instance of \String (rather than of a subclass of \String)
3434 * - +self+ has no instance variables set on it.
3435 *
3436 * Otherwise, the returned string is a frozen copy of +self+.
3437 *
3438 * Returning +self+, when possible, saves duplicating +self+;
3439 * see {Data deduplication}[https://en.wikipedia.org/wiki/Data_deduplication].
3440 *
3441 * It may also save duplicating other, already-existing, strings:
3442 *
3443 * s0 = 'foo'
3444 * s1 = 'foo'
3445 * s0.object_id == s1.object_id # => false
3446 * (-s0).object_id == (-s1).object_id # => true
3447 *
3448 * Note that method #-@ is convenient for defining a constant:
3449 *
3450 * FileName = -'config/database.yml'
3451 *
3452 * While its alias #dedup is better suited for chaining:
3453 *
3454 * 'foo'.dedup.gsub!('o')
3455 *
3456 * Related: see {Freezing/Unfreezing}[rdoc-ref:String@FreezingUnfreezing].
3457 */
3458static VALUE
3459str_uminus(VALUE str)
3460{
3461 if (!BARE_STRING_P(str) && !rb_obj_frozen_p(str)) {
3462 str = rb_str_dup(str);
3463 }
3464 return rb_fstring(str);
3465}
3466
3467RUBY_ALIAS_FUNCTION(rb_str_dup_frozen(VALUE str), rb_str_new_frozen, (str))
3468#define rb_str_dup_frozen rb_str_new_frozen
3469
3470VALUE
3472{
3473 rb_check_frozen(str);
3474 if (FL_TEST(str, STR_TMPLOCK)) {
3475 rb_raise(rb_eRuntimeError, "temporal locking already locked string");
3476 }
3477 FL_SET(str, STR_TMPLOCK);
3478 return str;
3479}
3480
3481VALUE
3483{
3484 rb_check_frozen(str);
3485 if (!FL_TEST(str, STR_TMPLOCK)) {
3486 rb_raise(rb_eRuntimeError, "temporal unlocking already unlocked string");
3487 }
3488 FL_UNSET(str, STR_TMPLOCK);
3489 return str;
3490}
3491
3492VALUE
3493rb_str_locktmp_ensure(VALUE str, VALUE (*func)(VALUE), VALUE arg)
3494{
3495 rb_str_locktmp(str);
3496 return rb_ensure(func, arg, rb_str_unlocktmp, str);
3497}
3498
3499void
3501{
3502 RUBY_ASSERT(ruby_thread_has_gvl_p());
3503
3504 long capa;
3505 const int termlen = TERM_LEN(str);
3506
3507 str_modifiable(str);
3508 if (STR_SHARED_P(str)) {
3509 rb_raise(rb_eRuntimeError, "can't set length of shared string");
3510 }
3511 if (len > (capa = (long)str_capacity(str, termlen)) || len < 0) {
3512 rb_bug("probable buffer overflow: %ld for %ld", len, capa);
3513 }
3514
3515 int cr = ENC_CODERANGE(str);
3516 if (len == 0) {
3517 /* Empty string does not contain non-ASCII */
3519 }
3520 else if (cr == ENC_CODERANGE_UNKNOWN) {
3521 /* Leave unknown. */
3522 }
3523 else if (len > RSTRING_LEN(str)) {
3524 if (ENC_CODERANGE_CLEAN_P(cr)) {
3525 /* Update the coderange regarding the extended part. */
3526 const char *const prev_end = RSTRING_END(str);
3527 const char *const new_end = RSTRING_PTR(str) + len;
3528 rb_encoding *enc = rb_enc_get(str);
3529 rb_str_coderange_scan_restartable(prev_end, new_end, enc, &cr);
3530 ENC_CODERANGE_SET(str, cr);
3531 }
3532 else if (cr == ENC_CODERANGE_BROKEN) {
3533 /* May be valid now, by appended part. */
3535 }
3536 }
3537 else if (len < RSTRING_LEN(str)) {
3538 if (cr != ENC_CODERANGE_7BIT) {
3539 /* ASCII-only string is keeping after truncated. Valid
3540 * and broken may be invalid or valid, leave unknown. */
3542 }
3543 }
3544
3545 STR_SET_LEN(str, len);
3546 TERM_FILL(&RSTRING_PTR(str)[len], termlen);
3547}
3548
3549VALUE
3550rb_str_resize(VALUE str, long len)
3551{
3552 if (len < 0) {
3553 rb_raise(rb_eArgError, "negative string size (or size too big)");
3554 }
3555
3556 int independent = str_independent(str);
3557 long slen = RSTRING_LEN(str);
3558 const int termlen = TERM_LEN(str);
3559
3560 if (slen > len || (termlen != 1 && slen < len)) {
3562 }
3563
3564 {
3565 long capa;
3566 if (STR_EMBED_P(str)) {
3567 if (len == slen) return str;
3568 if (str_embed_capa(str) >= len + termlen) {
3569 STR_SET_LEN(str, len);
3570 TERM_FILL(RSTRING(str)->as.embed.ary + len, termlen);
3571 return str;
3572 }
3573 str_make_independent_expand(str, slen, len - slen, termlen);
3574 }
3575 else if (str_embed_capa(str) >= len + termlen) {
3576 capa = RSTRING(str)->as.heap.aux.capa;
3577 char *ptr = STR_HEAP_PTR(str);
3578 STR_SET_EMBED(str);
3579 if (slen > len) slen = len;
3580 if (slen > 0) MEMCPY(RSTRING(str)->as.embed.ary, ptr, char, slen);
3581 TERM_FILL(RSTRING(str)->as.embed.ary + len, termlen);
3582 STR_SET_LEN(str, len);
3583 if (independent) {
3584 SIZED_FREE_N(ptr, capa + termlen);
3585 }
3586 return str;
3587 }
3588 else if (!independent) {
3589 if (len == slen) return str;
3590 str_make_independent_expand(str, slen, len - slen, termlen);
3591 }
3592 else if ((capa = RSTRING(str)->as.heap.aux.capa) < len ||
3593 (capa - len) > (len < 1024 ? len : 1024)) {
3594 SIZED_REALLOC_N(RSTRING(str)->as.heap.ptr, char,
3595 (size_t)len + termlen, STR_HEAP_SIZE(str));
3596 RSTRING(str)->as.heap.aux.capa = len;
3597 }
3598 else if (len == slen) return str;
3599 STR_SET_LEN(str, len);
3600 TERM_FILL(RSTRING(str)->as.heap.ptr + len, termlen); /* sentinel */
3601 }
3602 return str;
3603}
3604
3605static void
3606str_ensure_available_capa(VALUE str, long len)
3607{
3608 str_modify_keep_cr(str);
3609
3610 const int termlen = TERM_LEN(str);
3611 long olen = RSTRING_LEN(str);
3612
3613 if (RB_UNLIKELY(olen > LONG_MAX - len)) {
3614 rb_raise(rb_eArgError, "string sizes too big");
3615 }
3616
3617 long total = olen + len;
3618 long capa = str_capacity(str, termlen);
3619
3620 if (capa < total) {
3621 if (total >= LONG_MAX / 2) {
3622 capa = total;
3623 }
3624 while (total > capa) {
3625 capa = 2 * capa + termlen; /* == 2*(capa+termlen)-termlen */
3626 }
3627 RESIZE_CAPA_TERM(str, capa, termlen);
3628 }
3629}
3630
3631static VALUE
3632str_buf_cat4(VALUE str, const char *ptr, long len, bool keep_cr)
3633{
3634 if (keep_cr) {
3635 str_modify_keep_cr(str);
3636 }
3637 else {
3638 rb_str_modify(str);
3639 }
3640 if (len == 0) return 0;
3641
3642 long total, olen, off = -1;
3643 char *sptr;
3644 const int termlen = TERM_LEN(str);
3645
3646 RSTRING_GETMEM(str, sptr, olen);
3647 if (ptr >= sptr && ptr <= sptr + olen) {
3648 off = ptr - sptr;
3649 }
3650
3651 long capa = str_capacity(str, termlen);
3652
3653 if (olen > LONG_MAX - len) {
3654 rb_raise(rb_eArgError, "string sizes too big");
3655 }
3656 total = olen + len;
3657 if (capa < total) {
3658 if (total >= LONG_MAX / 2) {
3659 capa = total;
3660 }
3661 while (total > capa) {
3662 capa = 2 * capa + termlen; /* == 2*(capa+termlen)-termlen */
3663 }
3664 RESIZE_CAPA_TERM(str, capa, termlen);
3665 sptr = RSTRING_PTR(str);
3666 }
3667 if (off != -1) {
3668 ptr = sptr + off;
3669 }
3670 memcpy(sptr + olen, ptr, len);
3671 STR_SET_LEN(str, total);
3672 TERM_FILL(sptr + total, termlen); /* sentinel */
3673
3674 return str;
3675}
3676
3677#define str_buf_cat(str, ptr, len) str_buf_cat4((str), (ptr), len, false)
3678#define str_buf_cat2(str, ptr) str_buf_cat4((str), (ptr), rb_strlen_lit(ptr), false)
3679
3680VALUE
3681rb_str_cat(VALUE str, const char *ptr, long len)
3682{
3683 if (len == 0) return str;
3684 if (len < 0) {
3685 rb_raise(rb_eArgError, "negative string size (or size too big)");
3686 }
3687 return str_buf_cat(str, ptr, len);
3688}
3689
3690VALUE
3691rb_str_cat_cstr(VALUE str, const char *ptr)
3692{
3693 must_not_null(ptr);
3694 return rb_str_buf_cat(str, ptr, strlen(ptr));
3695}
3696
3697static void
3698rb_str_buf_cat_byte(VALUE str, unsigned char byte)
3699{
3700 RUBY_ASSERT(RB_ENCODING_GET_INLINED(str) == ENCINDEX_ASCII_8BIT || RB_ENCODING_GET_INLINED(str) == ENCINDEX_US_ASCII);
3701
3702 // We can't write directly to shared strings without impacting others, so we must make the string independent.
3703 if (UNLIKELY(!str_independent(str))) {
3704 str_make_independent(str);
3705 }
3706
3707 long string_length = -1;
3708 const int null_terminator_length = 1;
3709 char *sptr;
3710 RSTRING_GETMEM(str, sptr, string_length);
3711
3712 // Ensure the resulting string wouldn't be too long.
3713 if (UNLIKELY(string_length > LONG_MAX - 1)) {
3714 rb_raise(rb_eArgError, "string sizes too big");
3715 }
3716
3717 long string_capacity = str_capacity(str, null_terminator_length);
3718
3719 // Get the code range before any modifications since those might clear the code range.
3720 int cr = ENC_CODERANGE(str);
3721
3722 // Check if the string has spare string_capacity to write the new byte.
3723 if (LIKELY(string_capacity >= string_length + 1)) {
3724 // In fast path we can write the new byte and note the string's new length.
3725 sptr[string_length] = byte;
3726 STR_SET_LEN(str, string_length + 1);
3727 TERM_FILL(sptr + string_length + 1, null_terminator_length);
3728 }
3729 else {
3730 // If there's not enough string_capacity, make a call into the general string concatenation function.
3731 str_buf_cat(str, (char *)&byte, 1);
3732 }
3733
3734 // If the code range is already known, we can derive the resulting code range cheaply by looking at the byte we
3735 // just appended. If the code range is unknown, but the string was empty, then we can also derive the code range
3736 // by looking at the byte we just appended. Otherwise, we'd have to scan the bytes to determine the code range so
3737 // we leave it as unknown. It cannot be broken for binary strings so we don't need to handle that option.
3738 if (cr == ENC_CODERANGE_7BIT || string_length == 0) {
3739 if (ISASCII(byte)) {
3741 }
3742 else {
3744
3745 // Promote a US-ASCII string to ASCII-8BIT when a non-ASCII byte is appended.
3746 if (UNLIKELY(RB_ENCODING_GET_INLINED(str) == ENCINDEX_US_ASCII)) {
3747 rb_enc_associate_index(str, ENCINDEX_ASCII_8BIT);
3748 }
3749 }
3750 }
3751}
3752
3753RUBY_ALIAS_FUNCTION(rb_str_buf_cat(VALUE str, const char *ptr, long len), rb_str_cat, (str, ptr, len))
3754RUBY_ALIAS_FUNCTION(rb_str_buf_cat2(VALUE str, const char *ptr), rb_str_cat_cstr, (str, ptr))
3755RUBY_ALIAS_FUNCTION(rb_str_cat2(VALUE str, const char *ptr), rb_str_cat_cstr, (str, ptr))
3756
3757static VALUE
3758rb_enc_cr_str_buf_cat(VALUE str, const char *ptr, long len,
3759 int ptr_encindex, int ptr_cr, int *ptr_cr_ret)
3760{
3761 int str_encindex = ENCODING_GET(str);
3762 int res_encindex;
3763 int str_cr, res_cr;
3764 rb_encoding *str_enc, *ptr_enc;
3765
3766 str_cr = RSTRING_LEN(str) ? ENC_CODERANGE(str) : ENC_CODERANGE_7BIT;
3767
3768 if (str_encindex == ptr_encindex) {
3769 if (str_cr != ENC_CODERANGE_UNKNOWN && ptr_cr == ENC_CODERANGE_UNKNOWN) {
3770 ptr_cr = coderange_scan(ptr, len, rb_enc_from_index(ptr_encindex));
3771 }
3772 }
3773 else {
3774 str_enc = rb_enc_from_index(str_encindex);
3775 ptr_enc = rb_enc_from_index(ptr_encindex);
3776 if (!rb_enc_asciicompat(str_enc) || !rb_enc_asciicompat(ptr_enc)) {
3777 if (len == 0)
3778 return str;
3779 if (RSTRING_LEN(str) == 0) {
3780 rb_str_buf_cat(str, ptr, len);
3781 ENCODING_CODERANGE_SET(str, ptr_encindex, ptr_cr);
3782 rb_str_change_terminator_length(str, rb_enc_mbminlen(str_enc), rb_enc_mbminlen(ptr_enc));
3783 return str;
3784 }
3785 goto incompatible;
3786 }
3787 if (ptr_cr == ENC_CODERANGE_UNKNOWN) {
3788 ptr_cr = coderange_scan(ptr, len, ptr_enc);
3789 }
3790 if (str_cr == ENC_CODERANGE_UNKNOWN) {
3791 if (ENCODING_IS_ASCII8BIT(str) || ptr_cr != ENC_CODERANGE_7BIT) {
3792 str_cr = rb_enc_str_coderange(str);
3793 }
3794 }
3795 }
3796 if (ptr_cr_ret)
3797 *ptr_cr_ret = ptr_cr;
3798
3799 if (str_encindex != ptr_encindex &&
3800 str_cr != ENC_CODERANGE_7BIT &&
3801 ptr_cr != ENC_CODERANGE_7BIT) {
3802 str_enc = rb_enc_from_index(str_encindex);
3803 ptr_enc = rb_enc_from_index(ptr_encindex);
3804 goto incompatible;
3805 }
3806
3807 if (str_cr == ENC_CODERANGE_UNKNOWN) {
3808 res_encindex = str_encindex;
3809 res_cr = ENC_CODERANGE_UNKNOWN;
3810 }
3811 else if (str_cr == ENC_CODERANGE_7BIT) {
3812 if (ptr_cr == ENC_CODERANGE_7BIT) {
3813 res_encindex = str_encindex;
3814 res_cr = ENC_CODERANGE_7BIT;
3815 }
3816 else {
3817 res_encindex = ptr_encindex;
3818 res_cr = ptr_cr;
3819 }
3820 }
3821 else if (str_cr == ENC_CODERANGE_VALID) {
3822 res_encindex = str_encindex;
3823 if (ENC_CODERANGE_CLEAN_P(ptr_cr))
3824 res_cr = str_cr;
3825 else
3826 res_cr = ptr_cr;
3827 }
3828 else { /* str_cr == ENC_CODERANGE_BROKEN */
3829 res_encindex = str_encindex;
3830 res_cr = str_cr;
3831 if (0 < len) res_cr = ENC_CODERANGE_UNKNOWN;
3832 }
3833
3834 if (len < 0) {
3835 rb_raise(rb_eArgError, "negative string size (or size too big)");
3836 }
3837 str_buf_cat(str, ptr, len);
3838 ENCODING_CODERANGE_SET(str, res_encindex, res_cr);
3839 return str;
3840
3841 incompatible:
3842 rb_raise(rb_eEncCompatError, "incompatible character encodings: %s and %s",
3843 rb_enc_inspect_name(str_enc), rb_enc_inspect_name(ptr_enc));
3845}
3846
3847VALUE
3848rb_enc_str_buf_cat(VALUE str, const char *ptr, long len, rb_encoding *ptr_enc)
3849{
3850 return rb_enc_cr_str_buf_cat(str, ptr, len,
3851 rb_enc_to_index(ptr_enc), ENC_CODERANGE_UNKNOWN, NULL);
3852}
3853
3854VALUE
3856{
3857 /* ptr must reference NUL terminated ASCII string. */
3858 int encindex = ENCODING_GET(str);
3859 rb_encoding *enc = rb_enc_from_index(encindex);
3860 if (rb_enc_asciicompat(enc)) {
3861 return rb_enc_cr_str_buf_cat(str, ptr, strlen(ptr),
3862 encindex, ENC_CODERANGE_7BIT, 0);
3863 }
3864 else {
3865 char *buf = ALLOCA_N(char, rb_enc_mbmaxlen(enc));
3866 while (*ptr) {
3867 unsigned int c = (unsigned char)*ptr;
3868 int len = rb_enc_codelen(c, enc);
3869 rb_enc_mbcput(c, buf, enc);
3870 rb_enc_cr_str_buf_cat(str, buf, len,
3871 encindex, ENC_CODERANGE_VALID, 0);
3872 ptr++;
3873 }
3874 return str;
3875 }
3876}
3877
3878VALUE
3880{
3881 int str2_cr = rb_enc_str_coderange(str2);
3882
3883 if (rb_str_enc_fastpath(str)) {
3884 switch (str2_cr) {
3885 case ENC_CODERANGE_7BIT:
3886 // If RHS is 7bit we can do simple concatenation
3887 str_buf_cat4(str, RSTRING_PTR(str2), RSTRING_LEN(str2), true);
3888 RB_GC_GUARD(str2);
3889 return str;
3891 // If RHS is valid, we can do simple concatenation if encodings are the same
3892 if (ENCODING_GET_INLINED(str) == ENCODING_GET_INLINED(str2)) {
3893 str_buf_cat4(str, RSTRING_PTR(str2), RSTRING_LEN(str2), true);
3894 int str_cr = ENC_CODERANGE(str);
3895 if (UNLIKELY(str_cr != ENC_CODERANGE_VALID)) {
3896 ENC_CODERANGE_SET(str, RB_ENC_CODERANGE_AND(str_cr, str2_cr));
3897 }
3898 RB_GC_GUARD(str2);
3899 return str;
3900 }
3901 }
3902 }
3903
3904 rb_enc_cr_str_buf_cat(str, RSTRING_PTR(str2), RSTRING_LEN(str2),
3905 ENCODING_GET(str2), str2_cr, &str2_cr);
3906
3907 ENC_CODERANGE_SET(str2, str2_cr);
3908
3909 return str;
3910}
3911
3912VALUE
3914{
3915 StringValue(str2);
3916 return rb_str_buf_append(str, str2);
3917}
3918
3919VALUE
3920rb_str_concat_literals(size_t num, const VALUE *strary)
3921{
3922 VALUE str;
3923 size_t i, s = 0;
3924 unsigned long len = 1;
3925
3926 if (UNLIKELY(!num)) return rb_str_new(0, 0);
3927 if (UNLIKELY(num == 1)) return rb_str_resurrect(strary[0]);
3928
3929 for (i = 0; i < num; ++i) { len += RSTRING_LEN(strary[i]); }
3930 str = rb_str_buf_new(len);
3931 str_enc_copy_direct(str, strary[0]);
3932
3933 for (i = s; i < num; ++i) {
3934 const VALUE v = strary[i];
3935 int encidx = ENCODING_GET(v);
3936
3937 rb_str_buf_append(str, v);
3938 if (encidx != ENCINDEX_US_ASCII) {
3939 if (ENCODING_GET_INLINED(str) == ENCINDEX_US_ASCII)
3940 rb_enc_set_index(str, encidx);
3941 }
3942 }
3943 return str;
3944}
3945
3946/*
3947 * call-seq:
3948 * concat(*objects) -> string
3949 *
3950 * :include: doc/string/concat.rdoc
3951 */
3952static VALUE
3953rb_str_concat_multi(int argc, VALUE *argv, VALUE str)
3954{
3955 str_modifiable(str);
3956
3957 if (argc == 1) {
3958 return rb_str_concat(str, argv[0]);
3959 }
3960 else if (argc > 1) {
3961 int i;
3962 VALUE arg_str = rb_str_tmp_new(0);
3963 rb_enc_copy(arg_str, str);
3964 for (i = 0; i < argc; i++) {
3965 rb_str_concat(arg_str, argv[i]);
3966 }
3967 rb_str_buf_append(str, arg_str);
3968 }
3969
3970 return str;
3971}
3972
3973/*
3974 * call-seq:
3975 * append_as_bytes(*objects) -> self
3976 *
3977 * Concatenates each object in +objects+ into +self+; returns +self+;
3978 * performs no encoding validation or conversion:
3979 *
3980 * s = 'foo'
3981 * s.append_as_bytes(" \xE2\x82") # => "foo \xE2\x82"
3982 * s.valid_encoding? # => false
3983 * s.append_as_bytes("\xAC 12")
3984 * s.valid_encoding? # => true
3985 *
3986 * When a given object is an integer,
3987 * the value is considered an 8-bit byte;
3988 * if the integer occupies more than one byte (i.e,. is greater than 255),
3989 * appends only the low-order byte (similar to String#setbyte):
3990 *
3991 * s = ""
3992 * s.append_as_bytes(0, 257) # => "\u0000\u0001"
3993 * s.bytesize # => 2
3994 *
3995 * Related: see {Modifying}[rdoc-ref:String@Modifying].
3996 */
3997
3998VALUE
3999rb_str_append_as_bytes(int argc, VALUE *argv, VALUE str)
4000{
4001 long needed_capacity = 0;
4002 volatile VALUE t0;
4003 enum ruby_value_type *types = ALLOCV_N(enum ruby_value_type, t0, argc);
4004
4005 for (int index = 0; index < argc; index++) {
4006 VALUE obj = argv[index];
4007 enum ruby_value_type type = types[index] = rb_type(obj);
4008 switch (type) {
4009 case T_FIXNUM:
4010 case T_BIGNUM:
4011 needed_capacity++;
4012 break;
4013 case T_STRING:
4014 needed_capacity += RSTRING_LEN(obj);
4015 break;
4016 default:
4017 rb_raise(
4019 "wrong argument type %"PRIsVALUE" (expected String or Integer)",
4020 rb_obj_class(obj)
4021 );
4022 break;
4023 }
4024 }
4025
4026 str_ensure_available_capa(str, needed_capacity);
4027 char *sptr = RSTRING_END(str);
4028
4029 for (int index = 0; index < argc; index++) {
4030 VALUE obj = argv[index];
4031 enum ruby_value_type type = types[index];
4032 switch (type) {
4033 case T_FIXNUM:
4034 case T_BIGNUM: {
4035 argv[index] = obj = rb_int_and(obj, INT2FIX(0xff));
4036 char byte = (char)(NUM2INT(obj) & 0xFF);
4037 *sptr = byte;
4038 sptr++;
4039 break;
4040 }
4041 case T_STRING: {
4042 const char *ptr;
4043 long len;
4044 RSTRING_GETMEM(obj, ptr, len);
4045 memcpy(sptr, ptr, len);
4046 sptr += len;
4047 break;
4048 }
4049 default:
4050 rb_bug("append_as_bytes arguments should have been validated");
4051 }
4052 }
4053
4054 STR_SET_LEN(str, RSTRING_LEN(str) + needed_capacity);
4055 TERM_FILL(sptr, TERM_LEN(str)); /* sentinel */
4056
4057 int cr = ENC_CODERANGE(str);
4058 switch (cr) {
4059 case ENC_CODERANGE_7BIT: {
4060 for (int index = 0; index < argc; index++) {
4061 VALUE obj = argv[index];
4062 enum ruby_value_type type = types[index];
4063 switch (type) {
4064 case T_FIXNUM:
4065 case T_BIGNUM: {
4066 if (!ISASCII(NUM2INT(obj))) {
4067 goto clear_cr;
4068 }
4069 break;
4070 }
4071 case T_STRING: {
4072 if (ENC_CODERANGE(obj) != ENC_CODERANGE_7BIT) {
4073 goto clear_cr;
4074 }
4075 break;
4076 }
4077 default:
4078 rb_bug("append_as_bytes arguments should have been validated");
4079 }
4080 }
4081 break;
4082 }
4084 if (ENCODING_GET_INLINED(str) == ENCINDEX_ASCII_8BIT) {
4085 goto keep_cr;
4086 }
4087 else {
4088 goto clear_cr;
4089 }
4090 break;
4091 default:
4092 goto clear_cr;
4093 break;
4094 }
4095
4096 RB_GC_GUARD(t0);
4097
4098 clear_cr:
4099 // If no fast path was hit, we clear the coderange.
4100 // append_as_bytes is predominantly meant to be used in
4101 // buffering situation, hence it's likely the coderange
4102 // will never be scanned, so it's not worth spending time
4103 // precomputing the coderange except for simple and common
4104 // situations.
4106 keep_cr:
4107 return str;
4108}
4109
4110/*
4111 * call-seq:
4112 * self << object -> self
4113 *
4114 * Appends a string representation of +object+ to +self+;
4115 * returns +self+.
4116 *
4117 * If +object+ is a string, appends it to +self+:
4118 *
4119 * s = 'foo'
4120 * s << 'bar' # => "foobar"
4121 * s # => "foobar"
4122 *
4123 * If +object+ is an integer,
4124 * its value is considered a codepoint;
4125 * converts the value to a character before concatenating:
4126 *
4127 * s = 'foo'
4128 * s << 33 # => "foo!"
4129 *
4130 * Additionally, if the codepoint is in range <tt>0..0xff</tt>
4131 * and the encoding of +self+ is Encoding::US_ASCII,
4132 * changes the encoding to Encoding::ASCII_8BIT:
4133 *
4134 * s = 'foo'.encode(Encoding::US_ASCII)
4135 * s.encoding # => #<Encoding:US-ASCII>
4136 * s << 0xff # => "foo\xFF"
4137 * s.encoding # => #<Encoding:BINARY (ASCII-8BIT)>
4138 *
4139 * Raises RangeError if that codepoint is not representable in the encoding of +self+:
4140 *
4141 * s = 'foo'
4142 * s.encoding # => <Encoding:UTF-8>
4143 * s << 0x00110000 # 1114112 out of char range (RangeError)
4144 * s = 'foo'.encode(Encoding::EUC_JP)
4145 * s << 0x00800080 # invalid codepoint 0x800080 in EUC-JP (RangeError)
4146 *
4147 * Related: see {Modifying}[rdoc-ref:String@Modifying].
4148 */
4149VALUE
4151{
4152 unsigned int code;
4153 rb_encoding *enc = STR_ENC_GET(str1);
4154 int encidx;
4155
4156 if (RB_INTEGER_TYPE_P(str2)) {
4157 if (rb_num_to_uint(str2, &code) == 0) {
4158 }
4159 else if (FIXNUM_P(str2)) {
4160 rb_raise(rb_eRangeError, "%ld out of char range", FIX2LONG(str2));
4161 }
4162 else {
4163 rb_raise(rb_eRangeError, "bignum out of char range");
4164 }
4165 }
4166 else {
4167 return rb_str_append(str1, str2);
4168 }
4169
4170 encidx = rb_ascii8bit_appendable_encoding_index(enc, code);
4171
4172 if (encidx >= 0) {
4173 rb_str_buf_cat_byte(str1, (unsigned char)code);
4174 }
4175 else {
4176 long pos = RSTRING_LEN(str1);
4177 int cr = ENC_CODERANGE(str1);
4178 int len;
4179 char *buf;
4180
4181 switch (len = rb_enc_codelen(code, enc)) {
4182 case ONIGERR_INVALID_CODE_POINT_VALUE:
4183 rb_raise(rb_eRangeError, "invalid codepoint 0x%X in %s", code, rb_enc_name(enc));
4184 break;
4185 case ONIGERR_TOO_BIG_WIDE_CHAR_VALUE:
4186 case 0:
4187 rb_raise(rb_eRangeError, "%u out of char range", code);
4188 break;
4189 }
4190 buf = ALLOCA_N(char, len + 1);
4191 rb_enc_mbcput(code, buf, enc);
4192 if (rb_enc_precise_mbclen(buf, buf + len + 1, enc) != len) {
4193 rb_raise(rb_eRangeError, "invalid codepoint 0x%X in %s", code, rb_enc_name(enc));
4194 }
4195 rb_str_resize(str1, pos+len);
4196 memcpy(RSTRING_PTR(str1) + pos, buf, len);
4197 if (cr == ENC_CODERANGE_7BIT && code > 127) {
4199 }
4200 else if (cr == ENC_CODERANGE_BROKEN) {
4202 }
4203 ENC_CODERANGE_SET(str1, cr);
4204 }
4205 return str1;
4206}
4207
4208int
4209rb_ascii8bit_appendable_encoding_index(rb_encoding *enc, unsigned int code)
4210{
4211 int encidx = rb_enc_to_index(enc);
4212
4213 if (encidx == ENCINDEX_ASCII_8BIT || encidx == ENCINDEX_US_ASCII) {
4214 /* US-ASCII automatically extended to ASCII-8BIT */
4215 if (code > 0xFF) {
4216 rb_raise(rb_eRangeError, "%u out of char range", code);
4217 }
4218 if (encidx == ENCINDEX_US_ASCII && code > 127) {
4219 return ENCINDEX_ASCII_8BIT;
4220 }
4221 return encidx;
4222 }
4223 else {
4224 return -1;
4225 }
4226}
4227
4228/*
4229 * call-seq:
4230 * prepend(*other_strings) -> new_string
4231 *
4232 * Prefixes to +self+ the concatenation of the given +other_strings+; returns +self+:
4233 *
4234 * 'baz'.prepend('foo', 'bar') # => "foobarbaz"
4235 *
4236 * Related: see {Modifying}[rdoc-ref:String@Modifying].
4237 *
4238 */
4239
4240static VALUE
4241rb_str_prepend_multi(int argc, VALUE *argv, VALUE str)
4242{
4243 str_modifiable(str);
4244
4245 if (argc == 1) {
4246 rb_str_update(str, 0L, 0L, argv[0]);
4247 }
4248 else if (argc > 1) {
4249 int i;
4250 VALUE arg_str = rb_str_tmp_new(0);
4251 rb_enc_copy(arg_str, str);
4252 for (i = 0; i < argc; i++) {
4253 rb_str_append(arg_str, argv[i]);
4254 }
4255 rb_str_update(str, 0L, 0L, arg_str);
4256 }
4257
4258 return str;
4259}
4260
4261st_index_t
4263{
4264 if (FL_TEST_RAW(str, STR_PRECOMPUTED_HASH)) {
4265 st_index_t precomputed_hash;
4266 memcpy(&precomputed_hash, RSTRING_END(str) + TERM_LEN(str), sizeof(precomputed_hash));
4267
4268 RUBY_ASSERT(precomputed_hash == str_do_hash(str));
4269 return precomputed_hash;
4270 }
4271
4272 return str_do_hash(str);
4273}
4274
4275int
4277{
4278 long len1, len2;
4279 const char *ptr1, *ptr2;
4280 RSTRING_GETMEM(str1, ptr1, len1);
4281 RSTRING_GETMEM(str2, ptr2, len2);
4282 return (len1 != len2 ||
4283 !rb_str_comparable(str1, str2) ||
4284 memcmp(ptr1, ptr2, len1) != 0);
4285}
4286
4287/*
4288 * call-seq:
4289 * hash -> integer
4290 *
4291 * :include: doc/string/hash.rdoc
4292 *
4293 */
4294
4295static VALUE
4296rb_str_hash_m(VALUE str)
4297{
4298 st_index_t hval = rb_str_hash(str);
4299 return ST2FIX(hval);
4300}
4301
4302#define lesser(a,b) (((a)>(b))?(b):(a))
4303
4304int
4306{
4307 int idx1, idx2;
4308 int rc1, rc2;
4309
4310 if (RSTRING_LEN(str1) == 0) return TRUE;
4311 if (RSTRING_LEN(str2) == 0) return TRUE;
4312 idx1 = ENCODING_GET(str1);
4313 idx2 = ENCODING_GET(str2);
4314 if (idx1 == idx2) return TRUE;
4315 rc1 = rb_enc_str_coderange(str1);
4316 rc2 = rb_enc_str_coderange(str2);
4317 if (rc1 == ENC_CODERANGE_7BIT) {
4318 if (rc2 == ENC_CODERANGE_7BIT) return TRUE;
4319 if (rb_enc_asciicompat(rb_enc_from_index(idx2)))
4320 return TRUE;
4321 }
4322 if (rc2 == ENC_CODERANGE_7BIT) {
4323 if (rb_enc_asciicompat(rb_enc_from_index(idx1)))
4324 return TRUE;
4325 }
4326 return FALSE;
4327}
4328
4329int
4331{
4332 long len1, len2;
4333 const char *ptr1, *ptr2;
4334 int retval;
4335
4336 if (str1 == str2) return 0;
4337 RSTRING_GETMEM(str1, ptr1, len1);
4338 RSTRING_GETMEM(str2, ptr2, len2);
4339 if (ptr1 == ptr2 || (retval = memcmp(ptr1, ptr2, lesser(len1, len2))) == 0) {
4340 if (len1 == len2) {
4341 if (!rb_str_comparable(str1, str2)) {
4342 if (ENCODING_GET(str1) > ENCODING_GET(str2))
4343 return 1;
4344 return -1;
4345 }
4346 return 0;
4347 }
4348 if (len1 > len2) return 1;
4349 return -1;
4350 }
4351 if (retval > 0) return 1;
4352 return -1;
4353}
4354
4355/*
4356 * call-seq:
4357 * self == other -> true or false
4358 *
4359 * Returns whether +other+ is equal to +self+.
4360 *
4361 * When +other+ is a string, returns whether +other+ has the same length and content as +self+:
4362 *
4363 * s = 'foo'
4364 * s == 'foo' # => true
4365 * s == 'food' # => false
4366 * s == 'FOO' # => false
4367 *
4368 * Returns +false+ if the two strings' encodings are not compatible:
4369 *
4370 * "\u{e4 f6 fc}".encode(Encoding::ISO_8859_1) == ("\u{c4 d6 dc}") # => false
4371 *
4372 * When +other+ is not a string:
4373 *
4374 * - If +other+ responds to method <tt>to_str</tt>,
4375 * <tt>other == self</tt> is called and its return value is returned.
4376 * - If +other+ does not respond to <tt>to_str</tt>,
4377 * +false+ is returned.
4378 *
4379 * Related: {Comparing}[rdoc-ref:String@Comparing].
4380 */
4381
4382VALUE
4384{
4385 if (str1 == str2) return Qtrue;
4386 if (!RB_TYPE_P(str2, T_STRING)) {
4387 if (!rb_respond_to(str2, idTo_str)) {
4388 return Qfalse;
4389 }
4390 return rb_equal(str2, str1);
4391 }
4392 return rb_str_eql_internal(str1, str2);
4393}
4394
4395/*
4396 * call-seq:
4397 * eql?(object) -> true or false
4398 *
4399 * :include: doc/string/eql_p.rdoc
4400 *
4401 */
4402
4403VALUE
4404rb_str_eql(VALUE str1, VALUE str2)
4405{
4406 if (str1 == str2) return Qtrue;
4407 if (!RB_TYPE_P(str2, T_STRING)) return Qfalse;
4408 return rb_str_eql_internal(str1, str2);
4409}
4410
4411/*
4412 * call-seq:
4413 * self <=> other -> -1, 0, 1, or nil
4414 *
4415 * Compares +self+ and +other+,
4416 * evaluating their _contents_, not their _lengths_.
4417 *
4418 * Returns:
4419 *
4420 * - +-1+, if +self+ is smaller.
4421 * - +0+, if the two are equal.
4422 * - +1+, if +self+ is larger.
4423 * - +nil+, if the two are incomparable.
4424 *
4425 * Examples:
4426 *
4427 * 'a' <=> 'b' # => -1
4428 * 'a' <=> 'ab' # => -1
4429 * 'a' <=> 'a' # => 0
4430 * 'b' <=> 'a' # => 1
4431 * 'ab' <=> 'a' # => 1
4432 * 'a' <=> :a # => nil
4433 *
4434 * \Class \String includes module Comparable,
4435 * each of whose methods uses String#<=> for comparison.
4436 *
4437 * Related: see {Comparing}[rdoc-ref:String@Comparing].
4438 */
4439
4440static VALUE
4441rb_str_cmp_m(VALUE str1, VALUE str2)
4442{
4443 int result;
4444 VALUE s = rb_check_string_type(str2);
4445 if (NIL_P(s)) {
4446 return rb_invcmp(str1, str2);
4447 }
4448 result = rb_str_cmp(str1, s);
4449 return INT2FIX(result);
4450}
4451
4452static VALUE str_casecmp(VALUE str1, VALUE str2);
4453static VALUE str_casecmp_p(VALUE str1, VALUE str2);
4454
4455/*
4456 * call-seq:
4457 * casecmp(other_string) -> -1, 0, 1, or nil
4458 *
4459 * Ignoring case, compares +self+ and +other_string+; returns:
4460 *
4461 * - -1 if <tt>self.downcase</tt> is smaller than <tt>other_string.downcase</tt>.
4462 * - 0 if the two are equal.
4463 * - 1 if <tt>self.downcase</tt> is larger than <tt>other_string.downcase</tt>.
4464 * - +nil+ if the two are incomparable.
4465 *
4466 * See {Case Mapping}[rdoc-ref:case_mapping.rdoc].
4467 *
4468 * Examples:
4469 *
4470 * 'foo'.casecmp('goo') # => -1
4471 * 'goo'.casecmp('foo') # => 1
4472 * 'foo'.casecmp('food') # => -1
4473 * 'food'.casecmp('foo') # => 1
4474 * 'FOO'.casecmp('foo') # => 0
4475 * 'foo'.casecmp('FOO') # => 0
4476 * 'foo'.casecmp(1) # => nil
4477 *
4478 * Related: see {Comparing}[rdoc-ref:String@Comparing].
4479 */
4480
4481VALUE
4482rb_str_casecmp(VALUE str1, VALUE str2)
4483{
4484 VALUE s = rb_check_string_type(str2);
4485 if (NIL_P(s)) {
4486 return Qnil;
4487 }
4488 return str_casecmp(str1, s);
4489}
4490
4491static VALUE
4492str_casecmp(VALUE str1, VALUE str2)
4493{
4494 long len;
4495 rb_encoding *enc;
4496 const char *p1, *p1end, *p2, *p2end;
4497
4498 enc = rb_enc_compatible(str1, str2);
4499 if (!enc) {
4500 return Qnil;
4501 }
4502
4503 p1 = RSTRING_PTR(str1); p1end = RSTRING_END(str1);
4504 p2 = RSTRING_PTR(str2); p2end = RSTRING_END(str2);
4505 if (single_byte_optimizable(str1) && single_byte_optimizable(str2)) {
4506 while (p1 < p1end && p2 < p2end) {
4507 if (*p1 != *p2) {
4508 unsigned int c1 = TOLOWER(*p1 & 0xff);
4509 unsigned int c2 = TOLOWER(*p2 & 0xff);
4510 if (c1 != c2)
4511 return INT2FIX(c1 < c2 ? -1 : 1);
4512 }
4513 p1++;
4514 p2++;
4515 }
4516 }
4517 else {
4518 while (p1 < p1end && p2 < p2end) {
4519 int l1, c1 = rb_enc_ascget(p1, p1end, &l1, enc);
4520 int l2, c2 = rb_enc_ascget(p2, p2end, &l2, enc);
4521
4522 if (0 <= c1 && 0 <= c2) {
4523 c1 = TOLOWER(c1);
4524 c2 = TOLOWER(c2);
4525 if (c1 != c2)
4526 return INT2FIX(c1 < c2 ? -1 : 1);
4527 }
4528 else {
4529 int r;
4530 l1 = rb_enc_mbclen(p1, p1end, enc);
4531 l2 = rb_enc_mbclen(p2, p2end, enc);
4532 len = l1 < l2 ? l1 : l2;
4533 r = memcmp(p1, p2, len);
4534 if (r != 0)
4535 return INT2FIX(r < 0 ? -1 : 1);
4536 if (l1 != l2)
4537 return INT2FIX(l1 < l2 ? -1 : 1);
4538 }
4539 p1 += l1;
4540 p2 += l2;
4541 }
4542 }
4543 if (p1 == p1end && p2 == p2end) return INT2FIX(0);
4544 if (p1 == p1end) return INT2FIX(-1);
4545 return INT2FIX(1);
4546}
4547
4548/*
4549 * call-seq:
4550 * casecmp?(other_string) -> true, false, or nil
4551 *
4552 * Returns +true+ if +self+ and +other_string+ are equal after
4553 * Unicode case folding, +false+ if unequal, +nil+ if incomparable.
4554 *
4555 * See {Case Mapping}[rdoc-ref:case_mapping.rdoc].
4556 *
4557 * Examples:
4558 *
4559 * 'foo'.casecmp?('goo') # => false
4560 * 'goo'.casecmp?('foo') # => false
4561 * 'foo'.casecmp?('food') # => false
4562 * 'food'.casecmp?('foo') # => false
4563 * 'FOO'.casecmp?('foo') # => true
4564 * 'foo'.casecmp?('FOO') # => true
4565 * 'foo'.casecmp?(1) # => nil
4566 *
4567 * Related: see {Comparing}[rdoc-ref:String@Comparing].
4568 */
4569
4570static VALUE
4571rb_str_casecmp_p(VALUE str1, VALUE str2)
4572{
4573 VALUE s = rb_check_string_type(str2);
4574 if (NIL_P(s)) {
4575 return Qnil;
4576 }
4577 return str_casecmp_p(str1, s);
4578}
4579
4580static VALUE
4581str_casecmp_p(VALUE str1, VALUE str2)
4582{
4583 rb_encoding *enc;
4584 VALUE folded_str1, folded_str2;
4585 VALUE fold_opt = sym_fold;
4586
4587 enc = rb_enc_compatible(str1, str2);
4588 if (!enc) {
4589 return Qnil;
4590 }
4591
4592 if (is_ascii_string(str1) && is_ascii_string(str2)) {
4593 if (RSTRING_LEN(str1) != RSTRING_LEN(str2)) return Qfalse;
4594 const char *p1 = RSTRING_PTR(str1), *p1end = RSTRING_END(str1);
4595 const char *p2 = RSTRING_PTR(str2);
4596 while (p1 < p1end) {
4597 if (*p1 != *p2 && TOLOWER((unsigned char)*p1) != TOLOWER((unsigned char)*p2)) {
4598 return Qfalse;
4599 }
4600 p1++;
4601 p2++;
4602 }
4603 return Qtrue;
4604 }
4605
4606 folded_str1 = rb_str_downcase(1, &fold_opt, str1);
4607 folded_str2 = rb_str_downcase(1, &fold_opt, str2);
4608
4609 return rb_str_eql(folded_str1, folded_str2);
4610}
4611
4612static long
4613strseq_core(const char *str_ptr, const char *str_ptr_end, long str_len,
4614 const char *sub_ptr, long sub_len, long offset, rb_encoding *enc)
4615{
4616 const char *search_start = str_ptr;
4617 long pos, search_len = str_len - offset;
4618
4619 for (;;) {
4620 const char *t;
4621 pos = rb_memsearch(sub_ptr, sub_len, search_start, search_len, enc);
4622 if (pos < 0) return pos;
4623 t = rb_enc_right_char_head(search_start, search_start+pos, str_ptr_end, enc);
4624 if (t == search_start + pos) break;
4625 search_len -= t - search_start;
4626 if (search_len <= 0) return -1;
4627 offset += t - search_start;
4628 search_start = t;
4629 }
4630 return pos + offset;
4631}
4632
4633/* found index in byte */
4634#define rb_str_index(str, sub, offset) rb_strseq_index(str, sub, offset, 0)
4635#define rb_str_byteindex(str, sub, offset) rb_strseq_index(str, sub, offset, 1)
4636
4637static long
4638rb_strseq_index(VALUE str, VALUE sub, long offset, int in_byte)
4639{
4640 const char *str_ptr, *str_ptr_end, *sub_ptr;
4641 long str_len, sub_len;
4642 rb_encoding *enc;
4643
4644 enc = rb_enc_check(str, sub);
4645 if (is_broken_string(sub)) return -1;
4646
4647 str_ptr = RSTRING_PTR(str);
4648 str_ptr_end = RSTRING_END(str);
4649 str_len = RSTRING_LEN(str);
4650 sub_ptr = RSTRING_PTR(sub);
4651 sub_len = RSTRING_LEN(sub);
4652
4653 if (str_len < sub_len) return -1;
4654
4655 if (offset != 0) {
4656 long str_len_char, sub_len_char;
4657 int single_byte = single_byte_optimizable(str);
4658 str_len_char = (in_byte || single_byte) ? str_len : str_strlen(str, enc);
4659 sub_len_char = in_byte ? sub_len : str_strlen(sub, enc);
4660 if (offset < 0) {
4661 offset += str_len_char;
4662 if (offset < 0) return -1;
4663 }
4664 if (str_len_char - offset < sub_len_char) return -1;
4665 if (!in_byte) offset = str_offset(str_ptr, str_ptr_end, offset, enc, single_byte);
4666 str_ptr += offset;
4667 }
4668 if (sub_len == 0) return offset;
4669
4670 /* need proceed one character at a time */
4671 return strseq_core(str_ptr, str_ptr_end, str_len, sub_ptr, sub_len, offset, enc);
4672}
4673
4674
4675/*
4676 * call-seq:
4677 * index(pattern, offset = 0) -> integer or nil
4678 *
4679 * :include: doc/string/index.rdoc
4680 *
4681 */
4682
4683static VALUE
4684rb_str_index_m(int argc, VALUE *argv, VALUE str)
4685{
4686 VALUE sub;
4687 VALUE initpos;
4688 rb_encoding *enc = STR_ENC_GET(str);
4689 long pos;
4690
4691 if (rb_scan_args(argc, argv, "11", &sub, &initpos) == 2) {
4692 long slen = str_strlen(str, enc); /* str's enc */
4693 pos = NUM2LONG(initpos);
4694 if (pos < 0 ? (pos += slen) < 0 : pos > slen) {
4695 if (RB_TYPE_P(sub, T_REGEXP)) {
4697 }
4698 return Qnil;
4699 }
4700 }
4701 else {
4702 pos = 0;
4703 }
4704
4705 if (RB_TYPE_P(sub, T_REGEXP)) {
4706 pos = str_offset(RSTRING_PTR(str), RSTRING_END(str), pos,
4707 enc, single_byte_optimizable(str));
4708
4709 if (rb_reg_search(sub, str, pos, 0) >= 0) {
4710 VALUE match = rb_backref_get();
4711 pos = rb_str_sublen(str, RMATCH_BEG(match, 0));
4712 return LONG2NUM(pos);
4713 }
4714 }
4715 else {
4716 StringValue(sub);
4717 pos = rb_str_index(str, sub, pos);
4718 if (pos >= 0) {
4719 pos = rb_str_sublen(str, pos);
4720 return LONG2NUM(pos);
4721 }
4722 }
4723 return Qnil;
4724}
4725
4726/* Ensure that the given pos is a valid character boundary.
4727 * Note that in this function, "character" means a code point
4728 * (Unicode scalar value), not a grapheme cluster.
4729 */
4730static void
4731str_ensure_byte_pos(VALUE str, long pos)
4732{
4733 if (!single_byte_optimizable(str)) {
4734 const char *s = RSTRING_PTR(str);
4735 const char *e = RSTRING_END(str);
4736 const char *p = s + pos;
4737 if (!at_char_boundary(s, p, e, rb_enc_get(str))) {
4738 rb_raise(rb_eIndexError,
4739 "offset %ld does not land on character boundary", pos);
4740 }
4741 }
4742}
4743
4744/*
4745 * call-seq:
4746 * byteindex(object, offset = 0) -> integer or nil
4747 *
4748 * Returns the 0-based integer index of a substring of +self+
4749 * specified by +object+ (a string or Regexp) and +offset+,
4750 * or +nil+ if there is no such substring;
4751 * the returned index is the count of _bytes_ (not characters).
4752 *
4753 * When +object+ is a string,
4754 * returns the index of the first found substring equal to +object+:
4755 *
4756 * s = 'foo' # => "foo"
4757 * s.size # => 3 # Three 1-byte characters.
4758 * s.bytesize # => 3 # Three bytes.
4759 * s.byteindex('f') # => 0
4760 * s.byteindex('o') # => 1
4761 * s.byteindex('oo') # => 1
4762 * s.byteindex('ooo') # => nil
4763 *
4764 * When +object+ is a Regexp,
4765 * returns the index of the first found substring matching +object+;
4766 * updates {Regexp-related global variables}[rdoc-ref:Regexp@Global+Variables]:
4767 *
4768 * s = 'foo'
4769 * s.byteindex(/f/) # => 0
4770 * $~ # => #<MatchData "f">
4771 * s.byteindex(/o/) # => 1
4772 * s.byteindex(/oo/) # => 1
4773 * s.byteindex(/ooo/) # => nil
4774 * $~ # => nil
4775 *
4776 * \Integer argument +offset+, if given, specifies the 0-based index
4777 * of the byte where searching is to begin.
4778 *
4779 * When +offset+ is non-negative,
4780 * searching begins at byte position +offset+:
4781 *
4782 * s = 'foo'
4783 * s.byteindex('o', 1) # => 1
4784 * s.byteindex('o', 2) # => 2
4785 * s.byteindex('o', 3) # => nil
4786 *
4787 * When +offset+ is negative, counts backward from the end of +self+:
4788 *
4789 * s = 'foo'
4790 * s.byteindex('o', -1) # => 2
4791 * s.byteindex('o', -2) # => 1
4792 * s.byteindex('o', -3) # => 1
4793 * s.byteindex('o', -4) # => nil
4794 *
4795 * Raises IndexError if the byte at +offset+ is not the first byte of a character:
4796 *
4797 * s = "\uFFFF\uFFFF" # => "\uFFFF\uFFFF"
4798 * s.size # => 2 # Two 3-byte characters.
4799 * s.bytesize # => 6 # Six bytes.
4800 * s.byteindex("\uFFFF") # => 0
4801 * s.byteindex("\uFFFF", 1) # Raises IndexError
4802 * s.byteindex("\uFFFF", 2) # Raises IndexError
4803 * s.byteindex("\uFFFF", 3) # => 3
4804 * s.byteindex("\uFFFF", 4) # Raises IndexError
4805 * s.byteindex("\uFFFF", 5) # Raises IndexError
4806 * s.byteindex("\uFFFF", 6) # => nil
4807 *
4808 * Related: see {Querying}[rdoc-ref:String@Querying].
4809 */
4810
4811static VALUE
4812rb_str_byteindex_m(int argc, VALUE *argv, VALUE str)
4813{
4814 VALUE sub;
4815 VALUE initpos;
4816 long pos;
4817
4818 if (rb_scan_args(argc, argv, "11", &sub, &initpos) == 2) {
4819 pos = NUM2LONG(initpos);
4820 long slen = RSTRING_LEN(str);
4821 if (pos < 0 ? (pos += slen) < 0 : pos > slen) {
4822 if (RB_TYPE_P(sub, T_REGEXP)) {
4824 }
4825 return Qnil;
4826 }
4827 }
4828 else {
4829 pos = 0;
4830 }
4831
4832 str_ensure_byte_pos(str, pos);
4833
4834 if (RB_TYPE_P(sub, T_REGEXP)) {
4835 if (rb_reg_search(sub, str, pos, 0) >= 0) {
4836 VALUE match = rb_backref_get();
4837 pos = RMATCH_BEG(match, 0);
4838 return LONG2NUM(pos);
4839 }
4840 }
4841 else {
4842 StringValue(sub);
4843 pos = rb_str_byteindex(str, sub, pos);
4844 if (pos >= 0) return LONG2NUM(pos);
4845 }
4846 return Qnil;
4847}
4848
4849static long
4850str_rindex(VALUE str, VALUE sub, const char *s, rb_encoding *enc)
4851{
4852 const char *hit, *adjusted, *sbeg, *e, *t;
4853 int c;
4854 long slen, searchlen;
4855
4856 sbeg = RSTRING_PTR(str);
4857 slen = RSTRING_LEN(sub);
4858 if (slen == 0) return s - sbeg;
4859 e = RSTRING_END(str);
4860 t = RSTRING_PTR(sub);
4861 c = *t & 0xff;
4862 searchlen = s - sbeg + 1;
4863
4864 if (s + slen <= e && memcmp(s, t, slen) == 0) {
4865 return s - sbeg;
4866 }
4867
4868 do {
4869 hit = memrchr(sbeg, c, searchlen);
4870 if (!hit) break;
4871 adjusted = rb_enc_left_char_head(sbeg, hit, e, enc);
4872 if (hit != adjusted) {
4873 searchlen = adjusted - sbeg;
4874 continue;
4875 }
4876 if (hit + slen <= e && memcmp(hit, t, slen) == 0)
4877 return hit - sbeg;
4878 searchlen = adjusted - sbeg;
4879 } while (searchlen > 0);
4880
4881 return -1;
4882}
4883
4884/* found index in byte */
4885static long
4886rb_str_rindex(VALUE str, VALUE sub, long pos)
4887{
4888 long len, slen;
4889 const char *sbeg, *s;
4890 rb_encoding *enc;
4891 int singlebyte;
4892
4893 enc = rb_enc_check(str, sub);
4894 if (is_broken_string(sub)) return -1;
4895 singlebyte = single_byte_optimizable(str);
4896 len = singlebyte ? RSTRING_LEN(str) : str_strlen(str, enc); /* rb_enc_check */
4897 slen = str_strlen(sub, enc); /* rb_enc_check */
4898
4899 /* substring longer than string */
4900 if (len < slen) return -1;
4901 /* character counts, so the byte tail can still be shorter than sub */
4902 if (len - pos < slen) pos = len - slen;
4903 if (len == 0) return pos;
4904
4905 sbeg = RSTRING_PTR(str);
4906
4907 if (pos == 0) {
4908 if (RSTRING_LEN(sub) <= RSTRING_LEN(str) &&
4909 memcmp(sbeg, RSTRING_PTR(sub), RSTRING_LEN(sub)) == 0) {
4910 return 0;
4911 }
4912 else {
4913 return -1;
4914 }
4915 }
4916
4917 s = str_nth(sbeg, RSTRING_END(str), pos, enc, singlebyte);
4918 return str_rindex(str, sub, s, enc);
4919}
4920
4921/*
4922 * call-seq:
4923 * rindex(pattern, offset = self.length) -> integer or nil
4924 *
4925 * :include:doc/string/rindex.rdoc
4926 *
4927 */
4928
4929static VALUE
4930rb_str_rindex_m(int argc, VALUE *argv, VALUE str)
4931{
4932 VALUE sub;
4933 VALUE initpos;
4934 rb_encoding *enc = STR_ENC_GET(str);
4935 long pos, len = str_strlen(str, enc); /* str's enc */
4936
4937 if (rb_scan_args(argc, argv, "11", &sub, &initpos) == 2) {
4938 pos = NUM2LONG(initpos);
4939 if (pos < 0 && (pos += len) < 0) {
4940 if (RB_TYPE_P(sub, T_REGEXP)) {
4942 }
4943 return Qnil;
4944 }
4945 if (pos > len) pos = len;
4946 }
4947 else {
4948 pos = len;
4949 }
4950
4951 if (RB_TYPE_P(sub, T_REGEXP)) {
4952 /* enc = rb_enc_check(str, sub); */
4953 pos = str_offset(RSTRING_PTR(str), RSTRING_END(str), pos,
4954 enc, single_byte_optimizable(str));
4955
4956 if (rb_reg_search(sub, str, pos, 1) >= 0) {
4957 VALUE match = rb_backref_get();
4958 pos = rb_str_sublen(str, RMATCH_BEG(match, 0));
4959 return LONG2NUM(pos);
4960 }
4961 }
4962 else {
4963 StringValue(sub);
4964 pos = rb_str_rindex(str, sub, pos);
4965 if (pos >= 0) {
4966 pos = rb_str_sublen(str, pos);
4967 return LONG2NUM(pos);
4968 }
4969 }
4970 return Qnil;
4971}
4972
4973static long
4974rb_str_byterindex(VALUE str, VALUE sub, long pos)
4975{
4976 long len, slen;
4977 const char *sbeg, *s;
4978 rb_encoding *enc;
4979
4980 enc = rb_enc_check(str, sub);
4981 if (is_broken_string(sub)) return -1;
4982 len = RSTRING_LEN(str);
4983 slen = RSTRING_LEN(sub);
4984
4985 /* substring longer than string */
4986 if (len < slen) return -1;
4987 if (len - pos < slen) pos = len - slen;
4988 if (len == 0) return pos;
4989
4990 sbeg = RSTRING_PTR(str);
4991
4992 if (pos == 0) {
4993 if (memcmp(sbeg, RSTRING_PTR(sub), RSTRING_LEN(sub)) == 0)
4994 return 0;
4995 else
4996 return -1;
4997 }
4998
4999 s = sbeg + pos;
5000 return str_rindex(str, sub, s, enc);
5001}
5002
5003/*
5004 * call-seq:
5005 * byterindex(object, offset = self.bytesize) -> integer or nil
5006 *
5007 * Returns the 0-based integer index of a substring of +self+
5008 * that is the _last_ match for the given +object+ (a string or Regexp) and +offset+,
5009 * or +nil+ if there is no such substring;
5010 * the returned index is the count of _bytes_ (not characters).
5011 *
5012 * When +object+ is a string,
5013 * returns the index of the _last_ found substring equal to +object+:
5014 *
5015 * s = 'foo' # => "foo"
5016 * s.size # => 3 # Three 1-byte characters.
5017 * s.bytesize # => 3 # Three bytes.
5018 * s.byterindex('f') # => 0
5019 * s.byterindex('o') # => 2
5020 * s.byterindex('oo') # => 1
5021 * s.byterindex('ooo') # => nil
5022 *
5023 * When +object+ is a Regexp,
5024 * returns the index of the last found substring matching +object+;
5025 * updates {Regexp-related global variables}[rdoc-ref:Regexp@Global+Variables]:
5026 *
5027 * s = 'foo'
5028 * s.byterindex(/f/) # => 0
5029 * $~ # => #<MatchData "f">
5030 * s.byterindex(/o/) # => 2
5031 * s.byterindex(/oo/) # => 1
5032 * s.byterindex(/ooo/) # => nil
5033 * $~ # => nil
5034 *
5035 * The last match means starting at the possible last position,
5036 * not the last of the longest matches:
5037 *
5038 * s = 'foo'
5039 * s.byterindex(/o+/) # => 2
5040 * $~ #=> #<MatchData "o">
5041 *
5042 * To get the last longest match, use a negative lookbehind:
5043 *
5044 * s = 'foo'
5045 * s.byterindex(/(?<!o)o+/) # => 1
5046 * $~ # => #<MatchData "oo">
5047 *
5048 * Or use method #byteindex with negative lookahead:
5049 *
5050 * s = 'foo'
5051 * s.byteindex(/o+(?!.*o)/) # => 1
5052 * $~ #=> #<MatchData "oo">
5053 *
5054 * \Integer argument +offset+, if given, specifies the 0-based index
5055 * of the byte where searching is to end.
5056 *
5057 * When +offset+ is non-negative,
5058 * searching ends at byte position +offset+:
5059 *
5060 * s = 'foo'
5061 * s.byterindex('o', 0) # => nil
5062 * s.byterindex('o', 1) # => 1
5063 * s.byterindex('o', 2) # => 2
5064 * s.byterindex('o', 3) # => 2
5065 *
5066 * When +offset+ is negative, counts backward from the end of +self+:
5067 *
5068 * s = 'foo'
5069 * s.byterindex('o', -1) # => 2
5070 * s.byterindex('o', -2) # => 1
5071 * s.byterindex('o', -3) # => nil
5072 *
5073 * Raises IndexError if the byte at +offset+ is not the first byte of a character:
5074 *
5075 * s = "\uFFFF\uFFFF" # => "\uFFFF\uFFFF"
5076 * s.size # => 2 # Two 3-byte characters.
5077 * s.bytesize # => 6 # Six bytes.
5078 * s.byterindex("\uFFFF") # => 3
5079 * s.byterindex("\uFFFF", 1) # Raises IndexError
5080 * s.byterindex("\uFFFF", 2) # Raises IndexError
5081 * s.byterindex("\uFFFF", 3) # => 3
5082 * s.byterindex("\uFFFF", 4) # Raises IndexError
5083 * s.byterindex("\uFFFF", 5) # Raises IndexError
5084 * s.byterindex("\uFFFF", 6) # => nil
5085 *
5086 * Related: see {Querying}[rdoc-ref:String@Querying].
5087 */
5088
5089static VALUE
5090rb_str_byterindex_m(int argc, VALUE *argv, VALUE str)
5091{
5092 VALUE sub;
5093 VALUE initpos;
5094 long pos;
5095
5096 if (rb_scan_args(argc, argv, "11", &sub, &initpos) == 2) {
5097 pos = NUM2LONG(initpos);
5098 long len = RSTRING_LEN(str);
5099 if (pos < 0 && (pos += len) < 0) {
5100 if (RB_TYPE_P(sub, T_REGEXP)) {
5102 }
5103 return Qnil;
5104 }
5105 if (pos > len) pos = len;
5106 }
5107 else {
5108 pos = RSTRING_LEN(str);
5109 }
5110
5111 str_ensure_byte_pos(str, pos);
5112
5113 if (RB_TYPE_P(sub, T_REGEXP)) {
5114 if (rb_reg_search(sub, str, pos, 1) >= 0) {
5115 VALUE match = rb_backref_get();
5116 pos = RMATCH_BEG(match, 0);
5117 return LONG2NUM(pos);
5118 }
5119 }
5120 else {
5121 StringValue(sub);
5122 pos = rb_str_byterindex(str, sub, pos);
5123 if (pos >= 0) return LONG2NUM(pos);
5124 }
5125 return Qnil;
5126}
5127
5128/*
5129 * call-seq:
5130 * self =~ other -> integer or nil
5131 *
5132 * When +other+ is a Regexp:
5133 *
5134 * - Returns the integer index (in characters) of the first match
5135 * for +self+ and +other+, or +nil+ if none;
5136 * - Updates {Regexp-related global variables}[rdoc-ref:Regexp@Global+Variables].
5137 *
5138 * Examples:
5139 *
5140 * 'foo' =~ /f/ # => 0
5141 * $~ # => #<MatchData "f">
5142 * 'foo' =~ /o/ # => 1
5143 * $~ # => #<MatchData "o">
5144 * 'foo' =~ /x/ # => nil
5145 * $~ # => nil
5146 *
5147 * Note that <tt>string =~ regexp</tt> is different from <tt>regexp =~ string</tt>
5148 * (see Regexp#=~):
5149 *
5150 * number = nil
5151 * 'no. 9' =~ /(?<number>\d+)/ # => 4
5152 * number # => nil # Not assigned.
5153 * /(?<number>\d+)/ =~ 'no. 9' # => 4
5154 * number # => "9" # Assigned.
5155 *
5156 * When +other+ is not a Regexp, returns the value
5157 * returned by <tt>other =~ self</tt>.
5158 *
5159 * Related: see {Querying}[rdoc-ref:String@Querying].
5160 */
5161
5162static VALUE
5163rb_str_match(VALUE x, VALUE y)
5164{
5165 switch (OBJ_BUILTIN_TYPE(y)) {
5166 case T_STRING:
5167 rb_raise(rb_eTypeError, "type mismatch: String given");
5168
5169 case T_REGEXP:
5170 return rb_reg_match(y, x);
5171
5172 default:
5173 return rb_funcall(y, idEqTilde, 1, x);
5174 }
5175}
5176
5177
5178static VALUE get_pat(VALUE);
5179
5180
5181/*
5182 * call-seq:
5183 * match(pattern, offset = 0) -> matchdata or nil
5184 * match(pattern, offset = 0) {|matchdata| ... } -> object
5185 *
5186 * Creates a MatchData object based on +self+ and the given arguments;
5187 * updates {Regexp Global Variables}[rdoc-ref:Regexp@Global+Variables].
5188 *
5189 * - Computes +regexp+ by converting +pattern+ (if not already a Regexp).
5190 *
5191 * regexp = Regexp.new(pattern)
5192 *
5193 * - Calls <tt>regexp.match</tt> with +self+ to compute +matchdata+.
5194 * If +offset+ is given, it is also passed (see Regexp#match).
5195 *
5196 * With no block given, returns the computed +matchdata+ or +nil+:
5197 *
5198 * 'foo'.match('f') # => #<MatchData "f">
5199 * 'foo'.match('o') # => #<MatchData "o">
5200 * 'foo'.match('x') # => nil
5201 * 'foo'.match('f', 1) # => nil
5202 * 'foo'.match('o', 1) # => #<MatchData "o">
5203 *
5204 * With a block given and computed +matchdata+ non-nil, calls the block with +matchdata+;
5205 * returns the block's return value:
5206 *
5207 * 'foo'.match(/o/) {|matchdata| matchdata } # => #<MatchData "o">
5208 *
5209 * With a block given and +nil+ +matchdata+, does not call the block:
5210 *
5211 * 'foo'.match(/x/) {|matchdata| fail 'Cannot happen' } # => nil
5212 *
5213 * Related: see {Querying}[rdoc-ref:String@Querying].
5214 */
5215
5216static VALUE
5217rb_str_match_m(int argc, VALUE *argv, VALUE str)
5218{
5219 VALUE re, result;
5220 if (argc < 1)
5221 rb_check_arity(argc, 1, 2);
5222 re = argv[0];
5223 argv[0] = str;
5224 result = rb_funcallv(get_pat(re), rb_intern("match"), argc, argv);
5225 if (!NIL_P(result) && rb_block_given_p()) {
5226 return rb_yield(result);
5227 }
5228 return result;
5229}
5230
5231/*
5232 * call-seq:
5233 * match?(pattern, offset = 0) -> true or false
5234 *
5235 * Returns whether a match is found for +self+ and the given arguments;
5236 * does not update {Regexp Global Variables}[rdoc-ref:Regexp@Global+Variables].
5237 *
5238 * Computes +regexp+ by converting +pattern+ (if not already a Regexp):
5239 *
5240 * regexp = Regexp.new(pattern)
5241 *
5242 * The search for +regexp+ in +self+ begins at the given character +offset+.
5243 * Returns +true+ if a match is found, +false+ otherwise:
5244 *
5245 * 'foo'.match?(/o/) # => true
5246 * 'foo'.match?('o') # => true
5247 * 'foo'.match?(/x/) # => false
5248 * 'foo'.match?('f', 1) # => false
5249 * 'foo'.match?('o', 1) # => true
5250 *
5251 * Related: see {Querying}[rdoc-ref:String@Querying].
5252 */
5253
5254static VALUE
5255rb_str_match_m_p(int argc, VALUE *argv, VALUE str)
5256{
5257 VALUE re;
5258 rb_check_arity(argc, 1, 2);
5259 re = get_pat(argv[0]);
5260 return rb_reg_match_p(re, str, argc > 1 ? NUM2LONG(argv[1]) : 0);
5261}
5262
5263enum neighbor_char {
5264 NEIGHBOR_NOT_CHAR,
5265 NEIGHBOR_FOUND,
5266 NEIGHBOR_WRAPPED
5267};
5268
5269static enum neighbor_char
5270enc_succ_char(char *p, long len, rb_encoding *enc)
5271{
5272 long i;
5273 int l;
5274
5275 if (rb_enc_mbminlen(enc) > 1) {
5276 /* wchar, trivial case */
5277 int r = rb_enc_precise_mbclen(p, p + len, enc), c;
5278 if (!MBCLEN_CHARFOUND_P(r)) {
5279 return NEIGHBOR_NOT_CHAR;
5280 }
5281 c = rb_enc_mbc_to_codepoint(p, p + len, enc) + 1;
5282 l = rb_enc_code_to_mbclen(c, enc);
5283 if (!l) return NEIGHBOR_NOT_CHAR;
5284 if (l != len) return NEIGHBOR_WRAPPED;
5285 rb_enc_mbcput(c, p, enc);
5286 r = rb_enc_precise_mbclen(p, p + len, enc);
5287 if (!MBCLEN_CHARFOUND_P(r)) {
5288 return NEIGHBOR_NOT_CHAR;
5289 }
5290 return NEIGHBOR_FOUND;
5291 }
5292 while (1) {
5293 for (i = len-1; 0 <= i && (unsigned char)p[i] == 0xff; i--)
5294 p[i] = '\0';
5295 if (i < 0)
5296 return NEIGHBOR_WRAPPED;
5297 ++((unsigned char*)p)[i];
5298 l = rb_enc_precise_mbclen(p, p+len, enc);
5299 if (MBCLEN_CHARFOUND_P(l)) {
5300 l = MBCLEN_CHARFOUND_LEN(l);
5301 if (l == len) {
5302 return NEIGHBOR_FOUND;
5303 }
5304 else {
5305 memset(p+l, 0xff, len-l);
5306 }
5307 }
5308 if (MBCLEN_INVALID_P(l) && i < len-1) {
5309 long len2;
5310 int l2;
5311 for (len2 = len-1; 0 < len2; len2--) {
5312 l2 = rb_enc_precise_mbclen(p, p+len2, enc);
5313 if (!MBCLEN_INVALID_P(l2))
5314 break;
5315 }
5316 memset(p+len2+1, 0xff, len-(len2+1));
5317 }
5318 }
5319}
5320
5321static enum neighbor_char
5322enc_pred_char(char *p, long len, rb_encoding *enc)
5323{
5324 long i;
5325 int l;
5326 if (rb_enc_mbminlen(enc) > 1) {
5327 /* wchar, trivial case */
5328 int r = rb_enc_precise_mbclen(p, p + len, enc), c;
5329 if (!MBCLEN_CHARFOUND_P(r)) {
5330 return NEIGHBOR_NOT_CHAR;
5331 }
5332 c = rb_enc_mbc_to_codepoint(p, p + len, enc);
5333 if (!c) return NEIGHBOR_NOT_CHAR;
5334 --c;
5335 l = rb_enc_code_to_mbclen(c, enc);
5336 if (!l) return NEIGHBOR_NOT_CHAR;
5337 if (l != len) return NEIGHBOR_WRAPPED;
5338 rb_enc_mbcput(c, p, enc);
5339 r = rb_enc_precise_mbclen(p, p + len, enc);
5340 if (!MBCLEN_CHARFOUND_P(r)) {
5341 return NEIGHBOR_NOT_CHAR;
5342 }
5343 return NEIGHBOR_FOUND;
5344 }
5345 while (1) {
5346 for (i = len-1; 0 <= i && (unsigned char)p[i] == 0; i--)
5347 p[i] = '\xff';
5348 if (i < 0)
5349 return NEIGHBOR_WRAPPED;
5350 --((unsigned char*)p)[i];
5351 l = rb_enc_precise_mbclen(p, p+len, enc);
5352 if (MBCLEN_CHARFOUND_P(l)) {
5353 l = MBCLEN_CHARFOUND_LEN(l);
5354 if (l == len) {
5355 return NEIGHBOR_FOUND;
5356 }
5357 else {
5358 memset(p+l, 0, len-l);
5359 }
5360 }
5361 if (MBCLEN_INVALID_P(l) && i < len-1) {
5362 long len2;
5363 int l2;
5364 for (len2 = len-1; 0 < len2; len2--) {
5365 l2 = rb_enc_precise_mbclen(p, p+len2, enc);
5366 if (!MBCLEN_INVALID_P(l2))
5367 break;
5368 }
5369 memset(p+len2+1, 0, len-(len2+1));
5370 }
5371 }
5372}
5373
5374/*
5375 overwrite +p+ by succeeding letter in +enc+ and returns
5376 NEIGHBOR_FOUND or NEIGHBOR_WRAPPED.
5377 When NEIGHBOR_WRAPPED, carried-out letter is stored into carry.
5378 assuming each ranges are successive, and mbclen
5379 never change in each ranges.
5380 NEIGHBOR_NOT_CHAR is returned if invalid character or the range has only one
5381 character.
5382 */
5383static enum neighbor_char
5384enc_succ_alnum_char(char *p, long len, rb_encoding *enc, char *carry)
5385{
5386 enum neighbor_char ret;
5387 unsigned int c;
5388 int ctype;
5389 int range;
5390 char save[ONIGENC_CODE_TO_MBC_MAXLEN];
5391
5392 /* skip 03A2, invalid char between GREEK CAPITAL LETTERS */
5393 int try;
5394 const int max_gaps = 1;
5395
5396 c = rb_enc_mbc_to_codepoint(p, p+len, enc);
5397 if (rb_enc_isctype(c, ONIGENC_CTYPE_DIGIT, enc))
5398 ctype = ONIGENC_CTYPE_DIGIT;
5399 else if (rb_enc_isctype(c, ONIGENC_CTYPE_ALPHA, enc))
5400 ctype = ONIGENC_CTYPE_ALPHA;
5401 else
5402 return NEIGHBOR_NOT_CHAR;
5403
5404 MEMCPY(save, p, char, len);
5405 for (try = 0; try <= max_gaps; ++try) {
5406 ret = enc_succ_char(p, len, enc);
5407 if (ret == NEIGHBOR_FOUND) {
5408 c = rb_enc_mbc_to_codepoint(p, p+len, enc);
5409 if (rb_enc_isctype(c, ctype, enc))
5410 return NEIGHBOR_FOUND;
5411 }
5412 }
5413 MEMCPY(p, save, char, len);
5414 range = 1;
5415 while (1) {
5416 MEMCPY(save, p, char, len);
5417 ret = enc_pred_char(p, len, enc);
5418 if (ret == NEIGHBOR_FOUND) {
5419 c = rb_enc_mbc_to_codepoint(p, p+len, enc);
5420 if (!rb_enc_isctype(c, ctype, enc)) {
5421 MEMCPY(p, save, char, len);
5422 break;
5423 }
5424 }
5425 else {
5426 MEMCPY(p, save, char, len);
5427 break;
5428 }
5429 range++;
5430 }
5431 if (range == 1) {
5432 return NEIGHBOR_NOT_CHAR;
5433 }
5434
5435 if (ctype != ONIGENC_CTYPE_DIGIT) {
5436 MEMCPY(carry, p, char, len);
5437 return NEIGHBOR_WRAPPED;
5438 }
5439
5440 MEMCPY(carry, p, char, len);
5441 enc_succ_char(carry, len, enc);
5442 return NEIGHBOR_WRAPPED;
5443}
5444
5445
5446static VALUE str_succ(VALUE str);
5447
5448/*
5449 * call-seq:
5450 * succ -> new_str
5451 *
5452 * :include: doc/string/succ.rdoc
5453 *
5454 */
5455
5456VALUE
5458{
5459 VALUE str;
5460 str = rb_str_new(RSTRING_PTR(orig), RSTRING_LEN(orig));
5461 rb_enc_cr_str_copy_for_substr(str, orig);
5462 return str_succ(str);
5463}
5464
5465static VALUE
5466str_succ(VALUE str)
5467{
5468 rb_encoding *enc;
5469 char *sbeg, *s, *e, *last_alnum = 0;
5470 int found_alnum = 0;
5471 long l, slen;
5472 char carry[ONIGENC_CODE_TO_MBC_MAXLEN] = "\1";
5473 long carry_pos = 0, carry_len = 1;
5474 enum neighbor_char neighbor = NEIGHBOR_FOUND;
5475
5476 slen = RSTRING_LEN(str);
5477 if (slen == 0) return str;
5478
5479 enc = STR_ENC_GET(str);
5480 sbeg = RSTRING_PTR(str);
5481 s = e = sbeg + slen;
5482
5483 while ((s = rb_enc_prev_char(sbeg, s, e, enc)) != 0) {
5484 if (neighbor == NEIGHBOR_NOT_CHAR && last_alnum) {
5485 if (ISALPHA(*last_alnum) ? ISDIGIT(*s) :
5486 ISDIGIT(*last_alnum) ? ISALPHA(*s) : 0) {
5487 break;
5488 }
5489 }
5490 l = rb_enc_precise_mbclen(s, e, enc);
5491 if (!ONIGENC_MBCLEN_CHARFOUND_P(l)) continue;
5492 l = ONIGENC_MBCLEN_CHARFOUND_LEN(l);
5493 neighbor = enc_succ_alnum_char(s, l, enc, carry);
5494 switch (neighbor) {
5495 case NEIGHBOR_NOT_CHAR:
5496 continue;
5497 case NEIGHBOR_FOUND:
5498 return str;
5499 case NEIGHBOR_WRAPPED:
5500 last_alnum = s;
5501 break;
5502 }
5503 found_alnum = 1;
5504 carry_pos = s - sbeg;
5505 carry_len = l;
5506 }
5507 if (!found_alnum) { /* str contains no alnum */
5508 s = e;
5509 while ((s = rb_enc_prev_char(sbeg, s, e, enc)) != 0) {
5510 enum neighbor_char neighbor;
5511 char tmp[ONIGENC_CODE_TO_MBC_MAXLEN];
5512 l = rb_enc_precise_mbclen(s, e, enc);
5513 if (!ONIGENC_MBCLEN_CHARFOUND_P(l)) continue;
5514 l = ONIGENC_MBCLEN_CHARFOUND_LEN(l);
5515 MEMCPY(tmp, s, char, l);
5516 neighbor = enc_succ_char(tmp, l, enc);
5517 switch (neighbor) {
5518 case NEIGHBOR_FOUND:
5519 MEMCPY(s, tmp, char, l);
5520 return str;
5521 break;
5522 case NEIGHBOR_WRAPPED:
5523 MEMCPY(s, tmp, char, l);
5524 break;
5525 case NEIGHBOR_NOT_CHAR:
5526 break;
5527 }
5528 if (rb_enc_precise_mbclen(s, s+l, enc) != l) {
5529 /* wrapped to \0...\0. search next valid char. */
5530 enc_succ_char(s, l, enc);
5531 }
5532 if (!rb_enc_asciicompat(enc)) {
5533 MEMCPY(carry, s, char, l);
5534 carry_len = l;
5535 }
5536 carry_pos = s - sbeg;
5537 }
5539 }
5540 RESIZE_CAPA(str, slen + carry_len);
5541 sbeg = RSTRING_PTR(str);
5542 s = sbeg + carry_pos;
5543 memmove(s + carry_len, s, slen - carry_pos);
5544 memmove(s, carry, carry_len);
5545 slen += carry_len;
5546 STR_SET_LEN(str, slen);
5547 TERM_FILL(&sbeg[slen], rb_enc_mbminlen(enc));
5548 rb_enc_str_coderange(str);
5549 return str;
5550}
5551
5552
5553/*
5554 * call-seq:
5555 * succ! -> self
5556 *
5557 * Like String#succ, but modifies +self+ in place; returns +self+.
5558 *
5559 * Related: see {Modifying}[rdoc-ref:String@Modifying].
5560 */
5561
5562static VALUE
5563rb_str_succ_bang(VALUE str)
5564{
5565 rb_str_modify(str);
5566 str_succ(str);
5567 return str;
5568}
5569
5570static int
5571all_digits_p(const char *s, long len)
5572{
5573 while (len-- > 0) {
5574 if (!ISDIGIT(*s)) return 0;
5575 s++;
5576 }
5577 return 1;
5578}
5579
5580static int
5581str_upto_i(VALUE str, VALUE arg)
5582{
5583 rb_yield(str);
5584 return 0;
5585}
5586
5587/*
5588 * call-seq:
5589 * upto(other_string, exclusive = false) {|string| ... } -> self
5590 * upto(other_string, exclusive = false) -> new_enumerator
5591 *
5592 * :include: doc/string/upto.rdoc
5593 *
5594 */
5595
5596static VALUE
5597rb_str_upto(int argc, VALUE *argv, VALUE beg)
5598{
5599 VALUE end, exclusive;
5600
5601 rb_scan_args(argc, argv, "11", &end, &exclusive);
5602 RETURN_ENUMERATOR(beg, argc, argv);
5603 return rb_str_upto_each(beg, end, RTEST(exclusive), str_upto_i, Qnil);
5604}
5605
5606VALUE
5607rb_str_upto_each(VALUE beg, VALUE end, int excl, int (*each)(VALUE, VALUE), VALUE arg)
5608{
5609 VALUE current, after_end;
5610 ID succ;
5611 int n, ascii;
5612 rb_encoding *enc;
5613
5614 CONST_ID(succ, "succ");
5615 StringValue(end);
5616 enc = rb_enc_check(beg, end);
5617 ascii = (is_ascii_string(beg) && is_ascii_string(end));
5618 /* single character */
5619 if (RSTRING_LEN(beg) == 1 && RSTRING_LEN(end) == 1 && ascii) {
5620 char c = RSTRING_PTR(beg)[0];
5621 char e = RSTRING_PTR(end)[0];
5622
5623 if (c > e || (excl && c == e)) return beg;
5624 for (;;) {
5625 VALUE str = rb_enc_str_new(&c, 1, enc);
5627 if ((*each)(str, arg)) break;
5628 if (!excl && c == e) break;
5629 c++;
5630 if (excl && c == e) break;
5631 }
5632 return beg;
5633 }
5634 /* both edges are all digits */
5635 if (ascii && ISDIGIT(RSTRING_PTR(beg)[0]) && ISDIGIT(RSTRING_PTR(end)[0]) &&
5636 all_digits_p(RSTRING_PTR(beg), RSTRING_LEN(beg)) &&
5637 all_digits_p(RSTRING_PTR(end), RSTRING_LEN(end))) {
5638 VALUE b, e;
5639 int width;
5640
5641 width = RSTRING_LENINT(beg);
5642 b = rb_str_to_inum(beg, 10, FALSE);
5643 e = rb_str_to_inum(end, 10, FALSE);
5644 if (FIXNUM_P(b) && FIXNUM_P(e)) {
5645 long bi = FIX2LONG(b);
5646 long ei = FIX2LONG(e);
5647 rb_encoding *usascii = rb_usascii_encoding();
5648
5649 while (bi <= ei) {
5650 if (excl && bi == ei) break;
5651 if ((*each)(rb_enc_sprintf(usascii, "%.*ld", width, bi), arg)) break;
5652 bi++;
5653 }
5654 }
5655 else {
5656 ID op = excl ? '<' : idLE;
5657 VALUE args[2], fmt = rb_fstring_lit("%.*d");
5658
5659 args[0] = INT2FIX(width);
5660 while (rb_funcall(b, op, 1, e)) {
5661 args[1] = b;
5662 if ((*each)(rb_str_format(numberof(args), args, fmt), arg)) break;
5663 b = rb_funcallv(b, succ, 0, 0);
5664 }
5665 }
5666 return beg;
5667 }
5668 /* normal case */
5669 n = rb_str_cmp(beg, end);
5670 if (n > 0 || (excl && n == 0)) return beg;
5671
5672 after_end = rb_funcallv(end, succ, 0, 0);
5673 current = str_duplicate(rb_cString, beg);
5674 while (!rb_str_equal(current, after_end)) {
5675 VALUE next = Qnil;
5676 if (excl || !rb_str_equal(current, end))
5677 next = rb_funcallv(current, succ, 0, 0);
5678 if ((*each)(current, arg)) break;
5679 if (NIL_P(next)) break;
5680 current = next;
5681 StringValue(current);
5682 if (excl && rb_str_equal(current, end)) break;
5683 if (RSTRING_LEN(current) > RSTRING_LEN(end) || RSTRING_LEN(current) == 0)
5684 break;
5685 }
5686
5687 return beg;
5688}
5689
5690VALUE
5691rb_str_upto_endless_each(VALUE beg, int (*each)(VALUE, VALUE), VALUE arg)
5692{
5693 VALUE current;
5694 ID succ;
5695
5696 CONST_ID(succ, "succ");
5697 /* both edges are all digits */
5698 if (is_ascii_string(beg) && ISDIGIT(RSTRING_PTR(beg)[0]) &&
5699 all_digits_p(RSTRING_PTR(beg), RSTRING_LEN(beg))) {
5700 VALUE b, args[2], fmt = rb_fstring_lit("%.*d");
5701 int width = RSTRING_LENINT(beg);
5702 b = rb_str_to_inum(beg, 10, FALSE);
5703 if (FIXNUM_P(b)) {
5704 long bi = FIX2LONG(b);
5705 rb_encoding *usascii = rb_usascii_encoding();
5706
5707 while (FIXABLE(bi)) {
5708 if ((*each)(rb_enc_sprintf(usascii, "%.*ld", width, bi), arg)) break;
5709 bi++;
5710 }
5711 b = LONG2NUM(bi);
5712 }
5713 args[0] = INT2FIX(width);
5714 while (1) {
5715 args[1] = b;
5716 if ((*each)(rb_str_format(numberof(args), args, fmt), arg)) break;
5717 b = rb_funcallv(b, succ, 0, 0);
5718 }
5719 }
5720 /* normal case */
5721 current = str_duplicate(rb_cString, beg);
5722 while (1) {
5723 VALUE next = rb_funcallv(current, succ, 0, 0);
5724 if ((*each)(current, arg)) break;
5725 current = next;
5726 StringValue(current);
5727 if (RSTRING_LEN(current) == 0)
5728 break;
5729 }
5730
5731 return beg;
5732}
5733
5734static int
5735include_range_i(VALUE str, VALUE arg)
5736{
5737 VALUE *argp = (VALUE *)arg;
5738 if (!rb_equal(str, *argp)) return 0;
5739 *argp = Qnil;
5740 return 1;
5741}
5742
5743VALUE
5744rb_str_include_range_p(VALUE beg, VALUE end, VALUE val, VALUE exclusive)
5745{
5746 beg = rb_str_new_frozen(beg);
5747 StringValue(end);
5748 end = rb_str_new_frozen(end);
5749 if (NIL_P(val)) return Qfalse;
5750 val = rb_check_string_type(val);
5751 if (NIL_P(val)) return Qfalse;
5752 if (rb_enc_asciicompat(STR_ENC_GET(beg)) &&
5753 rb_enc_asciicompat(STR_ENC_GET(end)) &&
5754 rb_enc_asciicompat(STR_ENC_GET(val))) {
5755 const char *bp = RSTRING_PTR(beg);
5756 const char *ep = RSTRING_PTR(end);
5757 const char *vp = RSTRING_PTR(val);
5758 if (RSTRING_LEN(beg) == 1 && RSTRING_LEN(end) == 1) {
5759 if (RSTRING_LEN(val) == 0 || RSTRING_LEN(val) > 1)
5760 return Qfalse;
5761 else {
5762 char b = *bp;
5763 char e = *ep;
5764 char v = *vp;
5765
5766 if (ISASCII(b) && ISASCII(e) && ISASCII(v)) {
5767 if (b <= v && v < e) return Qtrue;
5768 return RBOOL(!RTEST(exclusive) && v == e);
5769 }
5770 }
5771 }
5772#if 0
5773 /* both edges are all digits */
5774 if (ISDIGIT(*bp) && ISDIGIT(*ep) &&
5775 all_digits_p(bp, RSTRING_LEN(beg)) &&
5776 all_digits_p(ep, RSTRING_LEN(end))) {
5777 /* TODO */
5778 }
5779#endif
5780 }
5781 rb_str_upto_each(beg, end, RTEST(exclusive), include_range_i, (VALUE)&val);
5782
5783 return RBOOL(NIL_P(val));
5784}
5785
5786static VALUE
5787rb_str_subpat(VALUE str, VALUE re, VALUE backref)
5788{
5789 if (rb_reg_search(re, str, 0, 0) >= 0) {
5790 VALUE match = rb_backref_get();
5791 int nth = rb_reg_backref_number(match, backref);
5792 return rb_reg_nth_match(nth, match);
5793 }
5794 return Qnil;
5795}
5796
5797static VALUE
5798rb_str_aref(VALUE str, VALUE indx)
5799{
5800 long idx;
5801
5802 if (FIXNUM_P(indx)) {
5803 idx = FIX2LONG(indx);
5804 }
5805 else if (RB_TYPE_P(indx, T_REGEXP)) {
5806 return rb_str_subpat(str, indx, INT2FIX(0));
5807 }
5808 else if (RB_TYPE_P(indx, T_STRING)) {
5809 if (rb_str_index(str, indx, 0) != -1)
5810 return str_duplicate(rb_cString, indx);
5811 return Qnil;
5812 }
5813 else {
5814 /* check if indx is Range */
5815 long beg, len = str_strlen(str, NULL);
5816 switch (rb_range_beg_len(indx, &beg, &len, len, 0)) {
5817 case Qfalse:
5818 break;
5819 case Qnil:
5820 return Qnil;
5821 default:
5822 return rb_str_substr(str, beg, len);
5823 }
5824 idx = NUM2LONG(indx);
5825 }
5826
5827 return str_substr(str, idx, 1, FALSE);
5828}
5829
5830
5831/*
5832 * call-seq:
5833 * self[offset] -> new_string or nil
5834 * self[offset, size] -> new_string or nil
5835 * self[range] -> new_string or nil
5836 * self[regexp, capture = 0] -> new_string or nil
5837 * self[substring] -> new_string or nil
5838 *
5839 * :include: doc/string/aref.rdoc
5840 *
5841 */
5842
5843static VALUE
5844rb_str_aref_m(int argc, VALUE *argv, VALUE str)
5845{
5846 if (argc == 2) {
5847 if (RB_TYPE_P(argv[0], T_REGEXP)) {
5848 return rb_str_subpat(str, argv[0], argv[1]);
5849 }
5850 else {
5851 return rb_str_substr_two_fixnums(str, argv[0], argv[1], TRUE);
5852 }
5853 }
5854 rb_check_arity(argc, 1, 2);
5855 return rb_str_aref(str, argv[0]);
5856}
5857
5858VALUE
5860{
5861 char *ptr = RSTRING_PTR(str);
5862 long olen = RSTRING_LEN(str), nlen;
5863
5864 str_modifiable(str);
5865 if (len > olen) len = olen;
5866 nlen = olen - len;
5867 if (str_embed_capa(str) >= nlen + TERM_LEN(str)) {
5868 char *oldptr = ptr;
5869 size_t old_capa = RSTRING(str)->as.heap.aux.capa + TERM_LEN(str);
5870 int fl = (int)(RBASIC(str)->flags & (STR_NOEMBED|STR_SHARED|STR_NOFREE));
5871 STR_SET_EMBED(str);
5872 ptr = RSTRING(str)->as.embed.ary;
5873 memmove(ptr, oldptr + len, nlen);
5874 if (fl == STR_NOEMBED) {
5875 SIZED_FREE_N(oldptr, old_capa);
5876 }
5877 }
5878 else {
5879 if (!STR_SHARED_P(str)) {
5880 VALUE shared = heap_str_make_shared(rb_obj_class(str), str, TERM_LEN(str));
5881 rb_enc_cr_str_exact_copy(shared, str);
5883 }
5884 ptr = RSTRING(str)->as.heap.ptr += len;
5885 }
5886 STR_SET_LEN(str, nlen);
5887
5888 if (!SHARABLE_MIDDLE_SUBSTRING) {
5889 TERM_FILL(ptr + nlen, TERM_LEN(str));
5890 }
5892 return str;
5893}
5894
5895static void
5896rb_str_update_1(VALUE str, long beg, long len, VALUE val, long vbeg, long vlen)
5897{
5898 char *sptr;
5899 long slen;
5900 int cr;
5901
5902 if (beg == 0 && vlen == 0) {
5903 rb_str_drop_bytes(str, len);
5904 return;
5905 }
5906
5907 str_modify_keep_cr(str);
5908 RSTRING_GETMEM(str, sptr, slen);
5909 if (len < vlen) {
5910 /* expand string */
5911 RESIZE_CAPA(str, slen + vlen - len);
5912 sptr = RSTRING_PTR(str);
5913 }
5914
5916 cr = rb_enc_str_coderange(val);
5917 else
5919
5920 if (vlen != len) {
5921 memmove(sptr + beg + vlen,
5922 sptr + beg + len,
5923 slen - (beg + len));
5924 }
5925 if (vlen < beg && len < 0) {
5926 MEMZERO(sptr + slen, char, -len);
5927 }
5928 if (vlen > 0) {
5929 memmove(sptr + beg, RSTRING_PTR(val) + vbeg, vlen);
5930 }
5931 slen += vlen - len;
5932 STR_SET_LEN(str, slen);
5933 TERM_FILL(&sptr[slen], TERM_LEN(str));
5934 ENC_CODERANGE_SET(str, cr);
5935}
5936
5937static inline void
5938rb_str_update_0(VALUE str, long beg, long len, VALUE val)
5939{
5940 rb_str_update_1(str, beg, len, val, 0, RSTRING_LEN(val));
5941}
5942
5943void
5944rb_str_update(VALUE str, long beg, long len, VALUE val)
5945{
5946 long slen;
5947 char *p, *e;
5948 rb_encoding *enc;
5949 int singlebyte = single_byte_optimizable(str);
5950 int cr;
5951
5952 if (len < 0) rb_raise(rb_eIndexError, "negative length %ld", len);
5953
5954 StringValue(val);
5955 enc = rb_enc_check(str, val);
5956 slen = str_strlen(str, enc); /* rb_enc_check */
5957
5958 if ((slen < beg) || ((beg < 0) && (beg + slen < 0))) {
5959 rb_raise(rb_eIndexError, "index %ld out of string", beg);
5960 }
5961 if (beg < 0) {
5962 beg += slen;
5963 }
5964 RUBY_ASSERT(beg >= 0);
5965 RUBY_ASSERT(beg <= slen);
5966
5967 if (len > slen - beg) {
5968 len = slen - beg;
5969 }
5970 p = str_nth(RSTRING_PTR(str), RSTRING_END(str), beg, enc, singlebyte);
5971 if (!p) p = RSTRING_END(str);
5972 e = str_nth(p, RSTRING_END(str), len, enc, singlebyte);
5973 if (!e) e = RSTRING_END(str);
5974 /* error check */
5975 beg = p - RSTRING_PTR(str); /* physical position */
5976 len = e - p; /* physical length */
5977 rb_str_update_0(str, beg, len, val);
5978 rb_enc_associate(str, enc);
5980 if (cr != ENC_CODERANGE_BROKEN)
5981 ENC_CODERANGE_SET(str, cr);
5982}
5983
5984static void
5985rb_str_subpat_set(VALUE str, VALUE re, VALUE backref, VALUE val)
5986{
5987 int nth;
5988 VALUE match;
5989 long start, end, len;
5990 rb_encoding *enc;
5991
5992 if (rb_reg_search(re, str, 0, 0) < 0) {
5993 rb_raise(rb_eIndexError, "regexp not matched");
5994 }
5995 match = rb_backref_get();
5996 nth = rb_reg_backref_number(match, backref);
5997 int num_regs = RMATCH_NREGS(match);
5998 if ((nth >= num_regs) || ((nth < 0) && (-nth >= num_regs))) {
5999 rb_raise(rb_eIndexError, "index %d out of regexp", nth);
6000 }
6001 if (nth < 0) {
6002 nth += num_regs;
6003 }
6004
6005 start = RMATCH_BEG(match, nth);
6006 if (start == -1) {
6007 rb_raise(rb_eIndexError, "regexp group %d not matched", nth);
6008 }
6009 end = RMATCH_END(match, nth);
6010 len = end - start;
6011
6012 StringValue(val);
6013 if (start + len > RSTRING_LEN(str)) {
6014 rb_raise(rb_eRuntimeError, "string modified");
6015 }
6016
6017 enc = rb_enc_check_str(str, val);
6018 rb_str_update_0(str, start, len, val);
6019 rb_enc_associate(str, enc);
6020}
6021
6022static VALUE
6023rb_str_aset(VALUE str, VALUE indx, VALUE val)
6024{
6025 long idx, beg;
6026
6027 switch (TYPE(indx)) {
6028 case T_REGEXP:
6029 rb_str_subpat_set(str, indx, INT2FIX(0), val);
6030 return val;
6031
6032 case T_STRING:
6033 beg = rb_str_index(str, indx, 0);
6034 if (beg < 0) {
6035 rb_raise(rb_eIndexError, "string not matched");
6036 }
6037 beg = rb_str_sublen(str, beg);
6038 rb_str_update(str, beg, str_strlen(indx, NULL), val);
6039 return val;
6040
6041 default:
6042 /* check if indx is Range */
6043 {
6044 long beg, len;
6045 if (rb_range_beg_len(indx, &beg, &len, str_strlen(str, NULL), 2)) {
6046 rb_str_update(str, beg, len, val);
6047 return val;
6048 }
6049 }
6050 /* FALLTHROUGH */
6051
6052 case T_FIXNUM:
6053 idx = NUM2LONG(indx);
6054 rb_str_update(str, idx, 1, val);
6055 return val;
6056 }
6057}
6058
6059/*
6060 * call-seq:
6061 * self[index] = other_string -> new_string
6062 * self[start, length] = other_string -> new_string
6063 * self[range] = other_string -> new_string
6064 * self[regexp, capture = 0] = other_string -> new_string
6065 * self[substring] = other_string -> new_string
6066 *
6067 * :include: doc/string/aset.rdoc
6068 *
6069 */
6070
6071static VALUE
6072rb_str_aset_m(int argc, VALUE *argv, VALUE str)
6073{
6074 if (argc == 3) {
6075 if (RB_TYPE_P(argv[0], T_REGEXP)) {
6076 rb_str_subpat_set(str, argv[0], argv[1], argv[2]);
6077 }
6078 else {
6079 rb_str_update(str, NUM2LONG(argv[0]), NUM2LONG(argv[1]), argv[2]);
6080 }
6081 return argv[2];
6082 }
6083 rb_check_arity(argc, 2, 3);
6084 return rb_str_aset(str, argv[0], argv[1]);
6085}
6086
6087/*
6088 * call-seq:
6089 * insert(offset, other_string) -> self
6090 *
6091 * :include: doc/string/insert.rdoc
6092 *
6093 */
6094
6095static VALUE
6096rb_str_insert(VALUE str, VALUE idx, VALUE str2)
6097{
6098 long pos = NUM2LONG(idx);
6099
6100 if (pos == -1) {
6101 return rb_str_append(str, str2);
6102 }
6103 else if (pos < 0) {
6104 pos++;
6105 }
6106 rb_str_update(str, pos, 0, str2);
6107 return str;
6108}
6109
6110
6111/*
6112 * call-seq:
6113 * slice!(index) -> new_string or nil
6114 * slice!(start, length) -> new_string or nil
6115 * slice!(range) -> new_string or nil
6116 * slice!(regexp, capture = 0) -> new_string or nil
6117 * slice!(substring) -> new_string or nil
6118 *
6119 * Like String#[] (and its alias String#slice), except that:
6120 *
6121 * - Performs substitutions in +self+ (not in a copy of +self+).
6122 * - Returns the removed substring if any modifications were made, +nil+ otherwise.
6123 *
6124 * A few examples:
6125 *
6126 * s = 'hello'
6127 * s.slice!('e') # => "e"
6128 * s # => "hllo"
6129 * s.slice!('e') # => nil
6130 * s # => "hllo"
6131 *
6132 * Related: see {Modifying}[rdoc-ref:String@Modifying].
6133 */
6134
6135static VALUE
6136rb_str_slice_bang(int argc, VALUE *argv, VALUE str)
6137{
6138 VALUE result = Qnil;
6139 VALUE indx;
6140 long beg, len = 1;
6141 char *p;
6142
6143 rb_check_arity(argc, 1, 2);
6144 str_modify_keep_cr(str);
6145 indx = argv[0];
6146 if (RB_TYPE_P(indx, T_REGEXP)) {
6147 if (rb_reg_search(indx, str, 0, 0) < 0) return Qnil;
6148 VALUE match = rb_backref_get();
6149 int num_regs = RMATCH_NREGS(match);
6150 int nth = 0;
6151 if (argc > 1 && (nth = rb_reg_backref_number(match, argv[1])) < 0) {
6152 if ((nth += num_regs) <= 0) return Qnil;
6153 }
6154 else if (nth >= num_regs) return Qnil;
6155 beg = RMATCH_BEG(match, nth);
6156 len = RMATCH_END(match, nth) - beg;
6157 /* Converting the backref may have modified the string. */
6158 if (beg > RSTRING_LEN(str)) return Qnil;
6159 if (len > RSTRING_LEN(str) - beg) len = RSTRING_LEN(str) - beg;
6160 goto subseq;
6161 }
6162 else if (argc == 2) {
6163 beg = NUM2LONG(indx);
6164 len = NUM2LONG(argv[1]);
6165 goto num_index;
6166 }
6167 else if (FIXNUM_P(indx)) {
6168 beg = FIX2LONG(indx);
6169 if (!(p = rb_str_subpos(str, beg, &len))) return Qnil;
6170 if (!len) return Qnil;
6171 beg = p - RSTRING_PTR(str);
6172 goto subseq;
6173 }
6174 else if (RB_TYPE_P(indx, T_STRING)) {
6175 beg = rb_str_index(str, indx, 0);
6176 if (beg == -1) return Qnil;
6177 len = RSTRING_LEN(indx);
6178 result = str_duplicate(rb_cString, indx);
6179 goto squash;
6180 }
6181 else {
6182 switch (rb_range_beg_len(indx, &beg, &len, str_strlen(str, NULL), 0)) {
6183 case Qnil:
6184 return Qnil;
6185 case Qfalse:
6186 beg = NUM2LONG(indx);
6187 if (!(p = rb_str_subpos(str, beg, &len))) return Qnil;
6188 if (!len) return Qnil;
6189 beg = p - RSTRING_PTR(str);
6190 goto subseq;
6191 default:
6192 goto num_index;
6193 }
6194 }
6195
6196 num_index:
6197 if (!(p = rb_str_subpos(str, beg, &len))) return Qnil;
6198 beg = p - RSTRING_PTR(str);
6199
6200 subseq:
6201 result = rb_str_new(RSTRING_PTR(str)+beg, len);
6202 rb_enc_cr_str_copy_for_substr(result, str);
6203
6204 squash:
6205 if (len > 0) {
6206 if (beg == 0) {
6207 rb_str_drop_bytes(str, len);
6208 }
6209 else {
6210 char *sptr = RSTRING_PTR(str);
6211 long slen = RSTRING_LEN(str);
6212 if (beg + len > slen) /* pathological check */
6213 len = slen - beg;
6214 memmove(sptr + beg,
6215 sptr + beg + len,
6216 slen - (beg + len));
6217 slen -= len;
6218 STR_SET_LEN(str, slen);
6219 TERM_FILL(&sptr[slen], TERM_LEN(str));
6220 }
6221 }
6222 return result;
6223}
6224
6225static VALUE
6226get_pat(VALUE pat)
6227{
6228 VALUE val;
6229
6230 switch (OBJ_BUILTIN_TYPE(pat)) {
6231 case T_REGEXP:
6232 return pat;
6233
6234 case T_STRING:
6235 break;
6236
6237 default:
6238 val = rb_check_string_type(pat);
6239 if (NIL_P(val)) {
6240 Check_Type(pat, T_REGEXP);
6241 }
6242 pat = val;
6243 }
6244
6245 return rb_reg_regcomp(pat);
6246}
6247
6248static VALUE
6249get_pat_quoted(VALUE pat, int check)
6250{
6251 VALUE val;
6252
6253 switch (OBJ_BUILTIN_TYPE(pat)) {
6254 case T_REGEXP:
6255 return pat;
6256
6257 case T_STRING:
6258 break;
6259
6260 default:
6261 val = rb_check_string_type(pat);
6262 if (NIL_P(val)) {
6263 Check_Type(pat, T_REGEXP);
6264 }
6265 pat = val;
6266 }
6267 if (check && is_broken_string(pat)) {
6268 rb_exc_raise(rb_reg_check_preprocess(pat));
6269 }
6270 return pat;
6271}
6272
6273static long
6274rb_pat_search0(VALUE pat, VALUE str, long pos, int set_backref_str, VALUE *match)
6275{
6276 if (BUILTIN_TYPE(pat) == T_STRING) {
6277 pos = rb_str_byteindex(str, pat, pos);
6278 if (set_backref_str) {
6279 if (pos >= 0) {
6280 str = rb_str_new_frozen_String(str);
6281 VALUE match_data = rb_backref_set_string(str, pos, RSTRING_LEN(pat));
6282 if (match) {
6283 *match = match_data;
6284 }
6285 }
6286 else {
6288 }
6289 }
6290 return pos;
6291 }
6292 else {
6293 return rb_reg_search0(pat, str, pos, 0, set_backref_str, match);
6294 }
6295}
6296
6297static long
6298rb_pat_search(VALUE pat, VALUE str, long pos, int set_backref_str)
6299{
6300 return rb_pat_search0(pat, str, pos, set_backref_str, NULL);
6301}
6302
6303
6304/*
6305 * call-seq:
6306 * sub!(pattern, replacement) -> self or nil
6307 * sub!(pattern) {|match| ... } -> self or nil
6308 *
6309 * Like String#sub, except that:
6310 *
6311 * - Changes are made to +self+, not to copy of +self+.
6312 * - Returns +self+ if any changes are made, +nil+ otherwise.
6313 *
6314 * Related: see {Modifying}[rdoc-ref:String@Modifying].
6315 */
6316
6317static VALUE
6318rb_str_sub_bang(int argc, VALUE *argv, VALUE str)
6319{
6320 VALUE pat, repl, hash = Qnil;
6321 int iter = 0;
6322 long plen;
6323 int min_arity = rb_block_given_p() ? 1 : 2;
6324 long beg;
6325
6326 rb_check_arity(argc, min_arity, 2);
6327 if (argc == 1) {
6328 iter = 1;
6329 }
6330 else {
6331 repl = argv[1];
6332 if (!RB_TYPE_P(repl, T_STRING)) {
6333 hash = rb_check_hash_type(repl);
6334 if (NIL_P(hash)) {
6335 StringValue(repl);
6336 }
6337 }
6338 }
6339
6340 pat = get_pat_quoted(argv[0], 1);
6341
6342 str_modifiable(str);
6343 beg = rb_pat_search(pat, str, 0, 1);
6344 if (beg >= 0) {
6345 rb_encoding *enc;
6346 int cr = ENC_CODERANGE(str);
6347 long beg0, end0;
6348 VALUE match, match0 = Qnil;
6349 char *p, *rp;
6350 long len, rlen;
6351
6352 match = rb_backref_get();
6353 if (RB_TYPE_P(pat, T_STRING)) {
6354 beg0 = beg;
6355 end0 = beg0 + RSTRING_LEN(pat);
6356 match0 = pat;
6357 }
6358 else {
6359 beg0 = RMATCH_BEG(match, 0);
6360 end0 = RMATCH_END(match, 0);
6361 if (iter) match0 = rb_reg_nth_match(0, match);
6362 }
6363
6364 if (iter || !NIL_P(hash)) {
6365 p = RSTRING_PTR(str); len = RSTRING_LEN(str);
6366
6367 if (iter) {
6368 repl = rb_obj_as_string(rb_yield(match0));
6369 }
6370 else {
6371 repl = rb_hash_aref(hash, rb_str_subseq(str, beg0, end0 - beg0));
6372 repl = rb_obj_as_string(repl);
6373 }
6374 str_mod_check(str, p, len);
6375 rb_check_frozen(str);
6376 }
6377 else {
6378 repl = rb_reg_regsub_match(repl, str, match);
6379 }
6380
6381 enc = rb_enc_compatible(str, repl);
6382 if (!enc) {
6383 rb_encoding *str_enc = STR_ENC_GET(str);
6384 p = RSTRING_PTR(str); len = RSTRING_LEN(str);
6385 if (coderange_scan(p, beg0, str_enc) != ENC_CODERANGE_7BIT ||
6386 coderange_scan(p+end0, len-end0, str_enc) != ENC_CODERANGE_7BIT) {
6387 rb_raise(rb_eEncCompatError, "incompatible character encodings: %s and %s",
6388 rb_enc_inspect_name(str_enc),
6389 rb_enc_inspect_name(STR_ENC_GET(repl)));
6390 }
6391 enc = STR_ENC_GET(repl);
6392 }
6393 rb_str_modify(str);
6394 rb_enc_associate(str, enc);
6396 int cr2 = ENC_CODERANGE(repl);
6397 if (cr2 == ENC_CODERANGE_BROKEN ||
6398 (cr == ENC_CODERANGE_VALID && cr2 == ENC_CODERANGE_7BIT))
6400 else
6401 cr = cr2;
6402 }
6403 plen = end0 - beg0;
6404 rlen = RSTRING_LEN(repl);
6405 len = RSTRING_LEN(str);
6406 if (rlen > plen) {
6407 RESIZE_CAPA(str, len + rlen - plen);
6408 }
6409 p = RSTRING_PTR(str);
6410 if (rlen != plen) {
6411 memmove(p + beg0 + rlen, p + beg0 + plen, len - beg0 - plen);
6412 }
6413 rp = RSTRING_PTR(repl);
6414 memmove(p + beg0, rp, rlen);
6415 len += rlen - plen;
6416 STR_SET_LEN(str, len);
6417 TERM_FILL(&RSTRING_PTR(str)[len], TERM_LEN(str));
6418 ENC_CODERANGE_SET(str, cr);
6419
6420 RB_GC_GUARD(match);
6421
6422 return str;
6423 }
6424 return Qnil;
6425}
6426
6427
6428/*
6429 * call-seq:
6430 * sub(pattern, replacement) -> new_string
6431 * sub(pattern) {|match| ... } -> new_string
6432 *
6433 * :include: doc/string/sub.rdoc
6434 */
6435
6436static VALUE
6437rb_str_sub(int argc, VALUE *argv, VALUE str)
6438{
6439 str = str_duplicate(rb_cString, str);
6440 rb_str_sub_bang(argc, argv, str);
6441 return str;
6442}
6443
6444static VALUE
6445str_gsub(int argc, VALUE *argv, VALUE str, int bang)
6446{
6447 VALUE pat, val = Qnil, repl, match0 = Qnil, dest, hash = Qnil, match = Qnil;
6448 long beg, beg0, end0;
6449 long offset, blen, slen, len, last;
6450 enum {STR, ITER, FAST_MAP, MAP} mode = STR;
6451 char *sp, *cp;
6452 int need_backref_str = -1;
6453 rb_encoding *str_enc;
6454
6455 switch (argc) {
6456 case 1:
6457 RETURN_ENUMERATOR(str, argc, argv);
6458 mode = ITER;
6459 break;
6460 case 2:
6461 repl = argv[1];
6462 if (!RB_TYPE_P(repl, T_STRING)) {
6463 hash = rb_check_hash_type(repl);
6464 if (NIL_P(hash)) {
6465 StringValue(repl);
6466 }
6467 else if (rb_hash_default_unredefined(hash) && !FL_TEST_RAW(hash, RHASH_PROC_DEFAULT)) {
6468 mode = FAST_MAP;
6469 }
6470 else {
6471 mode = MAP;
6472 }
6473 }
6474 break;
6475 default:
6476 rb_error_arity(argc, 1, 2);
6477 }
6478
6479 pat = get_pat_quoted(argv[0], 1);
6480 beg = rb_pat_search0(pat, str, 0, need_backref_str, &match);
6481
6482 if (beg < 0) {
6483 if (bang) return Qnil; /* no match, no substitution */
6484 return str_duplicate(rb_cString, str);
6485 }
6486 if (bang) str_modify_keep_cr(str);
6487
6488 offset = 0;
6489 blen = RSTRING_LEN(str) + 30; /* len + margin */
6490 dest = rb_str_buf_new(blen);
6491 sp = RSTRING_PTR(str);
6492 slen = RSTRING_LEN(str);
6493 cp = sp;
6494 str_enc = STR_ENC_GET(str);
6495 rb_enc_associate(dest, str_enc);
6496 ENC_CODERANGE_SET(dest, rb_enc_asciicompat(str_enc) ? ENC_CODERANGE_7BIT : ENC_CODERANGE_VALID);
6497
6498 do {
6499 if (RB_TYPE_P(pat, T_STRING)) {
6500 beg0 = beg;
6501 end0 = beg0 + RSTRING_LEN(pat);
6502 match0 = pat;
6503 }
6504 else {
6505 beg0 = RMATCH_BEG(match, 0);
6506 end0 = RMATCH_END(match, 0);
6507 if (mode == ITER) match0 = rb_reg_nth_match(0, match);
6508 }
6509
6510 if (mode != STR) {
6511 if (mode == ITER) {
6512 val = rb_obj_as_string(rb_yield(match0));
6513 }
6514 else {
6515 struct RString fake_str = {RBASIC_INIT};
6516 VALUE key;
6517 if (mode == FAST_MAP) {
6518 // It is safe to use a fake_str here because we established that it won't escape,
6519 // as it's only used for `rb_hash_aref` and we checked the hash doesn't have a
6520 // default proc.
6521 key = setup_fake_str(&fake_str, sp + beg0, end0 - beg0, ENCODING_GET_INLINED(str));
6522 }
6523 else {
6524 key = rb_str_subseq(str, beg0, end0 - beg0);
6525 }
6526 val = rb_hash_aref(hash, key);
6527 val = rb_obj_as_string(val);
6528 }
6529 str_mod_check(str, sp, slen);
6530 if (val == dest) { /* paranoid check [ruby-dev:24827] */
6531 rb_raise(rb_eRuntimeError, "block should not cheat");
6532 }
6533 }
6534 else if (need_backref_str) {
6535 val = rb_reg_regsub_match(repl, str, match);
6536 if (need_backref_str < 0) {
6537 need_backref_str = val != repl;
6538 }
6539 }
6540 else {
6541 val = repl;
6542 }
6543
6544 len = beg0 - offset; /* copy pre-match substr */
6545 if (len) {
6546 rb_enc_str_buf_cat(dest, cp, len, str_enc);
6547 }
6548
6549 rb_str_buf_append(dest, val);
6550
6551 last = offset;
6552 offset = end0;
6553 if (beg0 == end0) {
6554 /*
6555 * Always consume at least one character of the input string
6556 * in order to prevent infinite loops.
6557 */
6558 if (RSTRING_LEN(str) <= end0) break;
6559 len = rb_enc_fast_mbclen(RSTRING_PTR(str)+end0, RSTRING_END(str), str_enc);
6560 rb_enc_str_buf_cat(dest, RSTRING_PTR(str)+end0, len, str_enc);
6561 offset = end0 + len;
6562 }
6563 cp = RSTRING_PTR(str) + offset;
6564 if (offset > RSTRING_LEN(str)) break;
6565
6566 // In FAST_MAP and STR mode the backref can't escape so we can re-use the MatchData safely.
6567 if (mode != FAST_MAP && mode != STR) {
6568 match = Qnil;
6569 }
6570 beg = rb_pat_search0(pat, str, offset, need_backref_str, &match);
6571
6572 RB_GC_GUARD(match);
6573 } while (beg >= 0);
6574
6575 if (RSTRING_LEN(str) > offset) {
6576 rb_enc_str_buf_cat(dest, cp, RSTRING_LEN(str) - offset, str_enc);
6577 }
6578 rb_pat_search0(pat, str, last, 1, &match);
6579 if (bang) {
6580 str_shared_replace(str, dest);
6581 }
6582 else {
6583 str = dest;
6584 }
6585
6586 return str;
6587}
6588
6589
6590/*
6591 * call-seq:
6592 * gsub!(pattern, replacement) -> self or nil
6593 * gsub!(pattern) {|match| ... } -> self or nil
6594 * gsub!(pattern) -> an_enumerator
6595 *
6596 * Like String#gsub, except that:
6597 *
6598 * - Performs substitutions in +self+ (not in a copy of +self+).
6599 * - Returns +self+ if any substitutions were performed, +nil+ otherwise.
6600 *
6601 * Related: see {Modifying}[rdoc-ref:String@Modifying].
6602 */
6603
6604static VALUE
6605rb_str_gsub_bang(int argc, VALUE *argv, VALUE str)
6606{
6607 str_modifiable(str);
6608 return str_gsub(argc, argv, str, 1);
6609}
6610
6611
6612/*
6613 * call-seq:
6614 * gsub(pattern, replacement) -> new_string
6615 * gsub(pattern) {|match| ... } -> new_string
6616 * gsub(pattern) -> enumerator
6617 *
6618 * Returns a copy of +self+ with zero or more substrings replaced.
6619 *
6620 * Argument +pattern+ may be a string or a Regexp;
6621 * argument +replacement+ may be a string or a Hash.
6622 * Varying types for the argument values makes this method very versatile.
6623 *
6624 * Below are some simple examples;
6625 * for many more examples, see {Substitution Methods}[rdoc-ref:String@Substitution+Methods].
6626 *
6627 * With arguments +pattern+ and string +replacement+ given,
6628 * replaces each matching substring with the given +replacement+ string:
6629 *
6630 * s = 'abracadabra'
6631 * s.gsub('ab', 'AB') # => "ABracadABra"
6632 * s.gsub(/[a-c]/, 'X') # => "XXrXXXdXXrX"
6633 *
6634 * With arguments +pattern+ and hash +replacement+ given,
6635 * replaces each matching substring with a value from the given +replacement+ hash,
6636 * or removes it:
6637 *
6638 * h = {'a' => 'A', 'b' => 'B', 'c' => 'C'}
6639 * s.gsub(/[a-c]/, h) # => "ABrACAdABrA" # 'a', 'b', 'c' replaced.
6640 * s.gsub(/[a-d]/, h) # => "ABrACAABrA" # 'd' removed.
6641 *
6642 * With argument +pattern+ and a block given,
6643 * calls the block with each matching substring;
6644 * replaces that substring with the block's return value:
6645 *
6646 * s.gsub(/[a-d]/) {|substring| substring.upcase }
6647 * # => "ABrACADABrA"
6648 *
6649 * With argument +pattern+ and no block given,
6650 * returns a new Enumerator.
6651 *
6652 * Related: see {Converting to New String}[rdoc-ref:String@Converting+to+New+String].
6653 */
6654
6655static VALUE
6656rb_str_gsub(int argc, VALUE *argv, VALUE str)
6657{
6658 return str_gsub(argc, argv, str, 0);
6659}
6660
6661
6662/*
6663 * call-seq:
6664 * replace(other_string) -> self
6665 *
6666 * Replaces the contents of +self+ with the contents of +other_string+;
6667 * returns +self+:
6668 *
6669 * s = 'foo' # => "foo"
6670 * s.replace('bar') # => "bar"
6671 *
6672 * Related: see {Modifying}[rdoc-ref:String@Modifying].
6673 */
6674
6675VALUE
6677{
6678 str_modifiable(str);
6679 if (str == str2) return str;
6680
6681 StringValue(str2);
6682 str_discard(str);
6683 return str_replace(str, str2);
6684}
6685
6686/*
6687 * call-seq:
6688 * clear -> self
6689 *
6690 * Removes the contents of +self+:
6691 *
6692 * s = 'foo'
6693 * s.clear # => ""
6694 * s # => ""
6695 *
6696 * Related: see {Modifying}[rdoc-ref:String@Modifying].
6697 */
6698
6699static VALUE
6700rb_str_clear(VALUE str)
6701{
6702 str_discard(str);
6703 STR_SET_EMBED(str);
6704 STR_SET_LEN(str, 0);
6705 RSTRING_PTR(str)[0] = 0;
6706 if (rb_enc_asciicompat(STR_ENC_GET(str)))
6708 else
6710 return str;
6711}
6712
6713/*
6714 * call-seq:
6715 * chr -> string
6716 *
6717 * :include: doc/string/chr.rdoc
6718 *
6719 */
6720
6721static VALUE
6722rb_str_chr(VALUE str)
6723{
6724 return rb_str_substr(str, 0, 1);
6725}
6726
6727/*
6728 * call-seq:
6729 * getbyte(index) -> integer or nil
6730 *
6731 * :include: doc/string/getbyte.rdoc
6732 *
6733 */
6734VALUE
6735rb_str_getbyte(VALUE str, VALUE index)
6736{
6737 long pos = NUM2LONG(index);
6738
6739 if (pos < 0)
6740 pos += RSTRING_LEN(str);
6741 if (pos < 0 || RSTRING_LEN(str) <= pos)
6742 return Qnil;
6743
6744 return INT2FIX((unsigned char)RSTRING_PTR(str)[pos]);
6745}
6746
6747/*
6748 * call-seq:
6749 * setbyte(index, integer) -> integer
6750 *
6751 * Sets the byte at zero-based offset +index+ to the value of the given +integer+;
6752 * returns +integer+:
6753 *
6754 * s = 'xyzzy'
6755 * s.setbyte(2, 129) # => 129
6756 * s # => "xy\x81zy"
6757 *
6758 * Related: see {Modifying}[rdoc-ref:String@Modifying].
6759 */
6760VALUE
6761rb_str_setbyte(VALUE str, VALUE index, VALUE value)
6762{
6763 long pos = NUM2LONG(index);
6764 char *ptr, *head, *left = 0;
6765 rb_encoding *enc;
6766 int cr = ENC_CODERANGE_UNKNOWN, width, nlen;
6767
6768 VALUE v = rb_to_int(value);
6769 VALUE w = rb_int_and(v, INT2FIX(0xff));
6770 char byte = (char)(NUM2INT(w) & 0xFF);
6771
6772 long len = RSTRING_LEN(str);
6773 if (pos < -len || len <= pos)
6774 rb_raise(rb_eIndexError, "index %ld out of string", pos);
6775 if (pos < 0)
6776 pos += len;
6777
6778 if (!str_independent(str))
6779 str_make_independent(str);
6780 enc = STR_ENC_GET(str);
6781 head = RSTRING_PTR(str);
6782 ptr = &head[pos];
6783 if (!STR_EMBED_P(str)) {
6784 cr = ENC_CODERANGE(str);
6785 switch (cr) {
6786 case ENC_CODERANGE_7BIT:
6787 left = ptr;
6788 *ptr = byte;
6789 if (ISASCII(byte)) goto end;
6790 nlen = rb_enc_precise_mbclen(left, head+len, enc);
6791 if (!MBCLEN_CHARFOUND_P(nlen))
6793 else
6795 goto end;
6797 left = rb_enc_left_char_head(head, ptr, head+len, enc);
6798 width = rb_enc_precise_mbclen(left, head+len, enc);
6799 *ptr = byte;
6800 nlen = rb_enc_precise_mbclen(left, head+len, enc);
6801 if (!MBCLEN_CHARFOUND_P(nlen))
6803 else if (MBCLEN_CHARFOUND_LEN(nlen) != width || ISASCII(byte))
6805 goto end;
6806 }
6807 }
6809 *ptr = byte;
6810
6811 end:
6812 return value;
6813}
6814
6815static inline bool
6816str_bit_offset_out_of_range(long byte_len, uint64_t bit_offset)
6817{
6818 /* Compare byte indexes to avoid overflowing byte_len * CHAR_BIT. */
6819 return bit_offset / CHAR_BIT >= (uint64_t)byte_len;
6820}
6821
6822/*
6823 * Keep both the full bit offset and its long representation. Most calls use a
6824 * Fixnum-sized offset and can stay on the original long fast path; only large
6825 * Bignum offsets need the uint64_t path below. This matters on platforms
6826 * where long is narrower than the address space, such as 32-bit and LLP64.
6827 */
6829 uint64_t value;
6830 long long_value;
6831 bool fits_long;
6832};
6833
6834static inline struct str_bit_offset
6835str_bit_offset_from_index(VALUE index)
6836{
6837 VALUE integer = rb_to_int(index);
6838 struct str_bit_offset offset;
6839
6840 /*
6841 * FIXNUM_P only decides whether the common long path is immediately usable.
6842 * This covers practically all offsets on LP64 platforms; Bignum offsets
6843 * are still accepted below when they fit in uint64_t, mainly for platforms
6844 * with 32-bit long where large strings can have Bignum bit offsets.
6845 */
6846 if (FIXNUM_P(integer)) {
6847 offset.long_value = FIX2LONG(integer);
6848 if (offset.long_value < 0) {
6849 rb_raise(rb_eIndexError, "bit index out of range");
6850 }
6851 offset.value = (uint64_t)offset.long_value;
6852 offset.fits_long = true;
6853 return offset;
6854 }
6855
6856 RUBY_ASSERT(RB_TYPE_P(integer, T_BIGNUM));
6857 if (rb_int_negative_p(integer)) {
6858 rb_raise(rb_eIndexError, "bit index out of range");
6859 }
6860 if (rb_cmpint(rb_int_cmp(integer, ULL2NUM(UINT64_MAX)), integer, ULL2NUM(UINT64_MAX)) > 0) {
6861 rb_raise(rb_eArgError, "bit index out of representable range");
6862 }
6863
6864 offset.value = (uint64_t)NUM2ULL(integer);
6865 if (offset.value <= (uint64_t)LONG_MAX) {
6866 offset.long_value = (long)offset.value;
6867 offset.fits_long = true;
6868 }
6869 else {
6870 offset.long_value = 0;
6871 offset.fits_long = false;
6872 }
6873 return offset;
6874}
6875
6876/*
6877 * Bit lengths share the offset's representable range.
6878 * A negative length is an ArgumentError rather than an IndexError.
6879 */
6880static uint64_t
6881str_bit_length_from_index(VALUE index)
6882{
6883 VALUE integer = rb_to_int(index);
6884
6885 if (FIXNUM_P(integer)) {
6886 long value = FIX2LONG(integer);
6887 if (value < 0) {
6888 rb_raise(rb_eArgError, "negative bit length");
6889 }
6890 return (uint64_t)value;
6891 }
6892
6893 RUBY_ASSERT(RB_TYPE_P(integer, T_BIGNUM));
6894 if (rb_int_negative_p(integer)) {
6895 rb_raise(rb_eArgError, "negative bit length");
6896 }
6897 if (rb_cmpint(rb_int_cmp(integer, ULL2NUM(UINT64_MAX)), integer, ULL2NUM(UINT64_MAX)) > 0) {
6898 rb_raise(rb_eArgError, "bit length out of representable range");
6899 }
6900 return (uint64_t)NUM2ULL(integer);
6901}
6902
6903static inline uint64_t
6904str_bit_size(long byte_len)
6905{
6906 /*
6907 * byte_len * CHAR_BIT overflows uint64_t only for byte_len >= 2**61 which cannot
6908 * be allocated. Saturate so that unreachable cases cannot wrap.
6909 */
6910 if ((uint64_t)byte_len > UINT64_MAX / CHAR_BIT) return UINT64_MAX;
6911 return (uint64_t)byte_len * CHAR_BIT;
6912}
6913
6915 uint64_t beg;
6916 uint64_t end_exclusive; /* meaningful only when end_open is false */
6917 bool end_open; /* a nil end: the region runs to the end of self */
6918};
6919
6920/*
6921 * Coerce a bit Range's endpoints to bit offsets. This may run arbitrary Ruby
6922 * (Integer#to_int on the endpoints), so it does NOT read the string's length:
6923 * The caller must resolve the length only after this returns, otherwise
6924 * to_int that reallocates self would leave a stale size.
6925 */
6926static void
6927str_bit_range_to_offsets(VALUE range, struct str_bit_range *out)
6928{
6929 VALUE beg_v, end_v;
6930 int excl;
6931
6932 /*
6933 * We don't use rb_range_beg_len: it counts negative endpoints from the end,
6934 * which is an IndexError for bit positions, and it is limited to long instead
6935 * of uint64_t.
6936 */
6937 rb_range_values(range, &beg_v, &end_v, &excl);
6938
6939 out->beg = NIL_P(beg_v) ? 0 : str_bit_offset_from_index(beg_v).value;
6940 if (NIL_P(end_v)) {
6941 out->end_open = true;
6942 out->end_exclusive = 0;
6943 }
6944 else {
6945 uint64_t end = str_bit_offset_from_index(end_v).value;
6946 out->end_open = false;
6947 /*
6948 * The saturation loses one position only for an inclusive end of
6949 * 2**64-1, which lies beyond any real string either way.
6950 */
6951 out->end_exclusive = (excl || end == UINT64_MAX) ? end : end + 1;
6952 }
6953}
6954
6955/*
6956 * Turn a coerced Range into (beg, len) against the now-current total bit size.
6957 * The length is deliberately not clamped to the bits available, so a reading
6958 * caller can clamp while a writing caller detects the overrun and raises.
6959 */
6960static bool
6961str_bit_range_resolve(const struct str_bit_range *range, uint64_t total_bits, uint64_t *begp, uint64_t *lenp)
6962{
6963 uint64_t beg = range->beg;
6964 if (beg > total_bits) return false;
6965
6966 uint64_t end_exclusive = range->end_open ? total_bits : range->end_exclusive;
6967 if (end_exclusive < beg) end_exclusive = beg;
6968
6969 *begp = beg;
6970 *lenp = end_exclusive - beg;
6971 return true;
6972}
6973
6974static bool
6975str_lsb_first_from_opts(VALUE opts)
6976{
6977 static ID keywords[1];
6978 VALUE vlsb_first;
6979
6980 if (!keywords[0]) {
6981 keywords[0] = rb_intern_const("lsb_first");
6982 }
6983
6984 rb_get_kwargs(opts, keywords, 0, 1, &vlsb_first);
6985 if (vlsb_first == Qundef || vlsb_first == Qtrue) {
6986 return true;
6987 }
6988 if (vlsb_first == Qfalse) {
6989 return false;
6990 }
6991 rb_raise(rb_eArgError, "lsb_first must be true or false");
6992 UNREACHABLE_RETURN(false);
6993}
6994
6995static bool
6996str_lsb_first(int argc, VALUE *argv, VALUE *index)
6997{
6998 VALUE opts;
6999
7000 rb_scan_args(argc, argv, "1:", index, &opts);
7001 return str_lsb_first_from_opts(opts);
7002}
7003
7004static inline uint64_t
7005str_logical_to_physical_bit64(uint64_t logical, bool lsb_first)
7006{
7007 return lsb_first ? logical : ((logical & ~(uint64_t)7) | (7 - (logical & 7)));
7008}
7009
7010static inline long
7011str_logical_to_physical_bit(long logical, bool lsb_first)
7012{
7013 return lsb_first ? logical : ((logical & ~7L) | (7 - (logical & 7L)));
7014}
7015
7017 long byte_index;
7018 unsigned int bit_offset;
7019};
7020
7021static inline struct str_bit_location
7022str_bit_location_from_offset(uint64_t logical, bool lsb_first)
7023{
7024 /*
7025 * When long is 32-bit, a bit offset for a large string can be a Bignum
7026 * while the byte index still fits in long, which is RSTRING_LEN's type.
7027 */
7028 uint64_t physical = str_logical_to_physical_bit64(logical, lsb_first);
7029 struct str_bit_location location;
7030 location.byte_index = (long)(physical / CHAR_BIT);
7031 location.bit_offset = (unsigned int)(physical % CHAR_BIT);
7032 return location;
7033}
7034
7035static inline int
7036str_get_bit(const char *ptr, long bit_index)
7037{
7038 return (((unsigned char)ptr[bit_index / CHAR_BIT]) >> (bit_index % CHAR_BIT)) & 1;
7039}
7040
7041static inline int
7042str_get_bit_location(const char *ptr, struct str_bit_location location)
7043{
7044 return (((unsigned char)ptr[location.byte_index]) >> location.bit_offset) & 1;
7045}
7046
7047static int
7048str_bit_get(int argc, VALUE *argv, VALUE str)
7049{
7050 VALUE index;
7051 bool lsb_first = str_lsb_first(argc, argv, &index);
7052 struct str_bit_offset offset = str_bit_offset_from_index(index);
7053
7054 if (str_bit_offset_out_of_range(RSTRING_LEN(str), offset.value)) {
7055 return -1;
7056 }
7057
7058 if (offset.fits_long) {
7059 return str_get_bit(RSTRING_PTR(str), str_logical_to_physical_bit(offset.long_value, lsb_first));
7060 }
7061 else {
7062 return str_get_bit_location(RSTRING_PTR(str), str_bit_location_from_offset(offset.value, lsb_first));
7063 }
7064}
7065
7066/*
7067 * call-seq:
7068 * bit_get(offset, lsb_first: true) -> 0, 1, or nil
7069 *
7070 * :include: doc/string/bit_get.rdoc
7071 *
7072 */
7073static VALUE
7074rb_str_bit_get(int argc, VALUE *argv, VALUE str)
7075{
7076 int bit = str_bit_get(argc, argv, str);
7077 return bit < 0 ? Qnil : INT2FIX(bit);
7078}
7079
7080/*
7081 * call-seq:
7082 * bit_set?(offset, lsb_first: true) -> true, false, or nil
7083 *
7084 * :include: doc/string/bit_set_p.rdoc
7085 *
7086 */
7087static VALUE
7088rb_str_bit_set_p(int argc, VALUE *argv, VALUE str)
7089{
7090 int bit = str_bit_get(argc, argv, str);
7091 return bit < 0 ? Qnil : RBOOL(bit);
7092}
7093
7094enum str_bit_mutation {
7095 STR_BIT_SET,
7096 STR_BIT_CLEAR,
7097 STR_BIT_FLIP
7098};
7099
7100/*
7101 * Mask for the logical in-byte positions lo..hi (0 <= lo <= hi <= 7) of one
7102 * byte. A contiguous logical run stays contiguous within a byte under both
7103 * numbering conventions; MSB-first only mirrors it.
7104 */
7105static inline unsigned char
7106str_bit_region_byte_mask(unsigned int lo, unsigned int hi, bool lsb_first)
7107{
7108 if (lsb_first) {
7109 return (unsigned char)((0xFFu >> (7 - hi)) & (0xFFu << lo));
7110 }
7111 else {
7112 return (unsigned char)((0xFFu >> lo) & (0xFFu << (7 - hi)));
7113 }
7114}
7115
7116static inline void
7117str_apply_bit_mask(unsigned char *byte, unsigned char mask, enum str_bit_mutation mutation)
7118{
7119 switch (mutation) {
7120 case STR_BIT_SET:
7121 *byte |= mask;
7122 break;
7123 case STR_BIT_CLEAR:
7124 *byte &= (unsigned char)~mask;
7125 break;
7126 case STR_BIT_FLIP:
7127 *byte ^= mask;
7128 break;
7129 }
7130}
7131
7132/* The caller has bounds-checked [beg, beg+len) and called rb_str_modify. */
7133static void
7134str_mutate_bit_region(unsigned char *ptr, uint64_t beg, uint64_t len, bool lsb_first, enum str_bit_mutation mutation)
7135{
7136 uint64_t first_bit = beg;
7137 uint64_t last_bit = beg + len - 1;
7138 long first_byte = (long)(first_bit / CHAR_BIT);
7139 long last_byte = (long)(last_bit / CHAR_BIT);
7140 unsigned int first_off = (unsigned int)(first_bit % CHAR_BIT);
7141 unsigned int last_off = (unsigned int)(last_bit % CHAR_BIT);
7142
7143 if (first_byte == last_byte) {
7144 str_apply_bit_mask(ptr + first_byte, str_bit_region_byte_mask(first_off, last_off, lsb_first), mutation);
7145 return;
7146 }
7147
7148 str_apply_bit_mask(ptr + first_byte, str_bit_region_byte_mask(first_off, 7, lsb_first), mutation);
7149 long middle_len = last_byte - first_byte - 1;
7150 if (middle_len > 0) {
7151 unsigned char *middle = ptr + first_byte + 1;
7152 switch (mutation) {
7153 case STR_BIT_SET:
7154 memset(middle, 0xFF, middle_len);
7155 break;
7156 case STR_BIT_CLEAR:
7157 memset(middle, 0, middle_len);
7158 break;
7159 case STR_BIT_FLIP:
7160 /*
7161 * Byte loop on purpose: the compiler auto-vectorizes it (verified on gcc 13.3
7162 * and clang 18.1 with x86_64), and being read-modify-write, the flip is memory-bound,
7163 * so a manual word-at-a-time XOR loop was measured to be no faster.
7164 */
7165 for (long i = 0; i < middle_len; i++) {
7166 middle[i] ^= 0xFF;
7167 }
7168 break;
7169 }
7170 }
7171 str_apply_bit_mask(ptr + last_byte, str_bit_region_byte_mask(0, last_off, lsb_first), mutation);
7172}
7173
7174static VALUE
7175str_mutate_single_bit(VALUE str, VALUE index, bool lsb_first, enum str_bit_mutation mutation)
7176{
7177 struct str_bit_offset offset = str_bit_offset_from_index(index);
7178 struct str_bit_location location;
7179 long bit_index;
7180 unsigned char *ptr;
7181 unsigned char mask;
7182
7183 rb_check_frozen(str);
7184
7185 if (str_bit_offset_out_of_range(RSTRING_LEN(str), offset.value)) {
7186 rb_raise(rb_eIndexError, "bit index out of range");
7187 }
7188
7189 rb_str_modify(str);
7190 ptr = (unsigned char *)RSTRING_PTR(str);
7191 if (offset.fits_long) {
7192 bit_index = str_logical_to_physical_bit(offset.long_value, lsb_first);
7193 mask = (unsigned char)(1u << (bit_index % CHAR_BIT));
7194 location.byte_index = bit_index / CHAR_BIT;
7195 }
7196 else {
7197 location = str_bit_location_from_offset(offset.value, lsb_first);
7198 mask = (unsigned char)(1u << location.bit_offset);
7199 }
7200
7201 str_apply_bit_mask(ptr + location.byte_index, mask, mutation);
7202 return str;
7203}
7204
7205static VALUE
7206str_mutate_bit(int argc, VALUE *argv, VALUE str, enum str_bit_mutation mutation)
7207{
7208 VALUE target, length_v, opts;
7209 uint64_t beg = 0, len = 0;
7210
7211 /* Count positional arguments so that an explicit nil is not mistaken for an omitted one. */
7212 int nargs = rb_scan_args(argc, argv, "11:", &target, &length_v, &opts);
7213 bool lsb_first = str_lsb_first_from_opts(opts);
7214
7215 bool is_range = rb_obj_is_kind_of(target, rb_cRange);
7216 if (nargs == 1 && !is_range) {
7217 return str_mutate_single_bit(str, target, lsb_first, mutation);
7218 }
7219
7220 struct str_bit_range range = {0};
7221 struct str_bit_offset offset;
7222 if (is_range) {
7223 if (nargs == 2) {
7224 rb_raise(rb_eArgError, "bit length not allowed with a Range");
7225 }
7226 str_bit_range_to_offsets(target, &range);
7227 }
7228 else {
7229 offset = str_bit_offset_from_index(target);
7230 len = str_bit_length_from_index(length_v);
7231 }
7232
7233 /* Even a zero-length write requires a mutable receiver. */
7234 rb_check_frozen(str);
7235
7236 /*
7237 * A region that begins past the end is out of range even when it is
7238 * empty, and one that runs past the end is not allowed to silently
7239 * shrink: both are errors for a mutation, unlike the clamping reads.
7240 * An empty region whose start is within 0..bitsize writes nothing.
7241 */
7242 uint64_t total_bits = str_bit_size(RSTRING_LEN(str));
7243 if (is_range) {
7244 if (!str_bit_range_resolve(&range, total_bits, &beg, &len) || len > total_bits - beg) {
7245 rb_raise(rb_eIndexError, "bit range out of range");
7246 }
7247 }
7248 else {
7249 beg = offset.value;
7250 if (beg > total_bits || len > total_bits - beg) {
7251 rb_raise(rb_eIndexError, "bit range out of range");
7252 }
7253 }
7254
7255 if (len == 0) return str;
7256
7257 rb_str_modify(str);
7258 str_mutate_bit_region((unsigned char *)RSTRING_PTR(str), beg, len, lsb_first, mutation);
7259 return str;
7260}
7261
7262/*
7263 * call-seq:
7264 * bit_set(offset, lsb_first: true) -> self
7265 * bit_set(offset, length, lsb_first: true) -> self
7266 * bit_set(range, lsb_first: true) -> self
7267 *
7268 * :include: doc/string/bit_set.rdoc
7269 *
7270 */
7271static VALUE
7272rb_str_bit_set(int argc, VALUE *argv, VALUE str)
7273{
7274 return str_mutate_bit(argc, argv, str, STR_BIT_SET);
7275}
7276
7277/*
7278 * call-seq:
7279 * bit_clear(offset, lsb_first: true) -> self
7280 * bit_clear(offset, length, lsb_first: true) -> self
7281 * bit_clear(range, lsb_first: true) -> self
7282 *
7283 * :include: doc/string/bit_clear.rdoc
7284 *
7285 */
7286static VALUE
7287rb_str_bit_clear(int argc, VALUE *argv, VALUE str)
7288{
7289 return str_mutate_bit(argc, argv, str, STR_BIT_CLEAR);
7290}
7291
7292/*
7293 * call-seq:
7294 * bit_flip(offset, lsb_first: true) -> self
7295 * bit_flip(offset, length, lsb_first: true) -> self
7296 * bit_flip(range, lsb_first: true) -> self
7297 *
7298 * :include: doc/string/bit_flip.rdoc
7299 *
7300 */
7301static VALUE
7302rb_str_bit_flip(int argc, VALUE *argv, VALUE str)
7303{
7304 return str_mutate_bit(argc, argv, str, STR_BIT_FLIP);
7305}
7306
7307static uint64_t
7308str_count_bits(const unsigned char *ptr, long len)
7309{
7310 uint64_t count = 0;
7311 long off = 0;
7312 long unrolled_end = len & ~31L;
7313 long aligned_end = len & ~7L;
7314
7315 // 32 bytes (256 bits) at a time
7316 for (; off < unrolled_end; off += 32) {
7317 uint64_t w0, w1, w2, w3;
7318 memcpy(&w0, ptr + off, 8);
7319 memcpy(&w1, ptr + off + 8, 8);
7320 memcpy(&w2, ptr + off + 16, 8);
7321 memcpy(&w3, ptr + off + 24, 8);
7322 count += rb_popcount64(w0);
7323 count += rb_popcount64(w1);
7324 count += rb_popcount64(w2);
7325 count += rb_popcount64(w3);
7326 }
7327
7328 // 8 bytes (64 bits) at a time
7329 for (; off < aligned_end; off += 8) {
7330 uint64_t word;
7331 memcpy(&word, ptr + off, 8);
7332 count += rb_popcount64(word);
7333 }
7334
7335 // remaining bytes
7336 if (off < len) {
7337 uint64_t word = 0;
7338 int shift = 0;
7339 for (; off < len; off++, shift += CHAR_BIT) {
7340 word |= (uint64_t)ptr[off] << shift;
7341 }
7342 count += rb_popcount64(word);
7343 }
7344
7345 return count;
7346}
7347
7348static uint64_t
7349str_count_bits_region(const unsigned char *ptr, uint64_t beg, uint64_t len, bool lsb_first)
7350{
7351 uint64_t first_bit = beg;
7352 uint64_t last_bit = beg + len - 1;
7353 long first_byte = (long)(first_bit / CHAR_BIT);
7354 long last_byte = (long)(last_bit / CHAR_BIT);
7355 unsigned int first_off = (unsigned int)(first_bit % CHAR_BIT);
7356 unsigned int last_off = (unsigned int)(last_bit % CHAR_BIT);
7357
7358 if (first_byte == last_byte) {
7359 return rb_popcount32((uint32_t)(ptr[first_byte] & str_bit_region_byte_mask(first_off, last_off, lsb_first)));
7360 }
7361
7362 uint64_t count = rb_popcount32((uint32_t)(ptr[first_byte] & str_bit_region_byte_mask(first_off, 7, lsb_first)));
7363 count += str_count_bits(ptr + first_byte + 1, last_byte - first_byte - 1);
7364 count += rb_popcount32((uint32_t)(ptr[last_byte] & str_bit_region_byte_mask(0, last_off, lsb_first)));
7365 return count;
7366}
7367
7368/*
7369 * call-seq:
7370 * bit_count -> integer
7371 * bit_count(offset, length, lsb_first: true) -> integer
7372 * bit_count(range, lsb_first: true) -> integer
7373 *
7374 * :include: doc/string/bit_count.rdoc
7375 *
7376 */
7377static VALUE
7378rb_str_bit_count(int argc, VALUE *argv, VALUE str)
7379{
7380 VALUE v0, v1, opts;
7381 uint64_t beg = 0, len = 0;
7382
7383 /* Count positional arguments so that an explicit nil is not mistaken for an omitted one. */
7384 int nargs = rb_scan_args(argc, argv, "02:", &v0, &v1, &opts);
7385 /*
7386 * A whole-string popcount is independent of bit numbering.
7387 * no-(offset|range)-argument form only validates lsb_first.
7388 */
7389 bool lsb_first = str_lsb_first_from_opts(opts);
7390
7391 if (nargs == 0) {
7392 return ULL2NUM(str_count_bits((const unsigned char *)RSTRING_PTR(str), RSTRING_LEN(str)));
7393 }
7394
7395 bool is_range = rb_obj_is_kind_of(v0, rb_cRange);
7396 struct str_bit_range range = {0};
7397 if (is_range) {
7398 if (nargs == 2) {
7399 rb_raise(rb_eArgError, "bit length not allowed with a Range");
7400 }
7401 str_bit_range_to_offsets(v0, &range);
7402 }
7403 else if (nargs == 1) {
7404 rb_raise(rb_eArgError, "no bit length given");
7405 }
7406 else {
7407 beg = str_bit_offset_from_index(v0).value;
7408 len = str_bit_length_from_index(v1);
7409 }
7410
7411 const unsigned char *ptr = (const unsigned char *)RSTRING_PTR(str);
7412 uint64_t total_bits = str_bit_size(RSTRING_LEN(str));
7413 if (is_range) {
7414 if (!str_bit_range_resolve(&range, total_bits, &beg, &len)) {
7415 return INT2FIX(0);
7416 }
7417 }
7418 else if (beg >= total_bits) {
7419 return INT2FIX(0);
7420 }
7421
7422 /* Reads clamp: only the part of the region that exists is counted. */
7423 if (len > total_bits - beg) len = total_bits - beg;
7424 if (len == 0) return INT2FIX(0);
7425 return ULL2NUM(str_count_bits_region(ptr, beg, len, lsb_first));
7426}
7427
7428static void
7429str_check_bitwise_length(VALUE str, VALUE other)
7430{
7431 if (RSTRING_LEN(str) != RSTRING_LEN(other)) {
7432 rb_raise(rb_eArgError, "operands must have the same length (%ld vs %ld)",
7433 RSTRING_LEN(str), RSTRING_LEN(other));
7434 }
7435}
7436
7437static VALUE
7438str_bitwise_result(VALUE str)
7439{
7440 long len = RSTRING_LEN(str);
7441 VALUE result = rb_str_buf_new(len);
7442 rb_str_resize(result, len);
7443 rb_enc_associate(result, rb_ascii8bit_encoding());
7444 ENC_CODERANGE_CLEAR(result);
7445 return result;
7446}
7447
7448#define STR_DEFINE_UNARY_BITWISE_KERNEL(name, expr_word, expr_byte) \
7449 static void \
7450 name(unsigned char *dst, const unsigned char *src, long len) \
7451 { \
7452 long off = 0; \
7453 long unrolled_end = len & ~31L; \
7454 long aligned_end = len & ~7L; \
7455 for (; off < unrolled_end; off += 32) { \
7456 uint64_t s0, s1, s2, s3; \
7457 memcpy(&s0, src + off, 8); \
7458 memcpy(&s1, src + off + 8, 8); \
7459 memcpy(&s2, src + off + 16, 8); \
7460 memcpy(&s3, src + off + 24, 8); \
7461 s0 = (expr_word(s0)); \
7462 s1 = (expr_word(s1)); \
7463 s2 = (expr_word(s2)); \
7464 s3 = (expr_word(s3)); \
7465 memcpy(dst + off, &s0, 8); \
7466 memcpy(dst + off + 8, &s1, 8); \
7467 memcpy(dst + off + 16, &s2, 8); \
7468 memcpy(dst + off + 24, &s3, 8); \
7469 } \
7470 for (; off < aligned_end; off += 8) { \
7471 uint64_t word; \
7472 memcpy(&word, src + off, 8); \
7473 word = (expr_word(word)); \
7474 memcpy(dst + off, &word, 8); \
7475 } \
7476 for (; off < len; off++) dst[off] = (expr_byte(src[off])); \
7477 }
7478
7479#define STR_DEFINE_BINARY_BITWISE_KERNEL(name, expr_word, expr_byte) \
7480 static void \
7481 name(unsigned char *dst, const unsigned char *lhs, \
7482 const unsigned char *rhs, long len) \
7483 { \
7484 long off = 0; \
7485 long unrolled_end = len & ~31L; \
7486 long aligned_end = len & ~7L; \
7487 for (; off < unrolled_end; off += 32) { \
7488 uint64_t l0, l1, l2, l3, r0, r1, r2, r3; \
7489 memcpy(&l0, lhs + off, 8); memcpy(&r0, rhs + off, 8); \
7490 memcpy(&l1, lhs + off + 8, 8); memcpy(&r1, rhs + off + 8, 8); \
7491 memcpy(&l2, lhs + off + 16, 8); memcpy(&r2, rhs + off + 16, 8); \
7492 memcpy(&l3, lhs + off + 24, 8); memcpy(&r3, rhs + off + 24, 8); \
7493 l0 = expr_word(l0, r0); \
7494 l1 = expr_word(l1, r1); \
7495 l2 = expr_word(l2, r2); \
7496 l3 = expr_word(l3, r3); \
7497 memcpy(dst + off, &l0, 8); \
7498 memcpy(dst + off + 8, &l1, 8); \
7499 memcpy(dst + off + 16, &l2, 8); \
7500 memcpy(dst + off + 24, &l3, 8); \
7501 } \
7502 for (; off < aligned_end; off += 8) { \
7503 uint64_t lhs_word, rhs_word; \
7504 memcpy(&lhs_word, lhs + off, 8); \
7505 memcpy(&rhs_word, rhs + off, 8); \
7506 lhs_word = expr_word(lhs_word, rhs_word); \
7507 memcpy(dst + off, &lhs_word, 8); \
7508 } \
7509 for (; off < len; off++) dst[off] = expr_byte(lhs[off], rhs[off]); \
7510 }
7511
7512#define STR_BITWISE_NOT_WORD(x) (~(x))
7513#define STR_BITWISE_NOT_BYTE(x) ((unsigned char)~(x))
7514#define STR_BITWISE_AND_WORD(x, y) ((x) & (y))
7515#define STR_BITWISE_AND_BYTE(x, y) ((unsigned char)((x) & (y)))
7516#define STR_BITWISE_OR_WORD(x, y) ((x) | (y))
7517#define STR_BITWISE_OR_BYTE(x, y) ((unsigned char)((x) | (y)))
7518#define STR_BITWISE_XOR_WORD(x, y) ((x) ^ (y))
7519#define STR_BITWISE_XOR_BYTE(x, y) ((unsigned char)((x) ^ (y)))
7520
7521STR_DEFINE_UNARY_BITWISE_KERNEL(str_bitwise_not, STR_BITWISE_NOT_WORD, STR_BITWISE_NOT_BYTE)
7522STR_DEFINE_BINARY_BITWISE_KERNEL(str_bitwise_and, STR_BITWISE_AND_WORD, STR_BITWISE_AND_BYTE)
7523STR_DEFINE_BINARY_BITWISE_KERNEL(str_bitwise_or, STR_BITWISE_OR_WORD, STR_BITWISE_OR_BYTE)
7524STR_DEFINE_BINARY_BITWISE_KERNEL(str_bitwise_xor, STR_BITWISE_XOR_WORD, STR_BITWISE_XOR_BYTE)
7525
7526/*
7527 * call-seq:
7528 * bitwise_not -> string
7529 *
7530 * :include: doc/string/bitwise_not.rdoc
7531 *
7532 */
7533static VALUE
7534rb_str_bitwise_not(VALUE str)
7535{
7536 long len = RSTRING_LEN(str);
7537 VALUE result = str_bitwise_result(str);
7538 str_bitwise_not((unsigned char *)RSTRING_PTR(result),
7539 (const unsigned char *)RSTRING_PTR(str), len);
7540 return result;
7541}
7542
7543/*
7544 * call-seq:
7545 * bitwise_not! -> self
7546 *
7547 * :include: doc/string/bitwise_not_bang.rdoc
7548 *
7549 */
7550static VALUE
7551rb_str_bitwise_not_bang(VALUE str)
7552{
7553 long len;
7554 unsigned char *ptr;
7555
7556 rb_str_modify(str);
7557 len = RSTRING_LEN(str);
7558 ptr = (unsigned char *)RSTRING_PTR(str);
7559 str_bitwise_not(ptr, ptr, len);
7560 return str;
7561}
7562
7563#define STR_DEFINE_BINARY_BITWISE_METHOD(name) \
7564 static VALUE \
7565 rb_str_bitwise_##name(VALUE str, VALUE other) \
7566 { \
7567 long len; \
7568 VALUE result; \
7569 StringValue(other); \
7570 str_check_bitwise_length(str, other); \
7571 len = RSTRING_LEN(str); \
7572 result = str_bitwise_result(str); \
7573 str_bitwise_##name((unsigned char *)RSTRING_PTR(result), \
7574 (const unsigned char *)RSTRING_PTR(str), \
7575 (const unsigned char *)RSTRING_PTR(other), len); \
7576 return result; \
7577 } \
7578 static VALUE \
7579 rb_str_bitwise_##name##_bang(VALUE str, VALUE other) \
7580 { \
7581 long len; \
7582 unsigned char *ptr; \
7583 StringValue(other); \
7584 str_check_bitwise_length(str, other); \
7585 rb_str_modify(str); \
7586 len = RSTRING_LEN(str); \
7587 ptr = (unsigned char *)RSTRING_PTR(str); \
7588 str_bitwise_##name(ptr, ptr, \
7589 (const unsigned char *)RSTRING_PTR(other), len); \
7590 return str; \
7591 }
7592
7593STR_DEFINE_BINARY_BITWISE_METHOD(and)
7594STR_DEFINE_BINARY_BITWISE_METHOD(or)
7595STR_DEFINE_BINARY_BITWISE_METHOD(xor)
7596
7597static VALUE
7598str_byte_substr(VALUE str, long beg, long len, int empty)
7599{
7600 long n = RSTRING_LEN(str);
7601
7602 if (beg > n || len < 0) return Qnil;
7603 if (beg < 0) {
7604 beg += n;
7605 if (beg < 0) return Qnil;
7606 }
7607 if (len > n - beg)
7608 len = n - beg;
7609 if (len <= 0) {
7610 if (!empty) return Qnil;
7611 len = 0;
7612 }
7613
7614 VALUE str2 = str_subseq(str, beg, len);
7615
7616 str_enc_copy_direct(str2, str);
7617
7618 if (RSTRING_LEN(str2) == 0) {
7619 if (!rb_enc_asciicompat(STR_ENC_GET(str)))
7621 else
7623 }
7624 else {
7625 switch (ENC_CODERANGE(str)) {
7626 case ENC_CODERANGE_7BIT:
7628 break;
7629 default:
7631 break;
7632 }
7633 }
7634
7635 return str2;
7636}
7637
7638VALUE
7639rb_str_byte_substr(VALUE str, VALUE beg, VALUE len)
7640{
7641 return str_byte_substr(str, NUM2LONG(beg), NUM2LONG(len), TRUE);
7642}
7643
7644static VALUE
7645str_byte_aref(VALUE str, VALUE indx)
7646{
7647 long idx;
7648 if (FIXNUM_P(indx)) {
7649 idx = FIX2LONG(indx);
7650 }
7651 else {
7652 /* check if indx is Range */
7653 long beg, len = RSTRING_LEN(str);
7654
7655 switch (rb_range_beg_len(indx, &beg, &len, len, 0)) {
7656 case Qfalse:
7657 break;
7658 case Qnil:
7659 return Qnil;
7660 default:
7661 return str_byte_substr(str, beg, len, TRUE);
7662 }
7663
7664 idx = NUM2LONG(indx);
7665 }
7666 return str_byte_substr(str, idx, 1, FALSE);
7667}
7668
7669/*
7670 * call-seq:
7671 * byteslice(offset, length = 1) -> string or nil
7672 * byteslice(range) -> string or nil
7673 *
7674 * :include: doc/string/byteslice.rdoc
7675 */
7676
7677static VALUE
7678rb_str_byteslice(int argc, VALUE *argv, VALUE str)
7679{
7680 if (argc == 2) {
7681 long beg = NUM2LONG(argv[0]);
7682 long len = NUM2LONG(argv[1]);
7683 return str_byte_substr(str, beg, len, TRUE);
7684 }
7685 rb_check_arity(argc, 1, 2);
7686 return str_byte_aref(str, argv[0]);
7687}
7688
7689static void
7690str_check_beg_len(VALUE str, long *beg, long *len)
7691{
7692 long end, slen = RSTRING_LEN(str);
7693
7694 if (*len < 0) rb_raise(rb_eIndexError, "negative length %ld", *len);
7695 if ((slen < *beg) || ((*beg < 0) && (*beg + slen < 0))) {
7696 rb_raise(rb_eIndexError, "index %ld out of string", *beg);
7697 }
7698 if (*beg < 0) {
7699 *beg += slen;
7700 }
7701 RUBY_ASSERT(*beg >= 0);
7702 RUBY_ASSERT(*beg <= slen);
7703
7704 if (*len > slen - *beg) {
7705 *len = slen - *beg;
7706 }
7707 end = *beg + *len;
7708 str_ensure_byte_pos(str, *beg);
7709 str_ensure_byte_pos(str, end);
7710}
7711
7712/*
7713 * call-seq:
7714 * bytesplice(offset, length, str) -> self
7715 * bytesplice(offset, length, str, str_offset, str_length) -> self
7716 * bytesplice(range, str) -> self
7717 * bytesplice(range, str, str_range) -> self
7718 *
7719 * :include: doc/string/bytesplice.rdoc
7720 */
7721
7722static VALUE
7723rb_str_bytesplice(int argc, VALUE *argv, VALUE str)
7724{
7725 long beg, len, vbeg, vlen;
7726 VALUE val;
7727 int cr;
7728
7729 rb_check_arity(argc, 2, 5);
7730 if (!(argc == 2 || argc == 3 || argc == 5)) {
7731 rb_raise(rb_eArgError, "wrong number of arguments (given %d, expected 2, 3, or 5)", argc);
7732 }
7733 if (argc == 2 || (argc == 3 && !RB_INTEGER_TYPE_P(argv[0]))) {
7734 if (!rb_range_beg_len(argv[0], &beg, &len, RSTRING_LEN(str), 2)) {
7735 rb_raise(rb_eTypeError, "wrong argument type %s (expected Range)",
7736 rb_builtin_class_name(argv[0]));
7737 }
7738 val = argv[1];
7739 StringValue(val);
7740 if (argc == 2) {
7741 /* bytesplice(range, str) */
7742 vbeg = 0;
7743 vlen = RSTRING_LEN(val);
7744 }
7745 else {
7746 /* bytesplice(range, str, str_range) */
7747 if (!rb_range_beg_len(argv[2], &vbeg, &vlen, RSTRING_LEN(val), 2)) {
7748 rb_raise(rb_eTypeError, "wrong argument type %s (expected Range)",
7749 rb_builtin_class_name(argv[2]));
7750 }
7751 }
7752 }
7753 else {
7754 beg = NUM2LONG(argv[0]);
7755 len = NUM2LONG(argv[1]);
7756 val = argv[2];
7757 StringValue(val);
7758 if (argc == 3) {
7759 /* bytesplice(index, length, str) */
7760 vbeg = 0;
7761 vlen = RSTRING_LEN(val);
7762 }
7763 else {
7764 /* bytesplice(index, length, str, str_index, str_length) */
7765 vbeg = NUM2LONG(argv[3]);
7766 vlen = NUM2LONG(argv[4]);
7767 }
7768 }
7769 str_check_beg_len(str, &beg, &len);
7770 str_check_beg_len(val, &vbeg, &vlen);
7771 str_modify_keep_cr(str);
7772
7773 if (RB_UNLIKELY(ENCODING_GET_INLINED(str) != ENCODING_GET_INLINED(val))) {
7774 rb_enc_associate(str, rb_enc_check(str, val));
7775 }
7776
7777 rb_str_update_1(str, beg, len, val, vbeg, vlen);
7779 if (cr != ENC_CODERANGE_BROKEN)
7780 ENC_CODERANGE_SET(str, cr);
7781 return str;
7782}
7783
7784/*
7785 * call-seq:
7786 * reverse -> new_string
7787 *
7788 * Returns a new string with the characters from +self+ in reverse order.
7789 *
7790 * 'drawer'.reverse # => "reward"
7791 * 'reviled'.reverse # => "deliver"
7792 * 'stressed'.reverse # => "desserts"
7793 * 'semordnilaps'.reverse # => "spalindromes"
7794 *
7795 * Related: see {Converting to New String}[rdoc-ref:String@Converting+to+New+String].
7796 */
7797
7798static VALUE
7799rb_str_reverse(VALUE str)
7800{
7801 rb_encoding *enc;
7802 VALUE rev;
7803 char *s, *e, *p;
7804 int cr;
7805
7806 if (RSTRING_LEN(str) <= 1) return str_duplicate(rb_cString, str);
7807 enc = STR_ENC_GET(str);
7808 rev = rb_str_new(0, RSTRING_LEN(str));
7809 s = RSTRING_PTR(str); e = RSTRING_END(str);
7810 p = RSTRING_END(rev);
7811 cr = ENC_CODERANGE(str);
7812
7813 if (RSTRING_LEN(str) > 1) {
7814 if (single_byte_optimizable(str)) {
7815 while (s < e) {
7816 *--p = *s++;
7817 }
7818 }
7819 else if (cr == ENC_CODERANGE_VALID) {
7820 while (s < e) {
7821 int clen = rb_enc_fast_mbclen(s, e, enc);
7822
7823 p -= clen;
7824 memcpy(p, s, clen);
7825 s += clen;
7826 }
7827 }
7828 else {
7829 cr = rb_enc_asciicompat(enc) ?
7831 while (s < e) {
7832 int clen = rb_enc_mbclen(s, e, enc);
7833
7834 if (clen > 1 || (*s & 0x80)) cr = ENC_CODERANGE_UNKNOWN;
7835 p -= clen;
7836 memcpy(p, s, clen);
7837 s += clen;
7838 }
7839 }
7840 }
7841 STR_SET_LEN(rev, RSTRING_LEN(str));
7842 str_enc_copy_direct(rev, str);
7843 ENC_CODERANGE_SET(rev, cr);
7844
7845 return rev;
7846}
7847
7848
7849/*
7850 * call-seq:
7851 * reverse! -> self
7852 *
7853 * Returns +self+ with its characters reversed:
7854 *
7855 * 'drawer'.reverse! # => "reward"
7856 * 'reviled'.reverse! # => "deliver"
7857 * 'stressed'.reverse! # => "desserts"
7858 * 'semordnilaps'.reverse! # => "spalindromes"
7859 *
7860 * Related: see {Modifying}[rdoc-ref:String@Modifying].
7861 */
7862
7863static VALUE
7864rb_str_reverse_bang(VALUE str)
7865{
7866 if (RSTRING_LEN(str) > 1) {
7867 if (single_byte_optimizable(str)) {
7868 char *s, *e, c;
7869
7870 str_modify_keep_cr(str);
7871 s = RSTRING_PTR(str);
7872 e = RSTRING_END(str) - 1;
7873 while (s < e) {
7874 c = *s;
7875 *s++ = *e;
7876 *e-- = c;
7877 }
7878 }
7879 else {
7880 str_shared_replace(str, rb_str_reverse(str));
7881 }
7882 }
7883 else {
7884 str_modify_keep_cr(str);
7885 }
7886 return str;
7887}
7888
7889
7890/*
7891 * call-seq:
7892 * include?(other_string) -> true or false
7893 *
7894 * Returns whether +self+ contains +other_string+:
7895 *
7896 * s = 'bar'
7897 * s.include?('ba') # => true
7898 * s.include?('ar') # => true
7899 * s.include?('bar') # => true
7900 * s.include?('a') # => true
7901 * s.include?('') # => true
7902 * s.include?('foo') # => false
7903 *
7904 * Related: see {Querying}[rdoc-ref:String@Querying].
7905 */
7906
7907VALUE
7908rb_str_include(VALUE str, VALUE arg)
7909{
7910 long i;
7911
7912 StringValue(arg);
7913 i = rb_str_index(str, arg, 0);
7914
7915 return RBOOL(i != -1);
7916}
7917
7918
7919/*
7920 * call-seq:
7921 * to_i(base = 10) -> integer
7922 *
7923 * Returns the result of interpreting leading characters in +self+
7924 * as an integer in the given +base+;
7925 * +base+ must be either +0+ or in range <tt>(2..36)</tt>:
7926 *
7927 * '123456'.to_i # => 123456
7928 * '123def'.to_i(16) # => 1195503
7929 *
7930 * With +base+ zero given, string +object+ may contain leading characters
7931 * to specify the actual base:
7932 *
7933 * '123def'.to_i(0) # => 123
7934 * '0123def'.to_i(0) # => 83
7935 * '0b123def'.to_i(0) # => 1
7936 * '0o123def'.to_i(0) # => 83
7937 * '0d123def'.to_i(0) # => 123
7938 * '0x123def'.to_i(0) # => 1195503
7939 *
7940 * Characters past a leading valid number (in the given +base+) are ignored:
7941 *
7942 * '12.345'.to_i # => 12
7943 * '12345'.to_i(2) # => 1
7944 *
7945 * Returns zero if there is no leading valid number:
7946 *
7947 * 'abcdef'.to_i # => 0
7948 * '2'.to_i(2) # => 0
7949 *
7950 * Related: see {Converting to Non-String}[rdoc-ref:String@Converting+to+Non-String].
7951 */
7952
7953static VALUE
7954rb_str_to_i(int argc, VALUE *argv, VALUE str)
7955{
7956 int base = 10;
7957
7958 if (rb_check_arity(argc, 0, 1) && (base = NUM2INT(argv[0])) < 0) {
7959 rb_raise(rb_eArgError, "invalid radix %d", base);
7960 }
7961 return rb_str_to_inum(str, base, FALSE);
7962}
7963
7964
7965/*
7966 * call-seq:
7967 * to_f -> float
7968 *
7969 * Returns the result of interpreting leading characters in +self+ as a Float:
7970 *
7971 * '3.14159'.to_f # => 3.14159
7972 * '1.234e-2'.to_f # => 0.01234
7973 *
7974 * Characters past a leading valid number are ignored:
7975 *
7976 * '3.14 (pi to two places)'.to_f # => 3.14
7977 *
7978 * Returns zero if there is no leading valid number:
7979 *
7980 * 'abcdef'.to_f # => 0.0
7981 *
7982 * See {Converting to Non-String}[rdoc-ref:String@Converting+to+Non-String].
7983 */
7984
7985static VALUE
7986rb_str_to_f(VALUE str)
7987{
7988 return DBL2NUM(rb_str_to_dbl(str, FALSE));
7989}
7990
7991
7992/*
7993 * call-seq:
7994 * to_s -> self or new_string
7995 *
7996 * Returns +self+ if +self+ is a +String+,
7997 * or +self+ converted to a +String+ if +self+ is a subclass of +String+.
7998 *
7999 * Related: see {Converting to New String}[rdoc-ref:String@Converting+to+New+String].
8000 */
8001
8002static VALUE
8003rb_str_to_s(VALUE str)
8004{
8005 if (rb_obj_class(str) != rb_cString) {
8006 return str_duplicate(rb_cString, str);
8007 }
8008 return str;
8009}
8010
8011#if 0
8012static void
8013str_cat_char(VALUE str, unsigned int c, rb_encoding *enc)
8014{
8015 char s[RUBY_MAX_CHAR_LEN];
8016 int n = rb_enc_codelen(c, enc);
8017
8018 rb_enc_mbcput(c, s, enc);
8019 rb_enc_str_buf_cat(str, s, n, enc);
8020}
8021#endif
8022
8023#define CHAR_ESC_LEN 13 /* sizeof(\x{ hex of 32bit unsigned int } \0) */
8024
8025int
8026rb_str_buf_cat_escaped_char(VALUE result, unsigned int c, int unicode_p)
8027{
8028 char buf[CHAR_ESC_LEN + 1];
8029 int l;
8030
8031#if SIZEOF_INT > 4
8032 c &= 0xffffffff;
8033#endif
8034 if (unicode_p) {
8035 if (c < 0x7F && ISPRINT(c)) {
8036 snprintf(buf, CHAR_ESC_LEN, "%c", c);
8037 }
8038 else if (c < 0x10000) {
8039 snprintf(buf, CHAR_ESC_LEN, "\\u%04X", c);
8040 }
8041 else {
8042 snprintf(buf, CHAR_ESC_LEN, "\\u{%X}", c);
8043 }
8044 }
8045 else {
8046 if (c < 0x100) {
8047 snprintf(buf, CHAR_ESC_LEN, "\\x%02X", c);
8048 }
8049 else {
8050 snprintf(buf, CHAR_ESC_LEN, "\\x{%X}", c);
8051 }
8052 }
8053 l = (int)strlen(buf); /* CHAR_ESC_LEN cannot exceed INT_MAX */
8054 rb_str_buf_cat(result, buf, l);
8055 return l;
8056}
8057
8058const char *
8059ruby_escaped_char(int c)
8060{
8061 switch (c) {
8062 case '\0': return "\\0";
8063 case '\n': return "\\n";
8064 case '\r': return "\\r";
8065 case '\t': return "\\t";
8066 case '\f': return "\\f";
8067 case '\013': return "\\v";
8068 case '\010': return "\\b";
8069 case '\007': return "\\a";
8070 case '\033': return "\\e";
8071 case '\x7f': return "\\c?";
8072 }
8073 return NULL;
8074}
8075
8076VALUE
8077rb_str_escape(VALUE str)
8078{
8079 int encidx = ENCODING_GET(str);
8080 rb_encoding *enc = rb_enc_from_index(encidx);
8081 const char *p = RSTRING_PTR(str);
8082 const char *pend = RSTRING_END(str);
8083 const char *prev = p;
8084 char buf[CHAR_ESC_LEN + 1];
8085 VALUE result = rb_str_buf_new(0);
8086 int unicode_p = rb_enc_unicode_p(enc);
8087 int asciicompat = rb_enc_asciicompat(enc);
8088
8089 while (p < pend) {
8090 unsigned int c;
8091 const char *cc;
8092 int n = rb_enc_precise_mbclen(p, pend, enc);
8093 if (!MBCLEN_CHARFOUND_P(n)) {
8094 if (p > prev) str_buf_cat(result, prev, p - prev);
8095 n = rb_enc_mbminlen(enc);
8096 if (pend < p + n)
8097 n = (int)(pend - p);
8098 while (n--) {
8099 snprintf(buf, CHAR_ESC_LEN, "\\x%02X", *p & 0377);
8100 str_buf_cat(result, buf, strlen(buf));
8101 prev = ++p;
8102 }
8103 continue;
8104 }
8105 n = MBCLEN_CHARFOUND_LEN(n);
8106 c = rb_enc_mbc_to_codepoint(p, pend, enc);
8107 p += n;
8108 cc = ruby_escaped_char(c);
8109 if (cc) {
8110 if (p - n > prev) str_buf_cat(result, prev, p - n - prev);
8111 str_buf_cat(result, cc, strlen(cc));
8112 prev = p;
8113 }
8114 else if (asciicompat && rb_enc_isascii(c, enc) && ISPRINT(c)) {
8115 }
8116 else {
8117 if (p - n > prev) str_buf_cat(result, prev, p - n - prev);
8118 rb_str_buf_cat_escaped_char(result, c, unicode_p);
8119 prev = p;
8120 }
8121 }
8122 if (p > prev) str_buf_cat(result, prev, p - prev);
8123 ENCODING_CODERANGE_SET(result, rb_usascii_encindex(), ENC_CODERANGE_7BIT);
8124
8125 return result;
8126}
8127
8128/* Lookup table for the inspect fast path. 1 marks bytes that need
8129 * no escaping. 0 marks bytes that need escape inspection: 0x00-0x1F
8130 * (control), 0x22 ("), 0x23 (#), 0x5C (\‍), 0x7F (DEL), 0x80-0xFF
8131 * (non-ASCII). */
8132static const bool inspect_no_escape[256] = {
8133 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, /* 0x00-0x0F */
8134 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, /* 0x10-0x1F */
8135 1, 1, 0, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, /* 0x20-0x2F */
8136 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, /* 0x30-0x3F */
8137 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, /* 0x40-0x4F */
8138 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 1, 1, 1, /* 0x50-0x5F */
8139 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, /* 0x60-0x6F */
8140 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, /* 0x70-0x7F */
8141};
8142
8143/*
8144 * call-seq:
8145 * inspect -> string
8146 *
8147 * :include: doc/string/inspect.rdoc
8148 *
8149 */
8150
8151VALUE
8153{
8154 int encidx = ENCODING_GET(str);
8155 rb_encoding *enc = rb_enc_from_index(encidx);
8156 const char *p, *pend, *prev;
8157 char buf[CHAR_ESC_LEN + 1];
8158 VALUE result = rb_str_buf_new(RSTRING_LEN(str) + 2); /* string content + surrounding quotes */
8159 rb_encoding *resenc = rb_default_internal_encoding();
8160 int unicode_p = rb_enc_unicode_p(enc);
8161 int asciicompat = rb_enc_asciicompat(enc);
8162 int cr = rb_enc_str_coderange(str);
8163
8164 if (resenc == NULL) resenc = rb_default_external_encoding();
8165 if (!rb_enc_asciicompat(resenc)) resenc = rb_usascii_encoding();
8166 rb_enc_associate(result, resenc);
8167 str_buf_cat2(result, "\"");
8168
8169 p = RSTRING_PTR(str); pend = RSTRING_END(str);
8170 prev = p;
8171 while (p < pend) {
8172 unsigned int c, cc;
8173 int n;
8174
8175 /* Fast path: bulk-skip runs of safe ASCII bytes via a lookup table.
8176 * Only well-formed strings (CR=7BIT for any encoding, or UTF-8 VALID)
8177 * are eligible. */
8178 if (cr == ENC_CODERANGE_7BIT ||
8179 (encidx == ENCINDEX_UTF_8 && cr == ENC_CODERANGE_VALID)) {
8180 while (p < pend && inspect_no_escape[(unsigned char)*p]) p++;
8181 if (p >= pend) break;
8182 }
8183
8184 n = rb_enc_precise_mbclen(p, pend, enc);
8185 if (!MBCLEN_CHARFOUND_P(n)) {
8186 if (p > prev) str_buf_cat(result, prev, p - prev);
8187 n = rb_enc_mbminlen(enc);
8188 if (pend < p + n)
8189 n = (int)(pend - p);
8190 while (n--) {
8191 snprintf(buf, CHAR_ESC_LEN, "\\x%02X", *p & 0377);
8192 str_buf_cat(result, buf, strlen(buf));
8193 prev = ++p;
8194 }
8195 continue;
8196 }
8197 n = MBCLEN_CHARFOUND_LEN(n);
8198 c = rb_enc_mbc_to_codepoint(p, pend, enc);
8199 p += n;
8200 if ((asciicompat || unicode_p) &&
8201 (c == '"'|| c == '\\' ||
8202 (c == '#' &&
8203 p < pend &&
8204 MBCLEN_CHARFOUND_P(rb_enc_precise_mbclen(p,pend,enc)) &&
8205 (cc = rb_enc_codepoint(p,pend,enc),
8206 (cc == '$' || cc == '@' || cc == '{'))))) {
8207 if (p - n > prev) str_buf_cat(result, prev, p - n - prev);
8208 str_buf_cat2(result, "\\");
8209 if (asciicompat || enc == resenc) {
8210 prev = p - n;
8211 continue;
8212 }
8213 }
8214 switch (c) {
8215 case '\n': cc = 'n'; break;
8216 case '\r': cc = 'r'; break;
8217 case '\t': cc = 't'; break;
8218 case '\f': cc = 'f'; break;
8219 case '\013': cc = 'v'; break;
8220 case '\010': cc = 'b'; break;
8221 case '\007': cc = 'a'; break;
8222 case 033: cc = 'e'; break;
8223 default: cc = 0; break;
8224 }
8225 if (cc) {
8226 if (p - n > prev) str_buf_cat(result, prev, p - n - prev);
8227 buf[0] = '\\';
8228 buf[1] = (char)cc;
8229 str_buf_cat(result, buf, 2);
8230 prev = p;
8231 continue;
8232 }
8233 /* The special casing of 0x85 (NEXT_LINE) here is because
8234 * Oniguruma historically treats it as printable, but it
8235 * doesn't match the print POSIX bracket class or character
8236 * property in regexps.
8237 *
8238 * See Ruby Bug #16842 for details:
8239 * https://bugs.ruby-lang.org/issues/16842
8240 */
8241 if ((enc == resenc && rb_enc_isprint(c, enc) && c != 0x85) ||
8242 (asciicompat && rb_enc_isascii(c, enc) && ISPRINT(c))) {
8243 continue;
8244 }
8245 else {
8246 if (p - n > prev) str_buf_cat(result, prev, p - n - prev);
8247 rb_str_buf_cat_escaped_char(result, c, unicode_p);
8248 prev = p;
8249 continue;
8250 }
8251 }
8252 if (p > prev) str_buf_cat(result, prev, p - prev);
8253 str_buf_cat2(result, "\"");
8254
8255 return result;
8256}
8257
8258#define IS_EVSTR(p,e) ((p) < (e) && (*(p) == '$' || *(p) == '@' || *(p) == '{'))
8259
8260/*
8261 * call-seq:
8262 * dump -> new_string
8263 *
8264 * :include: doc/string/dump.rdoc
8265 *
8266 */
8267
8268VALUE
8270{
8271 int encidx = rb_enc_get_index(str);
8272 rb_encoding *enc = rb_enc_from_index(encidx);
8273 long len;
8274 const char *p, *pend;
8275 char *q, *qend;
8276 VALUE result;
8277 int u8 = (encidx == rb_utf8_encindex());
8278 static const char nonascii_suffix[] = ".dup.force_encoding(\"%s\")";
8279
8280 len = 2; /* "" */
8281 if (!rb_enc_asciicompat(enc)) {
8282 len += strlen(nonascii_suffix) - rb_strlen_lit("%s");
8283 len += strlen(enc->name);
8284 }
8285
8286 p = RSTRING_PTR(str); pend = p + RSTRING_LEN(str);
8287 while (p < pend) {
8288 int clen;
8289 unsigned char c = *p++;
8290
8291 switch (c) {
8292 case '"': case '\\':
8293 case '\n': case '\r':
8294 case '\t': case '\f':
8295 case '\013': case '\010': case '\007': case '\033':
8296 clen = 2;
8297 break;
8298
8299 case '#':
8300 clen = IS_EVSTR(p, pend) ? 2 : 1;
8301 break;
8302
8303 default:
8304 if (ISPRINT(c)) {
8305 clen = 1;
8306 }
8307 else {
8308 if (u8 && c > 0x7F) { /* \u notation */
8309 int n = rb_enc_precise_mbclen(p-1, pend, enc);
8310 if (MBCLEN_CHARFOUND_P(n)) {
8311 unsigned int cc = rb_enc_mbc_to_codepoint(p-1, pend, enc);
8312 if (cc <= 0xFFFF)
8313 clen = 6; /* \uXXXX */
8314 else if (cc <= 0xFFFFF)
8315 clen = 9; /* \u{XXXXX} */
8316 else
8317 clen = 10; /* \u{XXXXXX} */
8318 p += MBCLEN_CHARFOUND_LEN(n)-1;
8319 break;
8320 }
8321 }
8322 clen = 4; /* \xNN */
8323 }
8324 break;
8325 }
8326
8327 if (clen > LONG_MAX - len) {
8328 rb_raise(rb_eRuntimeError, "string size too big");
8329 }
8330 len += clen;
8331 }
8332
8333 result = rb_str_new(0, len);
8334 p = RSTRING_PTR(str); pend = p + RSTRING_LEN(str);
8335 q = RSTRING_PTR(result); qend = q + len + 1;
8336
8337 *q++ = '"';
8338 while (p < pend) {
8339 unsigned char c = *p++;
8340
8341 if (c == '"' || c == '\\') {
8342 *q++ = '\\';
8343 *q++ = c;
8344 }
8345 else if (c == '#') {
8346 if (IS_EVSTR(p, pend)) *q++ = '\\';
8347 *q++ = '#';
8348 }
8349 else if (c == '\n') {
8350 *q++ = '\\';
8351 *q++ = 'n';
8352 }
8353 else if (c == '\r') {
8354 *q++ = '\\';
8355 *q++ = 'r';
8356 }
8357 else if (c == '\t') {
8358 *q++ = '\\';
8359 *q++ = 't';
8360 }
8361 else if (c == '\f') {
8362 *q++ = '\\';
8363 *q++ = 'f';
8364 }
8365 else if (c == '\013') {
8366 *q++ = '\\';
8367 *q++ = 'v';
8368 }
8369 else if (c == '\010') {
8370 *q++ = '\\';
8371 *q++ = 'b';
8372 }
8373 else if (c == '\007') {
8374 *q++ = '\\';
8375 *q++ = 'a';
8376 }
8377 else if (c == '\033') {
8378 *q++ = '\\';
8379 *q++ = 'e';
8380 }
8381 else if (ISPRINT(c)) {
8382 *q++ = c;
8383 }
8384 else {
8385 *q++ = '\\';
8386 if (u8) {
8387 int n = rb_enc_precise_mbclen(p-1, pend, enc) - 1;
8388 if (MBCLEN_CHARFOUND_P(n)) {
8389 int cc = rb_enc_mbc_to_codepoint(p-1, pend, enc);
8390 p += n;
8391 if (cc <= 0xFFFF)
8392 snprintf(q, qend-q, "u%04X", cc); /* \uXXXX */
8393 else
8394 snprintf(q, qend-q, "u{%X}", cc); /* \u{XXXXX} or \u{XXXXXX} */
8395 q += strlen(q);
8396 continue;
8397 }
8398 }
8399 snprintf(q, qend-q, "x%02X", c);
8400 q += 3;
8401 }
8402 }
8403 *q++ = '"';
8404 *q = '\0';
8405 if (!rb_enc_asciicompat(enc)) {
8406 snprintf(q, qend-q, nonascii_suffix, enc->name);
8407 encidx = rb_ascii8bit_encindex();
8408 }
8409 /* result from dump is ASCII */
8410 rb_enc_associate_index(result, encidx);
8412 return result;
8413}
8414
8415static int
8416unescape_ascii(unsigned int c)
8417{
8418 switch (c) {
8419 case 'n':
8420 return '\n';
8421 case 'r':
8422 return '\r';
8423 case 't':
8424 return '\t';
8425 case 'f':
8426 return '\f';
8427 case 'v':
8428 return '\13';
8429 case 'b':
8430 return '\010';
8431 case 'a':
8432 return '\007';
8433 case 'e':
8434 return 033;
8435 }
8437}
8438
8439static void
8440undump_after_backslash(VALUE undumped, const char **ss, const char *s_end, rb_encoding **penc, bool *utf8, bool *binary)
8441{
8442 const char *s = *ss;
8443 unsigned int c;
8444 int codelen;
8445 size_t hexlen;
8446 unsigned char buf[6];
8447 static rb_encoding *enc_utf8 = NULL;
8448
8449 switch (*s) {
8450 case '\\':
8451 case '"':
8452 case '#':
8453 rb_str_cat(undumped, s, 1); /* cat itself */
8454 s++;
8455 break;
8456 case 'n':
8457 case 'r':
8458 case 't':
8459 case 'f':
8460 case 'v':
8461 case 'b':
8462 case 'a':
8463 case 'e':
8464 *buf = unescape_ascii(*s);
8465 rb_str_cat(undumped, (char *)buf, 1);
8466 s++;
8467 break;
8468 case 'u':
8469 if (*binary) {
8470 rb_raise(rb_eRuntimeError, "hex escape and Unicode escape are mixed");
8471 }
8472 *utf8 = true;
8473 if (++s >= s_end) {
8474 rb_raise(rb_eRuntimeError, "invalid Unicode escape");
8475 }
8476 if (enc_utf8 == NULL) enc_utf8 = rb_utf8_encoding();
8477 if (*penc != enc_utf8) {
8478 *penc = enc_utf8;
8479 rb_enc_associate(undumped, enc_utf8);
8480 }
8481 if (*s == '{') { /* handle \u{...} form */
8482 s++;
8483 for (;;) {
8484 if (s >= s_end) {
8485 rb_raise(rb_eRuntimeError, "unterminated Unicode escape");
8486 }
8487 if (*s == '}') {
8488 s++;
8489 break;
8490 }
8491 if (ISSPACE(*s)) {
8492 s++;
8493 continue;
8494 }
8495 c = scan_hex(s, s_end-s, &hexlen);
8496 if (hexlen == 0 || hexlen > 6) {
8497 rb_raise(rb_eRuntimeError, "invalid Unicode escape");
8498 }
8499 if (c > 0x10ffff) {
8500 rb_raise(rb_eRuntimeError, "invalid Unicode codepoint (too large)");
8501 }
8502 if (0xd800 <= c && c <= 0xdfff) {
8503 rb_raise(rb_eRuntimeError, "invalid Unicode codepoint");
8504 }
8505 codelen = rb_enc_mbcput(c, (char *)buf, *penc);
8506 rb_str_cat(undumped, (char *)buf, codelen);
8507 s += hexlen;
8508 }
8509 }
8510 else { /* handle \uXXXX form */
8511 c = scan_hex(s, 4, &hexlen);
8512 if (hexlen != 4) {
8513 rb_raise(rb_eRuntimeError, "invalid Unicode escape");
8514 }
8515 if (0xd800 <= c && c <= 0xdfff) {
8516 rb_raise(rb_eRuntimeError, "invalid Unicode codepoint");
8517 }
8518 codelen = rb_enc_mbcput(c, (char *)buf, *penc);
8519 rb_str_cat(undumped, (char *)buf, codelen);
8520 s += hexlen;
8521 }
8522 break;
8523 case 'x':
8524 if (++s >= s_end) {
8525 rb_raise(rb_eRuntimeError, "invalid hex escape");
8526 }
8527 *buf = scan_hex(s, 2, &hexlen);
8528 if (hexlen != 2) {
8529 rb_raise(rb_eRuntimeError, "invalid hex escape");
8530 }
8531 if (!ISASCII(*buf)) {
8532 if (*utf8) {
8533 rb_raise(rb_eRuntimeError, "hex escape and Unicode escape are mixed");
8534 }
8535 *binary = true;
8536 }
8537 rb_str_cat(undumped, (char *)buf, 1);
8538 s += hexlen;
8539 break;
8540 default:
8541 rb_str_cat(undumped, s-1, 2);
8542 s++;
8543 }
8544
8545 *ss = s;
8546}
8547
8548static VALUE rb_str_is_ascii_only_p(VALUE str);
8549
8550/*
8551 * call-seq:
8552 * undump -> new_string
8553 *
8554 * Inverse of String#dump; returns a copy of +self+ with changes of the kinds made by String#dump "undone."
8555 *
8556 * Related: see {Converting to New String}[rdoc-ref:String@Converting+to+New+String].
8557 */
8558
8559static VALUE
8560str_undump(VALUE str)
8561{
8562 const char *s = RSTRING_PTR(str);
8563 const char *s_end = RSTRING_END(str);
8564 rb_encoding *enc = rb_enc_get(str);
8565 VALUE undumped = rb_enc_str_new(s, 0L, enc);
8566 bool utf8 = false;
8567 bool binary = false;
8568 int w;
8569
8571 if (rb_str_is_ascii_only_p(str) == Qfalse) {
8572 rb_raise(rb_eRuntimeError, "non-ASCII character detected");
8573 }
8574 if (!str_null_check(str, &w)) {
8575 rb_raise(rb_eRuntimeError, "string contains null byte");
8576 }
8577 if (RSTRING_LEN(str) < 2) goto invalid_format;
8578 if (*s != '"') goto invalid_format;
8579
8580 /* strip '"' at the start */
8581 s++;
8582
8583 for (;;) {
8584 if (s >= s_end) {
8585 rb_raise(rb_eRuntimeError, "unterminated dumped string");
8586 }
8587
8588 if (*s == '"') {
8589 /* epilogue */
8590 s++;
8591 if (s == s_end) {
8592 /* ascii compatible dumped string */
8593 break;
8594 }
8595 else {
8596 static const char force_encoding_suffix[] = ".force_encoding(\""; /* "\")" */
8597 static const char dup_suffix[] = ".dup";
8598 const char *encname;
8599 int encidx;
8600 ptrdiff_t size;
8601
8602 /* check separately for strings dumped by older versions */
8603 size = sizeof(dup_suffix) - 1;
8604 if (s_end - s > size && memcmp(s, dup_suffix, size) == 0) s += size;
8605
8606 size = sizeof(force_encoding_suffix) - 1;
8607 if (s_end - s <= size) goto invalid_format;
8608 if (memcmp(s, force_encoding_suffix, size) != 0) goto invalid_format;
8609 s += size;
8610
8611 if (utf8) {
8612 rb_raise(rb_eRuntimeError, "dumped string contained Unicode escape but used force_encoding");
8613 }
8614
8615 encname = s;
8616 s = memchr(s, '"', s_end-s);
8617 size = s - encname;
8618 if (!s) goto invalid_format;
8619 if (s_end - s != 2) goto invalid_format;
8620 if (s[0] != '"' || s[1] != ')') goto invalid_format;
8621
8622 encidx = rb_enc_find_index2(encname, (long)size);
8623 if (encidx < 0) {
8624 rb_raise(rb_eRuntimeError, "dumped string has unknown encoding name");
8625 }
8626 rb_enc_associate_index(undumped, encidx);
8627 }
8628 break;
8629 }
8630
8631 if (*s == '\\') {
8632 s++;
8633 if (s >= s_end) {
8634 rb_raise(rb_eRuntimeError, "invalid escape");
8635 }
8636 undump_after_backslash(undumped, &s, s_end, &enc, &utf8, &binary);
8637 }
8638 else {
8639 rb_str_cat(undumped, s++, 1);
8640 }
8641 }
8642
8643 RB_GC_GUARD(str);
8644
8645 return undumped;
8646invalid_format:
8647 rb_raise(rb_eRuntimeError, "invalid dumped string; not wrapped with '\"' nor '\"...\".force_encoding(\"...\")' form");
8648}
8649
8650static void
8651rb_str_check_dummy_enc(rb_encoding *enc)
8652{
8653 if (rb_enc_dummy_p(enc)) {
8654 rb_raise(rb_eEncCompatError, "incompatible encoding with this operation: %s",
8655 rb_enc_name(enc));
8656 }
8657}
8658
8659static rb_encoding *
8660str_true_enc(VALUE str)
8661{
8662 rb_encoding *enc = STR_ENC_GET(str);
8663 rb_str_check_dummy_enc(enc);
8664 return enc;
8665}
8666
8667static OnigCaseFoldType
8668check_case_options(int argc, VALUE *argv, OnigCaseFoldType flags)
8669{
8670 if (argc==0)
8671 return flags;
8672 if (argc>2)
8673 rb_raise(rb_eArgError, "too many options");
8674 if (argv[0]==sym_turkic) {
8675 flags |= ONIGENC_CASE_FOLD_TURKISH_AZERI;
8676 if (argc==2) {
8677 if (argv[1]==sym_lithuanian)
8678 flags |= ONIGENC_CASE_FOLD_LITHUANIAN;
8679 else
8680 rb_raise(rb_eArgError, "invalid second option");
8681 }
8682 }
8683 else if (argv[0]==sym_lithuanian) {
8684 flags |= ONIGENC_CASE_FOLD_LITHUANIAN;
8685 if (argc==2) {
8686 if (argv[1]==sym_turkic)
8687 flags |= ONIGENC_CASE_FOLD_TURKISH_AZERI;
8688 else
8689 rb_raise(rb_eArgError, "invalid second option");
8690 }
8691 }
8692 else if (argc>1)
8693 rb_raise(rb_eArgError, "too many options");
8694 else if (argv[0]==sym_ascii)
8695 flags |= ONIGENC_CASE_ASCII_ONLY;
8696 else if (argv[0]==sym_fold) {
8697 if ((flags & (ONIGENC_CASE_UPCASE|ONIGENC_CASE_DOWNCASE)) == ONIGENC_CASE_DOWNCASE)
8698 flags ^= ONIGENC_CASE_FOLD|ONIGENC_CASE_DOWNCASE;
8699 else
8700 rb_raise(rb_eArgError, "option :fold only allowed for downcasing");
8701 }
8702 else
8703 rb_raise(rb_eArgError, "invalid option");
8704 return flags;
8705}
8706
8707static inline bool
8708case_option_single_p(OnigCaseFoldType flags, rb_encoding *enc, VALUE str)
8709{
8710 if ((flags & ONIGENC_CASE_ASCII_ONLY) && (enc==rb_utf8_encoding() || rb_enc_mbmaxlen(enc) == 1))
8711 return true;
8712 return !(flags & ONIGENC_CASE_FOLD_TURKISH_AZERI) &&
8713 (ENC_CODERANGE(str) == ENC_CODERANGE_7BIT || rb_is_ascii8bit_enc(enc));
8714}
8715
8716/* 16 should be long enough to absorb any kind of single character length increase */
8717#define CASE_MAPPING_ADDITIONAL_LENGTH 20
8718#ifndef CASEMAP_DEBUG
8719# define CASEMAP_DEBUG 0
8720#endif
8721
8722struct mapping_buffer;
8723typedef struct mapping_buffer {
8724 size_t capa;
8725 size_t used;
8726 struct mapping_buffer *next;
8727 OnigUChar space[FLEX_ARY_LEN];
8729
8730static void
8731mapping_buffer_free(void *p)
8732{
8733 mapping_buffer *previous_buffer;
8734 mapping_buffer *current_buffer = p;
8735 while (current_buffer) {
8736 previous_buffer = current_buffer;
8737 current_buffer = current_buffer->next;
8738 ruby_xfree_sized(previous_buffer, offsetof(mapping_buffer, space) + previous_buffer->capa);
8739 }
8740}
8741
8742static const rb_data_type_t mapping_buffer_type = {
8743 "mapping_buffer",
8744 {0, mapping_buffer_free,},
8745 0, 0, RUBY_TYPED_THREAD_SAFE_FREE | RUBY_TYPED_WB_PROTECTED
8746};
8747
8748static VALUE
8749rb_str_casemap(VALUE source, OnigCaseFoldType *flags, rb_encoding *enc)
8750{
8751 VALUE target;
8752
8753 const OnigUChar *source_current, *source_end;
8754 int target_length = 0;
8755 VALUE buffer_anchor;
8756 mapping_buffer *current_buffer = 0;
8757 mapping_buffer **pre_buffer;
8758 size_t buffer_count = 0;
8759 int buffer_length_or_invalid;
8760
8761 if (RSTRING_LEN(source) == 0) return str_duplicate(rb_cString, source);
8762
8763 source_current = (OnigUChar*)RSTRING_PTR(source);
8764 source_end = (OnigUChar*)RSTRING_END(source);
8765
8766 buffer_anchor = TypedData_Wrap_Struct(0, &mapping_buffer_type, 0);
8767 pre_buffer = (mapping_buffer **)&DATA_PTR(buffer_anchor);
8768 while (source_current < source_end) {
8769 /* increase multiplier using buffer count to converge quickly */
8770 size_t capa = (size_t)(source_end-source_current)*++buffer_count + CASE_MAPPING_ADDITIONAL_LENGTH;
8771 if (CASEMAP_DEBUG) {
8772 fprintf(stderr, "Buffer allocation, capa is %"PRIuSIZE"\n", capa); /* for tuning */
8773 }
8774 current_buffer = xmalloc(offsetof(mapping_buffer, space) + capa);
8775 *pre_buffer = current_buffer;
8776 pre_buffer = &current_buffer->next;
8777 current_buffer->next = NULL;
8778 current_buffer->capa = capa;
8779 buffer_length_or_invalid = enc->case_map(flags,
8780 &source_current, source_end,
8781 current_buffer->space,
8782 current_buffer->space+current_buffer->capa,
8783 enc);
8784 if (buffer_length_or_invalid < 0) {
8785 current_buffer = DATA_PTR(buffer_anchor);
8786 DATA_PTR(buffer_anchor) = 0;
8787 mapping_buffer_free(current_buffer);
8788 rb_raise(rb_eArgError, "input string invalid");
8789 }
8790 target_length += current_buffer->used = buffer_length_or_invalid;
8791 }
8792 if (CASEMAP_DEBUG) {
8793 fprintf(stderr, "Buffer count is %"PRIuSIZE"\n", buffer_count); /* for tuning */
8794 }
8795
8796 if (buffer_count==1) {
8797 target = rb_str_new((const char*)current_buffer->space, target_length);
8798 }
8799 else {
8800 char *target_current;
8801
8802 target = rb_str_new(0, target_length);
8803 target_current = RSTRING_PTR(target);
8804 current_buffer = DATA_PTR(buffer_anchor);
8805 while (current_buffer) {
8806 memcpy(target_current, current_buffer->space, current_buffer->used);
8807 target_current += current_buffer->used;
8808 current_buffer = current_buffer->next;
8809 }
8810 }
8811 current_buffer = DATA_PTR(buffer_anchor);
8812 DATA_PTR(buffer_anchor) = 0;
8813 mapping_buffer_free(current_buffer);
8814
8815 RB_GC_GUARD(buffer_anchor);
8816
8817 /* TODO: check about string terminator character */
8818 str_enc_copy_direct(target, source);
8819 /*ENC_CODERANGE_SET(mapped, cr);*/
8820
8821 return target;
8822}
8823
8824static VALUE
8825rb_str_ascii_casemap(VALUE source, VALUE target, OnigCaseFoldType *flags, rb_encoding *enc)
8826{
8827 const OnigUChar *source_current, *source_end;
8828 OnigUChar *target_current, *target_end;
8829 long old_length = RSTRING_LEN(source);
8830 int length_or_invalid;
8831
8832 if (old_length == 0) return Qnil;
8833
8834 source_current = (OnigUChar*)RSTRING_PTR(source);
8835 source_end = (OnigUChar*)RSTRING_END(source);
8836 if (source == target) {
8837 target_current = (OnigUChar*)source_current;
8838 target_end = (OnigUChar*)source_end;
8839 }
8840 else {
8841 target_current = (OnigUChar*)RSTRING_PTR(target);
8842 target_end = (OnigUChar*)RSTRING_END(target);
8843 }
8844
8845 length_or_invalid = onigenc_ascii_only_case_map(flags,
8846 &source_current, source_end,
8847 target_current, target_end, enc);
8848 if (length_or_invalid < 0)
8849 rb_raise(rb_eArgError, "input string invalid");
8850 if (CASEMAP_DEBUG && length_or_invalid != old_length) {
8851 fprintf(stderr, "problem with rb_str_ascii_casemap"
8852 "; old_length=%ld, new_length=%d\n", old_length, length_or_invalid);
8853 rb_raise(rb_eArgError, "internal problem with rb_str_ascii_casemap"
8854 "; old_length=%ld, new_length=%d\n", old_length, length_or_invalid);
8855 }
8856
8857 str_enc_copy(target, source);
8858
8859 return target;
8860}
8861
8862static bool
8863upcase_single(VALUE str)
8864{
8865 char *s = RSTRING_PTR(str), *send = RSTRING_END(str);
8866 bool modified = false;
8867
8868 while (s < send) {
8869 unsigned int c = *(unsigned char*)s;
8870
8871 if ('a' <= c && c <= 'z') {
8872 *s = 'A' + (c - 'a');
8873 modified = true;
8874 }
8875 s++;
8876 }
8877 return modified;
8878}
8879
8880/*
8881 * call-seq:
8882 * upcase!(mapping) -> self or nil
8883 *
8884 * Like String#upcase, except that:
8885 *
8886 * - Changes character casings in +self+ (not in a copy of +self+).
8887 * - Returns +self+ if any changes are made, +nil+ otherwise.
8888 *
8889 * Related: See {Modifying}[rdoc-ref:String@Modifying].
8890 */
8891
8892static VALUE
8893rb_str_upcase_bang(int argc, VALUE *argv, VALUE str)
8894{
8895 rb_encoding *enc;
8896 OnigCaseFoldType flags = ONIGENC_CASE_UPCASE;
8897
8898 flags = check_case_options(argc, argv, flags);
8899 str_modify_keep_cr(str);
8900 enc = str_true_enc(str);
8901 if (case_option_single_p(flags, enc, str)) {
8902 if (upcase_single(str))
8903 flags |= ONIGENC_CASE_MODIFIED;
8904 }
8905 else if (flags&ONIGENC_CASE_ASCII_ONLY)
8906 rb_str_ascii_casemap(str, str, &flags, enc);
8907 else
8908 str_shared_replace(str, rb_str_casemap(str, &flags, enc));
8909
8910 if (ONIGENC_CASE_MODIFIED&flags) return str;
8911 return Qnil;
8912}
8913
8914
8915/*
8916 * call-seq:
8917 * upcase(mapping = :ascii) -> new_string
8918 *
8919 * :include: doc/string/upcase.rdoc
8920 */
8921
8922static VALUE
8923rb_str_upcase(int argc, VALUE *argv, VALUE str)
8924{
8925 rb_encoding *enc;
8926 OnigCaseFoldType flags = ONIGENC_CASE_UPCASE;
8927 VALUE ret;
8928
8929 flags = check_case_options(argc, argv, flags);
8930 enc = str_true_enc(str);
8931 if (case_option_single_p(flags, enc, str)) {
8932 ret = rb_str_new(RSTRING_PTR(str), RSTRING_LEN(str));
8933 str_enc_copy_direct(ret, str);
8934 upcase_single(ret);
8935 }
8936 else if (flags&ONIGENC_CASE_ASCII_ONLY) {
8937 ret = rb_str_new(0, RSTRING_LEN(str));
8938 rb_str_ascii_casemap(str, ret, &flags, enc);
8939 }
8940 else {
8941 ret = rb_str_casemap(str, &flags, enc);
8942 }
8943
8944 return ret;
8945}
8946
8947static bool
8948downcase_single(VALUE str)
8949{
8950 char *s = RSTRING_PTR(str), *send = RSTRING_END(str);
8951 bool modified = false;
8952
8953 while (s < send) {
8954 unsigned int c = *(unsigned char*)s;
8955
8956 if ('A' <= c && c <= 'Z') {
8957 *s = 'a' + (c - 'A');
8958 modified = true;
8959 }
8960 s++;
8961 }
8962
8963 return modified;
8964}
8965
8966/*
8967 * call-seq:
8968 * downcase!(mapping) -> self or nil
8969 *
8970 * Like String#downcase, except that:
8971 *
8972 * - Changes character casings in +self+ (not in a copy of +self+).
8973 * - Returns +self+ if any changes are made, +nil+ otherwise.
8974 *
8975 * Related: See {Modifying}[rdoc-ref:String@Modifying].
8976 */
8977
8978static VALUE
8979rb_str_downcase_bang(int argc, VALUE *argv, VALUE str)
8980{
8981 rb_encoding *enc;
8982 OnigCaseFoldType flags = ONIGENC_CASE_DOWNCASE;
8983
8984 flags = check_case_options(argc, argv, flags);
8985 str_modify_keep_cr(str);
8986 enc = str_true_enc(str);
8987 if (case_option_single_p(flags, enc, str)) {
8988 if (downcase_single(str))
8989 flags |= ONIGENC_CASE_MODIFIED;
8990 }
8991 else if (flags&ONIGENC_CASE_ASCII_ONLY)
8992 rb_str_ascii_casemap(str, str, &flags, enc);
8993 else
8994 str_shared_replace(str, rb_str_casemap(str, &flags, enc));
8995
8996 if (ONIGENC_CASE_MODIFIED&flags) return str;
8997 return Qnil;
8998}
8999
9000
9001/*
9002 * call-seq:
9003 * downcase(mapping = :ascii) -> new_string
9004 *
9005 * :include: doc/string/downcase.rdoc
9006 *
9007 */
9008
9009static VALUE
9010rb_str_downcase(int argc, VALUE *argv, VALUE str)
9011{
9012 rb_encoding *enc;
9013 OnigCaseFoldType flags = ONIGENC_CASE_DOWNCASE;
9014 VALUE ret;
9015
9016 flags = check_case_options(argc, argv, flags);
9017 enc = str_true_enc(str);
9018 if (case_option_single_p(flags, enc, str)) {
9019 ret = rb_str_new(RSTRING_PTR(str), RSTRING_LEN(str));
9020 str_enc_copy_direct(ret, str);
9021 downcase_single(ret);
9022 }
9023 else if (flags&ONIGENC_CASE_ASCII_ONLY) {
9024 ret = rb_str_new(0, RSTRING_LEN(str));
9025 rb_str_ascii_casemap(str, ret, &flags, enc);
9026 }
9027 else {
9028 ret = rb_str_casemap(str, &flags, enc);
9029 }
9030
9031 return ret;
9032}
9033
9034static bool
9035capitalize_single(VALUE str)
9036{
9037 char *s = RSTRING_PTR(str), *send = RSTRING_END(str);
9038 bool modified = false;
9039
9040 if (s < send) {
9041 unsigned int c = (unsigned char)*s;
9042
9043 if ('a' <= c && c <= 'z') {
9044 *s = 'A' + (c - 'a');
9045 modified = true;
9046 }
9047 s++;
9048 }
9049 while (s < send) {
9050 unsigned int c = (unsigned char)*s;
9051
9052 if ('A' <= c && c <= 'Z') {
9053 *s = 'a' + (c - 'A');
9054 modified = true;
9055 }
9056 s++;
9057 }
9058
9059 return modified;
9060}
9061
9062/*
9063 * call-seq:
9064 * capitalize!(mapping = :ascii) -> self or nil
9065 *
9066 * Like String#capitalize, except that:
9067 *
9068 * - Changes character casings in +self+ (not in a copy of +self+).
9069 * - Returns +self+ if any changes are made, +nil+ otherwise.
9070 *
9071 * Related: See {Modifying}[rdoc-ref:String@Modifying].
9072 */
9073
9074static VALUE
9075rb_str_capitalize_bang(int argc, VALUE *argv, VALUE str)
9076{
9077 rb_encoding *enc;
9078 OnigCaseFoldType flags = ONIGENC_CASE_UPCASE | ONIGENC_CASE_TITLECASE;
9079
9080 flags = check_case_options(argc, argv, flags);
9081 str_modify_keep_cr(str);
9082 enc = str_true_enc(str);
9083 if (RSTRING_LEN(str) == 0 || !RSTRING_PTR(str)) return Qnil;
9084 if (case_option_single_p(flags, enc, str)) {
9085 if (capitalize_single(str))
9086 flags |= ONIGENC_CASE_MODIFIED;
9087 }
9088 else if (flags&ONIGENC_CASE_ASCII_ONLY)
9089 rb_str_ascii_casemap(str, str, &flags, enc);
9090 else
9091 str_shared_replace(str, rb_str_casemap(str, &flags, enc));
9092
9093 if (ONIGENC_CASE_MODIFIED&flags) return str;
9094 return Qnil;
9095}
9096
9097
9098/*
9099 * call-seq:
9100 * capitalize(mapping = :ascii) -> new_string
9101 *
9102 * :include: doc/string/capitalize.rdoc
9103 *
9104 */
9105
9106static VALUE
9107rb_str_capitalize(int argc, VALUE *argv, VALUE str)
9108{
9109 rb_encoding *enc;
9110 OnigCaseFoldType flags = ONIGENC_CASE_UPCASE | ONIGENC_CASE_TITLECASE;
9111 VALUE ret;
9112
9113 flags = check_case_options(argc, argv, flags);
9114 enc = str_true_enc(str);
9115 if (RSTRING_LEN(str) == 0 || !RSTRING_PTR(str)) return str;
9116 if (case_option_single_p(flags, enc, str)) {
9117 ret = rb_str_new(RSTRING_PTR(str), RSTRING_LEN(str));
9118 str_enc_copy_direct(ret, str);
9119 capitalize_single(ret);
9120 }
9121 else if (flags&ONIGENC_CASE_ASCII_ONLY) {
9122 ret = rb_str_new(0, RSTRING_LEN(str));
9123 rb_str_ascii_casemap(str, ret, &flags, enc);
9124 }
9125 else {
9126 ret = rb_str_casemap(str, &flags, enc);
9127 }
9128 return ret;
9129}
9130
9131
9132/*
9133 * call-seq:
9134 * swapcase!(mapping) -> self or nil
9135 *
9136 * Like String#swapcase, except that:
9137 *
9138 * - Changes are made to +self+, not to copy of +self+.
9139 * - Returns +self+ if any changes are made, +nil+ otherwise.
9140 *
9141 * Related: see {Modifying}[rdoc-ref:String@Modifying].
9142 */
9143
9144static VALUE
9145rb_str_swapcase_bang(int argc, VALUE *argv, VALUE str)
9146{
9147 rb_encoding *enc;
9148 OnigCaseFoldType flags = ONIGENC_CASE_UPCASE | ONIGENC_CASE_DOWNCASE;
9149
9150 flags = check_case_options(argc, argv, flags);
9151 str_modify_keep_cr(str);
9152 enc = str_true_enc(str);
9153 if (flags&ONIGENC_CASE_ASCII_ONLY)
9154 rb_str_ascii_casemap(str, str, &flags, enc);
9155 else
9156 str_shared_replace(str, rb_str_casemap(str, &flags, enc));
9157
9158 if (ONIGENC_CASE_MODIFIED&flags) return str;
9159 return Qnil;
9160}
9161
9162
9163/*
9164 * call-seq:
9165 * swapcase(mapping = :ascii) -> new_string
9166 *
9167 * :include: doc/string/swapcase.rdoc
9168 *
9169 */
9170
9171static VALUE
9172rb_str_swapcase(int argc, VALUE *argv, VALUE str)
9173{
9174 rb_encoding *enc;
9175 OnigCaseFoldType flags = ONIGENC_CASE_UPCASE | ONIGENC_CASE_DOWNCASE;
9176 VALUE ret;
9177
9178 flags = check_case_options(argc, argv, flags);
9179 enc = str_true_enc(str);
9180 if (RSTRING_LEN(str) == 0 || !RSTRING_PTR(str)) return str_duplicate(rb_cString, str);
9181 if (flags&ONIGENC_CASE_ASCII_ONLY) {
9182 ret = rb_str_new(0, RSTRING_LEN(str));
9183 rb_str_ascii_casemap(str, ret, &flags, enc);
9184 }
9185 else {
9186 ret = rb_str_casemap(str, &flags, enc);
9187 }
9188 return ret;
9189}
9190
9191typedef unsigned char *USTR;
9192
9193struct tr {
9194 int gen;
9195 unsigned int now, max;
9196 const char *p, *pend;
9197};
9198
9199static unsigned int
9200trnext(struct tr *t, rb_encoding *enc)
9201{
9202 int n;
9203
9204 for (;;) {
9205 nextpart:
9206 if (!t->gen) {
9207 if (t->p == t->pend) return -1;
9208 if (rb_enc_ascget(t->p, t->pend, &n, enc) == '\\' && t->p + n < t->pend) {
9209 t->p += n;
9210 }
9211 t->now = rb_enc_codepoint_len(t->p, t->pend, &n, enc);
9212 t->p += n;
9213 if (rb_enc_ascget(t->p, t->pend, &n, enc) == '-' && t->p + n < t->pend) {
9214 t->p += n;
9215 if (t->p < t->pend) {
9216 unsigned int c = rb_enc_codepoint_len(t->p, t->pend, &n, enc);
9217 t->p += n;
9218 if (t->now > c) {
9219 if (t->now < 0x80 && c < 0x80) {
9220 rb_raise(rb_eArgError,
9221 "invalid range \"%c-%c\" in string transliteration",
9222 t->now, c);
9223 }
9224 else {
9225 rb_raise(rb_eArgError, "invalid range in string transliteration");
9226 }
9227 continue; /* not reached */
9228 }
9229 else if (t->now < c) {
9230 t->gen = 1;
9231 t->max = c;
9232 }
9233 }
9234 }
9235 return t->now;
9236 }
9237 else {
9238 while (ONIGENC_CODE_TO_MBCLEN(enc, ++t->now) <= 0) {
9239 if (t->now == t->max) {
9240 t->gen = 0;
9241 goto nextpart;
9242 }
9243 }
9244 if (t->now < t->max) {
9245 return t->now;
9246 }
9247 else {
9248 t->gen = 0;
9249 return t->max;
9250 }
9251 }
9252 }
9253}
9254
9255static VALUE rb_str_delete_bang(int,VALUE*,VALUE);
9256
9257static VALUE
9258tr_trans(VALUE str, VALUE src, VALUE repl, int sflag)
9259{
9260 const unsigned int errc = -1;
9261 unsigned int trans[256];
9262 rb_encoding *enc, *e1, *e2;
9263 struct tr trsrc, trrepl;
9264 int cflag = 0;
9265 unsigned int c, c0, last = 0;
9266 int modify = 0, i, l;
9267 unsigned char *s, *send;
9268 VALUE hash = 0;
9269 int singlebyte = single_byte_optimizable(str);
9270 int termlen;
9271 int cr;
9272
9273#define CHECK_IF_ASCII(c) \
9274 (void)((cr == ENC_CODERANGE_7BIT && !rb_isascii(c)) ? \
9275 (cr = ENC_CODERANGE_VALID) : 0)
9276
9277 StringValue(src);
9278 StringValue(repl);
9279 if (RSTRING_LEN(str) == 0 || !RSTRING_PTR(str)) return Qnil;
9280 if (RSTRING_LEN(repl) == 0) {
9281 return rb_str_delete_bang(1, &src, str);
9282 }
9283
9284 cr = ENC_CODERANGE(str);
9285 e1 = rb_enc_check(str, src);
9286 e2 = rb_enc_check(str, repl);
9287 if (e1 == e2) {
9288 enc = e1;
9289 }
9290 else {
9291 enc = rb_enc_check(src, repl);
9292 }
9293 trsrc.p = RSTRING_PTR(src); trsrc.pend = trsrc.p + RSTRING_LEN(src);
9294 if (RSTRING_LEN(src) > 1 &&
9295 rb_enc_ascget(trsrc.p, trsrc.pend, &l, enc) == '^' &&
9296 trsrc.p + l < trsrc.pend) {
9297 cflag = 1;
9298 trsrc.p += l;
9299 }
9300 trrepl.p = RSTRING_PTR(repl);
9301 trrepl.pend = trrepl.p + RSTRING_LEN(repl);
9302 trsrc.gen = trrepl.gen = 0;
9303 trsrc.now = trrepl.now = 0;
9304 trsrc.max = trrepl.max = 0;
9305
9306 if (cflag) {
9307 for (i=0; i<256; i++) {
9308 trans[i] = 1;
9309 }
9310 while ((c = trnext(&trsrc, enc)) != errc) {
9311 if (c < 256) {
9312 trans[c] = errc;
9313 }
9314 else {
9315 if (!hash) hash = rb_hash_new();
9316 rb_hash_aset(hash, UINT2NUM(c), Qtrue);
9317 }
9318 }
9319 while ((c = trnext(&trrepl, enc)) != errc)
9320 /* retrieve last replacer */;
9321 last = trrepl.now;
9322 for (i=0; i<256; i++) {
9323 if (trans[i] != errc) {
9324 trans[i] = last;
9325 }
9326 }
9327 }
9328 else {
9329 unsigned int r;
9330
9331 for (i=0; i<256; i++) {
9332 trans[i] = errc;
9333 }
9334 while ((c = trnext(&trsrc, enc)) != errc) {
9335 r = trnext(&trrepl, enc);
9336 if (r == errc) r = trrepl.now;
9337 if (c < 256) {
9338 trans[c] = r;
9339 if (rb_enc_codelen(r, enc) != 1) singlebyte = 0;
9340 }
9341 else {
9342 if (!hash) hash = rb_hash_new();
9343 rb_hash_aset(hash, UINT2NUM(c), UINT2NUM(r));
9344 }
9345 }
9346 }
9347
9348 if (cr == ENC_CODERANGE_VALID && rb_enc_asciicompat(e1))
9349 cr = ENC_CODERANGE_7BIT;
9350 str_modify_keep_cr(str);
9351 s = (unsigned char *)RSTRING_PTR(str); send = (unsigned char *)RSTRING_END(str);
9352 termlen = rb_enc_mbminlen(enc);
9353 if (sflag) {
9354 int clen, tlen;
9355 long offset, max = RSTRING_LEN(str);
9356 unsigned int save = -1;
9357 unsigned char *buf = ALLOC_N(unsigned char, max + termlen), *t = buf;
9358
9359 while (s < send) {
9360 int may_modify = 0;
9361
9362 int r = rb_enc_precise_mbclen((char *)s, (char *)send, e1);
9363 if (!MBCLEN_CHARFOUND_P(r)) {
9364 SIZED_FREE_N(buf, max + termlen);
9365 rb_raise(rb_eArgError, "invalid byte sequence in %s", rb_enc_name(e1));
9366 }
9367 clen = MBCLEN_CHARFOUND_LEN(r);
9368 c0 = c = rb_enc_mbc_to_codepoint((char *)s, (char *)send, e1);
9369
9370 tlen = enc == e1 ? clen : rb_enc_codelen(c, enc);
9371
9372 s += clen;
9373 if (c < 256) {
9374 c = trans[c];
9375 }
9376 else if (hash) {
9377 VALUE tmp = rb_hash_lookup(hash, UINT2NUM(c));
9378 if (NIL_P(tmp)) {
9379 if (cflag) c = last;
9380 else c = errc;
9381 }
9382 else if (cflag) c = errc;
9383 else c = NUM2INT(tmp);
9384 }
9385 else {
9386 c = errc;
9387 }
9388 if (c != (unsigned int)-1) {
9389 if (save == c) {
9390 CHECK_IF_ASCII(c);
9391 continue;
9392 }
9393 save = c;
9394 tlen = rb_enc_codelen(c, enc);
9395 modify = 1;
9396 }
9397 else {
9398 save = -1;
9399 c = c0;
9400 if (enc != e1) may_modify = 1;
9401 }
9402 if ((offset = t - buf) + tlen > max) {
9403 size_t MAYBE_UNUSED(old) = max + termlen;
9404 max = offset + tlen + (send - s);
9405 SIZED_REALLOC_N(buf, unsigned char, max + termlen, old);
9406 t = buf + offset;
9407 }
9408 rb_enc_mbcput(c, t, enc);
9409 if (may_modify && memcmp(s, t, tlen) != 0) {
9410 modify = 1;
9411 }
9412 CHECK_IF_ASCII(c);
9413 t += tlen;
9414 }
9415 if (!STR_EMBED_P(str)) {
9416 SIZED_FREE_N(STR_HEAP_PTR(str), STR_HEAP_SIZE(str));
9417 }
9418 TERM_FILL((char *)t, termlen);
9419 RSTRING(str)->as.heap.ptr = (char *)buf;
9420 STR_SET_LEN(str, t - buf);
9421 STR_SET_NOEMBED(str);
9422 RSTRING(str)->as.heap.aux.capa = max;
9423 }
9424 else if (rb_enc_mbmaxlen(enc) == 1 || (singlebyte && !hash)) {
9425 while (s < send) {
9426 c = (unsigned char)*s;
9427 if (trans[c] != errc) {
9428 if (!cflag) {
9429 c = trans[c];
9430 *s = c;
9431 modify = 1;
9432 }
9433 else {
9434 *s = last;
9435 modify = 1;
9436 }
9437 }
9438 CHECK_IF_ASCII(c);
9439 s++;
9440 }
9441 }
9442 else {
9443 int clen, tlen;
9444 long offset, max = (long)((send - s) * 1.2);
9445 unsigned char *buf = ALLOC_N(unsigned char, max + termlen), *t = buf;
9446
9447 while (s < send) {
9448 int may_modify = 0;
9449
9450 int r = rb_enc_precise_mbclen((char *)s, (char *)send, e1);
9451 if (!MBCLEN_CHARFOUND_P(r)) {
9452 SIZED_FREE_N(buf, max + termlen);
9453 rb_raise(rb_eArgError, "invalid byte sequence in %s", rb_enc_name(e1));
9454 }
9455 clen = MBCLEN_CHARFOUND_LEN(r);
9456 c0 = c = rb_enc_mbc_to_codepoint((char *)s, (char *)send, e1);
9457
9458 tlen = enc == e1 ? clen : rb_enc_codelen(c, enc);
9459
9460 if (c < 256) {
9461 c = trans[c];
9462 }
9463 else if (hash) {
9464 VALUE tmp = rb_hash_lookup(hash, UINT2NUM(c));
9465 if (NIL_P(tmp)) {
9466 if (cflag) c = last;
9467 else c = errc;
9468 }
9469 else if (cflag) c = errc;
9470 else c = NUM2INT(tmp);
9471 }
9472 else {
9473 c = cflag ? last : errc;
9474 }
9475 if (c != errc) {
9476 tlen = rb_enc_codelen(c, enc);
9477 modify = 1;
9478 }
9479 else {
9480 c = c0;
9481 if (enc != e1) may_modify = 1;
9482 }
9483 if ((offset = t - buf) + tlen > max) {
9484 size_t MAYBE_UNUSED(old) = max + termlen;
9485 max = offset + tlen + (long)((send - s) * 1.2);
9486 SIZED_REALLOC_N(buf, unsigned char, max + termlen, old);
9487 t = buf + offset;
9488 }
9489
9490 rb_enc_mbcput(c, t, enc);
9491 if (may_modify && memcmp(s, t, tlen) != 0) {
9492 modify = 1;
9493 }
9494 CHECK_IF_ASCII(c);
9495 s += clen;
9496 t += tlen;
9497 }
9498 if (!STR_EMBED_P(str)) {
9499 SIZED_FREE_N(STR_HEAP_PTR(str), STR_HEAP_SIZE(str));
9500 }
9501 TERM_FILL((char *)t, termlen);
9502 RSTRING(str)->as.heap.ptr = (char *)buf;
9503 STR_SET_LEN(str, t - buf);
9504 STR_SET_NOEMBED(str);
9505 RSTRING(str)->as.heap.aux.capa = max;
9506 }
9507
9508 if (modify) {
9509 if (cr != ENC_CODERANGE_BROKEN)
9510 ENC_CODERANGE_SET(str, cr);
9511 rb_enc_associate(str, enc);
9512 return str;
9513 }
9514 return Qnil;
9515}
9516
9518 unsigned char *buf;
9519 unsigned char *ptr;
9520 size_t capa;
9521 size_t initial_capa;
9522};
9523
9524static inline void
9525tr_buffer_init(struct tr_buffer *buffer, size_t initial_capa)
9526{
9527 if (initial_capa < 32) {
9528 initial_capa = 32;
9529 }
9530 *buffer = (struct tr_buffer){ .initial_capa = initial_capa };
9531}
9532
9533static inline void
9534tr_buffer_ensure_capa(struct tr_buffer *buffer, size_t extra_capa)
9535{
9536 size_t offset = buffer->ptr - buffer->buf;
9537 size_t required_capa = offset + extra_capa;
9538 if (UNLIKELY(buffer->capa < required_capa)) {
9539 size_t new_capa = buffer->capa ? buffer->capa : buffer->initial_capa;
9540 RUBY_ASSERT(new_capa >= 32); // Lower would cause infinite loop
9541 while (new_capa < required_capa) {
9542 new_capa = (size_t)(new_capa * 1.2);
9543 }
9544 SIZED_REALLOC_N(buffer->buf, unsigned char, new_capa, buffer->capa);
9545 buffer->ptr = buffer->buf + offset;
9546 buffer->capa = new_capa;
9547 }
9548}
9549
9550static inline void
9551tr_buffer_append(struct tr_buffer *buffer, const unsigned char *ptr, size_t len)
9552{
9553 if (len) {
9554 tr_buffer_ensure_capa(buffer, len);
9555 memcpy(buffer->ptr, ptr, len);
9556 buffer->ptr += len;
9557 }
9558}
9559
9560static inline void
9561tr_buffer_append_str(struct tr_buffer *buffer, VALUE str)
9562{
9563 tr_buffer_append(buffer, (unsigned char *)RSTRING_PTR(str), RSTRING_LEN(str));
9564}
9565
9566static inline void
9567tr_buffer_mbcput(struct tr_buffer *buffer, int codepoint, rb_encoding *enc)
9568{
9569 tr_buffer_ensure_capa(buffer, 4);
9570 buffer->ptr += rb_enc_mbcput(codepoint, buffer->ptr, enc);
9571}
9572
9573static inline void
9574tr_buffer_free(struct tr_buffer *buffer)
9575{
9576 if (buffer->buf) {
9577 SIZED_FREE_N(buffer->buf, buffer->capa);
9578 }
9579}
9580
9581struct tr_pair {
9582 VALUE search;
9583 VALUE replace;
9584};
9585
9587 struct tr_pair *pairs;
9588 size_t index;
9589 rb_encoding *enc;
9590 int cr;
9591};
9592
9593static int
9594tr_trans_pairs_coerce_i(st_data_t key, st_data_t value, st_data_t _args)
9595{
9596 struct tr_trans_pairs_coerce_args *args = (struct tr_trans_pairs_coerce_args *)_args;
9597 struct tr_pair *pair = &args->pairs[args->index];
9598 args->index++;
9599
9600 VALUE search = (VALUE)key;
9601 VALUE replace = (VALUE)value;
9602 StringValue(search);
9603 StringValue(replace);
9604
9605 if (RSTRING_LEN(search) != 1 && str_strlen(search, NULL) != 1) {
9606 rb_raise(rb_eArgError, "keys must be of size 1"); // TODO: better error message
9607 }
9608
9609 args->enc = rb_enc_check_multi_str(args->enc, &args->cr, search);
9610 args->enc = rb_enc_check_multi_str(args->enc, &args->cr, replace);
9611
9612 pair->search = search;
9613 pair->replace = replace;
9614 return ST_CONTINUE;
9615}
9616
9617#define TR_TRANS_PAIRS_SIMD_MAX_NEEDLES 16
9618
9620 const unsigned char *s;
9621 const unsigned char *send;
9622
9623#ifdef HAVE_SIMD
9624 unsigned char needles[TR_TRANS_PAIRS_SIMD_MAX_NEEDLES];
9625 unsigned int needles_count;
9626#ifdef HAVE_SIMD_NEON
9627 uint64_t matches_bitmap;
9628#endif
9629#ifdef HAVE_SIMD_SSE2
9630 int matches_bitmap;
9631#endif
9632#endif
9633
9634 VALUE trans_table[256];
9635};
9636
9637static inline VALUE
9638tr_trans_pairs_search_basic(struct tr_trans_pairs_search *search)
9639{
9640 while (search->s < search->send) {
9641 VALUE repl = search->trans_table[*search->s];
9642 if (UNLIKELY(repl)) {
9643 return repl;
9644 }
9645
9646 search->s++;
9647 }
9648
9649 return 0;
9650}
9651
9652#ifdef HAVE_SIMD_SSE2
9653static inline VALUE
9654tr_trans_pairs_next_match_sse2(struct tr_trans_pairs_search *search)
9655{
9656 RUBY_ASSERT(search->matches_bitmap > 0);
9657 size_t trailing_zeros = (size_t)ntz_int32(search->matches_bitmap);
9658
9659 RUBY_ASSERT(trailing_zeros < (sizeof(search->matches_bitmap) * CHAR_BIT));
9660 search->matches_bitmap >>= trailing_zeros;
9661 search->s += trailing_zeros;
9662
9663 RUBY_ASSERT(search->s <= search->send);
9664 return search->trans_table[*search->s];
9665}
9666
9667static inline VALUE
9668tr_trans_pairs_search_sse2(struct tr_trans_pairs_search *search)
9669{
9670 const unsigned int needles_count = search->needles_count;
9671 if (needles_count) {
9672 RBIMPL_ASSERT_OR_ASSUME(needles_count <= TR_TRANS_PAIRS_SIMD_MAX_NEEDLES);
9673
9674 if (search->matches_bitmap) {
9675 return tr_trans_pairs_next_match_sse2(search);
9676 }
9677
9678 if ((size_t)(search->send - search->s) >= sizeof(__m128i)) {
9679 unsigned int i;
9680 __m128i masks[TR_TRANS_PAIRS_SIMD_MAX_NEEDLES];
9681 for (i = 0; i < needles_count; i++) {
9682 masks[i] = _mm_set1_epi8(search->needles[i]);
9683 }
9684
9685 do {
9686 const __m128i bytes = _mm_loadu_si128((__m128i const *)search->s);
9687
9688 __m128i matches[TR_TRANS_PAIRS_SIMD_MAX_NEEDLES];
9689 for (i = 0; i < needles_count; i++) {
9690 matches[i] = _mm_cmpeq_epi8(bytes, masks[i]);
9691 }
9692
9693 for (i = 1; i < needles_count; i++) {
9694 matches[0] = _mm_or_si128(matches[0], matches[i]);
9695 }
9696
9697 const int bitmap = _mm_movemask_epi8(matches[0]);
9698
9699 if (bitmap) {
9700 search->matches_bitmap = bitmap;
9701 return tr_trans_pairs_next_match_sse2(search);
9702 }
9703 search->s += sizeof(__m128i);
9704 } while ((size_t)(search->send - search->s) >= sizeof(__m128i));
9705 }
9706 }
9707 return tr_trans_pairs_search_basic(search);
9708}
9709
9710#define tr_trans_pairs_search_impl tr_trans_pairs_search_sse2
9711#endif
9712
9713#ifdef HAVE_SIMD_NEON
9714static inline VALUE
9715tr_trans_pairs_next_match_neon(struct tr_trans_pairs_search *search)
9716{
9717 RUBY_ASSERT(search->matches_bitmap > 0);
9718 size_t trailing_zeros = (size_t)ntz_int64(search->matches_bitmap);
9719
9720 // uint64_t >>= 64 would be undefined behaviour
9721 RUBY_ASSERT(trailing_zeros < (sizeof(search->matches_bitmap) * CHAR_BIT));
9722 search->matches_bitmap >>= trailing_zeros;
9723 search->s += trailing_zeros / 4;
9724
9725 RUBY_ASSERT(search->s <= search->send);
9726 return search->trans_table[*search->s];
9727}
9728
9729static inline VALUE
9730tr_trans_pairs_search_neon(struct tr_trans_pairs_search *search)
9731{
9732 const unsigned int needles_count = search->needles_count;
9733 if (needles_count) {
9734 RBIMPL_ASSERT_OR_ASSUME(needles_count <= TR_TRANS_PAIRS_SIMD_MAX_NEEDLES);
9735
9736 if (search->matches_bitmap) {
9737 return tr_trans_pairs_next_match_neon(search);
9738 }
9739
9740 if ((size_t)(search->send - search->s) >= sizeof(uint8x16_t)) {
9741 unsigned int i;
9742 uint8x16_t masks[TR_TRANS_PAIRS_SIMD_MAX_NEEDLES];
9743 for (i = 0; i < needles_count; i++) {
9744 masks[i] = vdupq_n_u8(search->needles[i]);
9745 }
9746
9747 do {
9748 const uint8x16_t bytes = vld1q_u8(search->s);
9749
9750 uint8x16_t matches[TR_TRANS_PAIRS_SIMD_MAX_NEEDLES];
9751 for (i = 0; i < needles_count; i++) {
9752 matches[i] = vceqq_u8(bytes, masks[i]);
9753 }
9754
9755 for (i = 1; i < needles_count; i++) {
9756 matches[0] = vorrq_u8(matches[0], matches[i]);
9757 }
9758
9759 const uint8x8_t res = vshrn_n_u16(vreinterpretq_u16_u8(matches[0]), 4);
9760 const uint64_t bitmap = vget_lane_u64(vreinterpret_u64_u8(res), 0);
9761
9762 if (bitmap) {
9763 search->matches_bitmap = bitmap & 0x8888888888888888ull;
9764 return tr_trans_pairs_next_match_neon(search);
9765 }
9766 search->s += sizeof(uint8x16_t);
9767 } while ((size_t)(search->send - search->s) >= sizeof(uint8x16_t));
9768 }
9769 }
9770 return tr_trans_pairs_search_basic(search);
9771}
9772
9773#define tr_trans_pairs_search_impl tr_trans_pairs_search_neon
9774#endif
9775
9776#ifndef tr_trans_pairs_search_impl
9777#define tr_trans_pairs_search_impl tr_trans_pairs_search_basic
9778#endif
9779
9780static inline void
9781tr_trans_pairs_consume_match(struct tr_trans_pairs_search *search)
9782{
9783 search->s++;
9784#ifdef HAVE_SIMD
9785 search->matches_bitmap >>= 1;
9786#endif
9787}
9788
9789static VALUE
9790tr_trans_pairs(VALUE str, VALUE pairs_val)
9791{
9792 Check_Type(pairs_val, T_HASH);
9793 size_t pairs_count = RHASH_SIZE(pairs_val);
9794 mustnot_broken(str);
9795 rb_str_modify(str);
9796
9797 if (RSTRING_LEN(str) == 0 || !RSTRING_PTR(str) || pairs_count == 0) return Qnil;
9798
9799 VALUE pairs_handle;
9800 struct tr_pair *pairs = ALLOCV_N(struct tr_pair, pairs_handle, pairs_count);
9801
9802 int cr = rb_enc_str_coderange(str);
9803 rb_encoding *enc = rb_str_enc_get(str);
9804
9805 struct tr_trans_pairs_coerce_args coerce_args = {
9806 .pairs = pairs,
9807 .enc = enc,
9808 .cr = cr,
9809 };
9810 rb_hash_foreach(pairs_val, tr_trans_pairs_coerce_i, (VALUE)&coerce_args);
9811 rb_encoding *e1 = coerce_args.enc;
9812
9813 /* Keys could be deleted from pairs_val during rb_hash_foreach when coercing
9814 * the keys/values, so we need to update pairs_count to the number of pairs we
9815 * were actually able to extract from pairs_val. */
9816 pairs_count = coerce_args.index;
9817
9818 VALUE hash = 0;
9819
9820 const unsigned char *sstart = (unsigned char *)RSTRING_PTR(str);
9821 long str_len = RSTRING_LEN(str);
9822 int termlen = rb_enc_mbminlen(e1);
9823
9824 struct tr_buffer buffer;
9825 tr_buffer_init(&buffer, str_len);
9826 bool modify = false;
9827
9828 if (RB_LIKELY(rb_str_encindex_fastpath(rb_enc_to_index(e1)))) {
9829
9830 struct tr_trans_pairs_search search = {
9831 .s = sstart,
9832 .send = sstart + str_len,
9833 };
9834
9835 for (size_t index = 0; index < pairs_count; index++) {
9836 struct tr_pair *pair = &pairs[index];
9837
9838 char *ptr = RSTRING_PTR(pair->search);
9839 unsigned int codepoint = rb_enc_mbc_to_codepoint(ptr, RSTRING_END(pair->search), e1);
9840
9841 const unsigned char first_byte = (unsigned char)*ptr;
9842
9843#ifdef HAVE_SIMD
9844 if (pairs_count <= TR_TRANS_PAIRS_SIMD_MAX_NEEDLES) {
9845 search.needles[index] = first_byte;
9846 search.needles_count++;
9847 }
9848#endif
9849
9850 if (rb_enc_codelen(codepoint, e1) == 1) {
9851 search.trans_table[first_byte] = pair->replace;
9852 }
9853 else {
9854 search.trans_table[first_byte] = Qundef;
9855 if (!hash) {
9856 hash = rb_obj_hide(rb_hash_new_capa(pairs_count));
9857 }
9858 rb_hash_aset(hash, UINT2NUM(codepoint), pair->replace);
9859 }
9860 }
9861
9862 const unsigned char *checkpoint = search.s;
9863 VALUE repl;
9864 while ((repl = tr_trans_pairs_search_impl(&search))) {
9865 int clen = 1;
9866
9867 if (UNLIKELY(repl == Qundef)) {
9868 unsigned int c = rb_enc_mbc_to_codepoint((char *)search.s, (char *)search.send, e1);
9869 clen = rb_enc_codelen(c, e1);
9870 repl = rb_hash_lookup2(hash, UINT2NUM(c), 0);
9871 if (!repl) {
9872 tr_trans_pairs_consume_match(&search);
9873 continue;
9874 }
9875 }
9877
9878 modify = true;
9879
9880 if (checkpoint < search.s) {
9881 tr_buffer_append(&buffer, checkpoint, search.s - checkpoint);
9882 }
9883 tr_buffer_append_str(&buffer, repl);
9884 checkpoint = search.s + clen;
9885 tr_trans_pairs_consume_match(&search);
9886
9887 if (cr == ENC_CODERANGE_7BIT && rb_enc_str_coderange(repl) != ENC_CODERANGE_7BIT) {
9889 }
9890 }
9891
9892 if (modify && checkpoint < search.s) {
9893 tr_buffer_append(&buffer, checkpoint, search.s - checkpoint);
9894 }
9895 }
9896 else {
9897 const unsigned char *s = sstart;
9898 const unsigned char *send = sstart + str_len;
9899
9900 hash = rb_obj_hide(rb_hash_new_capa(pairs_count));
9901
9902 for (size_t index = 0; index < pairs_count; index++) {
9903 struct tr_pair *pair = &pairs[index];
9904
9905 unsigned int codepoint = rb_enc_mbc_to_codepoint(RSTRING_PTR(pair->search), RSTRING_END(pair->search), e1);
9906 rb_hash_aset(hash, UINT2NUM(codepoint), pair->replace);
9907 }
9908
9909 while (s < send) {
9910 bool may_modify = false;
9911
9912 int r = rb_enc_precise_mbclen((char *)s, (char *)send, e1);
9913 if (!MBCLEN_CHARFOUND_P(r)) {
9914 tr_buffer_free(&buffer);
9915 rb_raise(rb_eArgError, "invalid byte sequence in %s", rb_enc_name(e1));
9916 }
9917 int clen = MBCLEN_CHARFOUND_LEN(r);
9918 unsigned int c = rb_enc_mbc_to_codepoint((char *)s, (char *)send, e1);
9919 unsigned int c0 = c;
9920
9921 long tlen = enc == e1 ? clen : rb_enc_codelen(c, e1);
9922
9923 VALUE replacement = rb_hash_lookup(hash, UINT2NUM(c));
9924 if (NIL_P(replacement)) {
9925 tlen = enc == e1 ? clen : rb_enc_codelen(c, enc);
9926 c = c0;
9927 if (enc != e1) may_modify = true;
9928 }
9929 else {
9930 tlen = RSTRING_LEN(replacement);
9931 modify = true;
9932 }
9933
9934 if (NIL_P(replacement)) {
9935 tr_buffer_mbcput(&buffer, c, enc);
9936 }
9937 else {
9938 tr_buffer_append_str(&buffer, replacement);
9939 }
9940
9941 if (may_modify && memcmp(s, buffer.ptr - tlen, tlen) != 0) {
9942 modify = true;
9943 }
9944
9945 if (cr == ENC_CODERANGE_7BIT && !rb_isascii(c)) {
9947 }
9948
9949 s += clen;
9950 }
9951 }
9952
9953 if (!modify) {
9954 return Qnil;
9955 }
9956
9957 if (!STR_EMBED_P(str)) {
9958 SIZED_FREE_N(STR_HEAP_PTR(str), STR_HEAP_SIZE(str));
9959 }
9960 tr_buffer_ensure_capa(&buffer, termlen);
9961 TERM_FILL((char *)buffer.ptr, termlen);
9962 RSTRING(str)->as.heap.ptr = (char *)buffer.buf;
9963 STR_SET_LEN(str, buffer.ptr - buffer.buf);
9964 STR_SET_NOEMBED(str);
9965 RSTRING(str)->as.heap.aux.capa = buffer.capa - termlen;
9966
9967 RB_GC_GUARD(hash);
9968
9969 if (cr != ENC_CODERANGE_BROKEN)
9970 ENC_CODERANGE_SET(str, cr);
9971 rb_enc_associate(str, e1);
9972 return str;
9973}
9974
9975/*
9976 * call-seq:
9977 * tr!(selector, replacements) -> self or nil
9978 * tr!(pairs) -> self or nil
9979 *
9980 * Like String#tr, except:
9981 *
9982 * - Performs substitutions in +self+ (not in a copy of +self+).
9983 * - Returns +self+ if any modifications were made, +nil+ otherwise.
9984 *
9985 * Related: {Modifying}[rdoc-ref:String@Modifying].
9986 */
9987
9988static VALUE
9989rb_str_tr_bang(int argc, VALUE *argv, VALUE str)
9990{
9991 rb_check_arity(argc, 1, 2);
9992
9993 if (argc == 1) {
9994 VALUE pairs = argv[0];
9995 return tr_trans_pairs(str, pairs);
9996 }
9997
9998 VALUE src = argv[0], repl = argv[1];
9999 return tr_trans(str, src, repl, 0);
10000}
10001
10002
10003/*
10004 * call-seq:
10005 * tr(selector, replacements) -> new_string
10006 * tr(pairs) -> new_string
10007 *
10008 * Accepts either a +selector+ and a +replacements+ string,
10009 * or a single +pairs+ Hash.
10010 *
10011 * When a +pairs+ Hash is provided the keys, returns a copy of +self+ with
10012 * the keys of the hash replaced by the values.
10013 *
10014 * - They keys must be strings containing a single codepoints.
10015 * - The values can be of any length.
10016 *
10017 * Example:
10018 *
10019 * 'hello'.tr('e' => 'er', 'l' => '', 'o' => 'o !') #=> "hero !"
10020 *
10021 * When +selector+ and +replacements+are provided, returns a copy of +self+
10022 * with each character specified by string +selector+ translated to the
10023 * corresponding character in string +replacements+.
10024 * The correspondence is _positional_:
10025 *
10026 * - Each occurrence of the first character specified by +selector+
10027 * is translated to the first character in +replacements+.
10028 * - Each occurrence of the second character specified by +selector+
10029 * is translated to the second character in +replacements+.
10030 * - And so on.
10031 *
10032 * Example:
10033 *
10034 * 'hello'.tr('el', 'ip') #=> "hippo"
10035 *
10036 * If +replacements+ is shorter than +selector+,
10037 * it is implicitly padded with its own last character:
10038 *
10039 * 'hello'.tr('aeiou', '-') # => "h-ll-"
10040 * 'hello'.tr('aeiou', 'AA-') # => "hAll-"
10041 *
10042 * Arguments +selector+ and +replacements+ must be valid character selectors
10043 * (see {Character Selectors}[rdoc-ref:character_selectors.rdoc]),
10044 * and may use any of its valid forms, including negation, ranges, and escapes:
10045 *
10046 * 'hello'.tr('^aeiou', '-') # => "-e--o" # Negation.
10047 * 'ibm'.tr('b-z', 'a-z') # => "hal" # Range.
10048 * 'hel^lo'.tr('\^aeiou', '-') # => "h-l-l-" # Escaped leading caret.
10049 * 'i-b-m'.tr('b\-z', 'a-z') # => "ibabm" # Escaped embedded hyphen.
10050 * 'foo\\bar'.tr('ab\\', 'XYZ') # => "fooZYXr" # Escaped backslash.
10051 *
10052 * Related: see {Converting to New String}[rdoc-ref:String@Converting+to+New+String].
10053 */
10054
10055static VALUE
10056rb_str_tr(int argc, VALUE *argv, VALUE str)
10057{
10058 rb_check_arity(argc, 1, 2);
10059
10060 str = str_duplicate(rb_cString, str);
10061
10062 if (argc == 1) {
10063 VALUE pairs = argv[0];
10064 VALUE result = tr_trans_pairs(str, pairs);
10065 if (NIL_P(result)) result = str;
10066 return str;
10067 }
10068
10069 VALUE src = argv[0], repl = argv[1];
10070 tr_trans(str, src, repl, 0);
10071 return str;
10072}
10073
10074#define TR_TABLE_MAX (UCHAR_MAX+1)
10075#define TR_TABLE_SIZE (TR_TABLE_MAX+1)
10076static void
10077tr_setup_table(VALUE str, char stable[TR_TABLE_SIZE], int first,
10078 VALUE *tablep, VALUE *ctablep, rb_encoding *enc)
10079{
10080 const unsigned int errc = -1;
10081 char buf[TR_TABLE_MAX];
10082 struct tr tr;
10083 unsigned int c;
10084 VALUE table = 0, ptable = 0;
10085 int i, l, cflag = 0;
10086
10087 tr.p = RSTRING_PTR(str); tr.pend = tr.p + RSTRING_LEN(str);
10088 tr.gen = tr.now = tr.max = 0;
10089
10090 if (RSTRING_LEN(str) > 1 && rb_enc_ascget(tr.p, tr.pend, &l, enc) == '^') {
10091 cflag = 1;
10092 tr.p += l;
10093 }
10094 if (first) {
10095 for (i=0; i<TR_TABLE_MAX; i++) {
10096 stable[i] = 1;
10097 }
10098 stable[TR_TABLE_MAX] = cflag;
10099 }
10100 else if (stable[TR_TABLE_MAX] && !cflag) {
10101 stable[TR_TABLE_MAX] = 0;
10102 }
10103 for (i=0; i<TR_TABLE_MAX; i++) {
10104 buf[i] = cflag;
10105 }
10106
10107 while ((c = trnext(&tr, enc)) != errc) {
10108 if (c < TR_TABLE_MAX) {
10109 buf[(unsigned char)c] = !cflag;
10110 }
10111 else {
10112 VALUE key = UINT2NUM(c);
10113
10114 if (!table && (first || *tablep || stable[TR_TABLE_MAX])) {
10115 if (cflag) {
10116 ptable = *ctablep;
10117 table = ptable ? ptable : rb_hash_new();
10118 *ctablep = table;
10119 }
10120 else {
10121 table = rb_hash_new();
10122 ptable = *tablep;
10123 *tablep = table;
10124 }
10125 }
10126 if (table && (!ptable || (cflag ^ !NIL_P(rb_hash_aref(ptable, key))))) {
10127 rb_hash_aset(table, key, Qtrue);
10128 }
10129 }
10130 }
10131 for (i=0; i<TR_TABLE_MAX; i++) {
10132 stable[i] = stable[i] && buf[i];
10133 }
10134 if (!table && !cflag) {
10135 *tablep = 0;
10136 }
10137}
10138
10139
10140static int
10141tr_find(unsigned int c, const char table[TR_TABLE_SIZE], VALUE del, VALUE nodel)
10142{
10143 if (c < TR_TABLE_MAX) {
10144 return table[c] != 0;
10145 }
10146 else {
10147 VALUE v = UINT2NUM(c);
10148
10149 if (del) {
10150 if (!NIL_P(rb_hash_lookup(del, v)) &&
10151 (!nodel || NIL_P(rb_hash_lookup(nodel, v)))) {
10152 return TRUE;
10153 }
10154 }
10155 else if (nodel && !NIL_P(rb_hash_lookup(nodel, v))) {
10156 return FALSE;
10157 }
10158 return table[TR_TABLE_MAX] ? TRUE : FALSE;
10159 }
10160}
10161
10162/*
10163 * call-seq:
10164 * delete!(*selectors) -> self or nil
10165 *
10166 * Like String#delete, but modifies +self+ in place;
10167 * returns +self+ if any characters were deleted, +nil+ otherwise.
10168 *
10169 * Related: see {Modifying}[rdoc-ref:String@Modifying].
10170 */
10171
10172static VALUE
10173rb_str_delete_bang(int argc, VALUE *argv, VALUE str)
10174{
10175 char squeez[TR_TABLE_SIZE];
10176 rb_encoding *enc = 0;
10177 char *s, *send, *t;
10178 VALUE del = 0, nodel = 0;
10179 int modify = 0;
10180 int i, ascompat, cr;
10181
10182 if (RSTRING_LEN(str) == 0 || !RSTRING_PTR(str)) return Qnil;
10184 for (i=0; i<argc; i++) {
10185 VALUE s = argv[i];
10186
10187 StringValue(s);
10188 enc = rb_enc_check(str, s);
10189 tr_setup_table(s, squeez, i==0, &del, &nodel, enc);
10190 }
10191
10192 str_modify_keep_cr(str);
10193 ascompat = rb_enc_asciicompat(enc);
10194 s = t = RSTRING_PTR(str);
10195 send = RSTRING_END(str);
10196 cr = ascompat ? ENC_CODERANGE_7BIT : ENC_CODERANGE_VALID;
10197 while (s < send) {
10198 unsigned int c;
10199 int clen;
10200
10201 if (ascompat && (c = *(unsigned char*)s) < 0x80) {
10202 if (squeez[c]) {
10203 modify = 1;
10204 }
10205 else {
10206 if (t != s) *t = c;
10207 t++;
10208 }
10209 s++;
10210 }
10211 else {
10212 c = rb_enc_codepoint_len(s, send, &clen, enc);
10213
10214 if (tr_find(c, squeez, del, nodel)) {
10215 modify = 1;
10216 }
10217 else {
10218 if (t != s) rb_enc_mbcput(c, t, enc);
10219 t += clen;
10221 }
10222 s += clen;
10223 }
10224 }
10225 TERM_FILL(t, TERM_LEN(str));
10226 STR_SET_LEN(str, t - RSTRING_PTR(str));
10227 ENC_CODERANGE_SET(str, cr);
10228
10229 if (modify) return str;
10230 return Qnil;
10231}
10232
10233
10234/*
10235 * call-seq:
10236 * delete(*selectors) -> new_string
10237 *
10238 * :include: doc/string/delete.rdoc
10239 *
10240 */
10241
10242static VALUE
10243rb_str_delete(int argc, VALUE *argv, VALUE str)
10244{
10245 str = str_duplicate(rb_cString, str);
10246 rb_str_delete_bang(argc, argv, str);
10247 return str;
10248}
10249
10250
10251/*
10252 * call-seq:
10253 * squeeze!(*selectors) -> self or nil
10254 *
10255 * Like String#squeeze, except that:
10256 *
10257 * - Characters are squeezed in +self+ (not in a copy of +self+).
10258 * - Returns +self+ if any changes are made, +nil+ otherwise.
10259 *
10260 * Related: See {Modifying}[rdoc-ref:String@Modifying].
10261 */
10262
10263static VALUE
10264rb_str_squeeze_bang(int argc, VALUE *argv, VALUE str)
10265{
10266 char squeez[TR_TABLE_SIZE];
10267 rb_encoding *enc = 0;
10268 VALUE del = 0, nodel = 0;
10269 unsigned char *s, *send, *t;
10270 int i, modify = 0;
10271 int ascompat, singlebyte = single_byte_optimizable(str);
10272 unsigned int save;
10273
10274 if (argc == 0) {
10275 enc = STR_ENC_GET(str);
10276 }
10277 else {
10278 for (i=0; i<argc; i++) {
10279 VALUE s = argv[i];
10280
10281 StringValue(s);
10282 enc = rb_enc_check(str, s);
10283 if (singlebyte && !single_byte_optimizable(s))
10284 singlebyte = 0;
10285 tr_setup_table(s, squeez, i==0, &del, &nodel, enc);
10286 }
10287 }
10288
10289 str_modify_keep_cr(str);
10290 s = t = (unsigned char *)RSTRING_PTR(str);
10291 if (!s || RSTRING_LEN(str) == 0) return Qnil;
10292 send = (unsigned char *)RSTRING_END(str);
10293 save = -1;
10294 ascompat = rb_enc_asciicompat(enc);
10295
10296 if (singlebyte) {
10297 while (s < send) {
10298 unsigned int c = *s++;
10299 if (c != save || (argc > 0 && !squeez[c])) {
10300 *t++ = save = c;
10301 }
10302 }
10303 }
10304 else {
10305 while (s < send) {
10306 unsigned int c;
10307 int clen;
10308
10309 if (ascompat && (c = *s) < 0x80) {
10310 if (c != save || (argc > 0 && !squeez[c])) {
10311 *t++ = save = c;
10312 }
10313 s++;
10314 }
10315 else {
10316 c = rb_enc_codepoint_len((char *)s, (char *)send, &clen, enc);
10317
10318 if (c != save || (argc > 0 && !tr_find(c, squeez, del, nodel))) {
10319 if (t != s) rb_enc_mbcput(c, t, enc);
10320 save = c;
10321 t += clen;
10322 }
10323 s += clen;
10324 }
10325 }
10326 }
10327
10328 TERM_FILL((char *)t, TERM_LEN(str));
10329 if ((char *)t - RSTRING_PTR(str) != RSTRING_LEN(str)) {
10330 STR_SET_LEN(str, (char *)t - RSTRING_PTR(str));
10331 modify = 1;
10332 }
10333
10334 if (modify) return str;
10335 return Qnil;
10336}
10337
10338
10339/*
10340 * call-seq:
10341 * squeeze(*selectors) -> new_string
10342 *
10343 * :include: doc/string/squeeze.rdoc
10344 *
10345 */
10346
10347static VALUE
10348rb_str_squeeze(int argc, VALUE *argv, VALUE str)
10349{
10350 str = str_duplicate(rb_cString, str);
10351 rb_str_squeeze_bang(argc, argv, str);
10352 return str;
10353}
10354
10355
10356/*
10357 * call-seq:
10358 * tr_s!(selector, replacements) -> self or nil
10359 *
10360 * Like String#tr_s, except:
10361 *
10362 * - Modifies +self+ in place (not a copy of +self+).
10363 * - Returns +self+ if any changes were made, +nil+ otherwise.
10364 *
10365 * Related: {Modifying}[rdoc-ref:String@Modifying].
10366 */
10367
10368static VALUE
10369rb_str_tr_s_bang(VALUE str, VALUE src, VALUE repl)
10370{
10371 return tr_trans(str, src, repl, 1);
10372}
10373
10374
10375/*
10376 * call-seq:
10377 * tr_s(selector, replacements) -> new_string
10378 *
10379 * Like String#tr, except:
10380 *
10381 * - Also squeezes the modified portions of the translated string;
10382 * see String#squeeze.
10383 * - Returns the translated and squeezed string.
10384 *
10385 * Examples:
10386 *
10387 * 'hello'.tr_s('l', 'r') #=> "hero"
10388 * 'hello'.tr_s('el', '-') #=> "h-o"
10389 * 'hello'.tr_s('el', 'hx') #=> "hhxo"
10390 *
10391 * Related: see {Converting to New String}[rdoc-ref:String@Converting+to+New+String].
10392 *
10393 */
10394
10395static VALUE
10396rb_str_tr_s(VALUE str, VALUE src, VALUE repl)
10397{
10398 str = str_duplicate(rb_cString, str);
10399 tr_trans(str, src, repl, 1);
10400 return str;
10401}
10402
10403
10404/*
10405 * call-seq:
10406 * count(*selectors) -> integer
10407 *
10408 * :include: doc/string/count.rdoc
10409 */
10410
10411static VALUE
10412rb_str_count(int argc, VALUE *argv, VALUE str)
10413{
10414 char table[TR_TABLE_SIZE];
10415 rb_encoding *enc = 0;
10416 VALUE del = 0, nodel = 0, tstr;
10417 const char *s, *send;
10418 int i;
10419 int ascompat;
10420 size_t n = 0;
10421
10423
10424 tstr = argv[0];
10425 StringValue(tstr);
10426 enc = rb_enc_check(str, tstr);
10427 if (argc == 1) {
10428 const char *ptstr;
10429 if (RSTRING_LEN(tstr) == 1 && rb_enc_asciicompat(enc) &&
10430 (ptstr = RSTRING_PTR(tstr),
10431 ONIGENC_IS_ALLOWED_REVERSE_MATCH(enc, (const unsigned char *)ptstr, (const unsigned char *)ptstr+1)) &&
10432 !is_broken_string(str)) {
10433 int clen;
10434 unsigned char c = rb_enc_codepoint_len(ptstr, ptstr+1, &clen, enc);
10435
10436 s = RSTRING_PTR(str);
10437 if (!s || RSTRING_LEN(str) == 0) return INT2FIX(0);
10438 send = RSTRING_END(str);
10439 while (s < send) {
10440 if (*(unsigned char*)s++ == c) n++;
10441 }
10442 return SIZET2NUM(n);
10443 }
10444 }
10445
10446 tr_setup_table(tstr, table, TRUE, &del, &nodel, enc);
10447 for (i=1; i<argc; i++) {
10448 tstr = argv[i];
10449 StringValue(tstr);
10450 enc = rb_enc_check(str, tstr);
10451 tr_setup_table(tstr, table, FALSE, &del, &nodel, enc);
10452 }
10453
10454 s = RSTRING_PTR(str);
10455 if (!s || RSTRING_LEN(str) == 0) return INT2FIX(0);
10456 send = RSTRING_END(str);
10457 ascompat = rb_enc_asciicompat(enc);
10458 while (s < send) {
10459 unsigned int c;
10460
10461 if (ascompat && (c = *(unsigned char*)s) < 0x80) {
10462 if (table[c]) {
10463 n++;
10464 }
10465 s++;
10466 }
10467 else {
10468 int clen;
10469 c = rb_enc_codepoint_len(s, send, &clen, enc);
10470 if (tr_find(c, table, del, nodel)) {
10471 n++;
10472 }
10473 s += clen;
10474 }
10475 }
10476
10477 return SIZET2NUM(n);
10478}
10479
10480static VALUE
10481rb_fs_check(VALUE val)
10482{
10483 if (!NIL_P(val) && !RB_TYPE_P(val, T_STRING) && !RB_TYPE_P(val, T_REGEXP)) {
10484 val = rb_check_string_type(val);
10485 if (NIL_P(val)) return 0;
10486 }
10487 return val;
10488}
10489
10490static const char isspacetable[256] = {
10491 0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 0, 0,
10492 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10493 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10494 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10495 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10496 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10497 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10498 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10499 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10500 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10501 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10502 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10503 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10504 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10505 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
10506 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0
10507};
10508
10509#define ascii_isspace(c) isspacetable[(unsigned char)(c)]
10510
10511static long
10512split_string(VALUE result, VALUE str, long beg, long len, long empty_count)
10513{
10514 if (empty_count >= 0 && len == 0) {
10515 return empty_count + 1;
10516 }
10517 if (empty_count > 0) {
10518 /* make different substrings */
10519 if (result) {
10520 do {
10521 rb_ary_push(result, str_new_empty_String(str));
10522 } while (--empty_count > 0);
10523 }
10524 else {
10525 do {
10526 rb_yield(str_new_empty_String(str));
10527 } while (--empty_count > 0);
10528 }
10529 }
10530 str = rb_str_subseq(str, beg, len);
10531 if (result) {
10532 rb_ary_push(result, str);
10533 }
10534 else {
10535 rb_yield(str);
10536 }
10537 return empty_count;
10538}
10539
10540typedef enum {
10541 SPLIT_TYPE_AWK, SPLIT_TYPE_STRING, SPLIT_TYPE_REGEXP, SPLIT_TYPE_CHARS
10542} split_type_t;
10543
10544static split_type_t
10545literal_split_pattern(VALUE spat, split_type_t default_type)
10546{
10547 rb_encoding *enc = STR_ENC_GET(spat);
10548 const char *ptr;
10549 long len;
10550 RSTRING_GETMEM(spat, ptr, len);
10551 if (len == 0) {
10552 /* Special case - split into chars */
10553 return SPLIT_TYPE_CHARS;
10554 }
10555 else if (rb_enc_asciicompat(enc)) {
10556 if (len == 1 && ptr[0] == ' ') {
10557 return SPLIT_TYPE_AWK;
10558 }
10559 }
10560 else {
10561 int l;
10562 if (rb_enc_ascget(ptr, ptr + len, &l, enc) == ' ' && len == l) {
10563 return SPLIT_TYPE_AWK;
10564 }
10565 }
10566 return default_type;
10567}
10568
10569/*
10570 * call-seq:
10571 * split(field_sep = $;, limit = 0) -> array_of_substrings
10572 * split(field_sep = $;, limit = 0) {|substring| ... } -> self
10573 *
10574 * :include: doc/string/split.rdoc
10575 *
10576 */
10577
10578static VALUE
10579rb_str_split_m(int argc, VALUE *argv, VALUE str)
10580{
10581 rb_encoding *enc;
10582 VALUE spat;
10583 VALUE limit;
10584 split_type_t split_type;
10585 long beg, end, i = 0, empty_count = -1;
10586 int lim = 0;
10587 VALUE result, tmp;
10588
10589 result = rb_block_given_p() ? Qfalse : Qnil;
10590 if (rb_scan_args(argc, argv, "02", &spat, &limit) == 2) {
10591 lim = NUM2INT(limit);
10592 if (lim <= 0) limit = Qnil;
10593 else if (lim == 1) {
10594 if (RSTRING_LEN(str) == 0)
10595 return result ? rb_ary_new2(0) : str;
10596 tmp = str_duplicate(rb_cString, str);
10597 if (!result) {
10598 rb_yield(tmp);
10599 return str;
10600 }
10601 return rb_ary_new3(1, tmp);
10602 }
10603 i = 1;
10604 }
10605 if (NIL_P(limit) && !lim) empty_count = 0;
10606
10607 enc = STR_ENC_GET(str);
10608 split_type = SPLIT_TYPE_REGEXP;
10609 if (!NIL_P(spat)) {
10610 spat = get_pat_quoted(spat, 0);
10611 }
10612 else if (NIL_P(spat = rb_fs)) {
10613 split_type = SPLIT_TYPE_AWK;
10614 }
10615 else if (!(spat = rb_fs_check(spat))) {
10616 rb_raise(rb_eTypeError, "value of $; must be String or Regexp");
10617 }
10618 else {
10619 rb_category_warn(RB_WARN_CATEGORY_DEPRECATED, "$; is set to non-nil value");
10620 }
10621 if (split_type != SPLIT_TYPE_AWK) {
10622 switch (BUILTIN_TYPE(spat)) {
10623 case T_REGEXP:
10624 rb_reg_options(spat); /* check if uninitialized */
10625 tmp = RREGEXP_SRC(spat);
10626 split_type = literal_split_pattern(tmp, SPLIT_TYPE_REGEXP);
10627 if (split_type == SPLIT_TYPE_AWK) {
10628 spat = tmp;
10629 split_type = SPLIT_TYPE_STRING;
10630 }
10631 break;
10632
10633 case T_STRING:
10634 mustnot_broken(spat);
10635 split_type = literal_split_pattern(spat, SPLIT_TYPE_STRING);
10636 break;
10637
10638 default:
10640 }
10641 }
10642
10643#define SPLIT_STR(beg, len) ( \
10644 empty_count = split_string(result, str, beg, len, empty_count), \
10645 str_mod_check(str, str_start, str_len))
10646
10647 beg = 0;
10648 const char *ptr = RSTRING_PTR(str);
10649 const char *const str_start = ptr;
10650 const long str_len = RSTRING_LEN(str);
10651 const char *const eptr = str_start + str_len;
10652 if (split_type == SPLIT_TYPE_AWK) {
10653 const char *bptr = ptr;
10654 int skip = 1;
10655 unsigned int c;
10656
10657 if (result) result = rb_ary_new();
10658 end = beg;
10659 if (is_ascii_string(str)) {
10660 while (ptr < eptr) {
10661 c = (unsigned char)*ptr++;
10662 if (skip) {
10663 if (ascii_isspace(c)) {
10664 beg = ptr - bptr;
10665 }
10666 else {
10667 end = ptr - bptr;
10668 skip = 0;
10669 if (!NIL_P(limit) && lim <= i) break;
10670 }
10671 }
10672 else if (ascii_isspace(c)) {
10673 SPLIT_STR(beg, end-beg);
10674 skip = 1;
10675 beg = ptr - bptr;
10676 if (!NIL_P(limit)) ++i;
10677 }
10678 else {
10679 end = ptr - bptr;
10680 }
10681 }
10682 }
10683 else {
10684 while (ptr < eptr) {
10685 int n;
10686
10687 c = rb_enc_codepoint_len(ptr, eptr, &n, enc);
10688 ptr += n;
10689 if (skip) {
10690 if (rb_isspace(c)) {
10691 beg = ptr - bptr;
10692 }
10693 else {
10694 end = ptr - bptr;
10695 skip = 0;
10696 if (!NIL_P(limit) && lim <= i) break;
10697 }
10698 }
10699 else if (rb_isspace(c)) {
10700 SPLIT_STR(beg, end-beg);
10701 skip = 1;
10702 beg = ptr - bptr;
10703 if (!NIL_P(limit)) ++i;
10704 }
10705 else {
10706 end = ptr - bptr;
10707 }
10708 }
10709 }
10710 }
10711 else if (split_type == SPLIT_TYPE_STRING) {
10712 const char *substr_start = ptr;
10713 const char *sptr = RSTRING_PTR(spat);
10714 long slen = RSTRING_LEN(spat);
10715
10716 if (result) result = rb_ary_new();
10717 mustnot_broken(str);
10718 enc = rb_enc_check(str, spat);
10719 while (ptr < eptr &&
10720 (end = rb_memsearch(sptr, slen, ptr, eptr - ptr, enc)) >= 0) {
10721 /* Check we are at the start of a char */
10722 const char *t = rb_enc_right_char_head(ptr, ptr + end, eptr, enc);
10723 if (t != ptr + end) {
10724 ptr = t;
10725 continue;
10726 }
10727 SPLIT_STR(substr_start - str_start, (ptr+end) - substr_start);
10728 str_mod_check(spat, sptr, slen);
10729 ptr += end + slen;
10730 substr_start = ptr;
10731 if (!NIL_P(limit) && lim <= ++i) break;
10732 }
10733 beg = ptr - str_start;
10734 }
10735 else if (split_type == SPLIT_TYPE_CHARS) {
10736 int n;
10737
10738 if (result) result = rb_ary_new_capa(RSTRING_LEN(str));
10739 mustnot_broken(str);
10740 enc = rb_enc_get(str);
10741 while (ptr < eptr &&
10742 (n = rb_enc_precise_mbclen(ptr, eptr, enc)) > 0) {
10743 SPLIT_STR(ptr - str_start, n);
10744 ptr += n;
10745 if (!NIL_P(limit) && lim <= ++i) break;
10746 }
10747 beg = ptr - str_start;
10748 }
10749 else {
10750 if (result) result = rb_ary_new();
10751 long len = RSTRING_LEN(str);
10752 long start = beg;
10753 int idx;
10754 int last_null = 0;
10755 VALUE match = 0;
10756
10757 for (; rb_reg_search(spat, str, start, 0) >= 0;
10758 (match ? (rb_match_unbusy(match), rb_backref_set(match)) : (void)0)) {
10759 match = rb_backref_get();
10760 if (!result) rb_match_busy(match);
10761 end = RMATCH_BEG(match, 0);
10762 if (start == end && RMATCH_BEG(match, 0) == RMATCH_END(match, 0)) {
10763 if (!ptr) {
10764 SPLIT_STR(0, 0);
10765 break;
10766 }
10767 else if (last_null == 1) {
10768 SPLIT_STR(beg, rb_enc_fast_mbclen(ptr+beg, eptr, enc));
10769 beg = start;
10770 }
10771 else {
10772 if (start == len)
10773 start++;
10774 else
10775 start += rb_enc_fast_mbclen(ptr+start,eptr,enc);
10776 last_null = 1;
10777 continue;
10778 }
10779 }
10780 else {
10781 SPLIT_STR(beg, end-beg);
10782 beg = start = RMATCH_END(match, 0);
10783 }
10784 last_null = 0;
10785
10786 for (idx = 1; idx < RMATCH_NREGS(match); idx++) {
10787 if (RMATCH_BEG(match, idx) == -1) continue;
10788 SPLIT_STR(RMATCH_BEG(match, idx), RMATCH_END(match, idx) - RMATCH_BEG(match, idx));
10789 }
10790 if (!NIL_P(limit) && lim <= ++i) break;
10791 }
10792 if (match) rb_match_unbusy(match);
10793 }
10794 if (RSTRING_LEN(str) > 0 && (!NIL_P(limit) || RSTRING_LEN(str) > beg || lim < 0)) {
10795 SPLIT_STR(beg, RSTRING_LEN(str)-beg);
10796 }
10797
10798 return result ? result : str;
10799}
10800
10801VALUE
10802rb_str_split(VALUE str, const char *sep0)
10803{
10804 VALUE sep;
10805
10806 StringValue(str);
10807 sep = rb_str_new_cstr(sep0);
10808 return rb_str_split_m(1, &sep, str);
10809}
10810
10811#define WANTARRAY(m, size) (!rb_block_given_p() ? rb_ary_new_capa(size) : 0)
10812
10813static inline int
10814enumerator_element(VALUE ary, VALUE e)
10815{
10816 if (ary) {
10817 rb_ary_push(ary, e);
10818 return 0;
10819 }
10820 else {
10821 rb_yield(e);
10822 return 1;
10823 }
10824}
10825
10826#define ENUM_ELEM(ary, e) enumerator_element(ary, e)
10827
10828static const char *
10829chomp_newline(const char *p, const char *e, rb_encoding *enc)
10830{
10831 const char *prev = rb_enc_prev_char(p, e, e, enc);
10832 if (rb_enc_is_newline(prev, e, enc)) {
10833 e = prev;
10834 prev = rb_enc_prev_char(p, e, e, enc);
10835 if (prev && rb_enc_ascget(prev, e, NULL, enc) == '\r')
10836 e = prev;
10837 }
10838 return e;
10839}
10840
10841static VALUE
10842get_rs(void)
10843{
10844 VALUE rs = rb_rs;
10845 if (!NIL_P(rs) &&
10846 (!RB_TYPE_P(rs, T_STRING) ||
10847 RSTRING_LEN(rs) != 1 ||
10848 RSTRING_PTR(rs)[0] != '\n')) {
10849 rb_category_warn(RB_WARN_CATEGORY_DEPRECATED, "$/ is set to non-default value");
10850 }
10851 return rs;
10852}
10853
10854#define rb_rs get_rs()
10855
10856static VALUE
10857rb_str_enumerate_lines(int argc, VALUE *argv, VALUE str, VALUE ary)
10858{
10859 rb_encoding *enc;
10860 VALUE line, rs, orig = str, opts = Qnil, chomp = Qfalse;
10861 const char *pend, *subptr, *subend, *rsptr, *hit, *adjusted;
10862 long pos, rslen;
10863 int rsnewline = 0;
10864
10865 if (rb_scan_args(argc, argv, "01:", &rs, &opts) == 0)
10866 rs = rb_rs;
10867 if (!NIL_P(opts)) {
10868 static ID keywords[1];
10869 if (!keywords[0]) {
10870 keywords[0] = rb_intern_const("chomp");
10871 }
10872 rb_get_kwargs(opts, keywords, 0, 1, &chomp);
10873 chomp = (!UNDEF_P(chomp) && RTEST(chomp));
10874 }
10875
10876 if (NIL_P(rs)) {
10877 if (!ENUM_ELEM(ary, str)) {
10878 return ary;
10879 }
10880 else {
10881 return orig;
10882 }
10883 }
10884
10885 if (!RSTRING_LEN(str)) goto end;
10886 str = rb_str_new_frozen(str);
10887 const char *const ptr = subptr = RSTRING_PTR(str);
10888 const long len = RSTRING_LEN(str);
10889 pend = RSTRING_END(str);
10890 StringValue(rs);
10891 rslen = RSTRING_LEN(rs);
10892
10893 if (rs == rb_default_rs)
10894 enc = rb_enc_get(str);
10895 else
10896 enc = rb_enc_check(str, rs);
10897
10898 if (rslen == 0) {
10899 /* paragraph mode */
10900 int n;
10901 const char *eol = NULL;
10902 subend = subptr;
10903 while (subend < pend) {
10904 long chomp_rslen = 0;
10905 do {
10906 if (rb_enc_ascget(subend, pend, &n, enc) != '\r')
10907 n = 0;
10908 rslen = n + rb_enc_mbclen(subend + n, pend, enc);
10909 if (rb_enc_is_newline(subend + n, pend, enc)) {
10910 if (eol == subend) break;
10911 subend += rslen;
10912 if (subptr) {
10913 eol = subend;
10914 chomp_rslen = -rslen;
10915 }
10916 }
10917 else {
10918 if (!subptr) subptr = subend;
10919 subend += rslen;
10920 }
10921 rslen = 0;
10922 } while (subend < pend);
10923 if (!subptr) break;
10924 if (rslen == 0) chomp_rslen = 0;
10925 line = rb_str_subseq(str, subptr - ptr,
10926 subend - subptr + (chomp ? chomp_rslen : rslen));
10927 if (ENUM_ELEM(ary, line)) {
10928 str_mod_check(str, ptr, len);
10929 }
10930 subptr = eol = NULL;
10931 }
10932 goto end;
10933 }
10934 else {
10935 rsptr = RSTRING_PTR(rs);
10936 if (RSTRING_LEN(rs) == rb_enc_mbminlen(enc) &&
10937 rb_enc_is_newline(rsptr, rsptr + RSTRING_LEN(rs), enc)) {
10938 rsnewline = 1;
10939 }
10940 }
10941
10942 if ((rs == rb_default_rs) && !rb_enc_asciicompat(enc)) {
10943 rs = rb_str_new(rsptr, rslen);
10944 rs = rb_str_encode(rs, rb_enc_from_encoding(enc), 0, Qnil);
10945 rsptr = RSTRING_PTR(rs);
10946 rslen = RSTRING_LEN(rs);
10947 }
10948
10949 while (subptr < pend) {
10950 pos = rb_memsearch(rsptr, rslen, subptr, pend - subptr, enc);
10951 if (pos < 0) break;
10952 hit = subptr + pos;
10953 adjusted = rb_enc_right_char_head(subptr, hit, pend, enc);
10954 if (hit != adjusted) {
10955 subptr = adjusted;
10956 continue;
10957 }
10958 subend = hit += rslen;
10959 if (chomp) {
10960 if (rsnewline) {
10961 subend = chomp_newline(subptr, subend, enc);
10962 }
10963 else {
10964 subend -= rslen;
10965 }
10966 }
10967 line = rb_str_subseq(str, subptr - ptr, subend - subptr);
10968 if (ENUM_ELEM(ary, line)) {
10969 str_mod_check(str, ptr, len);
10970 str_mod_check(rs, rsptr, rslen);
10971 }
10972 subptr = hit;
10973 }
10974
10975 if (subptr < pend) {
10976 if (chomp) {
10977 if (rsnewline) {
10978 pend = chomp_newline(subptr, pend, enc);
10979 }
10980 else if (pend - subptr >= rslen &&
10981 memcmp(pend - rslen, rsptr, rslen) == 0) {
10982 pend -= rslen;
10983 }
10984 }
10985 line = rb_str_subseq(str, subptr - ptr, pend - subptr);
10986 ENUM_ELEM(ary, line);
10987 RB_GC_GUARD(str);
10988 }
10989
10990 end:
10991 if (ary)
10992 return ary;
10993 else
10994 return orig;
10995}
10996
10997/*
10998 * call-seq:
10999 * each_line(record_separator = $/, chomp: false) {|substring| ... } -> self
11000 * each_line(record_separator = $/, chomp: false) -> enumerator
11001 *
11002 * :include: doc/string/each_line.rdoc
11003 *
11004 */
11005
11006static VALUE
11007rb_str_each_line(int argc, VALUE *argv, VALUE str)
11008{
11009 RETURN_SIZED_ENUMERATOR(str, argc, argv, 0);
11010 return rb_str_enumerate_lines(argc, argv, str, 0);
11011}
11012
11013/*
11014 * call-seq:
11015 * lines(record_separator = $/, chomp: false) -> array_of_strings
11016 *
11017 * Returns substrings ("lines") of +self+
11018 * according to the given arguments:
11019 *
11020 * s = <<~EOT
11021 * This is the first line.
11022 * This is line two.
11023 *
11024 * This is line four.
11025 * This is line five.
11026 * EOT
11027 *
11028 * With the default argument values:
11029 *
11030 * $/ # => "\n"
11031 * s.lines
11032 * # =>
11033 * ["This is the first line.\n",
11034 * "This is line two.\n",
11035 * "\n",
11036 * "This is line four.\n",
11037 * "This is line five.\n"]
11038 *
11039 * With a different +record_separator+:
11040 *
11041 * record_separator = ' is '
11042 * s.lines(record_separator)
11043 * # =>
11044 * ["This is ",
11045 * "the first line.\nThis is ",
11046 * "line two.\n\nThis is ",
11047 * "line four.\nThis is ",
11048 * "line five.\n"]
11049 *
11050 * With keyword argument +chomp+ as +true+,
11051 * removes the trailing newline from each line:
11052 *
11053 * s.lines(chomp: true)
11054 * # =>
11055 * ["This is the first line.",
11056 * "This is line two.",
11057 * "",
11058 * "This is line four.",
11059 * "This is line five."]
11060 *
11061 * Related: see {Converting to Non-String}[rdoc-ref:String@Converting+to+Non-String].
11062 */
11063
11064static VALUE
11065rb_str_lines(int argc, VALUE *argv, VALUE str)
11066{
11067 VALUE ary = WANTARRAY("lines", 0);
11068 return rb_str_enumerate_lines(argc, argv, str, ary);
11069}
11070
11071static VALUE
11072rb_str_each_byte_size(VALUE str, VALUE args, VALUE eobj)
11073{
11074 return LONG2FIX(RSTRING_LEN(str));
11075}
11076
11077static VALUE
11078rb_str_enumerate_bytes(VALUE str, VALUE ary)
11079{
11080 long i;
11081
11082 for (i=0; i<RSTRING_LEN(str); i++) {
11083 ENUM_ELEM(ary, INT2FIX((unsigned char)RSTRING_PTR(str)[i]));
11084 }
11085 if (ary)
11086 return ary;
11087 else
11088 return str;
11089}
11090
11091/*
11092 * call-seq:
11093 * each_byte {|byte| ... } -> self
11094 * each_byte -> enumerator
11095 *
11096 * :include: doc/string/each_byte.rdoc
11097 *
11098 */
11099
11100static VALUE
11101rb_str_each_byte(VALUE str)
11102{
11103 RETURN_SIZED_ENUMERATOR(str, 0, 0, rb_str_each_byte_size);
11104 return rb_str_enumerate_bytes(str, 0);
11105}
11106
11107/*
11108 * call-seq:
11109 * bytes -> array_of_bytes
11110 *
11111 * :include: doc/string/bytes.rdoc
11112 *
11113 */
11114
11115static VALUE
11116rb_str_bytes(VALUE str)
11117{
11118 VALUE ary = WANTARRAY("bytes", RSTRING_LEN(str));
11119 return rb_str_enumerate_bytes(str, ary);
11120}
11121
11122static VALUE
11123rb_str_each_char_size(VALUE str, VALUE args, VALUE eobj)
11124{
11125 return rb_str_length(str);
11126}
11127
11128static VALUE
11129rb_str_enumerate_chars(VALUE str, VALUE ary)
11130{
11131 VALUE orig = str;
11132 long i, len, n;
11133 const char *ptr;
11134 rb_encoding *enc;
11135
11136 str = rb_str_new_frozen(str);
11137 ptr = RSTRING_PTR(str);
11138 len = RSTRING_LEN(str);
11139 enc = rb_enc_get(str);
11140
11142 for (i = 0; i < len; i += n) {
11143 n = rb_enc_fast_mbclen(ptr + i, ptr + len, enc);
11144 ENUM_ELEM(ary, rb_str_subseq(str, i, n));
11145 }
11146 }
11147 else {
11148 for (i = 0; i < len; i += n) {
11149 n = rb_enc_mbclen(ptr + i, ptr + len, enc);
11150 ENUM_ELEM(ary, rb_str_subseq(str, i, n));
11151 }
11152 }
11153 RB_GC_GUARD(str);
11154 if (ary)
11155 return ary;
11156 else
11157 return orig;
11158}
11159
11160/*
11161 * call-seq:
11162 * each_char {|char| ... } -> self
11163 * each_char -> enumerator
11164 *
11165 * :include: doc/string/each_char.rdoc
11166 *
11167 */
11168
11169static VALUE
11170rb_str_each_char(VALUE str)
11171{
11172 RETURN_SIZED_ENUMERATOR(str, 0, 0, rb_str_each_char_size);
11173 return rb_str_enumerate_chars(str, 0);
11174}
11175
11176/*
11177 * call-seq:
11178 * chars -> array_of_characters
11179 *
11180 * :include: doc/string/chars.rdoc
11181 *
11182 */
11183
11184static VALUE
11185rb_str_chars(VALUE str)
11186{
11187 VALUE ary = WANTARRAY("chars", rb_str_strlen(str));
11188 return rb_str_enumerate_chars(str, ary);
11189}
11190
11191static VALUE
11192rb_str_enumerate_codepoints(VALUE str, VALUE ary)
11193{
11194 VALUE orig = str;
11195 int n;
11196 unsigned int c;
11197 const char *ptr, *end;
11198 rb_encoding *enc;
11199 int enc_asciicompat;
11200
11201 if (single_byte_optimizable(str))
11202 return rb_str_enumerate_bytes(str, ary);
11203
11204 str = rb_str_new_frozen(str);
11205 ptr = RSTRING_PTR(str);
11206 end = RSTRING_END(str);
11207 enc = STR_ENC_GET(str);
11208 enc_asciicompat = rb_enc_asciicompat(enc);
11209
11210 while (ptr < end) {
11211 /* Fast path: ASCII byte in an ASCII-compatible encoding is its own codepoint;
11212 * skip rb_enc_codepoint_len and return the byte directly.
11213 */
11214 n = 1;
11215 c = (enc_asciicompat && ISASCII(*ptr)) ?
11216 (unsigned char)*ptr : rb_enc_codepoint_len(ptr, end, &n, enc);
11217 ENUM_ELEM(ary, UINT2NUM(c));
11218 ptr += n;
11219 }
11220 RB_GC_GUARD(str);
11221 if (ary)
11222 return ary;
11223 else
11224 return orig;
11225}
11226
11227/*
11228 * call-seq:
11229 * each_codepoint {|codepoint| ... } -> self
11230 * each_codepoint -> enumerator
11231 *
11232 * :include: doc/string/each_codepoint.rdoc
11233 *
11234 */
11235
11236static VALUE
11237rb_str_each_codepoint(VALUE str)
11238{
11239 RETURN_SIZED_ENUMERATOR(str, 0, 0, rb_str_each_char_size);
11240 return rb_str_enumerate_codepoints(str, 0);
11241}
11242
11243/*
11244 * call-seq:
11245 * codepoints -> array_of_integers
11246 *
11247 * :include: doc/string/codepoints.rdoc
11248 *
11249 */
11250
11251static VALUE
11252rb_str_codepoints(VALUE str)
11253{
11254 VALUE ary = WANTARRAY("codepoints", rb_str_strlen(str));
11255 return rb_str_enumerate_codepoints(str, ary);
11256}
11257
11258static regex_t *
11259get_reg_grapheme_cluster(rb_encoding *enc)
11260{
11261 int encidx = rb_enc_to_index(enc);
11262
11263 const OnigUChar source_ascii[] = "\\X";
11264 const OnigUChar *source = source_ascii;
11265 size_t source_len = sizeof(source_ascii) - 1;
11266
11267 switch (encidx) {
11268#define CHARS_16BE(x) (OnigUChar)((x)>>8), (OnigUChar)(x)
11269#define CHARS_16LE(x) (OnigUChar)(x), (OnigUChar)((x)>>8)
11270#define CHARS_32BE(x) CHARS_16BE((x)>>16), CHARS_16BE(x)
11271#define CHARS_32LE(x) CHARS_16LE(x), CHARS_16LE((x)>>16)
11272#define CASE_UTF(e) \
11273 case ENCINDEX_UTF_##e: { \
11274 static const OnigUChar source_UTF_##e[] = {CHARS_##e('\\'), CHARS_##e('X')}; \
11275 source = source_UTF_##e; \
11276 source_len = sizeof(source_UTF_##e); \
11277 break; \
11278 }
11279 CASE_UTF(16BE); CASE_UTF(16LE); CASE_UTF(32BE); CASE_UTF(32LE);
11280#undef CASE_UTF
11281#undef CHARS_16BE
11282#undef CHARS_16LE
11283#undef CHARS_32BE
11284#undef CHARS_32LE
11285 }
11286
11287 regex_t *reg_grapheme_cluster;
11288 OnigErrorInfo einfo;
11289 int r = onig_new(&reg_grapheme_cluster, source, source + source_len,
11290 ONIG_OPTION_DEFAULT, enc, OnigDefaultSyntax, &einfo);
11291 if (r) {
11292 UChar message[ONIG_MAX_ERROR_MESSAGE_LEN];
11293 onig_error_code_to_str(message, r, &einfo);
11294 rb_fatal("cannot compile grapheme cluster regexp: %s", (char *)message);
11295 }
11296
11297 return reg_grapheme_cluster;
11298}
11299
11300static regex_t *
11301get_cached_reg_grapheme_cluster(rb_encoding *enc)
11302{
11303 int encidx = rb_enc_to_index(enc);
11304 static regex_t *reg_grapheme_cluster_utf8 = NULL;
11305
11306 if (encidx == rb_utf8_encindex()) {
11307 if (!reg_grapheme_cluster_utf8) {
11308 reg_grapheme_cluster_utf8 = get_reg_grapheme_cluster(enc);
11309 }
11310
11311 return reg_grapheme_cluster_utf8;
11312 }
11313
11314 return NULL;
11315}
11316
11317static VALUE
11318rb_str_each_grapheme_cluster_size(VALUE str, VALUE args, VALUE eobj)
11319{
11320 size_t grapheme_cluster_count = 0;
11321 rb_encoding *enc = get_encoding(str);
11322 const char *ptr, *end;
11323
11324 if (!rb_enc_unicode_p(enc)) {
11325 return rb_str_length(str);
11326 }
11327
11328 bool cached_reg_grapheme_cluster = true;
11329 regex_t *reg_grapheme_cluster = get_cached_reg_grapheme_cluster(enc);
11330 if (!reg_grapheme_cluster) {
11331 reg_grapheme_cluster = get_reg_grapheme_cluster(enc);
11332 cached_reg_grapheme_cluster = false;
11333 }
11334
11335 ptr = RSTRING_PTR(str);
11336 end = RSTRING_END(str);
11337
11338 while (ptr < end) {
11339 OnigPosition len = onig_match(reg_grapheme_cluster,
11340 (const OnigUChar *)ptr, (const OnigUChar *)end,
11341 (const OnigUChar *)ptr, NULL, 0);
11342 if (len <= 0) break;
11343 grapheme_cluster_count++;
11344 ptr += len;
11345 }
11346
11347 if (!cached_reg_grapheme_cluster) {
11348 onig_free(reg_grapheme_cluster);
11349 }
11350
11351 return SIZET2NUM(grapheme_cluster_count);
11352}
11353
11354static VALUE
11355rb_str_enumerate_grapheme_clusters(VALUE str, VALUE ary)
11356{
11357 VALUE orig = str;
11358 rb_encoding *enc = get_encoding(str);
11359 const char *ptr0, *ptr, *end;
11360
11361 if (!rb_enc_unicode_p(enc)) {
11362 return rb_str_enumerate_chars(str, ary);
11363 }
11364
11365 if (!ary) str = rb_str_new_frozen(str);
11366
11367 bool cached_reg_grapheme_cluster = true;
11368 regex_t *reg_grapheme_cluster = get_cached_reg_grapheme_cluster(enc);
11369 if (!reg_grapheme_cluster) {
11370 reg_grapheme_cluster = get_reg_grapheme_cluster(enc);
11371 cached_reg_grapheme_cluster = false;
11372 }
11373
11374 ptr0 = ptr = RSTRING_PTR(str);
11375 end = RSTRING_END(str);
11376
11377 while (ptr < end) {
11378 OnigPosition len = onig_match(reg_grapheme_cluster,
11379 (const OnigUChar *)ptr, (const OnigUChar *)end,
11380 (const OnigUChar *)ptr, NULL, 0);
11381 if (len <= 0) break;
11382 ENUM_ELEM(ary, rb_str_subseq(str, ptr-ptr0, len));
11383 ptr += len;
11384 }
11385
11386 if (!cached_reg_grapheme_cluster) {
11387 onig_free(reg_grapheme_cluster);
11388 }
11389
11390 RB_GC_GUARD(str);
11391 if (ary)
11392 return ary;
11393 else
11394 return orig;
11395}
11396
11397/*
11398 * call-seq:
11399 * each_grapheme_cluster {|grapheme_cluster| ... } -> self
11400 * each_grapheme_cluster -> enumerator
11401 *
11402 * :include: doc/string/each_grapheme_cluster.rdoc
11403 *
11404 */
11405
11406static VALUE
11407rb_str_each_grapheme_cluster(VALUE str)
11408{
11409 RETURN_SIZED_ENUMERATOR(str, 0, 0, rb_str_each_grapheme_cluster_size);
11410 return rb_str_enumerate_grapheme_clusters(str, 0);
11411}
11412
11413/*
11414 * call-seq:
11415 * grapheme_clusters -> array_of_grapheme_clusters
11416 *
11417 * :include: doc/string/grapheme_clusters.rdoc
11418 *
11419 */
11420
11421static VALUE
11422rb_str_grapheme_clusters(VALUE str)
11423{
11424 VALUE ary = WANTARRAY("grapheme_clusters", rb_str_strlen(str));
11425 return rb_str_enumerate_grapheme_clusters(str, ary);
11426}
11427
11428static long
11429chopped_length(VALUE str)
11430{
11431 rb_encoding *enc = STR_ENC_GET(str);
11432 const char *p, *p2, *beg, *end;
11433
11434 beg = RSTRING_PTR(str);
11435 end = beg + RSTRING_LEN(str);
11436 if (beg >= end) return 0;
11437 p = rb_enc_prev_char(beg, end, end, enc);
11438 if (!p) return 0;
11439 if (p > beg && rb_enc_ascget(p, end, 0, enc) == '\n') {
11440 p2 = rb_enc_prev_char(beg, p, end, enc);
11441 if (p2 && rb_enc_ascget(p2, end, 0, enc) == '\r') p = p2;
11442 }
11443 return p - beg;
11444}
11445
11446/*
11447 * call-seq:
11448 * chop! -> self or nil
11449 *
11450 * Like String#chop, except that:
11451 *
11452 * - Removes trailing characters from +self+ (not from a copy of +self+).
11453 * - Returns +self+ if any characters are removed, +nil+ otherwise.
11454 *
11455 * Related: see {Modifying}[rdoc-ref:String@Modifying].
11456 */
11457
11458static VALUE
11459rb_str_chop_bang(VALUE str)
11460{
11461 str_modify_keep_cr(str);
11462 if (RSTRING_LEN(str) > 0) {
11463 long len;
11464 len = chopped_length(str);
11465 STR_SET_LEN(str, len);
11466 TERM_FILL(&RSTRING_PTR(str)[len], TERM_LEN(str));
11467 if (ENC_CODERANGE(str) != ENC_CODERANGE_7BIT) {
11469 }
11470 return str;
11471 }
11472 return Qnil;
11473}
11474
11475
11476/*
11477 * call-seq:
11478 * chop -> new_string
11479 *
11480 * :include: doc/string/chop.rdoc
11481 *
11482 */
11483
11484static VALUE
11485rb_str_chop(VALUE str)
11486{
11487 return rb_str_subseq(str, 0, chopped_length(str));
11488}
11489
11490static long
11491smart_chomp(VALUE str, const char *e, const char *p)
11492{
11493 rb_encoding *enc = rb_enc_get(str);
11494 if (rb_enc_mbminlen(enc) > 1) {
11495 /* a receiver shorter than one character has nothing to chomp */
11496 if (e - p < rb_enc_mbminlen(enc)) return e - p;
11497 const char *pp = rb_enc_left_char_head(p, e-rb_enc_mbminlen(enc), e, enc);
11498 if (rb_enc_is_newline(pp, e, enc)) {
11499 e = pp;
11500 }
11501 pp = e - rb_enc_mbminlen(enc);
11502 if (pp >= p) {
11503 pp = rb_enc_left_char_head(p, pp, e, enc);
11504 if (rb_enc_ascget(pp, e, 0, enc) == '\r') {
11505 e = pp;
11506 }
11507 }
11508 }
11509 else {
11510 switch (*(e-1)) { /* not e[-1] to get rid of VC bug */
11511 case '\n':
11512 if (--e > p && *(e-1) == '\r') {
11513 --e;
11514 }
11515 break;
11516 case '\r':
11517 --e;
11518 break;
11519 }
11520 }
11521 return e - p;
11522}
11523
11524static long
11525chompped_length(VALUE str, VALUE rs)
11526{
11527 rb_encoding *enc;
11528 int newline;
11529 const char *pp, *e, *rsptr;
11530 long rslen;
11531 const char *const p = RSTRING_PTR(str);
11532 long len = RSTRING_LEN(str);
11533
11534 if (len == 0) return 0;
11535 e = p + len;
11536 if (rs == rb_default_rs) {
11537 return smart_chomp(str, e, p);
11538 }
11539
11540 enc = rb_enc_get(str);
11541 RSTRING_GETMEM(rs, rsptr, rslen);
11542 if (rslen == 0) {
11543 if (rb_enc_mbminlen(enc) > 1) {
11544 while (e - p >= rb_enc_mbminlen(enc)) {
11545 pp = rb_enc_left_char_head(p, e-rb_enc_mbminlen(enc), e, enc);
11546 if (!rb_enc_is_newline(pp, e, enc)) break;
11547 e = pp;
11548 pp -= rb_enc_mbminlen(enc);
11549 if (pp >= p) {
11550 pp = rb_enc_left_char_head(p, pp, e, enc);
11551 if (rb_enc_ascget(pp, e, 0, enc) == '\r') {
11552 e = pp;
11553 }
11554 }
11555 }
11556 }
11557 else {
11558 while (e > p && *(e-1) == '\n') {
11559 --e;
11560 if (e > p && *(e-1) == '\r')
11561 --e;
11562 }
11563 }
11564 return e - p;
11565 }
11566 if (rslen > len) return len;
11567
11568 enc = rb_enc_get(rs);
11569 newline = rsptr[rslen-1];
11570 if (rslen == rb_enc_mbminlen(enc)) {
11571 if (rslen == 1) {
11572 if (newline == '\n')
11573 return smart_chomp(str, e, p);
11574 }
11575 else {
11576 if (rb_enc_is_newline(rsptr, rsptr+rslen, enc))
11577 return smart_chomp(str, e, p);
11578 }
11579 }
11580
11581 enc = rb_enc_check(str, rs);
11582 if (is_broken_string(rs)) {
11583 return len;
11584 }
11585 pp = e - rslen;
11586 if (p[len-1] == newline &&
11587 (rslen <= 1 ||
11588 memcmp(rsptr, pp, rslen) == 0)) {
11589 if (at_char_boundary(p, pp, e, enc))
11590 return len - rslen;
11591 RB_GC_GUARD(rs);
11592 }
11593 return len;
11594}
11595
11601static VALUE
11602chomp_rs(int argc, const VALUE *argv)
11603{
11604 rb_check_arity(argc, 0, 1);
11605 if (argc > 0) {
11606 VALUE rs = argv[0];
11607 if (!NIL_P(rs)) StringValue(rs);
11608 return rs;
11609 }
11610 else {
11611 return rb_rs;
11612 }
11613}
11614
11615static VALUE
11616str_shrink(VALUE str, long len)
11617{
11618 str_modify_keep_cr(str);
11619 STR_SET_LEN(str, len);
11620 TERM_FILL(&RSTRING_PTR(str)[len], TERM_LEN(str));
11621 if (ENC_CODERANGE(str) != ENC_CODERANGE_7BIT) {
11623 }
11624 return str;
11625}
11626
11627VALUE
11628rb_str_chomp_string(VALUE str, VALUE rs)
11629{
11630 long olen = RSTRING_LEN(str);
11631 long len = chompped_length(str, rs);
11632 if (len >= olen) return Qnil;
11633 return str_shrink(str, len);
11634}
11635
11636/*
11637 * call-seq:
11638 * chomp!(line_sep = $/) -> self or nil
11639 *
11640 * Like String#chomp, except that:
11641 *
11642 * - Removes trailing characters from +self+ (not from a copy of +self+).
11643 * - Returns +self+ if any characters are removed, +nil+ otherwise.
11644 *
11645 * Related: see {Modifying}[rdoc-ref:String@Modifying].
11646 */
11647
11648static VALUE
11649rb_str_chomp_bang(int argc, VALUE *argv, VALUE str)
11650{
11651 VALUE rs;
11652 str_modifiable(str);
11653 if (RSTRING_LEN(str) == 0 && argc < 2) return Qnil;
11654 rs = chomp_rs(argc, argv);
11655 if (NIL_P(rs)) return Qnil;
11656 return rb_str_chomp_string(str, rs);
11657}
11658
11659
11660/*
11661 * call-seq:
11662 * chomp(line_sep = $/) -> new_string
11663 *
11664 * :include: doc/string/chomp.rdoc
11665 *
11666 */
11667
11668static VALUE
11669rb_str_chomp(int argc, VALUE *argv, VALUE str)
11670{
11671 VALUE rs = chomp_rs(argc, argv);
11672 if (NIL_P(rs)) return str_duplicate(rb_cString, str);
11673 return rb_str_subseq(str, 0, chompped_length(str, rs));
11674}
11675
11676static void
11677tr_setup_table_multi(char table[TR_TABLE_SIZE], VALUE *tablep, VALUE *ctablep,
11678 VALUE str, int num_selectors, VALUE *selectors)
11679{
11680 int i;
11681
11682 for (i=0; i<num_selectors; i++) {
11683 VALUE selector = selectors[i];
11684 rb_encoding *enc;
11685
11686 StringValue(selector);
11687 enc = rb_enc_check(str, selector);
11688 tr_setup_table(selector, table, i==0, tablep, ctablep, enc);
11689 }
11690}
11691
11692static long
11693lstrip_offset(VALUE str, const char *s, const char *e, rb_encoding *enc)
11694{
11695 const char *const start = s;
11696
11697 if (!s || s >= e) return 0;
11698
11699 /* remove spaces at head */
11700 if (single_byte_optimizable(str)) {
11701 while (s < e && (*s == '\0' || ascii_isspace(*s))) s++;
11702 }
11703 else {
11704 while (s < e) {
11705 int n;
11706 unsigned int cc = rb_enc_codepoint_len(s, e, &n, enc);
11707
11708 if (cc && !rb_isspace(cc)) break;
11709 s += n;
11710 }
11711 }
11712 return s - start;
11713}
11714
11715static long
11716lstrip_offset_table(VALUE str, const char *s, const char *e, rb_encoding *enc,
11717 char table[TR_TABLE_SIZE], VALUE del, VALUE nodel)
11718{
11719 const char *const start = s;
11720
11721 if (!s || s >= e) return 0;
11722
11723 /* remove leading characters in the table */
11724 while (s < e) {
11725 int n;
11726 unsigned int cc = rb_enc_codepoint_len(s, e, &n, enc);
11727
11728 if (!tr_find(cc, table, del, nodel)) break;
11729 s += n;
11730 }
11731 return s - start;
11732}
11733
11734/*
11735 * call-seq:
11736 * lstrip!(*selectors) -> self or nil
11737 *
11738 * Like String#lstrip, except that:
11739 *
11740 * - Performs stripping in +self+ (not in a copy of +self+).
11741 * - Returns +self+ if any characters are stripped, +nil+ otherwise.
11742 *
11743 * Related: see {Modifying}[rdoc-ref:String@Modifying].
11744 */
11745
11746static VALUE
11747rb_str_lstrip_bang(int argc, VALUE *argv, VALUE str)
11748{
11749 rb_encoding *enc;
11750 char *start;
11751 long olen, loffset;
11752
11753 str_modify_keep_cr(str);
11754 enc = STR_ENC_GET(str);
11755 RSTRING_GETMEM(str, start, olen);
11756 if (argc > 0) {
11757 char table[TR_TABLE_SIZE];
11758 VALUE del = 0, nodel = 0;
11759
11760 tr_setup_table_multi(table, &del, &nodel, str, argc, argv);
11761
11762 /* the selector conversion may have modified str */
11763 str_modify_keep_cr(str);
11764 enc = STR_ENC_GET(str);
11765 RSTRING_GETMEM(str, start, olen);
11766
11767 loffset = lstrip_offset_table(str, start, start+olen, enc, table, del, nodel);
11768 }
11769 else {
11770 loffset = lstrip_offset(str, start, start+olen, enc);
11771 }
11772
11773 if (loffset > 0) {
11774 long len = olen-loffset;
11775 memmove(start, start + loffset, len);
11776 STR_SET_LEN(str, len);
11777 TERM_FILL(start+len, rb_enc_mbminlen(enc));
11778 return str;
11779 }
11780 return Qnil;
11781}
11782
11783
11784/*
11785 * call-seq:
11786 * lstrip(*selectors) -> new_string
11787 *
11788 * Returns a copy of +self+ with leading whitespace removed;
11789 * see {Whitespace in Strings}[rdoc-ref:String@Whitespace+in+Strings]:
11790 *
11791 * whitespace = "\x00\t\n\v\f\r "
11792 * s = whitespace + 'abc' + whitespace
11793 * # => "\u0000\t\n\v\f\r abc\u0000\t\n\v\f\r "
11794 * s.lstrip
11795 * # => "abc\u0000\t\n\v\f\r "
11796 *
11797 * If +selectors+ are given, removes characters of +selectors+ from the beginning of +self+:
11798 *
11799 * s = "---abc+++"
11800 * s.lstrip("-") # => "abc+++"
11801 *
11802 * +selectors+ must be valid character selectors (see {Character Selectors}[rdoc-ref:character_selectors.rdoc]),
11803 * and may use any of its valid forms, including negation, ranges, and escapes:
11804 *
11805 * "01234abc56789".lstrip("0-9") # "abc56789"
11806 * "01234abc56789".lstrip("0-9", "^4-6") # "4abc56789"
11807 *
11808 * Related: see {Converting to New String}[rdoc-ref:String@Converting+to+New+String].
11809 */
11810
11811static VALUE
11812rb_str_lstrip(int argc, VALUE *argv, VALUE str)
11813{
11814 const char *start;
11815 long len, loffset;
11816
11817 RSTRING_GETMEM(str, start, len);
11818 if (argc > 0) {
11819 char table[TR_TABLE_SIZE];
11820 VALUE del = 0, nodel = 0;
11821
11822 tr_setup_table_multi(table, &del, &nodel, str, argc, argv);
11823
11824 /* the selector conversion may have modified str */
11825 RSTRING_GETMEM(str, start, len);
11826
11827 loffset = lstrip_offset_table(str, start, start+len, STR_ENC_GET(str), table, del, nodel);
11828 }
11829 else {
11830 loffset = lstrip_offset(str, start, start+len, STR_ENC_GET(str));
11831 }
11832 if (loffset <= 0) return str_duplicate(rb_cString, str);
11833 return rb_str_subseq(str, loffset, len - loffset);
11834}
11835
11836static long
11837rstrip_offset(VALUE str, const char *s, const char *e, rb_encoding *enc)
11838{
11839 const char *t;
11840
11841 rb_str_check_dummy_enc(enc);
11842 if (rb_enc_str_coderange(str) == ENC_CODERANGE_BROKEN) {
11843 rb_raise(rb_eEncCompatError, "invalid byte sequence in %s", rb_enc_name(enc));
11844 }
11845 if (!s || s >= e) return 0;
11846 t = e;
11847
11848 /* remove trailing spaces or '\0's */
11849 if (single_byte_optimizable(str)) {
11850 unsigned char c;
11851 while (s < t && ((c = *(t-1)) == '\0' || ascii_isspace(c))) t--;
11852 }
11853 else {
11854 const char *tp;
11855
11856 while ((tp = rb_enc_prev_char(s, t, e, enc)) != NULL) {
11857 unsigned int c = rb_enc_codepoint(tp, e, enc);
11858 if (c && !rb_isspace(c)) break;
11859 t = tp;
11860 }
11861 }
11862 return e - t;
11863}
11864
11865static long
11866rstrip_offset_table(VALUE str, const char *s, const char *e, rb_encoding *enc,
11867 char table[TR_TABLE_SIZE], VALUE del, VALUE nodel)
11868{
11869 const char *t, *tp;
11870
11871 rb_str_check_dummy_enc(enc);
11872 if (rb_enc_str_coderange(str) == ENC_CODERANGE_BROKEN) {
11873 rb_raise(rb_eEncCompatError, "invalid byte sequence in %s", rb_enc_name(enc));
11874 }
11875 if (!s || s >= e) return 0;
11876 t = e;
11877
11878 /* remove trailing characters in the table */
11879 while ((tp = rb_enc_prev_char(s, t, e, enc)) != NULL) {
11880 unsigned int c = rb_enc_codepoint(tp, e, enc);
11881 if (!tr_find(c, table, del, nodel)) break;
11882 t = tp;
11883 }
11884
11885 return e - t;
11886}
11887
11888/*
11889 * call-seq:
11890 * rstrip!(*selectors) -> self or nil
11891 *
11892 * Like String#rstrip, except that:
11893 *
11894 * - Performs stripping in +self+ (not in a copy of +self+).
11895 * - Returns +self+ if any characters are stripped, +nil+ otherwise.
11896 *
11897 * Related: see {Modifying}[rdoc-ref:String@Modifying].
11898 */
11899
11900static VALUE
11901rb_str_rstrip_bang(int argc, VALUE *argv, VALUE str)
11902{
11903 rb_encoding *enc;
11904 char *start;
11905 long olen, roffset;
11906
11907 str_modify_keep_cr(str);
11908 enc = STR_ENC_GET(str);
11909 RSTRING_GETMEM(str, start, olen);
11910 if (argc > 0) {
11911 char table[TR_TABLE_SIZE];
11912 VALUE del = 0, nodel = 0;
11913
11914 tr_setup_table_multi(table, &del, &nodel, str, argc, argv);
11915
11916 /* the selector conversion may have modified str */
11917 str_modify_keep_cr(str);
11918 enc = STR_ENC_GET(str);
11919 RSTRING_GETMEM(str, start, olen);
11920
11921 roffset = rstrip_offset_table(str, start, start+olen, enc, table, del, nodel);
11922 }
11923 else {
11924 roffset = rstrip_offset(str, start, start+olen, enc);
11925 }
11926 if (roffset > 0) {
11927 long len = olen - roffset;
11928
11929 STR_SET_LEN(str, len);
11930 TERM_FILL(start+len, rb_enc_mbminlen(enc));
11931 return str;
11932 }
11933 return Qnil;
11934}
11935
11936
11937/*
11938 * call-seq:
11939 * rstrip(*selectors) -> new_string
11940 *
11941 * Returns a copy of +self+ with trailing whitespace removed;
11942 * see {Whitespace in Strings}[rdoc-ref:String@Whitespace+in+Strings]:
11943 *
11944 * whitespace = "\x00\t\n\v\f\r "
11945 * s = whitespace + 'abc' + whitespace
11946 * s # => "\u0000\t\n\v\f\r abc\u0000\t\n\v\f\r "
11947 * s.rstrip # => "\u0000\t\n\v\f\r abc"
11948 *
11949 * If +selectors+ are given, removes characters of +selectors+ from the end of +self+:
11950 *
11951 * s = "---abc+++"
11952 * s.rstrip("+") # => "---abc"
11953 *
11954 * +selectors+ must be valid character selectors (see {Character Selectors}[rdoc-ref:character_selectors.rdoc]),
11955 * and may use any of its valid forms, including negation, ranges, and escapes:
11956 *
11957 * "01234abc56789".rstrip("0-9") # "01234abc"
11958 * "01234abc56789".rstrip("0-9", "^4-6") # "01234abc56"
11959 *
11960 * Related: see {Converting to New String}[rdoc-ref:String@Converting+to+New+String].
11961 */
11962
11963static VALUE
11964rb_str_rstrip(int argc, VALUE *argv, VALUE str)
11965{
11966 rb_encoding *enc;
11967 const char *start;
11968 long olen, roffset;
11969
11970 enc = STR_ENC_GET(str);
11971 RSTRING_GETMEM(str, start, olen);
11972 if (argc > 0) {
11973 char table[TR_TABLE_SIZE];
11974 VALUE del = 0, nodel = 0;
11975
11976 tr_setup_table_multi(table, &del, &nodel, str, argc, argv);
11977
11978 /* the selector conversion may have modified str */
11979 enc = STR_ENC_GET(str);
11980
11981 RSTRING_GETMEM(str, start, olen);
11982 roffset = rstrip_offset_table(str, start, start+olen, enc, table, del, nodel);
11983 }
11984 else {
11985 roffset = rstrip_offset(str, start, start+olen, enc);
11986 }
11987 if (roffset <= 0) return str_duplicate(rb_cString, str);
11988 return rb_str_subseq(str, 0, olen-roffset);
11989}
11990
11991
11992/*
11993 * call-seq:
11994 * strip!(*selectors) -> self or nil
11995 *
11996 * Like String#strip, except that:
11997 *
11998 * - Any modifications are made to +self+.
11999 * - Returns +self+ if any modification are made, +nil+ otherwise.
12000 *
12001 * Related: see {Modifying}[rdoc-ref:String@Modifying].
12002 */
12003
12004static VALUE
12005rb_str_strip_bang(int argc, VALUE *argv, VALUE str)
12006{
12007 char *start;
12008 long olen, loffset, roffset;
12009 rb_encoding *enc;
12010
12011 str_modify_keep_cr(str);
12012 enc = STR_ENC_GET(str);
12013 RSTRING_GETMEM(str, start, olen);
12014
12015 if (argc > 0) {
12016 char table[TR_TABLE_SIZE];
12017 VALUE del = 0, nodel = 0;
12018
12019 tr_setup_table_multi(table, &del, &nodel, str, argc, argv);
12020
12021 /* the selector conversion may have modified str */
12022 str_modify_keep_cr(str);
12023 enc = STR_ENC_GET(str);
12024 RSTRING_GETMEM(str, start, olen);
12025
12026 loffset = lstrip_offset_table(str, start, start+olen, enc, table, del, nodel);
12027 roffset = rstrip_offset_table(str, start+loffset, start+olen, enc, table, del, nodel);
12028 }
12029 else {
12030 loffset = lstrip_offset(str, start, start+olen, enc);
12031 roffset = rstrip_offset(str, start+loffset, start+olen, enc);
12032 }
12033
12034 if (loffset > 0 || roffset > 0) {
12035 long len = olen-roffset;
12036 if (loffset > 0) {
12037 len -= loffset;
12038 memmove(start, start + loffset, len);
12039 }
12040 STR_SET_LEN(str, len);
12041 TERM_FILL(start+len, rb_enc_mbminlen(enc));
12042 return str;
12043 }
12044 return Qnil;
12045}
12046
12047
12048/*
12049 * call-seq:
12050 * strip(*selectors) -> new_string
12051 *
12052 * Returns a copy of +self+ with leading and trailing whitespace removed;
12053 * see {Whitespace in Strings}[rdoc-ref:String@Whitespace+in+Strings]:
12054 *
12055 * whitespace = "\x00\t\n\v\f\r "
12056 * s = whitespace + 'abc' + whitespace
12057 * # => "\u0000\t\n\v\f\r abc\u0000\t\n\v\f\r "
12058 * s.strip # => "abc"
12059 *
12060 * If +selectors+ are given, removes characters of +selectors+ from both ends of +self+:
12061 *
12062 * s = "---abc+++"
12063 * s.strip("-+") # => "abc"
12064 * s.strip("+-") # => "abc"
12065 *
12066 * +selectors+ must be valid character selectors (see {Character Selectors}[rdoc-ref:character_selectors.rdoc]),
12067 * and may use any of its valid forms, including negation, ranges, and escapes:
12068 *
12069 * "01234abc56789".strip("0-9") # "abc"
12070 * "01234abc56789".strip("0-9", "^4-6") # "4abc56"
12071 *
12072 * Related: see {Converting to New String}[rdoc-ref:String@Converting+to+New+String].
12073 */
12074
12075static VALUE
12076rb_str_strip(int argc, VALUE *argv, VALUE str)
12077{
12078 const char *start;
12079 long olen, loffset, roffset;
12080 rb_encoding *enc = STR_ENC_GET(str);
12081
12082 RSTRING_GETMEM(str, start, olen);
12083
12084 if (argc > 0) {
12085 char table[TR_TABLE_SIZE];
12086 VALUE del = 0, nodel = 0;
12087
12088 tr_setup_table_multi(table, &del, &nodel, str, argc, argv);
12089
12090 /* the selector conversion may have modified str */
12091 enc = STR_ENC_GET(str);
12092 RSTRING_GETMEM(str, start, olen);
12093
12094 loffset = lstrip_offset_table(str, start, start+olen, enc, table, del, nodel);
12095 roffset = rstrip_offset_table(str, start+loffset, start+olen, enc, table, del, nodel);
12096 }
12097 else {
12098 loffset = lstrip_offset(str, start, start+olen, enc);
12099 roffset = rstrip_offset(str, start+loffset, start+olen, enc);
12100 }
12101
12102 if (loffset <= 0 && roffset <= 0) return str_duplicate(rb_cString, str);
12103 return rb_str_subseq(str, loffset, olen-loffset-roffset);
12104}
12105
12106static VALUE
12107scan_once(VALUE str, VALUE pat, long *start, int set_backref_str)
12108{
12109 VALUE result = Qnil;
12110 long end, pos = rb_pat_search(pat, str, *start, set_backref_str);
12111 if (pos >= 0) {
12112 VALUE match = Qnil;
12113 if (BUILTIN_TYPE(pat) == T_STRING) {
12114 end = pos + RSTRING_LEN(pat);
12115 }
12116 else {
12117 match = rb_backref_get();
12118 pos = RMATCH_BEG(match, 0);
12119 end = RMATCH_END(match, 0);
12120 }
12121
12122 if (pos == end) {
12123 rb_encoding *enc = STR_ENC_GET(str);
12124 /*
12125 * Always consume at least one character of the input string
12126 */
12127 if (RSTRING_LEN(str) > end)
12128 *start = end + rb_enc_fast_mbclen(RSTRING_PTR(str) + end,
12129 RSTRING_END(str), enc);
12130 else
12131 *start = end + 1;
12132 }
12133 else {
12134 *start = end;
12135 }
12136
12137 if (NIL_P(match) || RMATCH_NREGS(match) == 1) {
12138 result = rb_str_subseq(str, pos, end - pos);
12139 return result;
12140 }
12141 else {
12142 int num_regs = RMATCH_NREGS(match);
12143 result = rb_ary_new2(num_regs);
12144 for (int i = 1; i < num_regs; i++) {
12145 VALUE s = Qnil;
12146 if (RMATCH_BEG(match, i) >= 0) {
12147 s = rb_str_subseq(str, RMATCH_BEG(match, i), RMATCH_END(match, i) - RMATCH_BEG(match, i));
12148 }
12149
12150 rb_ary_push(result, s);
12151 }
12152 }
12153
12154 RB_GC_GUARD(match);
12155 }
12156
12157 return result;
12158}
12159
12160
12161/*
12162 * call-seq:
12163 * scan(pattern) -> array_of_results
12164 * scan(pattern) {|result| ... } -> self
12165 *
12166 * :include: doc/string/scan.rdoc
12167 *
12168 */
12169
12170static VALUE
12171rb_str_scan(VALUE str, VALUE pat)
12172{
12173 VALUE result;
12174 long start = 0;
12175 long last = -1, prev = 0;
12176 const char *p = RSTRING_PTR(str);
12177 long len = RSTRING_LEN(str);
12178
12179 pat = get_pat_quoted(pat, 1);
12180 mustnot_broken(str);
12181 if (!rb_block_given_p()) {
12182 VALUE ary = rb_ary_new();
12183
12184 while (!NIL_P(result = scan_once(str, pat, &start, 0))) {
12185 last = prev;
12186 prev = start;
12187 rb_ary_push(ary, result);
12188 }
12189 if (last >= 0) rb_pat_search(pat, str, last, 1);
12190 else rb_backref_set(Qnil);
12191 return ary;
12192 }
12193
12194 while (!NIL_P(result = scan_once(str, pat, &start, 1))) {
12195 last = prev;
12196 prev = start;
12197 rb_yield(result);
12198 str_mod_check(str, p, len);
12199 }
12200 if (last >= 0) rb_pat_search(pat, str, last, 1);
12201 return str;
12202}
12203
12204
12205/*
12206 * call-seq:
12207 * hex -> integer
12208 *
12209 * Interprets the leading substring of +self+ as hexadecimal, possibly signed;
12210 * returns its value as an integer.
12211 *
12212 * The leading substring is interpreted as hexadecimal when it begins with:
12213 *
12214 * - One or more character representing hexadecimal digits
12215 * (each in one of the ranges <tt>'0'..'9'</tt>, <tt>'a'..'f'</tt>, or <tt>'A'..'F'</tt>);
12216 * the string to be interpreted ends at the first character that does not represent a hexadecimal digit:
12217 *
12218 * 'f'.hex # => 15
12219 * '11'.hex # => 17
12220 * 'FFF'.hex # => 4095
12221 * 'fffg'.hex # => 4095
12222 * 'foo'.hex # => 15 # 'f' hexadecimal, 'oo' not.
12223 * 'bar'.hex # => 186 # 'ba' hexadecimal, 'r' not.
12224 * 'deadbeef'.hex # => 3735928559
12225 *
12226 * - <tt>'0x'</tt> or <tt>'0X'</tt>, followed by one or more hexadecimal digits:
12227 *
12228 * '0xfff'.hex # => 4095
12229 * '0xfffg'.hex # => 4095
12230 *
12231 * Any of the above may prefixed with <tt>'-'</tt>, which negates the interpreted value:
12232 *
12233 * '-fff'.hex # => -4095
12234 * '-0xFFF'.hex # => -4095
12235 *
12236 * For any substring not described above, returns zero:
12237 *
12238 * 'xxx'.hex # => 0
12239 * ''.hex # => 0
12240 *
12241 * Note that, unlike #oct, this method interprets only hexadecimal,
12242 * and not binary, octal, or decimal notations:
12243 *
12244 * '0b111'.hex # => 45329
12245 * '0o777'.hex # => 0
12246 * '0d999'.hex # => 55705
12247 *
12248 * Related: See {Converting to Non-String}[rdoc-ref:String@Converting+to+Non-String].
12249 */
12250
12251static VALUE
12252rb_str_hex(VALUE str)
12253{
12254 return rb_str_to_inum(str, 16, FALSE);
12255}
12256
12257
12258/*
12259 * call-seq:
12260 * oct -> integer
12261 *
12262 * Interprets the leading substring of +self+ as octal, binary, decimal, or hexadecimal, possibly signed;
12263 * returns their value as an integer.
12264 *
12265 * In brief:
12266 *
12267 * # Interpreted as octal.
12268 * '777'.oct # => 511
12269 * '777x'.oct # => 511
12270 * '0777'.oct # => 511
12271 * '0o777'.oct # => 511
12272 * '-777'.oct # => -511
12273 * # Not interpreted as octal.
12274 * '0b111'.oct # => 7 # Interpreted as binary.
12275 * '0d999'.oct # => 999 # Interpreted as decimal.
12276 * '0xfff'.oct # => 4095 # Interpreted as hexadecimal.
12277 *
12278 * The leading substring is interpreted as octal when it begins with:
12279 *
12280 * - One or more character representing octal digits
12281 * (each in the range <tt>'0'..'7'</tt>);
12282 * the string to be interpreted ends at the first character that does not represent an octal digit:
12283 *
12284 * '7'.oct @ => 7
12285 * '11'.oct # => 9
12286 * '777'.oct # => 511
12287 * '0777'.oct # => 511
12288 * '7778'.oct # => 511
12289 * '777x'.oct # => 511
12290 *
12291 * - <tt>'0o'</tt>, followed by one or more octal digits:
12292 *
12293 * '0o777'.oct # => 511
12294 * '0o7778'.oct # => 511
12295 *
12296 * The leading substring is _not_ interpreted as octal when it begins with:
12297 *
12298 * - <tt>'0b'</tt>, followed by one or more characters representing binary digits
12299 * (each in the range <tt>'0'..'1'</tt>);
12300 * the string to be interpreted ends at the first character that does not represent a binary digit.
12301 * the string is interpreted as binary digits (base 2):
12302 *
12303 * '0b111'.oct # => 7
12304 * '0b1112'.oct # => 7
12305 *
12306 * - <tt>'0d'</tt>, followed by one or more characters representing decimal digits
12307 * (each in the range <tt>'0'..'9'</tt>);
12308 * the string to be interpreted ends at the first character that does not represent a decimal digit.
12309 * the string is interpreted as decimal digits (base 10):
12310 *
12311 * '0d999'.oct # => 999
12312 * '0d999x'.oct # => 999
12313 *
12314 * - <tt>'0x'</tt>, followed by one or more characters representing hexadecimal digits
12315 * (each in one of the ranges <tt>'0'..'9'</tt>, <tt>'a'..'f'</tt>, or <tt>'A'..'F'</tt>);
12316 * the string to be interpreted ends at the first character that does not represent a hexadecimal digit.
12317 * the string is interpreted as hexadecimal digits (base 16):
12318 *
12319 * '0xfff'.oct # => 4095
12320 * '0xfffg'.oct # => 4095
12321 *
12322 * Any of the above may prefixed with <tt>'-'</tt>, which negates the interpreted value:
12323 *
12324 * '-777'.oct # => -511
12325 * '-0777'.oct # => -511
12326 * '-0b111'.oct # => -7
12327 * '-0xfff'.oct # => -4095
12328 *
12329 * For any substring not described above, returns zero:
12330 *
12331 * 'foo'.oct # => 0
12332 * ''.oct # => 0
12333 *
12334 * Related: see {Converting to Non-String}[rdoc-ref:String@Converting+to+Non-String].
12335 */
12336
12337static VALUE
12338rb_str_oct(VALUE str)
12339{
12340 return rb_str_to_inum(str, -8, FALSE);
12341}
12342
12343#ifndef HAVE_CRYPT_R
12344# include "ruby/thread_native.h"
12345# include "ruby/atomic.h"
12346
12347static struct {
12348 rb_nativethread_lock_t lock;
12349} crypt_mutex = {PTHREAD_MUTEX_INITIALIZER};
12350#endif
12351
12352/*
12353 * call-seq:
12354 * crypt(salt_str) -> new_string
12355 *
12356 * Returns the string generated by calling <code>crypt(3)</code>
12357 * standard library function with <code>str</code> and
12358 * <code>salt_str</code>, in this order, as its arguments. Please do
12359 * not use this method any longer. It is legacy; provided only for
12360 * backward compatibility with ruby scripts in earlier days. It is
12361 * bad to use in contemporary programs for several reasons:
12362 *
12363 * * Behaviour of C's <code>crypt(3)</code> depends on the OS it is
12364 * run. The generated string lacks data portability.
12365 *
12366 * * On some OSes such as Mac OS, <code>crypt(3)</code> never fails
12367 * (i.e. silently ends up in unexpected results).
12368 *
12369 * * On some OSes such as Mac OS, <code>crypt(3)</code> is not
12370 * thread safe.
12371 *
12372 * * So-called "traditional" usage of <code>crypt(3)</code> is very
12373 * very very weak. According to its manpage, Linux's traditional
12374 * <code>crypt(3)</code> output has only 2**56 variations; too
12375 * easy to brute force today. And this is the default behaviour.
12376 *
12377 * * In order to make things robust some OSes implement so-called
12378 * "modular" usage. To go through, you have to do a complex
12379 * build-up of the <code>salt_str</code> parameter, by hand.
12380 * Failure in generation of a proper salt string tends not to
12381 * yield any errors; typos in parameters are normally not
12382 * detectable.
12383 *
12384 * * For instance, in the following example, the second invocation
12385 * of String#crypt is wrong; it has a typo in "round=" (lacks
12386 * "s"). However the call does not fail and something unexpected
12387 * is generated.
12388 *
12389 * "foo".crypt("$5$rounds=1000$salt$") # OK, proper usage
12390 * "foo".crypt("$5$round=1000$salt$") # Typo not detected
12391 *
12392 * * Even in the "modular" mode, some hash functions are considered
12393 * archaic and no longer recommended at all; for instance module
12394 * <code>$1$</code> is officially abandoned by its author: see
12395 * http://phk.freebsd.dk/sagas/md5crypt_eol/ . For another
12396 * instance module <code>$3$</code> is considered completely
12397 * broken: see the manpage of FreeBSD.
12398 *
12399 * * On some OS such as Mac OS, there is no modular mode. Yet, as
12400 * written above, <code>crypt(3)</code> on Mac OS never fails.
12401 * This means even if you build up a proper salt string it
12402 * generates a traditional DES hash anyways, and there is no way
12403 * for you to be aware of.
12404 *
12405 * "foo".crypt("$5$rounds=1000$salt$") # => "$5fNPQMxC5j6."
12406 *
12407 * If for some reason you cannot migrate to other secure contemporary
12408 * password hashing algorithms, install the string-crypt gem and
12409 * <code>require 'string/crypt'</code> to continue using it.
12410 */
12411
12412static VALUE
12413rb_str_crypt(VALUE str, VALUE salt)
12414{
12415#ifdef HAVE_CRYPT_R
12416 VALUE databuf;
12417 struct crypt_data *data;
12418# define CRYPT_END() ALLOCV_END(databuf)
12419#else
12420 char *tmp_buf;
12421 extern char *crypt(const char *, const char *);
12422# define CRYPT_END() rb_nativethread_lock_unlock(&crypt_mutex.lock)
12423#endif
12424 VALUE result;
12425 const char *s, *saltp, *res;
12426#ifdef BROKEN_CRYPT
12427 char salt_8bit_clean[3];
12428#endif
12429
12430 StringValue(salt);
12431 mustnot_wchar(str);
12432 mustnot_wchar(salt);
12433 s = StringValueCStr(str);
12434 saltp = RSTRING_PTR(salt);
12435 if (RSTRING_LEN(salt) < 2 || !saltp[0] || !saltp[1]) {
12436 rb_raise(rb_eArgError, "salt too short (need >=2 bytes)");
12437 }
12438
12439#ifdef BROKEN_CRYPT
12440 if (!ISASCII((unsigned char)saltp[0]) || !ISASCII((unsigned char)saltp[1])) {
12441 salt_8bit_clean[0] = saltp[0] & 0x7f;
12442 salt_8bit_clean[1] = saltp[1] & 0x7f;
12443 salt_8bit_clean[2] = '\0';
12444 saltp = salt_8bit_clean;
12445 }
12446#endif
12447#ifdef HAVE_CRYPT_R
12448 data = ALLOCV(databuf, sizeof(struct crypt_data));
12449# ifdef HAVE_STRUCT_CRYPT_DATA_INITIALIZED
12450 data->initialized = 0;
12451# endif
12452 res = crypt_r(s, saltp, data);
12453#else
12454 rb_nativethread_lock_lock(&crypt_mutex.lock);
12455 res = crypt(s, saltp);
12456#endif
12457 if (!res) {
12458 int err = errno;
12459 CRYPT_END();
12460 rb_syserr_fail(err, "crypt");
12461 }
12462#ifdef HAVE_CRYPT_R
12463 result = rb_str_new_cstr(res);
12464 CRYPT_END();
12465#else
12466 // We need to copy this buffer because it's static and we need to unlock the mutex
12467 // before allocating a new object (the string to be returned). If we allocate while
12468 // holding the lock, we could run GC which fires the VM barrier and causes a deadlock
12469 // if other ractors are waiting on this lock.
12470 size_t res_size = strlen(res);
12471 tmp_buf = ALLOCA_N(char, res_size); // should be small enough to alloca
12472 memcpy(tmp_buf, res, res_size);
12473 CRYPT_END();
12474 result = rb_str_new(tmp_buf, res_size);
12475#endif
12476 return result;
12477}
12478
12479
12480/*
12481 * call-seq:
12482 * ord -> integer
12483 *
12484 * :include: doc/string/ord.rdoc
12485 *
12486 */
12487
12488static VALUE
12489rb_str_ord(VALUE s)
12490{
12491 unsigned int c;
12492
12493 c = rb_enc_codepoint(RSTRING_PTR(s), RSTRING_END(s), STR_ENC_GET(s));
12494 return UINT2NUM(c);
12495}
12496/*
12497 * call-seq:
12498 * sum(n = 16) -> integer
12499 *
12500 * :include: doc/string/sum.rdoc
12501 *
12502 */
12503
12504static VALUE
12505rb_str_sum(int argc, VALUE *argv, VALUE str)
12506{
12507 int bits = 16;
12508 char *ptr, *p, *pend;
12509 long len;
12510 VALUE sum = INT2FIX(0);
12511 unsigned long sum0 = 0;
12512
12513 if (rb_check_arity(argc, 0, 1) && (bits = NUM2INT(argv[0])) < 0) {
12514 bits = 0;
12515 }
12516 ptr = p = RSTRING_PTR(str);
12517 len = RSTRING_LEN(str);
12518 pend = p + len;
12519
12520 while (p < pend) {
12521 if (FIXNUM_MAX - UCHAR_MAX < sum0) {
12522 sum = rb_funcall(sum, '+', 1, LONG2FIX(sum0));
12523 str_mod_check(str, ptr, len);
12524 sum0 = 0;
12525 }
12526 sum0 += (unsigned char)*p;
12527 p++;
12528 }
12529
12530 if (bits == 0) {
12531 if (sum0) {
12532 sum = rb_funcall(sum, '+', 1, LONG2FIX(sum0));
12533 }
12534 }
12535 else {
12536 if (sum == INT2FIX(0)) {
12537 if (bits < (int)sizeof(long)*CHAR_BIT) {
12538 sum0 &= (((unsigned long)1)<<bits)-1;
12539 }
12540 sum = LONG2FIX(sum0);
12541 }
12542 else {
12543 VALUE mod;
12544
12545 if (sum0) {
12546 sum = rb_funcall(sum, '+', 1, LONG2FIX(sum0));
12547 }
12548
12549 mod = rb_funcall(INT2FIX(1), idLTLT, 1, INT2FIX(bits));
12550 mod = rb_funcall(mod, '-', 1, INT2FIX(1));
12551 sum = rb_funcall(sum, '&', 1, mod);
12552 }
12553 }
12554 return sum;
12555}
12556
12557static VALUE
12558rb_str_justify(int argc, VALUE *argv, VALUE str, char jflag)
12559{
12560 rb_encoding *enc;
12561 VALUE w;
12562 long width, len, flen = 1, fclen = 1;
12563 VALUE res;
12564 char *p;
12565 const char *f = " ";
12566 long n, size, llen, rlen, llen2 = 0, rlen2 = 0;
12567 VALUE pad;
12568 int singlebyte = 1, cr;
12569 int termlen;
12570
12571 rb_scan_args(argc, argv, "11", &w, &pad);
12572 enc = STR_ENC_GET(str);
12573 width = NUM2LONG(w);
12574 if (argc == 2) {
12575 StringValue(pad);
12576 enc = rb_enc_check(str, pad);
12577 f = RSTRING_PTR(pad);
12578 flen = RSTRING_LEN(pad);
12579 fclen = str_strlen(pad, enc); /* rb_enc_check */
12580 singlebyte = single_byte_optimizable(pad);
12581 if (flen == 0 || fclen == 0) {
12582 rb_raise(rb_eArgError, "zero width padding");
12583 }
12584 }
12585 termlen = rb_enc_mbminlen(enc);
12586 len = str_strlen(str, enc); /* rb_enc_check */
12587 if (width < 0 || len >= width) return str_duplicate(rb_cString, str);
12588 n = width - len;
12589 llen = (jflag == 'l') ? 0 : ((jflag == 'r') ? n : n/2);
12590 rlen = n - llen;
12591 cr = ENC_CODERANGE(str);
12592 if (flen > 1) {
12593 llen2 = str_offset(f, f + flen, llen % fclen, enc, singlebyte);
12594 rlen2 = str_offset(f, f + flen, rlen % fclen, enc, singlebyte);
12595 }
12596 size = RSTRING_LEN(str);
12597 if ((len = llen / fclen + rlen / fclen) >= LONG_MAX / flen ||
12598 (len *= flen) >= LONG_MAX - llen2 - rlen2 ||
12599 (len += llen2 + rlen2) >= LONG_MAX - size) {
12600 rb_raise(rb_eArgError, "argument too big");
12601 }
12602 len += size;
12603 res = str_enc_new(rb_cString, 0, len, enc);
12604 p = RSTRING_PTR(res);
12605 if (flen <= 1) {
12606 memset(p, *f, llen);
12607 p += llen;
12608 }
12609 else {
12610 while (llen >= fclen) {
12611 memcpy(p,f,flen);
12612 p += flen;
12613 llen -= fclen;
12614 }
12615 if (llen > 0) {
12616 memcpy(p, f, llen2);
12617 p += llen2;
12618 }
12619 }
12620 memcpy(p, RSTRING_PTR(str), size);
12621 p += size;
12622 if (flen <= 1) {
12623 memset(p, *f, rlen);
12624 p += rlen;
12625 }
12626 else {
12627 while (rlen >= fclen) {
12628 memcpy(p,f,flen);
12629 p += flen;
12630 rlen -= fclen;
12631 }
12632 if (rlen > 0) {
12633 memcpy(p, f, rlen2);
12634 p += rlen2;
12635 }
12636 }
12637 TERM_FILL(p, termlen);
12638 STR_SET_LEN(res, p-RSTRING_PTR(res));
12639
12640 if (argc == 2)
12641 cr = ENC_CODERANGE_AND(cr, ENC_CODERANGE(pad));
12642 if (cr != ENC_CODERANGE_BROKEN)
12643 ENC_CODERANGE_SET(res, cr);
12644
12645 RB_GC_GUARD(pad);
12646 return res;
12647}
12648
12649
12650/*
12651 * call-seq:
12652 * ljust(width, pad_string = ' ') -> new_string
12653 *
12654 * :include: doc/string/ljust.rdoc
12655 *
12656 */
12657
12658static VALUE
12659rb_str_ljust(int argc, VALUE *argv, VALUE str)
12660{
12661 return rb_str_justify(argc, argv, str, 'l');
12662}
12663
12664/*
12665 * call-seq:
12666 * rjust(width, pad_string = ' ') -> new_string
12667 *
12668 * :include: doc/string/rjust.rdoc
12669 *
12670 */
12671
12672static VALUE
12673rb_str_rjust(int argc, VALUE *argv, VALUE str)
12674{
12675 return rb_str_justify(argc, argv, str, 'r');
12676}
12677
12678
12679/*
12680 * call-seq:
12681 * center(size, pad_string = ' ') -> new_string
12682 *
12683 * :include: doc/string/center.rdoc
12684 *
12685 */
12686
12687static VALUE
12688rb_str_center(int argc, VALUE *argv, VALUE str)
12689{
12690 return rb_str_justify(argc, argv, str, 'c');
12691}
12692
12693/*
12694 * call-seq:
12695 * partition(pattern) -> [pre_match, first_match, post_match]
12696 *
12697 * :include: doc/string/partition.rdoc
12698 *
12699 */
12700
12701static VALUE
12702rb_str_partition(VALUE str, VALUE sep)
12703{
12704 long pos;
12705
12706 sep = get_pat_quoted(sep, 0);
12707 if (RB_TYPE_P(sep, T_REGEXP)) {
12708 if (rb_reg_search(sep, str, 0, 0) < 0) {
12709 goto failed;
12710 }
12711 VALUE match = rb_backref_get();
12712
12713 pos = RMATCH_BEG(match, 0);
12714 sep = rb_str_subseq(str, pos, RMATCH_END(match, 0) - pos);
12715 }
12716 else {
12717 pos = rb_str_index(str, sep, 0);
12718 if (pos < 0) goto failed;
12719 }
12720
12721 long rpos = pos + RSTRING_LEN(sep);
12722 if (rpos > RSTRING_LEN(str)) goto failed;
12723 return rb_ary_new3(3, rb_str_subseq(str, 0, pos),
12724 sep,
12725 rb_str_subseq(str, rpos, RSTRING_LEN(str)-rpos));
12726
12727 failed:
12728 return rb_ary_new3(3, str_duplicate(rb_cString, str), str_new_empty_String(str), str_new_empty_String(str));
12729}
12730
12731/*
12732 * call-seq:
12733 * rpartition(pattern) -> [pre_match, last_match, post_match]
12734 *
12735 * :include: doc/string/rpartition.rdoc
12736 *
12737 */
12738
12739static VALUE
12740rb_str_rpartition(VALUE str, VALUE sep)
12741{
12742 long pos;
12743
12744 sep = get_pat_quoted(sep, 0);
12745 if (RB_TYPE_P(sep, T_REGEXP)) {
12746 pos = RSTRING_LEN(str);
12747 if (rb_reg_search(sep, str, pos, 1) < 0) {
12748 goto failed;
12749 }
12750 VALUE match = rb_backref_get();
12751
12752 pos = RMATCH_BEG(match, 0);
12753 sep = rb_str_subseq(str, pos, RMATCH_END(match, 0) - pos);
12754 }
12755 else {
12756 /* str may have been modified by #to_str above */
12757 pos = rb_str_sublen(str, RSTRING_LEN(str));
12758 pos = rb_str_rindex(str, sep, pos);
12759 if (pos < 0) {
12760 goto failed;
12761 }
12762 }
12763
12764 long rpos = pos + RSTRING_LEN(sep);
12765 if (rpos > RSTRING_LEN(str)) goto failed;
12766 return rb_ary_new3(3, rb_str_subseq(str, 0, pos),
12767 sep,
12768 rb_str_subseq(str, rpos, RSTRING_LEN(str)-rpos));
12769 failed:
12770 return rb_ary_new3(3, str_new_empty_String(str), str_new_empty_String(str), str_duplicate(rb_cString, str));
12771}
12772
12773/*
12774 * call-seq:
12775 * start_with?(*patterns) -> true or false
12776 *
12777 * :include: doc/string/start_with_p.rdoc
12778 *
12779 */
12780
12781static VALUE
12782rb_str_start_with(int argc, VALUE *argv, VALUE str)
12783{
12784 int i;
12785
12786 for (i=0; i<argc; i++) {
12787 VALUE tmp = argv[i];
12788 if (RB_TYPE_P(tmp, T_REGEXP)) {
12789 if (rb_reg_start_with_p(tmp, str))
12790 return Qtrue;
12791 }
12792 else {
12793 const char *p, *s, *e;
12794 long slen, tlen;
12795 rb_encoding *enc;
12796
12797 StringValue(tmp);
12798 enc = rb_enc_check(str, tmp);
12799 if ((tlen = RSTRING_LEN(tmp)) == 0) return Qtrue;
12800 if ((slen = RSTRING_LEN(str)) < tlen) continue;
12801 p = RSTRING_PTR(str);
12802 e = p + slen;
12803 s = p + tlen;
12804 if (!at_char_right_boundary(p, s, e, enc))
12805 continue;
12806 if (memcmp(p, RSTRING_PTR(tmp), tlen) == 0)
12807 return Qtrue;
12808 }
12809 }
12810 return Qfalse;
12811}
12812
12813/*
12814 * call-seq:
12815 * end_with?(*strings) -> true or false
12816 *
12817 * :include: doc/string/end_with_p.rdoc
12818 *
12819 */
12820
12821static VALUE
12822rb_str_end_with(int argc, VALUE *argv, VALUE str)
12823{
12824 int i;
12825
12826 for (i=0; i<argc; i++) {
12827 VALUE tmp = argv[i];
12828 const char *p, *s, *e;
12829 long slen, tlen;
12830 rb_encoding *enc;
12831
12832 StringValue(tmp);
12833 enc = rb_enc_check(str, tmp);
12834 if ((tlen = RSTRING_LEN(tmp)) == 0) return Qtrue;
12835 if ((slen = RSTRING_LEN(str)) < tlen) continue;
12836 p = RSTRING_PTR(str);
12837 e = p + slen;
12838 s = e - tlen;
12839 if (!at_char_boundary(p, s, e, enc))
12840 continue;
12841 if (memcmp(s, RSTRING_PTR(tmp), tlen) == 0)
12842 return Qtrue;
12843 }
12844 return Qfalse;
12845}
12846
12856static long
12857deleted_prefix_length(VALUE str, VALUE prefix)
12858{
12859 const char *strptr, *prefixptr;
12860 long olen, prefixlen;
12861 rb_encoding *enc = rb_enc_get(str);
12862
12863 StringValue(prefix);
12864
12865 if (!is_broken_string(prefix) ||
12866 !rb_enc_asciicompat(enc) ||
12867 !rb_enc_asciicompat(rb_enc_get(prefix))) {
12868 enc = rb_enc_check(str, prefix);
12869 }
12870
12871 /* return 0 if not start with prefix */
12872 prefixlen = RSTRING_LEN(prefix);
12873 if (prefixlen <= 0) return 0;
12874 olen = RSTRING_LEN(str);
12875 if (olen < prefixlen) return 0;
12876 strptr = RSTRING_PTR(str);
12877 prefixptr = RSTRING_PTR(prefix);
12878 if (memcmp(strptr, prefixptr, prefixlen) != 0) return 0;
12879 if (is_broken_string(prefix)) {
12880 if (!is_broken_string(str)) {
12881 /* prefix in a valid string cannot be broken */
12882 return 0;
12883 }
12884 const char *strend = strptr + olen;
12885 const char *after_prefix = strptr + prefixlen;
12886 if (!at_char_right_boundary(strptr, after_prefix, strend, enc)) {
12887 /* prefix does not end at char-boundary */
12888 return 0;
12889 }
12890 }
12891 /* prefix part in `str` also should be valid. */
12892
12893 return prefixlen;
12894}
12895
12896/*
12897 * call-seq:
12898 * delete_prefix!(prefix) -> self or nil
12899 *
12900 * Like String#delete_prefix, except that +self+ is modified in place;
12901 * returns +self+ if the prefix is removed, +nil+ otherwise.
12902 *
12903 * Related: see {Modifying}[rdoc-ref:String@Modifying].
12904 */
12905
12906static VALUE
12907rb_str_delete_prefix_bang(VALUE str, VALUE prefix)
12908{
12909 long prefixlen;
12910 str_modify_keep_cr(str);
12911
12912 prefixlen = deleted_prefix_length(str, prefix);
12913 if (prefixlen <= 0) return Qnil;
12914
12915 return rb_str_drop_bytes(str, prefixlen);
12916}
12917
12918/*
12919 * call-seq:
12920 * delete_prefix(prefix) -> new_string
12921 *
12922 * :include: doc/string/delete_prefix.rdoc
12923 *
12924 */
12925
12926static VALUE
12927rb_str_delete_prefix(VALUE str, VALUE prefix)
12928{
12929 long prefixlen;
12930
12931 prefixlen = deleted_prefix_length(str, prefix);
12932 if (prefixlen <= 0) return str_duplicate(rb_cString, str);
12933
12934 return rb_str_subseq(str, prefixlen, RSTRING_LEN(str) - prefixlen);
12935}
12936
12946static long
12947deleted_suffix_length(VALUE str, VALUE suffix)
12948{
12949 const char *strptr, *suffixptr;
12950 long olen, suffixlen;
12951 rb_encoding *enc;
12952
12953 StringValue(suffix);
12954 if (is_broken_string(suffix)) return 0;
12955 enc = rb_enc_check(str, suffix);
12956
12957 /* return 0 if not start with suffix */
12958 suffixlen = RSTRING_LEN(suffix);
12959 if (suffixlen <= 0) return 0;
12960 olen = RSTRING_LEN(str);
12961 if (olen < suffixlen) return 0;
12962 strptr = RSTRING_PTR(str);
12963 suffixptr = RSTRING_PTR(suffix);
12964 const char *strend = strptr + olen;
12965 const char *before_suffix = strend - suffixlen;
12966 if (memcmp(before_suffix, suffixptr, suffixlen) != 0) return 0;
12967 if (!at_char_boundary(strptr, before_suffix, strend, enc)) return 0;
12968
12969 return suffixlen;
12970}
12971
12972/*
12973 * call-seq:
12974 * delete_suffix!(suffix) -> self or nil
12975 *
12976 * Like String#delete_suffix, except that +self+ is modified in place;
12977 * returns +self+ if the suffix is removed, +nil+ otherwise.
12978 *
12979 * Related: see {Modifying}[rdoc-ref:String@Modifying].
12980 */
12981
12982static VALUE
12983rb_str_delete_suffix_bang(VALUE str, VALUE suffix)
12984{
12985 long suffixlen;
12986 str_modifiable(str);
12987
12988 suffixlen = deleted_suffix_length(str, suffix);
12989 if (suffixlen <= 0) return Qnil;
12990
12991 return str_shrink(str, RSTRING_LEN(str) - suffixlen);
12992}
12993
12994/*
12995 * call-seq:
12996 * delete_suffix(suffix) -> new_string
12997 *
12998 * :include: doc/string/delete_suffix.rdoc
12999 *
13000 */
13001
13002static VALUE
13003rb_str_delete_suffix(VALUE str, VALUE suffix)
13004{
13005 long suffixlen;
13006
13007 suffixlen = deleted_suffix_length(str, suffix);
13008 if (suffixlen <= 0) return str_duplicate(rb_cString, str);
13009
13010 return rb_str_subseq(str, 0, RSTRING_LEN(str) - suffixlen);
13011}
13012
13013void
13014rb_str_setter(VALUE val, ID id, VALUE *var)
13015{
13016 if (!NIL_P(val) && !RB_TYPE_P(val, T_STRING)) {
13017 rb_raise(rb_eTypeError, "value of %"PRIsVALUE" must be String", rb_id2str(id));
13018 }
13019 *var = val;
13020}
13021
13022static void
13023nil_setter_warning(ID id)
13024{
13025 rb_warn_deprecated("non-nil '%"PRIsVALUE"'", NULL, rb_id2str(id));
13026}
13027
13028void
13029rb_deprecated_str_setter(VALUE val, ID id, VALUE *var)
13030{
13031 rb_str_setter(val, id, var);
13032 if (!NIL_P(*var)) {
13033 nil_setter_warning(id);
13034 }
13035}
13036
13037static void
13038rb_fs_setter(VALUE val, ID id, VALUE *var)
13039{
13040 val = rb_fs_check(val);
13041 if (!val) {
13042 rb_raise(rb_eTypeError,
13043 "value of %"PRIsVALUE" must be String or Regexp",
13044 rb_id2str(id));
13045 }
13046 if (!NIL_P(val)) {
13047 nil_setter_warning(id);
13048 }
13049 *var = val;
13050}
13051
13052
13053/*
13054 * call-seq:
13055 * force_encoding(encoding) -> self
13056 *
13057 * :include: doc/string/force_encoding.rdoc
13058 *
13059 */
13060
13061static VALUE
13062rb_str_force_encoding(VALUE str, VALUE enc)
13063{
13064 str_modifiable(str);
13065
13066 rb_encoding *encoding = rb_to_encoding(enc);
13067 int idx = rb_enc_to_index(encoding);
13068
13069 // If the encoding is unchanged, we do nothing.
13070 if (ENCODING_GET(str) == idx) {
13071 return str;
13072 }
13073
13074 rb_enc_associate_index(str, idx);
13075
13076 // If the coderange was 7bit and the new encoding is ASCII-compatible
13077 // we can keep the coderange.
13078 if (ENC_CODERANGE(str) == ENC_CODERANGE_7BIT && encoding && rb_enc_asciicompat(encoding)) {
13079 return str;
13080 }
13081
13083 return str;
13084}
13085
13086/*
13087 * call-seq:
13088 * b -> new_string
13089 *
13090 * :include: doc/string/b.rdoc
13091 *
13092 */
13093
13094static VALUE
13095rb_str_b(VALUE str)
13096{
13097 VALUE str2;
13098 if (STR_EMBED_P(str)) {
13099 str2 = str_alloc_embed(rb_cString, RSTRING_LEN(str) + TERM_LEN(str));
13100 }
13101 else {
13102 str2 = str_alloc_heap(rb_cString);
13103 }
13104 str_replace_shared_without_enc(str2, str);
13105
13106 if (rb_enc_asciicompat(STR_ENC_GET(str))) {
13107 // BINARY strings can never be broken; they're either 7-bit ASCII or VALID.
13108 // If we know the receiver's code range then we know the result's code range.
13109 int cr = ENC_CODERANGE(str);
13110 switch (cr) {
13111 case ENC_CODERANGE_7BIT:
13113 break;
13117 break;
13118 default:
13119 ENC_CODERANGE_CLEAR(str2);
13120 break;
13121 }
13122 }
13123
13124 return str2;
13125}
13126
13127/* Defined as a leaf builtin in string.rb, so this must never raise or call into Ruby. */
13128static VALUE
13129rb_str_valid_encoding_p(VALUE str)
13130{
13131 int cr = rb_enc_str_coderange(str);
13132
13133 return RBOOL(cr != ENC_CODERANGE_BROKEN);
13134}
13135
13136/* Defined as a leaf builtin in string.rb, so this must never raise or call into Ruby. */
13137static VALUE
13138rb_str_is_ascii_only_p(VALUE str)
13139{
13140 int cr = rb_enc_str_coderange(str);
13141
13142 return RBOOL(cr == ENC_CODERANGE_7BIT);
13143}
13144
13145VALUE
13147{
13148 static const char ellipsis[] = "...";
13149 const long ellipsislen = sizeof(ellipsis) - 1;
13150 rb_encoding *const enc = rb_enc_get(str);
13151 const long blen = RSTRING_LEN(str);
13152 const char *const p = RSTRING_PTR(str), *e = p + blen;
13153 VALUE estr, ret = 0;
13154
13155 if (len < 0) rb_raise(rb_eIndexError, "negative length %ld", len);
13156 if (len * rb_enc_mbminlen(enc) >= blen ||
13157 (e = rb_enc_nth(p, e, len, enc)) - p == blen) {
13158 ret = str;
13159 }
13160 else if (len <= ellipsislen ||
13161 !(e = rb_enc_step_back(p, e, e, len = ellipsislen, enc))) {
13162 if (rb_enc_asciicompat(enc)) {
13163 ret = rb_str_new(ellipsis, len);
13164 rb_enc_associate(ret, enc);
13165 }
13166 else {
13167 estr = rb_usascii_str_new(ellipsis, len);
13168 ret = rb_str_encode(estr, rb_enc_from_encoding(enc), 0, Qnil);
13169 }
13170 }
13171 else if (ret = rb_str_subseq(str, 0, e - p), rb_enc_asciicompat(enc)) {
13172 rb_str_cat(ret, ellipsis, ellipsislen);
13173 }
13174 else {
13175 estr = rb_str_encode(rb_usascii_str_new(ellipsis, ellipsislen),
13176 rb_enc_from_encoding(enc), 0, Qnil);
13177 rb_str_append(ret, estr);
13178 }
13179 return ret;
13180}
13181
13182static VALUE
13183str_compat_and_valid(VALUE str, rb_encoding *enc)
13184{
13185 int cr;
13186 str = StringValue(str);
13187 cr = rb_enc_str_coderange(str);
13188 if (cr == ENC_CODERANGE_BROKEN) {
13189 rb_raise(rb_eArgError, "replacement must be valid byte sequence '%+"PRIsVALUE"'", str);
13190 }
13191 else {
13192 rb_encoding *e = STR_ENC_GET(str);
13193 if (cr == ENC_CODERANGE_7BIT ? rb_enc_mbminlen(enc) != 1 : enc != e) {
13194 rb_raise(rb_eEncCompatError, "incompatible character encodings: %s and %s",
13195 rb_enc_inspect_name(enc), rb_enc_inspect_name(e));
13196 }
13197 }
13198 return str;
13199}
13200
13201static VALUE enc_str_scrub(rb_encoding *enc, VALUE str, VALUE repl, int cr);
13202
13203VALUE
13205{
13206 rb_encoding *enc = STR_ENC_GET(str);
13207 return enc_str_scrub(enc, str, repl, ENC_CODERANGE(str));
13208}
13209
13210VALUE
13211rb_enc_str_scrub(rb_encoding *enc, VALUE str, VALUE repl)
13212{
13213 int cr = ENC_CODERANGE_UNKNOWN;
13214 if (enc == STR_ENC_GET(str)) {
13215 /* cached coderange makes sense only when enc equals the
13216 * actual encoding of str */
13217 cr = ENC_CODERANGE(str);
13218 }
13219 return enc_str_scrub(enc, str, repl, cr);
13220}
13221
13222static VALUE
13223enc_str_scrub(rb_encoding *enc, VALUE str, VALUE repl, int cr)
13224{
13225 int encidx;
13226 VALUE buf = Qnil;
13227 const char *rep, *p, *e, *p1, *sp;
13228 long replen = -1;
13229 long slen;
13230
13231 if (rb_block_given_p()) {
13232 if (!NIL_P(repl))
13233 rb_raise(rb_eArgError, "both of block and replacement given");
13234 replen = 0;
13235 }
13236
13237 if (ENC_CODERANGE_CLEAN_P(cr))
13238 return Qnil;
13239
13240 if (!NIL_P(repl)) {
13241 repl = str_compat_and_valid(repl, enc);
13242 }
13243
13244 if (rb_enc_dummy_p(enc)) {
13245 return Qnil;
13246 }
13247 encidx = rb_enc_to_index(enc);
13248
13249#define DEFAULT_REPLACE_CHAR(str) do { \
13250 RBIMPL_ATTR_NONSTRING() static const char replace[sizeof(str)-1] = str; \
13251 rep = replace; replen = (int)sizeof(replace); \
13252 } while (0)
13253
13254 slen = RSTRING_LEN(str);
13255 p = RSTRING_PTR(str);
13256 e = RSTRING_END(str);
13257 p1 = p;
13258 sp = p;
13259
13260 if (rb_enc_asciicompat(enc)) {
13261 int rep7bit_p;
13262 if (!replen) {
13263 rep = NULL;
13264 rep7bit_p = FALSE;
13265 }
13266 else if (!NIL_P(repl)) {
13267 rep = RSTRING_PTR(repl);
13268 replen = RSTRING_LEN(repl);
13269 rep7bit_p = (ENC_CODERANGE(repl) == ENC_CODERANGE_7BIT);
13270 }
13271 else if (encidx == rb_utf8_encindex()) {
13272 DEFAULT_REPLACE_CHAR("\xEF\xBF\xBD");
13273 rep7bit_p = FALSE;
13274 }
13275 else {
13276 DEFAULT_REPLACE_CHAR("?");
13277 rep7bit_p = TRUE;
13278 }
13279 cr = ENC_CODERANGE_7BIT;
13280
13281 p = search_nonascii(p, e);
13282 if (!p) {
13283 p = e;
13284 }
13285 while (p < e) {
13286 int ret = rb_enc_precise_mbclen(p, e, enc);
13287 if (MBCLEN_NEEDMORE_P(ret)) {
13288 break;
13289 }
13290 else if (MBCLEN_CHARFOUND_P(ret)) {
13292 p += MBCLEN_CHARFOUND_LEN(ret);
13293 /* After a multibyte character, fast-skip the following ASCII run. */
13294 p = search_nonascii(p, e);
13295 if (!p) {
13296 p = e;
13297 break;
13298 }
13299 }
13300 else if (MBCLEN_INVALID_P(ret)) {
13301 /*
13302 * p1~p: valid ascii/multibyte chars
13303 * p ~e: invalid bytes + unknown bytes
13304 */
13305 long clen = rb_enc_mbmaxlen(enc);
13306 if (NIL_P(buf)) buf = rb_str_buf_new(RSTRING_LEN(str));
13307 if (p > p1) {
13308 rb_str_buf_cat(buf, p1, p - p1);
13309 }
13310
13311 if (e - p < clen) clen = e - p;
13312 if (clen <= 2) {
13313 clen = 1;
13314 }
13315 else {
13316 const char *q = p;
13317 clen--;
13318 for (; clen > 1; clen--) {
13319 ret = rb_enc_precise_mbclen(q, q + clen, enc);
13320 if (MBCLEN_NEEDMORE_P(ret)) break;
13321 if (MBCLEN_INVALID_P(ret)) continue;
13323 }
13324 }
13325 if (rep) {
13326 rb_str_buf_cat(buf, rep, replen);
13327 if (!rep7bit_p) cr = ENC_CODERANGE_VALID;
13328 }
13329 else {
13330 repl = rb_yield(rb_enc_str_new(p, clen, enc));
13331 str_mod_check(str, sp, slen);
13332 repl = str_compat_and_valid(repl, enc);
13333 rb_str_buf_cat(buf, RSTRING_PTR(repl), RSTRING_LEN(repl));
13336 }
13337 p += clen;
13338 p1 = p;
13339 p = search_nonascii(p, e);
13340 if (!p) {
13341 p = e;
13342 break;
13343 }
13344 }
13345 else {
13347 }
13348 }
13349 if (NIL_P(buf)) {
13350 if (p == e) {
13351 ENC_CODERANGE_SET(str, cr);
13352 return Qnil;
13353 }
13354 buf = rb_str_buf_new(RSTRING_LEN(str));
13355 }
13356 if (p1 < p) {
13357 rb_str_buf_cat(buf, p1, p - p1);
13358 }
13359 if (p < e) {
13360 if (rep) {
13361 rb_str_buf_cat(buf, rep, replen);
13362 if (!rep7bit_p) cr = ENC_CODERANGE_VALID;
13363 }
13364 else {
13365 repl = rb_yield(rb_enc_str_new(p, e-p, enc));
13366 str_mod_check(str, sp, slen);
13367 repl = str_compat_and_valid(repl, enc);
13368 rb_str_buf_cat(buf, RSTRING_PTR(repl), RSTRING_LEN(repl));
13371 }
13372 }
13373 }
13374 else {
13375 /* ASCII incompatible */
13376 long mbminlen = rb_enc_mbminlen(enc);
13377 if (!replen) {
13378 rep = NULL;
13379 }
13380 else if (!NIL_P(repl)) {
13381 rep = RSTRING_PTR(repl);
13382 replen = RSTRING_LEN(repl);
13383 }
13384 else if (encidx == ENCINDEX_UTF_16BE) {
13385 DEFAULT_REPLACE_CHAR("\xFF\xFD");
13386 }
13387 else if (encidx == ENCINDEX_UTF_16LE) {
13388 DEFAULT_REPLACE_CHAR("\xFD\xFF");
13389 }
13390 else if (encidx == ENCINDEX_UTF_32BE) {
13391 DEFAULT_REPLACE_CHAR("\x00\x00\xFF\xFD");
13392 }
13393 else if (encidx == ENCINDEX_UTF_32LE) {
13394 DEFAULT_REPLACE_CHAR("\xFD\xFF\x00\x00");
13395 }
13396 else {
13397 DEFAULT_REPLACE_CHAR("?");
13398 }
13399
13400 while (p < e) {
13401 int ret = rb_enc_precise_mbclen(p, e, enc);
13402 if (MBCLEN_NEEDMORE_P(ret)) {
13403 break;
13404 }
13405 else if (MBCLEN_CHARFOUND_P(ret)) {
13406 p += MBCLEN_CHARFOUND_LEN(ret);
13407 }
13408 else if (MBCLEN_INVALID_P(ret)) {
13409 const char *q = p;
13410 long clen = rb_enc_mbmaxlen(enc);
13411 if (NIL_P(buf)) buf = rb_str_buf_new(RSTRING_LEN(str));
13412 if (p > p1) rb_str_buf_cat(buf, p1, p - p1);
13413
13414 if (e - p < clen) clen = e - p;
13415 if (clen <= mbminlen * 2) {
13416 clen = mbminlen;
13417 }
13418 else {
13419 clen -= mbminlen;
13420 for (; clen > mbminlen; clen-=mbminlen) {
13421 ret = rb_enc_precise_mbclen(q, q + clen, enc);
13422 if (MBCLEN_NEEDMORE_P(ret)) break;
13423 if (MBCLEN_INVALID_P(ret)) continue;
13425 }
13426 }
13427 if (rep) {
13428 rb_str_buf_cat(buf, rep, replen);
13429 }
13430 else {
13431 repl = rb_yield(rb_enc_str_new(p, clen, enc));
13432 str_mod_check(str, sp, slen);
13433 repl = str_compat_and_valid(repl, enc);
13434 rb_str_buf_cat(buf, RSTRING_PTR(repl), RSTRING_LEN(repl));
13435 }
13436 p += clen;
13437 p1 = p;
13438 }
13439 else {
13441 }
13442 }
13443 if (NIL_P(buf)) {
13444 if (p == e) {
13446 return Qnil;
13447 }
13448 buf = rb_str_buf_new(RSTRING_LEN(str));
13449 }
13450 if (p1 < p) {
13451 rb_str_buf_cat(buf, p1, p - p1);
13452 }
13453 if (p < e) {
13454 if (rep) {
13455 rb_str_buf_cat(buf, rep, replen);
13456 }
13457 else {
13458 repl = rb_yield(rb_enc_str_new(p, e-p, enc));
13459 str_mod_check(str, sp, slen);
13460 repl = str_compat_and_valid(repl, enc);
13461 rb_str_buf_cat(buf, RSTRING_PTR(repl), RSTRING_LEN(repl));
13462 }
13463 }
13465 }
13466 ENCODING_CODERANGE_SET(buf, rb_enc_to_index(enc), cr);
13467 return buf;
13468}
13469
13470/*
13471 * call-seq:
13472 * scrub(replacement_string = default_replacement_string) -> new_string
13473 * scrub{|sequence| ... } -> new_string
13474 *
13475 * :include: doc/string/scrub.rdoc
13476 *
13477 */
13478static VALUE
13479str_scrub(int argc, VALUE *argv, VALUE str)
13480{
13481 VALUE repl = argc ? (rb_check_arity(argc, 0, 1), argv[0]) : Qnil;
13482 VALUE new = rb_str_scrub(str, repl);
13483 return NIL_P(new) ? str_duplicate(rb_cString, str): new;
13484}
13485
13486/*
13487 * call-seq:
13488 * scrub!(replacement_string = default_replacement_string) -> self
13489 * scrub!{|sequence| ... } -> self
13490 *
13491 * Like String#scrub, except that:
13492 *
13493 * - Any replacements are made in +self+.
13494 * - Returns +self+.
13495 *
13496 * Related: see {Modifying}[rdoc-ref:String@Modifying].
13497 *
13498 */
13499static VALUE
13500str_scrub_bang(int argc, VALUE *argv, VALUE str)
13501{
13502 VALUE repl = argc ? (rb_check_arity(argc, 0, 1), argv[0]) : Qnil;
13503 VALUE new = rb_str_scrub(str, repl);
13504 if (!NIL_P(new)) rb_str_replace(str, new);
13505 return str;
13506}
13507
13508static ID id_normalize;
13509static ID id_normalized_p;
13510static VALUE mUnicodeNormalize;
13511
13512static VALUE
13513unicode_normalize_common(int argc, VALUE *argv, VALUE str, ID id)
13514{
13515 static int UnicodeNormalizeRequired = 0;
13516 VALUE argv2[2];
13517
13518 if (!UnicodeNormalizeRequired) {
13519 rb_require("unicode_normalize/normalize.rb");
13520 UnicodeNormalizeRequired = 1;
13521 }
13522 argv2[0] = str;
13523 if (rb_check_arity(argc, 0, 1)) argv2[1] = argv[0];
13524 return rb_funcallv(mUnicodeNormalize, id, argc+1, argv2);
13525}
13526
13527/*
13528 * call-seq:
13529 * unicode_normalize(form = :nfc) -> string
13530 *
13531 * :include: doc/string/unicode_normalize.rdoc
13532 *
13533 */
13534static VALUE
13535rb_str_unicode_normalize(int argc, VALUE *argv, VALUE str)
13536{
13537 return unicode_normalize_common(argc, argv, str, id_normalize);
13538}
13539
13540/*
13541 * call-seq:
13542 * unicode_normalize!(form = :nfc) -> self
13543 *
13544 * Like String#unicode_normalize, except that the normalization
13545 * is performed on +self+ (not on a copy of +self+).
13546 *
13547 * Related: see {Modifying}[rdoc-ref:String@Modifying].
13548 *
13549 */
13550static VALUE
13551rb_str_unicode_normalize_bang(int argc, VALUE *argv, VALUE str)
13552{
13553 return rb_str_replace(str, unicode_normalize_common(argc, argv, str, id_normalize));
13554}
13555
13556/* call-seq:
13557 * unicode_normalized?(form = :nfc) -> true or false
13558 *
13559 * Returns whether +self+ is in the given +form+ of Unicode normalization;
13560 * see String#unicode_normalize.
13561 *
13562 * The +form+ must be one of +:nfc+, +:nfd+, +:nfkc+, or +:nfkd+.
13563 *
13564 * Examples:
13565 *
13566 * "a\u0300".unicode_normalized? # => false
13567 * "a\u0300".unicode_normalized?(:nfd) # => true
13568 * "\u00E0".unicode_normalized? # => true
13569 * "\u00E0".unicode_normalized?(:nfd) # => false
13570 *
13571 *
13572 * Raises an exception if +self+ is not in a Unicode encoding:
13573 *
13574 * s = "\xE0".force_encoding(Encoding::ISO_8859_1)
13575 * s.unicode_normalized? # Raises Encoding::CompatibilityError
13576 *
13577 * Related: see {Querying}[rdoc-ref:String@Querying].
13578 */
13579static VALUE
13580rb_str_unicode_normalized_p(int argc, VALUE *argv, VALUE str)
13581{
13582 return unicode_normalize_common(argc, argv, str, id_normalized_p);
13583}
13584
13585/**********************************************************************
13586 * Document-class: Symbol
13587 *
13588 * A +Symbol+ object represents a named identifier inside the Ruby interpreter.
13589 *
13590 * You can create a +Symbol+ object explicitly with:
13591 *
13592 * - A {symbol literal}[rdoc-ref:syntax/literals.rdoc@Symbol+Literals].
13593 *
13594 * The same +Symbol+ object will be
13595 * created for a given name or string for the duration of a program's
13596 * execution, regardless of the context or meaning of that name. Thus
13597 * if <code>Fred</code> is a constant in one context, a method in
13598 * another, and a class in a third, the +Symbol+ <code>:Fred</code>
13599 * will be the same object in all three contexts.
13600 *
13601 * module One
13602 * class Fred
13603 * end
13604 * $f1 = :Fred
13605 * end
13606 * module Two
13607 * Fred = 1
13608 * $f2 = :Fred
13609 * end
13610 * def Fred()
13611 * end
13612 * $f3 = :Fred
13613 * $f1.object_id #=> 2514190
13614 * $f2.object_id #=> 2514190
13615 * $f3.object_id #=> 2514190
13616 *
13617 * Constant, method, and variable names are returned as symbols:
13618 *
13619 * module One
13620 * Two = 2
13621 * def three; 3 end
13622 * @four = 4
13623 * @@five = 5
13624 * $six = 6
13625 * end
13626 * seven = 7
13627 *
13628 * One.constants
13629 * # => [:Two]
13630 * One.instance_methods(true)
13631 * # => [:three]
13632 * One.instance_variables
13633 * # => [:@four]
13634 * One.class_variables
13635 * # => [:@@five]
13636 * global_variables.grep(/six/)
13637 * # => [:$six]
13638 * local_variables
13639 * # => [:seven]
13640 *
13641 * A +Symbol+ object differs from a String object in that
13642 * a +Symbol+ object represents an identifier, while a String object
13643 * represents text or data.
13644 *
13645 * == What's Here
13646 *
13647 * First, what's elsewhere. Class +Symbol+:
13648 *
13649 * - Inherits from {class Object}[rdoc-ref:Object@Whats+Here].
13650 * - Includes {module Comparable}[rdoc-ref:Comparable@Whats+Here].
13651 *
13652 * Here, class +Symbol+ provides methods that are useful for:
13653 *
13654 * - {Querying}[rdoc-ref:Symbol@Methods+for+Querying]
13655 * - {Comparing}[rdoc-ref:Symbol@Methods+for+Comparing]
13656 * - {Converting}[rdoc-ref:Symbol@Methods+for+Converting]
13657 *
13658 * === Methods for Querying
13659 *
13660 * - ::all_symbols: Returns an array of the symbols currently in Ruby's symbol table.
13661 * - #=~: Returns the index of the first substring in symbol that matches a
13662 * given Regexp or other object; returns +nil+ if no match is found.
13663 * - #[], #slice : Returns a substring of symbol
13664 * determined by a given index, start/length, or range, or string.
13665 * - #empty?: Returns +true+ if +self.length+ is zero; +false+ otherwise.
13666 * - #encoding: Returns the Encoding object that represents the encoding
13667 * of symbol.
13668 * - #end_with?: Returns +true+ if symbol ends with
13669 * any of the given strings.
13670 * - #match: Returns a MatchData object if symbol
13671 * matches a given Regexp; +nil+ otherwise.
13672 * - #match?: Returns +true+ if symbol
13673 * matches a given Regexp; +false+ otherwise.
13674 * - #length, #size: Returns the number of characters in symbol.
13675 * - #start_with?: Returns +true+ if symbol starts with
13676 * any of the given strings.
13677 *
13678 * === Methods for Comparing
13679 *
13680 * - #<=>: Returns -1, 0, or 1 as a given symbol is smaller than, equal to,
13681 * or larger than symbol.
13682 * - #==, #===: Returns +true+ if a given symbol has the same content and
13683 * encoding.
13684 * - #casecmp: Ignoring case, returns -1, 0, or 1 as a given
13685 * symbol is smaller than, equal to, or larger than symbol.
13686 * - #casecmp?: Returns +true+ if symbol is equal to a given symbol
13687 * after Unicode case folding; +false+ otherwise.
13688 *
13689 * === Methods for Converting
13690 *
13691 * - #capitalize: Returns symbol with the first character upcased
13692 * and all other characters downcased.
13693 * - #downcase: Returns symbol with all characters downcased.
13694 * - #inspect: Returns the string representation of +self+ as a symbol literal.
13695 * - #name: Returns the frozen string corresponding to symbol.
13696 * - #succ, #next: Returns the symbol that is the successor to symbol.
13697 * - #swapcase: Returns symbol with all upcase characters downcased
13698 * and all downcase characters upcased.
13699 * - #to_proc: Returns a Proc object which responds to the method named by symbol.
13700 * - #to_s, #id2name: Returns the string corresponding to +self+.
13701 * - #to_sym, #intern: Returns +self+.
13702 * - #upcase: Returns symbol with all characters upcased.
13703 *
13704 */
13705
13706
13707/*
13708 * call-seq:
13709 * self == other -> true or false
13710 *
13711 * Returns whether +other+ is the same object as +self+.
13712 */
13713
13714#define sym_equal rb_obj_equal
13715
13716static int
13717sym_printable(const char *s, const char *send, rb_encoding *enc)
13718{
13719 while (s < send) {
13720 int n;
13721 int c = rb_enc_precise_mbclen(s, send, enc);
13722
13723 if (!MBCLEN_CHARFOUND_P(c)) return FALSE;
13724 n = MBCLEN_CHARFOUND_LEN(c);
13725 c = rb_enc_mbc_to_codepoint(s, send, enc);
13726 if (!rb_enc_isprint(c, enc)) return FALSE;
13727 s += n;
13728 }
13729 return TRUE;
13730}
13731
13732int
13733rb_str_symname_p(VALUE sym)
13734{
13735 rb_encoding *enc;
13736 const char *ptr;
13737 long len;
13738 rb_encoding *resenc = rb_default_internal_encoding();
13739
13740 if (resenc == NULL) resenc = rb_default_external_encoding();
13741 enc = STR_ENC_GET(sym);
13742 ptr = RSTRING_PTR(sym);
13743 len = RSTRING_LEN(sym);
13744 if ((resenc != enc && !rb_str_is_ascii_only_p(sym)) || len != (long)strlen(ptr) ||
13745 !rb_enc_symname2_p(ptr, len, enc) || !sym_printable(ptr, ptr + len, enc)) {
13746 return FALSE;
13747 }
13748 return TRUE;
13749}
13750
13751VALUE
13752rb_str_quote_unprintable(VALUE str)
13753{
13754 rb_encoding *enc;
13755 const char *ptr;
13756 long len;
13757 rb_encoding *resenc;
13758
13759 Check_Type(str, T_STRING);
13760 resenc = rb_default_internal_encoding();
13761 if (resenc == NULL) resenc = rb_default_external_encoding();
13762 enc = STR_ENC_GET(str);
13763 ptr = RSTRING_PTR(str);
13764 len = RSTRING_LEN(str);
13765 if ((resenc != enc && !rb_str_is_ascii_only_p(str)) ||
13766 !sym_printable(ptr, ptr + len, enc)) {
13767 return rb_str_escape(str);
13768 }
13769 return str;
13770}
13771
13772VALUE
13773rb_id_quote_unprintable(ID id)
13774{
13775 VALUE str = rb_id2str(id);
13776 if (!rb_str_symname_p(str)) {
13777 return rb_str_escape(str);
13778 }
13779 return str;
13780}
13781
13782/*
13783 * call-seq:
13784 * inspect -> string
13785 *
13786 * Returns a string representation of +self+ (including the leading colon):
13787 *
13788 * :foo.inspect # => ":foo"
13789 *
13790 * Related: Symbol#to_s, Symbol#name.
13791 *
13792 */
13793
13794static VALUE
13795sym_inspect(VALUE sym)
13796{
13797 VALUE str = rb_sym2str(sym);
13798 const char *ptr;
13799 long len;
13800 char *dest;
13801
13802 if (!rb_str_symname_p(str)) {
13803 str = rb_str_inspect(str);
13804 len = RSTRING_LEN(str);
13805 rb_str_resize(str, len + 1);
13806 dest = RSTRING_PTR(str);
13807 memmove(dest + 1, dest, len);
13808 }
13809 else {
13810 rb_encoding *enc = STR_ENC_GET(str);
13811 VALUE orig_str = str;
13812
13813 len = RSTRING_LEN(orig_str);
13814 str = rb_enc_str_new(0, len + 1, enc);
13815
13816 // Get data pointer after allocation
13817 ptr = RSTRING_PTR(orig_str);
13818 dest = RSTRING_PTR(str);
13819 memcpy(dest + 1, ptr, len);
13820
13821 RB_GC_GUARD(orig_str);
13822 }
13823 dest[0] = ':';
13824
13826
13827 return str;
13828}
13829
13830VALUE
13832{
13833 return rb_sym2str(sym);
13834}
13835
13836VALUE
13837rb_sym_proc_call(ID mid, int argc, const VALUE *argv, int kw_splat, VALUE passed_proc)
13838{
13839 VALUE obj;
13840
13841 if (argc < 1) {
13842 rb_raise(rb_eArgError, "no receiver given");
13843 }
13844 obj = argv[0];
13845 return rb_funcall_with_block_kw(obj, mid, argc - 1, argv + 1, passed_proc, kw_splat);
13846}
13847
13848/*
13849 * call-seq:
13850 * succ
13851 *
13852 * Equivalent to <tt>self.to_s.succ.to_sym</tt>:
13853 *
13854 * :foo.succ # => :fop
13855 *
13856 * Related: String#succ.
13857 */
13858
13859static VALUE
13860sym_succ(VALUE sym)
13861{
13862 return rb_str_intern(rb_str_succ(rb_sym2str(sym)));
13863}
13864
13865/*
13866 * call-seq:
13867 * self <=> other -> -1, 0, 1, or nil
13868 *
13869 * Compares +self+ and +other+, using String#<=>.
13870 *
13871 * Returns:
13872 *
13873 * - <tt>self.to_s <=> other.to_s</tt>, if +other+ is a symbol.
13874 * - +nil+, otherwise.
13875 *
13876 * Examples:
13877 *
13878 * :bar <=> :foo # => -1
13879 * :foo <=> :foo # => 0
13880 * :foo <=> :bar # => 1
13881 * :foo <=> 'bar' # => nil
13882 *
13883 * \Class \Symbol includes module Comparable,
13884 * each of whose methods uses Symbol#<=> for comparison.
13885 *
13886 * Related: String#<=>.
13887 */
13888
13889static VALUE
13890sym_cmp(VALUE sym, VALUE other)
13891{
13892 if (!SYMBOL_P(other)) {
13893 return Qnil;
13894 }
13895 return rb_str_cmp_m(rb_sym2str(sym), rb_sym2str(other));
13896}
13897
13898/*
13899 * call-seq:
13900 * casecmp(object) -> -1, 0, 1, or nil
13901 *
13902 * :include: doc/symbol/casecmp.rdoc
13903 *
13904 */
13905
13906static VALUE
13907sym_casecmp(VALUE sym, VALUE other)
13908{
13909 if (!SYMBOL_P(other)) {
13910 return Qnil;
13911 }
13912 return str_casecmp(rb_sym2str(sym), rb_sym2str(other));
13913}
13914
13915/*
13916 * call-seq:
13917 * casecmp?(object) -> true, false, or nil
13918 *
13919 * :include: doc/symbol/casecmp_p.rdoc
13920 *
13921 */
13922
13923static VALUE
13924sym_casecmp_p(VALUE sym, VALUE other)
13925{
13926 if (!SYMBOL_P(other)) {
13927 return Qnil;
13928 }
13929 return str_casecmp_p(rb_sym2str(sym), rb_sym2str(other));
13930}
13931
13932/*
13933 * call-seq:
13934 * self =~ other -> integer or nil
13935 *
13936 * Equivalent to <tt>self.to_s =~ other</tt>,
13937 * including possible updates to global variables;
13938 * see String#=~.
13939 *
13940 */
13941
13942static VALUE
13943sym_match(VALUE sym, VALUE other)
13944{
13945 return rb_str_match(rb_sym2str(sym), other);
13946}
13947
13948/*
13949 * call-seq:
13950 * match(pattern, offset = 0) -> matchdata or nil
13951 * match(pattern, offset = 0) {|matchdata| } -> object
13952 *
13953 * Equivalent to <tt>self.to_s.match</tt>,
13954 * including possible updates to global variables;
13955 * see String#match.
13956 *
13957 */
13958
13959static VALUE
13960sym_match_m(int argc, VALUE *argv, VALUE sym)
13961{
13962 return rb_str_match_m(argc, argv, rb_sym2str(sym));
13963}
13964
13965/*
13966 * call-seq:
13967 * match?(pattern, offset) -> true or false
13968 *
13969 * Equivalent to <tt>sym.to_s.match?</tt>;
13970 * see String#match.
13971 *
13972 */
13973
13974static VALUE
13975sym_match_m_p(int argc, VALUE *argv, VALUE sym)
13976{
13977 return rb_str_match_m_p(argc, argv, sym);
13978}
13979
13980/*
13981 * call-seq:
13982 * self[offset] -> string or nil
13983 * self[offset, size] -> string or nil
13984 * self[range] -> string or nil
13985 * self[regexp, capture = 0] -> string or nil
13986 * self[substring] -> string or nil
13987 *
13988 * Equivalent to <tt>symbol.to_s[]</tt>; see String#[].
13989 *
13990 */
13991
13992static VALUE
13993sym_aref(int argc, VALUE *argv, VALUE sym)
13994{
13995 return rb_str_aref_m(argc, argv, rb_sym2str(sym));
13996}
13997
13998/*
13999 * call-seq:
14000 * length -> integer
14001 *
14002 * Equivalent to <tt>self.to_s.length</tt>; see String#length.
14003 */
14004
14005static VALUE
14006sym_length(VALUE sym)
14007{
14008 return rb_str_length(rb_sym2str(sym));
14009}
14010
14011/*
14012 * call-seq:
14013 * upcase(mapping) -> symbol
14014 *
14015 * Equivalent to <tt>sym.to_s.upcase.to_sym</tt>.
14016 *
14017 * See String#upcase.
14018 *
14019 */
14020
14021static VALUE
14022sym_upcase(int argc, VALUE *argv, VALUE sym)
14023{
14024 return rb_str_intern(rb_str_upcase(argc, argv, rb_sym2str(sym)));
14025}
14026
14027/*
14028 * call-seq:
14029 * downcase(mapping) -> symbol
14030 *
14031 * Equivalent to <tt>sym.to_s.downcase.to_sym</tt>.
14032 *
14033 * See String#downcase.
14034 *
14035 * Related: Symbol#upcase.
14036 *
14037 */
14038
14039static VALUE
14040sym_downcase(int argc, VALUE *argv, VALUE sym)
14041{
14042 return rb_str_intern(rb_str_downcase(argc, argv, rb_sym2str(sym)));
14043}
14044
14045/*
14046 * call-seq:
14047 * capitalize(mapping) -> symbol
14048 *
14049 * Equivalent to <tt>sym.to_s.capitalize.to_sym</tt>.
14050 *
14051 * See String#capitalize.
14052 *
14053 */
14054
14055static VALUE
14056sym_capitalize(int argc, VALUE *argv, VALUE sym)
14057{
14058 return rb_str_intern(rb_str_capitalize(argc, argv, rb_sym2str(sym)));
14059}
14060
14061/*
14062 * call-seq:
14063 * swapcase(mapping) -> symbol
14064 *
14065 * Equivalent to <tt>sym.to_s.swapcase.to_sym</tt>.
14066 *
14067 * See String#swapcase.
14068 *
14069 */
14070
14071static VALUE
14072sym_swapcase(int argc, VALUE *argv, VALUE sym)
14073{
14074 return rb_str_intern(rb_str_swapcase(argc, argv, rb_sym2str(sym)));
14075}
14076
14077/*
14078 * call-seq:
14079 * start_with?(*string_or_regexp) -> true or false
14080 *
14081 * Equivalent to <tt>self.to_s.start_with?</tt>; see String#start_with?.
14082 *
14083 */
14084
14085static VALUE
14086sym_start_with(int argc, VALUE *argv, VALUE sym)
14087{
14088 return rb_str_start_with(argc, argv, rb_sym2str(sym));
14089}
14090
14091/*
14092 * call-seq:
14093 * end_with?(*strings) -> true or false
14094 *
14095 *
14096 * Equivalent to <tt>self.to_s.end_with?</tt>; see String#end_with?.
14097 *
14098 */
14099
14100static VALUE
14101sym_end_with(int argc, VALUE *argv, VALUE sym)
14102{
14103 return rb_str_end_with(argc, argv, rb_sym2str(sym));
14104}
14105
14106/*
14107 * call-seq:
14108 * encoding -> encoding
14109 *
14110 * Equivalent to <tt>self.to_s.encoding</tt>; see String#encoding.
14111 *
14112 */
14113
14114static VALUE
14115sym_encoding(VALUE sym)
14116{
14117 return rb_obj_encoding(rb_sym2str(sym));
14118}
14119
14120static VALUE
14121string_for_symbol(VALUE name)
14122{
14123 if (!RB_TYPE_P(name, T_STRING)) {
14124 VALUE tmp = rb_check_string_type(name);
14125 if (NIL_P(tmp)) {
14126 rb_raise(rb_eTypeError, "%+"PRIsVALUE" is not a symbol nor a string",
14127 name);
14128 }
14129 name = tmp;
14130 }
14131 return name;
14132}
14133
14134ID
14136{
14137 if (SYMBOL_P(name)) {
14138 return SYM2ID(name);
14139 }
14140 name = string_for_symbol(name);
14141 return rb_intern_str(name);
14142}
14143
14144VALUE
14146{
14147 if (SYMBOL_P(name)) {
14148 return name;
14149 }
14150 name = string_for_symbol(name);
14151 return rb_str_intern(name);
14152}
14153
14154/*
14155 * call-seq:
14156 * Symbol.all_symbols -> array_of_symbols
14157 *
14158 * Returns an array of all symbols currently in Ruby's symbol table:
14159 *
14160 * Symbol.all_symbols.size # => 9334
14161 * Symbol.all_symbols.take(3) # => [:!, :"\"", :"#"]
14162 *
14163 */
14164
14165static VALUE
14166sym_all_symbols(VALUE _)
14167{
14168 return rb_sym_all_symbols();
14169}
14170
14171VALUE
14172rb_str_to_interned_str(VALUE str)
14173{
14174 return rb_fstring(str);
14175}
14176
14177VALUE
14178rb_interned_str(const char *ptr, long len)
14179{
14180 struct RString fake_str = {RBASIC_INIT};
14181 int encidx = ENCINDEX_US_ASCII;
14182 int coderange = ENC_CODERANGE_7BIT;
14183 if (len > 0 && search_nonascii(ptr, ptr + len)) {
14184 encidx = ENCINDEX_ASCII_8BIT;
14185 coderange = ENC_CODERANGE_VALID;
14186 }
14187 VALUE str = setup_fake_str(&fake_str, ptr, len, encidx);
14188 ENC_CODERANGE_SET(str, coderange);
14189 return register_fstring(str, true, false);
14190}
14191
14192VALUE
14194{
14195 return rb_interned_str(ptr, strlen(ptr));
14196}
14197
14198VALUE
14199rb_enc_interned_str(const char *ptr, long len, rb_encoding *enc)
14200{
14201 if (enc != NULL && UNLIKELY(rb_enc_autoload_p(enc))) {
14202 rb_enc_autoload(enc);
14203 }
14204
14205 struct RString fake_str = {RBASIC_INIT};
14206 return register_fstring(rb_setup_fake_str(&fake_str, ptr, len, enc), true, false);
14207}
14208
14209VALUE
14210rb_enc_literal_str(const char *ptr, long len, rb_encoding *enc)
14211{
14212 if (enc != NULL && UNLIKELY(rb_enc_autoload_p(enc))) {
14213 rb_enc_autoload(enc);
14214 }
14215
14216 struct RString fake_str = {RBASIC_INIT};
14217 VALUE str = register_fstring(rb_setup_fake_str(&fake_str, ptr, len, enc), true, true);
14218 RUBY_ASSERT(RB_OBJ_SHAREABLE_P(str) && (rb_gc_verify_shareable(str), 1));
14219 return str;
14220}
14221
14222VALUE
14224{
14225 return rb_enc_interned_str(ptr, strlen(ptr), enc);
14226}
14227
14228#if USE_YJIT || USE_ZJIT
14229void
14230rb_jit_str_concat_codepoint(VALUE str, VALUE codepoint)
14231{
14232 if (RB_LIKELY(ENCODING_GET_INLINED(str) == rb_ascii8bit_encindex())) {
14233 ssize_t code = RB_NUM2SSIZE(codepoint);
14234
14235 if (RB_LIKELY(code >= 0 && code < 0xff)) {
14236 rb_str_buf_cat_byte(str, (char) code);
14237 return;
14238 }
14239 }
14240
14241 rb_str_concat(str, codepoint);
14242}
14243#endif
14244
14245static int
14246fstring_set_class_i(VALUE *str, void *data)
14247{
14248 RBASIC_SET_CLASS(*str, rb_cString);
14249
14250 return ST_CONTINUE;
14251}
14252
14253void
14254Init_String(void)
14255{
14256 rb_cString = rb_define_class("String", rb_cObject);
14257
14258 rb_concurrent_set_foreach_with_replace(fstring_table_obj, fstring_set_class_i, NULL);
14259
14261 rb_define_alloc_func(rb_cString, empty_str_alloc);
14262 rb_define_singleton_method(rb_cString, "new", rb_str_s_new, -1);
14263 rb_define_singleton_method(rb_cString, "try_convert", rb_str_s_try_convert, 1);
14264 rb_define_method(rb_cString, "initialize", rb_str_init, -1);
14266 rb_define_method(rb_cString, "initialize_copy", rb_str_replace, 1);
14267 rb_define_method(rb_cString, "<=>", rb_str_cmp_m, 1);
14270 rb_define_method(rb_cString, "eql?", rb_str_eql, 1);
14271 rb_define_method(rb_cString, "hash", rb_str_hash_m, 0);
14272 rb_define_method(rb_cString, "casecmp", rb_str_casecmp, 1);
14273 rb_define_method(rb_cString, "casecmp?", rb_str_casecmp_p, 1);
14276 rb_define_method(rb_cString, "%", rb_str_format_m, 1);
14277 rb_define_method(rb_cString, "[]", rb_str_aref_m, -1);
14278 rb_define_method(rb_cString, "[]=", rb_str_aset_m, -1);
14279 rb_define_method(rb_cString, "insert", rb_str_insert, 2);
14282 rb_define_method(rb_cString, "bytesize", rb_str_bytesize, 0);
14283 rb_define_method(rb_cString, "empty?", rb_str_empty, 0);
14284 rb_define_method(rb_cString, "=~", rb_str_match, 1);
14285 rb_define_method(rb_cString, "match", rb_str_match_m, -1);
14286 rb_define_method(rb_cString, "match?", rb_str_match_m_p, -1);
14288 rb_define_method(rb_cString, "succ!", rb_str_succ_bang, 0);
14290 rb_define_method(rb_cString, "next!", rb_str_succ_bang, 0);
14291 rb_define_method(rb_cString, "upto", rb_str_upto, -1);
14292 rb_define_method(rb_cString, "index", rb_str_index_m, -1);
14293 rb_define_method(rb_cString, "byteindex", rb_str_byteindex_m, -1);
14294 rb_define_method(rb_cString, "rindex", rb_str_rindex_m, -1);
14295 rb_define_method(rb_cString, "byterindex", rb_str_byterindex_m, -1);
14296 rb_define_method(rb_cString, "clear", rb_str_clear, 0);
14297 rb_define_method(rb_cString, "chr", rb_str_chr, 0);
14298 rb_define_method(rb_cString, "getbyte", rb_str_getbyte, 1);
14299 rb_define_method(rb_cString, "setbyte", rb_str_setbyte, 2);
14300 rb_define_method(rb_cString, "bit_get", rb_str_bit_get, -1);
14301 rb_define_method(rb_cString, "bit_set?", rb_str_bit_set_p, -1);
14302 rb_define_method(rb_cString, "bit_set", rb_str_bit_set, -1);
14303 rb_define_method(rb_cString, "bit_clear", rb_str_bit_clear, -1);
14304 rb_define_method(rb_cString, "bit_flip", rb_str_bit_flip, -1);
14305 rb_define_method(rb_cString, "bit_count", rb_str_bit_count, -1);
14306 rb_define_method(rb_cString, "bitwise_not", rb_str_bitwise_not, 0);
14307 rb_define_method(rb_cString, "bitwise_not!", rb_str_bitwise_not_bang, 0);
14308 rb_define_method(rb_cString, "bitwise_and", rb_str_bitwise_and, 1);
14309 rb_define_method(rb_cString, "bitwise_and!", rb_str_bitwise_and_bang, 1);
14310 rb_define_method(rb_cString, "bitwise_or", rb_str_bitwise_or, 1);
14311 rb_define_method(rb_cString, "bitwise_or!", rb_str_bitwise_or_bang, 1);
14312 rb_define_method(rb_cString, "bitwise_xor", rb_str_bitwise_xor, 1);
14313 rb_define_method(rb_cString, "bitwise_xor!", rb_str_bitwise_xor_bang, 1);
14314 rb_define_method(rb_cString, "byteslice", rb_str_byteslice, -1);
14315 rb_define_method(rb_cString, "bytesplice", rb_str_bytesplice, -1);
14316 rb_define_method(rb_cString, "scrub", str_scrub, -1);
14317 rb_define_method(rb_cString, "scrub!", str_scrub_bang, -1);
14319 rb_define_method(rb_cString, "+@", str_uplus, 0);
14320 rb_define_method(rb_cString, "-@", str_uminus, 0);
14321 rb_define_method(rb_cString, "dup", rb_str_dup_m, 0);
14322 rb_define_alias(rb_cString, "dedup", "-@");
14323
14324 rb_define_method(rb_cString, "to_i", rb_str_to_i, -1);
14325 rb_define_method(rb_cString, "to_f", rb_str_to_f, 0);
14326 rb_define_method(rb_cString, "to_s", rb_str_to_s, 0);
14327 rb_define_method(rb_cString, "to_str", rb_str_to_s, 0);
14330 rb_define_method(rb_cString, "undump", str_undump, 0);
14331
14332 sym_ascii = ID2SYM(rb_intern_const("ascii"));
14333 sym_turkic = ID2SYM(rb_intern_const("turkic"));
14334 sym_lithuanian = ID2SYM(rb_intern_const("lithuanian"));
14335 sym_fold = ID2SYM(rb_intern_const("fold"));
14336
14337 rb_define_method(rb_cString, "upcase", rb_str_upcase, -1);
14338 rb_define_method(rb_cString, "downcase", rb_str_downcase, -1);
14339 rb_define_method(rb_cString, "capitalize", rb_str_capitalize, -1);
14340 rb_define_method(rb_cString, "swapcase", rb_str_swapcase, -1);
14341
14342 rb_define_method(rb_cString, "upcase!", rb_str_upcase_bang, -1);
14343 rb_define_method(rb_cString, "downcase!", rb_str_downcase_bang, -1);
14344 rb_define_method(rb_cString, "capitalize!", rb_str_capitalize_bang, -1);
14345 rb_define_method(rb_cString, "swapcase!", rb_str_swapcase_bang, -1);
14346
14347 rb_define_method(rb_cString, "hex", rb_str_hex, 0);
14348 rb_define_method(rb_cString, "oct", rb_str_oct, 0);
14349 rb_define_method(rb_cString, "split", rb_str_split_m, -1);
14350 rb_define_method(rb_cString, "lines", rb_str_lines, -1);
14351 rb_define_method(rb_cString, "bytes", rb_str_bytes, 0);
14352 rb_define_method(rb_cString, "chars", rb_str_chars, 0);
14353 rb_define_method(rb_cString, "codepoints", rb_str_codepoints, 0);
14354 rb_define_method(rb_cString, "grapheme_clusters", rb_str_grapheme_clusters, 0);
14355 rb_define_method(rb_cString, "reverse", rb_str_reverse, 0);
14356 rb_define_method(rb_cString, "reverse!", rb_str_reverse_bang, 0);
14357 rb_define_method(rb_cString, "concat", rb_str_concat_multi, -1);
14358 rb_define_method(rb_cString, "append_as_bytes", rb_str_append_as_bytes, -1);
14360 rb_define_method(rb_cString, "prepend", rb_str_prepend_multi, -1);
14361 rb_define_method(rb_cString, "crypt", rb_str_crypt, 1);
14362 rb_define_method(rb_cString, "intern", rb_str_intern, 0); /* in symbol.c */
14363 rb_define_method(rb_cString, "to_sym", rb_str_intern, 0); /* in symbol.c */
14364 rb_define_method(rb_cString, "ord", rb_str_ord, 0);
14365
14366 rb_define_method(rb_cString, "include?", rb_str_include, 1);
14367 rb_define_method(rb_cString, "start_with?", rb_str_start_with, -1);
14368 rb_define_method(rb_cString, "end_with?", rb_str_end_with, -1);
14369
14370 rb_define_method(rb_cString, "scan", rb_str_scan, 1);
14371
14372 rb_define_method(rb_cString, "ljust", rb_str_ljust, -1);
14373 rb_define_method(rb_cString, "rjust", rb_str_rjust, -1);
14374 rb_define_method(rb_cString, "center", rb_str_center, -1);
14375
14376 rb_define_method(rb_cString, "sub", rb_str_sub, -1);
14377 rb_define_method(rb_cString, "gsub", rb_str_gsub, -1);
14378 rb_define_method(rb_cString, "chop", rb_str_chop, 0);
14379 rb_define_method(rb_cString, "chomp", rb_str_chomp, -1);
14380 rb_define_method(rb_cString, "strip", rb_str_strip, -1);
14381 rb_define_method(rb_cString, "lstrip", rb_str_lstrip, -1);
14382 rb_define_method(rb_cString, "rstrip", rb_str_rstrip, -1);
14383 rb_define_method(rb_cString, "delete_prefix", rb_str_delete_prefix, 1);
14384 rb_define_method(rb_cString, "delete_suffix", rb_str_delete_suffix, 1);
14385
14386 rb_define_method(rb_cString, "sub!", rb_str_sub_bang, -1);
14387 rb_define_method(rb_cString, "gsub!", rb_str_gsub_bang, -1);
14388 rb_define_method(rb_cString, "chop!", rb_str_chop_bang, 0);
14389 rb_define_method(rb_cString, "chomp!", rb_str_chomp_bang, -1);
14390 rb_define_method(rb_cString, "strip!", rb_str_strip_bang, -1);
14391 rb_define_method(rb_cString, "lstrip!", rb_str_lstrip_bang, -1);
14392 rb_define_method(rb_cString, "rstrip!", rb_str_rstrip_bang, -1);
14393 rb_define_method(rb_cString, "delete_prefix!", rb_str_delete_prefix_bang, 1);
14394 rb_define_method(rb_cString, "delete_suffix!", rb_str_delete_suffix_bang, 1);
14395
14396 rb_define_method(rb_cString, "tr", rb_str_tr, -1);
14397 rb_define_method(rb_cString, "tr_s", rb_str_tr_s, 2);
14398 rb_define_method(rb_cString, "delete", rb_str_delete, -1);
14399 rb_define_method(rb_cString, "squeeze", rb_str_squeeze, -1);
14400 rb_define_method(rb_cString, "count", rb_str_count, -1);
14401
14402 rb_define_method(rb_cString, "tr!", rb_str_tr_bang, -1);
14403 rb_define_method(rb_cString, "tr_s!", rb_str_tr_s_bang, 2);
14404 rb_define_method(rb_cString, "delete!", rb_str_delete_bang, -1);
14405 rb_define_method(rb_cString, "squeeze!", rb_str_squeeze_bang, -1);
14406
14407 rb_define_method(rb_cString, "each_line", rb_str_each_line, -1);
14408 rb_define_method(rb_cString, "each_byte", rb_str_each_byte, 0);
14409 rb_define_method(rb_cString, "each_char", rb_str_each_char, 0);
14410 rb_define_method(rb_cString, "each_codepoint", rb_str_each_codepoint, 0);
14411 rb_define_method(rb_cString, "each_grapheme_cluster", rb_str_each_grapheme_cluster, 0);
14412
14413 rb_define_method(rb_cString, "sum", rb_str_sum, -1);
14414
14415 rb_define_method(rb_cString, "slice", rb_str_aref_m, -1);
14416 rb_define_method(rb_cString, "slice!", rb_str_slice_bang, -1);
14417
14418 rb_define_method(rb_cString, "partition", rb_str_partition, 1);
14419 rb_define_method(rb_cString, "rpartition", rb_str_rpartition, 1);
14420
14421 rb_define_method(rb_cString, "encoding", rb_obj_encoding, 0); /* in encoding.c */
14422 rb_define_method(rb_cString, "force_encoding", rb_str_force_encoding, 1);
14423 rb_define_method(rb_cString, "b", rb_str_b, 0);
14424
14425 /* define UnicodeNormalize module here so that we don't have to look it up */
14426 mUnicodeNormalize = rb_define_module("UnicodeNormalize");
14427 id_normalize = rb_intern_const("normalize");
14428 id_normalized_p = rb_intern_const("normalized?");
14429
14430 rb_define_method(rb_cString, "unicode_normalize", rb_str_unicode_normalize, -1);
14431 rb_define_method(rb_cString, "unicode_normalize!", rb_str_unicode_normalize_bang, -1);
14432 rb_define_method(rb_cString, "unicode_normalized?", rb_str_unicode_normalized_p, -1);
14433
14434 rb_fs = Qnil;
14435 rb_define_hooked_variable("$;", &rb_fs, 0, rb_fs_setter);
14436 rb_define_hooked_variable("$-F", &rb_fs, 0, rb_fs_setter);
14437 rb_gc_register_address(&rb_fs);
14438
14439 rb_cSymbol = rb_define_class("Symbol", rb_cObject);
14443 rb_define_singleton_method(rb_cSymbol, "all_symbols", sym_all_symbols, 0);
14444
14445 rb_define_method(rb_cSymbol, "==", sym_equal, 1);
14446 rb_define_method(rb_cSymbol, "===", sym_equal, 1);
14447 rb_define_method(rb_cSymbol, "inspect", sym_inspect, 0);
14448 rb_define_method(rb_cSymbol, "to_proc", rb_sym_to_proc, 0); /* in proc.c */
14449 rb_define_method(rb_cSymbol, "succ", sym_succ, 0);
14450 rb_define_method(rb_cSymbol, "next", sym_succ, 0);
14451
14452 rb_define_method(rb_cSymbol, "<=>", sym_cmp, 1);
14453 rb_define_method(rb_cSymbol, "casecmp", sym_casecmp, 1);
14454 rb_define_method(rb_cSymbol, "casecmp?", sym_casecmp_p, 1);
14455 rb_define_method(rb_cSymbol, "=~", sym_match, 1);
14456
14457 rb_define_method(rb_cSymbol, "[]", sym_aref, -1);
14458 rb_define_method(rb_cSymbol, "slice", sym_aref, -1);
14459 rb_define_method(rb_cSymbol, "length", sym_length, 0);
14460 rb_define_method(rb_cSymbol, "size", sym_length, 0);
14461 rb_define_method(rb_cSymbol, "match", sym_match_m, -1);
14462 rb_define_method(rb_cSymbol, "match?", sym_match_m_p, -1);
14463
14464 rb_define_method(rb_cSymbol, "upcase", sym_upcase, -1);
14465 rb_define_method(rb_cSymbol, "downcase", sym_downcase, -1);
14466 rb_define_method(rb_cSymbol, "capitalize", sym_capitalize, -1);
14467 rb_define_method(rb_cSymbol, "swapcase", sym_swapcase, -1);
14468
14469 rb_define_method(rb_cSymbol, "start_with?", sym_start_with, -1);
14470 rb_define_method(rb_cSymbol, "end_with?", sym_end_with, -1);
14471
14472 rb_define_method(rb_cSymbol, "encoding", sym_encoding, 0);
14473}
14474
14475#include "string.rbinc"
#define RUBY_ASSERT_ALWAYS(expr,...)
A variant of RUBY_ASSERT that does not interface with RUBY_DEBUG.
Definition assert.h:199
#define RBIMPL_ASSERT_OR_ASSUME(...)
This is either RUBY_ASSERT or RBIMPL_ASSUME, depending on RUBY_DEBUG.
Definition assert.h:311
#define RUBY_ASSERT_BUILTIN_TYPE(obj, type)
A variant of RUBY_ASSERT that asserts when either RUBY_DEBUG or built-in type of obj is type.
Definition assert.h:291
#define RUBY_ASSERT(...)
Asserts that the given expression is truthy if and only if RUBY_DEBUG is truthy.
Definition assert.h:219
Atomic operations.
@ RUBY_ENC_CODERANGE_7BIT
The object holds 0 to 127 inclusive and nothing else.
Definition coderange.h:39
static enum ruby_coderange_type RB_ENC_CODERANGE_AND(enum ruby_coderange_type a, enum ruby_coderange_type b)
"Mix" two code ranges into one.
Definition coderange.h:162
static int rb_isspace(int c)
Our own locale-insensitive version of isspace(3).
Definition ctype.h:395
static int rb_isascii(int c)
Our own locale-insensitive version of isascii(3).
Definition ctype.h:209
#define rb_define_method(klass, mid, func, arity)
Defines klass#mid.
#define rb_define_singleton_method(klass, mid, func, arity)
Defines klass.mid.
static bool rb_enc_is_newline(const char *p, const char *e, rb_encoding *enc)
Queries if the passed pointer points to a newline character.
Definition ctype.h:43
static bool rb_enc_isprint(OnigCodePoint c, rb_encoding *enc)
Identical to rb_isprint(), except it additionally takes an encoding.
Definition ctype.h:180
static bool rb_enc_isctype(OnigCodePoint c, OnigCtype t, rb_encoding *enc)
Queries if the passed code point is of passed character type in the passed encoding.
Definition ctype.h:63
VALUE rb_enc_sprintf(rb_encoding *enc, const char *fmt,...)
Identical to rb_sprintf(), except it additionally takes an encoding.
Definition sprintf.c:1231
static VALUE RB_OBJ_FROZEN_RAW(VALUE obj)
This is an implementation detail of RB_OBJ_FROZEN().
Definition fl_type.h:699
static VALUE RB_FL_TEST_RAW(VALUE obj, VALUE flags)
This is an implementation detail of RB_FL_TEST().
Definition fl_type.h:407
void rb_include_module(VALUE klass, VALUE module)
Includes a module to a class.
Definition class.c:1769
void rb_define_alias(VALUE klass, const char *name1, const char *name2)
Defines an alias of a method.
Definition class.c:3094
void rb_undef_method(VALUE klass, const char *name)
Defines an undef of a method.
Definition class.c:2897
int rb_scan_args(int argc, const VALUE *argv, const char *fmt,...)
Retrieves argument from argc and argv to given VALUE references according to the format string.
Definition class.c:3384
int rb_block_given_p(void)
Determines if the current method is given a block.
Definition eval.c:1035
int rb_get_kwargs(VALUE keyword_hash, const ID *table, int required, int optional, VALUE *values)
Keyword argument deconstructor.
Definition class.c:3173
#define TYPE(_)
Old name of rb_type.
Definition value_type.h:108
#define ENCODING_SET_INLINED(obj, i)
Old name of RB_ENCODING_SET_INLINED.
Definition encoding.h:106
#define RB_INTEGER_TYPE_P
Old name of rb_integer_type_p.
Definition value_type.h:87
#define ENC_CODERANGE_7BIT
Old name of RUBY_ENC_CODERANGE_7BIT.
Definition coderange.h:180
#define ENC_CODERANGE_VALID
Old name of RUBY_ENC_CODERANGE_VALID.
Definition coderange.h:181
#define FL_UNSET_RAW
Old name of RB_FL_UNSET_RAW.
Definition fl_type.h:130
#define rb_str_buf_cat2
Old name of rb_usascii_str_new_cstr.
Definition string.h:1707
#define ALLOCV
Old name of RB_ALLOCV.
Definition memory.h:404
#define ISSPACE
Old name of rb_isspace.
Definition ctype.h:88
#define T_STRING
Old name of RUBY_T_STRING.
Definition value_type.h:78
#define ENC_CODERANGE_CLEAN_P(cr)
Old name of RB_ENC_CODERANGE_CLEAN_P.
Definition coderange.h:183
#define ENC_CODERANGE_AND(a, b)
Old name of RB_ENC_CODERANGE_AND.
Definition coderange.h:188
#define Qundef
Old name of RUBY_Qundef.
#define INT2FIX
Old name of RB_INT2FIX.
Definition long.h:48
#define OBJ_FROZEN
Old name of RB_OBJ_FROZEN.
Definition fl_type.h:133
#define rb_str_cat2
Old name of rb_str_cat_cstr.
Definition string.h:1708
#define UNREACHABLE
Old name of RBIMPL_UNREACHABLE.
Definition assume.h:28
#define ID2SYM
Old name of RB_ID2SYM.
Definition symbol.h:44
#define T_BIGNUM
Old name of RUBY_T_BIGNUM.
Definition value_type.h:57
#define OBJ_FREEZE
Old name of RB_OBJ_FREEZE.
Definition fl_type.h:131
#define T_FIXNUM
Old name of RUBY_T_FIXNUM.
Definition value_type.h:63
#define UNREACHABLE_RETURN
Old name of RBIMPL_UNREACHABLE_RETURN.
Definition assume.h:29
#define SYM2ID
Old name of RB_SYM2ID.
Definition symbol.h:45
#define ENC_CODERANGE(obj)
Old name of RB_ENC_CODERANGE.
Definition coderange.h:184
#define CLASS_OF
Old name of rb_class_of.
Definition globals.h:205
#define ENC_CODERANGE_UNKNOWN
Old name of RUBY_ENC_CODERANGE_UNKNOWN.
Definition coderange.h:179
#define SIZET2NUM
Old name of RB_SIZE2NUM.
Definition size_t.h:62
#define FIXABLE
Old name of RB_FIXABLE.
Definition fixnum.h:25
#define xmalloc
Old name of ruby_xmalloc.
Definition xmalloc.h:53
#define ENCODING_GET(obj)
Old name of RB_ENCODING_GET.
Definition encoding.h:109
#define LONG2FIX
Old name of RB_INT2FIX.
Definition long.h:49
#define ISDIGIT
Old name of rb_isdigit.
Definition ctype.h:93
#define ENC_CODERANGE_MASK
Old name of RUBY_ENC_CODERANGE_MASK.
Definition coderange.h:178
#define ZALLOC_N
Old name of RB_ZALLOC_N.
Definition memory.h:401
#define T_HASH
Old name of RUBY_T_HASH.
Definition value_type.h:65
#define ALLOC_N
Old name of RB_ALLOC_N.
Definition memory.h:399
#define MBCLEN_CHARFOUND_LEN(ret)
Old name of ONIGENC_MBCLEN_CHARFOUND_LEN.
Definition encoding.h:517
#define FL_TEST_RAW
Old name of RB_FL_TEST_RAW.
Definition fl_type.h:128
#define FL_SET
Old name of RB_FL_SET.
Definition fl_type.h:125
#define rb_ary_new3
Old name of rb_ary_new_from_args.
Definition array.h:658
#define ENCODING_INLINE_MAX
Old name of RUBY_ENCODING_INLINE_MAX.
Definition encoding.h:67
#define LONG2NUM
Old name of RB_LONG2NUM.
Definition long.h:50
#define FL_ANY_RAW
Old name of RB_FL_ANY_RAW.
Definition fl_type.h:122
#define ISALPHA
Old name of rb_isalpha.
Definition ctype.h:92
#define MBCLEN_INVALID_P(ret)
Old name of ONIGENC_MBCLEN_INVALID_P.
Definition encoding.h:518
#define ISASCII
Old name of rb_isascii.
Definition ctype.h:85
#define ULL2NUM
Old name of RB_ULL2NUM.
Definition long_long.h:31
#define TOLOWER
Old name of rb_tolower.
Definition ctype.h:101
#define Qtrue
Old name of RUBY_Qtrue.
#define ST2FIX
Old name of RB_ST2FIX.
Definition st_data_t.h:33
#define MBCLEN_NEEDMORE_P(ret)
Old name of ONIGENC_MBCLEN_NEEDMORE_P.
Definition encoding.h:519
#define FIXNUM_MAX
Old name of RUBY_FIXNUM_MAX.
Definition fixnum.h:26
#define NUM2INT
Old name of RB_NUM2INT.
Definition int.h:44
#define Qnil
Old name of RUBY_Qnil.
#define Qfalse
Old name of RUBY_Qfalse.
#define FIX2LONG
Old name of RB_FIX2LONG.
Definition long.h:46
#define ENC_CODERANGE_BROKEN
Old name of RUBY_ENC_CODERANGE_BROKEN.
Definition coderange.h:182
#define scan_hex(s, l, e)
Old name of ruby_scan_hex.
Definition util.h:108
#define NIL_P
Old name of RB_NIL_P.
#define ALLOCV_N
Old name of RB_ALLOCV_N.
Definition memory.h:405
#define MBCLEN_CHARFOUND_P(ret)
Old name of ONIGENC_MBCLEN_CHARFOUND_P.
Definition encoding.h:516
#define NUM2ULL
Old name of RB_NUM2ULL.
Definition long_long.h:35
#define DBL2NUM
Old name of rb_float_new.
Definition double.h:29
#define ISPRINT
Old name of rb_isprint.
Definition ctype.h:86
#define BUILTIN_TYPE
Old name of RB_BUILTIN_TYPE.
Definition value_type.h:85
#define ENCODING_SHIFT
Old name of RUBY_ENCODING_SHIFT.
Definition encoding.h:68
#define FL_TEST
Old name of RB_FL_TEST.
Definition fl_type.h:127
#define FL_FREEZE
Old name of RUBY_FL_FREEZE.
Definition fl_type.h:65
#define NUM2LONG
Old name of RB_NUM2LONG.
Definition long.h:51
#define ENCODING_GET_INLINED(obj)
Old name of RB_ENCODING_GET_INLINED.
Definition encoding.h:108
#define ENC_CODERANGE_CLEAR(obj)
Old name of RB_ENC_CODERANGE_CLEAR.
Definition coderange.h:187
#define FL_UNSET
Old name of RB_FL_UNSET.
Definition fl_type.h:129
#define UINT2NUM
Old name of RB_UINT2NUM.
Definition int.h:46
#define ENCODING_IS_ASCII8BIT(obj)
Old name of RB_ENCODING_IS_ASCII8BIT.
Definition encoding.h:110
#define FIXNUM_P
Old name of RB_FIXNUM_P.
#define CONST_ID
Old name of RUBY_CONST_ID.
Definition symbol.h:47
#define rb_ary_new2
Old name of rb_ary_new_capa.
Definition array.h:657
#define ENC_CODERANGE_SET(obj, cr)
Old name of RB_ENC_CODERANGE_SET.
Definition coderange.h:186
#define ENCODING_CODERANGE_SET(obj, encindex, cr)
Old name of RB_ENCODING_CODERANGE_SET.
Definition coderange.h:189
#define FL_SET_RAW
Old name of RB_FL_SET_RAW.
Definition fl_type.h:126
#define SYMBOL_P
Old name of RB_SYMBOL_P.
Definition value_type.h:88
#define OBJ_FROZEN_RAW
Old name of RB_OBJ_FROZEN_RAW.
Definition fl_type.h:134
#define T_REGEXP
Old name of RUBY_T_REGEXP.
Definition value_type.h:77
#define ENCODING_MASK
Old name of RUBY_ENCODING_MASK.
Definition encoding.h:69
void rb_category_warn(rb_warning_category_t category, const char *fmt,...)
Identical to rb_category_warning(), except it reports unless $VERBOSE is nil.
Definition error.c:478
void rb_exc_raise(VALUE mesg)
Raises an exception in the current thread.
Definition eval.c:678
void rb_syserr_fail(int e, const char *mesg)
Raises appropriate exception that represents a C errno.
Definition error.c:4084
VALUE rb_eRangeError
RangeError exception.
Definition error.c:1477
VALUE rb_eTypeError
TypeError exception.
Definition error.c:1473
VALUE rb_eEncCompatError
Encoding::CompatibilityError exception.
Definition error.c:1480
VALUE rb_eRuntimeError
RuntimeError exception.
Definition error.c:1471
VALUE rb_eIndexError
IndexError exception.
Definition error.c:1475
@ RB_WARN_CATEGORY_DEPRECATED
Warning is for deprecated features.
Definition error.h:48
VALUE rb_cObject
Object class.
Definition object.c:60
VALUE rb_any_to_s(VALUE obj)
Generates a textual representation of the given object.
Definition object.c:658
VALUE rb_obj_alloc(VALUE klass)
Allocates an instance of the given class.
Definition object.c:2252
VALUE rb_obj_hide(VALUE obj)
Make the object invisible from Ruby code.
Definition object.c:94
VALUE rb_class_new_instance_pass_kw(int argc, const VALUE *argv, VALUE klass)
Identical to rb_class_new_instance(), except it passes the passed keywords if any to the #initialize ...
Definition object.c:2270
VALUE rb_obj_frozen_p(VALUE obj)
Same as RB_OBJ_FROZEN(), but returns Qtrue/Qfalse instead of #bool.
Definition object.c:1316
double rb_str_to_dbl(VALUE str, int mode)
Identical to rb_cstr_to_dbl(), except it accepts a Ruby's string instead of C's.
Definition object.c:3642
VALUE rb_obj_class(VALUE obj)
Queries the class of an object.
Definition object.c:234
VALUE rb_obj_dup(VALUE obj)
Duplicates the given object.
Definition object.c:556
VALUE rb_cSymbol
Symbol class.
Definition string.c:86
VALUE rb_cRange
Range class.
Definition range.c:35
VALUE rb_equal(VALUE lhs, VALUE rhs)
This function is an optimised version of calling #==.
Definition object.c:140
VALUE rb_obj_is_kind_of(VALUE obj, VALUE klass)
Queries if the given object is an instance (of possibly descendants) of the given class.
Definition object.c:906
VALUE rb_obj_freeze(VALUE obj)
Same as RB_OBJ_FREEZE(), but returns the given object.
Definition object.c:1309
VALUE rb_mComparable
Comparable module.
Definition compar.c:19
VALUE rb_cString
String class.
Definition string.c:85
VALUE rb_to_int(VALUE val)
Identical to rb_check_to_int(), except it raises in case of conversion mismatch.
Definition object.c:3328
Encoding relates APIs.
static char * rb_enc_left_char_head(const char *s, const char *p, const char *e, rb_encoding *enc)
Queries the left boundary of a character.
Definition encoding.h:683
static char * rb_enc_right_char_head(const char *s, const char *p, const char *e, rb_encoding *enc)
Queries the right boundary of a character.
Definition encoding.h:704
static unsigned int rb_enc_codepoint(const char *p, const char *e, rb_encoding *enc)
Queries the code point of character pointed by the passed pointer.
Definition encoding.h:571
static int rb_enc_mbmaxlen(rb_encoding *enc)
Queries the maximum number of bytes that the passed encoding needs to represent a character.
Definition encoding.h:447
static int RB_ENCODING_GET_INLINED(VALUE obj)
Queries the encoding of the passed object.
Definition encoding.h:99
static int rb_enc_code_to_mbclen(int c, rb_encoding *enc)
Identical to rb_enc_codelen(), except it returns 0 for invalid code points.
Definition encoding.h:619
static char * rb_enc_step_back(const char *s, const char *p, const char *e, int n, rb_encoding *enc)
Scans the string backwards for n characters.
Definition encoding.h:726
VALUE rb_str_conv_enc(VALUE str, rb_encoding *from, rb_encoding *to)
Encoding conversion main routine.
Definition string.c:1379
VALUE rb_enc_str_new_static(const char *ptr, long len, rb_encoding *enc)
Identical to rb_enc_str_new(), except it takes a C string literal.
Definition string.c:1244
char * rb_enc_nth(const char *head, const char *tail, long nth, rb_encoding *enc)
Queries the n-th character.
Definition string.c:3123
VALUE rb_str_conv_enc_opts(VALUE str, rb_encoding *from, rb_encoding *to, int ecflags, VALUE ecopts)
Identical to rb_str_conv_enc(), except it additionally takes IO encoder options.
Definition string.c:1263
VALUE rb_enc_interned_str(const char *ptr, long len, rb_encoding *enc)
Identical to rb_enc_str_new(), except it returns a "f"string.
Definition string.c:14199
long rb_memsearch(const void *x, long m, const void *y, long n, rb_encoding *enc)
Looks for the passed string in the passed buffer.
Definition re.c:285
long rb_enc_strlen(const char *head, const char *tail, rb_encoding *enc)
Counts the number of characters of the passed string, according to the passed encoding.
Definition string.c:2405
VALUE rb_enc_str_buf_cat(VALUE str, const char *ptr, long len, rb_encoding *enc)
Identical to rb_str_cat(), except it additionally takes an encoding.
Definition string.c:3848
VALUE rb_enc_str_new_cstr(const char *ptr, rb_encoding *enc)
Identical to rb_enc_str_new(), except it assumes the passed pointer is a pointer to a C string.
Definition string.c:1175
VALUE rb_str_export_to_enc(VALUE obj, rb_encoding *enc)
Identical to rb_str_export(), except it additionally takes an encoding.
Definition string.c:1484
VALUE rb_external_str_new_with_enc(const char *ptr, long len, rb_encoding *enc)
Identical to rb_external_str_new(), except it additionally takes an encoding.
Definition string.c:1385
int rb_enc_str_asciionly_p(VALUE str)
Queries if the passed string is "ASCII only".
Definition string.c:988
VALUE rb_enc_interned_str_cstr(const char *ptr, rb_encoding *enc)
Identical to rb_enc_str_new_cstr(), except it returns a "f"string.
Definition string.c:14223
long rb_str_coderange_scan_restartable(const char *str, const char *end, rb_encoding *enc, int *cr)
Scans the passed string until it finds something odd.
Definition string.c:844
int rb_enc_symname2_p(const char *name, long len, rb_encoding *enc)
Identical to rb_enc_symname_p(), except it additionally takes the passed string's length.
Definition symbol.c:858
rb_econv_result_t rb_econv_convert(rb_econv_t *ec, const unsigned char **source_buffer_ptr, const unsigned char *source_buffer_end, unsigned char **destination_buffer_ptr, unsigned char *destination_buffer_end, int flags)
Converts a string from an encoding to another.
Definition transcode.c:1487
rb_econv_result_t
return value of rb_econv_convert()
Definition transcode.h:30
@ econv_finished
The conversion stopped after converting everything.
Definition transcode.h:57
@ econv_destination_buffer_full
The conversion stopped because there is no destination.
Definition transcode.h:46
rb_econv_t * rb_econv_open_opts(const char *source_encoding, const char *destination_encoding, int ecflags, VALUE ecopts)
Identical to rb_econv_open(), except it additionally takes a hash of optional strings.
Definition transcode.c:2730
VALUE rb_str_encode(VALUE str, VALUE to, int ecflags, VALUE ecopts)
Converts the contents of the passed string from its encoding to the passed one.
Definition transcode.c:2993
void rb_econv_close(rb_econv_t *ec)
Destructs a converter.
Definition transcode.c:1744
VALUE rb_funcall(VALUE recv, ID mid, int n,...)
Calls a method.
Definition vm_eval.c:1123
VALUE rb_funcallv(VALUE recv, ID mid, int argc, const VALUE *argv)
Identical to rb_funcall(), except it takes the method arguments as a C array.
Definition vm_eval.c:1081
VALUE rb_funcall_with_block_kw(VALUE recv, ID mid, int argc, const VALUE *argv, VALUE procval, int kw_splat)
Identical to rb_funcallv_with_block(), except you can specify how to handle the last element of the g...
Definition vm_eval.c:1210
VALUE rb_check_array_type(VALUE obj)
Try converting an object to its array representation using its to_ary method, if any.
VALUE rb_ary_new(void)
Allocates a new, empty array.
VALUE rb_ary_new_capa(long capa)
Identical to rb_ary_new(), except it additionally specifies how many rooms of objects it should alloc...
VALUE rb_ary_push(VALUE ary, VALUE elem)
Special case of rb_ary_cat() that it adds only one element.
VALUE rb_ary_freeze(VALUE obj)
Freeze an array, preventing further modifications.
#define RETURN_SIZED_ENUMERATOR(obj, argc, argv, size_fn)
This roughly resembles return enum_for(__callee__) unless block_given?.
Definition enumerator.h:208
#define RETURN_ENUMERATOR(obj, argc, argv)
Identical to RETURN_SIZED_ENUMERATOR(), except its size is unknown.
Definition enumerator.h:242
#define UNLIMITED_ARGUMENTS
This macro is used in conjunction with rb_check_arity().
Definition error.h:35
static int rb_check_arity(int argc, int min, int max)
Ensures that the passed integer is in the passed range.
Definition error.h:284
VALUE rb_fs
The field separator character for inputs, or the $;.
Definition string.c:723
VALUE rb_default_rs
This is the default value of rb_rs, i.e.
Definition io.c:209
VALUE rb_backref_get(void)
Queries the last match, or Regexp.last_match, or the $~.
Definition vm.c:2131
VALUE rb_sym_all_symbols(void)
Collects every single bits of symbols that have ever interned in the entire history of the current pr...
Definition symbol.c:1215
void rb_backref_set(VALUE md)
Updates $~.
Definition vm.c:2137
int rb_range_values(VALUE range, VALUE *begp, VALUE *endp, int *exclp)
Deconstructs a range into its components.
Definition range.c:1857
VALUE rb_range_beg_len(VALUE range, long *begp, long *lenp, long len, int err)
Deconstructs a numerical range.
Definition range.c:1945
int rb_reg_backref_number(VALUE match, VALUE backref)
Queries the index of the given named capture.
Definition re.c:1387
int rb_reg_options(VALUE re)
Queries the options of the passed regular expression.
Definition re.c:4476
VALUE rb_reg_match(VALUE re, VALUE str)
This is the match operator.
Definition re.c:3970
void rb_match_busy(VALUE md)
Asserts that the given MatchData is "occupied".
Definition re.c:1631
VALUE rb_reg_nth_match(int n, VALUE md)
Queries the nth captured substring.
Definition re.c:2071
void rb_str_free(VALUE str)
Destroys the given string for no reason.
Definition string.c:1803
VALUE rb_str_new_shared(VALUE str)
Identical to rb_str_new_cstr(), except it takes a Ruby's string instead of C's.
Definition string.c:1549
VALUE rb_str_plus(VALUE lhs, VALUE rhs)
Generates a new string, concatenating the former to the latter.
Definition string.c:2556
#define rb_utf8_str_new_cstr(str)
Identical to rb_str_new_cstr, except it generates a string of "UTF-8" encoding.
Definition string.h:1608
#define rb_hash_end(h)
Just another name of st_hash_end.
Definition string.h:970
#define rb_hash_uint32(h, i)
Just another name of st_hash_uint32.
Definition string.h:964
VALUE rb_str_append(VALUE dst, VALUE src)
Identical to rb_str_buf_append(), except it converts the right hand side before concatenating.
Definition string.c:3913
VALUE rb_filesystem_str_new(const char *ptr, long len)
Identical to rb_str_new(), except it generates a string of "filesystem" encoding.
Definition string.c:1460
VALUE rb_sym_to_s(VALUE sym)
This is an rb_sym2str() + rb_str_dup() combo.
Definition string.c:13831
VALUE rb_str_times(VALUE str, VALUE num)
Repetition of a string.
Definition string.c:2630
VALUE rb_external_str_new(const char *ptr, long len)
Identical to rb_str_new(), except it generates a string of "default external" encoding.
Definition string.c:1436
VALUE rb_str_tmp_new(long len)
Allocates a "temporary" string.
Definition string.c:1797
long rb_str_offset(VALUE str, long pos)
"Inverse" of rb_str_sublen().
Definition string.c:3151
VALUE rb_str_succ(VALUE orig)
Searches for the "successor" of a string.
Definition string.c:5457
int rb_str_hash_cmp(VALUE str1, VALUE str2)
Compares two strings.
Definition string.c:4276
VALUE rb_str_subseq(VALUE str, long beg, long len)
Identical to rb_str_substr(), except the numbers are interpreted as byte offsets instead of character...
Definition string.c:3266
VALUE rb_str_ellipsize(VALUE str, long len)
Shortens str and adds three dots, an ellipsis, if it is longer than len characters.
Definition string.c:13146
st_index_t rb_memhash(const void *ptr, long len)
This is a universal hash function.
Definition random.c:1720
#define rb_str_new(str, len)
Allocates an instance of rb_cString.
Definition string.h:1523
void rb_str_shared_replace(VALUE dst, VALUE src)
Replaces the contents of the former with the latter.
Definition string.c:1839
#define rb_str_buf_cat
Just another name of rb_str_cat.
Definition string.h:1706
VALUE rb_str_new_static(const char *ptr, long len)
Identical to rb_str_new(), except it takes a C string literal.
Definition string.c:1209
#define rb_usascii_str_new(str, len)
Identical to rb_str_new, except it generates a string of "US ASCII" encoding.
Definition string.h:1557
size_t rb_str_capacity(VALUE str)
Queries the capacity of the given string.
Definition string.c:1023
VALUE rb_str_new_frozen(VALUE str)
Creates a frozen copy of the string, if necessary.
Definition string.c:1555
VALUE rb_str_dup(VALUE str)
Duplicates a string.
Definition string.c:2038
st_index_t rb_str_hash(VALUE str)
Calculates a hash value of a string.
Definition string.c:4262
VALUE rb_str_cat(VALUE dst, const char *src, long srclen)
Destructively appends the passed contents to the string.
Definition string.c:3681
VALUE rb_str_locktmp(VALUE str)
Obtains a "temporary lock" of the string.
long rb_str_strlen(VALUE str)
Counts the number of characters (not bytes) that are stored inside of the given string.
Definition string.c:2492
VALUE rb_str_resurrect(VALUE str)
Like rb_str_dup(), but always create an instance of rb_cString regardless of the given object's class...
Definition string.c:2056
#define rb_str_buf_new_cstr(str)
Identical to rb_str_new_cstr, except done differently.
Definition string.h:1664
#define rb_usascii_str_new_cstr(str)
Identical to rb_str_new_cstr, except it generates a string of "US ASCII" encoding.
Definition string.h:1592
VALUE rb_str_replace(VALUE dst, VALUE src)
Replaces the contents of the former object with the stringised contents of the latter.
Definition string.c:6676
VALUE rb_str_no_gvl_safe_acquire(VALUE orig)
Creates a frozen copy of orig that guarantees the RSTRING_PTR is safe to use in operations that relea...
Definition string.c:1584
char * rb_str_subpos(VALUE str, long beg, long *len)
Identical to rb_str_substr(), except it returns a C's string instead of Ruby's.
Definition string.c:3274
rb_gvar_setter_t rb_str_setter
This is a rb_gvar_setter_t that refutes non-string assignments.
Definition string.h:1171
VALUE rb_interned_str_cstr(const char *ptr)
Identical to rb_interned_str(), except it assumes the passed pointer is a pointer to a C's string.
Definition string.c:14193
VALUE rb_filesystem_str_new_cstr(const char *ptr)
Identical to rb_filesystem_str_new(), except it assumes the passed pointer is a pointer to a C string...
Definition string.c:1466
#define rb_external_str_new_cstr(str)
Identical to rb_str_new_cstr, except it generates a string of "default external" encoding.
Definition string.h:1629
VALUE rb_str_buf_append(VALUE dst, VALUE src)
Identical to rb_str_cat_cstr(), except it takes Ruby's string instead of C's.
Definition string.c:3879
long rb_str_sublen(VALUE str, long pos)
Byte offset to character offset conversion.
Definition string.c:3198
VALUE rb_str_equal(VALUE str1, VALUE str2)
Equality of two strings.
Definition string.c:4383
void rb_str_no_gvl_safe_release(VALUE orig, VALUE tmp)
Releases a string created from rb_str_no_gvl_safe_acquire.
Definition string.c:1627
void rb_str_set_len(VALUE str, long len)
Overwrites the length of the string.
Definition string.c:3500
VALUE rb_str_inspect(VALUE str)
Generates a "readable" version of the receiver.
Definition string.c:8152
void rb_must_asciicompat(VALUE obj)
Asserts that the given string's encoding is (Ruby's definition of) ASCII compatible.
Definition string.c:2862
VALUE rb_interned_str(const char *ptr, long len)
Identical to rb_str_new(), except it returns an infamous "f"string.
Definition string.c:14178
int rb_str_cmp(VALUE lhs, VALUE rhs)
Compares two strings, as in strcmp(3).
Definition string.c:4330
VALUE rb_str_concat(VALUE dst, VALUE src)
Identical to rb_str_append(), except it also accepts an integer as a codepoint.
Definition string.c:4150
int rb_str_comparable(VALUE str1, VALUE str2)
Checks if two strings are comparable each other or not.
Definition string.c:4305
#define rb_strlen_lit(str)
Length of a string literal.
Definition string.h:1717
VALUE rb_str_buf_cat_ascii(VALUE dst, const char *src)
Identical to rb_str_cat_cstr(), except it additionally assumes the source string be a NUL terminated ...
Definition string.c:3855
VALUE rb_str_freeze(VALUE str)
This is the implementation of String#freeze.
Definition string.c:3391
void rb_str_update(VALUE dst, long beg, long len, VALUE src)
Replaces some (or all) of the contents of the given string.
Definition string.c:5944
VALUE rb_str_scrub(VALUE str, VALUE repl)
"Cleanses" the string.
Definition string.c:13204
#define rb_locale_str_new_cstr(str)
Identical to rb_external_str_new_cstr, except it generates a string of "locale" encoding instead of "...
Definition string.h:1650
VALUE rb_str_new_with_class(VALUE obj, const char *ptr, long len)
Identical to rb_str_new(), except it takes the class of the allocating object.
Definition string.c:1753
#define rb_str_dup_frozen
Just another name of rb_str_new_frozen.
Definition string.h:656
VALUE rb_check_string_type(VALUE obj)
Try converting an object to its stringised representation using its to_str method,...
Definition string.c:3047
VALUE rb_str_substr(VALUE str, long beg, long len)
This is the implementation of two-argumented String#slice.
Definition string.c:3363
#define rb_str_cat_cstr(buf, str)
Identical to rb_str_cat(), except it assumes the passed pointer is a pointer to a C string.
Definition string.h:1681
VALUE rb_str_unlocktmp(VALUE str)
Releases a lock formerly obtained by rb_str_locktmp().
Definition string.c:3482
VALUE rb_utf8_str_new_static(const char *ptr, long len)
Identical to rb_str_new_static(), except it generates a string of "UTF-8" encoding instead of "binary...
Definition string.c:1238
#define rb_utf8_str_new(str, len)
Identical to rb_str_new, except it generates a string of "UTF-8" encoding.
Definition string.h:1574
void rb_str_modify_expand(VALUE str, long capa)
Identical to rb_str_modify(), except it additionally expands the capacity of the receiver.
Definition string.c:2816
VALUE rb_str_dump(VALUE str)
"Inverse" of rb_eval_string().
Definition string.c:8269
VALUE rb_locale_str_new(const char *ptr, long len)
Identical to rb_str_new(), except it generates a string of "locale" encoding.
Definition string.c:1448
VALUE rb_str_buf_new(long capa)
Allocates a "string buffer".
Definition string.c:1769
VALUE rb_str_length(VALUE)
Identical to rb_str_strlen(), except it returns the value in rb_cInteger.
Definition string.c:2506
#define rb_str_new_cstr(str)
Identical to rb_str_new, except it assumes the passed pointer is a pointer to a C string.
Definition string.h:1539
VALUE rb_str_drop_bytes(VALUE str, long len)
Shrinks the given string for the given number of bytes.
Definition string.c:5859
VALUE rb_str_split(VALUE str, const char *delim)
Divides the given string based on the given delimiter.
Definition string.c:10802
VALUE rb_usascii_str_new_static(const char *ptr, long len)
Identical to rb_str_new_static(), except it generates a string of "US ASCII" encoding instead of "bin...
Definition string.c:1232
VALUE rb_str_intern(VALUE str)
Identical to rb_to_symbol(), except it assumes the receiver being an instance of RString.
Definition symbol.c:1085
VALUE rb_obj_as_string(VALUE obj)
Try converting an object to its stringised representation using its to_s method, if any.
Definition string.c:1902
VALUE rb_ivar_set(VALUE obj, ID name, VALUE val)
Identical to rb_iv_set(), except it accepts the name as an ID instead of a C string.
Definition variable.c:2141
VALUE rb_ivar_defined(VALUE obj, ID name)
Queries if the instance variable is defined at the object.
Definition variable.c:2201
int rb_respond_to(VALUE obj, ID mid)
Queries if the object responds to the method.
Definition vm_method.c:3693
void rb_undef_alloc_func(VALUE klass)
Deletes the allocator function of a class.
Definition vm_method.c:1846
void rb_define_alloc_func(VALUE klass, rb_alloc_func_t func)
Sets the allocator function of a class.
static ID rb_intern_const(const char *str)
This is a "tiny optimisation" over rb_intern().
Definition symbol.h:285
VALUE rb_sym2str(VALUE symbol)
Obtain a frozen string representation of a symbol (not including the leading colon).
Definition symbol.c:1148
VALUE rb_to_symbol(VALUE name)
Identical to rb_intern_str(), except it generates a dynamic symbol if necessary.
Definition string.c:14145
ID rb_to_id(VALUE str)
Identical to rb_intern_str(), except it tries to convert the parameter object to an instance of rb_cS...
Definition string.c:14135
int capa
Designed capacity of the buffer.
Definition io.h:11
int off
Offset inside of ptr.
Definition io.h:5
int len
Length of the buffer.
Definition io.h:8
#define RB_OBJ_SET_SHAREABLE(obj)
Wrapper of rb_obj_set_shareable().
Definition ractor.h:290
#define RB_OBJ_SHAREABLE_P(obj)
Queries if the passed object has previously classified as shareable or not.
Definition ractor.h:255
long rb_reg_search(VALUE re, VALUE str, long pos, int dir)
Runs the passed regular expression over the passed string.
Definition re.c:2000
VALUE rb_reg_regcomp(VALUE str)
Creates a new instance of rb_cRegexp.
Definition re.c:3675
VALUE rb_str_format(int argc, const VALUE *argv, VALUE fmt)
Formats a string.
Definition sprintf.c:974
VALUE rb_yield(VALUE val)
Yields the block.
Definition vm_eval.c:1378
#define MEMCPY(p1, p2, type, n)
Handy macro to call memcpy.
Definition memory.h:372
#define ALLOCA_N(type, n)
Definition memory.h:292
#define MEMZERO(p, type, n)
Handy macro to erase a region of memory.
Definition memory.h:360
#define RB_GC_GUARD(v)
Prevents premature destruction of local objects.
Definition memory.h:167
void rb_define_hooked_variable(const char *q, VALUE *w, type *e, void_type *r)
Define a function-backended global variable.
VALUE type(ANYARGS)
ANYARGS-ed function type.
void rb_hash_foreach(VALUE q, int_type *w, VALUE e)
Iteration over the given hash.
VALUE rb_ensure(type *q, VALUE w, type *e, VALUE r)
An equivalent of ensure clause.
Defines RBIMPL_ATTR_NONSTRING.
static int RARRAY_LENINT(VALUE ary)
Identical to rb_array_len(), except it differs for the return type.
Definition rarray.h:280
#define RARRAY_CONST_PTR
Just another name of rb_array_const_ptr.
Definition rarray.h:51
static VALUE RBASIC_CLASS(VALUE obj)
Queries the class of an object.
Definition rbasic.h:166
#define RBASIC(obj)
Convenient casting macro.
Definition rbasic.h:40
#define RHASH_SIZE(h)
Queries the size of the hash.
Definition rhash.h:57
static VALUE RREGEXP_SRC(VALUE rexp)
Convenient getter function.
Definition rregexp.h:102
#define StringValue(v)
Ensures that the parameter object is a String.
Definition rstring.h:66
VALUE rb_str_export_locale(VALUE obj)
Identical to rb_str_export(), except it converts into the locale encoding instead.
Definition string.c:1478
char * rb_string_value_cstr(volatile VALUE *ptr)
Identical to rb_string_value_ptr(), except it additionally checks for the contents for viability as a...
Definition string.c:3018
static int RSTRING_LENINT(VALUE str)
Identical to RSTRING_LEN(), except it differs for the return type.
Definition rstring.h:438
static char * RSTRING_END(VALUE str)
Queries the end of the contents pointer of the string.
Definition rstring.h:409
#define RSTRING_GETMEM(str, ptrvar, lenvar)
Convenient macro to obtain the contents and length at once.
Definition rstring.h:450
VALUE rb_string_value(volatile VALUE *ptr)
Identical to rb_str_to_str(), except it fills the passed pointer with the converted object.
Definition string.c:2881
#define RSTRING(obj)
Convenient casting macro.
Definition rstring.h:41
VALUE rb_str_export(VALUE obj)
Identical to rb_str_to_str(), except it additionally converts the string into default external encodi...
Definition string.c:1472
char * rb_string_value_ptr(volatile VALUE *ptr)
Identical to rb_str_to_str(), except it returns the converted string's backend memory region.
Definition string.c:2894
VALUE rb_str_to_str(VALUE obj)
Identical to rb_check_string_type(), except it raises exceptions in case of conversion failures.
Definition string.c:1830
#define StringValueCStr(v)
Identical to StringValuePtr, except it additionally checks for the contents for viability as a C stri...
Definition rstring.h:89
#define DATA_PTR(obj)
Convenient casting macro for backward compatibility.
Definition rtypeddata.h:439
#define TypedData_Wrap_Struct(klass, data_type, sval)
Converts sval, a pointer to your struct, into a Ruby object.
Definition rtypeddata.h:557
VALUE rb_require(const char *feature)
Identical to rb_require_string(), except it takes C's string instead of Ruby's.
Definition load.c:1528
#define errno
Ractor-aware version of errno.
Definition ruby.h:388
#define RB_NUM2SSIZE
Converts an instance of rb_cInteger into C's ssize_t.
Definition size_t.h:49
#define RTEST
This is an old name of RB_TEST.
#define _(args)
This was a transition path from K&R to ANSI.
Definition stdarg.h:35
VALUE flags
Per-object flags.
Definition rbasic.h:81
Ruby's String.
Definition rstring.h:196
struct RBasic basic
Basic part, including flags and class.
Definition rstring.h:199
union RString::@60::@61::@63 aux
Auxiliary info.
long capa
Capacity of *ptr.
Definition rstring.h:232
long len
Length of the string, not including terminating NUL character.
Definition rstring.h:206
struct RString::@60::@61 heap
Strings that use separated memory region for contents use this pattern.
struct RString::@60::@62 embed
Embedded contents.
VALUE shared
Parent of the string.
Definition rstring.h:240
char * ptr
Pointer to the contents of the string.
Definition rstring.h:222
union RString::@60 as
String's specific fields.
This is the struct that holds necessary info for a struct.
Definition rtypeddata.h:242
Definition string.c:9193
void rb_nativethread_lock_lock(rb_nativethread_lock_t *lock)
Blocks until the current thread obtains a lock.
Definition thread.c:319
uintptr_t ID
Type that represents a Ruby identifier such as a variable name.
Definition value.h:52
uintptr_t VALUE
Type that represents a Ruby object.
Definition value.h:40
static enum ruby_value_type rb_type(VALUE obj)
Identical to RB_BUILTIN_TYPE(), except it can also accept special constants.
Definition value_type.h:225
static void Check_Type(VALUE v, enum ruby_value_type t)
Identical to RB_TYPE_P(), except it raises exceptions on predication failure.
Definition value_type.h:425
static bool RB_TYPE_P(VALUE obj, enum ruby_value_type t)
Queries if the given object is of given type.
Definition value_type.h:376
ruby_value_type
C-level type of an object.
Definition value_type.h:113