Ruby  2.0.0p594(2014-10-27revision48167)
string.c
Go to the documentation of this file.
00001 /**********************************************************************
00002 
00003   string.c -
00004 
00005   $Author: usa $
00006   created at: Mon Aug  9 17:12:58 JST 1993
00007 
00008   Copyright (C) 1993-2007 Yukihiro Matsumoto
00009   Copyright (C) 2000  Network Applied Communication Laboratory, Inc.
00010   Copyright (C) 2000  Information-technology Promotion Agency, Japan
00011 
00012 **********************************************************************/
00013 
00014 #include "ruby/ruby.h"
00015 #include "ruby/re.h"
00016 #include "ruby/encoding.h"
00017 #include "vm_core.h"
00018 #include "internal.h"
00019 #include "probes.h"
00020 #include <assert.h>
00021 
00022 #define BEG(no) (regs->beg[(no)])
00023 #define END(no) (regs->end[(no)])
00024 
00025 #include <math.h>
00026 #include <ctype.h>
00027 
00028 #ifdef HAVE_UNISTD_H
00029 #include <unistd.h>
00030 #endif
00031 
00032 #define numberof(array) (int)(sizeof(array) / sizeof((array)[0]))
00033 
00034 #undef rb_str_new_cstr
00035 #undef rb_tainted_str_new_cstr
00036 #undef rb_usascii_str_new_cstr
00037 #undef rb_external_str_new_cstr
00038 #undef rb_locale_str_new_cstr
00039 #undef rb_str_new2
00040 #undef rb_str_new3
00041 #undef rb_str_new4
00042 #undef rb_str_new5
00043 #undef rb_tainted_str_new2
00044 #undef rb_usascii_str_new2
00045 #undef rb_str_dup_frozen
00046 #undef rb_str_buf_new_cstr
00047 #undef rb_str_buf_new2
00048 #undef rb_str_buf_cat2
00049 #undef rb_str_cat2
00050 
00051 static VALUE rb_str_clear(VALUE str);
00052 
00053 VALUE rb_cString;
00054 VALUE rb_cSymbol;
00055 
00056 #define RUBY_MAX_CHAR_LEN 16
00057 #define STR_TMPLOCK FL_USER7
00058 #define STR_NOEMBED FL_USER1
00059 #define STR_SHARED  FL_USER2 /* = ELTS_SHARED */
00060 #define STR_ASSOC   FL_USER3
00061 #define STR_SHARED_P(s) FL_ALL((s), STR_NOEMBED|ELTS_SHARED)
00062 #define STR_ASSOC_P(s)  FL_ALL((s), STR_NOEMBED|STR_ASSOC)
00063 #define STR_NOCAPA  (STR_NOEMBED|ELTS_SHARED|STR_ASSOC)
00064 #define STR_NOCAPA_P(s) (FL_TEST((s),STR_NOEMBED) && FL_ANY((s),ELTS_SHARED|STR_ASSOC))
00065 #define STR_UNSET_NOCAPA(s) do {\
00066     if (FL_TEST((s),STR_NOEMBED)) FL_UNSET((s),(ELTS_SHARED|STR_ASSOC));\
00067 } while (0)
00068 
00069 
00070 #define STR_SET_NOEMBED(str) do {\
00071     FL_SET((str), STR_NOEMBED);\
00072     STR_SET_EMBED_LEN((str), 0);\
00073 } while (0)
00074 #define STR_SET_EMBED(str) FL_UNSET((str), STR_NOEMBED)
00075 #define STR_EMBED_P(str) (!FL_TEST((str), STR_NOEMBED))
00076 #define STR_SET_EMBED_LEN(str, n) do { \
00077     long tmp_n = (n);\
00078     RBASIC(str)->flags &= ~RSTRING_EMBED_LEN_MASK;\
00079     RBASIC(str)->flags |= (tmp_n) << RSTRING_EMBED_LEN_SHIFT;\
00080 } while (0)
00081 
00082 #define STR_SET_LEN(str, n) do { \
00083     if (STR_EMBED_P(str)) {\
00084         STR_SET_EMBED_LEN((str), (n));\
00085     }\
00086     else {\
00087         RSTRING(str)->as.heap.len = (n);\
00088     }\
00089 } while (0)
00090 
00091 #define STR_DEC_LEN(str) do {\
00092     if (STR_EMBED_P(str)) {\
00093         long n = RSTRING_LEN(str);\
00094         n--;\
00095         STR_SET_EMBED_LEN((str), n);\
00096     }\
00097     else {\
00098         RSTRING(str)->as.heap.len--;\
00099     }\
00100 } while (0)
00101 
00102 #define RESIZE_CAPA(str,capacity) do {\
00103     if (STR_EMBED_P(str)) {\
00104         if ((capacity) > RSTRING_EMBED_LEN_MAX) {\
00105             char *tmp = ALLOC_N(char, (capacity)+1);\
00106             memcpy(tmp, RSTRING_PTR(str), RSTRING_LEN(str));\
00107             RSTRING(str)->as.heap.ptr = tmp;\
00108             RSTRING(str)->as.heap.len = RSTRING_LEN(str);\
00109             STR_SET_NOEMBED(str);\
00110             RSTRING(str)->as.heap.aux.capa = (capacity);\
00111         }\
00112     }\
00113     else {\
00114         REALLOC_N(RSTRING(str)->as.heap.ptr, char, (capacity)+1);\
00115         if (!STR_NOCAPA_P(str))\
00116             RSTRING(str)->as.heap.aux.capa = (capacity);\
00117     }\
00118 } while (0)
00119 
00120 #define is_ascii_string(str) (rb_enc_str_coderange(str) == ENC_CODERANGE_7BIT)
00121 #define is_broken_string(str) (rb_enc_str_coderange(str) == ENC_CODERANGE_BROKEN)
00122 
00123 #define STR_ENC_GET(str) rb_enc_from_index(ENCODING_GET(str))
00124 
00125 static inline int
00126 single_byte_optimizable(VALUE str)
00127 {
00128     rb_encoding *enc;
00129 
00130     /* Conservative.  It may be ENC_CODERANGE_UNKNOWN. */
00131     if (ENC_CODERANGE(str) == ENC_CODERANGE_7BIT)
00132         return 1;
00133 
00134     enc = STR_ENC_GET(str);
00135     if (rb_enc_mbmaxlen(enc) == 1)
00136         return 1;
00137 
00138     /* Conservative.  Possibly single byte.
00139      * "\xa1" in Shift_JIS for example. */
00140     return 0;
00141 }
00142 
00143 VALUE rb_fs;
00144 
00145 static inline const char *
00146 search_nonascii(const char *p, const char *e)
00147 {
00148 #if SIZEOF_VALUE == 8
00149 # define NONASCII_MASK 0x8080808080808080ULL
00150 #elif SIZEOF_VALUE == 4
00151 # define NONASCII_MASK 0x80808080UL
00152 #endif
00153 #ifdef NONASCII_MASK
00154     if ((int)sizeof(VALUE) * 2 < e - p) {
00155         const VALUE *s, *t;
00156         const VALUE lowbits = sizeof(VALUE) - 1;
00157         s = (const VALUE*)(~lowbits & ((VALUE)p + lowbits));
00158         while (p < (const char *)s) {
00159             if (!ISASCII(*p))
00160                 return p;
00161             p++;
00162         }
00163         t = (const VALUE*)(~lowbits & (VALUE)e);
00164         while (s < t) {
00165             if (*s & NONASCII_MASK) {
00166                 t = s;
00167                 break;
00168             }
00169             s++;
00170         }
00171         p = (const char *)t;
00172     }
00173 #endif
00174     while (p < e) {
00175         if (!ISASCII(*p))
00176             return p;
00177         p++;
00178     }
00179     return NULL;
00180 }
00181 
00182 static int
00183 coderange_scan(const char *p, long len, rb_encoding *enc)
00184 {
00185     const char *e = p + len;
00186 
00187     if (rb_enc_to_index(enc) == 0) {
00188         /* enc is ASCII-8BIT.  ASCII-8BIT string never be broken. */
00189         p = search_nonascii(p, e);
00190         return p ? ENC_CODERANGE_VALID : ENC_CODERANGE_7BIT;
00191     }
00192 
00193     if (rb_enc_asciicompat(enc)) {
00194         p = search_nonascii(p, e);
00195         if (!p) {
00196             return ENC_CODERANGE_7BIT;
00197         }
00198         while (p < e) {
00199             int ret = rb_enc_precise_mbclen(p, e, enc);
00200             if (!MBCLEN_CHARFOUND_P(ret)) {
00201                 return ENC_CODERANGE_BROKEN;
00202             }
00203             p += MBCLEN_CHARFOUND_LEN(ret);
00204             if (p < e) {
00205                 p = search_nonascii(p, e);
00206                 if (!p) {
00207                     return ENC_CODERANGE_VALID;
00208                 }
00209             }
00210         }
00211         if (e < p) {
00212             return ENC_CODERANGE_BROKEN;
00213         }
00214         return ENC_CODERANGE_VALID;
00215     }
00216 
00217     while (p < e) {
00218         int ret = rb_enc_precise_mbclen(p, e, enc);
00219 
00220         if (!MBCLEN_CHARFOUND_P(ret)) {
00221             return ENC_CODERANGE_BROKEN;
00222         }
00223         p += MBCLEN_CHARFOUND_LEN(ret);
00224     }
00225     if (e < p) {
00226         return ENC_CODERANGE_BROKEN;
00227     }
00228     return ENC_CODERANGE_VALID;
00229 }
00230 
00231 long
00232 rb_str_coderange_scan_restartable(const char *s, const char *e, rb_encoding *enc, int *cr)
00233 {
00234     const char *p = s;
00235 
00236     if (*cr == ENC_CODERANGE_BROKEN)
00237         return e - s;
00238 
00239     if (rb_enc_to_index(enc) == 0) {
00240         /* enc is ASCII-8BIT.  ASCII-8BIT string never be broken. */
00241         p = search_nonascii(p, e);
00242         *cr = (!p && *cr != ENC_CODERANGE_VALID) ? ENC_CODERANGE_7BIT : ENC_CODERANGE_VALID;
00243         return e - s;
00244     }
00245     else if (rb_enc_asciicompat(enc)) {
00246         p = search_nonascii(p, e);
00247         if (!p) {
00248             if (*cr != ENC_CODERANGE_VALID) *cr = ENC_CODERANGE_7BIT;
00249             return e - s;
00250         }
00251         while (p < e) {
00252             int ret = rb_enc_precise_mbclen(p, e, enc);
00253             if (!MBCLEN_CHARFOUND_P(ret)) {
00254                 *cr = MBCLEN_INVALID_P(ret) ? ENC_CODERANGE_BROKEN: ENC_CODERANGE_UNKNOWN;
00255                 return p - s;
00256             }
00257             p += MBCLEN_CHARFOUND_LEN(ret);
00258             if (p < e) {
00259                 p = search_nonascii(p, e);
00260                 if (!p) {
00261                     *cr = ENC_CODERANGE_VALID;
00262                     return e - s;
00263                 }
00264             }
00265         }
00266         *cr = e < p ? ENC_CODERANGE_BROKEN: ENC_CODERANGE_VALID;
00267         return p - s;
00268     }
00269     else {
00270         while (p < e) {
00271             int ret = rb_enc_precise_mbclen(p, e, enc);
00272             if (!MBCLEN_CHARFOUND_P(ret)) {
00273                 *cr = MBCLEN_INVALID_P(ret) ? ENC_CODERANGE_BROKEN: ENC_CODERANGE_UNKNOWN;
00274                 return p - s;
00275             }
00276             p += MBCLEN_CHARFOUND_LEN(ret);
00277         }
00278         *cr = e < p ? ENC_CODERANGE_BROKEN: ENC_CODERANGE_VALID;
00279         return p - s;
00280     }
00281 }
00282 
00283 static inline void
00284 str_enc_copy(VALUE str1, VALUE str2)
00285 {
00286     rb_enc_set_index(str1, ENCODING_GET(str2));
00287 }
00288 
00289 static void
00290 rb_enc_cr_str_copy_for_substr(VALUE dest, VALUE src)
00291 {
00292     /* this function is designed for copying encoding and coderange
00293      * from src to new string "dest" which is made from the part of src.
00294      */
00295     str_enc_copy(dest, src);
00296     if (RSTRING_LEN(dest) == 0) {
00297         if (!rb_enc_asciicompat(STR_ENC_GET(src)))
00298             ENC_CODERANGE_SET(dest, ENC_CODERANGE_VALID);
00299         else
00300             ENC_CODERANGE_SET(dest, ENC_CODERANGE_7BIT);
00301         return;
00302     }
00303     switch (ENC_CODERANGE(src)) {
00304       case ENC_CODERANGE_7BIT:
00305         ENC_CODERANGE_SET(dest, ENC_CODERANGE_7BIT);
00306         break;
00307       case ENC_CODERANGE_VALID:
00308         if (!rb_enc_asciicompat(STR_ENC_GET(src)) ||
00309             search_nonascii(RSTRING_PTR(dest), RSTRING_END(dest)))
00310             ENC_CODERANGE_SET(dest, ENC_CODERANGE_VALID);
00311         else
00312             ENC_CODERANGE_SET(dest, ENC_CODERANGE_7BIT);
00313         break;
00314       default:
00315         break;
00316     }
00317 }
00318 
00319 static void
00320 rb_enc_cr_str_exact_copy(VALUE dest, VALUE src)
00321 {
00322     str_enc_copy(dest, src);
00323     ENC_CODERANGE_SET(dest, ENC_CODERANGE(src));
00324 }
00325 
00326 int
00327 rb_enc_str_coderange(VALUE str)
00328 {
00329     int cr = ENC_CODERANGE(str);
00330 
00331     if (cr == ENC_CODERANGE_UNKNOWN) {
00332         rb_encoding *enc = STR_ENC_GET(str);
00333         cr = coderange_scan(RSTRING_PTR(str), RSTRING_LEN(str), enc);
00334         ENC_CODERANGE_SET(str, cr);
00335     }
00336     return cr;
00337 }
00338 
00339 int
00340 rb_enc_str_asciionly_p(VALUE str)
00341 {
00342     rb_encoding *enc = STR_ENC_GET(str);
00343 
00344     if (!rb_enc_asciicompat(enc))
00345         return FALSE;
00346     else if (rb_enc_str_coderange(str) == ENC_CODERANGE_7BIT)
00347         return TRUE;
00348     return FALSE;
00349 }
00350 
00351 static inline void
00352 str_mod_check(VALUE s, const char *p, long len)
00353 {
00354     if (RSTRING_PTR(s) != p || RSTRING_LEN(s) != len){
00355         rb_raise(rb_eRuntimeError, "string modified");
00356     }
00357 }
00358 
00359 size_t
00360 rb_str_capacity(VALUE str)
00361 {
00362     if (STR_EMBED_P(str)) {
00363         return RSTRING_EMBED_LEN_MAX;
00364     }
00365     else if (STR_NOCAPA_P(str)) {
00366         return RSTRING(str)->as.heap.len;
00367     }
00368     else {
00369         return RSTRING(str)->as.heap.aux.capa;
00370     }
00371 }
00372 
00373 static inline VALUE
00374 str_alloc(VALUE klass)
00375 {
00376     NEWOBJ_OF(str, struct RString, klass, T_STRING);
00377 
00378     str->as.heap.ptr = 0;
00379     str->as.heap.len = 0;
00380     str->as.heap.aux.capa = 0;
00381 
00382     return (VALUE)str;
00383 }
00384 
00385 static inline VALUE
00386 empty_str_alloc(VALUE klass)
00387 {
00388     if (RUBY_DTRACE_STRING_CREATE_ENABLED()) {
00389         RUBY_DTRACE_STRING_CREATE(0, rb_sourcefile(), rb_sourceline());
00390     }
00391     return str_alloc(klass);
00392 }
00393 
00394 static VALUE
00395 str_new(VALUE klass, const char *ptr, long len)
00396 {
00397     VALUE str;
00398 
00399     if (len < 0) {
00400         rb_raise(rb_eArgError, "negative string size (or size too big)");
00401     }
00402 
00403     if (RUBY_DTRACE_STRING_CREATE_ENABLED()) {
00404         RUBY_DTRACE_STRING_CREATE(len, rb_sourcefile(), rb_sourceline());
00405     }
00406 
00407     str = str_alloc(klass);
00408     if (len > RSTRING_EMBED_LEN_MAX) {
00409         RSTRING(str)->as.heap.aux.capa = len;
00410         RSTRING(str)->as.heap.ptr = ALLOC_N(char,len+1);
00411         STR_SET_NOEMBED(str);
00412     }
00413     else if (len == 0) {
00414         ENC_CODERANGE_SET(str, ENC_CODERANGE_7BIT);
00415     }
00416     if (ptr) {
00417         memcpy(RSTRING_PTR(str), ptr, len);
00418     }
00419     STR_SET_LEN(str, len);
00420     RSTRING_PTR(str)[len] = '\0';
00421     return str;
00422 }
00423 
00424 VALUE
00425 rb_str_new(const char *ptr, long len)
00426 {
00427     return str_new(rb_cString, ptr, len);
00428 }
00429 
00430 VALUE
00431 rb_usascii_str_new(const char *ptr, long len)
00432 {
00433     VALUE str = rb_str_new(ptr, len);
00434     ENCODING_CODERANGE_SET(str, rb_usascii_encindex(), ENC_CODERANGE_7BIT);
00435     return str;
00436 }
00437 
00438 VALUE
00439 rb_enc_str_new(const char *ptr, long len, rb_encoding *enc)
00440 {
00441     VALUE str = rb_str_new(ptr, len);
00442     rb_enc_associate(str, enc);
00443     return str;
00444 }
00445 
00446 VALUE
00447 rb_str_new_cstr(const char *ptr)
00448 {
00449     if (!ptr) {
00450         rb_raise(rb_eArgError, "NULL pointer given");
00451     }
00452     return rb_str_new(ptr, strlen(ptr));
00453 }
00454 
00455 RUBY_ALIAS_FUNCTION(rb_str_new2(const char *ptr), rb_str_new_cstr, (ptr))
00456 #define rb_str_new2 rb_str_new_cstr
00457 
00458 VALUE
00459 rb_usascii_str_new_cstr(const char *ptr)
00460 {
00461     VALUE str = rb_str_new2(ptr);
00462     ENCODING_CODERANGE_SET(str, rb_usascii_encindex(), ENC_CODERANGE_7BIT);
00463     return str;
00464 }
00465 
00466 RUBY_ALIAS_FUNCTION(rb_usascii_str_new2(const char *ptr), rb_usascii_str_new_cstr, (ptr))
00467 #define rb_usascii_str_new2 rb_usascii_str_new_cstr
00468 
00469 VALUE
00470 rb_tainted_str_new(const char *ptr, long len)
00471 {
00472     VALUE str = rb_str_new(ptr, len);
00473 
00474     OBJ_TAINT(str);
00475     return str;
00476 }
00477 
00478 VALUE
00479 rb_tainted_str_new_cstr(const char *ptr)
00480 {
00481     VALUE str = rb_str_new2(ptr);
00482 
00483     OBJ_TAINT(str);
00484     return str;
00485 }
00486 
00487 RUBY_ALIAS_FUNCTION(rb_tainted_str_new2(const char *ptr), rb_tainted_str_new_cstr, (ptr))
00488 #define rb_tainted_str_new2 rb_tainted_str_new_cstr
00489 
00490 VALUE
00491 rb_str_conv_enc_opts(VALUE str, rb_encoding *from, rb_encoding *to, int ecflags, VALUE ecopts)
00492 {
00493     extern VALUE rb_cEncodingConverter;
00494     rb_econv_t *ec;
00495     rb_econv_result_t ret;
00496     long len, olen;
00497     VALUE econv_wrapper;
00498     VALUE newstr;
00499     const unsigned char *start, *sp;
00500     unsigned char *dest, *dp;
00501     size_t converted_output = 0;
00502 
00503     if (!to) return str;
00504     if (!from) from = rb_enc_get(str);
00505     if (from == to) return str;
00506     if ((rb_enc_asciicompat(to) && ENC_CODERANGE(str) == ENC_CODERANGE_7BIT) ||
00507         to == rb_ascii8bit_encoding()) {
00508         if (STR_ENC_GET(str) != to) {
00509             str = rb_str_dup(str);
00510             rb_enc_associate(str, to);
00511         }
00512         return str;
00513     }
00514 
00515     len = RSTRING_LEN(str);
00516     newstr = rb_str_new(0, len);
00517     olen = len;
00518 
00519     econv_wrapper = rb_obj_alloc(rb_cEncodingConverter);
00520     RBASIC(econv_wrapper)->klass = 0;
00521     ec = rb_econv_open_opts(from->name, to->name, ecflags, ecopts);
00522     if (!ec) return str;
00523     DATA_PTR(econv_wrapper) = ec;
00524 
00525     sp = (unsigned char*)RSTRING_PTR(str);
00526     start = sp;
00527     while ((dest = (unsigned char*)RSTRING_PTR(newstr)),
00528            (dp = dest + converted_output),
00529            (ret = rb_econv_convert(ec, &sp, start + len, &dp, dest + olen, 0)),
00530            ret == econv_destination_buffer_full) {
00531         /* destination buffer short */
00532         size_t converted_input = sp - start;
00533         size_t rest = len - converted_input;
00534         converted_output = dp - dest;
00535         rb_str_set_len(newstr, converted_output);
00536         if (converted_input && converted_output &&
00537             rest < (LONG_MAX / converted_output)) {
00538             rest = (rest * converted_output) / converted_input;
00539         }
00540         else {
00541             rest = olen;
00542         }
00543         olen += rest < 2 ? 2 : rest;
00544         rb_str_resize(newstr, olen);
00545     }
00546     DATA_PTR(econv_wrapper) = 0;
00547     rb_econv_close(ec);
00548     rb_gc_force_recycle(econv_wrapper);
00549     switch (ret) {
00550       case econv_finished:
00551         len = dp - (unsigned char*)RSTRING_PTR(newstr);
00552         rb_str_set_len(newstr, len);
00553         rb_enc_associate(newstr, to);
00554         return newstr;
00555 
00556       default:
00557         /* some error, return original */
00558         return str;
00559     }
00560 }
00561 
00562 VALUE
00563 rb_str_conv_enc(VALUE str, rb_encoding *from, rb_encoding *to)
00564 {
00565     return rb_str_conv_enc_opts(str, from, to, 0, Qnil);
00566 }
00567 
00568 VALUE
00569 rb_external_str_new_with_enc(const char *ptr, long len, rb_encoding *eenc)
00570 {
00571     VALUE str;
00572 
00573     str = rb_tainted_str_new(ptr, len);
00574     if (eenc == rb_usascii_encoding() &&
00575         rb_enc_str_coderange(str) != ENC_CODERANGE_7BIT) {
00576         rb_enc_associate(str, rb_ascii8bit_encoding());
00577         return str;
00578     }
00579     rb_enc_associate(str, eenc);
00580     return rb_str_conv_enc(str, eenc, rb_default_internal_encoding());
00581 }
00582 
00583 VALUE
00584 rb_external_str_new(const char *ptr, long len)
00585 {
00586     return rb_external_str_new_with_enc(ptr, len, rb_default_external_encoding());
00587 }
00588 
00589 VALUE
00590 rb_external_str_new_cstr(const char *ptr)
00591 {
00592     return rb_external_str_new_with_enc(ptr, strlen(ptr), rb_default_external_encoding());
00593 }
00594 
00595 VALUE
00596 rb_locale_str_new(const char *ptr, long len)
00597 {
00598     return rb_external_str_new_with_enc(ptr, len, rb_locale_encoding());
00599 }
00600 
00601 VALUE
00602 rb_locale_str_new_cstr(const char *ptr)
00603 {
00604     return rb_external_str_new_with_enc(ptr, strlen(ptr), rb_locale_encoding());
00605 }
00606 
00607 VALUE
00608 rb_filesystem_str_new(const char *ptr, long len)
00609 {
00610     return rb_external_str_new_with_enc(ptr, len, rb_filesystem_encoding());
00611 }
00612 
00613 VALUE
00614 rb_filesystem_str_new_cstr(const char *ptr)
00615 {
00616     return rb_external_str_new_with_enc(ptr, strlen(ptr), rb_filesystem_encoding());
00617 }
00618 
00619 VALUE
00620 rb_str_export(VALUE str)
00621 {
00622     return rb_str_conv_enc(str, STR_ENC_GET(str), rb_default_external_encoding());
00623 }
00624 
00625 VALUE
00626 rb_str_export_locale(VALUE str)
00627 {
00628     return rb_str_conv_enc(str, STR_ENC_GET(str), rb_locale_encoding());
00629 }
00630 
00631 VALUE
00632 rb_str_export_to_enc(VALUE str, rb_encoding *enc)
00633 {
00634     return rb_str_conv_enc(str, STR_ENC_GET(str), enc);
00635 }
00636 
00637 static VALUE
00638 str_replace_shared_without_enc(VALUE str2, VALUE str)
00639 {
00640     if (RSTRING_LEN(str) <= RSTRING_EMBED_LEN_MAX) {
00641         STR_SET_EMBED(str2);
00642         memcpy(RSTRING_PTR(str2), RSTRING_PTR(str), RSTRING_LEN(str)+1);
00643         STR_SET_EMBED_LEN(str2, RSTRING_LEN(str));
00644     }
00645     else {
00646         str = rb_str_new_frozen(str);
00647         FL_SET(str2, STR_NOEMBED);
00648         RSTRING(str2)->as.heap.len = RSTRING_LEN(str);
00649         RSTRING(str2)->as.heap.ptr = RSTRING_PTR(str);
00650         RSTRING(str2)->as.heap.aux.shared = str;
00651         FL_SET(str2, ELTS_SHARED);
00652     }
00653     return str2;
00654 }
00655 
00656 static VALUE
00657 str_replace_shared(VALUE str2, VALUE str)
00658 {
00659     str_replace_shared_without_enc(str2, str);
00660     rb_enc_cr_str_exact_copy(str2, str);
00661     return str2;
00662 }
00663 
00664 static VALUE
00665 str_new_shared(VALUE klass, VALUE str)
00666 {
00667     return str_replace_shared(str_alloc(klass), str);
00668 }
00669 
00670 static VALUE
00671 str_new3(VALUE klass, VALUE str)
00672 {
00673     return str_new_shared(klass, str);
00674 }
00675 
00676 VALUE
00677 rb_str_new_shared(VALUE str)
00678 {
00679     VALUE str2 = str_new3(rb_obj_class(str), str);
00680 
00681     OBJ_INFECT(str2, str);
00682     return str2;
00683 }
00684 
00685 RUBY_ALIAS_FUNCTION(rb_str_new3(VALUE str), rb_str_new_shared, (str))
00686 #define rb_str_new3 rb_str_new_shared
00687 
00688 static VALUE
00689 str_new4(VALUE klass, VALUE str)
00690 {
00691     VALUE str2;
00692 
00693     str2 = str_alloc(klass);
00694     STR_SET_NOEMBED(str2);
00695     RSTRING(str2)->as.heap.len = RSTRING_LEN(str);
00696     RSTRING(str2)->as.heap.ptr = RSTRING_PTR(str);
00697     if (STR_SHARED_P(str)) {
00698         VALUE shared = RSTRING(str)->as.heap.aux.shared;
00699         assert(OBJ_FROZEN(shared));
00700         FL_SET(str2, ELTS_SHARED);
00701         RSTRING(str2)->as.heap.aux.shared = shared;
00702     }
00703     else {
00704         FL_SET(str, ELTS_SHARED);
00705         RSTRING(str)->as.heap.aux.shared = str2;
00706     }
00707     rb_enc_cr_str_exact_copy(str2, str);
00708     OBJ_INFECT(str2, str);
00709     return str2;
00710 }
00711 
00712 VALUE
00713 rb_str_new_frozen(VALUE orig)
00714 {
00715     VALUE klass, str;
00716 
00717     if (OBJ_FROZEN(orig)) return orig;
00718     klass = rb_obj_class(orig);
00719     if (STR_SHARED_P(orig) && (str = RSTRING(orig)->as.heap.aux.shared)) {
00720         long ofs;
00721         assert(OBJ_FROZEN(str));
00722         ofs = RSTRING_LEN(str) - RSTRING_LEN(orig);
00723         if ((ofs > 0) || (klass != RBASIC(str)->klass) ||
00724             ((RBASIC(str)->flags ^ RBASIC(orig)->flags) & (FL_TAINT|FL_UNTRUSTED)) ||
00725             ENCODING_GET(str) != ENCODING_GET(orig)) {
00726             str = str_new3(klass, str);
00727             RSTRING(str)->as.heap.ptr += ofs;
00728             RSTRING(str)->as.heap.len -= ofs;
00729             rb_enc_cr_str_exact_copy(str, orig);
00730             OBJ_INFECT(str, orig);
00731         }
00732     }
00733     else if (STR_EMBED_P(orig)) {
00734         str = str_new(klass, RSTRING_PTR(orig), RSTRING_LEN(orig));
00735         rb_enc_cr_str_exact_copy(str, orig);
00736         OBJ_INFECT(str, orig);
00737     }
00738     else if (STR_ASSOC_P(orig)) {
00739         VALUE assoc = RSTRING(orig)->as.heap.aux.shared;
00740         FL_UNSET(orig, STR_ASSOC);
00741         str = str_new4(klass, orig);
00742         FL_SET(str, STR_ASSOC);
00743         RSTRING(str)->as.heap.aux.shared = assoc;
00744     }
00745     else {
00746         str = str_new4(klass, orig);
00747     }
00748     OBJ_FREEZE(str);
00749     return str;
00750 }
00751 
00752 RUBY_ALIAS_FUNCTION(rb_str_new4(VALUE orig), rb_str_new_frozen, (orig))
00753 #define rb_str_new4 rb_str_new_frozen
00754 
00755 VALUE
00756 rb_str_new_with_class(VALUE obj, const char *ptr, long len)
00757 {
00758     return str_new(rb_obj_class(obj), ptr, len);
00759 }
00760 
00761 RUBY_ALIAS_FUNCTION(rb_str_new5(VALUE obj, const char *ptr, long len),
00762            rb_str_new_with_class, (obj, ptr, len))
00763 #define rb_str_new5 rb_str_new_with_class
00764 
00765 static VALUE
00766 str_new_empty(VALUE str)
00767 {
00768     VALUE v = rb_str_new5(str, 0, 0);
00769     rb_enc_copy(v, str);
00770     OBJ_INFECT(v, str);
00771     return v;
00772 }
00773 
00774 #define STR_BUF_MIN_SIZE 128
00775 
00776 VALUE
00777 rb_str_buf_new(long capa)
00778 {
00779     VALUE str = str_alloc(rb_cString);
00780 
00781     if (capa < STR_BUF_MIN_SIZE) {
00782         capa = STR_BUF_MIN_SIZE;
00783     }
00784     FL_SET(str, STR_NOEMBED);
00785     RSTRING(str)->as.heap.aux.capa = capa;
00786     RSTRING(str)->as.heap.ptr = ALLOC_N(char, capa+1);
00787     RSTRING(str)->as.heap.ptr[0] = '\0';
00788 
00789     return str;
00790 }
00791 
00792 VALUE
00793 rb_str_buf_new_cstr(const char *ptr)
00794 {
00795     VALUE str;
00796     long len = strlen(ptr);
00797 
00798     str = rb_str_buf_new(len);
00799     rb_str_buf_cat(str, ptr, len);
00800 
00801     return str;
00802 }
00803 
00804 RUBY_ALIAS_FUNCTION(rb_str_buf_new2(const char *ptr), rb_str_buf_new_cstr, (ptr))
00805 #define rb_str_buf_new2 rb_str_buf_new_cstr
00806 
00807 VALUE
00808 rb_str_tmp_new(long len)
00809 {
00810     return str_new(0, 0, len);
00811 }
00812 
00813 void *
00814 rb_alloc_tmp_buffer(volatile VALUE *store, long len)
00815 {
00816     VALUE s = rb_str_tmp_new(len);
00817     *store = s;
00818     return RSTRING_PTR(s);
00819 }
00820 
00821 void
00822 rb_free_tmp_buffer(volatile VALUE *store)
00823 {
00824     VALUE s = *store;
00825     *store = 0;
00826     if (s) rb_str_clear(s);
00827 }
00828 
00829 void
00830 rb_str_free(VALUE str)
00831 {
00832     if (!STR_EMBED_P(str) && !STR_SHARED_P(str)) {
00833         xfree(RSTRING(str)->as.heap.ptr);
00834     }
00835 }
00836 
00837 RUBY_FUNC_EXPORTED size_t
00838 rb_str_memsize(VALUE str)
00839 {
00840     if (!STR_EMBED_P(str) && !STR_SHARED_P(str)) {
00841         return RSTRING(str)->as.heap.aux.capa + 1; /* termlen */
00842     }
00843     else {
00844         return 0;
00845     }
00846 }
00847 
00848 VALUE
00849 rb_str_to_str(VALUE str)
00850 {
00851     return rb_convert_type(str, T_STRING, "String", "to_str");
00852 }
00853 
00854 static inline void str_discard(VALUE str);
00855 
00856 void
00857 rb_str_shared_replace(VALUE str, VALUE str2)
00858 {
00859     rb_encoding *enc;
00860     int cr;
00861     if (str == str2) return;
00862     enc = STR_ENC_GET(str2);
00863     cr = ENC_CODERANGE(str2);
00864     str_discard(str);
00865     OBJ_INFECT(str, str2);
00866     if (RSTRING_LEN(str2) <= RSTRING_EMBED_LEN_MAX) {
00867         STR_SET_EMBED(str);
00868         memcpy(RSTRING_PTR(str), RSTRING_PTR(str2), RSTRING_LEN(str2)+1);
00869         STR_SET_EMBED_LEN(str, RSTRING_LEN(str2));
00870         rb_enc_associate(str, enc);
00871         ENC_CODERANGE_SET(str, cr);
00872         return;
00873     }
00874     STR_SET_NOEMBED(str);
00875     STR_UNSET_NOCAPA(str);
00876     RSTRING(str)->as.heap.ptr = RSTRING_PTR(str2);
00877     RSTRING(str)->as.heap.len = RSTRING_LEN(str2);
00878     if (STR_NOCAPA_P(str2)) {
00879         FL_SET(str, RBASIC(str2)->flags & STR_NOCAPA);
00880         RSTRING(str)->as.heap.aux.shared = RSTRING(str2)->as.heap.aux.shared;
00881     }
00882     else {
00883         RSTRING(str)->as.heap.aux.capa = RSTRING(str2)->as.heap.aux.capa;
00884     }
00885     STR_SET_EMBED(str2);        /* abandon str2 */
00886     RSTRING_PTR(str2)[0] = 0;
00887     STR_SET_EMBED_LEN(str2, 0);
00888     rb_enc_associate(str, enc);
00889     ENC_CODERANGE_SET(str, cr);
00890 }
00891 
00892 static ID id_to_s;
00893 
00894 VALUE
00895 rb_obj_as_string(VALUE obj)
00896 {
00897     VALUE str;
00898 
00899     if (RB_TYPE_P(obj, T_STRING)) {
00900         return obj;
00901     }
00902     str = rb_funcall(obj, id_to_s, 0);
00903     if (!RB_TYPE_P(str, T_STRING))
00904         return rb_any_to_s(obj);
00905     if (OBJ_TAINTED(obj)) OBJ_TAINT(str);
00906     return str;
00907 }
00908 
00909 static VALUE
00910 str_replace(VALUE str, VALUE str2)
00911 {
00912     long len;
00913 
00914     len = RSTRING_LEN(str2);
00915     if (STR_ASSOC_P(str2)) {
00916         str2 = rb_str_new4(str2);
00917     }
00918     if (STR_SHARED_P(str2)) {
00919         VALUE shared = RSTRING(str2)->as.heap.aux.shared;
00920         assert(OBJ_FROZEN(shared));
00921         STR_SET_NOEMBED(str);
00922         RSTRING(str)->as.heap.len = len;
00923         RSTRING(str)->as.heap.ptr = RSTRING_PTR(str2);
00924         FL_SET(str, ELTS_SHARED);
00925         FL_UNSET(str, STR_ASSOC);
00926         RSTRING(str)->as.heap.aux.shared = shared;
00927     }
00928     else {
00929         str_replace_shared(str, str2);
00930     }
00931 
00932     OBJ_INFECT(str, str2);
00933     rb_enc_cr_str_exact_copy(str, str2);
00934     return str;
00935 }
00936 
00937 static VALUE
00938 str_duplicate(VALUE klass, VALUE str)
00939 {
00940     VALUE dup = str_alloc(klass);
00941     str_replace(dup, str);
00942     return dup;
00943 }
00944 
00945 VALUE
00946 rb_str_dup(VALUE str)
00947 {
00948     return str_duplicate(rb_obj_class(str), str);
00949 }
00950 
00951 VALUE
00952 rb_str_resurrect(VALUE str)
00953 {
00954     if (RUBY_DTRACE_STRING_CREATE_ENABLED()) {
00955         RUBY_DTRACE_STRING_CREATE(RSTRING_LEN(str),
00956                                   rb_sourcefile(), rb_sourceline());
00957     }
00958     return str_replace(str_alloc(rb_cString), str);
00959 }
00960 
00961 /*
00962  *  call-seq:
00963  *     String.new(str="")   -> new_str
00964  *
00965  *  Returns a new string object containing a copy of <i>str</i>.
00966  */
00967 
00968 static VALUE
00969 rb_str_init(int argc, VALUE *argv, VALUE str)
00970 {
00971     VALUE orig;
00972 
00973     if (argc > 0 && rb_scan_args(argc, argv, "01", &orig) == 1)
00974         rb_str_replace(str, orig);
00975     return str;
00976 }
00977 
00978 static inline long
00979 enc_strlen(const char *p, const char *e, rb_encoding *enc, int cr)
00980 {
00981     long c;
00982     const char *q;
00983 
00984     if (rb_enc_mbmaxlen(enc) == rb_enc_mbminlen(enc)) {
00985         return (e - p + rb_enc_mbminlen(enc) - 1) / rb_enc_mbminlen(enc);
00986     }
00987     else if (rb_enc_asciicompat(enc)) {
00988         c = 0;
00989         if (cr == ENC_CODERANGE_7BIT || cr == ENC_CODERANGE_VALID) {
00990             while (p < e) {
00991                 if (ISASCII(*p)) {
00992                     q = search_nonascii(p, e);
00993                     if (!q)
00994                         return c + (e - p);
00995                     c += q - p;
00996                     p = q;
00997                 }
00998                 p += rb_enc_fast_mbclen(p, e, enc);
00999                 c++;
01000             }
01001         }
01002         else {
01003             while (p < e) {
01004                 if (ISASCII(*p)) {
01005                     q = search_nonascii(p, e);
01006                     if (!q)
01007                         return c + (e - p);
01008                     c += q - p;
01009                     p = q;
01010                 }
01011                 p += rb_enc_mbclen(p, e, enc);
01012                 c++;
01013             }
01014         }
01015         return c;
01016     }
01017 
01018     for (c=0; p<e; c++) {
01019         p += rb_enc_mbclen(p, e, enc);
01020     }
01021     return c;
01022 }
01023 
01024 long
01025 rb_enc_strlen(const char *p, const char *e, rb_encoding *enc)
01026 {
01027     return enc_strlen(p, e, enc, ENC_CODERANGE_UNKNOWN);
01028 }
01029 
01030 long
01031 rb_enc_strlen_cr(const char *p, const char *e, rb_encoding *enc, int *cr)
01032 {
01033     long c;
01034     const char *q;
01035     int ret;
01036 
01037     *cr = 0;
01038     if (rb_enc_mbmaxlen(enc) == rb_enc_mbminlen(enc)) {
01039         return (e - p + rb_enc_mbminlen(enc) - 1) / rb_enc_mbminlen(enc);
01040     }
01041     else if (rb_enc_asciicompat(enc)) {
01042         c = 0;
01043         while (p < e) {
01044             if (ISASCII(*p)) {
01045                 q = search_nonascii(p, e);
01046                 if (!q) {
01047                     if (!*cr) *cr = ENC_CODERANGE_7BIT;
01048                     return c + (e - p);
01049                 }
01050                 c += q - p;
01051                 p = q;
01052             }
01053             ret = rb_enc_precise_mbclen(p, e, enc);
01054             if (MBCLEN_CHARFOUND_P(ret)) {
01055                 *cr |= ENC_CODERANGE_VALID;
01056                 p += MBCLEN_CHARFOUND_LEN(ret);
01057             }
01058             else {
01059                 *cr = ENC_CODERANGE_BROKEN;
01060                 p++;
01061             }
01062             c++;
01063         }
01064         if (!*cr) *cr = ENC_CODERANGE_7BIT;
01065         return c;
01066     }
01067 
01068     for (c=0; p<e; c++) {
01069         ret = rb_enc_precise_mbclen(p, e, enc);
01070         if (MBCLEN_CHARFOUND_P(ret)) {
01071             *cr |= ENC_CODERANGE_VALID;
01072             p += MBCLEN_CHARFOUND_LEN(ret);
01073         }
01074         else {
01075             *cr = ENC_CODERANGE_BROKEN;
01076             if (p + rb_enc_mbminlen(enc) <= e)
01077                 p += rb_enc_mbminlen(enc);
01078             else
01079                 p = e;
01080         }
01081     }
01082     if (!*cr) *cr = ENC_CODERANGE_7BIT;
01083     return c;
01084 }
01085 
01086 #ifdef NONASCII_MASK
01087 #define is_utf8_lead_byte(c) (((c)&0xC0) != 0x80)
01088 
01089 /*
01090  * UTF-8 leading bytes have either 0xxxxxxx or 11xxxxxx
01091  * bit represention. (see http://en.wikipedia.org/wiki/UTF-8)
01092  * Therefore, following pseudo code can detect UTF-8 leading byte.
01093  *
01094  * if (!(byte & 0x80))
01095  *   byte |= 0x40;          // turn on bit6
01096  * return ((byte>>6) & 1);  // bit6 represent it's leading byte or not.
01097  *
01098  * This function calculate every bytes in the argument word `s'
01099  * using the above logic concurrently. and gather every bytes result.
01100  */
01101 static inline VALUE
01102 count_utf8_lead_bytes_with_word(const VALUE *s)
01103 {
01104     VALUE d = *s;
01105 
01106     /* Transform into bit0 represent UTF-8 leading or not. */
01107     d |= ~(d>>1);
01108     d >>= 6;
01109     d &= NONASCII_MASK >> 7;
01110 
01111     /* Gather every bytes. */
01112     d += (d>>8);
01113     d += (d>>16);
01114 #if SIZEOF_VALUE == 8
01115     d += (d>>32);
01116 #endif
01117     return (d&0xF);
01118 }
01119 #endif
01120 
01121 static long
01122 str_strlen(VALUE str, rb_encoding *enc)
01123 {
01124     const char *p, *e;
01125     long n;
01126     int cr;
01127 
01128     if (single_byte_optimizable(str)) return RSTRING_LEN(str);
01129     if (!enc) enc = STR_ENC_GET(str);
01130     p = RSTRING_PTR(str);
01131     e = RSTRING_END(str);
01132     cr = ENC_CODERANGE(str);
01133 #ifdef NONASCII_MASK
01134     if (ENC_CODERANGE(str) == ENC_CODERANGE_VALID &&
01135         enc == rb_utf8_encoding()) {
01136 
01137         VALUE len = 0;
01138         if ((int)sizeof(VALUE) * 2 < e - p) {
01139             const VALUE *s, *t;
01140             const VALUE lowbits = sizeof(VALUE) - 1;
01141             s = (const VALUE*)(~lowbits & ((VALUE)p + lowbits));
01142             t = (const VALUE*)(~lowbits & (VALUE)e);
01143             while (p < (const char *)s) {
01144                 if (is_utf8_lead_byte(*p)) len++;
01145                 p++;
01146             }
01147             while (s < t) {
01148                 len += count_utf8_lead_bytes_with_word(s);
01149                 s++;
01150             }
01151             p = (const char *)s;
01152         }
01153         while (p < e) {
01154             if (is_utf8_lead_byte(*p)) len++;
01155             p++;
01156         }
01157         return (long)len;
01158     }
01159 #endif
01160     n = rb_enc_strlen_cr(p, e, enc, &cr);
01161     if (cr) {
01162         ENC_CODERANGE_SET(str, cr);
01163     }
01164     return n;
01165 }
01166 
01167 long
01168 rb_str_strlen(VALUE str)
01169 {
01170     return str_strlen(str, STR_ENC_GET(str));
01171 }
01172 
01173 /*
01174  *  call-seq:
01175  *     str.length   -> integer
01176  *     str.size     -> integer
01177  *
01178  *  Returns the character length of <i>str</i>.
01179  */
01180 
01181 VALUE
01182 rb_str_length(VALUE str)
01183 {
01184     long len;
01185 
01186     len = str_strlen(str, STR_ENC_GET(str));
01187     return LONG2NUM(len);
01188 }
01189 
01190 /*
01191  *  call-seq:
01192  *     str.bytesize  -> integer
01193  *
01194  *  Returns the length of +str+ in bytes.
01195  *
01196  *    "\x80\u3042".bytesize  #=> 4
01197  *    "hello".bytesize       #=> 5
01198  */
01199 
01200 static VALUE
01201 rb_str_bytesize(VALUE str)
01202 {
01203     return LONG2NUM(RSTRING_LEN(str));
01204 }
01205 
01206 /*
01207  *  call-seq:
01208  *     str.empty?   -> true or false
01209  *
01210  *  Returns <code>true</code> if <i>str</i> has a length of zero.
01211  *
01212  *     "hello".empty?   #=> false
01213  *     " ".empty?       #=> false
01214  *     "".empty?        #=> true
01215  */
01216 
01217 static VALUE
01218 rb_str_empty(VALUE str)
01219 {
01220     if (RSTRING_LEN(str) == 0)
01221         return Qtrue;
01222     return Qfalse;
01223 }
01224 
01225 /*
01226  *  call-seq:
01227  *     str + other_str   -> new_str
01228  *
01229  *  Concatenation---Returns a new <code>String</code> containing
01230  *  <i>other_str</i> concatenated to <i>str</i>.
01231  *
01232  *     "Hello from " + self.to_s   #=> "Hello from main"
01233  */
01234 
01235 VALUE
01236 rb_str_plus(VALUE str1, VALUE str2)
01237 {
01238     VALUE str3;
01239     rb_encoding *enc;
01240 
01241     StringValue(str2);
01242     enc = rb_enc_check(str1, str2);
01243     str3 = rb_str_new(0, RSTRING_LEN(str1)+RSTRING_LEN(str2));
01244     memcpy(RSTRING_PTR(str3), RSTRING_PTR(str1), RSTRING_LEN(str1));
01245     memcpy(RSTRING_PTR(str3) + RSTRING_LEN(str1),
01246            RSTRING_PTR(str2), RSTRING_LEN(str2));
01247     RSTRING_PTR(str3)[RSTRING_LEN(str3)] = '\0';
01248 
01249     if (OBJ_TAINTED(str1) || OBJ_TAINTED(str2))
01250         OBJ_TAINT(str3);
01251     ENCODING_CODERANGE_SET(str3, rb_enc_to_index(enc),
01252                            ENC_CODERANGE_AND(ENC_CODERANGE(str1), ENC_CODERANGE(str2)));
01253     return str3;
01254 }
01255 
01256 /*
01257  *  call-seq:
01258  *     str * integer   -> new_str
01259  *
01260  *  Copy --- Returns a new String containing +integer+ copies of the receiver.
01261  *  +integer+ must be greater than or equal to 0.
01262  *
01263  *     "Ho! " * 3   #=> "Ho! Ho! Ho! "
01264  *     "Ho! " * 0   #=> ""
01265  */
01266 
01267 VALUE
01268 rb_str_times(VALUE str, VALUE times)
01269 {
01270     VALUE str2;
01271     long n, len;
01272     char *ptr2;
01273 
01274     len = NUM2LONG(times);
01275     if (len < 0) {
01276         rb_raise(rb_eArgError, "negative argument");
01277     }
01278     if (len && LONG_MAX/len <  RSTRING_LEN(str)) {
01279         rb_raise(rb_eArgError, "argument too big");
01280     }
01281 
01282     str2 = rb_str_new5(str, 0, len *= RSTRING_LEN(str));
01283     ptr2 = RSTRING_PTR(str2);
01284     if (len) {
01285         n = RSTRING_LEN(str);
01286         memcpy(ptr2, RSTRING_PTR(str), n);
01287         while (n <= len/2) {
01288             memcpy(ptr2 + n, ptr2, n);
01289             n *= 2;
01290         }
01291         memcpy(ptr2 + n, ptr2, len-n);
01292     }
01293     ptr2[RSTRING_LEN(str2)] = '\0';
01294     OBJ_INFECT(str2, str);
01295     rb_enc_cr_str_copy_for_substr(str2, str);
01296 
01297     return str2;
01298 }
01299 
01300 /*
01301  *  call-seq:
01302  *     str % arg   -> new_str
01303  *
01304  *  Format---Uses <i>str</i> as a format specification, and returns the result
01305  *  of applying it to <i>arg</i>. If the format specification contains more than
01306  *  one substitution, then <i>arg</i> must be an <code>Array</code> or <code>Hash</code>
01307  *  containing the values to be substituted. See <code>Kernel::sprintf</code> for
01308  *  details of the format string.
01309  *
01310  *     "%05d" % 123                              #=> "00123"
01311  *     "%-5s: %08x" % [ "ID", self.object_id ]   #=> "ID   : 200e14d6"
01312  *     "foo = %{foo}" % { :foo => 'bar' }        #=> "foo = bar"
01313  */
01314 
01315 static VALUE
01316 rb_str_format_m(VALUE str, VALUE arg)
01317 {
01318     volatile VALUE tmp = rb_check_array_type(arg);
01319 
01320     if (!NIL_P(tmp)) {
01321         return rb_str_format(RARRAY_LENINT(tmp), RARRAY_PTR(tmp), str);
01322     }
01323     return rb_str_format(1, &arg, str);
01324 }
01325 
01326 static inline void
01327 str_modifiable(VALUE str)
01328 {
01329     if (FL_TEST(str, STR_TMPLOCK)) {
01330         rb_raise(rb_eRuntimeError, "can't modify string; temporarily locked");
01331     }
01332     rb_check_frozen(str);
01333     if (!OBJ_UNTRUSTED(str) && rb_safe_level() >= 4)
01334         rb_raise(rb_eSecurityError, "Insecure: can't modify string");
01335 }
01336 
01337 static inline int
01338 str_independent(VALUE str)
01339 {
01340     str_modifiable(str);
01341     if (!STR_SHARED_P(str)) return 1;
01342     if (STR_EMBED_P(str)) return 1;
01343     return 0;
01344 }
01345 
01346 static void
01347 str_make_independent_expand(VALUE str, long expand)
01348 {
01349     char *ptr;
01350     long len = RSTRING_LEN(str);
01351     long capa = len + expand;
01352 
01353     if (len > capa) len = capa;
01354     ptr = ALLOC_N(char, capa + 1);
01355     if (RSTRING_PTR(str)) {
01356         memcpy(ptr, RSTRING_PTR(str), len);
01357     }
01358     STR_SET_NOEMBED(str);
01359     STR_UNSET_NOCAPA(str);
01360     ptr[len] = 0;
01361     RSTRING(str)->as.heap.ptr = ptr;
01362     RSTRING(str)->as.heap.len = len;
01363     RSTRING(str)->as.heap.aux.capa = capa;
01364 }
01365 
01366 #define str_make_independent(str) str_make_independent_expand((str), 0L)
01367 
01368 void
01369 rb_str_modify(VALUE str)
01370 {
01371     if (!str_independent(str))
01372         str_make_independent(str);
01373     ENC_CODERANGE_CLEAR(str);
01374 }
01375 
01376 void
01377 rb_str_modify_expand(VALUE str, long expand)
01378 {
01379     if (expand < 0) {
01380         rb_raise(rb_eArgError, "negative expanding string size");
01381     }
01382     if (!str_independent(str)) {
01383         str_make_independent_expand(str, expand);
01384     }
01385     else if (expand > 0) {
01386         long len = RSTRING_LEN(str);
01387         long capa = len + expand;
01388         if (!STR_EMBED_P(str)) {
01389             REALLOC_N(RSTRING(str)->as.heap.ptr, char, capa+1);
01390             STR_UNSET_NOCAPA(str);
01391             RSTRING(str)->as.heap.aux.capa = capa;
01392         }
01393         else if (capa > RSTRING_EMBED_LEN_MAX) {
01394             str_make_independent_expand(str, expand);
01395         }
01396     }
01397     ENC_CODERANGE_CLEAR(str);
01398 }
01399 
01400 /* As rb_str_modify(), but don't clear coderange */
01401 static void
01402 str_modify_keep_cr(VALUE str)
01403 {
01404     if (!str_independent(str))
01405         str_make_independent(str);
01406     if (ENC_CODERANGE(str) == ENC_CODERANGE_BROKEN)
01407         /* Force re-scan later */
01408         ENC_CODERANGE_CLEAR(str);
01409 }
01410 
01411 static inline void
01412 str_discard(VALUE str)
01413 {
01414     str_modifiable(str);
01415     if (!STR_SHARED_P(str) && !STR_EMBED_P(str)) {
01416         xfree(RSTRING_PTR(str));
01417         RSTRING(str)->as.heap.ptr = 0;
01418         RSTRING(str)->as.heap.len = 0;
01419     }
01420 }
01421 
01422 void
01423 rb_str_associate(VALUE str, VALUE add)
01424 {
01425     /* sanity check */
01426     rb_check_frozen(str);
01427     if (STR_ASSOC_P(str)) {
01428         /* already associated */
01429         rb_ary_concat(RSTRING(str)->as.heap.aux.shared, add);
01430     }
01431     else {
01432         if (STR_SHARED_P(str)) {
01433             VALUE assoc = RSTRING(str)->as.heap.aux.shared;
01434             str_make_independent(str);
01435             if (STR_ASSOC_P(assoc)) {
01436                 assoc = RSTRING(assoc)->as.heap.aux.shared;
01437                 rb_ary_concat(assoc, add);
01438                 add = assoc;
01439             }
01440         }
01441         else if (STR_EMBED_P(str)) {
01442             str_make_independent(str);
01443         }
01444         else if (RSTRING(str)->as.heap.aux.capa != RSTRING_LEN(str)) {
01445             RESIZE_CAPA(str, RSTRING_LEN(str));
01446         }
01447         FL_SET(str, STR_ASSOC);
01448         RBASIC(add)->klass = 0;
01449         RSTRING(str)->as.heap.aux.shared = add;
01450     }
01451 }
01452 
01453 VALUE
01454 rb_str_associated(VALUE str)
01455 {
01456     if (STR_SHARED_P(str)) str = RSTRING(str)->as.heap.aux.shared;
01457     if (STR_ASSOC_P(str)) {
01458         return RSTRING(str)->as.heap.aux.shared;
01459     }
01460     return Qfalse;
01461 }
01462 
01463 void
01464 rb_must_asciicompat(VALUE str)
01465 {
01466     rb_encoding *enc = rb_enc_get(str);
01467     if (!rb_enc_asciicompat(enc)) {
01468         rb_raise(rb_eEncCompatError, "ASCII incompatible encoding: %s", rb_enc_name(enc));
01469     }
01470 }
01471 
01472 VALUE
01473 rb_string_value(volatile VALUE *ptr)
01474 {
01475     VALUE s = *ptr;
01476     if (!RB_TYPE_P(s, T_STRING)) {
01477         s = rb_str_to_str(s);
01478         *ptr = s;
01479     }
01480     return s;
01481 }
01482 
01483 char *
01484 rb_string_value_ptr(volatile VALUE *ptr)
01485 {
01486     VALUE str = rb_string_value(ptr);
01487     return RSTRING_PTR(str);
01488 }
01489 
01490 char *
01491 rb_string_value_cstr(volatile VALUE *ptr)
01492 {
01493     VALUE str = rb_string_value(ptr);
01494     char *s = RSTRING_PTR(str);
01495     long len = RSTRING_LEN(str);
01496 
01497     if (!s || memchr(s, 0, len)) {
01498         rb_raise(rb_eArgError, "string contains null byte");
01499     }
01500     if (s[len]) {
01501         rb_str_modify(str);
01502         s = RSTRING_PTR(str);
01503         s[RSTRING_LEN(str)] = 0;
01504     }
01505     return s;
01506 }
01507 
01508 VALUE
01509 rb_check_string_type(VALUE str)
01510 {
01511     str = rb_check_convert_type(str, T_STRING, "String", "to_str");
01512     return str;
01513 }
01514 
01515 /*
01516  *  call-seq:
01517  *     String.try_convert(obj) -> string or nil
01518  *
01519  *  Try to convert <i>obj</i> into a String, using to_str method.
01520  *  Returns converted string or nil if <i>obj</i> cannot be converted
01521  *  for any reason.
01522  *
01523  *     String.try_convert("str")     #=> "str"
01524  *     String.try_convert(/re/)      #=> nil
01525  */
01526 static VALUE
01527 rb_str_s_try_convert(VALUE dummy, VALUE str)
01528 {
01529     return rb_check_string_type(str);
01530 }
01531 
01532 static char*
01533 str_nth_len(const char *p, const char *e, long *nthp, rb_encoding *enc)
01534 {
01535     long nth = *nthp;
01536     if (rb_enc_mbmaxlen(enc) == 1) {
01537         p += nth;
01538     }
01539     else if (rb_enc_mbmaxlen(enc) == rb_enc_mbminlen(enc)) {
01540         p += nth * rb_enc_mbmaxlen(enc);
01541     }
01542     else if (rb_enc_asciicompat(enc)) {
01543         const char *p2, *e2;
01544         int n;
01545 
01546         while (p < e && 0 < nth) {
01547             e2 = p + nth;
01548             if (e < e2) {
01549                 *nthp = nth;
01550                 return (char *)e;
01551             }
01552             if (ISASCII(*p)) {
01553                 p2 = search_nonascii(p, e2);
01554                 if (!p2) {
01555                     nth -= e2 - p;
01556                     *nthp = nth;
01557                     return (char *)e2;
01558                 }
01559                 nth -= p2 - p;
01560                 p = p2;
01561             }
01562             n = rb_enc_mbclen(p, e, enc);
01563             p += n;
01564             nth--;
01565         }
01566         *nthp = nth;
01567         if (nth != 0) {
01568             return (char *)e;
01569         }
01570         return (char *)p;
01571     }
01572     else {
01573         while (p < e && nth--) {
01574             p += rb_enc_mbclen(p, e, enc);
01575         }
01576     }
01577     if (p > e) p = e;
01578     *nthp = nth;
01579     return (char*)p;
01580 }
01581 
01582 char*
01583 rb_enc_nth(const char *p, const char *e, long nth, rb_encoding *enc)
01584 {
01585     return str_nth_len(p, e, &nth, enc);
01586 }
01587 
01588 static char*
01589 str_nth(const char *p, const char *e, long nth, rb_encoding *enc, int singlebyte)
01590 {
01591     if (singlebyte)
01592         p += nth;
01593     else {
01594         p = str_nth_len(p, e, &nth, enc);
01595     }
01596     if (!p) return 0;
01597     if (p > e) p = e;
01598     return (char *)p;
01599 }
01600 
01601 /* char offset to byte offset */
01602 static long
01603 str_offset(const char *p, const char *e, long nth, rb_encoding *enc, int singlebyte)
01604 {
01605     const char *pp = str_nth(p, e, nth, enc, singlebyte);
01606     if (!pp) return e - p;
01607     return pp - p;
01608 }
01609 
01610 long
01611 rb_str_offset(VALUE str, long pos)
01612 {
01613     return str_offset(RSTRING_PTR(str), RSTRING_END(str), pos,
01614                       STR_ENC_GET(str), single_byte_optimizable(str));
01615 }
01616 
01617 #ifdef NONASCII_MASK
01618 static char *
01619 str_utf8_nth(const char *p, const char *e, long *nthp)
01620 {
01621     long nth = *nthp;
01622     if ((int)SIZEOF_VALUE * 2 < e - p && (int)SIZEOF_VALUE * 2 < nth) {
01623         const VALUE *s, *t;
01624         const VALUE lowbits = sizeof(VALUE) - 1;
01625         s = (const VALUE*)(~lowbits & ((VALUE)p + lowbits));
01626         t = (const VALUE*)(~lowbits & (VALUE)e);
01627         while (p < (const char *)s) {
01628             if (is_utf8_lead_byte(*p)) nth--;
01629             p++;
01630         }
01631         do {
01632             nth -= count_utf8_lead_bytes_with_word(s);
01633             s++;
01634         } while (s < t && (int)sizeof(VALUE) <= nth);
01635         p = (char *)s;
01636     }
01637     while (p < e) {
01638         if (is_utf8_lead_byte(*p)) {
01639             if (nth == 0) break;
01640             nth--;
01641         }
01642         p++;
01643     }
01644     *nthp = nth;
01645     return (char *)p;
01646 }
01647 
01648 static long
01649 str_utf8_offset(const char *p, const char *e, long nth)
01650 {
01651     const char *pp = str_utf8_nth(p, e, &nth);
01652     return pp - p;
01653 }
01654 #endif
01655 
01656 /* byte offset to char offset */
01657 long
01658 rb_str_sublen(VALUE str, long pos)
01659 {
01660     if (single_byte_optimizable(str) || pos < 0)
01661         return pos;
01662     else {
01663         char *p = RSTRING_PTR(str);
01664         return enc_strlen(p, p + pos, STR_ENC_GET(str), ENC_CODERANGE(str));
01665     }
01666 }
01667 
01668 VALUE
01669 rb_str_subseq(VALUE str, long beg, long len)
01670 {
01671     VALUE str2;
01672 
01673     if (RSTRING_LEN(str) == beg + len &&
01674         RSTRING_EMBED_LEN_MAX < len) {
01675         str2 = rb_str_new_shared(rb_str_new_frozen(str));
01676         rb_str_drop_bytes(str2, beg);
01677     }
01678     else {
01679         str2 = rb_str_new5(str, RSTRING_PTR(str)+beg, len);
01680         RB_GC_GUARD(str);
01681     }
01682 
01683     rb_enc_cr_str_copy_for_substr(str2, str);
01684     OBJ_INFECT(str2, str);
01685 
01686     return str2;
01687 }
01688 
01689 static char *
01690 rb_str_subpos(VALUE str, long beg, long *lenp)
01691 {
01692     long len = *lenp;
01693     long slen = -1L;
01694     long blen = RSTRING_LEN(str);
01695     rb_encoding *enc = STR_ENC_GET(str);
01696     char *p, *s = RSTRING_PTR(str), *e = s + blen;
01697 
01698     if (len < 0) return 0;
01699     if (!blen) {
01700         len = 0;
01701     }
01702     if (single_byte_optimizable(str)) {
01703         if (beg > blen) return 0;
01704         if (beg < 0) {
01705             beg += blen;
01706             if (beg < 0) return 0;
01707         }
01708         if (beg + len > blen)
01709             len = blen - beg;
01710         if (len < 0) return 0;
01711         p = s + beg;
01712         goto end;
01713     }
01714     if (beg < 0) {
01715         if (len > -beg) len = -beg;
01716         if (-beg * rb_enc_mbmaxlen(enc) < RSTRING_LEN(str) / 8) {
01717             beg = -beg;
01718             while (beg-- > len && (e = rb_enc_prev_char(s, e, e, enc)) != 0);
01719             p = e;
01720             if (!p) return 0;
01721             while (len-- > 0 && (p = rb_enc_prev_char(s, p, e, enc)) != 0);
01722             if (!p) return 0;
01723             len = e - p;
01724             goto end;
01725         }
01726         else {
01727             slen = str_strlen(str, enc);
01728             beg += slen;
01729             if (beg < 0) return 0;
01730             p = s + beg;
01731             if (len == 0) goto end;
01732         }
01733     }
01734     else if (beg > 0 && beg > RSTRING_LEN(str)) {
01735         return 0;
01736     }
01737     if (len == 0) {
01738         if (beg > str_strlen(str, enc)) return 0;
01739         p = s + beg;
01740     }
01741 #ifdef NONASCII_MASK
01742     else if (ENC_CODERANGE(str) == ENC_CODERANGE_VALID &&
01743         enc == rb_utf8_encoding()) {
01744         p = str_utf8_nth(s, e, &beg);
01745         if (beg > 0) return 0;
01746         len = str_utf8_offset(p, e, len);
01747     }
01748 #endif
01749     else if (rb_enc_mbmaxlen(enc) == rb_enc_mbminlen(enc)) {
01750         int char_sz = rb_enc_mbmaxlen(enc);
01751 
01752         p = s + beg * char_sz;
01753         if (p > e) {
01754             return 0;
01755         }
01756         else if (len * char_sz > e - p)
01757             len = e - p;
01758         else
01759             len *= char_sz;
01760     }
01761     else if ((p = str_nth_len(s, e, &beg, enc)) == e) {
01762         if (beg > 0) return 0;
01763         len = 0;
01764     }
01765     else {
01766         len = str_offset(p, e, len, enc, 0);
01767     }
01768   end:
01769     *lenp = len;
01770     RB_GC_GUARD(str);
01771     return p;
01772 }
01773 
01774 VALUE
01775 rb_str_substr(VALUE str, long beg, long len)
01776 {
01777     VALUE str2;
01778     char *p = rb_str_subpos(str, beg, &len);
01779 
01780     if (!p) return Qnil;
01781     if (len > RSTRING_EMBED_LEN_MAX && p + len == RSTRING_END(str)) {
01782         str2 = rb_str_new4(str);
01783         str2 = str_new3(rb_obj_class(str2), str2);
01784         RSTRING(str2)->as.heap.ptr += RSTRING(str2)->as.heap.len - len;
01785         RSTRING(str2)->as.heap.len = len;
01786     }
01787     else {
01788         str2 = rb_str_new5(str, p, len);
01789         OBJ_INFECT(str2, str);
01790         RB_GC_GUARD(str);
01791     }
01792     rb_enc_cr_str_copy_for_substr(str2, str);
01793 
01794     return str2;
01795 }
01796 
01797 VALUE
01798 rb_str_freeze(VALUE str)
01799 {
01800     if (STR_ASSOC_P(str)) {
01801         VALUE ary = RSTRING(str)->as.heap.aux.shared;
01802         OBJ_FREEZE(ary);
01803     }
01804     return rb_obj_freeze(str);
01805 }
01806 
01807 RUBY_ALIAS_FUNCTION(rb_str_dup_frozen(VALUE str), rb_str_new_frozen, (str))
01808 #define rb_str_dup_frozen rb_str_new_frozen
01809 
01810 VALUE
01811 rb_str_locktmp(VALUE str)
01812 {
01813     if (FL_TEST(str, STR_TMPLOCK)) {
01814         rb_raise(rb_eRuntimeError, "temporal locking already locked string");
01815     }
01816     FL_SET(str, STR_TMPLOCK);
01817     return str;
01818 }
01819 
01820 VALUE
01821 rb_str_unlocktmp(VALUE str)
01822 {
01823     if (!FL_TEST(str, STR_TMPLOCK)) {
01824         rb_raise(rb_eRuntimeError, "temporal unlocking already unlocked string");
01825     }
01826     FL_UNSET(str, STR_TMPLOCK);
01827     return str;
01828 }
01829 
01830 VALUE
01831 rb_str_locktmp_ensure(VALUE str, VALUE (*func)(VALUE), VALUE arg)
01832 {
01833     rb_str_locktmp(str);
01834     return rb_ensure(func, arg, rb_str_unlocktmp, str);
01835 }
01836 
01837 void
01838 rb_str_set_len(VALUE str, long len)
01839 {
01840     long capa;
01841 
01842     str_modifiable(str);
01843     if (STR_SHARED_P(str)) {
01844         rb_raise(rb_eRuntimeError, "can't set length of shared string");
01845     }
01846     if (len > (capa = (long)rb_str_capacity(str))) {
01847         rb_bug("probable buffer overflow: %ld for %ld", len, capa);
01848     }
01849     STR_SET_LEN(str, len);
01850     RSTRING_PTR(str)[len] = '\0';
01851 }
01852 
01853 VALUE
01854 rb_str_resize(VALUE str, long len)
01855 {
01856     long slen;
01857     int independent;
01858 
01859     if (len < 0) {
01860         rb_raise(rb_eArgError, "negative string size (or size too big)");
01861     }
01862 
01863     independent = str_independent(str);
01864     ENC_CODERANGE_CLEAR(str);
01865     slen = RSTRING_LEN(str);
01866     {
01867         long capa;
01868         if (STR_EMBED_P(str)) {
01869             if (len == slen) return str;
01870             if (len + 1 <= RSTRING_EMBED_LEN_MAX + 1) {
01871                 STR_SET_EMBED_LEN(str, len);
01872                 RSTRING(str)->as.ary[len] = '\0';
01873                 return str;
01874             }
01875             str_make_independent_expand(str, len - slen);
01876             STR_SET_NOEMBED(str);
01877         }
01878         else if (len <= RSTRING_EMBED_LEN_MAX) {
01879             char *ptr = RSTRING(str)->as.heap.ptr;
01880             STR_SET_EMBED(str);
01881             if (slen > len) slen = len;
01882             if (slen > 0) MEMCPY(RSTRING(str)->as.ary, ptr, char, slen);
01883             RSTRING(str)->as.ary[len] = '\0';
01884             STR_SET_EMBED_LEN(str, len);
01885             if (independent) xfree(ptr);
01886             return str;
01887         }
01888         else if (!independent) {
01889             if (len == slen) return str;
01890             str_make_independent_expand(str, len - slen);
01891         }
01892         else if ((capa = RSTRING(str)->as.heap.aux.capa) < len ||
01893                  (capa - len) > (len < 1024 ? len : 1024)) {
01894             REALLOC_N(RSTRING(str)->as.heap.ptr, char, len+1);
01895             RSTRING(str)->as.heap.aux.capa = len;
01896         }
01897         else if (len == slen) return str;
01898         RSTRING(str)->as.heap.len = len;
01899         RSTRING(str)->as.heap.ptr[len] = '\0';  /* sentinel */
01900     }
01901     return str;
01902 }
01903 
01904 static VALUE
01905 str_buf_cat(VALUE str, const char *ptr, long len)
01906 {
01907     long capa, total, off = -1;
01908 
01909     if (ptr >= RSTRING_PTR(str) && ptr <= RSTRING_END(str)) {
01910         off = ptr - RSTRING_PTR(str);
01911     }
01912     rb_str_modify(str);
01913     if (len == 0) return 0;
01914     if (STR_ASSOC_P(str)) {
01915         FL_UNSET(str, STR_ASSOC);
01916         capa = RSTRING(str)->as.heap.aux.capa = RSTRING_LEN(str);
01917     }
01918     else if (STR_EMBED_P(str)) {
01919         capa = RSTRING_EMBED_LEN_MAX;
01920     }
01921     else {
01922         capa = RSTRING(str)->as.heap.aux.capa;
01923     }
01924     if (RSTRING_LEN(str) >= LONG_MAX - len) {
01925         rb_raise(rb_eArgError, "string sizes too big");
01926     }
01927     total = RSTRING_LEN(str)+len;
01928     if (capa <= total) {
01929         while (total > capa) {
01930             if (capa + 1 >= LONG_MAX / 2) {
01931                 capa = (total + 4095) / 4096 * 4096;
01932                 break;
01933             }
01934             capa = (capa + 1) * 2;
01935         }
01936         RESIZE_CAPA(str, capa);
01937     }
01938     if (off != -1) {
01939         ptr = RSTRING_PTR(str) + off;
01940     }
01941     memcpy(RSTRING_PTR(str) + RSTRING_LEN(str), ptr, len);
01942     STR_SET_LEN(str, total);
01943     RSTRING_PTR(str)[total] = '\0'; /* sentinel */
01944 
01945     return str;
01946 }
01947 
01948 #define str_buf_cat2(str, ptr) str_buf_cat((str), (ptr), strlen(ptr))
01949 
01950 VALUE
01951 rb_str_buf_cat(VALUE str, const char *ptr, long len)
01952 {
01953     if (len == 0) return str;
01954     if (len < 0) {
01955         rb_raise(rb_eArgError, "negative string size (or size too big)");
01956     }
01957     return str_buf_cat(str, ptr, len);
01958 }
01959 
01960 VALUE
01961 rb_str_buf_cat2(VALUE str, const char *ptr)
01962 {
01963     return rb_str_buf_cat(str, ptr, strlen(ptr));
01964 }
01965 
01966 VALUE
01967 rb_str_cat(VALUE str, const char *ptr, long len)
01968 {
01969     if (len < 0) {
01970         rb_raise(rb_eArgError, "negative string size (or size too big)");
01971     }
01972     if (STR_ASSOC_P(str)) {
01973         char *p;
01974         rb_str_modify_expand(str, len);
01975         p = RSTRING(str)->as.heap.ptr;
01976         memcpy(p + RSTRING(str)->as.heap.len, ptr, len);
01977         len = RSTRING(str)->as.heap.len += len;
01978         p[len] = '\0'; /* sentinel */
01979         return str;
01980     }
01981 
01982     return rb_str_buf_cat(str, ptr, len);
01983 }
01984 
01985 VALUE
01986 rb_str_cat2(VALUE str, const char *ptr)
01987 {
01988     return rb_str_cat(str, ptr, strlen(ptr));
01989 }
01990 
01991 static VALUE
01992 rb_enc_cr_str_buf_cat(VALUE str, const char *ptr, long len,
01993     int ptr_encindex, int ptr_cr, int *ptr_cr_ret)
01994 {
01995     int str_encindex = ENCODING_GET(str);
01996     int res_encindex;
01997     int str_cr, res_cr;
01998 
01999     str_cr = ENC_CODERANGE(str);
02000 
02001     if (str_encindex == ptr_encindex) {
02002         if (str_cr == ENC_CODERANGE_UNKNOWN)
02003             ptr_cr = ENC_CODERANGE_UNKNOWN;
02004         else if (ptr_cr == ENC_CODERANGE_UNKNOWN) {
02005             ptr_cr = coderange_scan(ptr, len, rb_enc_from_index(ptr_encindex));
02006         }
02007     }
02008     else {
02009         rb_encoding *str_enc = rb_enc_from_index(str_encindex);
02010         rb_encoding *ptr_enc = rb_enc_from_index(ptr_encindex);
02011         if (!rb_enc_asciicompat(str_enc) || !rb_enc_asciicompat(ptr_enc)) {
02012             if (len == 0)
02013                 return str;
02014             if (RSTRING_LEN(str) == 0) {
02015                 rb_str_buf_cat(str, ptr, len);
02016                 ENCODING_CODERANGE_SET(str, ptr_encindex, ptr_cr);
02017                 return str;
02018             }
02019             goto incompatible;
02020         }
02021         if (ptr_cr == ENC_CODERANGE_UNKNOWN) {
02022             ptr_cr = coderange_scan(ptr, len, ptr_enc);
02023         }
02024         if (str_cr == ENC_CODERANGE_UNKNOWN) {
02025             if (ENCODING_IS_ASCII8BIT(str) || ptr_cr != ENC_CODERANGE_7BIT) {
02026                 str_cr = rb_enc_str_coderange(str);
02027             }
02028         }
02029     }
02030     if (ptr_cr_ret)
02031         *ptr_cr_ret = ptr_cr;
02032 
02033     if (str_encindex != ptr_encindex &&
02034         str_cr != ENC_CODERANGE_7BIT &&
02035         ptr_cr != ENC_CODERANGE_7BIT) {
02036       incompatible:
02037         rb_raise(rb_eEncCompatError, "incompatible character encodings: %s and %s",
02038             rb_enc_name(rb_enc_from_index(str_encindex)),
02039             rb_enc_name(rb_enc_from_index(ptr_encindex)));
02040     }
02041 
02042     if (str_cr == ENC_CODERANGE_UNKNOWN) {
02043         res_encindex = str_encindex;
02044         res_cr = ENC_CODERANGE_UNKNOWN;
02045     }
02046     else if (str_cr == ENC_CODERANGE_7BIT) {
02047         if (ptr_cr == ENC_CODERANGE_7BIT) {
02048             res_encindex = str_encindex;
02049             res_cr = ENC_CODERANGE_7BIT;
02050         }
02051         else {
02052             res_encindex = ptr_encindex;
02053             res_cr = ptr_cr;
02054         }
02055     }
02056     else if (str_cr == ENC_CODERANGE_VALID) {
02057         res_encindex = str_encindex;
02058         if (ptr_cr == ENC_CODERANGE_7BIT || ptr_cr == ENC_CODERANGE_VALID)
02059             res_cr = str_cr;
02060         else
02061             res_cr = ptr_cr;
02062     }
02063     else { /* str_cr == ENC_CODERANGE_BROKEN */
02064         res_encindex = str_encindex;
02065         res_cr = str_cr;
02066         if (0 < len) res_cr = ENC_CODERANGE_UNKNOWN;
02067     }
02068 
02069     if (len < 0) {
02070         rb_raise(rb_eArgError, "negative string size (or size too big)");
02071     }
02072     str_buf_cat(str, ptr, len);
02073     ENCODING_CODERANGE_SET(str, res_encindex, res_cr);
02074     return str;
02075 }
02076 
02077 VALUE
02078 rb_enc_str_buf_cat(VALUE str, const char *ptr, long len, rb_encoding *ptr_enc)
02079 {
02080     return rb_enc_cr_str_buf_cat(str, ptr, len,
02081         rb_enc_to_index(ptr_enc), ENC_CODERANGE_UNKNOWN, NULL);
02082 }
02083 
02084 VALUE
02085 rb_str_buf_cat_ascii(VALUE str, const char *ptr)
02086 {
02087     /* ptr must reference NUL terminated ASCII string. */
02088     int encindex = ENCODING_GET(str);
02089     rb_encoding *enc = rb_enc_from_index(encindex);
02090     if (rb_enc_asciicompat(enc)) {
02091         return rb_enc_cr_str_buf_cat(str, ptr, strlen(ptr),
02092             encindex, ENC_CODERANGE_7BIT, 0);
02093     }
02094     else {
02095         char *buf = ALLOCA_N(char, rb_enc_mbmaxlen(enc));
02096         while (*ptr) {
02097             unsigned int c = (unsigned char)*ptr;
02098             int len = rb_enc_codelen(c, enc);
02099             rb_enc_mbcput(c, buf, enc);
02100             rb_enc_cr_str_buf_cat(str, buf, len,
02101                 encindex, ENC_CODERANGE_VALID, 0);
02102             ptr++;
02103         }
02104         return str;
02105     }
02106 }
02107 
02108 VALUE
02109 rb_str_buf_append(VALUE str, VALUE str2)
02110 {
02111     int str2_cr;
02112 
02113     str2_cr = ENC_CODERANGE(str2);
02114 
02115     rb_enc_cr_str_buf_cat(str, RSTRING_PTR(str2), RSTRING_LEN(str2),
02116         ENCODING_GET(str2), str2_cr, &str2_cr);
02117 
02118     OBJ_INFECT(str, str2);
02119     ENC_CODERANGE_SET(str2, str2_cr);
02120 
02121     return str;
02122 }
02123 
02124 VALUE
02125 rb_str_append(VALUE str, VALUE str2)
02126 {
02127     rb_encoding *enc;
02128     int cr, cr2;
02129     long len2;
02130 
02131     StringValue(str2);
02132     if ((len2 = RSTRING_LEN(str2)) > 0 && STR_ASSOC_P(str)) {
02133         long len = RSTRING_LEN(str) + len2;
02134         enc = rb_enc_check(str, str2);
02135         cr = ENC_CODERANGE(str);
02136         if ((cr2 = ENC_CODERANGE(str2)) > cr) cr = cr2;
02137         rb_str_modify_expand(str, len2);
02138         memcpy(RSTRING(str)->as.heap.ptr + RSTRING(str)->as.heap.len,
02139                RSTRING_PTR(str2), len2+1);
02140         RSTRING(str)->as.heap.len = len;
02141         rb_enc_associate(str, enc);
02142         ENC_CODERANGE_SET(str, cr);
02143         OBJ_INFECT(str, str2);
02144         return str;
02145     }
02146     return rb_str_buf_append(str, str2);
02147 }
02148 
02149 /*
02150  *  call-seq:
02151  *     str << integer       -> str
02152  *     str.concat(integer)  -> str
02153  *     str << obj           -> str
02154  *     str.concat(obj)      -> str
02155  *
02156  *  Append---Concatenates the given object to <i>str</i>. If the object is a
02157  *  <code>Integer</code>, it is considered as a codepoint, and is converted
02158  *  to a character before concatenation.
02159  *
02160  *     a = "hello "
02161  *     a << "world"   #=> "hello world"
02162  *     a.concat(33)   #=> "hello world!"
02163  */
02164 
02165 VALUE
02166 rb_str_concat(VALUE str1, VALUE str2)
02167 {
02168     unsigned int code;
02169     rb_encoding *enc = STR_ENC_GET(str1);
02170 
02171     if (FIXNUM_P(str2) || RB_TYPE_P(str2, T_BIGNUM)) {
02172         if (rb_num_to_uint(str2, &code) == 0) {
02173         }
02174         else if (FIXNUM_P(str2)) {
02175             rb_raise(rb_eRangeError, "%ld out of char range", FIX2LONG(str2));
02176         }
02177         else {
02178             rb_raise(rb_eRangeError, "bignum out of char range");
02179         }
02180     }
02181     else {
02182         return rb_str_append(str1, str2);
02183     }
02184 
02185     if (enc == rb_usascii_encoding()) {
02186         /* US-ASCII automatically extended to ASCII-8BIT */
02187         char buf[1];
02188         buf[0] = (char)code;
02189         if (code > 0xFF) {
02190             rb_raise(rb_eRangeError, "%u out of char range", code);
02191         }
02192         rb_str_cat(str1, buf, 1);
02193         if (code > 127) {
02194             rb_enc_associate(str1, rb_ascii8bit_encoding());
02195             ENC_CODERANGE_SET(str1, ENC_CODERANGE_VALID);
02196         }
02197     }
02198     else {
02199         long pos = RSTRING_LEN(str1);
02200         int cr = ENC_CODERANGE(str1);
02201         int len;
02202         char *buf;
02203 
02204         switch (len = rb_enc_codelen(code, enc)) {
02205           case ONIGERR_INVALID_CODE_POINT_VALUE:
02206             rb_raise(rb_eRangeError, "invalid codepoint 0x%X in %s", code, rb_enc_name(enc));
02207             break;
02208           case ONIGERR_TOO_BIG_WIDE_CHAR_VALUE:
02209           case 0:
02210             rb_raise(rb_eRangeError, "%u out of char range", code);
02211             break;
02212         }
02213         buf = ALLOCA_N(char, len + 1);
02214         rb_enc_mbcput(code, buf, enc);
02215         if (rb_enc_precise_mbclen(buf, buf + len + 1, enc) != len) {
02216             rb_raise(rb_eRangeError, "invalid codepoint 0x%X in %s", code, rb_enc_name(enc));
02217         }
02218         rb_str_resize(str1, pos+len);
02219         memcpy(RSTRING_PTR(str1) + pos, buf, len);
02220         if (cr == ENC_CODERANGE_7BIT && code > 127)
02221             cr = ENC_CODERANGE_VALID;
02222         ENC_CODERANGE_SET(str1, cr);
02223     }
02224     return str1;
02225 }
02226 
02227 /*
02228  *  call-seq:
02229  *     str.prepend(other_str)  -> str
02230  *
02231  *  Prepend---Prepend the given string to <i>str</i>.
02232  *
02233  *     a = "world"
02234  *     a.prepend("hello ") #=> "hello world"
02235  *     a                   #=> "hello world"
02236  */
02237 
02238 static VALUE
02239 rb_str_prepend(VALUE str, VALUE str2)
02240 {
02241     StringValue(str2);
02242     StringValue(str);
02243     rb_str_update(str, 0L, 0L, str2);
02244     return str;
02245 }
02246 
02247 st_index_t
02248 rb_str_hash(VALUE str)
02249 {
02250     int e = ENCODING_GET(str);
02251     if (e && rb_enc_str_coderange(str) == ENC_CODERANGE_7BIT) {
02252         e = 0;
02253     }
02254     return rb_memhash((const void *)RSTRING_PTR(str), RSTRING_LEN(str)) ^ e;
02255 }
02256 
02257 int
02258 rb_str_hash_cmp(VALUE str1, VALUE str2)
02259 {
02260     long len;
02261 
02262     if (!rb_str_comparable(str1, str2)) return 1;
02263     if (RSTRING_LEN(str1) == (len = RSTRING_LEN(str2)) &&
02264         memcmp(RSTRING_PTR(str1), RSTRING_PTR(str2), len) == 0) {
02265         return 0;
02266     }
02267     return 1;
02268 }
02269 
02270 /*
02271  * call-seq:
02272  *    str.hash   -> fixnum
02273  *
02274  * Return a hash based on the string's length and content.
02275  */
02276 
02277 static VALUE
02278 rb_str_hash_m(VALUE str)
02279 {
02280     st_index_t hval = rb_str_hash(str);
02281     return INT2FIX(hval);
02282 }
02283 
02284 #define lesser(a,b) (((a)>(b))?(b):(a))
02285 
02286 int
02287 rb_str_comparable(VALUE str1, VALUE str2)
02288 {
02289     int idx1, idx2;
02290     int rc1, rc2;
02291 
02292     if (RSTRING_LEN(str1) == 0) return TRUE;
02293     if (RSTRING_LEN(str2) == 0) return TRUE;
02294     idx1 = ENCODING_GET(str1);
02295     idx2 = ENCODING_GET(str2);
02296     if (idx1 == idx2) return TRUE;
02297     rc1 = rb_enc_str_coderange(str1);
02298     rc2 = rb_enc_str_coderange(str2);
02299     if (rc1 == ENC_CODERANGE_7BIT) {
02300         if (rc2 == ENC_CODERANGE_7BIT) return TRUE;
02301         if (rb_enc_asciicompat(rb_enc_from_index(idx2)))
02302             return TRUE;
02303     }
02304     if (rc2 == ENC_CODERANGE_7BIT) {
02305         if (rb_enc_asciicompat(rb_enc_from_index(idx1)))
02306             return TRUE;
02307     }
02308     return FALSE;
02309 }
02310 
02311 int
02312 rb_str_cmp(VALUE str1, VALUE str2)
02313 {
02314     long len1, len2;
02315     const char *ptr1, *ptr2;
02316     int retval;
02317 
02318     if (str1 == str2) return 0;
02319     RSTRING_GETMEM(str1, ptr1, len1);
02320     RSTRING_GETMEM(str2, ptr2, len2);
02321     if (ptr1 == ptr2 || (retval = memcmp(ptr1, ptr2, lesser(len1, len2))) == 0) {
02322         if (len1 == len2) {
02323             if (!rb_str_comparable(str1, str2)) {
02324                 if (ENCODING_GET(str1) > ENCODING_GET(str2))
02325                     return 1;
02326                 return -1;
02327             }
02328             return 0;
02329         }
02330         if (len1 > len2) return 1;
02331         return -1;
02332     }
02333     if (retval > 0) return 1;
02334     return -1;
02335 }
02336 
02337 /* expect tail call optimization */
02338 static VALUE
02339 str_eql(const VALUE str1, const VALUE str2)
02340 {
02341     const long len = RSTRING_LEN(str1);
02342     const char *ptr1, *ptr2;
02343 
02344     if (len != RSTRING_LEN(str2)) return Qfalse;
02345     if (!rb_str_comparable(str1, str2)) return Qfalse;
02346     if ((ptr1 = RSTRING_PTR(str1)) == (ptr2 = RSTRING_PTR(str2)))
02347         return Qtrue;
02348     if (memcmp(ptr1, ptr2, len) == 0)
02349         return Qtrue;
02350     return Qfalse;
02351 }
02352 
02353 /*
02354  *  call-seq:
02355  *     str == obj   -> true or false
02356  *
02357  *  Equality---If <i>obj</i> is not a <code>String</code>, returns
02358  *  <code>false</code>. Otherwise, returns <code>true</code> if <i>str</i>
02359  *  <code><=></code> <i>obj</i> returns zero.
02360  */
02361 
02362 VALUE
02363 rb_str_equal(VALUE str1, VALUE str2)
02364 {
02365     if (str1 == str2) return Qtrue;
02366     if (!RB_TYPE_P(str2, T_STRING)) {
02367         if (!rb_respond_to(str2, rb_intern("to_str"))) {
02368             return Qfalse;
02369         }
02370         return rb_equal(str2, str1);
02371     }
02372     return str_eql(str1, str2);
02373 }
02374 
02375 /*
02376  * call-seq:
02377  *   str.eql?(other)   -> true or false
02378  *
02379  * Two strings are equal if they have the same length and content.
02380  */
02381 
02382 static VALUE
02383 rb_str_eql(VALUE str1, VALUE str2)
02384 {
02385     if (str1 == str2) return Qtrue;
02386     if (!RB_TYPE_P(str2, T_STRING)) return Qfalse;
02387     return str_eql(str1, str2);
02388 }
02389 
02390 /*
02391  *  call-seq:
02392  *     string <=> other_string   -> -1, 0, +1 or nil
02393  *
02394  *
02395  *  Comparison---Returns -1, 0, +1 or nil depending on whether +string+ is less
02396  *  than, equal to, or greater than +other_string+.
02397  *
02398  *  +nil+ is returned if the two values are incomparable.
02399  *
02400  *  If the strings are of different lengths, and the strings are equal when
02401  *  compared up to the shortest length, then the longer string is considered
02402  *  greater than the shorter one.
02403  *
02404  *  <code><=></code> is the basis for the methods <code><</code>,
02405  *  <code><=</code>, <code>></code>, <code>>=</code>, and
02406  *  <code>between?</code>, included from module Comparable. The method
02407  *  String#== does not use Comparable#==.
02408  *
02409  *     "abcdef" <=> "abcde"     #=> 1
02410  *     "abcdef" <=> "abcdef"    #=> 0
02411  *     "abcdef" <=> "abcdefg"   #=> -1
02412  *     "abcdef" <=> "ABCDEF"    #=> 1
02413  */
02414 
02415 static VALUE
02416 rb_str_cmp_m(VALUE str1, VALUE str2)
02417 {
02418     int result;
02419 
02420     if (!RB_TYPE_P(str2, T_STRING)) {
02421         VALUE tmp = rb_check_funcall(str2, rb_intern("to_str"), 0, 0);
02422         if (RB_TYPE_P(tmp, T_STRING)) {
02423             result = rb_str_cmp(str1, tmp);
02424         }
02425         else {
02426             return rb_invcmp(str1, str2);
02427         }
02428     }
02429     else {
02430         result = rb_str_cmp(str1, str2);
02431     }
02432     return INT2FIX(result);
02433 }
02434 
02435 /*
02436  *  call-seq:
02437  *     str.casecmp(other_str)   -> -1, 0, +1 or nil
02438  *
02439  *  Case-insensitive version of <code>String#<=></code>.
02440  *
02441  *     "abcdef".casecmp("abcde")     #=> 1
02442  *     "aBcDeF".casecmp("abcdef")    #=> 0
02443  *     "abcdef".casecmp("abcdefg")   #=> -1
02444  *     "abcdef".casecmp("ABCDEF")    #=> 0
02445  */
02446 
02447 static VALUE
02448 rb_str_casecmp(VALUE str1, VALUE str2)
02449 {
02450     long len;
02451     rb_encoding *enc;
02452     char *p1, *p1end, *p2, *p2end;
02453 
02454     StringValue(str2);
02455     enc = rb_enc_compatible(str1, str2);
02456     if (!enc) {
02457         return Qnil;
02458     }
02459 
02460     p1 = RSTRING_PTR(str1); p1end = RSTRING_END(str1);
02461     p2 = RSTRING_PTR(str2); p2end = RSTRING_END(str2);
02462     if (single_byte_optimizable(str1) && single_byte_optimizable(str2)) {
02463         while (p1 < p1end && p2 < p2end) {
02464             if (*p1 != *p2) {
02465                 unsigned int c1 = TOUPPER(*p1 & 0xff);
02466                 unsigned int c2 = TOUPPER(*p2 & 0xff);
02467                 if (c1 != c2)
02468                     return INT2FIX(c1 < c2 ? -1 : 1);
02469             }
02470             p1++;
02471             p2++;
02472         }
02473     }
02474     else {
02475         while (p1 < p1end && p2 < p2end) {
02476             int l1, c1 = rb_enc_ascget(p1, p1end, &l1, enc);
02477             int l2, c2 = rb_enc_ascget(p2, p2end, &l2, enc);
02478 
02479             if (0 <= c1 && 0 <= c2) {
02480                 c1 = TOUPPER(c1);
02481                 c2 = TOUPPER(c2);
02482                 if (c1 != c2)
02483                     return INT2FIX(c1 < c2 ? -1 : 1);
02484             }
02485             else {
02486                 int r;
02487                 l1 = rb_enc_mbclen(p1, p1end, enc);
02488                 l2 = rb_enc_mbclen(p2, p2end, enc);
02489                 len = l1 < l2 ? l1 : l2;
02490                 r = memcmp(p1, p2, len);
02491                 if (r != 0)
02492                     return INT2FIX(r < 0 ? -1 : 1);
02493                 if (l1 != l2)
02494                     return INT2FIX(l1 < l2 ? -1 : 1);
02495             }
02496             p1 += l1;
02497             p2 += l2;
02498         }
02499     }
02500     if (RSTRING_LEN(str1) == RSTRING_LEN(str2)) return INT2FIX(0);
02501     if (RSTRING_LEN(str1) > RSTRING_LEN(str2)) return INT2FIX(1);
02502     return INT2FIX(-1);
02503 }
02504 
02505 static long
02506 rb_str_index(VALUE str, VALUE sub, long offset)
02507 {
02508     long pos;
02509     char *s, *sptr, *e;
02510     long len, slen;
02511     rb_encoding *enc;
02512 
02513     enc = rb_enc_check(str, sub);
02514     if (is_broken_string(sub)) {
02515         return -1;
02516     }
02517     len = str_strlen(str, enc);
02518     slen = str_strlen(sub, enc);
02519     if (offset < 0) {
02520         offset += len;
02521         if (offset < 0) return -1;
02522     }
02523     if (len - offset < slen) return -1;
02524     s = RSTRING_PTR(str);
02525     e = s + RSTRING_LEN(str);
02526     if (offset) {
02527         offset = str_offset(s, RSTRING_END(str), offset, enc, single_byte_optimizable(str));
02528         s += offset;
02529     }
02530     if (slen == 0) return offset;
02531     /* need proceed one character at a time */
02532     sptr = RSTRING_PTR(sub);
02533     slen = RSTRING_LEN(sub);
02534     len = RSTRING_LEN(str) - offset;
02535     for (;;) {
02536         char *t;
02537         pos = rb_memsearch(sptr, slen, s, len, enc);
02538         if (pos < 0) return pos;
02539         t = rb_enc_right_char_head(s, s+pos, e, enc);
02540         if (t == s + pos) break;
02541         if ((len -= t - s) <= 0) return -1;
02542         offset += t - s;
02543         s = t;
02544     }
02545     return pos + offset;
02546 }
02547 
02548 
02549 /*
02550  *  call-seq:
02551  *     str.index(substring [, offset])   -> fixnum or nil
02552  *     str.index(regexp [, offset])      -> fixnum or nil
02553  *
02554  *  Returns the index of the first occurrence of the given <i>substring</i> or
02555  *  pattern (<i>regexp</i>) in <i>str</i>. Returns <code>nil</code> if not
02556  *  found. If the second parameter is present, it specifies the position in the
02557  *  string to begin the search.
02558  *
02559  *     "hello".index('e')             #=> 1
02560  *     "hello".index('lo')            #=> 3
02561  *     "hello".index('a')             #=> nil
02562  *     "hello".index(?e)              #=> 1
02563  *     "hello".index(/[aeiou]/, -3)   #=> 4
02564  */
02565 
02566 static VALUE
02567 rb_str_index_m(int argc, VALUE *argv, VALUE str)
02568 {
02569     VALUE sub;
02570     VALUE initpos;
02571     long pos;
02572 
02573     if (rb_scan_args(argc, argv, "11", &sub, &initpos) == 2) {
02574         pos = NUM2LONG(initpos);
02575     }
02576     else {
02577         pos = 0;
02578     }
02579     if (pos < 0) {
02580         pos += str_strlen(str, STR_ENC_GET(str));
02581         if (pos < 0) {
02582             if (RB_TYPE_P(sub, T_REGEXP)) {
02583                 rb_backref_set(Qnil);
02584             }
02585             return Qnil;
02586         }
02587     }
02588 
02589     if (SPECIAL_CONST_P(sub)) goto generic;
02590     switch (BUILTIN_TYPE(sub)) {
02591       case T_REGEXP:
02592         if (pos > str_strlen(str, STR_ENC_GET(str)))
02593             return Qnil;
02594         pos = str_offset(RSTRING_PTR(str), RSTRING_END(str), pos,
02595                          rb_enc_check(str, sub), single_byte_optimizable(str));
02596 
02597         pos = rb_reg_search(sub, str, pos, 0);
02598         pos = rb_str_sublen(str, pos);
02599         break;
02600 
02601       generic:
02602       default: {
02603         VALUE tmp;
02604 
02605         tmp = rb_check_string_type(sub);
02606         if (NIL_P(tmp)) {
02607             rb_raise(rb_eTypeError, "type mismatch: %s given",
02608                      rb_obj_classname(sub));
02609         }
02610         sub = tmp;
02611       }
02612         /* fall through */
02613       case T_STRING:
02614         pos = rb_str_index(str, sub, pos);
02615         pos = rb_str_sublen(str, pos);
02616         break;
02617     }
02618 
02619     if (pos == -1) return Qnil;
02620     return LONG2NUM(pos);
02621 }
02622 
02623 static long
02624 rb_str_rindex(VALUE str, VALUE sub, long pos)
02625 {
02626     long len, slen;
02627     char *s, *sbeg, *e, *t;
02628     rb_encoding *enc;
02629     int singlebyte = single_byte_optimizable(str);
02630 
02631     enc = rb_enc_check(str, sub);
02632     if (is_broken_string(sub)) {
02633         return -1;
02634     }
02635     len = str_strlen(str, enc);
02636     slen = str_strlen(sub, enc);
02637     /* substring longer than string */
02638     if (len < slen) return -1;
02639     if (len - pos < slen) {
02640         pos = len - slen;
02641     }
02642     if (len == 0) {
02643         return pos;
02644     }
02645     sbeg = RSTRING_PTR(str);
02646     e = RSTRING_END(str);
02647     t = RSTRING_PTR(sub);
02648     slen = RSTRING_LEN(sub);
02649     s = str_nth(sbeg, e, pos, enc, singlebyte);
02650     while (s) {
02651         if (memcmp(s, t, slen) == 0) {
02652             return pos;
02653         }
02654         if (pos == 0) break;
02655         pos--;
02656         s = rb_enc_prev_char(sbeg, s, e, enc);
02657     }
02658     return -1;
02659 }
02660 
02661 
02662 /*
02663  *  call-seq:
02664  *     str.rindex(substring [, fixnum])   -> fixnum or nil
02665  *     str.rindex(regexp [, fixnum])   -> fixnum or nil
02666  *
02667  *  Returns the index of the last occurrence of the given <i>substring</i> or
02668  *  pattern (<i>regexp</i>) in <i>str</i>. Returns <code>nil</code> if not
02669  *  found. If the second parameter is present, it specifies the position in the
02670  *  string to end the search---characters beyond this point will not be
02671  *  considered.
02672  *
02673  *     "hello".rindex('e')             #=> 1
02674  *     "hello".rindex('l')             #=> 3
02675  *     "hello".rindex('a')             #=> nil
02676  *     "hello".rindex(?e)              #=> 1
02677  *     "hello".rindex(/[aeiou]/, -2)   #=> 1
02678  */
02679 
02680 static VALUE
02681 rb_str_rindex_m(int argc, VALUE *argv, VALUE str)
02682 {
02683     VALUE sub;
02684     VALUE vpos;
02685     rb_encoding *enc = STR_ENC_GET(str);
02686     long pos, len = str_strlen(str, enc);
02687 
02688     if (rb_scan_args(argc, argv, "11", &sub, &vpos) == 2) {
02689         pos = NUM2LONG(vpos);
02690         if (pos < 0) {
02691             pos += len;
02692             if (pos < 0) {
02693                 if (RB_TYPE_P(sub, T_REGEXP)) {
02694                     rb_backref_set(Qnil);
02695                 }
02696                 return Qnil;
02697             }
02698         }
02699         if (pos > len) pos = len;
02700     }
02701     else {
02702         pos = len;
02703     }
02704 
02705     if (SPECIAL_CONST_P(sub)) goto generic;
02706     switch (BUILTIN_TYPE(sub)) {
02707       case T_REGEXP:
02708         /* enc = rb_get_check(str, sub); */
02709         pos = str_offset(RSTRING_PTR(str), RSTRING_END(str), pos,
02710                          STR_ENC_GET(str), single_byte_optimizable(str));
02711 
02712         if (!RREGEXP(sub)->ptr || RREGEXP_SRC_LEN(sub)) {
02713             pos = rb_reg_search(sub, str, pos, 1);
02714             pos = rb_str_sublen(str, pos);
02715         }
02716         if (pos >= 0) return LONG2NUM(pos);
02717         break;
02718 
02719       generic:
02720       default: {
02721         VALUE tmp;
02722 
02723         tmp = rb_check_string_type(sub);
02724         if (NIL_P(tmp)) {
02725             rb_raise(rb_eTypeError, "type mismatch: %s given",
02726                      rb_obj_classname(sub));
02727         }
02728         sub = tmp;
02729       }
02730         /* fall through */
02731       case T_STRING:
02732         pos = rb_str_rindex(str, sub, pos);
02733         if (pos >= 0) return LONG2NUM(pos);
02734         break;
02735     }
02736     return Qnil;
02737 }
02738 
02739 /*
02740  *  call-seq:
02741  *     str =~ obj   -> fixnum or nil
02742  *
02743  *  Match---If <i>obj</i> is a <code>Regexp</code>, use it as a pattern to match
02744  *  against <i>str</i>,and returns the position the match starts, or
02745  *  <code>nil</code> if there is no match. Otherwise, invokes
02746  *  <i>obj.=~</i>, passing <i>str</i> as an argument. The default
02747  *  <code>=~</code> in <code>Object</code> returns <code>nil</code>.
02748  *
02749  *  Note: <code>str =~ regexp</code> is not the same as
02750  *  <code>regexp =~ str</code>. Strings captured from named capture groups
02751  *  are assigned to local variables only in the second case.
02752  *
02753  *     "cat o' 9 tails" =~ /\d/   #=> 7
02754  *     "cat o' 9 tails" =~ 9      #=> nil
02755  */
02756 
02757 static VALUE
02758 rb_str_match(VALUE x, VALUE y)
02759 {
02760     if (SPECIAL_CONST_P(y)) goto generic;
02761     switch (BUILTIN_TYPE(y)) {
02762       case T_STRING:
02763         rb_raise(rb_eTypeError, "type mismatch: String given");
02764 
02765       case T_REGEXP:
02766         return rb_reg_match(y, x);
02767 
02768       generic:
02769       default:
02770         return rb_funcall(y, rb_intern("=~"), 1, x);
02771     }
02772 }
02773 
02774 
02775 static VALUE get_pat(VALUE, int);
02776 
02777 
02778 /*
02779  *  call-seq:
02780  *     str.match(pattern)        -> matchdata or nil
02781  *     str.match(pattern, pos)   -> matchdata or nil
02782  *
02783  *  Converts <i>pattern</i> to a <code>Regexp</code> (if it isn't already one),
02784  *  then invokes its <code>match</code> method on <i>str</i>.  If the second
02785  *  parameter is present, it specifies the position in the string to begin the
02786  *  search.
02787  *
02788  *     'hello'.match('(.)\1')      #=> #<MatchData "ll" 1:"l">
02789  *     'hello'.match('(.)\1')[0]   #=> "ll"
02790  *     'hello'.match(/(.)\1/)[0]   #=> "ll"
02791  *     'hello'.match('xx')         #=> nil
02792  *
02793  *  If a block is given, invoke the block with MatchData if match succeed, so
02794  *  that you can write
02795  *
02796  *     str.match(pat) {|m| ...}
02797  *
02798  *  instead of
02799  *
02800  *     if m = str.match(pat)
02801  *       ...
02802  *     end
02803  *
02804  *  The return value is a value from block execution in this case.
02805  */
02806 
02807 static VALUE
02808 rb_str_match_m(int argc, VALUE *argv, VALUE str)
02809 {
02810     VALUE re, result;
02811     if (argc < 1)
02812         rb_check_arity(argc, 1, 2);
02813     re = argv[0];
02814     argv[0] = str;
02815     result = rb_funcall2(get_pat(re, 0), rb_intern("match"), argc, argv);
02816     if (!NIL_P(result) && rb_block_given_p()) {
02817         return rb_yield(result);
02818     }
02819     return result;
02820 }
02821 
02822 enum neighbor_char {
02823     NEIGHBOR_NOT_CHAR,
02824     NEIGHBOR_FOUND,
02825     NEIGHBOR_WRAPPED
02826 };
02827 
02828 static enum neighbor_char
02829 enc_succ_char(char *p, long len, rb_encoding *enc)
02830 {
02831     long i;
02832     int l;
02833     while (1) {
02834         for (i = len-1; 0 <= i && (unsigned char)p[i] == 0xff; i--)
02835             p[i] = '\0';
02836         if (i < 0)
02837             return NEIGHBOR_WRAPPED;
02838         ++((unsigned char*)p)[i];
02839         l = rb_enc_precise_mbclen(p, p+len, enc);
02840         if (MBCLEN_CHARFOUND_P(l)) {
02841             l = MBCLEN_CHARFOUND_LEN(l);
02842             if (l == len) {
02843                 return NEIGHBOR_FOUND;
02844             }
02845             else {
02846                 memset(p+l, 0xff, len-l);
02847             }
02848         }
02849         if (MBCLEN_INVALID_P(l) && i < len-1) {
02850             long len2;
02851             int l2;
02852             for (len2 = len-1; 0 < len2; len2--) {
02853                 l2 = rb_enc_precise_mbclen(p, p+len2, enc);
02854                 if (!MBCLEN_INVALID_P(l2))
02855                     break;
02856             }
02857             memset(p+len2+1, 0xff, len-(len2+1));
02858         }
02859     }
02860 }
02861 
02862 static enum neighbor_char
02863 enc_pred_char(char *p, long len, rb_encoding *enc)
02864 {
02865     long i;
02866     int l;
02867     while (1) {
02868         for (i = len-1; 0 <= i && (unsigned char)p[i] == 0; i--)
02869             p[i] = '\xff';
02870         if (i < 0)
02871             return NEIGHBOR_WRAPPED;
02872         --((unsigned char*)p)[i];
02873         l = rb_enc_precise_mbclen(p, p+len, enc);
02874         if (MBCLEN_CHARFOUND_P(l)) {
02875             l = MBCLEN_CHARFOUND_LEN(l);
02876             if (l == len) {
02877                 return NEIGHBOR_FOUND;
02878             }
02879             else {
02880                 memset(p+l, 0, len-l);
02881             }
02882         }
02883         if (MBCLEN_INVALID_P(l) && i < len-1) {
02884             long len2;
02885             int l2;
02886             for (len2 = len-1; 0 < len2; len2--) {
02887                 l2 = rb_enc_precise_mbclen(p, p+len2, enc);
02888                 if (!MBCLEN_INVALID_P(l2))
02889                     break;
02890             }
02891             memset(p+len2+1, 0, len-(len2+1));
02892         }
02893     }
02894 }
02895 
02896 /*
02897   overwrite +p+ by succeeding letter in +enc+ and returns
02898   NEIGHBOR_FOUND or NEIGHBOR_WRAPPED.
02899   When NEIGHBOR_WRAPPED, carried-out letter is stored into carry.
02900   assuming each ranges are successive, and mbclen
02901   never change in each ranges.
02902   NEIGHBOR_NOT_CHAR is returned if invalid character or the range has only one
02903   character.
02904  */
02905 static enum neighbor_char
02906 enc_succ_alnum_char(char *p, long len, rb_encoding *enc, char *carry)
02907 {
02908     enum neighbor_char ret;
02909     unsigned int c;
02910     int ctype;
02911     int range;
02912     char save[ONIGENC_CODE_TO_MBC_MAXLEN];
02913 
02914     c = rb_enc_mbc_to_codepoint(p, p+len, enc);
02915     if (rb_enc_isctype(c, ONIGENC_CTYPE_DIGIT, enc))
02916         ctype = ONIGENC_CTYPE_DIGIT;
02917     else if (rb_enc_isctype(c, ONIGENC_CTYPE_ALPHA, enc))
02918         ctype = ONIGENC_CTYPE_ALPHA;
02919     else
02920         return NEIGHBOR_NOT_CHAR;
02921 
02922     MEMCPY(save, p, char, len);
02923     ret = enc_succ_char(p, len, enc);
02924     if (ret == NEIGHBOR_FOUND) {
02925         c = rb_enc_mbc_to_codepoint(p, p+len, enc);
02926         if (rb_enc_isctype(c, ctype, enc))
02927             return NEIGHBOR_FOUND;
02928     }
02929     MEMCPY(p, save, char, len);
02930     range = 1;
02931     while (1) {
02932         MEMCPY(save, p, char, len);
02933         ret = enc_pred_char(p, len, enc);
02934         if (ret == NEIGHBOR_FOUND) {
02935             c = rb_enc_mbc_to_codepoint(p, p+len, enc);
02936             if (!rb_enc_isctype(c, ctype, enc)) {
02937                 MEMCPY(p, save, char, len);
02938                 break;
02939             }
02940         }
02941         else {
02942             MEMCPY(p, save, char, len);
02943             break;
02944         }
02945         range++;
02946     }
02947     if (range == 1) {
02948         return NEIGHBOR_NOT_CHAR;
02949     }
02950 
02951     if (ctype != ONIGENC_CTYPE_DIGIT) {
02952         MEMCPY(carry, p, char, len);
02953         return NEIGHBOR_WRAPPED;
02954     }
02955 
02956     MEMCPY(carry, p, char, len);
02957     enc_succ_char(carry, len, enc);
02958     return NEIGHBOR_WRAPPED;
02959 }
02960 
02961 
02962 /*
02963  *  call-seq:
02964  *     str.succ   -> new_str
02965  *     str.next   -> new_str
02966  *
02967  *  Returns the successor to <i>str</i>. The successor is calculated by
02968  *  incrementing characters starting from the rightmost alphanumeric (or
02969  *  the rightmost character if there are no alphanumerics) in the
02970  *  string. Incrementing a digit always results in another digit, and
02971  *  incrementing a letter results in another letter of the same case.
02972  *  Incrementing nonalphanumerics uses the underlying character set's
02973  *  collating sequence.
02974  *
02975  *  If the increment generates a ``carry,'' the character to the left of
02976  *  it is incremented. This process repeats until there is no carry,
02977  *  adding an additional character if necessary.
02978  *
02979  *     "abcd".succ        #=> "abce"
02980  *     "THX1138".succ     #=> "THX1139"
02981  *     "<<koala>>".succ   #=> "<<koalb>>"
02982  *     "1999zzz".succ     #=> "2000aaa"
02983  *     "ZZZ9999".succ     #=> "AAAA0000"
02984  *     "***".succ         #=> "**+"
02985  */
02986 
02987 VALUE
02988 rb_str_succ(VALUE orig)
02989 {
02990     rb_encoding *enc;
02991     VALUE str;
02992     char *sbeg, *s, *e, *last_alnum = 0;
02993     int c = -1;
02994     long l;
02995     char carry[ONIGENC_CODE_TO_MBC_MAXLEN] = "\1";
02996     long carry_pos = 0, carry_len = 1;
02997     enum neighbor_char neighbor = NEIGHBOR_FOUND;
02998 
02999     str = rb_str_new5(orig, RSTRING_PTR(orig), RSTRING_LEN(orig));
03000     rb_enc_cr_str_copy_for_substr(str, orig);
03001     OBJ_INFECT(str, orig);
03002     if (RSTRING_LEN(str) == 0) return str;
03003 
03004     enc = STR_ENC_GET(orig);
03005     sbeg = RSTRING_PTR(str);
03006     s = e = sbeg + RSTRING_LEN(str);
03007 
03008     while ((s = rb_enc_prev_char(sbeg, s, e, enc)) != 0) {
03009         if (neighbor == NEIGHBOR_NOT_CHAR && last_alnum) {
03010             if (ISALPHA(*last_alnum) ? ISDIGIT(*s) :
03011                 ISDIGIT(*last_alnum) ? ISALPHA(*s) : 0) {
03012                 s = last_alnum;
03013                 break;
03014             }
03015         }
03016         if ((l = rb_enc_precise_mbclen(s, e, enc)) <= 0) continue;
03017         neighbor = enc_succ_alnum_char(s, l, enc, carry);
03018         switch (neighbor) {
03019           case NEIGHBOR_NOT_CHAR:
03020             continue;
03021           case NEIGHBOR_FOUND:
03022             return str;
03023           case NEIGHBOR_WRAPPED:
03024             last_alnum = s;
03025             break;
03026         }
03027         c = 1;
03028         carry_pos = s - sbeg;
03029         carry_len = l;
03030     }
03031     if (c == -1) {              /* str contains no alnum */
03032         s = e;
03033         while ((s = rb_enc_prev_char(sbeg, s, e, enc)) != 0) {
03034             enum neighbor_char neighbor;
03035             if ((l = rb_enc_precise_mbclen(s, e, enc)) <= 0) continue;
03036             neighbor = enc_succ_char(s, l, enc);
03037             if (neighbor == NEIGHBOR_FOUND)
03038                 return str;
03039             if (rb_enc_precise_mbclen(s, s+l, enc) != l) {
03040                 /* wrapped to \0...\0.  search next valid char. */
03041                 enc_succ_char(s, l, enc);
03042             }
03043             if (!rb_enc_asciicompat(enc)) {
03044                 MEMCPY(carry, s, char, l);
03045                 carry_len = l;
03046             }
03047             carry_pos = s - sbeg;
03048         }
03049     }
03050     RESIZE_CAPA(str, RSTRING_LEN(str) + carry_len);
03051     s = RSTRING_PTR(str) + carry_pos;
03052     memmove(s + carry_len, s, RSTRING_LEN(str) - carry_pos);
03053     memmove(s, carry, carry_len);
03054     STR_SET_LEN(str, RSTRING_LEN(str) + carry_len);
03055     RSTRING_PTR(str)[RSTRING_LEN(str)] = '\0';
03056     rb_enc_str_coderange(str);
03057     return str;
03058 }
03059 
03060 
03061 /*
03062  *  call-seq:
03063  *     str.succ!   -> str
03064  *     str.next!   -> str
03065  *
03066  *  Equivalent to <code>String#succ</code>, but modifies the receiver in
03067  *  place.
03068  */
03069 
03070 static VALUE
03071 rb_str_succ_bang(VALUE str)
03072 {
03073     rb_str_shared_replace(str, rb_str_succ(str));
03074 
03075     return str;
03076 }
03077 
03078 
03079 /*
03080  *  call-seq:
03081  *     str.upto(other_str, exclusive=false) {|s| block }   -> str
03082  *     str.upto(other_str, exclusive=false)                -> an_enumerator
03083  *
03084  *  Iterates through successive values, starting at <i>str</i> and
03085  *  ending at <i>other_str</i> inclusive, passing each value in turn to
03086  *  the block. The <code>String#succ</code> method is used to generate
03087  *  each value.  If optional second argument exclusive is omitted or is false,
03088  *  the last value will be included; otherwise it will be excluded.
03089  *
03090  *  If no block is given, an enumerator is returned instead.
03091  *
03092  *     "a8".upto("b6") {|s| print s, ' ' }
03093  *     for s in "a8".."b6"
03094  *       print s, ' '
03095  *     end
03096  *
03097  *  <em>produces:</em>
03098  *
03099  *     a8 a9 b0 b1 b2 b3 b4 b5 b6
03100  *     a8 a9 b0 b1 b2 b3 b4 b5 b6
03101  *
03102  *  If <i>str</i> and <i>other_str</i> contains only ascii numeric characters,
03103  *  both are recognized as decimal numbers. In addition, the width of
03104  *  string (e.g. leading zeros) is handled appropriately.
03105  *
03106  *     "9".upto("11").to_a   #=> ["9", "10", "11"]
03107  *     "25".upto("5").to_a   #=> []
03108  *     "07".upto("11").to_a  #=> ["07", "08", "09", "10", "11"]
03109  */
03110 
03111 static VALUE
03112 rb_str_upto(int argc, VALUE *argv, VALUE beg)
03113 {
03114     VALUE end, exclusive;
03115     VALUE current, after_end;
03116     ID succ;
03117     int n, excl, ascii;
03118     rb_encoding *enc;
03119 
03120     rb_scan_args(argc, argv, "11", &end, &exclusive);
03121     RETURN_ENUMERATOR(beg, argc, argv);
03122     excl = RTEST(exclusive);
03123     CONST_ID(succ, "succ");
03124     StringValue(end);
03125     enc = rb_enc_check(beg, end);
03126     ascii = (is_ascii_string(beg) && is_ascii_string(end));
03127     /* single character */
03128     if (RSTRING_LEN(beg) == 1 && RSTRING_LEN(end) == 1 && ascii) {
03129         char c = RSTRING_PTR(beg)[0];
03130         char e = RSTRING_PTR(end)[0];
03131 
03132         if (c > e || (excl && c == e)) return beg;
03133         for (;;) {
03134             rb_yield(rb_enc_str_new(&c, 1, enc));
03135             if (!excl && c == e) break;
03136             c++;
03137             if (excl && c == e) break;
03138         }
03139         return beg;
03140     }
03141     /* both edges are all digits */
03142     if (ascii && ISDIGIT(RSTRING_PTR(beg)[0]) && ISDIGIT(RSTRING_PTR(end)[0])) {
03143         char *s, *send;
03144         VALUE b, e;
03145         int width;
03146 
03147         s = RSTRING_PTR(beg); send = RSTRING_END(beg);
03148         width = rb_long2int(send - s);
03149         while (s < send) {
03150             if (!ISDIGIT(*s)) goto no_digits;
03151             s++;
03152         }
03153         s = RSTRING_PTR(end); send = RSTRING_END(end);
03154         while (s < send) {
03155             if (!ISDIGIT(*s)) goto no_digits;
03156             s++;
03157         }
03158         b = rb_str_to_inum(beg, 10, FALSE);
03159         e = rb_str_to_inum(end, 10, FALSE);
03160         if (FIXNUM_P(b) && FIXNUM_P(e)) {
03161             long bi = FIX2LONG(b);
03162             long ei = FIX2LONG(e);
03163             rb_encoding *usascii = rb_usascii_encoding();
03164 
03165             while (bi <= ei) {
03166                 if (excl && bi == ei) break;
03167                 rb_yield(rb_enc_sprintf(usascii, "%.*ld", width, bi));
03168                 bi++;
03169             }
03170         }
03171         else {
03172             ID op = excl ? '<' : rb_intern("<=");
03173             VALUE args[2], fmt = rb_obj_freeze(rb_usascii_str_new_cstr("%.*d"));
03174 
03175             args[0] = INT2FIX(width);
03176             while (rb_funcall(b, op, 1, e)) {
03177                 args[1] = b;
03178                 rb_yield(rb_str_format(numberof(args), args, fmt));
03179                 b = rb_funcall(b, succ, 0, 0);
03180             }
03181         }
03182         return beg;
03183     }
03184     /* normal case */
03185   no_digits:
03186     n = rb_str_cmp(beg, end);
03187     if (n > 0 || (excl && n == 0)) return beg;
03188 
03189     after_end = rb_funcall(end, succ, 0, 0);
03190     current = rb_str_dup(beg);
03191     while (!rb_str_equal(current, after_end)) {
03192         VALUE next = Qnil;
03193         if (excl || !rb_str_equal(current, end))
03194             next = rb_funcall(current, succ, 0, 0);
03195         rb_yield(current);
03196         if (NIL_P(next)) break;
03197         current = next;
03198         StringValue(current);
03199         if (excl && rb_str_equal(current, end)) break;
03200         if (RSTRING_LEN(current) > RSTRING_LEN(end) || RSTRING_LEN(current) == 0)
03201             break;
03202     }
03203 
03204     return beg;
03205 }
03206 
03207 static VALUE
03208 rb_str_subpat(VALUE str, VALUE re, VALUE backref)
03209 {
03210     if (rb_reg_search(re, str, 0, 0) >= 0) {
03211         VALUE match = rb_backref_get();
03212         int nth = rb_reg_backref_number(match, backref);
03213         return rb_reg_nth_match(nth, match);
03214     }
03215     return Qnil;
03216 }
03217 
03218 static VALUE
03219 rb_str_aref(VALUE str, VALUE indx)
03220 {
03221     long idx;
03222 
03223     if (FIXNUM_P(indx)) {
03224         idx = FIX2LONG(indx);
03225 
03226       num_index:
03227         str = rb_str_substr(str, idx, 1);
03228         if (!NIL_P(str) && RSTRING_LEN(str) == 0) return Qnil;
03229         return str;
03230     }
03231 
03232     if (SPECIAL_CONST_P(indx)) goto generic;
03233     switch (BUILTIN_TYPE(indx)) {
03234       case T_REGEXP:
03235         return rb_str_subpat(str, indx, INT2FIX(0));
03236 
03237       case T_STRING:
03238         if (rb_str_index(str, indx, 0) != -1)
03239             return rb_str_dup(indx);
03240         return Qnil;
03241 
03242       generic:
03243       default:
03244         /* check if indx is Range */
03245         {
03246             long beg, len;
03247             VALUE tmp;
03248 
03249             len = str_strlen(str, STR_ENC_GET(str));
03250             switch (rb_range_beg_len(indx, &beg, &len, len, 0)) {
03251               case Qfalse:
03252                 break;
03253               case Qnil:
03254                 return Qnil;
03255               default:
03256                 tmp = rb_str_substr(str, beg, len);
03257                 return tmp;
03258             }
03259         }
03260         idx = NUM2LONG(indx);
03261         goto num_index;
03262     }
03263 
03264     UNREACHABLE;
03265 }
03266 
03267 
03268 /*
03269  *  call-seq:
03270  *     str[index]                 -> new_str or nil
03271  *     str[start, length]         -> new_str or nil
03272  *     str[range]                 -> new_str or nil
03273  *     str[regexp]                -> new_str or nil
03274  *     str[regexp, capture]       -> new_str or nil
03275  *     str[match_str]             -> new_str or nil
03276  *     str.slice(index)           -> new_str or nil
03277  *     str.slice(start, length)   -> new_str or nil
03278  *     str.slice(range)           -> new_str or nil
03279  *     str.slice(regexp)          -> new_str or nil
03280  *     str.slice(regexp, capture) -> new_str or nil
03281  *     str.slice(match_str)       -> new_str or nil
03282  *
03283  *  Element Reference --- If passed a single +index+, returns a substring of
03284  *  one character at that index. If passed a +start+ index and a +length+,
03285  *  returns a substring containing +length+ characters starting at the
03286  *  +index+. If passed a +range+, its beginning and end are interpreted as
03287  *  offsets delimiting the substring to be returned.
03288  *
03289  *  In these three cases, if an index is negative, it is counted from the end
03290  *  of the string.  For the +start+ and +range+ cases the starting index
03291  *  is just before a character and an index matching the string's size.
03292  *  Additionally, an empty string is returned when the starting index for a
03293  *  character range is at the end of the string.
03294  *
03295  *  Returns +nil+ if the initial index falls outside the string or the length
03296  *  is negative.
03297  *
03298  *  If a +Regexp+ is supplied, the matching portion of the string is
03299  *  returned.  If a +capture+ follows the regular expression, which may be a
03300  *  capture group index or name, follows the regular expression that component
03301  *  of the MatchData is returned instead.
03302  *
03303  *  If a +match_str+ is given, that string is returned if it occurs in
03304  *  the string.
03305  *
03306  *  Returns +nil+ if the regular expression does not match or the match string
03307  *  cannot be found.
03308  *
03309  *     a = "hello there"
03310  *
03311  *     a[1]                   #=> "e"
03312  *     a[2, 3]                #=> "llo"
03313  *     a[2..3]                #=> "ll"
03314  *
03315  *     a[-3, 2]               #=> "er"
03316  *     a[7..-2]               #=> "her"
03317  *     a[-4..-2]              #=> "her"
03318  *     a[-2..-4]              #=> ""
03319  *
03320  *     a[11, 0]               #=> ""
03321  *     a[11]                  #=> nil
03322  *     a[12, 0]               #=> nil
03323  *     a[12..-1]              #=> nil
03324  *
03325  *     a[/[aeiou](.)\1/]      #=> "ell"
03326  *     a[/[aeiou](.)\1/, 0]   #=> "ell"
03327  *     a[/[aeiou](.)\1/, 1]   #=> "l"
03328  *     a[/[aeiou](.)\1/, 2]   #=> nil
03329  *
03330  *     a[/(?<vowel>[aeiou])(?<non_vowel>[^aeiou])/, "non_vowel"] #=> "l"
03331  *     a[/(?<vowel>[aeiou])(?<non_vowel>[^aeiou])/, "vowel"]     #=> "e"
03332  *
03333  *     a["lo"]                #=> "lo"
03334  *     a["bye"]               #=> nil
03335  */
03336 
03337 static VALUE
03338 rb_str_aref_m(int argc, VALUE *argv, VALUE str)
03339 {
03340     if (argc == 2) {
03341         if (RB_TYPE_P(argv[0], T_REGEXP)) {
03342             return rb_str_subpat(str, argv[0], argv[1]);
03343         }
03344         return rb_str_substr(str, NUM2LONG(argv[0]), NUM2LONG(argv[1]));
03345     }
03346     rb_check_arity(argc, 1, 2);
03347     return rb_str_aref(str, argv[0]);
03348 }
03349 
03350 VALUE
03351 rb_str_drop_bytes(VALUE str, long len)
03352 {
03353     char *ptr = RSTRING_PTR(str);
03354     long olen = RSTRING_LEN(str), nlen;
03355 
03356     str_modifiable(str);
03357     if (len > olen) len = olen;
03358     nlen = olen - len;
03359     if (nlen <= RSTRING_EMBED_LEN_MAX) {
03360         char *oldptr = ptr;
03361         int fl = (int)(RBASIC(str)->flags & (STR_NOEMBED|ELTS_SHARED));
03362         STR_SET_EMBED(str);
03363         STR_SET_EMBED_LEN(str, nlen);
03364         ptr = RSTRING(str)->as.ary;
03365         memmove(ptr, oldptr + len, nlen);
03366         if (fl == STR_NOEMBED) xfree(oldptr);
03367     }
03368     else {
03369         if (!STR_SHARED_P(str)) rb_str_new4(str);
03370         ptr = RSTRING(str)->as.heap.ptr += len;
03371         RSTRING(str)->as.heap.len = nlen;
03372     }
03373     ptr[nlen] = 0;
03374     ENC_CODERANGE_CLEAR(str);
03375     return str;
03376 }
03377 
03378 static void
03379 rb_str_splice_0(VALUE str, long beg, long len, VALUE val)
03380 {
03381     if (beg == 0 && RSTRING_LEN(val) == 0) {
03382         rb_str_drop_bytes(str, len);
03383         OBJ_INFECT(str, val);
03384         return;
03385     }
03386 
03387     rb_str_modify(str);
03388     if (len < RSTRING_LEN(val)) {
03389         /* expand string */
03390         RESIZE_CAPA(str, RSTRING_LEN(str) + RSTRING_LEN(val) - len + 1);
03391     }
03392 
03393     if (RSTRING_LEN(val) != len) {
03394         memmove(RSTRING_PTR(str) + beg + RSTRING_LEN(val),
03395                 RSTRING_PTR(str) + beg + len,
03396                 RSTRING_LEN(str) - (beg + len));
03397     }
03398     if (RSTRING_LEN(val) < beg && len < 0) {
03399         MEMZERO(RSTRING_PTR(str) + RSTRING_LEN(str), char, -len);
03400     }
03401     if (RSTRING_LEN(val) > 0) {
03402         memmove(RSTRING_PTR(str)+beg, RSTRING_PTR(val), RSTRING_LEN(val));
03403     }
03404     STR_SET_LEN(str, RSTRING_LEN(str) + RSTRING_LEN(val) - len);
03405     if (RSTRING_PTR(str)) {
03406         RSTRING_PTR(str)[RSTRING_LEN(str)] = '\0';
03407     }
03408     OBJ_INFECT(str, val);
03409 }
03410 
03411 static void
03412 rb_str_splice(VALUE str, long beg, long len, VALUE val)
03413 {
03414     long slen;
03415     char *p, *e;
03416     rb_encoding *enc;
03417     int singlebyte = single_byte_optimizable(str);
03418     int cr;
03419 
03420     if (len < 0) rb_raise(rb_eIndexError, "negative length %ld", len);
03421 
03422     StringValue(val);
03423     enc = rb_enc_check(str, val);
03424     slen = str_strlen(str, enc);
03425 
03426     if (slen < beg) {
03427       out_of_range:
03428         rb_raise(rb_eIndexError, "index %ld out of string", beg);
03429     }
03430     if (beg < 0) {
03431         if (-beg > slen) {
03432             goto out_of_range;
03433         }
03434         beg += slen;
03435     }
03436     if (slen < len || slen < beg + len) {
03437         len = slen - beg;
03438     }
03439     str_modify_keep_cr(str);
03440     p = str_nth(RSTRING_PTR(str), RSTRING_END(str), beg, enc, singlebyte);
03441     if (!p) p = RSTRING_END(str);
03442     e = str_nth(p, RSTRING_END(str), len, enc, singlebyte);
03443     if (!e) e = RSTRING_END(str);
03444     /* error check */
03445     beg = p - RSTRING_PTR(str); /* physical position */
03446     len = e - p;                /* physical length */
03447     rb_str_splice_0(str, beg, len, val);
03448     rb_enc_associate(str, enc);
03449     cr = ENC_CODERANGE_AND(ENC_CODERANGE(str), ENC_CODERANGE(val));
03450     if (cr != ENC_CODERANGE_BROKEN)
03451         ENC_CODERANGE_SET(str, cr);
03452 }
03453 
03454 void
03455 rb_str_update(VALUE str, long beg, long len, VALUE val)
03456 {
03457     rb_str_splice(str, beg, len, val);
03458 }
03459 
03460 static void
03461 rb_str_subpat_set(VALUE str, VALUE re, VALUE backref, VALUE val)
03462 {
03463     int nth;
03464     VALUE match;
03465     long start, end, len;
03466     rb_encoding *enc;
03467     struct re_registers *regs;
03468 
03469     if (rb_reg_search(re, str, 0, 0) < 0) {
03470         rb_raise(rb_eIndexError, "regexp not matched");
03471     }
03472     match = rb_backref_get();
03473     nth = rb_reg_backref_number(match, backref);
03474     regs = RMATCH_REGS(match);
03475     if (nth >= regs->num_regs) {
03476       out_of_range:
03477         rb_raise(rb_eIndexError, "index %d out of regexp", nth);
03478     }
03479     if (nth < 0) {
03480         if (-nth >= regs->num_regs) {
03481             goto out_of_range;
03482         }
03483         nth += regs->num_regs;
03484     }
03485 
03486     start = BEG(nth);
03487     if (start == -1) {
03488         rb_raise(rb_eIndexError, "regexp group %d not matched", nth);
03489     }
03490     end = END(nth);
03491     len = end - start;
03492     StringValue(val);
03493     enc = rb_enc_check(str, val);
03494     rb_str_splice_0(str, start, len, val);
03495     rb_enc_associate(str, enc);
03496 }
03497 
03498 static VALUE
03499 rb_str_aset(VALUE str, VALUE indx, VALUE val)
03500 {
03501     long idx, beg;
03502 
03503     if (FIXNUM_P(indx)) {
03504         idx = FIX2LONG(indx);
03505       num_index:
03506         rb_str_splice(str, idx, 1, val);
03507         return val;
03508     }
03509 
03510     if (SPECIAL_CONST_P(indx)) goto generic;
03511     switch (TYPE(indx)) {
03512       case T_REGEXP:
03513         rb_str_subpat_set(str, indx, INT2FIX(0), val);
03514         return val;
03515 
03516       case T_STRING:
03517         beg = rb_str_index(str, indx, 0);
03518         if (beg < 0) {
03519             rb_raise(rb_eIndexError, "string not matched");
03520         }
03521         beg = rb_str_sublen(str, beg);
03522         rb_str_splice(str, beg, str_strlen(indx, 0), val);
03523         return val;
03524 
03525       generic:
03526       default:
03527         /* check if indx is Range */
03528         {
03529             long beg, len;
03530             if (rb_range_beg_len(indx, &beg, &len, str_strlen(str, 0), 2)) {
03531                 rb_str_splice(str, beg, len, val);
03532                 return val;
03533             }
03534         }
03535         idx = NUM2LONG(indx);
03536         goto num_index;
03537     }
03538 }
03539 
03540 /*
03541  *  call-seq:
03542  *     str[fixnum] = new_str
03543  *     str[fixnum, fixnum] = new_str
03544  *     str[range] = aString
03545  *     str[regexp] = new_str
03546  *     str[regexp, fixnum] = new_str
03547  *     str[regexp, name] = new_str
03548  *     str[other_str] = new_str
03549  *
03550  *  Element Assignment---Replaces some or all of the content of <i>str</i>. The
03551  *  portion of the string affected is determined using the same criteria as
03552  *  <code>String#[]</code>. If the replacement string is not the same length as
03553  *  the text it is replacing, the string will be adjusted accordingly. If the
03554  *  regular expression or string is used as the index doesn't match a position
03555  *  in the string, <code>IndexError</code> is raised. If the regular expression
03556  *  form is used, the optional second <code>Fixnum</code> allows you to specify
03557  *  which portion of the match to replace (effectively using the
03558  *  <code>MatchData</code> indexing rules. The forms that take a
03559  *  <code>Fixnum</code> will raise an <code>IndexError</code> if the value is
03560  *  out of range; the <code>Range</code> form will raise a
03561  *  <code>RangeError</code>, and the <code>Regexp</code> and <code>String</code>
03562  *  will raise an <code>IndexError</code> on negative match.
03563  */
03564 
03565 static VALUE
03566 rb_str_aset_m(int argc, VALUE *argv, VALUE str)
03567 {
03568     if (argc == 3) {
03569         if (RB_TYPE_P(argv[0], T_REGEXP)) {
03570             rb_str_subpat_set(str, argv[0], argv[1], argv[2]);
03571         }
03572         else {
03573             rb_str_splice(str, NUM2LONG(argv[0]), NUM2LONG(argv[1]), argv[2]);
03574         }
03575         return argv[2];
03576     }
03577     rb_check_arity(argc, 2, 3);
03578     return rb_str_aset(str, argv[0], argv[1]);
03579 }
03580 
03581 /*
03582  *  call-seq:
03583  *     str.insert(index, other_str)   -> str
03584  *
03585  *  Inserts <i>other_str</i> before the character at the given
03586  *  <i>index</i>, modifying <i>str</i>. Negative indices count from the
03587  *  end of the string, and insert <em>after</em> the given character.
03588  *  The intent is insert <i>aString</i> so that it starts at the given
03589  *  <i>index</i>.
03590  *
03591  *     "abcd".insert(0, 'X')    #=> "Xabcd"
03592  *     "abcd".insert(3, 'X')    #=> "abcXd"
03593  *     "abcd".insert(4, 'X')    #=> "abcdX"
03594  *     "abcd".insert(-3, 'X')   #=> "abXcd"
03595  *     "abcd".insert(-1, 'X')   #=> "abcdX"
03596  */
03597 
03598 static VALUE
03599 rb_str_insert(VALUE str, VALUE idx, VALUE str2)
03600 {
03601     long pos = NUM2LONG(idx);
03602 
03603     if (pos == -1) {
03604         return rb_str_append(str, str2);
03605     }
03606     else if (pos < 0) {
03607         pos++;
03608     }
03609     rb_str_splice(str, pos, 0, str2);
03610     return str;
03611 }
03612 
03613 
03614 /*
03615  *  call-seq:
03616  *     str.slice!(fixnum)           -> fixnum or nil
03617  *     str.slice!(fixnum, fixnum)   -> new_str or nil
03618  *     str.slice!(range)            -> new_str or nil
03619  *     str.slice!(regexp)           -> new_str or nil
03620  *     str.slice!(other_str)        -> new_str or nil
03621  *
03622  *  Deletes the specified portion from <i>str</i>, and returns the portion
03623  *  deleted.
03624  *
03625  *     string = "this is a string"
03626  *     string.slice!(2)        #=> "i"
03627  *     string.slice!(3..6)     #=> " is "
03628  *     string.slice!(/s.*t/)   #=> "sa st"
03629  *     string.slice!("r")      #=> "r"
03630  *     string                  #=> "thing"
03631  */
03632 
03633 static VALUE
03634 rb_str_slice_bang(int argc, VALUE *argv, VALUE str)
03635 {
03636     VALUE result;
03637     VALUE buf[3];
03638     int i;
03639 
03640     rb_check_arity(argc, 1, 2);
03641     for (i=0; i<argc; i++) {
03642         buf[i] = argv[i];
03643     }
03644     str_modify_keep_cr(str);
03645     result = rb_str_aref_m(argc, buf, str);
03646     if (!NIL_P(result)) {
03647         buf[i] = rb_str_new(0,0);
03648         rb_str_aset_m(argc+1, buf, str);
03649     }
03650     return result;
03651 }
03652 
03653 static VALUE
03654 get_pat(VALUE pat, int quote)
03655 {
03656     VALUE val;
03657 
03658     switch (TYPE(pat)) {
03659       case T_REGEXP:
03660         return pat;
03661 
03662       case T_STRING:
03663         break;
03664 
03665       default:
03666         val = rb_check_string_type(pat);
03667         if (NIL_P(val)) {
03668             Check_Type(pat, T_REGEXP);
03669         }
03670         pat = val;
03671     }
03672 
03673     if (quote) {
03674         pat = rb_reg_quote(pat);
03675     }
03676 
03677     return rb_reg_regcomp(pat);
03678 }
03679 
03680 
03681 /*
03682  *  call-seq:
03683  *     str.sub!(pattern, replacement)          -> str or nil
03684  *     str.sub!(pattern) {|match| block }      -> str or nil
03685  *
03686  *  Performs the same substitution as String#sub in-place.
03687  *
03688  *  Returns +str+ if a substitution was performed or +nil+ if no substitution
03689  *  was performed.
03690  */
03691 
03692 static VALUE
03693 rb_str_sub_bang(int argc, VALUE *argv, VALUE str)
03694 {
03695     VALUE pat, repl, hash = Qnil;
03696     int iter = 0;
03697     int tainted = 0;
03698     int untrusted = 0;
03699     long plen;
03700     int min_arity = rb_block_given_p() ? 1 : 2;
03701 
03702     rb_check_arity(argc, min_arity, 2);
03703     if (argc == 1) {
03704         iter = 1;
03705     }
03706     else {
03707         repl = argv[1];
03708         hash = rb_check_hash_type(argv[1]);
03709         if (NIL_P(hash)) {
03710             StringValue(repl);
03711         }
03712         if (OBJ_TAINTED(repl)) tainted = 1;
03713         if (OBJ_UNTRUSTED(repl)) untrusted = 1;
03714     }
03715 
03716     pat = get_pat(argv[0], 1);
03717     str_modifiable(str);
03718     if (rb_reg_search(pat, str, 0, 0) >= 0) {
03719         rb_encoding *enc;
03720         int cr = ENC_CODERANGE(str);
03721         VALUE match = rb_backref_get();
03722         struct re_registers *regs = RMATCH_REGS(match);
03723         long beg0 = BEG(0);
03724         long end0 = END(0);
03725         char *p, *rp;
03726         long len, rlen;
03727 
03728         if (iter || !NIL_P(hash)) {
03729             p = RSTRING_PTR(str); len = RSTRING_LEN(str);
03730 
03731             if (iter) {
03732                 repl = rb_obj_as_string(rb_yield(rb_reg_nth_match(0, match)));
03733             }
03734             else {
03735                 repl = rb_hash_aref(hash, rb_str_subseq(str, beg0, end0 - beg0));
03736                 repl = rb_obj_as_string(repl);
03737             }
03738             str_mod_check(str, p, len);
03739             rb_check_frozen(str);
03740         }
03741         else {
03742             repl = rb_reg_regsub(repl, str, regs, pat);
03743         }
03744         enc = rb_enc_compatible(str, repl);
03745         if (!enc) {
03746             rb_encoding *str_enc = STR_ENC_GET(str);
03747             p = RSTRING_PTR(str); len = RSTRING_LEN(str);
03748             if (coderange_scan(p, beg0, str_enc) != ENC_CODERANGE_7BIT ||
03749                 coderange_scan(p+end0, len-end0, str_enc) != ENC_CODERANGE_7BIT) {
03750                 rb_raise(rb_eEncCompatError, "incompatible character encodings: %s and %s",
03751                          rb_enc_name(str_enc),
03752                          rb_enc_name(STR_ENC_GET(repl)));
03753             }
03754             enc = STR_ENC_GET(repl);
03755         }
03756         rb_str_modify(str);
03757         rb_enc_associate(str, enc);
03758         if (OBJ_TAINTED(repl)) tainted = 1;
03759         if (OBJ_UNTRUSTED(repl)) untrusted = 1;
03760         if (ENC_CODERANGE_UNKNOWN < cr && cr < ENC_CODERANGE_BROKEN) {
03761             int cr2 = ENC_CODERANGE(repl);
03762             if (cr2 == ENC_CODERANGE_BROKEN ||
03763                 (cr == ENC_CODERANGE_VALID && cr2 == ENC_CODERANGE_7BIT))
03764                 cr = ENC_CODERANGE_UNKNOWN;
03765             else
03766                 cr = cr2;
03767         }
03768         plen = end0 - beg0;
03769         rp = RSTRING_PTR(repl); rlen = RSTRING_LEN(repl);
03770         len = RSTRING_LEN(str);
03771         if (rlen > plen) {
03772             RESIZE_CAPA(str, len + rlen - plen);
03773         }
03774         p = RSTRING_PTR(str);
03775         if (rlen != plen) {
03776             memmove(p + beg0 + rlen, p + beg0 + plen, len - beg0 - plen);
03777         }
03778         memcpy(p + beg0, rp, rlen);
03779         len += rlen - plen;
03780         STR_SET_LEN(str, len);
03781         RSTRING_PTR(str)[len] = '\0';
03782         ENC_CODERANGE_SET(str, cr);
03783         if (tainted) OBJ_TAINT(str);
03784         if (untrusted) OBJ_UNTRUST(str);
03785 
03786         return str;
03787     }
03788     return Qnil;
03789 }
03790 
03791 
03792 /*
03793  *  call-seq:
03794  *     str.sub(pattern, replacement)         -> new_str
03795  *     str.sub(pattern, hash)                -> new_str
03796  *     str.sub(pattern) {|match| block }     -> new_str
03797  *
03798  *  Returns a copy of +str+ with the _first_ occurrence of +pattern+
03799  *  replaced by the second argument. The +pattern+ is typically a Regexp; if
03800  *  given as a String, any regular expression metacharacters it contains will
03801  *  be interpreted literally, e.g. <code>'\\\d'</code> will match a backlash
03802  *  followed by 'd', instead of a digit.
03803  *
03804  *  If +replacement+ is a String it will be substituted for the matched text.
03805  *  It may contain back-references to the pattern's capture groups of the form
03806  *  <code>"\\d"</code>, where <i>d</i> is a group number, or
03807  *  <code>"\\k<n>"</code>, where <i>n</i> is a group name. If it is a
03808  *  double-quoted string, both back-references must be preceded by an
03809  *  additional backslash. However, within +replacement+ the special match
03810  *  variables, such as <code>&$</code>, will not refer to the current match.
03811  *
03812  *  If the second argument is a Hash, and the matched text is one of its keys,
03813  *  the corresponding value is the replacement string.
03814  *
03815  *  In the block form, the current match string is passed in as a parameter,
03816  *  and variables such as <code>$1</code>, <code>$2</code>, <code>$`</code>,
03817  *  <code>$&</code>, and <code>$'</code> will be set appropriately. The value
03818  *  returned by the block will be substituted for the match on each call.
03819  *
03820  *  The result inherits any tainting in the original string or any supplied
03821  *  replacement string.
03822  *
03823  *     "hello".sub(/[aeiou]/, '*')                  #=> "h*llo"
03824  *     "hello".sub(/([aeiou])/, '<\1>')             #=> "h<e>llo"
03825  *     "hello".sub(/./) {|s| s.ord.to_s + ' ' }     #=> "104 ello"
03826  *     "hello".sub(/(?<foo>[aeiou])/, '*\k<foo>*')  #=> "h*e*llo"
03827  *     'Is SHELL your preferred shell?'.sub(/[[:upper:]]{2,}/, ENV)
03828  *      #=> "Is /bin/bash your preferred shell?"
03829  */
03830 
03831 static VALUE
03832 rb_str_sub(int argc, VALUE *argv, VALUE str)
03833 {
03834     str = rb_str_dup(str);
03835     rb_str_sub_bang(argc, argv, str);
03836     return str;
03837 }
03838 
03839 static VALUE
03840 str_gsub(int argc, VALUE *argv, VALUE str, int bang)
03841 {
03842     VALUE pat, val, repl, match, dest, hash = Qnil;
03843     struct re_registers *regs;
03844     long beg, n;
03845     long beg0, end0;
03846     long offset, blen, slen, len, last;
03847     int iter = 0;
03848     char *sp, *cp;
03849     int tainted = 0;
03850     rb_encoding *str_enc;
03851 
03852     switch (argc) {
03853       case 1:
03854         RETURN_ENUMERATOR(str, argc, argv);
03855         iter = 1;
03856         break;
03857       case 2:
03858         repl = argv[1];
03859         hash = rb_check_hash_type(argv[1]);
03860         if (NIL_P(hash)) {
03861             StringValue(repl);
03862         }
03863         if (OBJ_TAINTED(repl)) tainted = 1;
03864         break;
03865       default:
03866         rb_check_arity(argc, 1, 2);
03867     }
03868 
03869     pat = get_pat(argv[0], 1);
03870     beg = rb_reg_search(pat, str, 0, 0);
03871     if (beg < 0) {
03872         if (bang) return Qnil;  /* no match, no substitution */
03873         return rb_str_dup(str);
03874     }
03875 
03876     offset = 0;
03877     n = 0;
03878     blen = RSTRING_LEN(str) + 30; /* len + margin */
03879     dest = rb_str_buf_new(blen);
03880     sp = RSTRING_PTR(str);
03881     slen = RSTRING_LEN(str);
03882     cp = sp;
03883     str_enc = STR_ENC_GET(str);
03884     rb_enc_associate(dest, str_enc);
03885     ENC_CODERANGE_SET(dest, rb_enc_asciicompat(str_enc) ? ENC_CODERANGE_7BIT : ENC_CODERANGE_VALID);
03886 
03887     do {
03888         n++;
03889         match = rb_backref_get();
03890         regs = RMATCH_REGS(match);
03891         beg0 = BEG(0);
03892         end0 = END(0);
03893         if (iter || !NIL_P(hash)) {
03894             if (iter) {
03895                 val = rb_obj_as_string(rb_yield(rb_reg_nth_match(0, match)));
03896             }
03897             else {
03898                 val = rb_hash_aref(hash, rb_str_subseq(str, BEG(0), END(0) - BEG(0)));
03899                 val = rb_obj_as_string(val);
03900             }
03901             str_mod_check(str, sp, slen);
03902             if (val == dest) {  /* paranoid check [ruby-dev:24827] */
03903                 rb_raise(rb_eRuntimeError, "block should not cheat");
03904             }
03905         }
03906         else {
03907             val = rb_reg_regsub(repl, str, regs, pat);
03908         }
03909 
03910         if (OBJ_TAINTED(val)) tainted = 1;
03911 
03912         len = beg0 - offset;    /* copy pre-match substr */
03913         if (len) {
03914             rb_enc_str_buf_cat(dest, cp, len, str_enc);
03915         }
03916 
03917         rb_str_buf_append(dest, val);
03918 
03919         last = offset;
03920         offset = end0;
03921         if (beg0 == end0) {
03922             /*
03923              * Always consume at least one character of the input string
03924              * in order to prevent infinite loops.
03925              */
03926             if (RSTRING_LEN(str) <= end0) break;
03927             len = rb_enc_fast_mbclen(RSTRING_PTR(str)+end0, RSTRING_END(str), str_enc);
03928             rb_enc_str_buf_cat(dest, RSTRING_PTR(str)+end0, len, str_enc);
03929             offset = end0 + len;
03930         }
03931         cp = RSTRING_PTR(str) + offset;
03932         if (offset > RSTRING_LEN(str)) break;
03933         beg = rb_reg_search(pat, str, offset, 0);
03934     } while (beg >= 0);
03935     if (RSTRING_LEN(str) > offset) {
03936         rb_enc_str_buf_cat(dest, cp, RSTRING_LEN(str) - offset, str_enc);
03937     }
03938     rb_reg_search(pat, str, last, 0);
03939     if (bang) {
03940         rb_str_shared_replace(str, dest);
03941     }
03942     else {
03943         RBASIC(dest)->klass = rb_obj_class(str);
03944         OBJ_INFECT(dest, str);
03945         str = dest;
03946     }
03947 
03948     if (tainted) OBJ_TAINT(str);
03949     return str;
03950 }
03951 
03952 
03953 /*
03954  *  call-seq:
03955  *     str.gsub!(pattern, replacement)        -> str or nil
03956  *     str.gsub!(pattern) {|match| block }    -> str or nil
03957  *     str.gsub!(pattern)                     -> an_enumerator
03958  *
03959  *  Performs the substitutions of <code>String#gsub</code> in place, returning
03960  *  <i>str</i>, or <code>nil</code> if no substitutions were performed.
03961  *  If no block and no <i>replacement</i> is given, an enumerator is returned instead.
03962  */
03963 
03964 static VALUE
03965 rb_str_gsub_bang(int argc, VALUE *argv, VALUE str)
03966 {
03967     str_modify_keep_cr(str);
03968     return str_gsub(argc, argv, str, 1);
03969 }
03970 
03971 
03972 /*
03973  *  call-seq:
03974  *     str.gsub(pattern, replacement)       -> new_str
03975  *     str.gsub(pattern, hash)              -> new_str
03976  *     str.gsub(pattern) {|match| block }   -> new_str
03977  *     str.gsub(pattern)                    -> enumerator
03978  *
03979  *  Returns a copy of <i>str</i> with the <em>all</em> occurrences of
03980  *  <i>pattern</i> substituted for the second argument. The <i>pattern</i> is
03981  *  typically a <code>Regexp</code>; if given as a <code>String</code>, any
03982  *  regular expression metacharacters it contains will be interpreted
03983  *  literally, e.g. <code>'\\\d'</code> will match a backlash followed by 'd',
03984  *  instead of a digit.
03985  *
03986  *  If <i>replacement</i> is a <code>String</code> it will be substituted for
03987  *  the matched text. It may contain back-references to the pattern's capture
03988  *  groups of the form <code>\\\d</code>, where <i>d</i> is a group number, or
03989  *  <code>\\\k<n></code>, where <i>n</i> is a group name. If it is a
03990  *  double-quoted string, both back-references must be preceded by an
03991  *  additional backslash. However, within <i>replacement</i> the special match
03992  *  variables, such as <code>$&</code>, will not refer to the current match.
03993  *
03994  *  If the second argument is a <code>Hash</code>, and the matched text is one
03995  *  of its keys, the corresponding value is the replacement string.
03996  *
03997  *  In the block form, the current match string is passed in as a parameter,
03998  *  and variables such as <code>$1</code>, <code>$2</code>, <code>$`</code>,
03999  *  <code>$&</code>, and <code>$'</code> will be set appropriately. The value
04000  *  returned by the block will be substituted for the match on each call.
04001  *
04002  *  The result inherits any tainting in the original string or any supplied
04003  *  replacement string.
04004  *
04005  *  When neither a block nor a second argument is supplied, an
04006  *  <code>Enumerator</code> is returned.
04007  *
04008  *     "hello".gsub(/[aeiou]/, '*')                  #=> "h*ll*"
04009  *     "hello".gsub(/([aeiou])/, '<\1>')             #=> "h<e>ll<o>"
04010  *     "hello".gsub(/./) {|s| s.ord.to_s + ' '}      #=> "104 101 108 108 111 "
04011  *     "hello".gsub(/(?<foo>[aeiou])/, '{\k<foo>}')  #=> "h{e}ll{o}"
04012  *     'hello'.gsub(/[eo]/, 'e' => 3, 'o' => '*')    #=> "h3ll*"
04013  */
04014 
04015 static VALUE
04016 rb_str_gsub(int argc, VALUE *argv, VALUE str)
04017 {
04018     return str_gsub(argc, argv, str, 0);
04019 }
04020 
04021 
04022 /*
04023  *  call-seq:
04024  *     str.replace(other_str)   -> str
04025  *
04026  *  Replaces the contents and taintedness of <i>str</i> with the corresponding
04027  *  values in <i>other_str</i>.
04028  *
04029  *     s = "hello"         #=> "hello"
04030  *     s.replace "world"   #=> "world"
04031  */
04032 
04033 VALUE
04034 rb_str_replace(VALUE str, VALUE str2)
04035 {
04036     str_modifiable(str);
04037     if (str == str2) return str;
04038 
04039     StringValue(str2);
04040     str_discard(str);
04041     return str_replace(str, str2);
04042 }
04043 
04044 /*
04045  *  call-seq:
04046  *     string.clear    ->  string
04047  *
04048  *  Makes string empty.
04049  *
04050  *     a = "abcde"
04051  *     a.clear    #=> ""
04052  */
04053 
04054 static VALUE
04055 rb_str_clear(VALUE str)
04056 {
04057     str_discard(str);
04058     STR_SET_EMBED(str);
04059     STR_SET_EMBED_LEN(str, 0);
04060     RSTRING_PTR(str)[0] = 0;
04061     if (rb_enc_asciicompat(STR_ENC_GET(str)))
04062         ENC_CODERANGE_SET(str, ENC_CODERANGE_7BIT);
04063     else
04064         ENC_CODERANGE_SET(str, ENC_CODERANGE_VALID);
04065     return str;
04066 }
04067 
04068 /*
04069  *  call-seq:
04070  *     string.chr    ->  string
04071  *
04072  *  Returns a one-character string at the beginning of the string.
04073  *
04074  *     a = "abcde"
04075  *     a.chr    #=> "a"
04076  */
04077 
04078 static VALUE
04079 rb_str_chr(VALUE str)
04080 {
04081     return rb_str_substr(str, 0, 1);
04082 }
04083 
04084 /*
04085  *  call-seq:
04086  *     str.getbyte(index)          -> 0 .. 255
04087  *
04088  *  returns the <i>index</i>th byte as an integer.
04089  */
04090 static VALUE
04091 rb_str_getbyte(VALUE str, VALUE index)
04092 {
04093     long pos = NUM2LONG(index);
04094 
04095     if (pos < 0)
04096         pos += RSTRING_LEN(str);
04097     if (pos < 0 ||  RSTRING_LEN(str) <= pos)
04098         return Qnil;
04099 
04100     return INT2FIX((unsigned char)RSTRING_PTR(str)[pos]);
04101 }
04102 
04103 /*
04104  *  call-seq:
04105  *     str.setbyte(index, integer) -> integer
04106  *
04107  *  modifies the <i>index</i>th byte as <i>integer</i>.
04108  */
04109 static VALUE
04110 rb_str_setbyte(VALUE str, VALUE index, VALUE value)
04111 {
04112     long pos = NUM2LONG(index);
04113     int byte = NUM2INT(value);
04114 
04115     rb_str_modify(str);
04116 
04117     if (pos < -RSTRING_LEN(str) || RSTRING_LEN(str) <= pos)
04118         rb_raise(rb_eIndexError, "index %ld out of string", pos);
04119     if (pos < 0)
04120         pos += RSTRING_LEN(str);
04121 
04122     RSTRING_PTR(str)[pos] = byte;
04123 
04124     return value;
04125 }
04126 
04127 static VALUE
04128 str_byte_substr(VALUE str, long beg, long len)
04129 {
04130     char *p, *s = RSTRING_PTR(str);
04131     long n = RSTRING_LEN(str);
04132     VALUE str2;
04133 
04134     if (beg > n || len < 0) return Qnil;
04135     if (beg < 0) {
04136         beg += n;
04137         if (beg < 0) return Qnil;
04138     }
04139     if (beg + len > n)
04140         len = n - beg;
04141     if (len <= 0) {
04142         len = 0;
04143         p = 0;
04144     }
04145     else
04146         p = s + beg;
04147 
04148     if (len > RSTRING_EMBED_LEN_MAX && beg + len == n) {
04149         str2 = rb_str_new4(str);
04150         str2 = str_new3(rb_obj_class(str2), str2);
04151         RSTRING(str2)->as.heap.ptr += RSTRING(str2)->as.heap.len - len;
04152         RSTRING(str2)->as.heap.len = len;
04153     }
04154     else {
04155         str2 = rb_str_new5(str, p, len);
04156     }
04157 
04158     str_enc_copy(str2, str);
04159 
04160     if (RSTRING_LEN(str2) == 0) {
04161         if (!rb_enc_asciicompat(STR_ENC_GET(str)))
04162             ENC_CODERANGE_SET(str2, ENC_CODERANGE_VALID);
04163         else
04164             ENC_CODERANGE_SET(str2, ENC_CODERANGE_7BIT);
04165     }
04166     else {
04167         switch (ENC_CODERANGE(str)) {
04168           case ENC_CODERANGE_7BIT:
04169             ENC_CODERANGE_SET(str2, ENC_CODERANGE_7BIT);
04170             break;
04171           default:
04172             ENC_CODERANGE_SET(str2, ENC_CODERANGE_UNKNOWN);
04173             break;
04174         }
04175     }
04176 
04177     OBJ_INFECT(str2, str);
04178 
04179     return str2;
04180 }
04181 
04182 static VALUE
04183 str_byte_aref(VALUE str, VALUE indx)
04184 {
04185     long idx;
04186     switch (TYPE(indx)) {
04187       case T_FIXNUM:
04188         idx = FIX2LONG(indx);
04189 
04190       num_index:
04191         str = str_byte_substr(str, idx, 1);
04192         if (NIL_P(str) || RSTRING_LEN(str) == 0) return Qnil;
04193         return str;
04194 
04195       default:
04196         /* check if indx is Range */
04197         {
04198             long beg, len = RSTRING_LEN(str);
04199 
04200             switch (rb_range_beg_len(indx, &beg, &len, len, 0)) {
04201               case Qfalse:
04202                 break;
04203               case Qnil:
04204                 return Qnil;
04205               default:
04206                 return str_byte_substr(str, beg, len);
04207             }
04208         }
04209         idx = NUM2LONG(indx);
04210         goto num_index;
04211     }
04212 
04213     UNREACHABLE;
04214 }
04215 
04216 /*
04217  *  call-seq:
04218  *     str.byteslice(fixnum)           -> new_str or nil
04219  *     str.byteslice(fixnum, fixnum)   -> new_str or nil
04220  *     str.byteslice(range)            -> new_str or nil
04221  *
04222  *  Byte Reference---If passed a single <code>Fixnum</code>, returns a
04223  *  substring of one byte at that position. If passed two <code>Fixnum</code>
04224  *  objects, returns a substring starting at the offset given by the first, and
04225  *  a length given by the second. If given a <code>Range</code>, a substring containing
04226  *  bytes at offsets given by the range is returned. In all three cases, if
04227  *  an offset is negative, it is counted from the end of <i>str</i>. Returns
04228  *  <code>nil</code> if the initial offset falls outside the string, the length
04229  *  is negative, or the beginning of the range is greater than the end.
04230  *  The encoding of the resulted string keeps original encoding.
04231  *
04232  *     "hello".byteslice(1)     #=> "e"
04233  *     "hello".byteslice(-1)    #=> "o"
04234  *     "hello".byteslice(1, 2)  #=> "el"
04235  *     "\x80\u3042".byteslice(1, 3) #=> "\u3042"
04236  *     "\x03\u3042\xff".byteslice(1..3) #=> "\u3042"
04237  */
04238 
04239 static VALUE
04240 rb_str_byteslice(int argc, VALUE *argv, VALUE str)
04241 {
04242     if (argc == 2) {
04243         return str_byte_substr(str, NUM2LONG(argv[0]), NUM2LONG(argv[1]));
04244     }
04245     rb_check_arity(argc, 1, 2);
04246     return str_byte_aref(str, argv[0]);
04247 }
04248 
04249 /*
04250  *  call-seq:
04251  *     str.reverse   -> new_str
04252  *
04253  *  Returns a new string with the characters from <i>str</i> in reverse order.
04254  *
04255  *     "stressed".reverse   #=> "desserts"
04256  */
04257 
04258 static VALUE
04259 rb_str_reverse(VALUE str)
04260 {
04261     rb_encoding *enc;
04262     VALUE rev;
04263     char *s, *e, *p;
04264     int single = 1;
04265 
04266     if (RSTRING_LEN(str) <= 1) return rb_str_dup(str);
04267     enc = STR_ENC_GET(str);
04268     rev = rb_str_new5(str, 0, RSTRING_LEN(str));
04269     s = RSTRING_PTR(str); e = RSTRING_END(str);
04270     p = RSTRING_END(rev);
04271 
04272     if (RSTRING_LEN(str) > 1) {
04273         if (single_byte_optimizable(str)) {
04274             while (s < e) {
04275                 *--p = *s++;
04276             }
04277         }
04278         else if (ENC_CODERANGE(str) == ENC_CODERANGE_VALID) {
04279             while (s < e) {
04280                 int clen = rb_enc_fast_mbclen(s, e, enc);
04281 
04282                 if (clen > 1 || (*s & 0x80)) single = 0;
04283                 p -= clen;
04284                 memcpy(p, s, clen);
04285                 s += clen;
04286             }
04287         }
04288         else {
04289             while (s < e) {
04290                 int clen = rb_enc_mbclen(s, e, enc);
04291 
04292                 if (clen > 1 || (*s & 0x80)) single = 0;
04293                 p -= clen;
04294                 memcpy(p, s, clen);
04295                 s += clen;
04296             }
04297         }
04298     }
04299     STR_SET_LEN(rev, RSTRING_LEN(str));
04300     OBJ_INFECT(rev, str);
04301     if (ENC_CODERANGE(str) == ENC_CODERANGE_UNKNOWN) {
04302         if (single) {
04303             ENC_CODERANGE_SET(str, ENC_CODERANGE_7BIT);
04304         }
04305         else {
04306             ENC_CODERANGE_SET(str, ENC_CODERANGE_VALID);
04307         }
04308     }
04309     rb_enc_cr_str_copy_for_substr(rev, str);
04310 
04311     return rev;
04312 }
04313 
04314 
04315 /*
04316  *  call-seq:
04317  *     str.reverse!   -> str
04318  *
04319  *  Reverses <i>str</i> in place.
04320  */
04321 
04322 static VALUE
04323 rb_str_reverse_bang(VALUE str)
04324 {
04325     if (RSTRING_LEN(str) > 1) {
04326         if (single_byte_optimizable(str)) {
04327             char *s, *e, c;
04328 
04329             str_modify_keep_cr(str);
04330             s = RSTRING_PTR(str);
04331             e = RSTRING_END(str) - 1;
04332             while (s < e) {
04333                 c = *s;
04334                 *s++ = *e;
04335                 *e-- = c;
04336             }
04337         }
04338         else {
04339             rb_str_shared_replace(str, rb_str_reverse(str));
04340         }
04341     }
04342     else {
04343         str_modify_keep_cr(str);
04344     }
04345     return str;
04346 }
04347 
04348 
04349 /*
04350  *  call-seq:
04351  *     str.include? other_str   -> true or false
04352  *
04353  *  Returns <code>true</code> if <i>str</i> contains the given string or
04354  *  character.
04355  *
04356  *     "hello".include? "lo"   #=> true
04357  *     "hello".include? "ol"   #=> false
04358  *     "hello".include? ?h     #=> true
04359  */
04360 
04361 static VALUE
04362 rb_str_include(VALUE str, VALUE arg)
04363 {
04364     long i;
04365 
04366     StringValue(arg);
04367     i = rb_str_index(str, arg, 0);
04368 
04369     if (i == -1) return Qfalse;
04370     return Qtrue;
04371 }
04372 
04373 
04374 /*
04375  *  call-seq:
04376  *     str.to_i(base=10)   -> integer
04377  *
04378  *  Returns the result of interpreting leading characters in <i>str</i> as an
04379  *  integer base <i>base</i> (between 2 and 36). Extraneous characters past the
04380  *  end of a valid number are ignored. If there is not a valid number at the
04381  *  start of <i>str</i>, <code>0</code> is returned. This method never raises an
04382  *  exception when <i>base</i> is valid.
04383  *
04384  *     "12345".to_i             #=> 12345
04385  *     "99 red balloons".to_i   #=> 99
04386  *     "0a".to_i                #=> 0
04387  *     "0a".to_i(16)            #=> 10
04388  *     "hello".to_i             #=> 0
04389  *     "1100101".to_i(2)        #=> 101
04390  *     "1100101".to_i(8)        #=> 294977
04391  *     "1100101".to_i(10)       #=> 1100101
04392  *     "1100101".to_i(16)       #=> 17826049
04393  */
04394 
04395 static VALUE
04396 rb_str_to_i(int argc, VALUE *argv, VALUE str)
04397 {
04398     int base;
04399 
04400     if (argc == 0) base = 10;
04401     else {
04402         VALUE b;
04403 
04404         rb_scan_args(argc, argv, "01", &b);
04405         base = NUM2INT(b);
04406     }
04407     if (base < 0) {
04408         rb_raise(rb_eArgError, "invalid radix %d", base);
04409     }
04410     return rb_str_to_inum(str, base, FALSE);
04411 }
04412 
04413 
04414 /*
04415  *  call-seq:
04416  *     str.to_f   -> float
04417  *
04418  *  Returns the result of interpreting leading characters in <i>str</i> as a
04419  *  floating point number. Extraneous characters past the end of a valid number
04420  *  are ignored. If there is not a valid number at the start of <i>str</i>,
04421  *  <code>0.0</code> is returned. This method never raises an exception.
04422  *
04423  *     "123.45e1".to_f        #=> 1234.5
04424  *     "45.67 degrees".to_f   #=> 45.67
04425  *     "thx1138".to_f         #=> 0.0
04426  */
04427 
04428 static VALUE
04429 rb_str_to_f(VALUE str)
04430 {
04431     return DBL2NUM(rb_str_to_dbl(str, FALSE));
04432 }
04433 
04434 
04435 /*
04436  *  call-seq:
04437  *     str.to_s     -> str
04438  *     str.to_str   -> str
04439  *
04440  *  Returns the receiver.
04441  */
04442 
04443 static VALUE
04444 rb_str_to_s(VALUE str)
04445 {
04446     if (rb_obj_class(str) != rb_cString) {
04447         return str_duplicate(rb_cString, str);
04448     }
04449     return str;
04450 }
04451 
04452 #if 0
04453 static void
04454 str_cat_char(VALUE str, unsigned int c, rb_encoding *enc)
04455 {
04456     char s[RUBY_MAX_CHAR_LEN];
04457     int n = rb_enc_codelen(c, enc);
04458 
04459     rb_enc_mbcput(c, s, enc);
04460     rb_enc_str_buf_cat(str, s, n, enc);
04461 }
04462 #endif
04463 
04464 #define CHAR_ESC_LEN 13 /* sizeof(\x{ hex of 32bit unsigned int } \0) */
04465 
04466 int
04467 rb_str_buf_cat_escaped_char(VALUE result, unsigned int c, int unicode_p)
04468 {
04469     char buf[CHAR_ESC_LEN + 1];
04470     int l;
04471 
04472 #if SIZEOF_INT > 4
04473     c &= 0xffffffff;
04474 #endif
04475     if (unicode_p) {
04476         if (c < 0x7F && ISPRINT(c)) {
04477             snprintf(buf, CHAR_ESC_LEN, "%c", c);
04478         }
04479         else if (c < 0x10000) {
04480             snprintf(buf, CHAR_ESC_LEN, "\\u%04X", c);
04481         }
04482         else {
04483             snprintf(buf, CHAR_ESC_LEN, "\\u{%X}", c);
04484         }
04485     }
04486     else {
04487         if (c < 0x100) {
04488             snprintf(buf, CHAR_ESC_LEN, "\\x%02X", c);
04489         }
04490         else {
04491             snprintf(buf, CHAR_ESC_LEN, "\\x{%X}", c);
04492         }
04493     }
04494     l = (int)strlen(buf);       /* CHAR_ESC_LEN cannot exceed INT_MAX */
04495     rb_str_buf_cat(result, buf, l);
04496     return l;
04497 }
04498 
04499 /*
04500  * call-seq:
04501  *   str.inspect   -> string
04502  *
04503  * Returns a printable version of _str_, surrounded by quote marks,
04504  * with special characters escaped.
04505  *
04506  *    str = "hello"
04507  *    str[3] = "\b"
04508  *    str.inspect       #=> "\"hel\\bo\""
04509  */
04510 
04511 VALUE
04512 rb_str_inspect(VALUE str)
04513 {
04514     rb_encoding *enc = STR_ENC_GET(str);
04515     const char *p, *pend, *prev;
04516     char buf[CHAR_ESC_LEN + 1];
04517     VALUE result = rb_str_buf_new(0);
04518     rb_encoding *resenc = rb_default_internal_encoding();
04519     int unicode_p = rb_enc_unicode_p(enc);
04520     int asciicompat = rb_enc_asciicompat(enc);
04521     static rb_encoding *utf16, *utf32;
04522 
04523     if (!utf16) utf16 = rb_enc_find("UTF-16");
04524     if (!utf32) utf32 = rb_enc_find("UTF-32");
04525     if (resenc == NULL) resenc = rb_default_external_encoding();
04526     if (!rb_enc_asciicompat(resenc)) resenc = rb_usascii_encoding();
04527     rb_enc_associate(result, resenc);
04528     str_buf_cat2(result, "\"");
04529 
04530     p = RSTRING_PTR(str); pend = RSTRING_END(str);
04531     prev = p;
04532     if (enc == utf16) {
04533         const unsigned char *q = (const unsigned char *)p;
04534         if (q[0] == 0xFE && q[1] == 0xFF)
04535             enc = rb_enc_find("UTF-16BE");
04536         else if (q[0] == 0xFF && q[1] == 0xFE)
04537             enc = rb_enc_find("UTF-16LE");
04538         else
04539             unicode_p = 0;
04540     }
04541     else if (enc == utf32) {
04542         const unsigned char *q = (const unsigned char *)p;
04543         if (q[0] == 0 && q[1] == 0 && q[2] == 0xFE && q[3] == 0xFF)
04544             enc = rb_enc_find("UTF-32BE");
04545         else if (q[3] == 0 && q[2] == 0 && q[1] == 0xFE && q[0] == 0xFF)
04546             enc = rb_enc_find("UTF-32LE");
04547         else
04548             unicode_p = 0;
04549     }
04550     while (p < pend) {
04551         unsigned int c, cc;
04552         int n;
04553 
04554         n = rb_enc_precise_mbclen(p, pend, enc);
04555         if (!MBCLEN_CHARFOUND_P(n)) {
04556             if (p > prev) str_buf_cat(result, prev, p - prev);
04557             n = rb_enc_mbminlen(enc);
04558             if (pend < p + n)
04559                 n = (int)(pend - p);
04560             while (n--) {
04561                 snprintf(buf, CHAR_ESC_LEN, "\\x%02X", *p & 0377);
04562                 str_buf_cat(result, buf, strlen(buf));
04563                 prev = ++p;
04564             }
04565             continue;
04566         }
04567         n = MBCLEN_CHARFOUND_LEN(n);
04568         c = rb_enc_mbc_to_codepoint(p, pend, enc);
04569         p += n;
04570         if ((asciicompat || unicode_p) &&
04571           (c == '"'|| c == '\\' ||
04572             (c == '#' &&
04573              p < pend &&
04574              MBCLEN_CHARFOUND_P(rb_enc_precise_mbclen(p,pend,enc)) &&
04575              (cc = rb_enc_codepoint(p,pend,enc),
04576               (cc == '$' || cc == '@' || cc == '{'))))) {
04577             if (p - n > prev) str_buf_cat(result, prev, p - n - prev);
04578             str_buf_cat2(result, "\\");
04579             if (asciicompat || enc == resenc) {
04580                 prev = p - n;
04581                 continue;
04582             }
04583         }
04584         switch (c) {
04585           case '\n': cc = 'n'; break;
04586           case '\r': cc = 'r'; break;
04587           case '\t': cc = 't'; break;
04588           case '\f': cc = 'f'; break;
04589           case '\013': cc = 'v'; break;
04590           case '\010': cc = 'b'; break;
04591           case '\007': cc = 'a'; break;
04592           case 033: cc = 'e'; break;
04593           default: cc = 0; break;
04594         }
04595         if (cc) {
04596             if (p - n > prev) str_buf_cat(result, prev, p - n - prev);
04597             buf[0] = '\\';
04598             buf[1] = (char)cc;
04599             str_buf_cat(result, buf, 2);
04600             prev = p;
04601             continue;
04602         }
04603         if ((enc == resenc && rb_enc_isprint(c, enc)) ||
04604             (asciicompat && rb_enc_isascii(c, enc) && ISPRINT(c))) {
04605             continue;
04606         }
04607         else {
04608             if (p - n > prev) str_buf_cat(result, prev, p - n - prev);
04609             rb_str_buf_cat_escaped_char(result, c, unicode_p);
04610             prev = p;
04611             continue;
04612         }
04613     }
04614     if (p > prev) str_buf_cat(result, prev, p - prev);
04615     str_buf_cat2(result, "\"");
04616 
04617     OBJ_INFECT(result, str);
04618     return result;
04619 }
04620 
04621 #define IS_EVSTR(p,e) ((p) < (e) && (*(p) == '$' || *(p) == '@' || *(p) == '{'))
04622 
04623 /*
04624  *  call-seq:
04625  *     str.dump   -> new_str
04626  *
04627  *  Produces a version of +str+ with all non-printing characters replaced by
04628  *  <code>\nnn</code> notation and all special characters escaped.
04629  *
04630  *    "hello \n ''".dump  #=> "\"hello \\n ''\"
04631  */
04632 
04633 VALUE
04634 rb_str_dump(VALUE str)
04635 {
04636     rb_encoding *enc = rb_enc_get(str);
04637     long len;
04638     const char *p, *pend;
04639     char *q, *qend;
04640     VALUE result;
04641     int u8 = (enc == rb_utf8_encoding());
04642 
04643     len = 2;                    /* "" */
04644     p = RSTRING_PTR(str); pend = p + RSTRING_LEN(str);
04645     while (p < pend) {
04646         unsigned char c = *p++;
04647         switch (c) {
04648           case '"':  case '\\':
04649           case '\n': case '\r':
04650           case '\t': case '\f':
04651           case '\013': case '\010': case '\007': case '\033':
04652             len += 2;
04653             break;
04654 
04655           case '#':
04656             len += IS_EVSTR(p, pend) ? 2 : 1;
04657             break;
04658 
04659           default:
04660             if (ISPRINT(c)) {
04661                 len++;
04662             }
04663             else {
04664                 if (u8) {       /* \u{NN} */
04665                     int n = rb_enc_precise_mbclen(p-1, pend, enc);
04666                     if (MBCLEN_CHARFOUND_P(n-1)) {
04667                         unsigned int cc = rb_enc_mbc_to_codepoint(p-1, pend, enc);
04668                         while (cc >>= 4) len++;
04669                         len += 5;
04670                         p += MBCLEN_CHARFOUND_LEN(n)-1;
04671                         break;
04672                     }
04673                 }
04674                 len += 4;       /* \xNN */
04675             }
04676             break;
04677         }
04678     }
04679     if (!rb_enc_asciicompat(enc)) {
04680         len += 19;              /* ".force_encoding('')" */
04681         len += strlen(enc->name);
04682     }
04683 
04684     result = rb_str_new5(str, 0, len);
04685     p = RSTRING_PTR(str); pend = p + RSTRING_LEN(str);
04686     q = RSTRING_PTR(result); qend = q + len + 1;
04687 
04688     *q++ = '"';
04689     while (p < pend) {
04690         unsigned char c = *p++;
04691 
04692         if (c == '"' || c == '\\') {
04693             *q++ = '\\';
04694             *q++ = c;
04695         }
04696         else if (c == '#') {
04697             if (IS_EVSTR(p, pend)) *q++ = '\\';
04698             *q++ = '#';
04699         }
04700         else if (c == '\n') {
04701             *q++ = '\\';
04702             *q++ = 'n';
04703         }
04704         else if (c == '\r') {
04705             *q++ = '\\';
04706             *q++ = 'r';
04707         }
04708         else if (c == '\t') {
04709             *q++ = '\\';
04710             *q++ = 't';
04711         }
04712         else if (c == '\f') {
04713             *q++ = '\\';
04714             *q++ = 'f';
04715         }
04716         else if (c == '\013') {
04717             *q++ = '\\';
04718             *q++ = 'v';
04719         }
04720         else if (c == '\010') {
04721             *q++ = '\\';
04722             *q++ = 'b';
04723         }
04724         else if (c == '\007') {
04725             *q++ = '\\';
04726             *q++ = 'a';
04727         }
04728         else if (c == '\033') {
04729             *q++ = '\\';
04730             *q++ = 'e';
04731         }
04732         else if (ISPRINT(c)) {
04733             *q++ = c;
04734         }
04735         else {
04736             *q++ = '\\';
04737             if (u8) {
04738                 int n = rb_enc_precise_mbclen(p-1, pend, enc) - 1;
04739                 if (MBCLEN_CHARFOUND_P(n)) {
04740                     int cc = rb_enc_mbc_to_codepoint(p-1, pend, enc);
04741                     p += n;
04742                     snprintf(q, qend-q, "u{%x}", cc);
04743                     q += strlen(q);
04744                     continue;
04745                 }
04746             }
04747             snprintf(q, qend-q, "x%02X", c);
04748             q += 3;
04749         }
04750     }
04751     *q++ = '"';
04752     *q = '\0';
04753     if (!rb_enc_asciicompat(enc)) {
04754         snprintf(q, qend-q, ".force_encoding(\"%s\")", enc->name);
04755         enc = rb_ascii8bit_encoding();
04756     }
04757     OBJ_INFECT(result, str);
04758     /* result from dump is ASCII */
04759     rb_enc_associate(result, enc);
04760     ENC_CODERANGE_SET(result, ENC_CODERANGE_7BIT);
04761     return result;
04762 }
04763 
04764 
04765 static void
04766 rb_str_check_dummy_enc(rb_encoding *enc)
04767 {
04768     if (rb_enc_dummy_p(enc)) {
04769         rb_raise(rb_eEncCompatError, "incompatible encoding with this operation: %s",
04770                  rb_enc_name(enc));
04771     }
04772 }
04773 
04774 /*
04775  *  call-seq:
04776  *     str.upcase!   -> str or nil
04777  *
04778  *  Upcases the contents of <i>str</i>, returning <code>nil</code> if no changes
04779  *  were made.
04780  *  Note: case replacement is effective only in ASCII region.
04781  */
04782 
04783 static VALUE
04784 rb_str_upcase_bang(VALUE str)
04785 {
04786     rb_encoding *enc;
04787     char *s, *send;
04788     int modify = 0;
04789     int n;
04790 
04791     str_modify_keep_cr(str);
04792     enc = STR_ENC_GET(str);
04793     rb_str_check_dummy_enc(enc);
04794     s = RSTRING_PTR(str); send = RSTRING_END(str);
04795     if (single_byte_optimizable(str)) {
04796         while (s < send) {
04797             unsigned int c = *(unsigned char*)s;
04798 
04799             if (rb_enc_isascii(c, enc) && 'a' <= c && c <= 'z') {
04800                 *s = 'A' + (c - 'a');
04801                 modify = 1;
04802             }
04803             s++;
04804         }
04805     }
04806     else {
04807         int ascompat = rb_enc_asciicompat(enc);
04808 
04809         while (s < send) {
04810             unsigned int c;
04811 
04812             if (ascompat && (c = *(unsigned char*)s) < 0x80) {
04813                 if (rb_enc_isascii(c, enc) && 'a' <= c && c <= 'z') {
04814                     *s = 'A' + (c - 'a');
04815                     modify = 1;
04816                 }
04817                 s++;
04818             }
04819             else {
04820                 c = rb_enc_codepoint_len(s, send, &n, enc);
04821                 if (rb_enc_islower(c, enc)) {
04822                     /* assuming toupper returns codepoint with same size */
04823                     rb_enc_mbcput(rb_enc_toupper(c, enc), s, enc);
04824                     modify = 1;
04825                 }
04826                 s += n;
04827             }
04828         }
04829     }
04830 
04831     if (modify) return str;
04832     return Qnil;
04833 }
04834 
04835 
04836 /*
04837  *  call-seq:
04838  *     str.upcase   -> new_str
04839  *
04840  *  Returns a copy of <i>str</i> with all lowercase letters replaced with their
04841  *  uppercase counterparts. The operation is locale insensitive---only
04842  *  characters ``a'' to ``z'' are affected.
04843  *  Note: case replacement is effective only in ASCII region.
04844  *
04845  *     "hEllO".upcase   #=> "HELLO"
04846  */
04847 
04848 static VALUE
04849 rb_str_upcase(VALUE str)
04850 {
04851     str = rb_str_dup(str);
04852     rb_str_upcase_bang(str);
04853     return str;
04854 }
04855 
04856 
04857 /*
04858  *  call-seq:
04859  *     str.downcase!   -> str or nil
04860  *
04861  *  Downcases the contents of <i>str</i>, returning <code>nil</code> if no
04862  *  changes were made.
04863  *  Note: case replacement is effective only in ASCII region.
04864  */
04865 
04866 static VALUE
04867 rb_str_downcase_bang(VALUE str)
04868 {
04869     rb_encoding *enc;
04870     char *s, *send;
04871     int modify = 0;
04872 
04873     str_modify_keep_cr(str);
04874     enc = STR_ENC_GET(str);
04875     rb_str_check_dummy_enc(enc);
04876     s = RSTRING_PTR(str); send = RSTRING_END(str);
04877     if (single_byte_optimizable(str)) {
04878         while (s < send) {
04879             unsigned int c = *(unsigned char*)s;
04880 
04881             if (rb_enc_isascii(c, enc) && 'A' <= c && c <= 'Z') {
04882                 *s = 'a' + (c - 'A');
04883                 modify = 1;
04884             }
04885             s++;
04886         }
04887     }
04888     else {
04889         int ascompat = rb_enc_asciicompat(enc);
04890 
04891         while (s < send) {
04892             unsigned int c;
04893             int n;
04894 
04895             if (ascompat && (c = *(unsigned char*)s) < 0x80) {
04896                 if (rb_enc_isascii(c, enc) && 'A' <= c && c <= 'Z') {
04897                     *s = 'a' + (c - 'A');
04898                     modify = 1;
04899                 }
04900                 s++;
04901             }
04902             else {
04903                 c = rb_enc_codepoint_len(s, send, &n, enc);
04904                 if (rb_enc_isupper(c, enc)) {
04905                     /* assuming toupper returns codepoint with same size */
04906                     rb_enc_mbcput(rb_enc_tolower(c, enc), s, enc);
04907                     modify = 1;
04908                 }
04909                 s += n;
04910             }
04911         }
04912     }
04913 
04914     if (modify) return str;
04915     return Qnil;
04916 }
04917 
04918 
04919 /*
04920  *  call-seq:
04921  *     str.downcase   -> new_str
04922  *
04923  *  Returns a copy of <i>str</i> with all uppercase letters replaced with their
04924  *  lowercase counterparts. The operation is locale insensitive---only
04925  *  characters ``A'' to ``Z'' are affected.
04926  *  Note: case replacement is effective only in ASCII region.
04927  *
04928  *     "hEllO".downcase   #=> "hello"
04929  */
04930 
04931 static VALUE
04932 rb_str_downcase(VALUE str)
04933 {
04934     str = rb_str_dup(str);
04935     rb_str_downcase_bang(str);
04936     return str;
04937 }
04938 
04939 
04940 /*
04941  *  call-seq:
04942  *     str.capitalize!   -> str or nil
04943  *
04944  *  Modifies <i>str</i> by converting the first character to uppercase and the
04945  *  remainder to lowercase. Returns <code>nil</code> if no changes are made.
04946  *  Note: case conversion is effective only in ASCII region.
04947  *
04948  *     a = "hello"
04949  *     a.capitalize!   #=> "Hello"
04950  *     a               #=> "Hello"
04951  *     a.capitalize!   #=> nil
04952  */
04953 
04954 static VALUE
04955 rb_str_capitalize_bang(VALUE str)
04956 {
04957     rb_encoding *enc;
04958     char *s, *send;
04959     int modify = 0;
04960     unsigned int c;
04961     int n;
04962 
04963     str_modify_keep_cr(str);
04964     enc = STR_ENC_GET(str);
04965     rb_str_check_dummy_enc(enc);
04966     if (RSTRING_LEN(str) == 0 || !RSTRING_PTR(str)) return Qnil;
04967     s = RSTRING_PTR(str); send = RSTRING_END(str);
04968 
04969     c = rb_enc_codepoint_len(s, send, &n, enc);
04970     if (rb_enc_islower(c, enc)) {
04971         rb_enc_mbcput(rb_enc_toupper(c, enc), s, enc);
04972         modify = 1;
04973     }
04974     s += n;
04975     while (s < send) {
04976         c = rb_enc_codepoint_len(s, send, &n, enc);
04977         if (rb_enc_isupper(c, enc)) {
04978             rb_enc_mbcput(rb_enc_tolower(c, enc), s, enc);
04979             modify = 1;
04980         }
04981         s += n;
04982     }
04983 
04984     if (modify) return str;
04985     return Qnil;
04986 }
04987 
04988 
04989 /*
04990  *  call-seq:
04991  *     str.capitalize   -> new_str
04992  *
04993  *  Returns a copy of <i>str</i> with the first character converted to uppercase
04994  *  and the remainder to lowercase.
04995  *  Note: case conversion is effective only in ASCII region.
04996  *
04997  *     "hello".capitalize    #=> "Hello"
04998  *     "HELLO".capitalize    #=> "Hello"
04999  *     "123ABC".capitalize   #=> "123abc"
05000  */
05001 
05002 static VALUE
05003 rb_str_capitalize(VALUE str)
05004 {
05005     str = rb_str_dup(str);
05006     rb_str_capitalize_bang(str);
05007     return str;
05008 }
05009 
05010 
05011 /*
05012  *  call-seq:
05013  *     str.swapcase!   -> str or nil
05014  *
05015  *  Equivalent to <code>String#swapcase</code>, but modifies the receiver in
05016  *  place, returning <i>str</i>, or <code>nil</code> if no changes were made.
05017  *  Note: case conversion is effective only in ASCII region.
05018  */
05019 
05020 static VALUE
05021 rb_str_swapcase_bang(VALUE str)
05022 {
05023     rb_encoding *enc;
05024     char *s, *send;
05025     int modify = 0;
05026     int n;
05027 
05028     str_modify_keep_cr(str);
05029     enc = STR_ENC_GET(str);
05030     rb_str_check_dummy_enc(enc);
05031     s = RSTRING_PTR(str); send = RSTRING_END(str);
05032     while (s < send) {
05033         unsigned int c = rb_enc_codepoint_len(s, send, &n, enc);
05034 
05035         if (rb_enc_isupper(c, enc)) {
05036             /* assuming toupper returns codepoint with same size */
05037             rb_enc_mbcput(rb_enc_tolower(c, enc), s, enc);
05038             modify = 1;
05039         }
05040         else if (rb_enc_islower(c, enc)) {
05041             /* assuming tolower returns codepoint with same size */
05042             rb_enc_mbcput(rb_enc_toupper(c, enc), s, enc);
05043             modify = 1;
05044         }
05045         s += n;
05046     }
05047 
05048     if (modify) return str;
05049     return Qnil;
05050 }
05051 
05052 
05053 /*
05054  *  call-seq:
05055  *     str.swapcase   -> new_str
05056  *
05057  *  Returns a copy of <i>str</i> with uppercase alphabetic characters converted
05058  *  to lowercase and lowercase characters converted to uppercase.
05059  *  Note: case conversion is effective only in ASCII region.
05060  *
05061  *     "Hello".swapcase          #=> "hELLO"
05062  *     "cYbEr_PuNk11".swapcase   #=> "CyBeR_pUnK11"
05063  */
05064 
05065 static VALUE
05066 rb_str_swapcase(VALUE str)
05067 {
05068     str = rb_str_dup(str);
05069     rb_str_swapcase_bang(str);
05070     return str;
05071 }
05072 
05073 typedef unsigned char *USTR;
05074 
05075 struct tr {
05076     int gen;
05077     unsigned int now, max;
05078     char *p, *pend;
05079 };
05080 
05081 static unsigned int
05082 trnext(struct tr *t, rb_encoding *enc)
05083 {
05084     int n;
05085 
05086     for (;;) {
05087         if (!t->gen) {
05088 nextpart:
05089             if (t->p == t->pend) return -1;
05090             if (rb_enc_ascget(t->p, t->pend, &n, enc) == '\\' && t->p + n < t->pend) {
05091                 t->p += n;
05092             }
05093             t->now = rb_enc_codepoint_len(t->p, t->pend, &n, enc);
05094             t->p += n;
05095             if (rb_enc_ascget(t->p, t->pend, &n, enc) == '-' && t->p + n < t->pend) {
05096                 t->p += n;
05097                 if (t->p < t->pend) {
05098                     unsigned int c = rb_enc_codepoint_len(t->p, t->pend, &n, enc);
05099                     t->p += n;
05100                     if (t->now > c) {
05101                         if (t->now < 0x80 && c < 0x80) {
05102                             rb_raise(rb_eArgError,
05103                                      "invalid range \"%c-%c\" in string transliteration",
05104                                      t->now, c);
05105                         }
05106                         else {
05107                             rb_raise(rb_eArgError, "invalid range in string transliteration");
05108                         }
05109                         continue; /* not reached */
05110                     }
05111                     t->gen = 1;
05112                     t->max = c;
05113                 }
05114             }
05115             return t->now;
05116         }
05117         else {
05118             while (ONIGENC_CODE_TO_MBCLEN(enc, ++t->now) <= 0) {
05119                 if (t->now == t->max) {
05120                     t->gen = 0;
05121                     goto nextpart;
05122                 }
05123             }
05124             if (t->now < t->max) {
05125                 return t->now;
05126             }
05127             else {
05128                 t->gen = 0;
05129                 return t->max;
05130             }
05131         }
05132     }
05133 }
05134 
05135 static VALUE rb_str_delete_bang(int,VALUE*,VALUE);
05136 
05137 static VALUE
05138 tr_trans(VALUE str, VALUE src, VALUE repl, int sflag)
05139 {
05140     const unsigned int errc = -1;
05141     unsigned int trans[256];
05142     rb_encoding *enc, *e1, *e2;
05143     struct tr trsrc, trrepl;
05144     int cflag = 0;
05145     unsigned int c, c0, last = 0;
05146     int modify = 0, i, l;
05147     char *s, *send;
05148     VALUE hash = 0;
05149     int singlebyte = single_byte_optimizable(str);
05150     int cr;
05151 
05152 #define CHECK_IF_ASCII(c) \
05153     (void)((cr == ENC_CODERANGE_7BIT && !rb_isascii(c)) ? \
05154            (cr = ENC_CODERANGE_VALID) : 0)
05155 
05156     StringValue(src);
05157     StringValue(repl);
05158     if (RSTRING_LEN(str) == 0 || !RSTRING_PTR(str)) return Qnil;
05159     if (RSTRING_LEN(repl) == 0) {
05160         return rb_str_delete_bang(1, &src, str);
05161     }
05162 
05163     cr = ENC_CODERANGE(str);
05164     e1 = rb_enc_check(str, src);
05165     e2 = rb_enc_check(str, repl);
05166     if (e1 == e2) {
05167         enc = e1;
05168     }
05169     else {
05170         enc = rb_enc_check(src, repl);
05171     }
05172     trsrc.p = RSTRING_PTR(src); trsrc.pend = trsrc.p + RSTRING_LEN(src);
05173     if (RSTRING_LEN(src) > 1 &&
05174         rb_enc_ascget(trsrc.p, trsrc.pend, &l, enc) == '^' &&
05175         trsrc.p + l < trsrc.pend) {
05176         cflag = 1;
05177         trsrc.p += l;
05178     }
05179     trrepl.p = RSTRING_PTR(repl);
05180     trrepl.pend = trrepl.p + RSTRING_LEN(repl);
05181     trsrc.gen = trrepl.gen = 0;
05182     trsrc.now = trrepl.now = 0;
05183     trsrc.max = trrepl.max = 0;
05184 
05185     if (cflag) {
05186         for (i=0; i<256; i++) {
05187             trans[i] = 1;
05188         }
05189         while ((c = trnext(&trsrc, enc)) != errc) {
05190             if (c < 256) {
05191                 trans[c] = errc;
05192             }
05193             else {
05194                 if (!hash) hash = rb_hash_new();
05195                 rb_hash_aset(hash, UINT2NUM(c), Qtrue);
05196             }
05197         }
05198         while ((c = trnext(&trrepl, enc)) != errc)
05199             /* retrieve last replacer */;
05200         last = trrepl.now;
05201         for (i=0; i<256; i++) {
05202             if (trans[i] != errc) {
05203                 trans[i] = last;
05204             }
05205         }
05206     }
05207     else {
05208         unsigned int r;
05209 
05210         for (i=0; i<256; i++) {
05211             trans[i] = errc;
05212         }
05213         while ((c = trnext(&trsrc, enc)) != errc) {
05214             r = trnext(&trrepl, enc);
05215             if (r == errc) r = trrepl.now;
05216             if (c < 256) {
05217                 trans[c] = r;
05218                 if (rb_enc_codelen(r, enc) != 1) singlebyte = 0;
05219             }
05220             else {
05221                 if (!hash) hash = rb_hash_new();
05222                 rb_hash_aset(hash, UINT2NUM(c), UINT2NUM(r));
05223             }
05224         }
05225     }
05226 
05227     if (cr == ENC_CODERANGE_VALID)
05228         cr = ENC_CODERANGE_7BIT;
05229     str_modify_keep_cr(str);
05230     s = RSTRING_PTR(str); send = RSTRING_END(str);
05231     if (sflag) {
05232         int clen, tlen;
05233         long offset, max = RSTRING_LEN(str);
05234         unsigned int save = -1;
05235         char *buf = ALLOC_N(char, max), *t = buf;
05236 
05237         while (s < send) {
05238             int may_modify = 0;
05239 
05240             c0 = c = rb_enc_codepoint_len(s, send, &clen, e1);
05241             tlen = enc == e1 ? clen : rb_enc_codelen(c, enc);
05242 
05243             s += clen;
05244             if (c < 256) {
05245                 c = trans[c];
05246             }
05247             else if (hash) {
05248                 VALUE tmp = rb_hash_lookup(hash, UINT2NUM(c));
05249                 if (NIL_P(tmp)) {
05250                     if (cflag) c = last;
05251                     else c = errc;
05252                 }
05253                 else if (cflag) c = errc;
05254                 else c = NUM2INT(tmp);
05255             }
05256             else {
05257                 c = errc;
05258             }
05259             if (c != (unsigned int)-1) {
05260                 if (save == c) {
05261                     CHECK_IF_ASCII(c);
05262                     continue;
05263                 }
05264                 save = c;
05265                 tlen = rb_enc_codelen(c, enc);
05266                 modify = 1;
05267             }
05268             else {
05269                 save = -1;
05270                 c = c0;
05271                 if (enc != e1) may_modify = 1;
05272             }
05273             while (t - buf + tlen >= max) {
05274                 offset = t - buf;
05275                 max *= 2;
05276                 REALLOC_N(buf, char, max);
05277                 t = buf + offset;
05278             }
05279             rb_enc_mbcput(c, t, enc);
05280             if (may_modify && memcmp(s, t, tlen) != 0) {
05281                 modify = 1;
05282             }
05283             CHECK_IF_ASCII(c);
05284             t += tlen;
05285         }
05286         if (!STR_EMBED_P(str)) {
05287             xfree(RSTRING(str)->as.heap.ptr);
05288         }
05289         *t = '\0';
05290         RSTRING(str)->as.heap.ptr = buf;
05291         RSTRING(str)->as.heap.len = t - buf;
05292         STR_SET_NOEMBED(str);
05293         RSTRING(str)->as.heap.aux.capa = max;
05294     }
05295     else if (rb_enc_mbmaxlen(enc) == 1 || (singlebyte && !hash)) {
05296         while (s < send) {
05297             c = (unsigned char)*s;
05298             if (trans[c] != errc) {
05299                 if (!cflag) {
05300                     c = trans[c];
05301                     *s = c;
05302                     modify = 1;
05303                 }
05304                 else {
05305                     *s = last;
05306                     modify = 1;
05307                 }
05308             }
05309             CHECK_IF_ASCII(c);
05310             s++;
05311         }
05312     }
05313     else {
05314         int clen, tlen, max = (int)(RSTRING_LEN(str) * 1.2);
05315         long offset;
05316         char *buf = ALLOC_N(char, max), *t = buf;
05317 
05318         while (s < send) {
05319             int may_modify = 0;
05320             c0 = c = rb_enc_codepoint_len(s, send, &clen, e1);
05321             tlen = enc == e1 ? clen : rb_enc_codelen(c, enc);
05322 
05323             if (c < 256) {
05324                 c = trans[c];
05325             }
05326             else if (hash) {
05327                 VALUE tmp = rb_hash_lookup(hash, UINT2NUM(c));
05328                 if (NIL_P(tmp)) {
05329                     if (cflag) c = last;
05330                     else c = errc;
05331                 }
05332                 else if (cflag) c = errc;
05333                 else c = NUM2INT(tmp);
05334             }
05335             else {
05336                 c = cflag ? last : errc;
05337             }
05338             if (c != errc) {
05339                 tlen = rb_enc_codelen(c, enc);
05340                 modify = 1;
05341             }
05342             else {
05343                 c = c0;
05344                 if (enc != e1) may_modify = 1;
05345             }
05346             while (t - buf + tlen >= max) {
05347                 offset = t - buf;
05348                 max *= 2;
05349                 REALLOC_N(buf, char, max);
05350                 t = buf + offset;
05351             }
05352             if (s != t) {
05353                 rb_enc_mbcput(c, t, enc);
05354                 if (may_modify && memcmp(s, t, tlen) != 0) {
05355                     modify = 1;
05356                 }
05357             }
05358             CHECK_IF_ASCII(c);
05359             s += clen;
05360             t += tlen;
05361         }
05362         if (!STR_EMBED_P(str)) {
05363             xfree(RSTRING(str)->as.heap.ptr);
05364         }
05365         *t = '\0';
05366         RSTRING(str)->as.heap.ptr = buf;
05367         RSTRING(str)->as.heap.len = t - buf;
05368         STR_SET_NOEMBED(str);
05369         RSTRING(str)->as.heap.aux.capa = max;
05370     }
05371 
05372     if (modify) {
05373         if (cr != ENC_CODERANGE_BROKEN)
05374             ENC_CODERANGE_SET(str, cr);
05375         rb_enc_associate(str, enc);
05376         return str;
05377     }
05378     return Qnil;
05379 }
05380 
05381 
05382 /*
05383  *  call-seq:
05384  *     str.tr!(from_str, to_str)   -> str or nil
05385  *
05386  *  Translates <i>str</i> in place, using the same rules as
05387  *  <code>String#tr</code>. Returns <i>str</i>, or <code>nil</code> if no
05388  *  changes were made.
05389  */
05390 
05391 static VALUE
05392 rb_str_tr_bang(VALUE str, VALUE src, VALUE repl)
05393 {
05394     return tr_trans(str, src, repl, 0);
05395 }
05396 
05397 
05398 /*
05399  *  call-seq:
05400  *     str.tr(from_str, to_str)   => new_str
05401  *
05402  *  Returns a copy of +str+ with the characters in +from_str+ replaced by the
05403  *  corresponding characters in +to_str+.  If +to_str+ is shorter than
05404  *  +from_str+, it is padded with its last character in order to maintain the
05405  *  correspondence.
05406  *
05407  *     "hello".tr('el', 'ip')      #=> "hippo"
05408  *     "hello".tr('aeiou', '*')    #=> "h*ll*"
05409  *     "hello".tr('aeiou', 'AA*')  #=> "hAll*"
05410  *
05411  *  Both strings may use the <code>c1-c2</code> notation to denote ranges of
05412  *  characters, and +from_str+ may start with a <code>^</code>, which denotes
05413  *  all characters except those listed.
05414  *
05415  *     "hello".tr('a-y', 'b-z')    #=> "ifmmp"
05416  *     "hello".tr('^aeiou', '*')   #=> "*e**o"
05417  *
05418  *  The backslash character <code></code> can be used to escape
05419  *  <code>^</code> or <code>-</code> and is otherwise ignored unless it
05420  *  appears at the end of a range or the end of the +from_str+ or +to_str+:
05421  *
05422  *     "hello^world".tr("\\^aeiou", "*") #=> "h*ll**w*rld"
05423  *     "hello-world".tr("a\\-eo", "*")   #=> "h*ll**w*rld"
05424  *
05425  *     "hello\r\nworld".tr("\r", "")   #=> "hello\nworld"
05426  *     "hello\r\nworld".tr("\\r", "")  #=> "hello\r\nwold"
05427  *     "hello\r\nworld".tr("\\\r", "") #=> "hello\nworld"
05428  *
05429  *     "X['\\b']".tr("X\\", "")   #=> "['b']"
05430  *     "X['\\b']".tr("X-\\]", "") #=> "'b'"
05431  */
05432 
05433 static VALUE
05434 rb_str_tr(VALUE str, VALUE src, VALUE repl)
05435 {
05436     str = rb_str_dup(str);
05437     tr_trans(str, src, repl, 0);
05438     return str;
05439 }
05440 
05441 #define TR_TABLE_SIZE 257
05442 static void
05443 tr_setup_table(VALUE str, char stable[TR_TABLE_SIZE], int first,
05444                VALUE *tablep, VALUE *ctablep, rb_encoding *enc)
05445 {
05446     const unsigned int errc = -1;
05447     char buf[256];
05448     struct tr tr;
05449     unsigned int c;
05450     VALUE table = 0, ptable = 0;
05451     int i, l, cflag = 0;
05452 
05453     tr.p = RSTRING_PTR(str); tr.pend = tr.p + RSTRING_LEN(str);
05454     tr.gen = tr.now = tr.max = 0;
05455 
05456     if (RSTRING_LEN(str) > 1 && rb_enc_ascget(tr.p, tr.pend, &l, enc) == '^') {
05457         cflag = 1;
05458         tr.p += l;
05459     }
05460     if (first) {
05461         for (i=0; i<256; i++) {
05462             stable[i] = 1;
05463         }
05464         stable[256] = cflag;
05465     }
05466     else if (stable[256] && !cflag) {
05467         stable[256] = 0;
05468     }
05469     for (i=0; i<256; i++) {
05470         buf[i] = cflag;
05471     }
05472 
05473     while ((c = trnext(&tr, enc)) != errc) {
05474         if (c < 256) {
05475             buf[c & 0xff] = !cflag;
05476         }
05477         else {
05478             VALUE key = UINT2NUM(c);
05479 
05480             if (!table && (first || *tablep || stable[256])) {
05481                 if (cflag) {
05482                     ptable = *ctablep;
05483                     table = ptable ? ptable : rb_hash_new();
05484                     *ctablep = table;
05485                 }
05486                 else {
05487                     table = rb_hash_new();
05488                     ptable = *tablep;
05489                     *tablep = table;
05490                 }
05491             }
05492             if (table && (!ptable || (cflag ^ !NIL_P(rb_hash_aref(ptable, key))))) {
05493                 rb_hash_aset(table, key, Qtrue);
05494             }
05495         }
05496     }
05497     for (i=0; i<256; i++) {
05498         stable[i] = stable[i] && buf[i];
05499     }
05500     if (!table && !cflag) {
05501         *tablep = 0;
05502     }
05503 }
05504 
05505 
05506 static int
05507 tr_find(unsigned int c, char table[TR_TABLE_SIZE], VALUE del, VALUE nodel)
05508 {
05509     if (c < 256) {
05510         return table[c] != 0;
05511     }
05512     else {
05513         VALUE v = UINT2NUM(c);
05514 
05515         if (del) {
05516             if (!NIL_P(rb_hash_lookup(del, v)) &&
05517                     (!nodel || NIL_P(rb_hash_lookup(nodel, v)))) {
05518                 return TRUE;
05519             }
05520         }
05521         else if (nodel && !NIL_P(rb_hash_lookup(nodel, v))) {
05522             return FALSE;
05523         }
05524         return table[256] ? TRUE : FALSE;
05525     }
05526 }
05527 
05528 /*
05529  *  call-seq:
05530  *     str.delete!([other_str]+)   -> str or nil
05531  *
05532  *  Performs a <code>delete</code> operation in place, returning <i>str</i>, or
05533  *  <code>nil</code> if <i>str</i> was not modified.
05534  */
05535 
05536 static VALUE
05537 rb_str_delete_bang(int argc, VALUE *argv, VALUE str)
05538 {
05539     char squeez[TR_TABLE_SIZE];
05540     rb_encoding *enc = 0;
05541     char *s, *send, *t;
05542     VALUE del = 0, nodel = 0;
05543     int modify = 0;
05544     int i, ascompat, cr;
05545 
05546     if (RSTRING_LEN(str) == 0 || !RSTRING_PTR(str)) return Qnil;
05547     rb_check_arity(argc, 1, UNLIMITED_ARGUMENTS);
05548     for (i=0; i<argc; i++) {
05549         VALUE s = argv[i];
05550 
05551         StringValue(s);
05552         enc = rb_enc_check(str, s);
05553         tr_setup_table(s, squeez, i==0, &del, &nodel, enc);
05554     }
05555 
05556     str_modify_keep_cr(str);
05557     ascompat = rb_enc_asciicompat(enc);
05558     s = t = RSTRING_PTR(str);
05559     send = RSTRING_END(str);
05560     cr = ascompat ? ENC_CODERANGE_7BIT : ENC_CODERANGE_VALID;
05561     while (s < send) {
05562         unsigned int c;
05563         int clen;
05564 
05565         if (ascompat && (c = *(unsigned char*)s) < 0x80) {
05566             if (squeez[c]) {
05567                 modify = 1;
05568             }
05569             else {
05570                 if (t != s) *t = c;
05571                 t++;
05572             }
05573             s++;
05574         }
05575         else {
05576             c = rb_enc_codepoint_len(s, send, &clen, enc);
05577 
05578             if (tr_find(c, squeez, del, nodel)) {
05579                 modify = 1;
05580             }
05581             else {
05582                 if (t != s) rb_enc_mbcput(c, t, enc);
05583                 t += clen;
05584                 if (cr == ENC_CODERANGE_7BIT) cr = ENC_CODERANGE_VALID;
05585             }
05586             s += clen;
05587         }
05588     }
05589     *t = '\0';
05590     STR_SET_LEN(str, t - RSTRING_PTR(str));
05591     ENC_CODERANGE_SET(str, cr);
05592 
05593     if (modify) return str;
05594     return Qnil;
05595 }
05596 
05597 
05598 /*
05599  *  call-seq:
05600  *     str.delete([other_str]+)   -> new_str
05601  *
05602  *  Returns a copy of <i>str</i> with all characters in the intersection of its
05603  *  arguments deleted. Uses the same rules for building the set of characters as
05604  *  <code>String#count</code>.
05605  *
05606  *     "hello".delete "l","lo"        #=> "heo"
05607  *     "hello".delete "lo"            #=> "he"
05608  *     "hello".delete "aeiou", "^e"   #=> "hell"
05609  *     "hello".delete "ej-m"          #=> "ho"
05610  */
05611 
05612 static VALUE
05613 rb_str_delete(int argc, VALUE *argv, VALUE str)
05614 {
05615     str = rb_str_dup(str);
05616     rb_str_delete_bang(argc, argv, str);
05617     return str;
05618 }
05619 
05620 
05621 /*
05622  *  call-seq:
05623  *     str.squeeze!([other_str]*)   -> str or nil
05624  *
05625  *  Squeezes <i>str</i> in place, returning either <i>str</i>, or
05626  *  <code>nil</code> if no changes were made.
05627  */
05628 
05629 static VALUE
05630 rb_str_squeeze_bang(int argc, VALUE *argv, VALUE str)
05631 {
05632     char squeez[TR_TABLE_SIZE];
05633     rb_encoding *enc = 0;
05634     VALUE del = 0, nodel = 0;
05635     char *s, *send, *t;
05636     int i, modify = 0;
05637     int ascompat, singlebyte = single_byte_optimizable(str);
05638     unsigned int save;
05639 
05640     if (argc == 0) {
05641         enc = STR_ENC_GET(str);
05642     }
05643     else {
05644         for (i=0; i<argc; i++) {
05645             VALUE s = argv[i];
05646 
05647             StringValue(s);
05648             enc = rb_enc_check(str, s);
05649             if (singlebyte && !single_byte_optimizable(s))
05650                 singlebyte = 0;
05651             tr_setup_table(s, squeez, i==0, &del, &nodel, enc);
05652         }
05653     }
05654 
05655     str_modify_keep_cr(str);
05656     s = t = RSTRING_PTR(str);
05657     if (!s || RSTRING_LEN(str) == 0) return Qnil;
05658     send = RSTRING_END(str);
05659     save = -1;
05660     ascompat = rb_enc_asciicompat(enc);
05661 
05662     if (singlebyte) {
05663         while (s < send) {
05664             unsigned int c = *(unsigned char*)s++;
05665             if (c != save || (argc > 0 && !squeez[c])) {
05666                 *t++ = save = c;
05667             }
05668         }
05669     } else {
05670         while (s < send) {
05671             unsigned int c;
05672             int clen;
05673 
05674             if (ascompat && (c = *(unsigned char*)s) < 0x80) {
05675                 if (c != save || (argc > 0 && !squeez[c])) {
05676                     *t++ = save = c;
05677                 }
05678                 s++;
05679             }
05680             else {
05681                 c = rb_enc_codepoint_len(s, send, &clen, enc);
05682 
05683                 if (c != save || (argc > 0 && !tr_find(c, squeez, del, nodel))) {
05684                     if (t != s) rb_enc_mbcput(c, t, enc);
05685                     save = c;
05686                     t += clen;
05687                 }
05688                 s += clen;
05689             }
05690         }
05691     }
05692 
05693     *t = '\0';
05694     if (t - RSTRING_PTR(str) != RSTRING_LEN(str)) {
05695         STR_SET_LEN(str, t - RSTRING_PTR(str));
05696         modify = 1;
05697     }
05698 
05699     if (modify) return str;
05700     return Qnil;
05701 }
05702 
05703 
05704 /*
05705  *  call-seq:
05706  *     str.squeeze([other_str]*)    -> new_str
05707  *
05708  *  Builds a set of characters from the <i>other_str</i> parameter(s) using the
05709  *  procedure described for <code>String#count</code>. Returns a new string
05710  *  where runs of the same character that occur in this set are replaced by a
05711  *  single character. If no arguments are given, all runs of identical
05712  *  characters are replaced by a single character.
05713  *
05714  *     "yellow moon".squeeze                  #=> "yelow mon"
05715  *     "  now   is  the".squeeze(" ")         #=> " now is the"
05716  *     "putters shoot balls".squeeze("m-z")   #=> "puters shot balls"
05717  */
05718 
05719 static VALUE
05720 rb_str_squeeze(int argc, VALUE *argv, VALUE str)
05721 {
05722     str = rb_str_dup(str);
05723     rb_str_squeeze_bang(argc, argv, str);
05724     return str;
05725 }
05726 
05727 
05728 /*
05729  *  call-seq:
05730  *     str.tr_s!(from_str, to_str)   -> str or nil
05731  *
05732  *  Performs <code>String#tr_s</code> processing on <i>str</i> in place,
05733  *  returning <i>str</i>, or <code>nil</code> if no changes were made.
05734  */
05735 
05736 static VALUE
05737 rb_str_tr_s_bang(VALUE str, VALUE src, VALUE repl)
05738 {
05739     return tr_trans(str, src, repl, 1);
05740 }
05741 
05742 
05743 /*
05744  *  call-seq:
05745  *     str.tr_s(from_str, to_str)   -> new_str
05746  *
05747  *  Processes a copy of <i>str</i> as described under <code>String#tr</code>,
05748  *  then removes duplicate characters in regions that were affected by the
05749  *  translation.
05750  *
05751  *     "hello".tr_s('l', 'r')     #=> "hero"
05752  *     "hello".tr_s('el', '*')    #=> "h*o"
05753  *     "hello".tr_s('el', 'hx')   #=> "hhxo"
05754  */
05755 
05756 static VALUE
05757 rb_str_tr_s(VALUE str, VALUE src, VALUE repl)
05758 {
05759     str = rb_str_dup(str);
05760     tr_trans(str, src, repl, 1);
05761     return str;
05762 }
05763 
05764 
05765 /*
05766  *  call-seq:
05767  *     str.count([other_str]+)   -> fixnum
05768  *
05769  *  Each +other_str+ parameter defines a set of characters to count.  The
05770  *  intersection of these sets defines the characters to count in +str+.  Any
05771  *  +other_str+ that starts with a caret <code>^</code> is negated.  The
05772  *  sequence <code>c1-c2</code> means all characters between c1 and c2.  The
05773  *  backslash character <code></code> can be used to escape <code>^</code> or
05774  *  <code>-</code> and is otherwise ignored unless it appears at the end of a
05775  *  sequence or the end of a +other_str+.
05776  *
05777  *     a = "hello world"
05778  *     a.count "lo"                   #=> 5
05779  *     a.count "lo", "o"              #=> 2
05780  *     a.count "hello", "^l"          #=> 4
05781  *     a.count "ej-m"                 #=> 4
05782  *
05783  *     "hello^world".count "\\^aeiou" #=> 4
05784  *     "hello-world".count "a\\-eo"   #=> 4
05785  *
05786  *     c = "hello world\\r\\n"
05787  *     c.count "\\"                   #=> 2
05788  *     c.count "\\A"                  #=> 0
05789  *     c.count "X-\\w"                #=> 3
05790  */
05791 
05792 static VALUE
05793 rb_str_count(int argc, VALUE *argv, VALUE str)
05794 {
05795     char table[TR_TABLE_SIZE];
05796     rb_encoding *enc = 0;
05797     VALUE del = 0, nodel = 0, tstr;
05798     char *s, *send;
05799     int i;
05800     int ascompat;
05801 
05802     rb_check_arity(argc, 1, UNLIMITED_ARGUMENTS);
05803 
05804     tstr = argv[0];
05805     StringValue(tstr);
05806     enc = rb_enc_check(str, tstr);
05807     if (argc == 1) {
05808         const char *ptstr;
05809         if (RSTRING_LEN(tstr) == 1 && rb_enc_asciicompat(enc) &&
05810             (ptstr = RSTRING_PTR(tstr),
05811              ONIGENC_IS_ALLOWED_REVERSE_MATCH(enc, (const unsigned char *)ptstr, (const unsigned char *)ptstr+1)) &&
05812             !is_broken_string(str)) {
05813             int n = 0;
05814             int clen;
05815             unsigned char c = rb_enc_codepoint_len(ptstr, ptstr+1, &clen, enc);
05816 
05817             s = RSTRING_PTR(str);
05818             if (!s || RSTRING_LEN(str) == 0) return INT2FIX(0);
05819             send = RSTRING_END(str);
05820             while (s < send) {
05821                 if (*(unsigned char*)s++ == c) n++;
05822             }
05823             return INT2NUM(n);
05824         }
05825     }
05826 
05827     tr_setup_table(tstr, table, TRUE, &del, &nodel, enc);
05828     for (i=1; i<argc; i++) {
05829         tstr = argv[i];
05830         StringValue(tstr);
05831         enc = rb_enc_check(str, tstr);
05832         tr_setup_table(tstr, table, FALSE, &del, &nodel, enc);
05833     }
05834 
05835     s = RSTRING_PTR(str);
05836     if (!s || RSTRING_LEN(str) == 0) return INT2FIX(0);
05837     send = RSTRING_END(str);
05838     ascompat = rb_enc_asciicompat(enc);
05839     i = 0;
05840     while (s < send) {
05841         unsigned int c;
05842 
05843         if (ascompat && (c = *(unsigned char*)s) < 0x80) {
05844             if (table[c]) {
05845                 i++;
05846             }
05847             s++;
05848         }
05849         else {
05850             int clen;
05851             c = rb_enc_codepoint_len(s, send, &clen, enc);
05852             if (tr_find(c, table, del, nodel)) {
05853                 i++;
05854             }
05855             s += clen;
05856         }
05857     }
05858 
05859     return INT2NUM(i);
05860 }
05861 
05862 static const char isspacetable[256] = {
05863     0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 0, 0,
05864     0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
05865     1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
05866     0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
05867     0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
05868     0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
05869     0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
05870     0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
05871     0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
05872     0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
05873     0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
05874     0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
05875     0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
05876     0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
05877     0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
05878     0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0
05879 };
05880 
05881 #define ascii_isspace(c) isspacetable[(unsigned char)(c)]
05882 
05883 /*
05884  *  call-seq:
05885  *     str.split(pattern=$;, [limit])   -> anArray
05886  *
05887  *  Divides <i>str</i> into substrings based on a delimiter, returning an array
05888  *  of these substrings.
05889  *
05890  *  If <i>pattern</i> is a <code>String</code>, then its contents are used as
05891  *  the delimiter when splitting <i>str</i>. If <i>pattern</i> is a single
05892  *  space, <i>str</i> is split on whitespace, with leading whitespace and runs
05893  *  of contiguous whitespace characters ignored.
05894  *
05895  *  If <i>pattern</i> is a <code>Regexp</code>, <i>str</i> is divided where the
05896  *  pattern matches. Whenever the pattern matches a zero-length string,
05897  *  <i>str</i> is split into individual characters. If <i>pattern</i> contains
05898  *  groups, the respective matches will be returned in the array as well.
05899  *
05900  *  If <i>pattern</i> is omitted, the value of <code>$;</code> is used.  If
05901  *  <code>$;</code> is <code>nil</code> (which is the default), <i>str</i> is
05902  *  split on whitespace as if ` ' were specified.
05903  *
05904  *  If the <i>limit</i> parameter is omitted, trailing null fields are
05905  *  suppressed. If <i>limit</i> is a positive number, at most that number of
05906  *  fields will be returned (if <i>limit</i> is <code>1</code>, the entire
05907  *  string is returned as the only entry in an array). If negative, there is no
05908  *  limit to the number of fields returned, and trailing null fields are not
05909  *  suppressed.
05910  *
05911  *  When the input +str+ is empty an empty Array is returned as the string is
05912  *  considered to have no fields to split.
05913  *
05914  *     " now's  the time".split        #=> ["now's", "the", "time"]
05915  *     " now's  the time".split(' ')   #=> ["now's", "the", "time"]
05916  *     " now's  the time".split(/ /)   #=> ["", "now's", "", "the", "time"]
05917  *     "1, 2.34,56, 7".split(%r{,\s*}) #=> ["1", "2.34", "56", "7"]
05918  *     "hello".split(//)               #=> ["h", "e", "l", "l", "o"]
05919  *     "hello".split(//, 3)            #=> ["h", "e", "llo"]
05920  *     "hi mom".split(%r{\s*})         #=> ["h", "i", "m", "o", "m"]
05921  *
05922  *     "mellow yellow".split("ello")   #=> ["m", "w y", "w"]
05923  *     "1,2,,3,4,,".split(',')         #=> ["1", "2", "", "3", "4"]
05924  *     "1,2,,3,4,,".split(',', 4)      #=> ["1", "2", "", "3,4,,"]
05925  *     "1,2,,3,4,,".split(',', -4)     #=> ["1", "2", "", "3", "4", "", ""]
05926  *
05927  *     "".split(',', -1)               #=> []
05928  */
05929 
05930 static VALUE
05931 rb_str_split_m(int argc, VALUE *argv, VALUE str)
05932 {
05933     rb_encoding *enc;
05934     VALUE spat;
05935     VALUE limit;
05936     enum {awk, string, regexp} split_type;
05937     long beg, end, i = 0;
05938     int lim = 0;
05939     VALUE result, tmp;
05940 
05941     if (rb_scan_args(argc, argv, "02", &spat, &limit) == 2) {
05942         lim = NUM2INT(limit);
05943         if (lim <= 0) limit = Qnil;
05944         else if (lim == 1) {
05945             if (RSTRING_LEN(str) == 0)
05946                 return rb_ary_new2(0);
05947             return rb_ary_new3(1, str);
05948         }
05949         i = 1;
05950     }
05951 
05952     enc = STR_ENC_GET(str);
05953     if (NIL_P(spat)) {
05954         if (!NIL_P(rb_fs)) {
05955             spat = rb_fs;
05956             goto fs_set;
05957         }
05958         split_type = awk;
05959     }
05960     else {
05961       fs_set:
05962         if (RB_TYPE_P(spat, T_STRING)) {
05963             rb_encoding *enc2 = STR_ENC_GET(spat);
05964 
05965             split_type = string;
05966             if (RSTRING_LEN(spat) == 0) {
05967                 /* Special case - split into chars */
05968                 spat = rb_reg_regcomp(spat);
05969                 split_type = regexp;
05970             }
05971             else if (rb_enc_asciicompat(enc2) == 1) {
05972                 if (RSTRING_LEN(spat) == 1 && RSTRING_PTR(spat)[0] == ' '){
05973                     split_type = awk;
05974                 }
05975             }
05976             else {
05977                 int l;
05978                 if (rb_enc_ascget(RSTRING_PTR(spat), RSTRING_END(spat), &l, enc2) == ' ' &&
05979                     RSTRING_LEN(spat) == l) {
05980                     split_type = awk;
05981                 }
05982             }
05983         }
05984         else {
05985             spat = get_pat(spat, 1);
05986             split_type = regexp;
05987         }
05988     }
05989 
05990     result = rb_ary_new();
05991     beg = 0;
05992     if (split_type == awk) {
05993         char *ptr = RSTRING_PTR(str);
05994         char *eptr = RSTRING_END(str);
05995         char *bptr = ptr;
05996         int skip = 1;
05997         unsigned int c;
05998 
05999         end = beg;
06000         if (is_ascii_string(str)) {
06001             while (ptr < eptr) {
06002                 c = (unsigned char)*ptr++;
06003                 if (skip) {
06004                     if (ascii_isspace(c)) {
06005                         beg = ptr - bptr;
06006                     }
06007                     else {
06008                         end = ptr - bptr;
06009                         skip = 0;
06010                         if (!NIL_P(limit) && lim <= i) break;
06011                     }
06012                 }
06013                 else if (ascii_isspace(c)) {
06014                     rb_ary_push(result, rb_str_subseq(str, beg, end-beg));
06015                     skip = 1;
06016                     beg = ptr - bptr;
06017                     if (!NIL_P(limit)) ++i;
06018                 }
06019                 else {
06020                     end = ptr - bptr;
06021                 }
06022             }
06023         }
06024         else {
06025             while (ptr < eptr) {
06026                 int n;
06027 
06028                 c = rb_enc_codepoint_len(ptr, eptr, &n, enc);
06029                 ptr += n;
06030                 if (skip) {
06031                     if (rb_isspace(c)) {
06032                         beg = ptr - bptr;
06033                     }
06034                     else {
06035                         end = ptr - bptr;
06036                         skip = 0;
06037                         if (!NIL_P(limit) && lim <= i) break;
06038                     }
06039                 }
06040                 else if (rb_isspace(c)) {
06041                     rb_ary_push(result, rb_str_subseq(str, beg, end-beg));
06042                     skip = 1;
06043                     beg = ptr - bptr;
06044                     if (!NIL_P(limit)) ++i;
06045                 }
06046                 else {
06047                     end = ptr - bptr;
06048                 }
06049             }
06050         }
06051     }
06052     else if (split_type == string) {
06053         char *ptr = RSTRING_PTR(str);
06054         char *temp = ptr;
06055         char *eptr = RSTRING_END(str);
06056         char *sptr = RSTRING_PTR(spat);
06057         long slen = RSTRING_LEN(spat);
06058 
06059         if (is_broken_string(str)) {
06060             rb_raise(rb_eArgError, "invalid byte sequence in %s", rb_enc_name(STR_ENC_GET(str)));
06061         }
06062         if (is_broken_string(spat)) {
06063             rb_raise(rb_eArgError, "invalid byte sequence in %s", rb_enc_name(STR_ENC_GET(spat)));
06064         }
06065         enc = rb_enc_check(str, spat);
06066         while (ptr < eptr &&
06067                (end = rb_memsearch(sptr, slen, ptr, eptr - ptr, enc)) >= 0) {
06068             /* Check we are at the start of a char */
06069             char *t = rb_enc_right_char_head(ptr, ptr + end, eptr, enc);
06070             if (t != ptr + end) {
06071                 ptr = t;
06072                 continue;
06073             }
06074             rb_ary_push(result, rb_str_subseq(str, ptr - temp, end));
06075             ptr += end + slen;
06076             if (!NIL_P(limit) && lim <= ++i) break;
06077         }
06078         beg = ptr - temp;
06079     }
06080     else {
06081         char *ptr = RSTRING_PTR(str);
06082         long len = RSTRING_LEN(str);
06083         long start = beg;
06084         long idx;
06085         int last_null = 0;
06086         struct re_registers *regs;
06087 
06088         while ((end = rb_reg_search(spat, str, start, 0)) >= 0) {
06089             regs = RMATCH_REGS(rb_backref_get());
06090             if (start == end && BEG(0) == END(0)) {
06091                 if (!ptr) {
06092                     rb_ary_push(result, str_new_empty(str));
06093                     break;
06094                 }
06095                 else if (last_null == 1) {
06096                     rb_ary_push(result, rb_str_subseq(str, beg,
06097                                                       rb_enc_fast_mbclen(ptr+beg,
06098                                                                          ptr+len,
06099                                                                          enc)));
06100                     beg = start;
06101                 }
06102                 else {
06103                     if (ptr+start == ptr+len)
06104                         start++;
06105                     else
06106                         start += rb_enc_fast_mbclen(ptr+start,ptr+len,enc);
06107                     last_null = 1;
06108                     continue;
06109                 }
06110             }
06111             else {
06112                 rb_ary_push(result, rb_str_subseq(str, beg, end-beg));
06113                 beg = start = END(0);
06114             }
06115             last_null = 0;
06116 
06117             for (idx=1; idx < regs->num_regs; idx++) {
06118                 if (BEG(idx) == -1) continue;
06119                 if (BEG(idx) == END(idx))
06120                     tmp = str_new_empty(str);
06121                 else
06122                     tmp = rb_str_subseq(str, BEG(idx), END(idx)-BEG(idx));
06123                 rb_ary_push(result, tmp);
06124             }
06125             if (!NIL_P(limit) && lim <= ++i) break;
06126         }
06127     }
06128     if (RSTRING_LEN(str) > 0 && (!NIL_P(limit) || RSTRING_LEN(str) > beg || lim < 0)) {
06129         if (RSTRING_LEN(str) == beg)
06130             tmp = str_new_empty(str);
06131         else
06132             tmp = rb_str_subseq(str, beg, RSTRING_LEN(str)-beg);
06133         rb_ary_push(result, tmp);
06134     }
06135     if (NIL_P(limit) && lim == 0) {
06136         long len;
06137         while ((len = RARRAY_LEN(result)) > 0 &&
06138                (tmp = RARRAY_PTR(result)[len-1], RSTRING_LEN(tmp) == 0))
06139             rb_ary_pop(result);
06140     }
06141 
06142     return result;
06143 }
06144 
06145 VALUE
06146 rb_str_split(VALUE str, const char *sep0)
06147 {
06148     VALUE sep;
06149 
06150     StringValue(str);
06151     sep = rb_str_new2(sep0);
06152     return rb_str_split_m(1, &sep, str);
06153 }
06154 
06155 
06156 static VALUE
06157 rb_str_enumerate_lines(int argc, VALUE *argv, VALUE str, int wantarray)
06158 {
06159     rb_encoding *enc;
06160     VALUE rs;
06161     unsigned int newline;
06162     const char *p, *pend, *s, *ptr;
06163     long len, rslen;
06164     VALUE line;
06165     int n;
06166     VALUE orig = str;
06167     VALUE UNINITIALIZED_VAR(ary);
06168 
06169     if (argc == 0) {
06170         rs = rb_rs;
06171     }
06172     else {
06173         rb_scan_args(argc, argv, "01", &rs);
06174     }
06175 
06176     if (rb_block_given_p()) {
06177         if (wantarray) {
06178 #if 0 /* next major */
06179             rb_warn("given block not used");
06180             ary = rb_ary_new();
06181 #else
06182             rb_warning("passing a block to String#lines is deprecated");
06183             wantarray = 0;
06184 #endif
06185         }
06186     }
06187     else {
06188         if (wantarray)
06189             ary = rb_ary_new();
06190         else
06191             RETURN_ENUMERATOR(str, argc, argv);
06192     }
06193 
06194     if (NIL_P(rs)) {
06195         if (wantarray) {
06196             rb_ary_push(ary, str);
06197             return ary;
06198         }
06199         else {
06200             rb_yield(str);
06201             return orig;
06202         }
06203     }
06204     str = rb_str_new4(str);
06205     ptr = p = s = RSTRING_PTR(str);
06206     pend = p + RSTRING_LEN(str);
06207     len = RSTRING_LEN(str);
06208     StringValue(rs);
06209     if (rs == rb_default_rs) {
06210         enc = rb_enc_get(str);
06211         while (p < pend) {
06212             char *p0;
06213 
06214             p = memchr(p, '\n', pend - p);
06215             if (!p) break;
06216             p0 = rb_enc_left_char_head(s, p, pend, enc);
06217             if (!rb_enc_is_newline(p0, pend, enc)) {
06218                 p++;
06219                 continue;
06220             }
06221             p = p0 + rb_enc_mbclen(p0, pend, enc);
06222             line = rb_str_subseq(str, s - ptr, p - s);
06223             if (wantarray)
06224                 rb_ary_push(ary, line);
06225             else
06226                 rb_yield(line);
06227             str_mod_check(str, ptr, len);
06228             s = p;
06229         }
06230         goto finish;
06231     }
06232 
06233     enc = rb_enc_check(str, rs);
06234     rslen = RSTRING_LEN(rs);
06235     if (rslen == 0) {
06236         newline = '\n';
06237     }
06238     else {
06239         newline = rb_enc_codepoint(RSTRING_PTR(rs), RSTRING_END(rs), enc);
06240     }
06241 
06242     while (p < pend) {
06243         unsigned int c = rb_enc_codepoint_len(p, pend, &n, enc);
06244 
06245       again:
06246         if (rslen == 0 && c == newline) {
06247             p += n;
06248             if (p < pend && (c = rb_enc_codepoint_len(p, pend, &n, enc)) != newline) {
06249                 goto again;
06250             }
06251             while (p < pend && rb_enc_codepoint(p, pend, enc) == newline) {
06252                 p += n;
06253             }
06254             p -= n;
06255         }
06256         if (c == newline &&
06257             (rslen <= 1 ||
06258              (pend - p >= rslen && memcmp(RSTRING_PTR(rs), p, rslen) == 0))) {
06259             const char *pp = p + (rslen ? rslen : n);
06260             line = rb_str_subseq(str, s - ptr, pp - s);
06261             if (wantarray)
06262                 rb_ary_push(ary, line);
06263             else
06264                 rb_yield(line);
06265             str_mod_check(str, ptr, len);
06266             s = pp;
06267         }
06268         p += n;
06269     }
06270 
06271   finish:
06272     if (s != pend) {
06273         line = rb_str_subseq(str, s - ptr, pend - s);
06274         if (wantarray)
06275             rb_ary_push(ary, line);
06276         else
06277             rb_yield(line);
06278         RB_GC_GUARD(str);
06279     }
06280 
06281     if (wantarray)
06282         return ary;
06283     else
06284         return orig;
06285 }
06286 
06287 /*
06288  *  call-seq:
06289  *     str.each_line(separator=$/) {|substr| block }   -> str
06290  *     str.each_line(separator=$/)                     -> an_enumerator
06291  *
06292  *  Splits <i>str</i> using the supplied parameter as the record
06293  *  separator (<code>$/</code> by default), passing each substring in
06294  *  turn to the supplied block.  If a zero-length record separator is
06295  *  supplied, the string is split into paragraphs delimited by
06296  *  multiple successive newlines.
06297  *
06298  *  If no block is given, an enumerator is returned instead.
06299  *
06300  *     print "Example one\n"
06301  *     "hello\nworld".each_line {|s| p s}
06302  *     print "Example two\n"
06303  *     "hello\nworld".each_line('l') {|s| p s}
06304  *     print "Example three\n"
06305  *     "hello\n\n\nworld".each_line('') {|s| p s}
06306  *
06307  *  <em>produces:</em>
06308  *
06309  *     Example one
06310  *     "hello\n"
06311  *     "world"
06312  *     Example two
06313  *     "hel"
06314  *     "l"
06315  *     "o\nworl"
06316  *     "d"
06317  *     Example three
06318  *     "hello\n\n\n"
06319  *     "world"
06320  */
06321 
06322 static VALUE
06323 rb_str_each_line(int argc, VALUE *argv, VALUE str)
06324 {
06325     return rb_str_enumerate_lines(argc, argv, str, 0);
06326 }
06327 
06328 /*
06329  *  call-seq:
06330  *     str.lines(separator=$/)  -> an_array
06331  *
06332  *  Returns an array of lines in <i>str</i> split using the supplied
06333  *  record separator (<code>$/</code> by default).  This is a
06334  *  shorthand for <code>str.each_line(separator).to_a</code>.
06335  *
06336  *  If a block is given, which is a deprecated form, works the same as
06337  *  <code>each_line</code>.
06338  */
06339 
06340 static VALUE
06341 rb_str_lines(int argc, VALUE *argv, VALUE str)
06342 {
06343     return rb_str_enumerate_lines(argc, argv, str, 1);
06344 }
06345 
06346 static VALUE
06347 rb_str_each_byte_size(VALUE str, VALUE args)
06348 {
06349     return LONG2FIX(RSTRING_LEN(str));
06350 }
06351 
06352 static VALUE
06353 rb_str_enumerate_bytes(VALUE str, int wantarray)
06354 {
06355     long i;
06356     VALUE UNINITIALIZED_VAR(ary);
06357 
06358     if (rb_block_given_p()) {
06359         if (wantarray) {
06360 #if 0 /* next major */
06361             rb_warn("given block not used");
06362             ary = rb_ary_new();
06363 #else
06364             rb_warning("passing a block to String#bytes is deprecated");
06365             wantarray = 0;
06366 #endif
06367         }
06368     }
06369     else {
06370         if (wantarray)
06371             ary = rb_ary_new2(RSTRING_LEN(str));
06372         else
06373             RETURN_SIZED_ENUMERATOR(str, 0, 0, rb_str_each_byte_size);
06374     }
06375 
06376     for (i=0; i<RSTRING_LEN(str); i++) {
06377         if (wantarray)
06378             rb_ary_push(ary, INT2FIX(RSTRING_PTR(str)[i] & 0xff));
06379         else
06380             rb_yield(INT2FIX(RSTRING_PTR(str)[i] & 0xff));
06381     }
06382     if (wantarray)
06383         return ary;
06384     else
06385         return str;
06386 }
06387 
06388 /*
06389  *  call-seq:
06390  *     str.each_byte {|fixnum| block }    -> str
06391  *     str.each_byte                      -> an_enumerator
06392  *
06393  *  Passes each byte in <i>str</i> to the given block, or returns an
06394  *  enumerator if no block is given.
06395  *
06396  *     "hello".each_byte {|c| print c, ' ' }
06397  *
06398  *  <em>produces:</em>
06399  *
06400  *     104 101 108 108 111
06401  */
06402 
06403 static VALUE
06404 rb_str_each_byte(VALUE str)
06405 {
06406     return rb_str_enumerate_bytes(str, 0);
06407 }
06408 
06409 /*
06410  *  call-seq:
06411  *     str.bytes    -> an_array
06412  *
06413  *  Returns an array of bytes in <i>str</i>.  This is a shorthand for
06414  *  <code>str.each_byte.to_a</code>.
06415  *
06416  *  If a block is given, which is a deprecated form, works the same as
06417  *  <code>each_byte</code>.
06418  */
06419 
06420 static VALUE
06421 rb_str_bytes(VALUE str)
06422 {
06423     return rb_str_enumerate_bytes(str, 1);
06424 }
06425 
06426 static VALUE
06427 rb_str_each_char_size(VALUE str)
06428 {
06429     long len = RSTRING_LEN(str);
06430     if (!single_byte_optimizable(str)) {
06431         const char *ptr = RSTRING_PTR(str);
06432         rb_encoding *enc = rb_enc_get(str);
06433         const char *end_ptr = ptr + len;
06434         for (len = 0; ptr < end_ptr; ++len) {
06435             ptr += rb_enc_mbclen(ptr, end_ptr, enc);
06436         }
06437     }
06438     return LONG2FIX(len);
06439 }
06440 
06441 static VALUE
06442 rb_str_enumerate_chars(VALUE str, int wantarray)
06443 {
06444     VALUE orig = str;
06445     VALUE substr;
06446     long i, len, n;
06447     const char *ptr;
06448     rb_encoding *enc;
06449     VALUE UNINITIALIZED_VAR(ary);
06450 
06451     if (rb_block_given_p()) {
06452         if (wantarray) {
06453 #if 0 /* next major */
06454             rb_warn("given block not used");
06455             ary = rb_ary_new();
06456 #else
06457             rb_warning("passing a block to String#chars is deprecated");
06458             wantarray = 0;
06459 #endif
06460         }
06461     }
06462     else {
06463         if (wantarray)
06464             ary = rb_ary_new();
06465         else
06466             RETURN_SIZED_ENUMERATOR(str, 0, 0, rb_str_each_char_size);
06467     }
06468 
06469     str = rb_str_new4(str);
06470     ptr = RSTRING_PTR(str);
06471     len = RSTRING_LEN(str);
06472     enc = rb_enc_get(str);
06473     switch (ENC_CODERANGE(str)) {
06474       case ENC_CODERANGE_VALID:
06475       case ENC_CODERANGE_7BIT:
06476         for (i = 0; i < len; i += n) {
06477             n = rb_enc_fast_mbclen(ptr + i, ptr + len, enc);
06478             substr = rb_str_subseq(str, i, n);
06479             if (wantarray)
06480                 rb_ary_push(ary, substr);
06481             else
06482                 rb_yield(substr);
06483         }
06484         break;
06485       default:
06486         for (i = 0; i < len; i += n) {
06487             n = rb_enc_mbclen(ptr + i, ptr + len, enc);
06488             substr = rb_str_subseq(str, i, n);
06489             if (wantarray)
06490                 rb_ary_push(ary, substr);
06491             else
06492                 rb_yield(substr);
06493         }
06494     }
06495     RB_GC_GUARD(str);
06496     if (wantarray)
06497         return ary;
06498     else
06499         return orig;
06500 }
06501 
06502 /*
06503  *  call-seq:
06504  *     str.each_char {|cstr| block }    -> str
06505  *     str.each_char                    -> an_enumerator
06506  *
06507  *  Passes each character in <i>str</i> to the given block, or returns
06508  *  an enumerator if no block is given.
06509  *
06510  *     "hello".each_char {|c| print c, ' ' }
06511  *
06512  *  <em>produces:</em>
06513  *
06514  *     h e l l o
06515  */
06516 
06517 static VALUE
06518 rb_str_each_char(VALUE str)
06519 {
06520     return rb_str_enumerate_chars(str, 0);
06521 }
06522 
06523 /*
06524  *  call-seq:
06525  *     str.chars    -> an_array
06526  *
06527  *  Returns an array of characters in <i>str</i>.  This is a shorthand
06528  *  for <code>str.each_char.to_a</code>.
06529  *
06530  *  If a block is given, which is a deprecated form, works the same as
06531  *  <code>each_char</code>.
06532  */
06533 
06534 static VALUE
06535 rb_str_chars(VALUE str)
06536 {
06537     return rb_str_enumerate_chars(str, 1);
06538 }
06539 
06540 
06541 static VALUE
06542 rb_str_enumerate_codepoints(VALUE str, int wantarray)
06543 {
06544     VALUE orig = str;
06545     int n;
06546     unsigned int c;
06547     const char *ptr, *end;
06548     rb_encoding *enc;
06549     VALUE UNINITIALIZED_VAR(ary);
06550 
06551     if (single_byte_optimizable(str))
06552         return rb_str_enumerate_bytes(str, wantarray);
06553 
06554     if (rb_block_given_p()) {
06555         if (wantarray) {
06556 #if 0 /* next major */
06557             rb_warn("given block not used");
06558             ary = rb_ary_new();
06559 #else
06560             rb_warning("passing a block to String#codepoints is deprecated");
06561             wantarray = 0;
06562 #endif
06563         }
06564     }
06565     else {
06566         if (wantarray)
06567             ary = rb_ary_new();
06568         else
06569             RETURN_SIZED_ENUMERATOR(str, 0, 0, rb_str_each_char_size);
06570     }
06571 
06572     str = rb_str_new4(str);
06573     ptr = RSTRING_PTR(str);
06574     end = RSTRING_END(str);
06575     enc = STR_ENC_GET(str);
06576     while (ptr < end) {
06577         c = rb_enc_codepoint_len(ptr, end, &n, enc);
06578         if (wantarray)
06579             rb_ary_push(ary, UINT2NUM(c));
06580         else
06581             rb_yield(UINT2NUM(c));
06582         ptr += n;
06583     }
06584     RB_GC_GUARD(str);
06585     if (wantarray)
06586         return ary;
06587     else
06588         return orig;
06589 }
06590 
06591 /*
06592  *  call-seq:
06593  *     str.each_codepoint {|integer| block }    -> str
06594  *     str.each_codepoint                       -> an_enumerator
06595  *
06596  *  Passes the <code>Integer</code> ordinal of each character in <i>str</i>,
06597  *  also known as a <i>codepoint</i> when applied to Unicode strings to the
06598  *  given block.
06599  *
06600  *  If no block is given, an enumerator is returned instead.
06601  *
06602  *     "hello\u0639".each_codepoint {|c| print c, ' ' }
06603  *
06604  *  <em>produces:</em>
06605  *
06606  *     104 101 108 108 111 1593
06607  */
06608 
06609 static VALUE
06610 rb_str_each_codepoint(VALUE str)
06611 {
06612     return rb_str_enumerate_codepoints(str, 0);
06613 }
06614 
06615 /*
06616  *  call-seq:
06617  *     str.codepoints   -> an_array
06618  *
06619  *  Returns an array of the <code>Integer</code> ordinals of the
06620  *  characters in <i>str</i>.  This is a shorthand for
06621  *  <code>str.each_codepoint.to_a</code>.
06622  *
06623  *  If a block is given, which is a deprecated form, works the same as
06624  *  <code>each_codepoint</code>.
06625  */
06626 
06627 static VALUE
06628 rb_str_codepoints(VALUE str)
06629 {
06630     return rb_str_enumerate_codepoints(str, 1);
06631 }
06632 
06633 
06634 static long
06635 chopped_length(VALUE str)
06636 {
06637     rb_encoding *enc = STR_ENC_GET(str);
06638     const char *p, *p2, *beg, *end;
06639 
06640     beg = RSTRING_PTR(str);
06641     end = beg + RSTRING_LEN(str);
06642     if (beg > end) return 0;
06643     p = rb_enc_prev_char(beg, end, end, enc);
06644     if (!p) return 0;
06645     if (p > beg && rb_enc_ascget(p, end, 0, enc) == '\n') {
06646         p2 = rb_enc_prev_char(beg, p, end, enc);
06647         if (p2 && rb_enc_ascget(p2, end, 0, enc) == '\r') p = p2;
06648     }
06649     return p - beg;
06650 }
06651 
06652 /*
06653  *  call-seq:
06654  *     str.chop!   -> str or nil
06655  *
06656  *  Processes <i>str</i> as for <code>String#chop</code>, returning <i>str</i>,
06657  *  or <code>nil</code> if <i>str</i> is the empty string.  See also
06658  *  <code>String#chomp!</code>.
06659  */
06660 
06661 static VALUE
06662 rb_str_chop_bang(VALUE str)
06663 {
06664     str_modify_keep_cr(str);
06665     if (RSTRING_LEN(str) > 0) {
06666         long len;
06667         len = chopped_length(str);
06668         STR_SET_LEN(str, len);
06669         RSTRING_PTR(str)[len] = '\0';
06670         if (ENC_CODERANGE(str) != ENC_CODERANGE_7BIT) {
06671             ENC_CODERANGE_CLEAR(str);
06672         }
06673         return str;
06674     }
06675     return Qnil;
06676 }
06677 
06678 
06679 /*
06680  *  call-seq:
06681  *     str.chop   -> new_str
06682  *
06683  *  Returns a new <code>String</code> with the last character removed.  If the
06684  *  string ends with <code>\r\n</code>, both characters are removed. Applying
06685  *  <code>chop</code> to an empty string returns an empty
06686  *  string. <code>String#chomp</code> is often a safer alternative, as it leaves
06687  *  the string unchanged if it doesn't end in a record separator.
06688  *
06689  *     "string\r\n".chop   #=> "string"
06690  *     "string\n\r".chop   #=> "string\n"
06691  *     "string\n".chop     #=> "string"
06692  *     "string".chop       #=> "strin"
06693  *     "x".chop.chop       #=> ""
06694  */
06695 
06696 static VALUE
06697 rb_str_chop(VALUE str)
06698 {
06699     return rb_str_subseq(str, 0, chopped_length(str));
06700 }
06701 
06702 
06703 /*
06704  *  call-seq:
06705  *     str.chomp!(separator=$/)   -> str or nil
06706  *
06707  *  Modifies <i>str</i> in place as described for <code>String#chomp</code>,
06708  *  returning <i>str</i>, or <code>nil</code> if no modifications were made.
06709  */
06710 
06711 static VALUE
06712 rb_str_chomp_bang(int argc, VALUE *argv, VALUE str)
06713 {
06714     rb_encoding *enc;
06715     VALUE rs;
06716     int newline;
06717     char *p, *pp, *e;
06718     long len, rslen;
06719 
06720     str_modify_keep_cr(str);
06721     len = RSTRING_LEN(str);
06722     if (len == 0) return Qnil;
06723     p = RSTRING_PTR(str);
06724     e = p + len;
06725     if (argc == 0) {
06726         rs = rb_rs;
06727         if (rs == rb_default_rs) {
06728           smart_chomp:
06729             enc = rb_enc_get(str);
06730             if (rb_enc_mbminlen(enc) > 1) {
06731                 pp = rb_enc_left_char_head(p, e-rb_enc_mbminlen(enc), e, enc);
06732                 if (rb_enc_is_newline(pp, e, enc)) {
06733                     e = pp;
06734                 }
06735                 pp = e - rb_enc_mbminlen(enc);
06736                 if (pp >= p) {
06737                     pp = rb_enc_left_char_head(p, pp, e, enc);
06738                     if (rb_enc_ascget(pp, e, 0, enc) == '\r') {
06739                         e = pp;
06740                     }
06741                 }
06742                 if (e == RSTRING_END(str)) {
06743                     return Qnil;
06744                 }
06745                 len = e - RSTRING_PTR(str);
06746                 STR_SET_LEN(str, len);
06747             }
06748             else {
06749                 if (RSTRING_PTR(str)[len-1] == '\n') {
06750                     STR_DEC_LEN(str);
06751                     if (RSTRING_LEN(str) > 0 &&
06752                         RSTRING_PTR(str)[RSTRING_LEN(str)-1] == '\r') {
06753                         STR_DEC_LEN(str);
06754                     }
06755                 }
06756                 else if (RSTRING_PTR(str)[len-1] == '\r') {
06757                     STR_DEC_LEN(str);
06758                 }
06759                 else {
06760                     return Qnil;
06761                 }
06762             }
06763             RSTRING_PTR(str)[RSTRING_LEN(str)] = '\0';
06764             return str;
06765         }
06766     }
06767     else {
06768         rb_scan_args(argc, argv, "01", &rs);
06769     }
06770     if (NIL_P(rs)) return Qnil;
06771     StringValue(rs);
06772     rslen = RSTRING_LEN(rs);
06773     if (rslen == 0) {
06774         while (len>0 && p[len-1] == '\n') {
06775             len--;
06776             if (len>0 && p[len-1] == '\r')
06777                 len--;
06778         }
06779         if (len < RSTRING_LEN(str)) {
06780             STR_SET_LEN(str, len);
06781             RSTRING_PTR(str)[len] = '\0';
06782             return str;
06783         }
06784         return Qnil;
06785     }
06786     if (rslen > len) return Qnil;
06787     newline = RSTRING_PTR(rs)[rslen-1];
06788     if (rslen == 1 && newline == '\n')
06789         goto smart_chomp;
06790 
06791     enc = rb_enc_check(str, rs);
06792     if (is_broken_string(rs)) {
06793         return Qnil;
06794     }
06795     pp = e - rslen;
06796     if (p[len-1] == newline &&
06797         (rslen <= 1 ||
06798          memcmp(RSTRING_PTR(rs), pp, rslen) == 0)) {
06799         if (rb_enc_left_char_head(p, pp, e, enc) != pp)
06800             return Qnil;
06801         if (ENC_CODERANGE(str) != ENC_CODERANGE_7BIT) {
06802             ENC_CODERANGE_CLEAR(str);
06803         }
06804         STR_SET_LEN(str, RSTRING_LEN(str) - rslen);
06805         RSTRING_PTR(str)[RSTRING_LEN(str)] = '\0';
06806         return str;
06807     }
06808     return Qnil;
06809 }
06810 
06811 
06812 /*
06813  *  call-seq:
06814  *     str.chomp(separator=$/)   -> new_str
06815  *
06816  *  Returns a new <code>String</code> with the given record separator removed
06817  *  from the end of <i>str</i> (if present). If <code>$/</code> has not been
06818  *  changed from the default Ruby record separator, then <code>chomp</code> also
06819  *  removes carriage return characters (that is it will remove <code>\n</code>,
06820  *  <code>\r</code>, and <code>\r\n</code>).
06821  *
06822  *     "hello".chomp            #=> "hello"
06823  *     "hello\n".chomp          #=> "hello"
06824  *     "hello\r\n".chomp        #=> "hello"
06825  *     "hello\n\r".chomp        #=> "hello\n"
06826  *     "hello\r".chomp          #=> "hello"
06827  *     "hello \n there".chomp   #=> "hello \n there"
06828  *     "hello".chomp("llo")     #=> "he"
06829  */
06830 
06831 static VALUE
06832 rb_str_chomp(int argc, VALUE *argv, VALUE str)
06833 {
06834     str = rb_str_dup(str);
06835     rb_str_chomp_bang(argc, argv, str);
06836     return str;
06837 }
06838 
06839 /*
06840  *  call-seq:
06841  *     str.lstrip!   -> self or nil
06842  *
06843  *  Removes leading whitespace from <i>str</i>, returning <code>nil</code> if no
06844  *  change was made. See also <code>String#rstrip!</code> and
06845  *  <code>String#strip!</code>.
06846  *
06847  *     "  hello  ".lstrip   #=> "hello  "
06848  *     "hello".lstrip!      #=> nil
06849  */
06850 
06851 static VALUE
06852 rb_str_lstrip_bang(VALUE str)
06853 {
06854     rb_encoding *enc;
06855     char *s, *t, *e;
06856 
06857     str_modify_keep_cr(str);
06858     enc = STR_ENC_GET(str);
06859     s = RSTRING_PTR(str);
06860     if (!s || RSTRING_LEN(str) == 0) return Qnil;
06861     e = t = RSTRING_END(str);
06862     /* remove spaces at head */
06863     while (s < e) {
06864         int n;
06865         unsigned int cc = rb_enc_codepoint_len(s, e, &n, enc);
06866 
06867         if (!rb_isspace(cc)) break;
06868         s += n;
06869     }
06870 
06871     if (s > RSTRING_PTR(str)) {
06872         STR_SET_LEN(str, t-s);
06873         memmove(RSTRING_PTR(str), s, RSTRING_LEN(str));
06874         RSTRING_PTR(str)[RSTRING_LEN(str)] = '\0';
06875         return str;
06876     }
06877     return Qnil;
06878 }
06879 
06880 
06881 /*
06882  *  call-seq:
06883  *     str.lstrip   -> new_str
06884  *
06885  *  Returns a copy of <i>str</i> with leading whitespace removed. See also
06886  *  <code>String#rstrip</code> and <code>String#strip</code>.
06887  *
06888  *     "  hello  ".lstrip   #=> "hello  "
06889  *     "hello".lstrip       #=> "hello"
06890  */
06891 
06892 static VALUE
06893 rb_str_lstrip(VALUE str)
06894 {
06895     str = rb_str_dup(str);
06896     rb_str_lstrip_bang(str);
06897     return str;
06898 }
06899 
06900 
06901 /*
06902  *  call-seq:
06903  *     str.rstrip!   -> self or nil
06904  *
06905  *  Removes trailing whitespace from <i>str</i>, returning <code>nil</code> if
06906  *  no change was made. See also <code>String#lstrip!</code> and
06907  *  <code>String#strip!</code>.
06908  *
06909  *     "  hello  ".rstrip   #=> "  hello"
06910  *     "hello".rstrip!      #=> nil
06911  */
06912 
06913 static VALUE
06914 rb_str_rstrip_bang(VALUE str)
06915 {
06916     rb_encoding *enc;
06917     char *s, *t, *e;
06918 
06919     str_modify_keep_cr(str);
06920     enc = STR_ENC_GET(str);
06921     rb_str_check_dummy_enc(enc);
06922     s = RSTRING_PTR(str);
06923     if (!s || RSTRING_LEN(str) == 0) return Qnil;
06924     t = e = RSTRING_END(str);
06925 
06926     /* remove trailing spaces or '\0's */
06927     if (single_byte_optimizable(str)) {
06928         unsigned char c;
06929         while (s < t && ((c = *(t-1)) == '\0' || ascii_isspace(c))) t--;
06930     }
06931     else {
06932         char *tp;
06933 
06934         while ((tp = rb_enc_prev_char(s, t, e, enc)) != NULL) {
06935             unsigned int c = rb_enc_codepoint(tp, e, enc);
06936             if (c && !rb_isspace(c)) break;
06937             t = tp;
06938         }
06939     }
06940     if (t < e) {
06941         long len = t-RSTRING_PTR(str);
06942 
06943         STR_SET_LEN(str, len);
06944         RSTRING_PTR(str)[len] = '\0';
06945         return str;
06946     }
06947     return Qnil;
06948 }
06949 
06950 
06951 /*
06952  *  call-seq:
06953  *     str.rstrip   -> new_str
06954  *
06955  *  Returns a copy of <i>str</i> with trailing whitespace removed. See also
06956  *  <code>String#lstrip</code> and <code>String#strip</code>.
06957  *
06958  *     "  hello  ".rstrip   #=> "  hello"
06959  *     "hello".rstrip       #=> "hello"
06960  */
06961 
06962 static VALUE
06963 rb_str_rstrip(VALUE str)
06964 {
06965     str = rb_str_dup(str);
06966     rb_str_rstrip_bang(str);
06967     return str;
06968 }
06969 
06970 
06971 /*
06972  *  call-seq:
06973  *     str.strip!   -> str or nil
06974  *
06975  *  Removes leading and trailing whitespace from <i>str</i>. Returns
06976  *  <code>nil</code> if <i>str</i> was not altered.
06977  */
06978 
06979 static VALUE
06980 rb_str_strip_bang(VALUE str)
06981 {
06982     VALUE l = rb_str_lstrip_bang(str);
06983     VALUE r = rb_str_rstrip_bang(str);
06984 
06985     if (NIL_P(l) && NIL_P(r)) return Qnil;
06986     return str;
06987 }
06988 
06989 
06990 /*
06991  *  call-seq:
06992  *     str.strip   -> new_str
06993  *
06994  *  Returns a copy of <i>str</i> with leading and trailing whitespace removed.
06995  *
06996  *     "    hello    ".strip   #=> "hello"
06997  *     "\tgoodbye\r\n".strip   #=> "goodbye"
06998  */
06999 
07000 static VALUE
07001 rb_str_strip(VALUE str)
07002 {
07003     str = rb_str_dup(str);
07004     rb_str_strip_bang(str);
07005     return str;
07006 }
07007 
07008 static VALUE
07009 scan_once(VALUE str, VALUE pat, long *start)
07010 {
07011     VALUE result, match;
07012     struct re_registers *regs;
07013     int i;
07014 
07015     if (rb_reg_search(pat, str, *start, 0) >= 0) {
07016         match = rb_backref_get();
07017         regs = RMATCH_REGS(match);
07018         if (BEG(0) == END(0)) {
07019             rb_encoding *enc = STR_ENC_GET(str);
07020             /*
07021              * Always consume at least one character of the input string
07022              */
07023             if (RSTRING_LEN(str) > END(0))
07024                 *start = END(0)+rb_enc_fast_mbclen(RSTRING_PTR(str)+END(0),
07025                                                    RSTRING_END(str), enc);
07026             else
07027                 *start = END(0)+1;
07028         }
07029         else {
07030             *start = END(0);
07031         }
07032         if (regs->num_regs == 1) {
07033             return rb_reg_nth_match(0, match);
07034         }
07035         result = rb_ary_new2(regs->num_regs);
07036         for (i=1; i < regs->num_regs; i++) {
07037             rb_ary_push(result, rb_reg_nth_match(i, match));
07038         }
07039 
07040         return result;
07041     }
07042     return Qnil;
07043 }
07044 
07045 
07046 /*
07047  *  call-seq:
07048  *     str.scan(pattern)                         -> array
07049  *     str.scan(pattern) {|match, ...| block }   -> str
07050  *
07051  *  Both forms iterate through <i>str</i>, matching the pattern (which may be a
07052  *  <code>Regexp</code> or a <code>String</code>). For each match, a result is
07053  *  generated and either added to the result array or passed to the block. If
07054  *  the pattern contains no groups, each individual result consists of the
07055  *  matched string, <code>$&</code>.  If the pattern contains groups, each
07056  *  individual result is itself an array containing one entry per group.
07057  *
07058  *     a = "cruel world"
07059  *     a.scan(/\w+/)        #=> ["cruel", "world"]
07060  *     a.scan(/.../)        #=> ["cru", "el ", "wor"]
07061  *     a.scan(/(...)/)      #=> [["cru"], ["el "], ["wor"]]
07062  *     a.scan(/(..)(..)/)   #=> [["cr", "ue"], ["l ", "wo"]]
07063  *
07064  *  And the block form:
07065  *
07066  *     a.scan(/\w+/) {|w| print "<<#{w}>> " }
07067  *     print "\n"
07068  *     a.scan(/(.)(.)/) {|x,y| print y, x }
07069  *     print "\n"
07070  *
07071  *  <em>produces:</em>
07072  *
07073  *     <<cruel>> <<world>>
07074  *     rceu lowlr
07075  */
07076 
07077 static VALUE
07078 rb_str_scan(VALUE str, VALUE pat)
07079 {
07080     VALUE result;
07081     long start = 0;
07082     long last = -1, prev = 0;
07083     char *p = RSTRING_PTR(str); long len = RSTRING_LEN(str);
07084 
07085     pat = get_pat(pat, 1);
07086     if (!rb_block_given_p()) {
07087         VALUE ary = rb_ary_new();
07088 
07089         while (!NIL_P(result = scan_once(str, pat, &start))) {
07090             last = prev;
07091             prev = start;
07092             rb_ary_push(ary, result);
07093         }
07094         if (last >= 0) rb_reg_search(pat, str, last, 0);
07095         return ary;
07096     }
07097 
07098     while (!NIL_P(result = scan_once(str, pat, &start))) {
07099         last = prev;
07100         prev = start;
07101         rb_yield(result);
07102         str_mod_check(str, p, len);
07103     }
07104     if (last >= 0) rb_reg_search(pat, str, last, 0);
07105     return str;
07106 }
07107 
07108 
07109 /*
07110  *  call-seq:
07111  *     str.hex   -> integer
07112  *
07113  *  Treats leading characters from <i>str</i> as a string of hexadecimal digits
07114  *  (with an optional sign and an optional <code>0x</code>) and returns the
07115  *  corresponding number. Zero is returned on error.
07116  *
07117  *     "0x0a".hex     #=> 10
07118  *     "-1234".hex    #=> -4660
07119  *     "0".hex        #=> 0
07120  *     "wombat".hex   #=> 0
07121  */
07122 
07123 static VALUE
07124 rb_str_hex(VALUE str)
07125 {
07126     return rb_str_to_inum(str, 16, FALSE);
07127 }
07128 
07129 
07130 /*
07131  *  call-seq:
07132  *     str.oct   -> integer
07133  *
07134  *  Treats leading characters of <i>str</i> as a string of octal digits (with an
07135  *  optional sign) and returns the corresponding number.  Returns 0 if the
07136  *  conversion fails.
07137  *
07138  *     "123".oct       #=> 83
07139  *     "-377".oct      #=> -255
07140  *     "bad".oct       #=> 0
07141  *     "0377bad".oct   #=> 255
07142  */
07143 
07144 static VALUE
07145 rb_str_oct(VALUE str)
07146 {
07147     return rb_str_to_inum(str, -8, FALSE);
07148 }
07149 
07150 
07151 /*
07152  *  call-seq:
07153  *     str.crypt(salt_str)   -> new_str
07154  *
07155  *  Applies a one-way cryptographic hash to <i>str</i> by invoking the
07156  *  standard library function <code>crypt(3)</code> with the given
07157  *  salt string.  While the format and the result are system and
07158  *  implementation dependent, using a salt matching the regular
07159  *  expression <code>\A[a-zA-Z0-9./]{2}</code> should be valid and
07160  *  safe on any platform, in which only the first two characters are
07161  *  significant.
07162  *
07163  *  This method is for use in system specific scripts, so if you want
07164  *  a cross-platform hash function consider using Digest or OpenSSL
07165  *  instead.
07166  */
07167 
07168 static VALUE
07169 rb_str_crypt(VALUE str, VALUE salt)
07170 {
07171     extern char *crypt(const char *, const char *);
07172     VALUE result;
07173     const char *s, *saltp;
07174     char *res;
07175 #ifdef BROKEN_CRYPT
07176     char salt_8bit_clean[3];
07177 #endif
07178 
07179     StringValue(salt);
07180     if (RSTRING_LEN(salt) < 2)
07181         rb_raise(rb_eArgError, "salt too short (need >=2 bytes)");
07182 
07183     s = RSTRING_PTR(str);
07184     if (!s) s = "";
07185     saltp = RSTRING_PTR(salt);
07186 #ifdef BROKEN_CRYPT
07187     if (!ISASCII((unsigned char)saltp[0]) || !ISASCII((unsigned char)saltp[1])) {
07188         salt_8bit_clean[0] = saltp[0] & 0x7f;
07189         salt_8bit_clean[1] = saltp[1] & 0x7f;
07190         salt_8bit_clean[2] = '\0';
07191         saltp = salt_8bit_clean;
07192     }
07193 #endif
07194     res = crypt(s, saltp);
07195     if (!res) {
07196         rb_sys_fail("crypt");
07197     }
07198     result = rb_str_new2(res);
07199     OBJ_INFECT(result, str);
07200     OBJ_INFECT(result, salt);
07201     return result;
07202 }
07203 
07204 
07205 /*
07206  *  call-seq:
07207  *     str.intern   -> symbol
07208  *     str.to_sym   -> symbol
07209  *
07210  *  Returns the <code>Symbol</code> corresponding to <i>str</i>, creating the
07211  *  symbol if it did not previously exist. See <code>Symbol#id2name</code>.
07212  *
07213  *     "Koala".intern         #=> :Koala
07214  *     s = 'cat'.to_sym       #=> :cat
07215  *     s == :cat              #=> true
07216  *     s = '@cat'.to_sym      #=> :@cat
07217  *     s == :@cat             #=> true
07218  *
07219  *  This can also be used to create symbols that cannot be represented using the
07220  *  <code>:xxx</code> notation.
07221  *
07222  *     'cat and dog'.to_sym   #=> :"cat and dog"
07223  */
07224 
07225 VALUE
07226 rb_str_intern(VALUE s)
07227 {
07228     VALUE str = RB_GC_GUARD(s);
07229     ID id;
07230 
07231     id = rb_intern_str(str);
07232     return ID2SYM(id);
07233 }
07234 
07235 
07236 /*
07237  *  call-seq:
07238  *     str.ord   -> integer
07239  *
07240  *  Return the <code>Integer</code> ordinal of a one-character string.
07241  *
07242  *     "a".ord         #=> 97
07243  */
07244 
07245 VALUE
07246 rb_str_ord(VALUE s)
07247 {
07248     unsigned int c;
07249 
07250     c = rb_enc_codepoint(RSTRING_PTR(s), RSTRING_END(s), STR_ENC_GET(s));
07251     return UINT2NUM(c);
07252 }
07253 /*
07254  *  call-seq:
07255  *     str.sum(n=16)   -> integer
07256  *
07257  *  Returns a basic <em>n</em>-bit checksum of the characters in <i>str</i>,
07258  *  where <em>n</em> is the optional <code>Fixnum</code> parameter, defaulting
07259  *  to 16. The result is simply the sum of the binary value of each character in
07260  *  <i>str</i> modulo <code>2**n - 1</code>. This is not a particularly good
07261  *  checksum.
07262  */
07263 
07264 static VALUE
07265 rb_str_sum(int argc, VALUE *argv, VALUE str)
07266 {
07267     VALUE vbits;
07268     int bits;
07269     char *ptr, *p, *pend;
07270     long len;
07271     VALUE sum = INT2FIX(0);
07272     unsigned long sum0 = 0;
07273 
07274     if (argc == 0) {
07275         bits = 16;
07276     }
07277     else {
07278         rb_scan_args(argc, argv, "01", &vbits);
07279         bits = NUM2INT(vbits);
07280     }
07281     ptr = p = RSTRING_PTR(str);
07282     len = RSTRING_LEN(str);
07283     pend = p + len;
07284 
07285     while (p < pend) {
07286         if (FIXNUM_MAX - UCHAR_MAX < sum0) {
07287             sum = rb_funcall(sum, '+', 1, LONG2FIX(sum0));
07288             str_mod_check(str, ptr, len);
07289             sum0 = 0;
07290         }
07291         sum0 += (unsigned char)*p;
07292         p++;
07293     }
07294 
07295     if (bits == 0) {
07296         if (sum0) {
07297             sum = rb_funcall(sum, '+', 1, LONG2FIX(sum0));
07298         }
07299     }
07300     else {
07301         if (sum == INT2FIX(0)) {
07302             if (bits < (int)sizeof(long)*CHAR_BIT) {
07303                 sum0 &= (((unsigned long)1)<<bits)-1;
07304             }
07305             sum = LONG2FIX(sum0);
07306         }
07307         else {
07308             VALUE mod;
07309 
07310             if (sum0) {
07311                 sum = rb_funcall(sum, '+', 1, LONG2FIX(sum0));
07312             }
07313 
07314             mod = rb_funcall(INT2FIX(1), rb_intern("<<"), 1, INT2FIX(bits));
07315             mod = rb_funcall(mod, '-', 1, INT2FIX(1));
07316             sum = rb_funcall(sum, '&', 1, mod);
07317         }
07318     }
07319     return sum;
07320 }
07321 
07322 static VALUE
07323 rb_str_justify(int argc, VALUE *argv, VALUE str, char jflag)
07324 {
07325     rb_encoding *enc;
07326     VALUE w;
07327     long width, len, flen = 1, fclen = 1;
07328     VALUE res;
07329     char *p;
07330     const char *f = " ";
07331     long n, size, llen, rlen, llen2 = 0, rlen2 = 0;
07332     volatile VALUE pad;
07333     int singlebyte = 1, cr;
07334 
07335     rb_scan_args(argc, argv, "11", &w, &pad);
07336     enc = STR_ENC_GET(str);
07337     width = NUM2LONG(w);
07338     if (argc == 2) {
07339         StringValue(pad);
07340         enc = rb_enc_check(str, pad);
07341         f = RSTRING_PTR(pad);
07342         flen = RSTRING_LEN(pad);
07343         fclen = str_strlen(pad, enc);
07344         singlebyte = single_byte_optimizable(pad);
07345         if (flen == 0 || fclen == 0) {
07346             rb_raise(rb_eArgError, "zero width padding");
07347         }
07348     }
07349     len = str_strlen(str, enc);
07350     if (width < 0 || len >= width) return rb_str_dup(str);
07351     n = width - len;
07352     llen = (jflag == 'l') ? 0 : ((jflag == 'r') ? n : n/2);
07353     rlen = n - llen;
07354     cr = ENC_CODERANGE(str);
07355     if (flen > 1) {
07356        llen2 = str_offset(f, f + flen, llen % fclen, enc, singlebyte);
07357        rlen2 = str_offset(f, f + flen, rlen % fclen, enc, singlebyte);
07358     }
07359     size = RSTRING_LEN(str);
07360     if ((len = llen / fclen + rlen / fclen) >= LONG_MAX / flen ||
07361        (len *= flen) >= LONG_MAX - llen2 - rlen2 ||
07362        (len += llen2 + rlen2) >= LONG_MAX - size) {
07363        rb_raise(rb_eArgError, "argument too big");
07364     }
07365     len += size;
07366     res = rb_str_new5(str, 0, len);
07367     p = RSTRING_PTR(res);
07368     if (flen <= 1) {
07369        memset(p, *f, llen);
07370        p += llen;
07371     }
07372     else {
07373        while (llen >= fclen) {
07374             memcpy(p,f,flen);
07375             p += flen;
07376             llen -= fclen;
07377         }
07378        if (llen > 0) {
07379            memcpy(p, f, llen2);
07380            p += llen2;
07381         }
07382     }
07383     memcpy(p, RSTRING_PTR(str), size);
07384     p += size;
07385     if (flen <= 1) {
07386        memset(p, *f, rlen);
07387        p += rlen;
07388     }
07389     else {
07390        while (rlen >= fclen) {
07391             memcpy(p,f,flen);
07392             p += flen;
07393             rlen -= fclen;
07394         }
07395        if (rlen > 0) {
07396            memcpy(p, f, rlen2);
07397            p += rlen2;
07398         }
07399     }
07400     *p = '\0';
07401     STR_SET_LEN(res, p-RSTRING_PTR(res));
07402     OBJ_INFECT(res, str);
07403     if (!NIL_P(pad)) OBJ_INFECT(res, pad);
07404     rb_enc_associate(res, enc);
07405     if (argc == 2)
07406         cr = ENC_CODERANGE_AND(cr, ENC_CODERANGE(pad));
07407     if (cr != ENC_CODERANGE_BROKEN)
07408         ENC_CODERANGE_SET(res, cr);
07409     return res;
07410 }
07411 
07412 
07413 /*
07414  *  call-seq:
07415  *     str.ljust(integer, padstr=' ')   -> new_str
07416  *
07417  *  If <i>integer</i> is greater than the length of <i>str</i>, returns a new
07418  *  <code>String</code> of length <i>integer</i> with <i>str</i> left justified
07419  *  and padded with <i>padstr</i>; otherwise, returns <i>str</i>.
07420  *
07421  *     "hello".ljust(4)            #=> "hello"
07422  *     "hello".ljust(20)           #=> "hello               "
07423  *     "hello".ljust(20, '1234')   #=> "hello123412341234123"
07424  */
07425 
07426 static VALUE
07427 rb_str_ljust(int argc, VALUE *argv, VALUE str)
07428 {
07429     return rb_str_justify(argc, argv, str, 'l');
07430 }
07431 
07432 
07433 /*
07434  *  call-seq:
07435  *     str.rjust(integer, padstr=' ')   -> new_str
07436  *
07437  *  If <i>integer</i> is greater than the length of <i>str</i>, returns a new
07438  *  <code>String</code> of length <i>integer</i> with <i>str</i> right justified
07439  *  and padded with <i>padstr</i>; otherwise, returns <i>str</i>.
07440  *
07441  *     "hello".rjust(4)            #=> "hello"
07442  *     "hello".rjust(20)           #=> "               hello"
07443  *     "hello".rjust(20, '1234')   #=> "123412341234123hello"
07444  */
07445 
07446 static VALUE
07447 rb_str_rjust(int argc, VALUE *argv, VALUE str)
07448 {
07449     return rb_str_justify(argc, argv, str, 'r');
07450 }
07451 
07452 
07453 /*
07454  *  call-seq:
07455  *     str.center(width, padstr=' ')   -> new_str
07456  *
07457  *  Centers +str+ in +width+.  If +width+ is greater than the length of +str+,
07458  *  returns a new String of length +width+ with +str+ centered and padded with
07459  *  +padstr+; otherwise, returns +str+.
07460  *
07461  *     "hello".center(4)         #=> "hello"
07462  *     "hello".center(20)        #=> "       hello        "
07463  *     "hello".center(20, '123') #=> "1231231hello12312312"
07464  */
07465 
07466 static VALUE
07467 rb_str_center(int argc, VALUE *argv, VALUE str)
07468 {
07469     return rb_str_justify(argc, argv, str, 'c');
07470 }
07471 
07472 /*
07473  *  call-seq:
07474  *     str.partition(sep)              -> [head, sep, tail]
07475  *     str.partition(regexp)           -> [head, match, tail]
07476  *
07477  *  Searches <i>sep</i> or pattern (<i>regexp</i>) in the string
07478  *  and returns the part before it, the match, and the part
07479  *  after it.
07480  *  If it is not found, returns two empty strings and <i>str</i>.
07481  *
07482  *     "hello".partition("l")         #=> ["he", "l", "lo"]
07483  *     "hello".partition("x")         #=> ["hello", "", ""]
07484  *     "hello".partition(/.l/)        #=> ["h", "el", "lo"]
07485  */
07486 
07487 static VALUE
07488 rb_str_partition(VALUE str, VALUE sep)
07489 {
07490     long pos;
07491     int regex = FALSE;
07492 
07493     if (RB_TYPE_P(sep, T_REGEXP)) {
07494         pos = rb_reg_search(sep, str, 0, 0);
07495         regex = TRUE;
07496     }
07497     else {
07498         VALUE tmp;
07499 
07500         tmp = rb_check_string_type(sep);
07501         if (NIL_P(tmp)) {
07502             rb_raise(rb_eTypeError, "type mismatch: %s given",
07503                      rb_obj_classname(sep));
07504         }
07505         sep = tmp;
07506         pos = rb_str_index(str, sep, 0);
07507     }
07508     if (pos < 0) {
07509       failed:
07510         return rb_ary_new3(3, str, str_new_empty(str), str_new_empty(str));
07511     }
07512     if (regex) {
07513         sep = rb_str_subpat(str, sep, INT2FIX(0));
07514         if (pos == 0 && RSTRING_LEN(sep) == 0) goto failed;
07515     }
07516     return rb_ary_new3(3, rb_str_subseq(str, 0, pos),
07517                           sep,
07518                           rb_str_subseq(str, pos+RSTRING_LEN(sep),
07519                                              RSTRING_LEN(str)-pos-RSTRING_LEN(sep)));
07520 }
07521 
07522 /*
07523  *  call-seq:
07524  *     str.rpartition(sep)             -> [head, sep, tail]
07525  *     str.rpartition(regexp)          -> [head, match, tail]
07526  *
07527  *  Searches <i>sep</i> or pattern (<i>regexp</i>) in the string from the end
07528  *  of the string, and returns the part before it, the match, and the part
07529  *  after it.
07530  *  If it is not found, returns two empty strings and <i>str</i>.
07531  *
07532  *     "hello".rpartition("l")         #=> ["hel", "l", "o"]
07533  *     "hello".rpartition("x")         #=> ["", "", "hello"]
07534  *     "hello".rpartition(/.l/)        #=> ["he", "ll", "o"]
07535  */
07536 
07537 static VALUE
07538 rb_str_rpartition(VALUE str, VALUE sep)
07539 {
07540     long pos = RSTRING_LEN(str);
07541     int regex = FALSE;
07542 
07543     if (RB_TYPE_P(sep, T_REGEXP)) {
07544         pos = rb_reg_search(sep, str, pos, 1);
07545         regex = TRUE;
07546     }
07547     else {
07548         VALUE tmp;
07549 
07550         tmp = rb_check_string_type(sep);
07551         if (NIL_P(tmp)) {
07552             rb_raise(rb_eTypeError, "type mismatch: %s given",
07553                      rb_obj_classname(sep));
07554         }
07555         sep = tmp;
07556         pos = rb_str_sublen(str, pos);
07557         pos = rb_str_rindex(str, sep, pos);
07558     }
07559     if (pos < 0) {
07560         return rb_ary_new3(3, str_new_empty(str), str_new_empty(str), str);
07561     }
07562     if (regex) {
07563         sep = rb_reg_nth_match(0, rb_backref_get());
07564     }
07565     return rb_ary_new3(3, rb_str_substr(str, 0, pos),
07566                           sep,
07567                           rb_str_substr(str,pos+str_strlen(sep,STR_ENC_GET(sep)),RSTRING_LEN(str)));
07568 }
07569 
07570 /*
07571  *  call-seq:
07572  *     str.start_with?([prefixes]+)   -> true or false
07573  *
07574  *  Returns true if +str+ starts with one of the +prefixes+ given.
07575  *
07576  *    "hello".start_with?("hell")               #=> true
07577  *
07578  *    # returns true if one of the prefixes matches.
07579  *    "hello".start_with?("heaven", "hell")     #=> true
07580  *    "hello".start_with?("heaven", "paradise") #=> false
07581  */
07582 
07583 static VALUE
07584 rb_str_start_with(int argc, VALUE *argv, VALUE str)
07585 {
07586     int i;
07587 
07588     for (i=0; i<argc; i++) {
07589         VALUE tmp = argv[i];
07590         StringValue(tmp);
07591         rb_enc_check(str, tmp);
07592         if (RSTRING_LEN(str) < RSTRING_LEN(tmp)) continue;
07593         if (memcmp(RSTRING_PTR(str), RSTRING_PTR(tmp), RSTRING_LEN(tmp)) == 0)
07594             return Qtrue;
07595     }
07596     return Qfalse;
07597 }
07598 
07599 /*
07600  *  call-seq:
07601  *     str.end_with?([suffixes]+)   -> true or false
07602  *
07603  *  Returns true if +str+ ends with one of the +suffixes+ given.
07604  */
07605 
07606 static VALUE
07607 rb_str_end_with(int argc, VALUE *argv, VALUE str)
07608 {
07609     int i;
07610     char *p, *s, *e;
07611     rb_encoding *enc;
07612 
07613     for (i=0; i<argc; i++) {
07614         VALUE tmp = argv[i];
07615         StringValue(tmp);
07616         enc = rb_enc_check(str, tmp);
07617         if (RSTRING_LEN(str) < RSTRING_LEN(tmp)) continue;
07618         p = RSTRING_PTR(str);
07619         e = p + RSTRING_LEN(str);
07620         s = e - RSTRING_LEN(tmp);
07621         if (rb_enc_left_char_head(p, s, e, enc) != s)
07622             continue;
07623         if (memcmp(s, RSTRING_PTR(tmp), RSTRING_LEN(tmp)) == 0)
07624             return Qtrue;
07625     }
07626     return Qfalse;
07627 }
07628 
07629 void
07630 rb_str_setter(VALUE val, ID id, VALUE *var)
07631 {
07632     if (!NIL_P(val) && !RB_TYPE_P(val, T_STRING)) {
07633         rb_raise(rb_eTypeError, "value of %s must be String", rb_id2name(id));
07634     }
07635     *var = val;
07636 }
07637 
07638 
07639 /*
07640  *  call-seq:
07641  *     str.force_encoding(encoding)   -> str
07642  *
07643  *  Changes the encoding to +encoding+ and returns self.
07644  */
07645 
07646 static VALUE
07647 rb_str_force_encoding(VALUE str, VALUE enc)
07648 {
07649     str_modifiable(str);
07650     rb_enc_associate(str, rb_to_encoding(enc));
07651     ENC_CODERANGE_CLEAR(str);
07652     return str;
07653 }
07654 
07655 /*
07656  *  call-seq:
07657  *     str.b   -> str
07658  *
07659  *  Returns a copied string whose encoding is ASCII-8BIT.
07660  */
07661 
07662 static VALUE
07663 rb_str_b(VALUE str)
07664 {
07665     VALUE str2 = str_alloc(rb_cString);
07666     str_replace_shared_without_enc(str2, str);
07667     OBJ_INFECT(str2, str);
07668     ENC_CODERANGE_SET(str2, ENC_CODERANGE_VALID);
07669     return str2;
07670 }
07671 
07672 /*
07673  *  call-seq:
07674  *     str.valid_encoding?  -> true or false
07675  *
07676  *  Returns true for a string which encoded correctly.
07677  *
07678  *    "\xc2\xa1".force_encoding("UTF-8").valid_encoding?  #=> true
07679  *    "\xc2".force_encoding("UTF-8").valid_encoding?      #=> false
07680  *    "\x80".force_encoding("UTF-8").valid_encoding?      #=> false
07681  */
07682 
07683 static VALUE
07684 rb_str_valid_encoding_p(VALUE str)
07685 {
07686     int cr = rb_enc_str_coderange(str);
07687 
07688     return cr == ENC_CODERANGE_BROKEN ? Qfalse : Qtrue;
07689 }
07690 
07691 /*
07692  *  call-seq:
07693  *     str.ascii_only?  -> true or false
07694  *
07695  *  Returns true for a string which has only ASCII characters.
07696  *
07697  *    "abc".force_encoding("UTF-8").ascii_only?          #=> true
07698  *    "abc\u{6666}".force_encoding("UTF-8").ascii_only?  #=> false
07699  */
07700 
07701 static VALUE
07702 rb_str_is_ascii_only_p(VALUE str)
07703 {
07704     int cr = rb_enc_str_coderange(str);
07705 
07706     return cr == ENC_CODERANGE_7BIT ? Qtrue : Qfalse;
07707 }
07708 
07723 VALUE
07724 rb_str_ellipsize(VALUE str, long len)
07725 {
07726     static const char ellipsis[] = "...";
07727     const long ellipsislen = sizeof(ellipsis) - 1;
07728     rb_encoding *const enc = rb_enc_get(str);
07729     const long blen = RSTRING_LEN(str);
07730     const char *const p = RSTRING_PTR(str), *e = p + blen;
07731     VALUE estr, ret = 0;
07732 
07733     if (len < 0) rb_raise(rb_eIndexError, "negative length %ld", len);
07734     if (len * rb_enc_mbminlen(enc) >= blen ||
07735         (e = rb_enc_nth(p, e, len, enc)) - p == blen) {
07736         ret = str;
07737     }
07738     else if (len <= ellipsislen ||
07739              !(e = rb_enc_step_back(p, e, e, len = ellipsislen, enc))) {
07740         if (rb_enc_asciicompat(enc)) {
07741             ret = rb_str_new_with_class(str, ellipsis, len);
07742             rb_enc_associate(ret, enc);
07743         }
07744         else {
07745             estr = rb_usascii_str_new(ellipsis, len);
07746             ret = rb_str_encode(estr, rb_enc_from_encoding(enc), 0, Qnil);
07747         }
07748     }
07749     else if (ret = rb_str_subseq(str, 0, e - p), rb_enc_asciicompat(enc)) {
07750         rb_str_cat(ret, ellipsis, ellipsislen);
07751     }
07752     else {
07753         estr = rb_str_encode(rb_usascii_str_new(ellipsis, ellipsislen),
07754                              rb_enc_from_encoding(enc), 0, Qnil);
07755         rb_str_append(ret, estr);
07756     }
07757     return ret;
07758 }
07759 
07760 /**********************************************************************
07761  * Document-class: Symbol
07762  *
07763  *  <code>Symbol</code> objects represent names and some strings
07764  *  inside the Ruby
07765  *  interpreter. They are generated using the <code>:name</code> and
07766  *  <code>:"string"</code> literals
07767  *  syntax, and by the various <code>to_sym</code> methods. The same
07768  *  <code>Symbol</code> object will be created for a given name or string
07769  *  for the duration of a program's execution, regardless of the context
07770  *  or meaning of that name. Thus if <code>Fred</code> is a constant in
07771  *  one context, a method in another, and a class in a third, the
07772  *  <code>Symbol</code> <code>:Fred</code> will be the same object in
07773  *  all three contexts.
07774  *
07775  *     module One
07776  *       class Fred
07777  *       end
07778  *       $f1 = :Fred
07779  *     end
07780  *     module Two
07781  *       Fred = 1
07782  *       $f2 = :Fred
07783  *     end
07784  *     def Fred()
07785  *     end
07786  *     $f3 = :Fred
07787  *     $f1.object_id   #=> 2514190
07788  *     $f2.object_id   #=> 2514190
07789  *     $f3.object_id   #=> 2514190
07790  *
07791  */
07792 
07793 
07794 /*
07795  *  call-seq:
07796  *     sym == obj   -> true or false
07797  *
07798  *  Equality---If <i>sym</i> and <i>obj</i> are exactly the same
07799  *  symbol, returns <code>true</code>.
07800  */
07801 
07802 static VALUE
07803 sym_equal(VALUE sym1, VALUE sym2)
07804 {
07805     if (sym1 == sym2) return Qtrue;
07806     return Qfalse;
07807 }
07808 
07809 
07810 static int
07811 sym_printable(const char *s, const char *send, rb_encoding *enc)
07812 {
07813     while (s < send) {
07814         int n;
07815         int c = rb_enc_codepoint_len(s, send, &n, enc);
07816 
07817         if (!rb_enc_isprint(c, enc)) return FALSE;
07818         s += n;
07819     }
07820     return TRUE;
07821 }
07822 
07823 int
07824 rb_str_symname_p(VALUE sym)
07825 {
07826     rb_encoding *enc;
07827     const char *ptr;
07828     long len;
07829     rb_encoding *resenc = rb_default_internal_encoding();
07830 
07831     if (resenc == NULL) resenc = rb_default_external_encoding();
07832     enc = STR_ENC_GET(sym);
07833     ptr = RSTRING_PTR(sym);
07834     len = RSTRING_LEN(sym);
07835     if ((resenc != enc && !rb_str_is_ascii_only_p(sym)) || len != (long)strlen(ptr) ||
07836         !rb_enc_symname_p(ptr, enc) || !sym_printable(ptr, ptr + len, enc)) {
07837         return FALSE;
07838     }
07839     return TRUE;
07840 }
07841 
07842 VALUE
07843 rb_str_quote_unprintable(VALUE str)
07844 {
07845     rb_encoding *enc;
07846     const char *ptr;
07847     long len;
07848     rb_encoding *resenc;
07849 
07850     Check_Type(str, T_STRING);
07851     resenc = rb_default_internal_encoding();
07852     if (resenc == NULL) resenc = rb_default_external_encoding();
07853     enc = STR_ENC_GET(str);
07854     ptr = RSTRING_PTR(str);
07855     len = RSTRING_LEN(str);
07856     if ((resenc != enc && !rb_str_is_ascii_only_p(str)) ||
07857         !sym_printable(ptr, ptr + len, enc)) {
07858         return rb_str_inspect(str);
07859     }
07860     return str;
07861 }
07862 
07863 VALUE
07864 rb_id_quote_unprintable(ID id)
07865 {
07866     return rb_str_quote_unprintable(rb_id2str(id));
07867 }
07868 
07869 /*
07870  *  call-seq:
07871  *     sym.inspect    -> string
07872  *
07873  *  Returns the representation of <i>sym</i> as a symbol literal.
07874  *
07875  *     :fred.inspect   #=> ":fred"
07876  */
07877 
07878 static VALUE
07879 sym_inspect(VALUE sym)
07880 {
07881     VALUE str;
07882     const char *ptr;
07883     long len;
07884     ID id = SYM2ID(sym);
07885     char *dest;
07886 
07887     sym = rb_id2str(id);
07888     if (!rb_str_symname_p(sym)) {
07889         str = rb_str_inspect(sym);
07890         len = RSTRING_LEN(str);
07891         rb_str_resize(str, len + 1);
07892         dest = RSTRING_PTR(str);
07893         memmove(dest + 1, dest, len);
07894         dest[0] = ':';
07895     }
07896     else {
07897         rb_encoding *enc = STR_ENC_GET(sym);
07898         ptr = RSTRING_PTR(sym);
07899         len = RSTRING_LEN(sym);
07900         str = rb_enc_str_new(0, len + 1, enc);
07901         dest = RSTRING_PTR(str);
07902         dest[0] = ':';
07903         memcpy(dest + 1, ptr, len);
07904     }
07905     return str;
07906 }
07907 
07908 
07909 /*
07910  *  call-seq:
07911  *     sym.id2name   -> string
07912  *     sym.to_s      -> string
07913  *
07914  *  Returns the name or string corresponding to <i>sym</i>.
07915  *
07916  *     :fred.id2name   #=> "fred"
07917  */
07918 
07919 
07920 VALUE
07921 rb_sym_to_s(VALUE sym)
07922 {
07923     ID id = SYM2ID(sym);
07924 
07925     return str_new3(rb_cString, rb_id2str(id));
07926 }
07927 
07928 
07929 /*
07930  * call-seq:
07931  *   sym.to_sym   -> sym
07932  *   sym.intern   -> sym
07933  *
07934  * In general, <code>to_sym</code> returns the <code>Symbol</code> corresponding
07935  * to an object. As <i>sym</i> is already a symbol, <code>self</code> is returned
07936  * in this case.
07937  */
07938 
07939 static VALUE
07940 sym_to_sym(VALUE sym)
07941 {
07942     return sym;
07943 }
07944 
07945 static VALUE
07946 sym_call(VALUE args, VALUE sym, int argc, VALUE *argv, VALUE passed_proc)
07947 {
07948     VALUE obj;
07949 
07950     if (argc < 1) {
07951         rb_raise(rb_eArgError, "no receiver given");
07952     }
07953     obj = argv[0];
07954     return rb_funcall_with_block(obj, (ID)sym, argc - 1, argv + 1, passed_proc);
07955 }
07956 
07957 /*
07958  * call-seq:
07959  *   sym.to_proc
07960  *
07961  * Returns a _Proc_ object which respond to the given method by _sym_.
07962  *
07963  *   (1..3).collect(&:to_s)  #=> ["1", "2", "3"]
07964  */
07965 
07966 static VALUE
07967 sym_to_proc(VALUE sym)
07968 {
07969     static VALUE sym_proc_cache = Qfalse;
07970     enum {SYM_PROC_CACHE_SIZE = 67};
07971     VALUE proc;
07972     long id, index;
07973     VALUE *aryp;
07974 
07975     if (!sym_proc_cache) {
07976         sym_proc_cache = rb_ary_tmp_new(SYM_PROC_CACHE_SIZE * 2);
07977         rb_gc_register_mark_object(sym_proc_cache);
07978         rb_ary_store(sym_proc_cache, SYM_PROC_CACHE_SIZE*2 - 1, Qnil);
07979     }
07980 
07981     id = SYM2ID(sym);
07982     index = (id % SYM_PROC_CACHE_SIZE) << 1;
07983 
07984     aryp = RARRAY_PTR(sym_proc_cache);
07985     if (aryp[index] == sym) {
07986         return aryp[index + 1];
07987     }
07988     else {
07989         proc = rb_proc_new(sym_call, (VALUE)id);
07990         aryp[index] = sym;
07991         aryp[index + 1] = proc;
07992         return proc;
07993     }
07994 }
07995 
07996 /*
07997  * call-seq:
07998  *
07999  *   sym.succ
08000  *
08001  * Same as <code>sym.to_s.succ.intern</code>.
08002  */
08003 
08004 static VALUE
08005 sym_succ(VALUE sym)
08006 {
08007     return rb_str_intern(rb_str_succ(rb_sym_to_s(sym)));
08008 }
08009 
08010 /*
08011  * call-seq:
08012  *
08013  *   symbol <=> other_symbol       -> -1, 0, +1 or nil
08014  *
08015  * Compares +symbol+ with +other_symbol+ after calling #to_s on each of the
08016  * symbols. Returns -1, 0, +1 or nil depending on whether +symbol+ is less
08017  * than, equal to, or greater than +other_symbol+.
08018  *
08019  *  +nil+ is returned if the two values are incomparable.
08020  *
08021  * See String#<=> for more information.
08022  */
08023 
08024 static VALUE
08025 sym_cmp(VALUE sym, VALUE other)
08026 {
08027     if (!SYMBOL_P(other)) {
08028         return Qnil;
08029     }
08030     return rb_str_cmp_m(rb_sym_to_s(sym), rb_sym_to_s(other));
08031 }
08032 
08033 /*
08034  * call-seq:
08035  *
08036  *   sym.casecmp(other)  -> -1, 0, +1 or nil
08037  *
08038  * Case-insensitive version of <code>Symbol#<=></code>.
08039  */
08040 
08041 static VALUE
08042 sym_casecmp(VALUE sym, VALUE other)
08043 {
08044     if (!SYMBOL_P(other)) {
08045         return Qnil;
08046     }
08047     return rb_str_casecmp(rb_sym_to_s(sym), rb_sym_to_s(other));
08048 }
08049 
08050 /*
08051  * call-seq:
08052  *   sym =~ obj   -> fixnum or nil
08053  *
08054  * Returns <code>sym.to_s =~ obj</code>.
08055  */
08056 
08057 static VALUE
08058 sym_match(VALUE sym, VALUE other)
08059 {
08060     return rb_str_match(rb_sym_to_s(sym), other);
08061 }
08062 
08063 /*
08064  * call-seq:
08065  *   sym[idx]      -> char
08066  *   sym[b, n]     -> char
08067  *
08068  * Returns <code>sym.to_s[]</code>.
08069  */
08070 
08071 static VALUE
08072 sym_aref(int argc, VALUE *argv, VALUE sym)
08073 {
08074     return rb_str_aref_m(argc, argv, rb_sym_to_s(sym));
08075 }
08076 
08077 /*
08078  * call-seq:
08079  *   sym.length    -> integer
08080  *
08081  * Same as <code>sym.to_s.length</code>.
08082  */
08083 
08084 static VALUE
08085 sym_length(VALUE sym)
08086 {
08087     return rb_str_length(rb_id2str(SYM2ID(sym)));
08088 }
08089 
08090 /*
08091  * call-seq:
08092  *   sym.empty?   -> true or false
08093  *
08094  * Returns that _sym_ is :"" or not.
08095  */
08096 
08097 static VALUE
08098 sym_empty(VALUE sym)
08099 {
08100     return rb_str_empty(rb_id2str(SYM2ID(sym)));
08101 }
08102 
08103 /*
08104  * call-seq:
08105  *   sym.upcase    -> symbol
08106  *
08107  * Same as <code>sym.to_s.upcase.intern</code>.
08108  */
08109 
08110 static VALUE
08111 sym_upcase(VALUE sym)
08112 {
08113     return rb_str_intern(rb_str_upcase(rb_id2str(SYM2ID(sym))));
08114 }
08115 
08116 /*
08117  * call-seq:
08118  *   sym.downcase  -> symbol
08119  *
08120  * Same as <code>sym.to_s.downcase.intern</code>.
08121  */
08122 
08123 static VALUE
08124 sym_downcase(VALUE sym)
08125 {
08126     return rb_str_intern(rb_str_downcase(rb_id2str(SYM2ID(sym))));
08127 }
08128 
08129 /*
08130  * call-seq:
08131  *   sym.capitalize  -> symbol
08132  *
08133  * Same as <code>sym.to_s.capitalize.intern</code>.
08134  */
08135 
08136 static VALUE
08137 sym_capitalize(VALUE sym)
08138 {
08139     return rb_str_intern(rb_str_capitalize(rb_id2str(SYM2ID(sym))));
08140 }
08141 
08142 /*
08143  * call-seq:
08144  *   sym.swapcase  -> symbol
08145  *
08146  * Same as <code>sym.to_s.swapcase.intern</code>.
08147  */
08148 
08149 static VALUE
08150 sym_swapcase(VALUE sym)
08151 {
08152     return rb_str_intern(rb_str_swapcase(rb_id2str(SYM2ID(sym))));
08153 }
08154 
08155 /*
08156  * call-seq:
08157  *   sym.encoding   -> encoding
08158  *
08159  * Returns the Encoding object that represents the encoding of _sym_.
08160  */
08161 
08162 static VALUE
08163 sym_encoding(VALUE sym)
08164 {
08165     return rb_obj_encoding(rb_id2str(SYM2ID(sym)));
08166 }
08167 
08168 ID
08169 rb_to_id(VALUE name)
08170 {
08171     VALUE tmp;
08172 
08173     switch (TYPE(name)) {
08174       default:
08175         tmp = rb_check_string_type(name);
08176         if (NIL_P(tmp)) {
08177             tmp = rb_inspect(name);
08178             rb_raise(rb_eTypeError, "%s is not a symbol",
08179                      RSTRING_PTR(tmp));
08180         }
08181         name = tmp;
08182         /* fall through */
08183       case T_STRING:
08184         name = rb_str_intern(name);
08185         /* fall through */
08186       case T_SYMBOL:
08187         return SYM2ID(name);
08188     }
08189 
08190     UNREACHABLE;
08191 }
08192 
08193 /*
08194  *  A <code>String</code> object holds and manipulates an arbitrary sequence of
08195  *  bytes, typically representing characters. String objects may be created
08196  *  using <code>String::new</code> or as literals.
08197  *
08198  *  Because of aliasing issues, users of strings should be aware of the methods
08199  *  that modify the contents of a <code>String</code> object.  Typically,
08200  *  methods with names ending in ``!'' modify their receiver, while those
08201  *  without a ``!'' return a new <code>String</code>.  However, there are
08202  *  exceptions, such as <code>String#[]=</code>.
08203  *
08204  */
08205 
08206 void
08207 Init_String(void)
08208 {
08209 #undef rb_intern
08210 #define rb_intern(str) rb_intern_const(str)
08211 
08212     rb_cString  = rb_define_class("String", rb_cObject);
08213     rb_include_module(rb_cString, rb_mComparable);
08214     rb_define_alloc_func(rb_cString, empty_str_alloc);
08215     rb_define_singleton_method(rb_cString, "try_convert", rb_str_s_try_convert, 1);
08216     rb_define_method(rb_cString, "initialize", rb_str_init, -1);
08217     rb_define_method(rb_cString, "initialize_copy", rb_str_replace, 1);
08218     rb_define_method(rb_cString, "<=>", rb_str_cmp_m, 1);
08219     rb_define_method(rb_cString, "==", rb_str_equal, 1);
08220     rb_define_method(rb_cString, "===", rb_str_equal, 1);
08221     rb_define_method(rb_cString, "eql?", rb_str_eql, 1);
08222     rb_define_method(rb_cString, "hash", rb_str_hash_m, 0);
08223     rb_define_method(rb_cString, "casecmp", rb_str_casecmp, 1);
08224     rb_define_method(rb_cString, "+", rb_str_plus, 1);
08225     rb_define_method(rb_cString, "*", rb_str_times, 1);
08226     rb_define_method(rb_cString, "%", rb_str_format_m, 1);
08227     rb_define_method(rb_cString, "[]", rb_str_aref_m, -1);
08228     rb_define_method(rb_cString, "[]=", rb_str_aset_m, -1);
08229     rb_define_method(rb_cString, "insert", rb_str_insert, 2);
08230     rb_define_method(rb_cString, "length", rb_str_length, 0);
08231     rb_define_method(rb_cString, "size", rb_str_length, 0);
08232     rb_define_method(rb_cString, "bytesize", rb_str_bytesize, 0);
08233     rb_define_method(rb_cString, "empty?", rb_str_empty, 0);
08234     rb_define_method(rb_cString, "=~", rb_str_match, 1);
08235     rb_define_method(rb_cString, "match", rb_str_match_m, -1);
08236     rb_define_method(rb_cString, "succ", rb_str_succ, 0);
08237     rb_define_method(rb_cString, "succ!", rb_str_succ_bang, 0);
08238     rb_define_method(rb_cString, "next", rb_str_succ, 0);
08239     rb_define_method(rb_cString, "next!", rb_str_succ_bang, 0);
08240     rb_define_method(rb_cString, "upto", rb_str_upto, -1);
08241     rb_define_method(rb_cString, "index", rb_str_index_m, -1);
08242     rb_define_method(rb_cString, "rindex", rb_str_rindex_m, -1);
08243     rb_define_method(rb_cString, "replace", rb_str_replace, 1);
08244     rb_define_method(rb_cString, "clear", rb_str_clear, 0);
08245     rb_define_method(rb_cString, "chr", rb_str_chr, 0);
08246     rb_define_method(rb_cString, "getbyte", rb_str_getbyte, 1);
08247     rb_define_method(rb_cString, "setbyte", rb_str_setbyte, 2);
08248     rb_define_method(rb_cString, "byteslice", rb_str_byteslice, -1);
08249 
08250     rb_define_method(rb_cString, "to_i", rb_str_to_i, -1);
08251     rb_define_method(rb_cString, "to_f", rb_str_to_f, 0);
08252     rb_define_method(rb_cString, "to_s", rb_str_to_s, 0);
08253     rb_define_method(rb_cString, "to_str", rb_str_to_s, 0);
08254     rb_define_method(rb_cString, "inspect", rb_str_inspect, 0);
08255     rb_define_method(rb_cString, "dump", rb_str_dump, 0);
08256 
08257     rb_define_method(rb_cString, "upcase", rb_str_upcase, 0);
08258     rb_define_method(rb_cString, "downcase", rb_str_downcase, 0);
08259     rb_define_method(rb_cString, "capitalize", rb_str_capitalize, 0);
08260     rb_define_method(rb_cString, "swapcase", rb_str_swapcase, 0);
08261 
08262     rb_define_method(rb_cString, "upcase!", rb_str_upcase_bang, 0);
08263     rb_define_method(rb_cString, "downcase!", rb_str_downcase_bang, 0);
08264     rb_define_method(rb_cString, "capitalize!", rb_str_capitalize_bang, 0);
08265     rb_define_method(rb_cString, "swapcase!", rb_str_swapcase_bang, 0);
08266 
08267     rb_define_method(rb_cString, "hex", rb_str_hex, 0);
08268     rb_define_method(rb_cString, "oct", rb_str_oct, 0);
08269     rb_define_method(rb_cString, "split", rb_str_split_m, -1);
08270     rb_define_method(rb_cString, "lines", rb_str_lines, -1);
08271     rb_define_method(rb_cString, "bytes", rb_str_bytes, 0);
08272     rb_define_method(rb_cString, "chars", rb_str_chars, 0);
08273     rb_define_method(rb_cString, "codepoints", rb_str_codepoints, 0);
08274     rb_define_method(rb_cString, "reverse", rb_str_reverse, 0);
08275     rb_define_method(rb_cString, "reverse!", rb_str_reverse_bang, 0);
08276     rb_define_method(rb_cString, "concat", rb_str_concat, 1);
08277     rb_define_method(rb_cString, "<<", rb_str_concat, 1);
08278     rb_define_method(rb_cString, "prepend", rb_str_prepend, 1);
08279     rb_define_method(rb_cString, "crypt", rb_str_crypt, 1);
08280     rb_define_method(rb_cString, "intern", rb_str_intern, 0);
08281     rb_define_method(rb_cString, "to_sym", rb_str_intern, 0);
08282     rb_define_method(rb_cString, "ord", rb_str_ord, 0);
08283 
08284     rb_define_method(rb_cString, "include?", rb_str_include, 1);
08285     rb_define_method(rb_cString, "start_with?", rb_str_start_with, -1);
08286     rb_define_method(rb_cString, "end_with?", rb_str_end_with, -1);
08287 
08288     rb_define_method(rb_cString, "scan", rb_str_scan, 1);
08289 
08290     rb_define_method(rb_cString, "ljust", rb_str_ljust, -1);
08291     rb_define_method(rb_cString, "rjust", rb_str_rjust, -1);
08292     rb_define_method(rb_cString, "center", rb_str_center, -1);
08293 
08294     rb_define_method(rb_cString, "sub", rb_str_sub, -1);
08295     rb_define_method(rb_cString, "gsub", rb_str_gsub, -1);
08296     rb_define_method(rb_cString, "chop", rb_str_chop, 0);
08297     rb_define_method(rb_cString, "chomp", rb_str_chomp, -1);
08298     rb_define_method(rb_cString, "strip", rb_str_strip, 0);
08299     rb_define_method(rb_cString, "lstrip", rb_str_lstrip, 0);
08300     rb_define_method(rb_cString, "rstrip", rb_str_rstrip, 0);
08301 
08302     rb_define_method(rb_cString, "sub!", rb_str_sub_bang, -1);
08303     rb_define_method(rb_cString, "gsub!", rb_str_gsub_bang, -1);
08304     rb_define_method(rb_cString, "chop!", rb_str_chop_bang, 0);
08305     rb_define_method(rb_cString, "chomp!", rb_str_chomp_bang, -1);
08306     rb_define_method(rb_cString, "strip!", rb_str_strip_bang, 0);
08307     rb_define_method(rb_cString, "lstrip!", rb_str_lstrip_bang, 0);
08308     rb_define_method(rb_cString, "rstrip!", rb_str_rstrip_bang, 0);
08309 
08310     rb_define_method(rb_cString, "tr", rb_str_tr, 2);
08311     rb_define_method(rb_cString, "tr_s", rb_str_tr_s, 2);
08312     rb_define_method(rb_cString, "delete", rb_str_delete, -1);
08313     rb_define_method(rb_cString, "squeeze", rb_str_squeeze, -1);
08314     rb_define_method(rb_cString, "count", rb_str_count, -1);
08315 
08316     rb_define_method(rb_cString, "tr!", rb_str_tr_bang, 2);
08317     rb_define_method(rb_cString, "tr_s!", rb_str_tr_s_bang, 2);
08318     rb_define_method(rb_cString, "delete!", rb_str_delete_bang, -1);
08319     rb_define_method(rb_cString, "squeeze!", rb_str_squeeze_bang, -1);
08320 
08321     rb_define_method(rb_cString, "each_line", rb_str_each_line, -1);
08322     rb_define_method(rb_cString, "each_byte", rb_str_each_byte, 0);
08323     rb_define_method(rb_cString, "each_char", rb_str_each_char, 0);
08324     rb_define_method(rb_cString, "each_codepoint", rb_str_each_codepoint, 0);
08325 
08326     rb_define_method(rb_cString, "sum", rb_str_sum, -1);
08327 
08328     rb_define_method(rb_cString, "slice", rb_str_aref_m, -1);
08329     rb_define_method(rb_cString, "slice!", rb_str_slice_bang, -1);
08330 
08331     rb_define_method(rb_cString, "partition", rb_str_partition, 1);
08332     rb_define_method(rb_cString, "rpartition", rb_str_rpartition, 1);
08333 
08334     rb_define_method(rb_cString, "encoding", rb_obj_encoding, 0); /* in encoding.c */
08335     rb_define_method(rb_cString, "force_encoding", rb_str_force_encoding, 1);
08336     rb_define_method(rb_cString, "b", rb_str_b, 0);
08337     rb_define_method(rb_cString, "valid_encoding?", rb_str_valid_encoding_p, 0);
08338     rb_define_method(rb_cString, "ascii_only?", rb_str_is_ascii_only_p, 0);
08339 
08340     id_to_s = rb_intern("to_s");
08341 
08342     rb_fs = Qnil;
08343     rb_define_variable("$;", &rb_fs);
08344     rb_define_variable("$-F", &rb_fs);
08345 
08346     rb_cSymbol = rb_define_class("Symbol", rb_cObject);
08347     rb_include_module(rb_cSymbol, rb_mComparable);
08348     rb_undef_alloc_func(rb_cSymbol);
08349     rb_undef_method(CLASS_OF(rb_cSymbol), "new");
08350     rb_define_singleton_method(rb_cSymbol, "all_symbols", rb_sym_all_symbols, 0); /* in parse.y */
08351 
08352     rb_define_method(rb_cSymbol, "==", sym_equal, 1);
08353     rb_define_method(rb_cSymbol, "===", sym_equal, 1);
08354     rb_define_method(rb_cSymbol, "inspect", sym_inspect, 0);
08355     rb_define_method(rb_cSymbol, "to_s", rb_sym_to_s, 0);
08356     rb_define_method(rb_cSymbol, "id2name", rb_sym_to_s, 0);
08357     rb_define_method(rb_cSymbol, "intern", sym_to_sym, 0);
08358     rb_define_method(rb_cSymbol, "to_sym", sym_to_sym, 0);
08359     rb_define_method(rb_cSymbol, "to_proc", sym_to_proc, 0);
08360     rb_define_method(rb_cSymbol, "succ", sym_succ, 0);
08361     rb_define_method(rb_cSymbol, "next", sym_succ, 0);
08362 
08363     rb_define_method(rb_cSymbol, "<=>", sym_cmp, 1);
08364     rb_define_method(rb_cSymbol, "casecmp", sym_casecmp, 1);
08365     rb_define_method(rb_cSymbol, "=~", sym_match, 1);
08366 
08367     rb_define_method(rb_cSymbol, "[]", sym_aref, -1);
08368     rb_define_method(rb_cSymbol, "slice", sym_aref, -1);
08369     rb_define_method(rb_cSymbol, "length", sym_length, 0);
08370     rb_define_method(rb_cSymbol, "size", sym_length, 0);
08371     rb_define_method(rb_cSymbol, "empty?", sym_empty, 0);
08372     rb_define_method(rb_cSymbol, "match", sym_match, 1);
08373 
08374     rb_define_method(rb_cSymbol, "upcase", sym_upcase, 0);
08375     rb_define_method(rb_cSymbol, "downcase", sym_downcase, 0);
08376     rb_define_method(rb_cSymbol, "capitalize", sym_capitalize, 0);
08377     rb_define_method(rb_cSymbol, "swapcase", sym_swapcase, 0);
08378 
08379     rb_define_method(rb_cSymbol, "encoding", sym_encoding, 0);
08380 }
08381