Ruby  1.9.3p286(2012-10-12revision37165)
string.c
Go to the documentation of this file.
00001 /**********************************************************************
00002 
00003   string.c -
00004 
00005   $Author: naruse $
00006   created at: Mon Aug  9 17:12:58 JST 1993
00007 
00008   Copyright (C) 1993-2007 Yukihiro Matsumoto
00009   Copyright (C) 2000  Network Applied Communication Laboratory, Inc.
00010   Copyright (C) 2000  Information-technology Promotion Agency, Japan
00011 
00012 **********************************************************************/
00013 
00014 #include "ruby/ruby.h"
00015 #include "ruby/re.h"
00016 #include "ruby/encoding.h"
00017 #include "internal.h"
00018 #include <assert.h>
00019 
00020 #define BEG(no) (regs->beg[(no)])
00021 #define END(no) (regs->end[(no)])
00022 
00023 #include <math.h>
00024 #include <ctype.h>
00025 
00026 #ifdef HAVE_UNISTD_H
00027 #include <unistd.h>
00028 #endif
00029 
00030 #define numberof(array) (int)(sizeof(array) / sizeof((array)[0]))
00031 
00032 #undef rb_str_new_cstr
00033 #undef rb_tainted_str_new_cstr
00034 #undef rb_usascii_str_new_cstr
00035 #undef rb_external_str_new_cstr
00036 #undef rb_locale_str_new_cstr
00037 #undef rb_str_new2
00038 #undef rb_str_new3
00039 #undef rb_str_new4
00040 #undef rb_str_new5
00041 #undef rb_tainted_str_new2
00042 #undef rb_usascii_str_new2
00043 #undef rb_str_dup_frozen
00044 #undef rb_str_buf_new_cstr
00045 #undef rb_str_buf_new2
00046 #undef rb_str_buf_cat2
00047 #undef rb_str_cat2
00048 
00049 static VALUE rb_str_clear(VALUE str);
00050 
00051 VALUE rb_cString;
00052 VALUE rb_cSymbol;
00053 
00054 #define RUBY_MAX_CHAR_LEN 16
00055 #define STR_TMPLOCK FL_USER7
00056 #define STR_NOEMBED FL_USER1
00057 #define STR_SHARED  FL_USER2 /* = ELTS_SHARED */
00058 #define STR_ASSOC   FL_USER3
00059 #define STR_SHARED_P(s) FL_ALL((s), STR_NOEMBED|ELTS_SHARED)
00060 #define STR_ASSOC_P(s)  FL_ALL((s), STR_NOEMBED|STR_ASSOC)
00061 #define STR_NOCAPA  (STR_NOEMBED|ELTS_SHARED|STR_ASSOC)
00062 #define STR_NOCAPA_P(s) (FL_TEST((s),STR_NOEMBED) && FL_ANY((s),ELTS_SHARED|STR_ASSOC))
00063 #define STR_UNSET_NOCAPA(s) do {\
00064     if (FL_TEST((s),STR_NOEMBED)) FL_UNSET((s),(ELTS_SHARED|STR_ASSOC));\
00065 } while (0)
00066 
00067 
00068 #define STR_SET_NOEMBED(str) do {\
00069     FL_SET((str), STR_NOEMBED);\
00070     STR_SET_EMBED_LEN((str), 0);\
00071 } while (0)
00072 #define STR_SET_EMBED(str) FL_UNSET((str), STR_NOEMBED)
00073 #define STR_EMBED_P(str) (!FL_TEST((str), STR_NOEMBED))
00074 #define STR_SET_EMBED_LEN(str, n) do { \
00075     long tmp_n = (n);\
00076     RBASIC(str)->flags &= ~RSTRING_EMBED_LEN_MASK;\
00077     RBASIC(str)->flags |= (tmp_n) << RSTRING_EMBED_LEN_SHIFT;\
00078 } while (0)
00079 
00080 #define STR_SET_LEN(str, n) do { \
00081     if (STR_EMBED_P(str)) {\
00082         STR_SET_EMBED_LEN((str), (n));\
00083     }\
00084     else {\
00085         RSTRING(str)->as.heap.len = (n);\
00086     }\
00087 } while (0)
00088 
00089 #define STR_DEC_LEN(str) do {\
00090     if (STR_EMBED_P(str)) {\
00091         long n = RSTRING_LEN(str);\
00092         n--;\
00093         STR_SET_EMBED_LEN((str), n);\
00094     }\
00095     else {\
00096         RSTRING(str)->as.heap.len--;\
00097     }\
00098 } while (0)
00099 
00100 #define RESIZE_CAPA(str,capacity) do {\
00101     if (STR_EMBED_P(str)) {\
00102         if ((capacity) > RSTRING_EMBED_LEN_MAX) {\
00103             char *tmp = ALLOC_N(char, (capacity)+1);\
00104             memcpy(tmp, RSTRING_PTR(str), RSTRING_LEN(str));\
00105             RSTRING(str)->as.heap.ptr = tmp;\
00106             RSTRING(str)->as.heap.len = RSTRING_LEN(str);\
00107             STR_SET_NOEMBED(str);\
00108             RSTRING(str)->as.heap.aux.capa = (capacity);\
00109         }\
00110     }\
00111     else {\
00112         REALLOC_N(RSTRING(str)->as.heap.ptr, char, (capacity)+1);\
00113         if (!STR_NOCAPA_P(str))\
00114             RSTRING(str)->as.heap.aux.capa = (capacity);\
00115     }\
00116 } while (0)
00117 
00118 #define is_ascii_string(str) (rb_enc_str_coderange(str) == ENC_CODERANGE_7BIT)
00119 #define is_broken_string(str) (rb_enc_str_coderange(str) == ENC_CODERANGE_BROKEN)
00120 
00121 #define STR_ENC_GET(str) rb_enc_from_index(ENCODING_GET(str))
00122 
00123 static inline int
00124 single_byte_optimizable(VALUE str)
00125 {
00126     rb_encoding *enc;
00127 
00128     /* Conservative.  It may be ENC_CODERANGE_UNKNOWN. */
00129     if (ENC_CODERANGE(str) == ENC_CODERANGE_7BIT)
00130         return 1;
00131 
00132     enc = STR_ENC_GET(str);
00133     if (rb_enc_mbmaxlen(enc) == 1)
00134         return 1;
00135 
00136     /* Conservative.  Possibly single byte.
00137      * "\xa1" in Shift_JIS for example. */
00138     return 0;
00139 }
00140 
00141 VALUE rb_fs;
00142 
00143 static inline const char *
00144 search_nonascii(const char *p, const char *e)
00145 {
00146 #if SIZEOF_VALUE == 8
00147 # define NONASCII_MASK 0x8080808080808080ULL
00148 #elif SIZEOF_VALUE == 4
00149 # define NONASCII_MASK 0x80808080UL
00150 #endif
00151 #ifdef NONASCII_MASK
00152     if ((int)sizeof(VALUE) * 2 < e - p) {
00153         const VALUE *s, *t;
00154         const VALUE lowbits = sizeof(VALUE) - 1;
00155         s = (const VALUE*)(~lowbits & ((VALUE)p + lowbits));
00156         while (p < (const char *)s) {
00157             if (!ISASCII(*p))
00158                 return p;
00159             p++;
00160         }
00161         t = (const VALUE*)(~lowbits & (VALUE)e);
00162         while (s < t) {
00163             if (*s & NONASCII_MASK) {
00164                 t = s;
00165                 break;
00166             }
00167             s++;
00168         }
00169         p = (const char *)t;
00170     }
00171 #endif
00172     while (p < e) {
00173         if (!ISASCII(*p))
00174             return p;
00175         p++;
00176     }
00177     return NULL;
00178 }
00179 
00180 static int
00181 coderange_scan(const char *p, long len, rb_encoding *enc)
00182 {
00183     const char *e = p + len;
00184 
00185     if (rb_enc_to_index(enc) == 0) {
00186         /* enc is ASCII-8BIT.  ASCII-8BIT string never be broken. */
00187         p = search_nonascii(p, e);
00188         return p ? ENC_CODERANGE_VALID : ENC_CODERANGE_7BIT;
00189     }
00190 
00191     if (rb_enc_asciicompat(enc)) {
00192         p = search_nonascii(p, e);
00193         if (!p) {
00194             return ENC_CODERANGE_7BIT;
00195         }
00196         while (p < e) {
00197             int ret = rb_enc_precise_mbclen(p, e, enc);
00198             if (!MBCLEN_CHARFOUND_P(ret)) {
00199                 return ENC_CODERANGE_BROKEN;
00200             }
00201             p += MBCLEN_CHARFOUND_LEN(ret);
00202             if (p < e) {
00203                 p = search_nonascii(p, e);
00204                 if (!p) {
00205                     return ENC_CODERANGE_VALID;
00206                 }
00207             }
00208         }
00209         if (e < p) {
00210             return ENC_CODERANGE_BROKEN;
00211         }
00212         return ENC_CODERANGE_VALID;
00213     }
00214 
00215     while (p < e) {
00216         int ret = rb_enc_precise_mbclen(p, e, enc);
00217 
00218         if (!MBCLEN_CHARFOUND_P(ret)) {
00219             return ENC_CODERANGE_BROKEN;
00220         }
00221         p += MBCLEN_CHARFOUND_LEN(ret);
00222     }
00223     if (e < p) {
00224         return ENC_CODERANGE_BROKEN;
00225     }
00226     return ENC_CODERANGE_VALID;
00227 }
00228 
00229 long
00230 rb_str_coderange_scan_restartable(const char *s, const char *e, rb_encoding *enc, int *cr)
00231 {
00232     const char *p = s;
00233 
00234     if (*cr == ENC_CODERANGE_BROKEN)
00235         return e - s;
00236 
00237     if (rb_enc_to_index(enc) == 0) {
00238         /* enc is ASCII-8BIT.  ASCII-8BIT string never be broken. */
00239         p = search_nonascii(p, e);
00240         *cr = (!p && *cr != ENC_CODERANGE_VALID) ? ENC_CODERANGE_7BIT : ENC_CODERANGE_VALID;
00241         return e - s;
00242     }
00243     else if (rb_enc_asciicompat(enc)) {
00244         p = search_nonascii(p, e);
00245         if (!p) {
00246             if (*cr != ENC_CODERANGE_VALID) *cr = ENC_CODERANGE_7BIT;
00247             return e - s;
00248         }
00249         while (p < e) {
00250             int ret = rb_enc_precise_mbclen(p, e, enc);
00251             if (!MBCLEN_CHARFOUND_P(ret)) {
00252                 *cr = MBCLEN_INVALID_P(ret) ? ENC_CODERANGE_BROKEN: ENC_CODERANGE_UNKNOWN;
00253                 return p - s;
00254             }
00255             p += MBCLEN_CHARFOUND_LEN(ret);
00256             if (p < e) {
00257                 p = search_nonascii(p, e);
00258                 if (!p) {
00259                     *cr = ENC_CODERANGE_VALID;
00260                     return e - s;
00261                 }
00262             }
00263         }
00264         *cr = e < p ? ENC_CODERANGE_BROKEN: ENC_CODERANGE_VALID;
00265         return p - s;
00266     }
00267     else {
00268         while (p < e) {
00269             int ret = rb_enc_precise_mbclen(p, e, enc);
00270             if (!MBCLEN_CHARFOUND_P(ret)) {
00271                 *cr = MBCLEN_INVALID_P(ret) ? ENC_CODERANGE_BROKEN: ENC_CODERANGE_UNKNOWN;
00272                 return p - s;
00273             }
00274             p += MBCLEN_CHARFOUND_LEN(ret);
00275         }
00276         *cr = e < p ? ENC_CODERANGE_BROKEN: ENC_CODERANGE_VALID;
00277         return p - s;
00278     }
00279 }
00280 
00281 static inline void
00282 str_enc_copy(VALUE str1, VALUE str2)
00283 {
00284     rb_enc_set_index(str1, ENCODING_GET(str2));
00285 }
00286 
00287 static void
00288 rb_enc_cr_str_copy_for_substr(VALUE dest, VALUE src)
00289 {
00290     /* this function is designed for copying encoding and coderange
00291      * from src to new string "dest" which is made from the part of src.
00292      */
00293     str_enc_copy(dest, src);
00294     switch (ENC_CODERANGE(src)) {
00295       case ENC_CODERANGE_7BIT:
00296         ENC_CODERANGE_SET(dest, ENC_CODERANGE_7BIT);
00297         break;
00298       case ENC_CODERANGE_VALID:
00299         if (!rb_enc_asciicompat(STR_ENC_GET(src)) ||
00300             search_nonascii(RSTRING_PTR(dest), RSTRING_END(dest)))
00301             ENC_CODERANGE_SET(dest, ENC_CODERANGE_VALID);
00302         else
00303             ENC_CODERANGE_SET(dest, ENC_CODERANGE_7BIT);
00304         break;
00305       default:
00306         if (RSTRING_LEN(dest) == 0) {
00307             if (!rb_enc_asciicompat(STR_ENC_GET(src)))
00308                 ENC_CODERANGE_SET(dest, ENC_CODERANGE_VALID);
00309             else
00310                 ENC_CODERANGE_SET(dest, ENC_CODERANGE_7BIT);
00311         }
00312         break;
00313     }
00314 }
00315 
00316 static void
00317 rb_enc_cr_str_exact_copy(VALUE dest, VALUE src)
00318 {
00319     str_enc_copy(dest, src);
00320     ENC_CODERANGE_SET(dest, ENC_CODERANGE(src));
00321 }
00322 
00323 int
00324 rb_enc_str_coderange(VALUE str)
00325 {
00326     int cr = ENC_CODERANGE(str);
00327 
00328     if (cr == ENC_CODERANGE_UNKNOWN) {
00329         rb_encoding *enc = STR_ENC_GET(str);
00330         cr = coderange_scan(RSTRING_PTR(str), RSTRING_LEN(str), enc);
00331         ENC_CODERANGE_SET(str, cr);
00332     }
00333     return cr;
00334 }
00335 
00336 int
00337 rb_enc_str_asciionly_p(VALUE str)
00338 {
00339     rb_encoding *enc = STR_ENC_GET(str);
00340 
00341     if (!rb_enc_asciicompat(enc))
00342         return FALSE;
00343     else if (rb_enc_str_coderange(str) == ENC_CODERANGE_7BIT)
00344         return TRUE;
00345     return FALSE;
00346 }
00347 
00348 static inline void
00349 str_mod_check(VALUE s, const char *p, long len)
00350 {
00351     if (RSTRING_PTR(s) != p || RSTRING_LEN(s) != len){
00352         rb_raise(rb_eRuntimeError, "string modified");
00353     }
00354 }
00355 
00356 size_t
00357 rb_str_capacity(VALUE str)
00358 {
00359     if (STR_EMBED_P(str)) {
00360         return RSTRING_EMBED_LEN_MAX;
00361     }
00362     else if (STR_NOCAPA_P(str)) {
00363         return RSTRING(str)->as.heap.len;
00364     }
00365     else {
00366         return RSTRING(str)->as.heap.aux.capa;
00367     }
00368 }
00369 
00370 static inline VALUE
00371 str_alloc(VALUE klass)
00372 {
00373     NEWOBJ(str, struct RString);
00374     OBJSETUP(str, klass, T_STRING);
00375 
00376     str->as.heap.ptr = 0;
00377     str->as.heap.len = 0;
00378     str->as.heap.aux.capa = 0;
00379 
00380     return (VALUE)str;
00381 }
00382 
00383 static VALUE
00384 str_new(VALUE klass, const char *ptr, long len)
00385 {
00386     VALUE str;
00387 
00388     if (len < 0) {
00389         rb_raise(rb_eArgError, "negative string size (or size too big)");
00390     }
00391 
00392     str = str_alloc(klass);
00393     if (len > RSTRING_EMBED_LEN_MAX) {
00394         RSTRING(str)->as.heap.aux.capa = len;
00395         RSTRING(str)->as.heap.ptr = ALLOC_N(char,len+1);
00396         STR_SET_NOEMBED(str);
00397     }
00398     else if (len == 0) {
00399         ENC_CODERANGE_SET(str, ENC_CODERANGE_7BIT);
00400     }
00401     if (ptr) {
00402         memcpy(RSTRING_PTR(str), ptr, len);
00403     }
00404     STR_SET_LEN(str, len);
00405     RSTRING_PTR(str)[len] = '\0';
00406     return str;
00407 }
00408 
00409 VALUE
00410 rb_str_new(const char *ptr, long len)
00411 {
00412     return str_new(rb_cString, ptr, len);
00413 }
00414 
00415 VALUE
00416 rb_usascii_str_new(const char *ptr, long len)
00417 {
00418     VALUE str = rb_str_new(ptr, len);
00419     ENCODING_CODERANGE_SET(str, rb_usascii_encindex(), ENC_CODERANGE_7BIT);
00420     return str;
00421 }
00422 
00423 VALUE
00424 rb_enc_str_new(const char *ptr, long len, rb_encoding *enc)
00425 {
00426     VALUE str = rb_str_new(ptr, len);
00427     rb_enc_associate(str, enc);
00428     return str;
00429 }
00430 
00431 VALUE
00432 rb_str_new_cstr(const char *ptr)
00433 {
00434     if (!ptr) {
00435         rb_raise(rb_eArgError, "NULL pointer given");
00436     }
00437     return rb_str_new(ptr, strlen(ptr));
00438 }
00439 
00440 RUBY_ALIAS_FUNCTION(rb_str_new2(const char *ptr), rb_str_new_cstr, (ptr))
00441 #define rb_str_new2 rb_str_new_cstr
00442 
00443 VALUE
00444 rb_usascii_str_new_cstr(const char *ptr)
00445 {
00446     VALUE str = rb_str_new2(ptr);
00447     ENCODING_CODERANGE_SET(str, rb_usascii_encindex(), ENC_CODERANGE_7BIT);
00448     return str;
00449 }
00450 
00451 RUBY_ALIAS_FUNCTION(rb_usascii_str_new2(const char *ptr), rb_usascii_str_new_cstr, (ptr))
00452 #define rb_usascii_str_new2 rb_usascii_str_new_cstr
00453 
00454 VALUE
00455 rb_tainted_str_new(const char *ptr, long len)
00456 {
00457     VALUE str = rb_str_new(ptr, len);
00458 
00459     OBJ_TAINT(str);
00460     return str;
00461 }
00462 
00463 VALUE
00464 rb_tainted_str_new_cstr(const char *ptr)
00465 {
00466     VALUE str = rb_str_new2(ptr);
00467 
00468     OBJ_TAINT(str);
00469     return str;
00470 }
00471 
00472 RUBY_ALIAS_FUNCTION(rb_tainted_str_new2(const char *ptr), rb_tainted_str_new_cstr, (ptr))
00473 #define rb_tainted_str_new2 rb_tainted_str_new_cstr
00474 
00475 VALUE
00476 rb_str_conv_enc_opts(VALUE str, rb_encoding *from, rb_encoding *to, int ecflags, VALUE ecopts)
00477 {
00478     rb_econv_t *ec;
00479     rb_econv_result_t ret;
00480     long len;
00481     VALUE newstr;
00482     const unsigned char *sp;
00483     unsigned char *dp;
00484 
00485     if (!to) return str;
00486     if (from == to) return str;
00487     if ((rb_enc_asciicompat(to) && ENC_CODERANGE(str) == ENC_CODERANGE_7BIT) ||
00488         to == rb_ascii8bit_encoding()) {
00489         if (STR_ENC_GET(str) != to) {
00490             str = rb_str_dup(str);
00491             rb_enc_associate(str, to);
00492         }
00493         return str;
00494     }
00495 
00496     len = RSTRING_LEN(str);
00497     newstr = rb_str_new(0, len);
00498 
00499   retry:
00500     ec = rb_econv_open_opts(from->name, to->name, ecflags, ecopts);
00501     if (!ec) return str;
00502 
00503     sp = (unsigned char*)RSTRING_PTR(str);
00504     dp = (unsigned char*)RSTRING_PTR(newstr);
00505     ret = rb_econv_convert(ec, &sp, (unsigned char*)RSTRING_END(str),
00506                            &dp, (unsigned char*)RSTRING_END(newstr), 0);
00507     rb_econv_close(ec);
00508     switch (ret) {
00509       case econv_destination_buffer_full:
00510         /* destination buffer short */
00511         len = len < 2 ? 2 : len * 2;
00512         rb_str_resize(newstr, len);
00513         goto retry;
00514 
00515       case econv_finished:
00516         len = dp - (unsigned char*)RSTRING_PTR(newstr);
00517         rb_str_set_len(newstr, len);
00518         rb_enc_associate(newstr, to);
00519         return newstr;
00520 
00521       default:
00522         /* some error, return original */
00523         return str;
00524     }
00525 }
00526 
00527 VALUE
00528 rb_str_conv_enc(VALUE str, rb_encoding *from, rb_encoding *to)
00529 {
00530     return rb_str_conv_enc_opts(str, from, to, 0, Qnil);
00531 }
00532 
00533 VALUE
00534 rb_external_str_new_with_enc(const char *ptr, long len, rb_encoding *eenc)
00535 {
00536     VALUE str;
00537 
00538     str = rb_tainted_str_new(ptr, len);
00539     if (eenc == rb_usascii_encoding() &&
00540         rb_enc_str_coderange(str) != ENC_CODERANGE_7BIT) {
00541         rb_enc_associate(str, rb_ascii8bit_encoding());
00542         return str;
00543     }
00544     rb_enc_associate(str, eenc);
00545     return rb_str_conv_enc(str, eenc, rb_default_internal_encoding());
00546 }
00547 
00548 VALUE
00549 rb_external_str_new(const char *ptr, long len)
00550 {
00551     return rb_external_str_new_with_enc(ptr, len, rb_default_external_encoding());
00552 }
00553 
00554 VALUE
00555 rb_external_str_new_cstr(const char *ptr)
00556 {
00557     return rb_external_str_new_with_enc(ptr, strlen(ptr), rb_default_external_encoding());
00558 }
00559 
00560 VALUE
00561 rb_locale_str_new(const char *ptr, long len)
00562 {
00563     return rb_external_str_new_with_enc(ptr, len, rb_locale_encoding());
00564 }
00565 
00566 VALUE
00567 rb_locale_str_new_cstr(const char *ptr)
00568 {
00569     return rb_external_str_new_with_enc(ptr, strlen(ptr), rb_locale_encoding());
00570 }
00571 
00572 VALUE
00573 rb_filesystem_str_new(const char *ptr, long len)
00574 {
00575     return rb_external_str_new_with_enc(ptr, len, rb_filesystem_encoding());
00576 }
00577 
00578 VALUE
00579 rb_filesystem_str_new_cstr(const char *ptr)
00580 {
00581     return rb_external_str_new_with_enc(ptr, strlen(ptr), rb_filesystem_encoding());
00582 }
00583 
00584 VALUE
00585 rb_str_export(VALUE str)
00586 {
00587     return rb_str_conv_enc(str, STR_ENC_GET(str), rb_default_external_encoding());
00588 }
00589 
00590 VALUE
00591 rb_str_export_locale(VALUE str)
00592 {
00593     return rb_str_conv_enc(str, STR_ENC_GET(str), rb_locale_encoding());
00594 }
00595 
00596 VALUE
00597 rb_str_export_to_enc(VALUE str, rb_encoding *enc)
00598 {
00599     return rb_str_conv_enc(str, STR_ENC_GET(str), enc);
00600 }
00601 
00602 static VALUE
00603 str_replace_shared(VALUE str2, VALUE str)
00604 {
00605     if (RSTRING_LEN(str) <= RSTRING_EMBED_LEN_MAX) {
00606         STR_SET_EMBED(str2);
00607         memcpy(RSTRING_PTR(str2), RSTRING_PTR(str), RSTRING_LEN(str)+1);
00608         STR_SET_EMBED_LEN(str2, RSTRING_LEN(str));
00609     }
00610     else {
00611         str = rb_str_new_frozen(str);
00612         FL_SET(str2, STR_NOEMBED);
00613         RSTRING(str2)->as.heap.len = RSTRING_LEN(str);
00614         RSTRING(str2)->as.heap.ptr = RSTRING_PTR(str);
00615         RSTRING(str2)->as.heap.aux.shared = str;
00616         FL_SET(str2, ELTS_SHARED);
00617     }
00618     rb_enc_cr_str_exact_copy(str2, str);
00619 
00620     return str2;
00621 }
00622 
00623 static VALUE
00624 str_new_shared(VALUE klass, VALUE str)
00625 {
00626     return str_replace_shared(str_alloc(klass), str);
00627 }
00628 
00629 static VALUE
00630 str_new3(VALUE klass, VALUE str)
00631 {
00632     return str_new_shared(klass, str);
00633 }
00634 
00635 VALUE
00636 rb_str_new_shared(VALUE str)
00637 {
00638     VALUE str2 = str_new3(rb_obj_class(str), str);
00639 
00640     OBJ_INFECT(str2, str);
00641     return str2;
00642 }
00643 
00644 RUBY_ALIAS_FUNCTION(rb_str_new3(VALUE str), rb_str_new_shared, (str))
00645 #define rb_str_new3 rb_str_new_shared
00646 
00647 static VALUE
00648 str_new4(VALUE klass, VALUE str)
00649 {
00650     VALUE str2;
00651 
00652     str2 = str_alloc(klass);
00653     STR_SET_NOEMBED(str2);
00654     RSTRING(str2)->as.heap.len = RSTRING_LEN(str);
00655     RSTRING(str2)->as.heap.ptr = RSTRING_PTR(str);
00656     if (STR_SHARED_P(str)) {
00657         VALUE shared = RSTRING(str)->as.heap.aux.shared;
00658         assert(OBJ_FROZEN(shared));
00659         FL_SET(str2, ELTS_SHARED);
00660         RSTRING(str2)->as.heap.aux.shared = shared;
00661     }
00662     else {
00663         FL_SET(str, ELTS_SHARED);
00664         RSTRING(str)->as.heap.aux.shared = str2;
00665     }
00666     rb_enc_cr_str_exact_copy(str2, str);
00667     OBJ_INFECT(str2, str);
00668     return str2;
00669 }
00670 
00671 VALUE
00672 rb_str_new_frozen(VALUE orig)
00673 {
00674     VALUE klass, str;
00675 
00676     if (OBJ_FROZEN(orig)) return orig;
00677     klass = rb_obj_class(orig);
00678     if (STR_SHARED_P(orig) && (str = RSTRING(orig)->as.heap.aux.shared)) {
00679         long ofs;
00680         assert(OBJ_FROZEN(str));
00681         ofs = RSTRING_LEN(str) - RSTRING_LEN(orig);
00682         if ((ofs > 0) || (klass != RBASIC(str)->klass) ||
00683             (!OBJ_TAINTED(str) && OBJ_TAINTED(orig)) ||
00684             ENCODING_GET(str) != ENCODING_GET(orig)) {
00685             str = str_new3(klass, str);
00686             RSTRING(str)->as.heap.ptr += ofs;
00687             RSTRING(str)->as.heap.len -= ofs;
00688             rb_enc_cr_str_exact_copy(str, orig);
00689             OBJ_INFECT(str, orig);
00690         }
00691     }
00692     else if (STR_EMBED_P(orig)) {
00693         str = str_new(klass, RSTRING_PTR(orig), RSTRING_LEN(orig));
00694         rb_enc_cr_str_exact_copy(str, orig);
00695         OBJ_INFECT(str, orig);
00696     }
00697     else if (STR_ASSOC_P(orig)) {
00698         VALUE assoc = RSTRING(orig)->as.heap.aux.shared;
00699         FL_UNSET(orig, STR_ASSOC);
00700         str = str_new4(klass, orig);
00701         FL_SET(str, STR_ASSOC);
00702         RSTRING(str)->as.heap.aux.shared = assoc;
00703     }
00704     else {
00705         str = str_new4(klass, orig);
00706     }
00707     OBJ_FREEZE(str);
00708     return str;
00709 }
00710 
00711 RUBY_ALIAS_FUNCTION(rb_str_new4(VALUE orig), rb_str_new_frozen, (orig))
00712 #define rb_str_new4 rb_str_new_frozen
00713 
00714 VALUE
00715 rb_str_new_with_class(VALUE obj, const char *ptr, long len)
00716 {
00717     return str_new(rb_obj_class(obj), ptr, len);
00718 }
00719 
00720 RUBY_ALIAS_FUNCTION(rb_str_new5(VALUE obj, const char *ptr, long len),
00721            rb_str_new_with_class, (obj, ptr, len))
00722 #define rb_str_new5 rb_str_new_with_class
00723 
00724 static VALUE
00725 str_new_empty(VALUE str)
00726 {
00727     VALUE v = rb_str_new5(str, 0, 0);
00728     rb_enc_copy(v, str);
00729     OBJ_INFECT(v, str);
00730     return v;
00731 }
00732 
00733 #define STR_BUF_MIN_SIZE 128
00734 
00735 VALUE
00736 rb_str_buf_new(long capa)
00737 {
00738     VALUE str = str_alloc(rb_cString);
00739 
00740     if (capa < STR_BUF_MIN_SIZE) {
00741         capa = STR_BUF_MIN_SIZE;
00742     }
00743     FL_SET(str, STR_NOEMBED);
00744     RSTRING(str)->as.heap.aux.capa = capa;
00745     RSTRING(str)->as.heap.ptr = ALLOC_N(char, capa+1);
00746     RSTRING(str)->as.heap.ptr[0] = '\0';
00747 
00748     return str;
00749 }
00750 
00751 VALUE
00752 rb_str_buf_new_cstr(const char *ptr)
00753 {
00754     VALUE str;
00755     long len = strlen(ptr);
00756 
00757     str = rb_str_buf_new(len);
00758     rb_str_buf_cat(str, ptr, len);
00759 
00760     return str;
00761 }
00762 
00763 RUBY_ALIAS_FUNCTION(rb_str_buf_new2(const char *ptr), rb_str_buf_new_cstr, (ptr))
00764 #define rb_str_buf_new2 rb_str_buf_new_cstr
00765 
00766 VALUE
00767 rb_str_tmp_new(long len)
00768 {
00769     return str_new(0, 0, len);
00770 }
00771 
00772 void *
00773 rb_alloc_tmp_buffer(volatile VALUE *store, long len)
00774 {
00775     VALUE s = rb_str_tmp_new(len);
00776     *store = s;
00777     return RSTRING_PTR(s);
00778 }
00779 
00780 void
00781 rb_free_tmp_buffer(volatile VALUE *store)
00782 {
00783     VALUE s = *store;
00784     *store = 0;
00785     if (s) rb_str_clear(s);
00786 }
00787 
00788 void
00789 rb_str_free(VALUE str)
00790 {
00791     if (!STR_EMBED_P(str) && !STR_SHARED_P(str)) {
00792         xfree(RSTRING(str)->as.heap.ptr);
00793     }
00794 }
00795 
00796 RUBY_FUNC_EXPORTED size_t
00797 rb_str_memsize(VALUE str)
00798 {
00799     if (!STR_EMBED_P(str) && !STR_SHARED_P(str)) {
00800         return RSTRING(str)->as.heap.aux.capa;
00801     }
00802     else {
00803         return 0;
00804     }
00805 }
00806 
00807 VALUE
00808 rb_str_to_str(VALUE str)
00809 {
00810     return rb_convert_type(str, T_STRING, "String", "to_str");
00811 }
00812 
00813 static inline void str_discard(VALUE str);
00814 
00815 void
00816 rb_str_shared_replace(VALUE str, VALUE str2)
00817 {
00818     rb_encoding *enc;
00819     int cr;
00820     if (str == str2) return;
00821     enc = STR_ENC_GET(str2);
00822     cr = ENC_CODERANGE(str2);
00823     str_discard(str);
00824     OBJ_INFECT(str, str2);
00825     if (RSTRING_LEN(str2) <= RSTRING_EMBED_LEN_MAX) {
00826         STR_SET_EMBED(str);
00827         memcpy(RSTRING_PTR(str), RSTRING_PTR(str2), RSTRING_LEN(str2)+1);
00828         STR_SET_EMBED_LEN(str, RSTRING_LEN(str2));
00829         rb_enc_associate(str, enc);
00830         ENC_CODERANGE_SET(str, cr);
00831         return;
00832     }
00833     STR_SET_NOEMBED(str);
00834     STR_UNSET_NOCAPA(str);
00835     RSTRING(str)->as.heap.ptr = RSTRING_PTR(str2);
00836     RSTRING(str)->as.heap.len = RSTRING_LEN(str2);
00837     if (STR_NOCAPA_P(str2)) {
00838         FL_SET(str, RBASIC(str2)->flags & STR_NOCAPA);
00839         RSTRING(str)->as.heap.aux.shared = RSTRING(str2)->as.heap.aux.shared;
00840     }
00841     else {
00842         RSTRING(str)->as.heap.aux.capa = RSTRING(str2)->as.heap.aux.capa;
00843     }
00844     STR_SET_EMBED(str2);        /* abandon str2 */
00845     RSTRING_PTR(str2)[0] = 0;
00846     STR_SET_EMBED_LEN(str2, 0);
00847     rb_enc_associate(str, enc);
00848     ENC_CODERANGE_SET(str, cr);
00849 }
00850 
00851 static ID id_to_s;
00852 
00853 VALUE
00854 rb_obj_as_string(VALUE obj)
00855 {
00856     VALUE str;
00857 
00858     if (TYPE(obj) == T_STRING) {
00859         return obj;
00860     }
00861     str = rb_funcall(obj, id_to_s, 0);
00862     if (TYPE(str) != T_STRING)
00863         return rb_any_to_s(obj);
00864     if (OBJ_TAINTED(obj)) OBJ_TAINT(str);
00865     return str;
00866 }
00867 
00868 static VALUE
00869 str_replace(VALUE str, VALUE str2)
00870 {
00871     long len;
00872 
00873     len = RSTRING_LEN(str2);
00874     if (STR_ASSOC_P(str2)) {
00875         str2 = rb_str_new4(str2);
00876     }
00877     if (STR_SHARED_P(str2)) {
00878         VALUE shared = RSTRING(str2)->as.heap.aux.shared;
00879         assert(OBJ_FROZEN(shared));
00880         STR_SET_NOEMBED(str);
00881         RSTRING(str)->as.heap.len = len;
00882         RSTRING(str)->as.heap.ptr = RSTRING_PTR(str2);
00883         FL_SET(str, ELTS_SHARED);
00884         FL_UNSET(str, STR_ASSOC);
00885         RSTRING(str)->as.heap.aux.shared = shared;
00886     }
00887     else {
00888         str_replace_shared(str, str2);
00889     }
00890 
00891     OBJ_INFECT(str, str2);
00892     rb_enc_cr_str_exact_copy(str, str2);
00893     return str;
00894 }
00895 
00896 static VALUE
00897 str_duplicate(VALUE klass, VALUE str)
00898 {
00899     VALUE dup = str_alloc(klass);
00900     str_replace(dup, str);
00901     return dup;
00902 }
00903 
00904 VALUE
00905 rb_str_dup(VALUE str)
00906 {
00907     return str_duplicate(rb_obj_class(str), str);
00908 }
00909 
00910 VALUE
00911 rb_str_resurrect(VALUE str)
00912 {
00913     return str_replace(str_alloc(rb_cString), str);
00914 }
00915 
00916 /*
00917  *  call-seq:
00918  *     String.new(str="")   -> new_str
00919  *
00920  *  Returns a new string object containing a copy of <i>str</i>.
00921  */
00922 
00923 static VALUE
00924 rb_str_init(int argc, VALUE *argv, VALUE str)
00925 {
00926     VALUE orig;
00927 
00928     if (argc > 0 && rb_scan_args(argc, argv, "01", &orig) == 1)
00929         rb_str_replace(str, orig);
00930     return str;
00931 }
00932 
00933 static inline long
00934 enc_strlen(const char *p, const char *e, rb_encoding *enc, int cr)
00935 {
00936     long c;
00937     const char *q;
00938 
00939     if (rb_enc_mbmaxlen(enc) == rb_enc_mbminlen(enc)) {
00940         return (e - p + rb_enc_mbminlen(enc) - 1) / rb_enc_mbminlen(enc);
00941     }
00942     else if (rb_enc_asciicompat(enc)) {
00943         c = 0;
00944         if (cr == ENC_CODERANGE_7BIT || cr == ENC_CODERANGE_VALID) {
00945             while (p < e) {
00946                 if (ISASCII(*p)) {
00947                     q = search_nonascii(p, e);
00948                     if (!q)
00949                         return c + (e - p);
00950                     c += q - p;
00951                     p = q;
00952                 }
00953                 p += rb_enc_fast_mbclen(p, e, enc);
00954                 c++;
00955             }
00956         }
00957         else {
00958             while (p < e) {
00959                 if (ISASCII(*p)) {
00960                     q = search_nonascii(p, e);
00961                     if (!q)
00962                         return c + (e - p);
00963                     c += q - p;
00964                     p = q;
00965                 }
00966                 p += rb_enc_mbclen(p, e, enc);
00967                 c++;
00968             }
00969         }
00970         return c;
00971     }
00972 
00973     for (c=0; p<e; c++) {
00974         p += rb_enc_mbclen(p, e, enc);
00975     }
00976     return c;
00977 }
00978 
00979 long
00980 rb_enc_strlen(const char *p, const char *e, rb_encoding *enc)
00981 {
00982     return enc_strlen(p, e, enc, ENC_CODERANGE_UNKNOWN);
00983 }
00984 
00985 long
00986 rb_enc_strlen_cr(const char *p, const char *e, rb_encoding *enc, int *cr)
00987 {
00988     long c;
00989     const char *q;
00990     int ret;
00991 
00992     *cr = 0;
00993     if (rb_enc_mbmaxlen(enc) == rb_enc_mbminlen(enc)) {
00994         return (e - p + rb_enc_mbminlen(enc) - 1) / rb_enc_mbminlen(enc);
00995     }
00996     else if (rb_enc_asciicompat(enc)) {
00997         c = 0;
00998         while (p < e) {
00999             if (ISASCII(*p)) {
01000                 q = search_nonascii(p, e);
01001                 if (!q) {
01002                     if (!*cr) *cr = ENC_CODERANGE_7BIT;
01003                     return c + (e - p);
01004                 }
01005                 c += q - p;
01006                 p = q;
01007             }
01008             ret = rb_enc_precise_mbclen(p, e, enc);
01009             if (MBCLEN_CHARFOUND_P(ret)) {
01010                 *cr |= ENC_CODERANGE_VALID;
01011                 p += MBCLEN_CHARFOUND_LEN(ret);
01012             }
01013             else {
01014                 *cr = ENC_CODERANGE_BROKEN;
01015                 p++;
01016             }
01017             c++;
01018         }
01019         if (!*cr) *cr = ENC_CODERANGE_7BIT;
01020         return c;
01021     }
01022 
01023     for (c=0; p<e; c++) {
01024         ret = rb_enc_precise_mbclen(p, e, enc);
01025         if (MBCLEN_CHARFOUND_P(ret)) {
01026             *cr |= ENC_CODERANGE_VALID;
01027             p += MBCLEN_CHARFOUND_LEN(ret);
01028         }
01029         else {
01030             *cr = ENC_CODERANGE_BROKEN;
01031             if (p + rb_enc_mbminlen(enc) <= e)
01032                 p += rb_enc_mbminlen(enc);
01033             else
01034                 p = e;
01035         }
01036     }
01037     if (!*cr) *cr = ENC_CODERANGE_7BIT;
01038     return c;
01039 }
01040 
01041 #ifdef NONASCII_MASK
01042 #define is_utf8_lead_byte(c) (((c)&0xC0) != 0x80)
01043 
01044 /*
01045  * UTF-8 leading bytes have either 0xxxxxxx or 11xxxxxx
01046  * bit represention. (see http://en.wikipedia.org/wiki/UTF-8)
01047  * Therefore, following pseudo code can detect UTF-8 leading byte.
01048  *
01049  * if (!(byte & 0x80))
01050  *   byte |= 0x40;          // turn on bit6
01051  * return ((byte>>6) & 1);  // bit6 represent it's leading byte or not.
01052  *
01053  * This function calculate every bytes in the argument word `s'
01054  * using the above logic concurrently. and gather every bytes result.
01055  */
01056 static inline VALUE
01057 count_utf8_lead_bytes_with_word(const VALUE *s)
01058 {
01059     VALUE d = *s;
01060 
01061     /* Transform into bit0 represent UTF-8 leading or not. */
01062     d |= ~(d>>1);
01063     d >>= 6;
01064     d &= NONASCII_MASK >> 7;
01065 
01066     /* Gather every bytes. */
01067     d += (d>>8);
01068     d += (d>>16);
01069 #if SIZEOF_VALUE == 8
01070     d += (d>>32);
01071 #endif
01072     return (d&0xF);
01073 }
01074 #endif
01075 
01076 static long
01077 str_strlen(VALUE str, rb_encoding *enc)
01078 {
01079     const char *p, *e;
01080     long n;
01081     int cr;
01082 
01083     if (single_byte_optimizable(str)) return RSTRING_LEN(str);
01084     if (!enc) enc = STR_ENC_GET(str);
01085     p = RSTRING_PTR(str);
01086     e = RSTRING_END(str);
01087     cr = ENC_CODERANGE(str);
01088 #ifdef NONASCII_MASK
01089     if (ENC_CODERANGE(str) == ENC_CODERANGE_VALID &&
01090         enc == rb_utf8_encoding()) {
01091 
01092         VALUE len = 0;
01093         if ((int)sizeof(VALUE) * 2 < e - p) {
01094             const VALUE *s, *t;
01095             const VALUE lowbits = sizeof(VALUE) - 1;
01096             s = (const VALUE*)(~lowbits & ((VALUE)p + lowbits));
01097             t = (const VALUE*)(~lowbits & (VALUE)e);
01098             while (p < (const char *)s) {
01099                 if (is_utf8_lead_byte(*p)) len++;
01100                 p++;
01101             }
01102             while (s < t) {
01103                 len += count_utf8_lead_bytes_with_word(s);
01104                 s++;
01105             }
01106             p = (const char *)s;
01107         }
01108         while (p < e) {
01109             if (is_utf8_lead_byte(*p)) len++;
01110             p++;
01111         }
01112         return (long)len;
01113     }
01114 #endif
01115     n = rb_enc_strlen_cr(p, e, enc, &cr);
01116     if (cr) {
01117         ENC_CODERANGE_SET(str, cr);
01118     }
01119     return n;
01120 }
01121 
01122 long
01123 rb_str_strlen(VALUE str)
01124 {
01125     return str_strlen(str, STR_ENC_GET(str));
01126 }
01127 
01128 /*
01129  *  call-seq:
01130  *     str.length   -> integer
01131  *     str.size     -> integer
01132  *
01133  *  Returns the character length of <i>str</i>.
01134  */
01135 
01136 VALUE
01137 rb_str_length(VALUE str)
01138 {
01139     long len;
01140 
01141     len = str_strlen(str, STR_ENC_GET(str));
01142     return LONG2NUM(len);
01143 }
01144 
01145 /*
01146  *  call-seq:
01147  *     str.bytesize  -> integer
01148  *
01149  *  Returns the length of <i>str</i> in bytes.
01150  */
01151 
01152 static VALUE
01153 rb_str_bytesize(VALUE str)
01154 {
01155     return LONG2NUM(RSTRING_LEN(str));
01156 }
01157 
01158 /*
01159  *  call-seq:
01160  *     str.empty?   -> true or false
01161  *
01162  *  Returns <code>true</code> if <i>str</i> has a length of zero.
01163  *
01164  *     "hello".empty?   #=> false
01165  *     "".empty?        #=> true
01166  */
01167 
01168 static VALUE
01169 rb_str_empty(VALUE str)
01170 {
01171     if (RSTRING_LEN(str) == 0)
01172         return Qtrue;
01173     return Qfalse;
01174 }
01175 
01176 /*
01177  *  call-seq:
01178  *     str + other_str   -> new_str
01179  *
01180  *  Concatenation---Returns a new <code>String</code> containing
01181  *  <i>other_str</i> concatenated to <i>str</i>.
01182  *
01183  *     "Hello from " + self.to_s   #=> "Hello from main"
01184  */
01185 
01186 VALUE
01187 rb_str_plus(VALUE str1, VALUE str2)
01188 {
01189     VALUE str3;
01190     rb_encoding *enc;
01191 
01192     StringValue(str2);
01193     enc = rb_enc_check(str1, str2);
01194     str3 = rb_str_new(0, RSTRING_LEN(str1)+RSTRING_LEN(str2));
01195     memcpy(RSTRING_PTR(str3), RSTRING_PTR(str1), RSTRING_LEN(str1));
01196     memcpy(RSTRING_PTR(str3) + RSTRING_LEN(str1),
01197            RSTRING_PTR(str2), RSTRING_LEN(str2));
01198     RSTRING_PTR(str3)[RSTRING_LEN(str3)] = '\0';
01199 
01200     if (OBJ_TAINTED(str1) || OBJ_TAINTED(str2))
01201         OBJ_TAINT(str3);
01202     ENCODING_CODERANGE_SET(str3, rb_enc_to_index(enc),
01203                            ENC_CODERANGE_AND(ENC_CODERANGE(str1), ENC_CODERANGE(str2)));
01204     return str3;
01205 }
01206 
01207 /*
01208  *  call-seq:
01209  *     str * integer   -> new_str
01210  *
01211  *  Copy---Returns a new <code>String</code> containing <i>integer</i> copies of
01212  *  the receiver.
01213  *
01214  *     "Ho! " * 3   #=> "Ho! Ho! Ho! "
01215  */
01216 
01217 VALUE
01218 rb_str_times(VALUE str, VALUE times)
01219 {
01220     VALUE str2;
01221     long n, len;
01222     char *ptr2;
01223 
01224     len = NUM2LONG(times);
01225     if (len < 0) {
01226         rb_raise(rb_eArgError, "negative argument");
01227     }
01228     if (len && LONG_MAX/len <  RSTRING_LEN(str)) {
01229         rb_raise(rb_eArgError, "argument too big");
01230     }
01231 
01232     str2 = rb_str_new5(str, 0, len *= RSTRING_LEN(str));
01233     ptr2 = RSTRING_PTR(str2);
01234     if (len) {
01235         n = RSTRING_LEN(str);
01236         memcpy(ptr2, RSTRING_PTR(str), n);
01237         while (n <= len/2) {
01238             memcpy(ptr2 + n, ptr2, n);
01239             n *= 2;
01240         }
01241         memcpy(ptr2 + n, ptr2, len-n);
01242     }
01243     ptr2[RSTRING_LEN(str2)] = '\0';
01244     OBJ_INFECT(str2, str);
01245     rb_enc_cr_str_copy_for_substr(str2, str);
01246 
01247     return str2;
01248 }
01249 
01250 /*
01251  *  call-seq:
01252  *     str % arg   -> new_str
01253  *
01254  *  Format---Uses <i>str</i> as a format specification, and returns the result
01255  *  of applying it to <i>arg</i>. If the format specification contains more than
01256  *  one substitution, then <i>arg</i> must be an <code>Array</code> or <code>Hash</code>
01257  *  containing the values to be substituted. See <code>Kernel::sprintf</code> for
01258  *  details of the format string.
01259  *
01260  *     "%05d" % 123                              #=> "00123"
01261  *     "%-5s: %08x" % [ "ID", self.object_id ]   #=> "ID   : 200e14d6"
01262  *     "foo = %{foo}" % { :foo => 'bar' }        #=> "foo = bar"
01263  */
01264 
01265 static VALUE
01266 rb_str_format_m(VALUE str, VALUE arg)
01267 {
01268     volatile VALUE tmp = rb_check_array_type(arg);
01269 
01270     if (!NIL_P(tmp)) {
01271         return rb_str_format(RARRAY_LENINT(tmp), RARRAY_PTR(tmp), str);
01272     }
01273     return rb_str_format(1, &arg, str);
01274 }
01275 
01276 static inline void
01277 str_modifiable(VALUE str)
01278 {
01279     if (FL_TEST(str, STR_TMPLOCK)) {
01280         rb_raise(rb_eRuntimeError, "can't modify string; temporarily locked");
01281     }
01282     rb_check_frozen(str);
01283     if (!OBJ_UNTRUSTED(str) && rb_safe_level() >= 4)
01284         rb_raise(rb_eSecurityError, "Insecure: can't modify string");
01285 }
01286 
01287 static inline int
01288 str_independent(VALUE str)
01289 {
01290     str_modifiable(str);
01291     if (!STR_SHARED_P(str)) return 1;
01292     if (STR_EMBED_P(str)) return 1;
01293     return 0;
01294 }
01295 
01296 static void
01297 str_make_independent_expand(VALUE str, long expand)
01298 {
01299     char *ptr;
01300     long len = RSTRING_LEN(str);
01301     long capa = len + expand;
01302 
01303     if (len > capa) len = capa;
01304     ptr = ALLOC_N(char, capa + 1);
01305     if (RSTRING_PTR(str)) {
01306         memcpy(ptr, RSTRING_PTR(str), len);
01307     }
01308     STR_SET_NOEMBED(str);
01309     STR_UNSET_NOCAPA(str);
01310     ptr[len] = 0;
01311     RSTRING(str)->as.heap.ptr = ptr;
01312     RSTRING(str)->as.heap.len = len;
01313     RSTRING(str)->as.heap.aux.capa = capa;
01314 }
01315 
01316 #define str_make_independent(str) str_make_independent_expand((str), 0L)
01317 
01318 void
01319 rb_str_modify(VALUE str)
01320 {
01321     if (!str_independent(str))
01322         str_make_independent(str);
01323     ENC_CODERANGE_CLEAR(str);
01324 }
01325 
01326 void
01327 rb_str_modify_expand(VALUE str, long expand)
01328 {
01329     if (expand < 0) {
01330         rb_raise(rb_eArgError, "negative expanding string size");
01331     }
01332     if (!str_independent(str)) {
01333         str_make_independent_expand(str, expand);
01334     }
01335     else if (expand > 0) {
01336         long len = RSTRING_LEN(str);
01337         long capa = len + expand;
01338         if (!STR_EMBED_P(str)) {
01339             REALLOC_N(RSTRING(str)->as.heap.ptr, char, capa+1);
01340             RSTRING(str)->as.heap.aux.capa = capa;
01341         }
01342         else if (capa > RSTRING_EMBED_LEN_MAX) {
01343             str_make_independent_expand(str, expand);
01344         }
01345     }
01346     ENC_CODERANGE_CLEAR(str);
01347 }
01348 
01349 /* As rb_str_modify(), but don't clear coderange */
01350 static void
01351 str_modify_keep_cr(VALUE str)
01352 {
01353     if (!str_independent(str))
01354         str_make_independent(str);
01355     if (ENC_CODERANGE(str) == ENC_CODERANGE_BROKEN)
01356         /* Force re-scan later */
01357         ENC_CODERANGE_CLEAR(str);
01358 }
01359 
01360 static inline void
01361 str_discard(VALUE str)
01362 {
01363     str_modifiable(str);
01364     if (!STR_SHARED_P(str) && !STR_EMBED_P(str)) {
01365         xfree(RSTRING_PTR(str));
01366         RSTRING(str)->as.heap.ptr = 0;
01367         RSTRING(str)->as.heap.len = 0;
01368     }
01369 }
01370 
01371 void
01372 rb_str_associate(VALUE str, VALUE add)
01373 {
01374     /* sanity check */
01375     rb_check_frozen(str);
01376     if (STR_ASSOC_P(str)) {
01377         /* already associated */
01378         rb_ary_concat(RSTRING(str)->as.heap.aux.shared, add);
01379     }
01380     else {
01381         if (STR_SHARED_P(str)) {
01382             VALUE assoc = RSTRING(str)->as.heap.aux.shared;
01383             str_make_independent(str);
01384             if (STR_ASSOC_P(assoc)) {
01385                 assoc = RSTRING(assoc)->as.heap.aux.shared;
01386                 rb_ary_concat(assoc, add);
01387                 add = assoc;
01388             }
01389         }
01390         else if (STR_EMBED_P(str)) {
01391             str_make_independent(str);
01392         }
01393         else if (RSTRING(str)->as.heap.aux.capa != RSTRING_LEN(str)) {
01394             RESIZE_CAPA(str, RSTRING_LEN(str));
01395         }
01396         FL_SET(str, STR_ASSOC);
01397         RBASIC(add)->klass = 0;
01398         RSTRING(str)->as.heap.aux.shared = add;
01399     }
01400 }
01401 
01402 VALUE
01403 rb_str_associated(VALUE str)
01404 {
01405     if (STR_SHARED_P(str)) str = RSTRING(str)->as.heap.aux.shared;
01406     if (STR_ASSOC_P(str)) {
01407         return RSTRING(str)->as.heap.aux.shared;
01408     }
01409     return Qfalse;
01410 }
01411 
01412 VALUE
01413 rb_string_value(volatile VALUE *ptr)
01414 {
01415     VALUE s = *ptr;
01416     if (TYPE(s) != T_STRING) {
01417         s = rb_str_to_str(s);
01418         *ptr = s;
01419     }
01420     return s;
01421 }
01422 
01423 char *
01424 rb_string_value_ptr(volatile VALUE *ptr)
01425 {
01426     VALUE str = rb_string_value(ptr);
01427     return RSTRING_PTR(str);
01428 }
01429 
01430 char *
01431 rb_string_value_cstr(volatile VALUE *ptr)
01432 {
01433     VALUE str = rb_string_value(ptr);
01434     char *s = RSTRING_PTR(str);
01435     long len = RSTRING_LEN(str);
01436 
01437     if (!s || memchr(s, 0, len)) {
01438         rb_raise(rb_eArgError, "string contains null byte");
01439     }
01440     if (s[len]) {
01441         rb_str_modify(str);
01442         s = RSTRING_PTR(str);
01443         s[RSTRING_LEN(str)] = 0;
01444     }
01445     return s;
01446 }
01447 
01448 VALUE
01449 rb_check_string_type(VALUE str)
01450 {
01451     str = rb_check_convert_type(str, T_STRING, "String", "to_str");
01452     return str;
01453 }
01454 
01455 /*
01456  *  call-seq:
01457  *     String.try_convert(obj) -> string or nil
01458  *
01459  *  Try to convert <i>obj</i> into a String, using to_str method.
01460  *  Returns converted string or nil if <i>obj</i> cannot be converted
01461  *  for any reason.
01462  *
01463  *     String.try_convert("str")     #=> "str"
01464  *     String.try_convert(/re/)      #=> nil
01465  */
01466 static VALUE
01467 rb_str_s_try_convert(VALUE dummy, VALUE str)
01468 {
01469     return rb_check_string_type(str);
01470 }
01471 
01472 static char*
01473 str_nth_len(const char *p, const char *e, long *nthp, rb_encoding *enc)
01474 {
01475     long nth = *nthp;
01476     if (rb_enc_mbmaxlen(enc) == 1) {
01477         p += nth;
01478     }
01479     else if (rb_enc_mbmaxlen(enc) == rb_enc_mbminlen(enc)) {
01480         p += nth * rb_enc_mbmaxlen(enc);
01481     }
01482     else if (rb_enc_asciicompat(enc)) {
01483         const char *p2, *e2;
01484         int n;
01485 
01486         while (p < e && 0 < nth) {
01487             e2 = p + nth;
01488             if (e < e2) {
01489                 *nthp = nth;
01490                 return (char *)e;
01491             }
01492             if (ISASCII(*p)) {
01493                 p2 = search_nonascii(p, e2);
01494                 if (!p2) {
01495                     *nthp = nth;
01496                     return (char *)e2;
01497                 }
01498                 nth -= p2 - p;
01499                 p = p2;
01500             }
01501             n = rb_enc_mbclen(p, e, enc);
01502             p += n;
01503             nth--;
01504         }
01505         *nthp = nth;
01506         if (nth != 0) {
01507             return (char *)e;
01508         }
01509         return (char *)p;
01510     }
01511     else {
01512         while (p < e && nth--) {
01513             p += rb_enc_mbclen(p, e, enc);
01514         }
01515     }
01516     if (p > e) p = e;
01517     *nthp = nth;
01518     return (char*)p;
01519 }
01520 
01521 char*
01522 rb_enc_nth(const char *p, const char *e, long nth, rb_encoding *enc)
01523 {
01524     return str_nth_len(p, e, &nth, enc);
01525 }
01526 
01527 static char*
01528 str_nth(const char *p, const char *e, long nth, rb_encoding *enc, int singlebyte)
01529 {
01530     if (singlebyte)
01531         p += nth;
01532     else {
01533         p = str_nth_len(p, e, &nth, enc);
01534     }
01535     if (!p) return 0;
01536     if (p > e) p = e;
01537     return (char *)p;
01538 }
01539 
01540 /* char offset to byte offset */
01541 static long
01542 str_offset(const char *p, const char *e, long nth, rb_encoding *enc, int singlebyte)
01543 {
01544     const char *pp = str_nth(p, e, nth, enc, singlebyte);
01545     if (!pp) return e - p;
01546     return pp - p;
01547 }
01548 
01549 long
01550 rb_str_offset(VALUE str, long pos)
01551 {
01552     return str_offset(RSTRING_PTR(str), RSTRING_END(str), pos,
01553                       STR_ENC_GET(str), single_byte_optimizable(str));
01554 }
01555 
01556 #ifdef NONASCII_MASK
01557 static char *
01558 str_utf8_nth(const char *p, const char *e, long *nthp)
01559 {
01560     long nth = *nthp;
01561     if ((int)SIZEOF_VALUE * 2 < e - p && (int)SIZEOF_VALUE * 2 < nth) {
01562         const VALUE *s, *t;
01563         const VALUE lowbits = sizeof(VALUE) - 1;
01564         s = (const VALUE*)(~lowbits & ((VALUE)p + lowbits));
01565         t = (const VALUE*)(~lowbits & (VALUE)e);
01566         while (p < (const char *)s) {
01567             if (is_utf8_lead_byte(*p)) nth--;
01568             p++;
01569         }
01570         do {
01571             nth -= count_utf8_lead_bytes_with_word(s);
01572             s++;
01573         } while (s < t && (int)sizeof(VALUE) <= nth);
01574         p = (char *)s;
01575     }
01576     while (p < e) {
01577         if (is_utf8_lead_byte(*p)) {
01578             if (nth == 0) break;
01579             nth--;
01580         }
01581         p++;
01582     }
01583     *nthp = nth;
01584     return (char *)p;
01585 }
01586 
01587 static long
01588 str_utf8_offset(const char *p, const char *e, long nth)
01589 {
01590     const char *pp = str_utf8_nth(p, e, &nth);
01591     return pp - p;
01592 }
01593 #endif
01594 
01595 /* byte offset to char offset */
01596 long
01597 rb_str_sublen(VALUE str, long pos)
01598 {
01599     if (single_byte_optimizable(str) || pos < 0)
01600         return pos;
01601     else {
01602         char *p = RSTRING_PTR(str);
01603         return enc_strlen(p, p + pos, STR_ENC_GET(str), ENC_CODERANGE(str));
01604     }
01605 }
01606 
01607 VALUE
01608 rb_str_subseq(VALUE str, long beg, long len)
01609 {
01610     VALUE str2;
01611 
01612     if (RSTRING_LEN(str) == beg + len &&
01613         RSTRING_EMBED_LEN_MAX < len) {
01614         str2 = rb_str_new_shared(rb_str_new_frozen(str));
01615         rb_str_drop_bytes(str2, beg);
01616     }
01617     else {
01618         str2 = rb_str_new5(str, RSTRING_PTR(str)+beg, len);
01619     }
01620 
01621     rb_enc_cr_str_copy_for_substr(str2, str);
01622     OBJ_INFECT(str2, str);
01623 
01624     return str2;
01625 }
01626 
01627 VALUE
01628 rb_str_substr(VALUE str, long beg, long len)
01629 {
01630     rb_encoding *enc = STR_ENC_GET(str);
01631     VALUE str2;
01632     char *p, *s = RSTRING_PTR(str), *e = s + RSTRING_LEN(str);
01633 
01634     if (len < 0) return Qnil;
01635     if (!RSTRING_LEN(str)) {
01636         len = 0;
01637     }
01638     if (single_byte_optimizable(str)) {
01639         if (beg > RSTRING_LEN(str)) return Qnil;
01640         if (beg < 0) {
01641             beg += RSTRING_LEN(str);
01642             if (beg < 0) return Qnil;
01643         }
01644         if (beg + len > RSTRING_LEN(str))
01645             len = RSTRING_LEN(str) - beg;
01646         if (len <= 0) {
01647             len = 0;
01648             p = 0;
01649         }
01650         else
01651             p = s + beg;
01652         goto sub;
01653     }
01654     if (beg < 0) {
01655         if (len > -beg) len = -beg;
01656         if (-beg * rb_enc_mbmaxlen(enc) < RSTRING_LEN(str) / 8) {
01657             beg = -beg;
01658             while (beg-- > len && (e = rb_enc_prev_char(s, e, e, enc)) != 0);
01659             p = e;
01660             if (!p) return Qnil;
01661             while (len-- > 0 && (p = rb_enc_prev_char(s, p, e, enc)) != 0);
01662             if (!p) return Qnil;
01663             len = e - p;
01664             goto sub;
01665         }
01666         else {
01667             beg += str_strlen(str, enc);
01668             if (beg < 0) return Qnil;
01669         }
01670     }
01671     else if (beg > 0 && beg > RSTRING_LEN(str)) {
01672         return Qnil;
01673     }
01674     if (len == 0) {
01675         if (beg > str_strlen(str, enc)) return Qnil;
01676         p = 0;
01677     }
01678 #ifdef NONASCII_MASK
01679     else if (ENC_CODERANGE(str) == ENC_CODERANGE_VALID &&
01680         enc == rb_utf8_encoding()) {
01681         p = str_utf8_nth(s, e, &beg);
01682         if (beg > 0) return Qnil;
01683         len = str_utf8_offset(p, e, len);
01684     }
01685 #endif
01686     else if (rb_enc_mbmaxlen(enc) == rb_enc_mbminlen(enc)) {
01687         int char_sz = rb_enc_mbmaxlen(enc);
01688 
01689         p = s + beg * char_sz;
01690         if (p > e) {
01691             return Qnil;
01692         }
01693         else if (len * char_sz > e - p)
01694             len = e - p;
01695         else
01696             len *= char_sz;
01697     }
01698     else if ((p = str_nth_len(s, e, &beg, enc)) == e) {
01699         if (beg > 0) return Qnil;
01700         len = 0;
01701     }
01702     else {
01703         len = str_offset(p, e, len, enc, 0);
01704     }
01705   sub:
01706     if (len > RSTRING_EMBED_LEN_MAX && beg + len == RSTRING_LEN(str)) {
01707         str2 = rb_str_new4(str);
01708         str2 = str_new3(rb_obj_class(str2), str2);
01709         RSTRING(str2)->as.heap.ptr += RSTRING(str2)->as.heap.len - len;
01710         RSTRING(str2)->as.heap.len = len;
01711     }
01712     else {
01713         str2 = rb_str_new5(str, p, len);
01714         rb_enc_cr_str_copy_for_substr(str2, str);
01715         OBJ_INFECT(str2, str);
01716     }
01717 
01718     return str2;
01719 }
01720 
01721 VALUE
01722 rb_str_freeze(VALUE str)
01723 {
01724     if (STR_ASSOC_P(str)) {
01725         VALUE ary = RSTRING(str)->as.heap.aux.shared;
01726         OBJ_FREEZE(ary);
01727     }
01728     return rb_obj_freeze(str);
01729 }
01730 
01731 RUBY_ALIAS_FUNCTION(rb_str_dup_frozen(VALUE str), rb_str_new_frozen, (str))
01732 #define rb_str_dup_frozen rb_str_new_frozen
01733 
01734 VALUE
01735 rb_str_locktmp(VALUE str)
01736 {
01737     if (FL_TEST(str, STR_TMPLOCK)) {
01738         rb_raise(rb_eRuntimeError, "temporal locking already locked string");
01739     }
01740     FL_SET(str, STR_TMPLOCK);
01741     return str;
01742 }
01743 
01744 VALUE
01745 rb_str_unlocktmp(VALUE str)
01746 {
01747     if (!FL_TEST(str, STR_TMPLOCK)) {
01748         rb_raise(rb_eRuntimeError, "temporal unlocking already unlocked string");
01749     }
01750     FL_UNSET(str, STR_TMPLOCK);
01751     return str;
01752 }
01753 
01754 void
01755 rb_str_set_len(VALUE str, long len)
01756 {
01757     long capa;
01758 
01759     str_modifiable(str);
01760     if (STR_SHARED_P(str)) {
01761         rb_raise(rb_eRuntimeError, "can't set length of shared string");
01762     }
01763     if (len > (capa = (long)rb_str_capacity(str))) {
01764         rb_bug("probable buffer overflow: %ld for %ld", len, capa);
01765     }
01766     STR_SET_LEN(str, len);
01767     RSTRING_PTR(str)[len] = '\0';
01768 }
01769 
01770 VALUE
01771 rb_str_resize(VALUE str, long len)
01772 {
01773     long slen;
01774     int independent;
01775 
01776     if (len < 0) {
01777         rb_raise(rb_eArgError, "negative string size (or size too big)");
01778     }
01779 
01780     independent = str_independent(str);
01781     ENC_CODERANGE_CLEAR(str);
01782     slen = RSTRING_LEN(str);
01783     if (len != slen) {
01784         if (STR_EMBED_P(str)) {
01785             if (len <= RSTRING_EMBED_LEN_MAX) {
01786                 STR_SET_EMBED_LEN(str, len);
01787                 RSTRING(str)->as.ary[len] = '\0';
01788                 return str;
01789             }
01790             str_make_independent_expand(str, len - slen);
01791             STR_SET_NOEMBED(str);
01792         }
01793         else if (len <= RSTRING_EMBED_LEN_MAX) {
01794             char *ptr = RSTRING(str)->as.heap.ptr;
01795             STR_SET_EMBED(str);
01796             if (slen > len) slen = len;
01797             if (slen > 0) MEMCPY(RSTRING(str)->as.ary, ptr, char, slen);
01798             RSTRING(str)->as.ary[len] = '\0';
01799             STR_SET_EMBED_LEN(str, len);
01800             if (independent) xfree(ptr);
01801             return str;
01802         }
01803         else if (!independent) {
01804             str_make_independent_expand(str, len - slen);
01805         }
01806         else if (slen < len || slen - len > 1024) {
01807             REALLOC_N(RSTRING(str)->as.heap.ptr, char, len+1);
01808         }
01809         if (!STR_NOCAPA_P(str)) {
01810             RSTRING(str)->as.heap.aux.capa = len;
01811         }
01812         RSTRING(str)->as.heap.len = len;
01813         RSTRING(str)->as.heap.ptr[len] = '\0';  /* sentinel */
01814     }
01815     return str;
01816 }
01817 
01818 static VALUE
01819 str_buf_cat(VALUE str, const char *ptr, long len)
01820 {
01821     long capa, total, off = -1;
01822 
01823     if (ptr >= RSTRING_PTR(str) && ptr <= RSTRING_END(str)) {
01824         off = ptr - RSTRING_PTR(str);
01825     }
01826     rb_str_modify(str);
01827     if (len == 0) return 0;
01828     if (STR_ASSOC_P(str)) {
01829         FL_UNSET(str, STR_ASSOC);
01830         capa = RSTRING(str)->as.heap.aux.capa = RSTRING_LEN(str);
01831     }
01832     else if (STR_EMBED_P(str)) {
01833         capa = RSTRING_EMBED_LEN_MAX;
01834     }
01835     else {
01836         capa = RSTRING(str)->as.heap.aux.capa;
01837     }
01838     if (RSTRING_LEN(str) >= LONG_MAX - len) {
01839         rb_raise(rb_eArgError, "string sizes too big");
01840     }
01841     total = RSTRING_LEN(str)+len;
01842     if (capa <= total) {
01843         while (total > capa) {
01844             if (capa + 1 >= LONG_MAX / 2) {
01845                 capa = (total + 4095) / 4096;
01846                 break;
01847             }
01848             capa = (capa + 1) * 2;
01849         }
01850         RESIZE_CAPA(str, capa);
01851     }
01852     if (off != -1) {
01853         ptr = RSTRING_PTR(str) + off;
01854     }
01855     memcpy(RSTRING_PTR(str) + RSTRING_LEN(str), ptr, len);
01856     STR_SET_LEN(str, total);
01857     RSTRING_PTR(str)[total] = '\0'; /* sentinel */
01858 
01859     return str;
01860 }
01861 
01862 #define str_buf_cat2(str, ptr) str_buf_cat((str), (ptr), strlen(ptr))
01863 
01864 VALUE
01865 rb_str_buf_cat(VALUE str, const char *ptr, long len)
01866 {
01867     if (len == 0) return str;
01868     if (len < 0) {
01869         rb_raise(rb_eArgError, "negative string size (or size too big)");
01870     }
01871     return str_buf_cat(str, ptr, len);
01872 }
01873 
01874 VALUE
01875 rb_str_buf_cat2(VALUE str, const char *ptr)
01876 {
01877     return rb_str_buf_cat(str, ptr, strlen(ptr));
01878 }
01879 
01880 VALUE
01881 rb_str_cat(VALUE str, const char *ptr, long len)
01882 {
01883     if (len < 0) {
01884         rb_raise(rb_eArgError, "negative string size (or size too big)");
01885     }
01886     if (STR_ASSOC_P(str)) {
01887         char *p;
01888         rb_str_modify_expand(str, len);
01889         p = RSTRING(str)->as.heap.ptr;
01890         memcpy(p + RSTRING(str)->as.heap.len, ptr, len);
01891         len = RSTRING(str)->as.heap.len += len;
01892         p[len] = '\0'; /* sentinel */
01893         return str;
01894     }
01895 
01896     return rb_str_buf_cat(str, ptr, len);
01897 }
01898 
01899 VALUE
01900 rb_str_cat2(VALUE str, const char *ptr)
01901 {
01902     return rb_str_cat(str, ptr, strlen(ptr));
01903 }
01904 
01905 static VALUE
01906 rb_enc_cr_str_buf_cat(VALUE str, const char *ptr, long len,
01907     int ptr_encindex, int ptr_cr, int *ptr_cr_ret)
01908 {
01909     int str_encindex = ENCODING_GET(str);
01910     int res_encindex;
01911     int str_cr, res_cr;
01912 
01913     str_cr = ENC_CODERANGE(str);
01914 
01915     if (str_encindex == ptr_encindex) {
01916         if (str_cr == ENC_CODERANGE_UNKNOWN)
01917             ptr_cr = ENC_CODERANGE_UNKNOWN;
01918         else if (ptr_cr == ENC_CODERANGE_UNKNOWN) {
01919             ptr_cr = coderange_scan(ptr, len, rb_enc_from_index(ptr_encindex));
01920         }
01921     }
01922     else {
01923         rb_encoding *str_enc = rb_enc_from_index(str_encindex);
01924         rb_encoding *ptr_enc = rb_enc_from_index(ptr_encindex);
01925         if (!rb_enc_asciicompat(str_enc) || !rb_enc_asciicompat(ptr_enc)) {
01926             if (len == 0)
01927                 return str;
01928             if (RSTRING_LEN(str) == 0) {
01929                 rb_str_buf_cat(str, ptr, len);
01930                 ENCODING_CODERANGE_SET(str, ptr_encindex, ptr_cr);
01931                 return str;
01932             }
01933             goto incompatible;
01934         }
01935         if (ptr_cr == ENC_CODERANGE_UNKNOWN) {
01936             ptr_cr = coderange_scan(ptr, len, ptr_enc);
01937         }
01938         if (str_cr == ENC_CODERANGE_UNKNOWN) {
01939             if (ENCODING_IS_ASCII8BIT(str) || ptr_cr != ENC_CODERANGE_7BIT) {
01940                 str_cr = rb_enc_str_coderange(str);
01941             }
01942         }
01943     }
01944     if (ptr_cr_ret)
01945         *ptr_cr_ret = ptr_cr;
01946 
01947     if (str_encindex != ptr_encindex &&
01948         str_cr != ENC_CODERANGE_7BIT &&
01949         ptr_cr != ENC_CODERANGE_7BIT) {
01950       incompatible:
01951         rb_raise(rb_eEncCompatError, "incompatible character encodings: %s and %s",
01952             rb_enc_name(rb_enc_from_index(str_encindex)),
01953             rb_enc_name(rb_enc_from_index(ptr_encindex)));
01954     }
01955 
01956     if (str_cr == ENC_CODERANGE_UNKNOWN) {
01957         res_encindex = str_encindex;
01958         res_cr = ENC_CODERANGE_UNKNOWN;
01959     }
01960     else if (str_cr == ENC_CODERANGE_7BIT) {
01961         if (ptr_cr == ENC_CODERANGE_7BIT) {
01962             res_encindex = str_encindex;
01963             res_cr = ENC_CODERANGE_7BIT;
01964         }
01965         else {
01966             res_encindex = ptr_encindex;
01967             res_cr = ptr_cr;
01968         }
01969     }
01970     else if (str_cr == ENC_CODERANGE_VALID) {
01971         res_encindex = str_encindex;
01972         if (ptr_cr == ENC_CODERANGE_7BIT || ptr_cr == ENC_CODERANGE_VALID)
01973             res_cr = str_cr;
01974         else
01975             res_cr = ptr_cr;
01976     }
01977     else { /* str_cr == ENC_CODERANGE_BROKEN */
01978         res_encindex = str_encindex;
01979         res_cr = str_cr;
01980         if (0 < len) res_cr = ENC_CODERANGE_UNKNOWN;
01981     }
01982 
01983     if (len < 0) {
01984         rb_raise(rb_eArgError, "negative string size (or size too big)");
01985     }
01986     str_buf_cat(str, ptr, len);
01987     ENCODING_CODERANGE_SET(str, res_encindex, res_cr);
01988     return str;
01989 }
01990 
01991 VALUE
01992 rb_enc_str_buf_cat(VALUE str, const char *ptr, long len, rb_encoding *ptr_enc)
01993 {
01994     return rb_enc_cr_str_buf_cat(str, ptr, len,
01995         rb_enc_to_index(ptr_enc), ENC_CODERANGE_UNKNOWN, NULL);
01996 }
01997 
01998 VALUE
01999 rb_str_buf_cat_ascii(VALUE str, const char *ptr)
02000 {
02001     /* ptr must reference NUL terminated ASCII string. */
02002     int encindex = ENCODING_GET(str);
02003     rb_encoding *enc = rb_enc_from_index(encindex);
02004     if (rb_enc_asciicompat(enc)) {
02005         return rb_enc_cr_str_buf_cat(str, ptr, strlen(ptr),
02006             encindex, ENC_CODERANGE_7BIT, 0);
02007     }
02008     else {
02009         char *buf = ALLOCA_N(char, rb_enc_mbmaxlen(enc));
02010         while (*ptr) {
02011             unsigned int c = (unsigned char)*ptr;
02012             int len = rb_enc_codelen(c, enc);
02013             rb_enc_mbcput(c, buf, enc);
02014             rb_enc_cr_str_buf_cat(str, buf, len,
02015                 encindex, ENC_CODERANGE_VALID, 0);
02016             ptr++;
02017         }
02018         return str;
02019     }
02020 }
02021 
02022 VALUE
02023 rb_str_buf_append(VALUE str, VALUE str2)
02024 {
02025     int str2_cr;
02026 
02027     str2_cr = ENC_CODERANGE(str2);
02028 
02029     rb_enc_cr_str_buf_cat(str, RSTRING_PTR(str2), RSTRING_LEN(str2),
02030         ENCODING_GET(str2), str2_cr, &str2_cr);
02031 
02032     OBJ_INFECT(str, str2);
02033     ENC_CODERANGE_SET(str2, str2_cr);
02034 
02035     return str;
02036 }
02037 
02038 VALUE
02039 rb_str_append(VALUE str, VALUE str2)
02040 {
02041     rb_encoding *enc;
02042     int cr, cr2;
02043     long len2;
02044 
02045     StringValue(str2);
02046     if ((len2 = RSTRING_LEN(str2)) > 0 && STR_ASSOC_P(str)) {
02047         long len = RSTRING_LEN(str) + len2;
02048         enc = rb_enc_check(str, str2);
02049         cr = ENC_CODERANGE(str);
02050         if ((cr2 = ENC_CODERANGE(str2)) > cr) cr = cr2;
02051         rb_str_modify_expand(str, len2);
02052         memcpy(RSTRING(str)->as.heap.ptr + RSTRING(str)->as.heap.len,
02053                RSTRING_PTR(str2), len2+1);
02054         RSTRING(str)->as.heap.len = len;
02055         rb_enc_associate(str, enc);
02056         ENC_CODERANGE_SET(str, cr);
02057         OBJ_INFECT(str, str2);
02058         return str;
02059     }
02060     return rb_str_buf_append(str, str2);
02061 }
02062 
02063 /*
02064  *  call-seq:
02065  *     str << integer       -> str
02066  *     str.concat(integer)  -> str
02067  *     str << obj           -> str
02068  *     str.concat(obj)      -> str
02069  *
02070  *  Append---Concatenates the given object to <i>str</i>. If the object is a
02071  *  <code>Integer</code>, it is considered as a codepoint, and is converted
02072  *  to a character before concatenation.
02073  *
02074  *     a = "hello "
02075  *     a << "world"   #=> "hello world"
02076  *     a.concat(33)   #=> "hello world!"
02077  */
02078 
02079 VALUE
02080 rb_str_concat(VALUE str1, VALUE str2)
02081 {
02082     unsigned int code;
02083     rb_encoding *enc = STR_ENC_GET(str1);
02084 
02085     if (FIXNUM_P(str2) || TYPE(str2) == T_BIGNUM) {
02086         if (rb_num_to_uint(str2, &code) == 0) {
02087         }
02088         else if (FIXNUM_P(str2)) {
02089             rb_raise(rb_eRangeError, "%ld out of char range", FIX2LONG(str2));
02090         }
02091         else {
02092             rb_raise(rb_eRangeError, "bignum out of char range");
02093         }
02094     }
02095     else {
02096         return rb_str_append(str1, str2);
02097     }
02098 
02099     if (enc == rb_usascii_encoding()) {
02100         /* US-ASCII automatically extended to ASCII-8BIT */
02101         char buf[1] = {(char)code};
02102         if (code > 0xFF) {
02103             rb_raise(rb_eRangeError, "%u out of char range", code);
02104         }
02105         rb_str_cat(str1, buf, 1);
02106         if (code > 127) {
02107             rb_enc_associate(str1, rb_ascii8bit_encoding());
02108             ENC_CODERANGE_SET(str1, ENC_CODERANGE_VALID);
02109         }
02110     }
02111     else {
02112         long pos = RSTRING_LEN(str1);
02113         int cr = ENC_CODERANGE(str1);
02114         int len;
02115         char *buf;
02116 
02117         switch (len = rb_enc_codelen(code, enc)) {
02118           case ONIGERR_INVALID_CODE_POINT_VALUE:
02119             rb_raise(rb_eRangeError, "invalid codepoint 0x%X in %s", code, rb_enc_name(enc));
02120             break;
02121           case ONIGERR_TOO_BIG_WIDE_CHAR_VALUE:
02122           case 0:
02123             rb_raise(rb_eRangeError, "%u out of char range", code);
02124             break;
02125         }
02126         buf = ALLOCA_N(char, len + 1);
02127         rb_enc_mbcput(code, buf, enc);
02128         if (rb_enc_precise_mbclen(buf, buf + len + 1, enc) != len) {
02129             rb_raise(rb_eRangeError, "invalid codepoint 0x%X in %s", code, rb_enc_name(enc));
02130         }
02131         rb_str_resize(str1, pos+len);
02132         strncpy(RSTRING_PTR(str1) + pos, buf, len);
02133         if (cr == ENC_CODERANGE_7BIT && code > 127)
02134             cr = ENC_CODERANGE_VALID;
02135         ENC_CODERANGE_SET(str1, cr);
02136     }
02137     return str1;
02138 }
02139 
02140 /*
02141  *  call-seq:
02142  *     str.prepend(other_str)  -> str
02143  *
02144  *  Prepend---Prepend the given string to <i>str</i>.
02145  *
02146  *  a = "world"
02147  *  a.prepend("hello ") #=> "hello world"
02148  *  a                   #=> "hello world"
02149  */
02150 
02151 static VALUE
02152 rb_str_prepend(VALUE str, VALUE str2)
02153 {
02154     StringValue(str2);
02155     StringValue(str);
02156     rb_str_update(str, 0L, 0L, str2);
02157     return str;
02158 }
02159 
02160 st_index_t
02161 rb_memhash(const void *ptr, long len)
02162 {
02163     return st_hash(ptr, len, rb_hash_start((st_index_t)len));
02164 }
02165 
02166 st_index_t
02167 rb_str_hash(VALUE str)
02168 {
02169     int e = ENCODING_GET(str);
02170     if (e && rb_enc_str_coderange(str) == ENC_CODERANGE_7BIT) {
02171         e = 0;
02172     }
02173     return rb_memhash((const void *)RSTRING_PTR(str), RSTRING_LEN(str)) ^ e;
02174 }
02175 
02176 int
02177 rb_str_hash_cmp(VALUE str1, VALUE str2)
02178 {
02179     long len;
02180 
02181     if (!rb_str_comparable(str1, str2)) return 1;
02182     if (RSTRING_LEN(str1) == (len = RSTRING_LEN(str2)) &&
02183         memcmp(RSTRING_PTR(str1), RSTRING_PTR(str2), len) == 0) {
02184         return 0;
02185     }
02186     return 1;
02187 }
02188 
02189 /*
02190  * call-seq:
02191  *    str.hash   -> fixnum
02192  *
02193  * Return a hash based on the string's length and content.
02194  */
02195 
02196 static VALUE
02197 rb_str_hash_m(VALUE str)
02198 {
02199     st_index_t hval = rb_str_hash(str);
02200     return INT2FIX(hval);
02201 }
02202 
02203 #define lesser(a,b) (((a)>(b))?(b):(a))
02204 
02205 int
02206 rb_str_comparable(VALUE str1, VALUE str2)
02207 {
02208     int idx1, idx2;
02209     int rc1, rc2;
02210 
02211     if (RSTRING_LEN(str1) == 0) return TRUE;
02212     if (RSTRING_LEN(str2) == 0) return TRUE;
02213     idx1 = ENCODING_GET(str1);
02214     idx2 = ENCODING_GET(str2);
02215     if (idx1 == idx2) return TRUE;
02216     rc1 = rb_enc_str_coderange(str1);
02217     rc2 = rb_enc_str_coderange(str2);
02218     if (rc1 == ENC_CODERANGE_7BIT) {
02219         if (rc2 == ENC_CODERANGE_7BIT) return TRUE;
02220         if (rb_enc_asciicompat(rb_enc_from_index(idx2)))
02221             return TRUE;
02222     }
02223     if (rc2 == ENC_CODERANGE_7BIT) {
02224         if (rb_enc_asciicompat(rb_enc_from_index(idx1)))
02225             return TRUE;
02226     }
02227     return FALSE;
02228 }
02229 
02230 int
02231 rb_str_cmp(VALUE str1, VALUE str2)
02232 {
02233     long len1, len2;
02234     const char *ptr1, *ptr2;
02235     int retval;
02236 
02237     if (str1 == str2) return 0;
02238     RSTRING_GETMEM(str1, ptr1, len1);
02239     RSTRING_GETMEM(str2, ptr2, len2);
02240     if (ptr1 == ptr2 || (retval = memcmp(ptr1, ptr2, lesser(len1, len2))) == 0) {
02241         if (len1 == len2) {
02242             if (!rb_str_comparable(str1, str2)) {
02243                 if (ENCODING_GET(str1) > ENCODING_GET(str2))
02244                     return 1;
02245                 return -1;
02246             }
02247             return 0;
02248         }
02249         if (len1 > len2) return 1;
02250         return -1;
02251     }
02252     if (retval > 0) return 1;
02253     return -1;
02254 }
02255 
02256 /* expect tail call optimization */
02257 static VALUE
02258 str_eql(const VALUE str1, const VALUE str2)
02259 {
02260     const long len = RSTRING_LEN(str1);
02261     const char *ptr1, *ptr2;
02262 
02263     if (len != RSTRING_LEN(str2)) return Qfalse;
02264     if (!rb_str_comparable(str1, str2)) return Qfalse;
02265     if ((ptr1 = RSTRING_PTR(str1)) == (ptr2 = RSTRING_PTR(str2)))
02266         return Qtrue;
02267     if (memcmp(ptr1, ptr2, len) == 0)
02268         return Qtrue;
02269     return Qfalse;
02270 }
02271 /*
02272  *  call-seq:
02273  *     str == obj   -> true or false
02274  *
02275  *  Equality---If <i>obj</i> is not a <code>String</code>, returns
02276  *  <code>false</code>. Otherwise, returns <code>true</code> if <i>str</i>
02277  *  <code><=></code> <i>obj</i> returns zero.
02278  */
02279 
02280 VALUE
02281 rb_str_equal(VALUE str1, VALUE str2)
02282 {
02283     if (str1 == str2) return Qtrue;
02284     if (TYPE(str2) != T_STRING) {
02285         if (!rb_respond_to(str2, rb_intern("to_str"))) {
02286             return Qfalse;
02287         }
02288         return rb_equal(str2, str1);
02289     }
02290     return str_eql(str1, str2);
02291 }
02292 
02293 /*
02294  * call-seq:
02295  *   str.eql?(other)   -> true or false
02296  *
02297  * Two strings are equal if they have the same length and content.
02298  */
02299 
02300 static VALUE
02301 rb_str_eql(VALUE str1, VALUE str2)
02302 {
02303     if (str1 == str2) return Qtrue;
02304     if (TYPE(str2) != T_STRING) return Qfalse;
02305     return str_eql(str1, str2);
02306 }
02307 
02308 /*
02309  *  call-seq:
02310  *     str <=> other_str   -> -1, 0, +1 or nil
02311  *
02312  *  Comparison---Returns -1 if <i>other_str</i> is greater than, 0 if
02313  *  <i>other_str</i> is equal to, and +1 if <i>other_str</i> is less than
02314  *  <i>str</i>. If the strings are of different lengths, and the strings are
02315  *  equal when compared up to the shortest length, then the longer string is
02316  *  considered greater than the shorter one. In older versions of Ruby, setting
02317  *  <code>$=</code> allowed case-insensitive comparisons; this is now deprecated
02318  *  in favor of using <code>String#casecmp</code>.
02319  *
02320  *  <code><=></code> is the basis for the methods <code><</code>,
02321  *  <code><=</code>, <code>></code>, <code>>=</code>, and <code>between?</code>,
02322  *  included from module <code>Comparable</code>.  The method
02323  *  <code>String#==</code> does not use <code>Comparable#==</code>.
02324  *
02325  *     "abcdef" <=> "abcde"     #=> 1
02326  *     "abcdef" <=> "abcdef"    #=> 0
02327  *     "abcdef" <=> "abcdefg"   #=> -1
02328  *     "abcdef" <=> "ABCDEF"    #=> 1
02329  */
02330 
02331 static VALUE
02332 rb_str_cmp_m(VALUE str1, VALUE str2)
02333 {
02334     long result;
02335 
02336     if (TYPE(str2) != T_STRING) {
02337         if (!rb_respond_to(str2, rb_intern("to_str"))) {
02338             return Qnil;
02339         }
02340         else if (!rb_respond_to(str2, rb_intern("<=>"))) {
02341             return Qnil;
02342         }
02343         else {
02344             VALUE tmp = rb_funcall(str2, rb_intern("<=>"), 1, str1);
02345 
02346             if (NIL_P(tmp)) return Qnil;
02347             if (!FIXNUM_P(tmp)) {
02348                 return rb_funcall(LONG2FIX(0), '-', 1, tmp);
02349             }
02350             result = -FIX2LONG(tmp);
02351         }
02352     }
02353     else {
02354         result = rb_str_cmp(str1, str2);
02355     }
02356     return LONG2NUM(result);
02357 }
02358 
02359 /*
02360  *  call-seq:
02361  *     str.casecmp(other_str)   -> -1, 0, +1 or nil
02362  *
02363  *  Case-insensitive version of <code>String#<=></code>.
02364  *
02365  *     "abcdef".casecmp("abcde")     #=> 1
02366  *     "aBcDeF".casecmp("abcdef")    #=> 0
02367  *     "abcdef".casecmp("abcdefg")   #=> -1
02368  *     "abcdef".casecmp("ABCDEF")    #=> 0
02369  */
02370 
02371 static VALUE
02372 rb_str_casecmp(VALUE str1, VALUE str2)
02373 {
02374     long len;
02375     rb_encoding *enc;
02376     char *p1, *p1end, *p2, *p2end;
02377 
02378     StringValue(str2);
02379     enc = rb_enc_compatible(str1, str2);
02380     if (!enc) {
02381         return Qnil;
02382     }
02383 
02384     p1 = RSTRING_PTR(str1); p1end = RSTRING_END(str1);
02385     p2 = RSTRING_PTR(str2); p2end = RSTRING_END(str2);
02386     if (single_byte_optimizable(str1) && single_byte_optimizable(str2)) {
02387         while (p1 < p1end && p2 < p2end) {
02388             if (*p1 != *p2) {
02389                 unsigned int c1 = TOUPPER(*p1 & 0xff);
02390                 unsigned int c2 = TOUPPER(*p2 & 0xff);
02391                 if (c1 != c2)
02392                     return INT2FIX(c1 < c2 ? -1 : 1);
02393             }
02394             p1++;
02395             p2++;
02396         }
02397     }
02398     else {
02399         while (p1 < p1end && p2 < p2end) {
02400             int l1, c1 = rb_enc_ascget(p1, p1end, &l1, enc);
02401             int l2, c2 = rb_enc_ascget(p2, p2end, &l2, enc);
02402 
02403             if (0 <= c1 && 0 <= c2) {
02404                 c1 = TOUPPER(c1);
02405                 c2 = TOUPPER(c2);
02406                 if (c1 != c2)
02407                     return INT2FIX(c1 < c2 ? -1 : 1);
02408             }
02409             else {
02410                 int r;
02411                 l1 = rb_enc_mbclen(p1, p1end, enc);
02412                 l2 = rb_enc_mbclen(p2, p2end, enc);
02413                 len = l1 < l2 ? l1 : l2;
02414                 r = memcmp(p1, p2, len);
02415                 if (r != 0)
02416                     return INT2FIX(r < 0 ? -1 : 1);
02417                 if (l1 != l2)
02418                     return INT2FIX(l1 < l2 ? -1 : 1);
02419             }
02420             p1 += l1;
02421             p2 += l2;
02422         }
02423     }
02424     if (RSTRING_LEN(str1) == RSTRING_LEN(str2)) return INT2FIX(0);
02425     if (RSTRING_LEN(str1) > RSTRING_LEN(str2)) return INT2FIX(1);
02426     return INT2FIX(-1);
02427 }
02428 
02429 static long
02430 rb_str_index(VALUE str, VALUE sub, long offset)
02431 {
02432     long pos;
02433     char *s, *sptr, *e;
02434     long len, slen;
02435     rb_encoding *enc;
02436 
02437     enc = rb_enc_check(str, sub);
02438     if (is_broken_string(sub)) {
02439         return -1;
02440     }
02441     len = str_strlen(str, enc);
02442     slen = str_strlen(sub, enc);
02443     if (offset < 0) {
02444         offset += len;
02445         if (offset < 0) return -1;
02446     }
02447     if (len - offset < slen) return -1;
02448     s = RSTRING_PTR(str);
02449     e = s + RSTRING_LEN(str);
02450     if (offset) {
02451         offset = str_offset(s, RSTRING_END(str), offset, enc, single_byte_optimizable(str));
02452         s += offset;
02453     }
02454     if (slen == 0) return offset;
02455     /* need proceed one character at a time */
02456     sptr = RSTRING_PTR(sub);
02457     slen = RSTRING_LEN(sub);
02458     len = RSTRING_LEN(str) - offset;
02459     for (;;) {
02460         char *t;
02461         pos = rb_memsearch(sptr, slen, s, len, enc);
02462         if (pos < 0) return pos;
02463         t = rb_enc_right_char_head(s, s+pos, e, enc);
02464         if (t == s + pos) break;
02465         if ((len -= t - s) <= 0) return -1;
02466         offset += t - s;
02467         s = t;
02468     }
02469     return pos + offset;
02470 }
02471 
02472 
02473 /*
02474  *  call-seq:
02475  *     str.index(substring [, offset])   -> fixnum or nil
02476  *     str.index(regexp [, offset])      -> fixnum or nil
02477  *
02478  *  Returns the index of the first occurrence of the given <i>substring</i> or
02479  *  pattern (<i>regexp</i>) in <i>str</i>. Returns <code>nil</code> if not
02480  *  found. If the second parameter is present, it specifies the position in the
02481  *  string to begin the search.
02482  *
02483  *     "hello".index('e')             #=> 1
02484  *     "hello".index('lo')            #=> 3
02485  *     "hello".index('a')             #=> nil
02486  *     "hello".index(?e)              #=> 1
02487  *     "hello".index(/[aeiou]/, -3)   #=> 4
02488  */
02489 
02490 static VALUE
02491 rb_str_index_m(int argc, VALUE *argv, VALUE str)
02492 {
02493     VALUE sub;
02494     VALUE initpos;
02495     long pos;
02496 
02497     if (rb_scan_args(argc, argv, "11", &sub, &initpos) == 2) {
02498         pos = NUM2LONG(initpos);
02499     }
02500     else {
02501         pos = 0;
02502     }
02503     if (pos < 0) {
02504         pos += str_strlen(str, STR_ENC_GET(str));
02505         if (pos < 0) {
02506             if (TYPE(sub) == T_REGEXP) {
02507                 rb_backref_set(Qnil);
02508             }
02509             return Qnil;
02510         }
02511     }
02512 
02513     switch (TYPE(sub)) {
02514       case T_REGEXP:
02515         if (pos > str_strlen(str, STR_ENC_GET(str)))
02516             return Qnil;
02517         pos = str_offset(RSTRING_PTR(str), RSTRING_END(str), pos,
02518                          rb_enc_check(str, sub), single_byte_optimizable(str));
02519 
02520         pos = rb_reg_search(sub, str, pos, 0);
02521         pos = rb_str_sublen(str, pos);
02522         break;
02523 
02524       default: {
02525         VALUE tmp;
02526 
02527         tmp = rb_check_string_type(sub);
02528         if (NIL_P(tmp)) {
02529             rb_raise(rb_eTypeError, "type mismatch: %s given",
02530                      rb_obj_classname(sub));
02531         }
02532         sub = tmp;
02533       }
02534         /* fall through */
02535       case T_STRING:
02536         pos = rb_str_index(str, sub, pos);
02537         pos = rb_str_sublen(str, pos);
02538         break;
02539     }
02540 
02541     if (pos == -1) return Qnil;
02542     return LONG2NUM(pos);
02543 }
02544 
02545 static long
02546 rb_str_rindex(VALUE str, VALUE sub, long pos)
02547 {
02548     long len, slen;
02549     char *s, *sbeg, *e, *t;
02550     rb_encoding *enc;
02551     int singlebyte = single_byte_optimizable(str);
02552 
02553     enc = rb_enc_check(str, sub);
02554     if (is_broken_string(sub)) {
02555         return -1;
02556     }
02557     len = str_strlen(str, enc);
02558     slen = str_strlen(sub, enc);
02559     /* substring longer than string */
02560     if (len < slen) return -1;
02561     if (len - pos < slen) {
02562         pos = len - slen;
02563     }
02564     if (len == 0) {
02565         return pos;
02566     }
02567     sbeg = RSTRING_PTR(str);
02568     e = RSTRING_END(str);
02569     t = RSTRING_PTR(sub);
02570     slen = RSTRING_LEN(sub);
02571     s = str_nth(sbeg, e, pos, enc, singlebyte);
02572     while (s) {
02573         if (memcmp(s, t, slen) == 0) {
02574             return pos;
02575         }
02576         if (pos == 0) break;
02577         pos--;
02578         s = rb_enc_prev_char(sbeg, s, e, enc);
02579     }
02580     return -1;
02581 }
02582 
02583 
02584 /*
02585  *  call-seq:
02586  *     str.rindex(substring [, fixnum])   -> fixnum or nil
02587  *     str.rindex(regexp [, fixnum])   -> fixnum or nil
02588  *
02589  *  Returns the index of the last occurrence of the given <i>substring</i> or
02590  *  pattern (<i>regexp</i>) in <i>str</i>. Returns <code>nil</code> if not
02591  *  found. If the second parameter is present, it specifies the position in the
02592  *  string to end the search---characters beyond this point will not be
02593  *  considered.
02594  *
02595  *     "hello".rindex('e')             #=> 1
02596  *     "hello".rindex('l')             #=> 3
02597  *     "hello".rindex('a')             #=> nil
02598  *     "hello".rindex(?e)              #=> 1
02599  *     "hello".rindex(/[aeiou]/, -2)   #=> 1
02600  */
02601 
02602 static VALUE
02603 rb_str_rindex_m(int argc, VALUE *argv, VALUE str)
02604 {
02605     VALUE sub;
02606     VALUE vpos;
02607     rb_encoding *enc = STR_ENC_GET(str);
02608     long pos, len = str_strlen(str, enc);
02609 
02610     if (rb_scan_args(argc, argv, "11", &sub, &vpos) == 2) {
02611         pos = NUM2LONG(vpos);
02612         if (pos < 0) {
02613             pos += len;
02614             if (pos < 0) {
02615                 if (TYPE(sub) == T_REGEXP) {
02616                     rb_backref_set(Qnil);
02617                 }
02618                 return Qnil;
02619             }
02620         }
02621         if (pos > len) pos = len;
02622     }
02623     else {
02624         pos = len;
02625     }
02626 
02627     switch (TYPE(sub)) {
02628       case T_REGEXP:
02629         /* enc = rb_get_check(str, sub); */
02630         pos = str_offset(RSTRING_PTR(str), RSTRING_END(str), pos,
02631                          STR_ENC_GET(str), single_byte_optimizable(str));
02632 
02633         if (!RREGEXP(sub)->ptr || RREGEXP_SRC_LEN(sub)) {
02634             pos = rb_reg_search(sub, str, pos, 1);
02635             pos = rb_str_sublen(str, pos);
02636         }
02637         if (pos >= 0) return LONG2NUM(pos);
02638         break;
02639 
02640       default: {
02641         VALUE tmp;
02642 
02643         tmp = rb_check_string_type(sub);
02644         if (NIL_P(tmp)) {
02645             rb_raise(rb_eTypeError, "type mismatch: %s given",
02646                      rb_obj_classname(sub));
02647         }
02648         sub = tmp;
02649       }
02650         /* fall through */
02651       case T_STRING:
02652         pos = rb_str_rindex(str, sub, pos);
02653         if (pos >= 0) return LONG2NUM(pos);
02654         break;
02655     }
02656     return Qnil;
02657 }
02658 
02659 /*
02660  *  call-seq:
02661  *     str =~ obj   -> fixnum or nil
02662  *
02663  *  Match---If <i>obj</i> is a <code>Regexp</code>, use it as a pattern to match
02664  *  against <i>str</i>,and returns the position the match starts, or
02665  *  <code>nil</code> if there is no match. Otherwise, invokes
02666  *  <i>obj.=~</i>, passing <i>str</i> as an argument. The default
02667  *  <code>=~</code> in <code>Object</code> returns <code>nil</code>.
02668  *
02669  *     "cat o' 9 tails" =~ /\d/   #=> 7
02670  *     "cat o' 9 tails" =~ 9      #=> nil
02671  */
02672 
02673 static VALUE
02674 rb_str_match(VALUE x, VALUE y)
02675 {
02676     switch (TYPE(y)) {
02677       case T_STRING:
02678         rb_raise(rb_eTypeError, "type mismatch: String given");
02679 
02680       case T_REGEXP:
02681         return rb_reg_match(y, x);
02682 
02683       default:
02684         return rb_funcall(y, rb_intern("=~"), 1, x);
02685     }
02686 }
02687 
02688 
02689 static VALUE get_pat(VALUE, int);
02690 
02691 
02692 /*
02693  *  call-seq:
02694  *     str.match(pattern)        -> matchdata or nil
02695  *     str.match(pattern, pos)   -> matchdata or nil
02696  *
02697  *  Converts <i>pattern</i> to a <code>Regexp</code> (if it isn't already one),
02698  *  then invokes its <code>match</code> method on <i>str</i>.  If the second
02699  *  parameter is present, it specifies the position in the string to begin the
02700  *  search.
02701  *
02702  *     'hello'.match('(.)\1')      #=> #<MatchData "ll" 1:"l">
02703  *     'hello'.match('(.)\1')[0]   #=> "ll"
02704  *     'hello'.match(/(.)\1/)[0]   #=> "ll"
02705  *     'hello'.match('xx')         #=> nil
02706  *
02707  *  If a block is given, invoke the block with MatchData if match succeed, so
02708  *  that you can write
02709  *
02710  *     str.match(pat) {|m| ...}
02711  *
02712  *  instead of
02713  *
02714  *     if m = str.match(pat)
02715  *       ...
02716  *     end
02717  *
02718  *  The return value is a value from block execution in this case.
02719  */
02720 
02721 static VALUE
02722 rb_str_match_m(int argc, VALUE *argv, VALUE str)
02723 {
02724     VALUE re, result;
02725     if (argc < 1)
02726        rb_raise(rb_eArgError, "wrong number of arguments (%d for 1..2)", argc);
02727     re = argv[0];
02728     argv[0] = str;
02729     result = rb_funcall2(get_pat(re, 0), rb_intern("match"), argc, argv);
02730     if (!NIL_P(result) && rb_block_given_p()) {
02731         return rb_yield(result);
02732     }
02733     return result;
02734 }
02735 
02736 enum neighbor_char {
02737     NEIGHBOR_NOT_CHAR,
02738     NEIGHBOR_FOUND,
02739     NEIGHBOR_WRAPPED
02740 };
02741 
02742 static enum neighbor_char
02743 enc_succ_char(char *p, long len, rb_encoding *enc)
02744 {
02745     long i;
02746     int l;
02747     while (1) {
02748         for (i = len-1; 0 <= i && (unsigned char)p[i] == 0xff; i--)
02749             p[i] = '\0';
02750         if (i < 0)
02751             return NEIGHBOR_WRAPPED;
02752         ++((unsigned char*)p)[i];
02753         l = rb_enc_precise_mbclen(p, p+len, enc);
02754         if (MBCLEN_CHARFOUND_P(l)) {
02755             l = MBCLEN_CHARFOUND_LEN(l);
02756             if (l == len) {
02757                 return NEIGHBOR_FOUND;
02758             }
02759             else {
02760                 memset(p+l, 0xff, len-l);
02761             }
02762         }
02763         if (MBCLEN_INVALID_P(l) && i < len-1) {
02764             long len2;
02765             int l2;
02766             for (len2 = len-1; 0 < len2; len2--) {
02767                 l2 = rb_enc_precise_mbclen(p, p+len2, enc);
02768                 if (!MBCLEN_INVALID_P(l2))
02769                     break;
02770             }
02771             memset(p+len2+1, 0xff, len-(len2+1));
02772         }
02773     }
02774 }
02775 
02776 static enum neighbor_char
02777 enc_pred_char(char *p, long len, rb_encoding *enc)
02778 {
02779     long i;
02780     int l;
02781     while (1) {
02782         for (i = len-1; 0 <= i && (unsigned char)p[i] == 0; i--)
02783             p[i] = '\xff';
02784         if (i < 0)
02785             return NEIGHBOR_WRAPPED;
02786         --((unsigned char*)p)[i];
02787         l = rb_enc_precise_mbclen(p, p+len, enc);
02788         if (MBCLEN_CHARFOUND_P(l)) {
02789             l = MBCLEN_CHARFOUND_LEN(l);
02790             if (l == len) {
02791                 return NEIGHBOR_FOUND;
02792             }
02793             else {
02794                 memset(p+l, 0, len-l);
02795             }
02796         }
02797         if (MBCLEN_INVALID_P(l) && i < len-1) {
02798             long len2;
02799             int l2;
02800             for (len2 = len-1; 0 < len2; len2--) {
02801                 l2 = rb_enc_precise_mbclen(p, p+len2, enc);
02802                 if (!MBCLEN_INVALID_P(l2))
02803                     break;
02804             }
02805             memset(p+len2+1, 0, len-(len2+1));
02806         }
02807     }
02808 }
02809 
02810 /*
02811   overwrite +p+ by succeeding letter in +enc+ and returns
02812   NEIGHBOR_FOUND or NEIGHBOR_WRAPPED.
02813   When NEIGHBOR_WRAPPED, carried-out letter is stored into carry.
02814   assuming each ranges are successive, and mbclen
02815   never change in each ranges.
02816   NEIGHBOR_NOT_CHAR is returned if invalid character or the range has only one
02817   character.
02818  */
02819 static enum neighbor_char
02820 enc_succ_alnum_char(char *p, long len, rb_encoding *enc, char *carry)
02821 {
02822     enum neighbor_char ret;
02823     unsigned int c;
02824     int ctype;
02825     int range;
02826     char save[ONIGENC_CODE_TO_MBC_MAXLEN];
02827 
02828     c = rb_enc_mbc_to_codepoint(p, p+len, enc);
02829     if (rb_enc_isctype(c, ONIGENC_CTYPE_DIGIT, enc))
02830         ctype = ONIGENC_CTYPE_DIGIT;
02831     else if (rb_enc_isctype(c, ONIGENC_CTYPE_ALPHA, enc))
02832         ctype = ONIGENC_CTYPE_ALPHA;
02833     else
02834         return NEIGHBOR_NOT_CHAR;
02835 
02836     MEMCPY(save, p, char, len);
02837     ret = enc_succ_char(p, len, enc);
02838     if (ret == NEIGHBOR_FOUND) {
02839         c = rb_enc_mbc_to_codepoint(p, p+len, enc);
02840         if (rb_enc_isctype(c, ctype, enc))
02841             return NEIGHBOR_FOUND;
02842     }
02843     MEMCPY(p, save, char, len);
02844     range = 1;
02845     while (1) {
02846         MEMCPY(save, p, char, len);
02847         ret = enc_pred_char(p, len, enc);
02848         if (ret == NEIGHBOR_FOUND) {
02849             c = rb_enc_mbc_to_codepoint(p, p+len, enc);
02850             if (!rb_enc_isctype(c, ctype, enc)) {
02851                 MEMCPY(p, save, char, len);
02852                 break;
02853             }
02854         }
02855         else {
02856             MEMCPY(p, save, char, len);
02857             break;
02858         }
02859         range++;
02860     }
02861     if (range == 1) {
02862         return NEIGHBOR_NOT_CHAR;
02863     }
02864 
02865     if (ctype != ONIGENC_CTYPE_DIGIT) {
02866         MEMCPY(carry, p, char, len);
02867         return NEIGHBOR_WRAPPED;
02868     }
02869 
02870     MEMCPY(carry, p, char, len);
02871     enc_succ_char(carry, len, enc);
02872     return NEIGHBOR_WRAPPED;
02873 }
02874 
02875 
02876 /*
02877  *  call-seq:
02878  *     str.succ   -> new_str
02879  *     str.next   -> new_str
02880  *
02881  *  Returns the successor to <i>str</i>. The successor is calculated by
02882  *  incrementing characters starting from the rightmost alphanumeric (or
02883  *  the rightmost character if there are no alphanumerics) in the
02884  *  string. Incrementing a digit always results in another digit, and
02885  *  incrementing a letter results in another letter of the same case.
02886  *  Incrementing nonalphanumerics uses the underlying character set's
02887  *  collating sequence.
02888  *
02889  *  If the increment generates a ``carry,'' the character to the left of
02890  *  it is incremented. This process repeats until there is no carry,
02891  *  adding an additional character if necessary.
02892  *
02893  *     "abcd".succ        #=> "abce"
02894  *     "THX1138".succ     #=> "THX1139"
02895  *     "<<koala>>".succ   #=> "<<koalb>>"
02896  *     "1999zzz".succ     #=> "2000aaa"
02897  *     "ZZZ9999".succ     #=> "AAAA0000"
02898  *     "***".succ         #=> "**+"
02899  */
02900 
02901 VALUE
02902 rb_str_succ(VALUE orig)
02903 {
02904     rb_encoding *enc;
02905     VALUE str;
02906     char *sbeg, *s, *e, *last_alnum = 0;
02907     int c = -1;
02908     long l;
02909     char carry[ONIGENC_CODE_TO_MBC_MAXLEN] = "\1";
02910     long carry_pos = 0, carry_len = 1;
02911     enum neighbor_char neighbor = NEIGHBOR_FOUND;
02912 
02913     str = rb_str_new5(orig, RSTRING_PTR(orig), RSTRING_LEN(orig));
02914     rb_enc_cr_str_copy_for_substr(str, orig);
02915     OBJ_INFECT(str, orig);
02916     if (RSTRING_LEN(str) == 0) return str;
02917 
02918     enc = STR_ENC_GET(orig);
02919     sbeg = RSTRING_PTR(str);
02920     s = e = sbeg + RSTRING_LEN(str);
02921 
02922     while ((s = rb_enc_prev_char(sbeg, s, e, enc)) != 0) {
02923         if (neighbor == NEIGHBOR_NOT_CHAR && last_alnum) {
02924             if (ISALPHA(*last_alnum) ? ISDIGIT(*s) :
02925                 ISDIGIT(*last_alnum) ? ISALPHA(*s) : 0) {
02926                 s = last_alnum;
02927                 break;
02928             }
02929         }
02930         if ((l = rb_enc_precise_mbclen(s, e, enc)) <= 0) continue;
02931         neighbor = enc_succ_alnum_char(s, l, enc, carry);
02932         switch (neighbor) {
02933           case NEIGHBOR_NOT_CHAR:
02934             continue;
02935           case NEIGHBOR_FOUND:
02936             return str;
02937           case NEIGHBOR_WRAPPED:
02938             last_alnum = s;
02939             break;
02940         }
02941         c = 1;
02942         carry_pos = s - sbeg;
02943         carry_len = l;
02944     }
02945     if (c == -1) {              /* str contains no alnum */
02946         s = e;
02947         while ((s = rb_enc_prev_char(sbeg, s, e, enc)) != 0) {
02948             enum neighbor_char neighbor;
02949             if ((l = rb_enc_precise_mbclen(s, e, enc)) <= 0) continue;
02950             neighbor = enc_succ_char(s, l, enc);
02951             if (neighbor == NEIGHBOR_FOUND)
02952                 return str;
02953             if (rb_enc_precise_mbclen(s, s+l, enc) != l) {
02954                 /* wrapped to \0...\0.  search next valid char. */
02955                 enc_succ_char(s, l, enc);
02956             }
02957             if (!rb_enc_asciicompat(enc)) {
02958                 MEMCPY(carry, s, char, l);
02959                 carry_len = l;
02960             }
02961             carry_pos = s - sbeg;
02962         }
02963     }
02964     RESIZE_CAPA(str, RSTRING_LEN(str) + carry_len);
02965     s = RSTRING_PTR(str) + carry_pos;
02966     memmove(s + carry_len, s, RSTRING_LEN(str) - carry_pos);
02967     memmove(s, carry, carry_len);
02968     STR_SET_LEN(str, RSTRING_LEN(str) + carry_len);
02969     RSTRING_PTR(str)[RSTRING_LEN(str)] = '\0';
02970     rb_enc_str_coderange(str);
02971     return str;
02972 }
02973 
02974 
02975 /*
02976  *  call-seq:
02977  *     str.succ!   -> str
02978  *     str.next!   -> str
02979  *
02980  *  Equivalent to <code>String#succ</code>, but modifies the receiver in
02981  *  place.
02982  */
02983 
02984 static VALUE
02985 rb_str_succ_bang(VALUE str)
02986 {
02987     rb_str_shared_replace(str, rb_str_succ(str));
02988 
02989     return str;
02990 }
02991 
02992 
02993 /*
02994  *  call-seq:
02995  *     str.upto(other_str, exclusive=false) {|s| block }   -> str
02996  *     str.upto(other_str, exclusive=false)                -> an_enumerator
02997  *
02998  *  Iterates through successive values, starting at <i>str</i> and
02999  *  ending at <i>other_str</i> inclusive, passing each value in turn to
03000  *  the block. The <code>String#succ</code> method is used to generate
03001  *  each value.  If optional second argument exclusive is omitted or is false,
03002  *  the last value will be included; otherwise it will be excluded.
03003  *
03004  *  If no block is given, an enumerator is returned instead.
03005  *
03006  *     "a8".upto("b6") {|s| print s, ' ' }
03007  *     for s in "a8".."b6"
03008  *       print s, ' '
03009  *     end
03010  *
03011  *  <em>produces:</em>
03012  *
03013  *     a8 a9 b0 b1 b2 b3 b4 b5 b6
03014  *     a8 a9 b0 b1 b2 b3 b4 b5 b6
03015  *
03016  *  If <i>str</i> and <i>other_str</i> contains only ascii numeric characters,
03017  *  both are recognized as decimal numbers. In addition, the width of
03018  *  string (e.g. leading zeros) is handled appropriately.
03019  *
03020  *     "9".upto("11").to_a   #=> ["9", "10", "11"]
03021  *     "25".upto("5").to_a   #=> []
03022  *     "07".upto("11").to_a  #=> ["07", "08", "09", "10", "11"]
03023  */
03024 
03025 static VALUE
03026 rb_str_upto(int argc, VALUE *argv, VALUE beg)
03027 {
03028     VALUE end, exclusive;
03029     VALUE current, after_end;
03030     ID succ;
03031     int n, excl, ascii;
03032     rb_encoding *enc;
03033 
03034     rb_scan_args(argc, argv, "11", &end, &exclusive);
03035     RETURN_ENUMERATOR(beg, argc, argv);
03036     excl = RTEST(exclusive);
03037     CONST_ID(succ, "succ");
03038     StringValue(end);
03039     enc = rb_enc_check(beg, end);
03040     ascii = (is_ascii_string(beg) && is_ascii_string(end));
03041     /* single character */
03042     if (RSTRING_LEN(beg) == 1 && RSTRING_LEN(end) == 1 && ascii) {
03043         char c = RSTRING_PTR(beg)[0];
03044         char e = RSTRING_PTR(end)[0];
03045 
03046         if (c > e || (excl && c == e)) return beg;
03047         for (;;) {
03048             rb_yield(rb_enc_str_new(&c, 1, enc));
03049             if (!excl && c == e) break;
03050             c++;
03051             if (excl && c == e) break;
03052         }
03053         return beg;
03054     }
03055     /* both edges are all digits */
03056     if (ascii && ISDIGIT(RSTRING_PTR(beg)[0]) && ISDIGIT(RSTRING_PTR(end)[0])) {
03057         char *s, *send;
03058         VALUE b, e;
03059         int width;
03060 
03061         s = RSTRING_PTR(beg); send = RSTRING_END(beg);
03062         width = rb_long2int(send - s);
03063         while (s < send) {
03064             if (!ISDIGIT(*s)) goto no_digits;
03065             s++;
03066         }
03067         s = RSTRING_PTR(end); send = RSTRING_END(end);
03068         while (s < send) {
03069             if (!ISDIGIT(*s)) goto no_digits;
03070             s++;
03071         }
03072         b = rb_str_to_inum(beg, 10, FALSE);
03073         e = rb_str_to_inum(end, 10, FALSE);
03074         if (FIXNUM_P(b) && FIXNUM_P(e)) {
03075             long bi = FIX2LONG(b);
03076             long ei = FIX2LONG(e);
03077             rb_encoding *usascii = rb_usascii_encoding();
03078 
03079             while (bi <= ei) {
03080                 if (excl && bi == ei) break;
03081                 rb_yield(rb_enc_sprintf(usascii, "%.*ld", width, bi));
03082                 bi++;
03083             }
03084         }
03085         else {
03086             ID op = excl ? '<' : rb_intern("<=");
03087             VALUE args[2], fmt = rb_obj_freeze(rb_usascii_str_new_cstr("%.*d"));
03088 
03089             args[0] = INT2FIX(width);
03090             while (rb_funcall(b, op, 1, e)) {
03091                 args[1] = b;
03092                 rb_yield(rb_str_format(numberof(args), args, fmt));
03093                 b = rb_funcall(b, succ, 0, 0);
03094             }
03095         }
03096         return beg;
03097     }
03098     /* normal case */
03099   no_digits:
03100     n = rb_str_cmp(beg, end);
03101     if (n > 0 || (excl && n == 0)) return beg;
03102 
03103     after_end = rb_funcall(end, succ, 0, 0);
03104     current = rb_str_dup(beg);
03105     while (!rb_str_equal(current, after_end)) {
03106         VALUE next = Qnil;
03107         if (excl || !rb_str_equal(current, end))
03108             next = rb_funcall(current, succ, 0, 0);
03109         rb_yield(current);
03110         if (NIL_P(next)) break;
03111         current = next;
03112         StringValue(current);
03113         if (excl && rb_str_equal(current, end)) break;
03114         if (RSTRING_LEN(current) > RSTRING_LEN(end) || RSTRING_LEN(current) == 0)
03115             break;
03116     }
03117 
03118     return beg;
03119 }
03120 
03121 static VALUE
03122 rb_str_subpat(VALUE str, VALUE re, VALUE backref)
03123 {
03124     if (rb_reg_search(re, str, 0, 0) >= 0) {
03125         VALUE match = rb_backref_get();
03126         int nth = rb_reg_backref_number(match, backref);
03127         return rb_reg_nth_match(nth, match);
03128     }
03129     return Qnil;
03130 }
03131 
03132 static VALUE
03133 rb_str_aref(VALUE str, VALUE indx)
03134 {
03135     long idx;
03136 
03137     switch (TYPE(indx)) {
03138       case T_FIXNUM:
03139         idx = FIX2LONG(indx);
03140 
03141       num_index:
03142         str = rb_str_substr(str, idx, 1);
03143         if (!NIL_P(str) && RSTRING_LEN(str) == 0) return Qnil;
03144         return str;
03145 
03146       case T_REGEXP:
03147         return rb_str_subpat(str, indx, INT2FIX(0));
03148 
03149       case T_STRING:
03150         if (rb_str_index(str, indx, 0) != -1)
03151             return rb_str_dup(indx);
03152         return Qnil;
03153 
03154       default:
03155         /* check if indx is Range */
03156         {
03157             long beg, len;
03158             VALUE tmp;
03159 
03160             len = str_strlen(str, STR_ENC_GET(str));
03161             switch (rb_range_beg_len(indx, &beg, &len, len, 0)) {
03162               case Qfalse:
03163                 break;
03164               case Qnil:
03165                 return Qnil;
03166               default:
03167                 tmp = rb_str_substr(str, beg, len);
03168                 return tmp;
03169             }
03170         }
03171         idx = NUM2LONG(indx);
03172         goto num_index;
03173     }
03174     return Qnil;                /* not reached */
03175 }
03176 
03177 
03178 /*
03179  *  call-seq:
03180  *     str[fixnum]                 -> new_str or nil
03181  *     str[fixnum, fixnum]         -> new_str or nil
03182  *     str[range]                  -> new_str or nil
03183  *     str[regexp]                 -> new_str or nil
03184  *     str[regexp, fixnum]         -> new_str or nil
03185  *     str[other_str]              -> new_str or nil
03186  *     str.slice(fixnum)           -> new_str or nil
03187  *     str.slice(fixnum, fixnum)   -> new_str or nil
03188  *     str.slice(range)            -> new_str or nil
03189  *     str.slice(regexp)           -> new_str or nil
03190  *     str.slice(regexp, fixnum)   -> new_str or nil
03191  *     str.slice(regexp, capname)  -> new_str or nil
03192  *     str.slice(other_str)        -> new_str or nil
03193  *
03194  *  Element Reference---If passed a single <code>Fixnum</code>, returns a
03195  *  substring of one character at that position. If passed two <code>Fixnum</code>
03196  *  objects, returns a substring starting at the offset given by the first, and
03197  *  with a length given by the second. If passed a range, its beginning and end
03198  *  are interpreted as offsets delimiting the substring to be returned. In all
03199  *  three cases, if an offset is negative, it is counted from the end of <i>str</i>.
03200  *  Returns <code>nil</code> if the initial offset falls outside the string or
03201  *  the length is negative.
03202  *
03203  *  If a <code>Regexp</code> is supplied, the matching portion of <i>str</i> is
03204  *  returned. If a numeric or name parameter follows the regular expression, that
03205  *  component of the <code>MatchData</code> is returned instead. If a
03206  *  <code>String</code> is given, that string is returned if it occurs in
03207  *  <i>str</i>. In both cases, <code>nil</code> is returned if there is no
03208  *  match.
03209  *
03210  *     a = "hello there"
03211  *     a[1]                   #=> "e"
03212  *     a[2, 3]                #=> "llo"
03213  *     a[2..3]                #=> "ll"
03214  *     a[-3, 2]               #=> "er"
03215  *     a[7..-2]               #=> "her"
03216  *     a[-4..-2]              #=> "her"
03217  *     a[-2..-4]              #=> ""
03218  *     a[12..-1]              #=> nil
03219  *     a[/[aeiou](.)\1/]      #=> "ell"
03220  *     a[/[aeiou](.)\1/, 0]   #=> "ell"
03221  *     a[/[aeiou](.)\1/, 1]   #=> "l"
03222  *     a[/[aeiou](.)\1/, 2]   #=> nil
03223  *     a["lo"]                #=> "lo"
03224  *     a["bye"]               #=> nil
03225  */
03226 
03227 static VALUE
03228 rb_str_aref_m(int argc, VALUE *argv, VALUE str)
03229 {
03230     if (argc == 2) {
03231         if (TYPE(argv[0]) == T_REGEXP) {
03232             return rb_str_subpat(str, argv[0], argv[1]);
03233         }
03234         return rb_str_substr(str, NUM2LONG(argv[0]), NUM2LONG(argv[1]));
03235     }
03236     if (argc != 1) {
03237         rb_raise(rb_eArgError, "wrong number of arguments (%d for 1..2)", argc);
03238     }
03239     return rb_str_aref(str, argv[0]);
03240 }
03241 
03242 VALUE
03243 rb_str_drop_bytes(VALUE str, long len)
03244 {
03245     char *ptr = RSTRING_PTR(str);
03246     long olen = RSTRING_LEN(str), nlen;
03247 
03248     str_modifiable(str);
03249     if (len > olen) len = olen;
03250     nlen = olen - len;
03251     if (nlen <= RSTRING_EMBED_LEN_MAX) {
03252         char *oldptr = ptr;
03253         int fl = (int)(RBASIC(str)->flags & (STR_NOEMBED|ELTS_SHARED));
03254         STR_SET_EMBED(str);
03255         STR_SET_EMBED_LEN(str, nlen);
03256         ptr = RSTRING(str)->as.ary;
03257         memmove(ptr, oldptr + len, nlen);
03258         if (fl == STR_NOEMBED) xfree(oldptr);
03259     }
03260     else {
03261         if (!STR_SHARED_P(str)) rb_str_new4(str);
03262         ptr = RSTRING(str)->as.heap.ptr += len;
03263         RSTRING(str)->as.heap.len = nlen;
03264     }
03265     ptr[nlen] = 0;
03266     ENC_CODERANGE_CLEAR(str);
03267     return str;
03268 }
03269 
03270 static void
03271 rb_str_splice_0(VALUE str, long beg, long len, VALUE val)
03272 {
03273     if (beg == 0 && RSTRING_LEN(val) == 0) {
03274         rb_str_drop_bytes(str, len);
03275         OBJ_INFECT(str, val);
03276         return;
03277     }
03278 
03279     rb_str_modify(str);
03280     if (len < RSTRING_LEN(val)) {
03281         /* expand string */
03282         RESIZE_CAPA(str, RSTRING_LEN(str) + RSTRING_LEN(val) - len + 1);
03283     }
03284 
03285     if (RSTRING_LEN(val) != len) {
03286         memmove(RSTRING_PTR(str) + beg + RSTRING_LEN(val),
03287                 RSTRING_PTR(str) + beg + len,
03288                 RSTRING_LEN(str) - (beg + len));
03289     }
03290     if (RSTRING_LEN(val) < beg && len < 0) {
03291         MEMZERO(RSTRING_PTR(str) + RSTRING_LEN(str), char, -len);
03292     }
03293     if (RSTRING_LEN(val) > 0) {
03294         memmove(RSTRING_PTR(str)+beg, RSTRING_PTR(val), RSTRING_LEN(val));
03295     }
03296     STR_SET_LEN(str, RSTRING_LEN(str) + RSTRING_LEN(val) - len);
03297     if (RSTRING_PTR(str)) {
03298         RSTRING_PTR(str)[RSTRING_LEN(str)] = '\0';
03299     }
03300     OBJ_INFECT(str, val);
03301 }
03302 
03303 static void
03304 rb_str_splice(VALUE str, long beg, long len, VALUE val)
03305 {
03306     long slen;
03307     char *p, *e;
03308     rb_encoding *enc;
03309     int singlebyte = single_byte_optimizable(str);
03310     int cr;
03311 
03312     if (len < 0) rb_raise(rb_eIndexError, "negative length %ld", len);
03313 
03314     StringValue(val);
03315     enc = rb_enc_check(str, val);
03316     slen = str_strlen(str, enc);
03317 
03318     if (slen < beg) {
03319       out_of_range:
03320         rb_raise(rb_eIndexError, "index %ld out of string", beg);
03321     }
03322     if (beg < 0) {
03323         if (-beg > slen) {
03324             goto out_of_range;
03325         }
03326         beg += slen;
03327     }
03328     if (slen < len || slen < beg + len) {
03329         len = slen - beg;
03330     }
03331     str_modify_keep_cr(str);
03332     p = str_nth(RSTRING_PTR(str), RSTRING_END(str), beg, enc, singlebyte);
03333     if (!p) p = RSTRING_END(str);
03334     e = str_nth(p, RSTRING_END(str), len, enc, singlebyte);
03335     if (!e) e = RSTRING_END(str);
03336     /* error check */
03337     beg = p - RSTRING_PTR(str); /* physical position */
03338     len = e - p;                /* physical length */
03339     rb_str_splice_0(str, beg, len, val);
03340     rb_enc_associate(str, enc);
03341     cr = ENC_CODERANGE_AND(ENC_CODERANGE(str), ENC_CODERANGE(val));
03342     if (cr != ENC_CODERANGE_BROKEN)
03343         ENC_CODERANGE_SET(str, cr);
03344 }
03345 
03346 void
03347 rb_str_update(VALUE str, long beg, long len, VALUE val)
03348 {
03349     rb_str_splice(str, beg, len, val);
03350 }
03351 
03352 static void
03353 rb_str_subpat_set(VALUE str, VALUE re, VALUE backref, VALUE val)
03354 {
03355     int nth;
03356     VALUE match;
03357     long start, end, len;
03358     rb_encoding *enc;
03359     struct re_registers *regs;
03360 
03361     if (rb_reg_search(re, str, 0, 0) < 0) {
03362         rb_raise(rb_eIndexError, "regexp not matched");
03363     }
03364     match = rb_backref_get();
03365     nth = rb_reg_backref_number(match, backref);
03366     regs = RMATCH_REGS(match);
03367     if (nth >= regs->num_regs) {
03368       out_of_range:
03369         rb_raise(rb_eIndexError, "index %d out of regexp", nth);
03370     }
03371     if (nth < 0) {
03372         if (-nth >= regs->num_regs) {
03373             goto out_of_range;
03374         }
03375         nth += regs->num_regs;
03376     }
03377 
03378     start = BEG(nth);
03379     if (start == -1) {
03380         rb_raise(rb_eIndexError, "regexp group %d not matched", nth);
03381     }
03382     end = END(nth);
03383     len = end - start;
03384     StringValue(val);
03385     enc = rb_enc_check(str, val);
03386     rb_str_splice_0(str, start, len, val);
03387     rb_enc_associate(str, enc);
03388 }
03389 
03390 static VALUE
03391 rb_str_aset(VALUE str, VALUE indx, VALUE val)
03392 {
03393     long idx, beg;
03394 
03395     switch (TYPE(indx)) {
03396       case T_FIXNUM:
03397         idx = FIX2LONG(indx);
03398       num_index:
03399         rb_str_splice(str, idx, 1, val);
03400         return val;
03401 
03402       case T_REGEXP:
03403         rb_str_subpat_set(str, indx, INT2FIX(0), val);
03404         return val;
03405 
03406       case T_STRING:
03407         beg = rb_str_index(str, indx, 0);
03408         if (beg < 0) {
03409             rb_raise(rb_eIndexError, "string not matched");
03410         }
03411         beg = rb_str_sublen(str, beg);
03412         rb_str_splice(str, beg, str_strlen(indx, 0), val);
03413         return val;
03414 
03415       default:
03416         /* check if indx is Range */
03417         {
03418             long beg, len;
03419             if (rb_range_beg_len(indx, &beg, &len, str_strlen(str, 0), 2)) {
03420                 rb_str_splice(str, beg, len, val);
03421                 return val;
03422             }
03423         }
03424         idx = NUM2LONG(indx);
03425         goto num_index;
03426     }
03427 }
03428 
03429 /*
03430  *  call-seq:
03431  *     str[fixnum] = new_str
03432  *     str[fixnum, fixnum] = new_str
03433  *     str[range] = aString
03434  *     str[regexp] = new_str
03435  *     str[regexp, fixnum] = new_str
03436  *     str[regexp, name] = new_str
03437  *     str[other_str] = new_str
03438  *
03439  *  Element Assignment---Replaces some or all of the content of <i>str</i>. The
03440  *  portion of the string affected is determined using the same criteria as
03441  *  <code>String#[]</code>. If the replacement string is not the same length as
03442  *  the text it is replacing, the string will be adjusted accordingly. If the
03443  *  regular expression or string is used as the index doesn't match a position
03444  *  in the string, <code>IndexError</code> is raised. If the regular expression
03445  *  form is used, the optional second <code>Fixnum</code> allows you to specify
03446  *  which portion of the match to replace (effectively using the
03447  *  <code>MatchData</code> indexing rules. The forms that take a
03448  *  <code>Fixnum</code> will raise an <code>IndexError</code> if the value is
03449  *  out of range; the <code>Range</code> form will raise a
03450  *  <code>RangeError</code>, and the <code>Regexp</code> and <code>String</code>
03451  *  forms will silently ignore the assignment.
03452  */
03453 
03454 static VALUE
03455 rb_str_aset_m(int argc, VALUE *argv, VALUE str)
03456 {
03457     if (argc == 3) {
03458         if (TYPE(argv[0]) == T_REGEXP) {
03459             rb_str_subpat_set(str, argv[0], argv[1], argv[2]);
03460         }
03461         else {
03462             rb_str_splice(str, NUM2LONG(argv[0]), NUM2LONG(argv[1]), argv[2]);
03463         }
03464         return argv[2];
03465     }
03466     if (argc != 2) {
03467         rb_raise(rb_eArgError, "wrong number of arguments (%d for 2..3)", argc);
03468     }
03469     return rb_str_aset(str, argv[0], argv[1]);
03470 }
03471 
03472 /*
03473  *  call-seq:
03474  *     str.insert(index, other_str)   -> str
03475  *
03476  *  Inserts <i>other_str</i> before the character at the given
03477  *  <i>index</i>, modifying <i>str</i>. Negative indices count from the
03478  *  end of the string, and insert <em>after</em> the given character.
03479  *  The intent is insert <i>aString</i> so that it starts at the given
03480  *  <i>index</i>.
03481  *
03482  *     "abcd".insert(0, 'X')    #=> "Xabcd"
03483  *     "abcd".insert(3, 'X')    #=> "abcXd"
03484  *     "abcd".insert(4, 'X')    #=> "abcdX"
03485  *     "abcd".insert(-3, 'X')   #=> "abXcd"
03486  *     "abcd".insert(-1, 'X')   #=> "abcdX"
03487  */
03488 
03489 static VALUE
03490 rb_str_insert(VALUE str, VALUE idx, VALUE str2)
03491 {
03492     long pos = NUM2LONG(idx);
03493 
03494     if (pos == -1) {
03495         return rb_str_append(str, str2);
03496     }
03497     else if (pos < 0) {
03498         pos++;
03499     }
03500     rb_str_splice(str, pos, 0, str2);
03501     return str;
03502 }
03503 
03504 
03505 /*
03506  *  call-seq:
03507  *     str.slice!(fixnum)           -> fixnum or nil
03508  *     str.slice!(fixnum, fixnum)   -> new_str or nil
03509  *     str.slice!(range)            -> new_str or nil
03510  *     str.slice!(regexp)           -> new_str or nil
03511  *     str.slice!(other_str)        -> new_str or nil
03512  *
03513  *  Deletes the specified portion from <i>str</i>, and returns the portion
03514  *  deleted.
03515  *
03516  *     string = "this is a string"
03517  *     string.slice!(2)        #=> "i"
03518  *     string.slice!(3..6)     #=> " is "
03519  *     string.slice!(/s.*t/)   #=> "sa st"
03520  *     string.slice!("r")      #=> "r"
03521  *     string                  #=> "thing"
03522  */
03523 
03524 static VALUE
03525 rb_str_slice_bang(int argc, VALUE *argv, VALUE str)
03526 {
03527     VALUE result;
03528     VALUE buf[3];
03529     int i;
03530 
03531     if (argc < 1 || 2 < argc) {
03532         rb_raise(rb_eArgError, "wrong number of arguments (%d for 1..2)", argc);
03533     }
03534     for (i=0; i<argc; i++) {
03535         buf[i] = argv[i];
03536     }
03537     str_modify_keep_cr(str);
03538     result = rb_str_aref_m(argc, buf, str);
03539     if (!NIL_P(result)) {
03540         buf[i] = rb_str_new(0,0);
03541         rb_str_aset_m(argc+1, buf, str);
03542     }
03543     return result;
03544 }
03545 
03546 static VALUE
03547 get_pat(VALUE pat, int quote)
03548 {
03549     VALUE val;
03550 
03551     switch (TYPE(pat)) {
03552       case T_REGEXP:
03553         return pat;
03554 
03555       case T_STRING:
03556         break;
03557 
03558       default:
03559         val = rb_check_string_type(pat);
03560         if (NIL_P(val)) {
03561             Check_Type(pat, T_REGEXP);
03562         }
03563         pat = val;
03564     }
03565 
03566     if (quote) {
03567         pat = rb_reg_quote(pat);
03568     }
03569 
03570     return rb_reg_regcomp(pat);
03571 }
03572 
03573 
03574 /*
03575  *  call-seq:
03576  *     str.sub!(pattern, replacement)          -> str or nil
03577  *     str.sub!(pattern) {|match| block }      -> str or nil
03578  *
03579  *  Performs the substitutions of <code>String#sub</code> in place,
03580  *  returning <i>str</i>, or <code>nil</code> if no substitutions were
03581  *  performed.
03582  */
03583 
03584 static VALUE
03585 rb_str_sub_bang(int argc, VALUE *argv, VALUE str)
03586 {
03587     VALUE pat, repl, hash = Qnil;
03588     int iter = 0;
03589     int tainted = 0;
03590     int untrusted = 0;
03591     long plen;
03592 
03593     if (argc == 1 && rb_block_given_p()) {
03594         iter = 1;
03595     }
03596     else if (argc == 2) {
03597         repl = argv[1];
03598         hash = rb_check_convert_type(argv[1], T_HASH, "Hash", "to_hash");
03599         if (NIL_P(hash)) {
03600             StringValue(repl);
03601         }
03602         if (OBJ_TAINTED(repl)) tainted = 1;
03603         if (OBJ_UNTRUSTED(repl)) untrusted = 1;
03604     }
03605     else {
03606         rb_raise(rb_eArgError, "wrong number of arguments (%d for 1..2)", argc);
03607     }
03608 
03609     pat = get_pat(argv[0], 1);
03610     str_modifiable(str);
03611     if (rb_reg_search(pat, str, 0, 0) >= 0) {
03612         rb_encoding *enc;
03613         int cr = ENC_CODERANGE(str);
03614         VALUE match = rb_backref_get();
03615         struct re_registers *regs = RMATCH_REGS(match);
03616         long beg0 = BEG(0);
03617         long end0 = END(0);
03618         char *p, *rp;
03619         long len, rlen;
03620 
03621         if (iter || !NIL_P(hash)) {
03622             p = RSTRING_PTR(str); len = RSTRING_LEN(str);
03623 
03624             if (iter) {
03625                 repl = rb_obj_as_string(rb_yield(rb_reg_nth_match(0, match)));
03626             }
03627             else {
03628                 repl = rb_hash_aref(hash, rb_str_subseq(str, beg0, end0 - beg0));
03629                 repl = rb_obj_as_string(repl);
03630             }
03631             str_mod_check(str, p, len);
03632             rb_check_frozen(str);
03633         }
03634         else {
03635             repl = rb_reg_regsub(repl, str, regs, pat);
03636         }
03637         enc = rb_enc_compatible(str, repl);
03638         if (!enc) {
03639             rb_encoding *str_enc = STR_ENC_GET(str);
03640             p = RSTRING_PTR(str); len = RSTRING_LEN(str);
03641             if (coderange_scan(p, beg0, str_enc) != ENC_CODERANGE_7BIT ||
03642                 coderange_scan(p+end0, len-end0, str_enc) != ENC_CODERANGE_7BIT) {
03643                 rb_raise(rb_eEncCompatError, "incompatible character encodings: %s and %s",
03644                          rb_enc_name(str_enc),
03645                          rb_enc_name(STR_ENC_GET(repl)));
03646             }
03647             enc = STR_ENC_GET(repl);
03648         }
03649         rb_str_modify(str);
03650         rb_enc_associate(str, enc);
03651         if (OBJ_TAINTED(repl)) tainted = 1;
03652         if (OBJ_UNTRUSTED(repl)) untrusted = 1;
03653         if (ENC_CODERANGE_UNKNOWN < cr && cr < ENC_CODERANGE_BROKEN) {
03654             int cr2 = ENC_CODERANGE(repl);
03655             if (cr2 == ENC_CODERANGE_BROKEN ||
03656                 (cr == ENC_CODERANGE_VALID && cr2 == ENC_CODERANGE_7BIT))
03657                 cr = ENC_CODERANGE_UNKNOWN;
03658             else
03659                 cr = cr2;
03660         }
03661         plen = end0 - beg0;
03662         rp = RSTRING_PTR(repl); rlen = RSTRING_LEN(repl);
03663         len = RSTRING_LEN(str);
03664         if (rlen > plen) {
03665             RESIZE_CAPA(str, len + rlen - plen);
03666         }
03667         p = RSTRING_PTR(str);
03668         if (rlen != plen) {
03669             memmove(p + beg0 + rlen, p + beg0 + plen, len - beg0 - plen);
03670         }
03671         memcpy(p + beg0, rp, rlen);
03672         len += rlen - plen;
03673         STR_SET_LEN(str, len);
03674         RSTRING_PTR(str)[len] = '\0';
03675         ENC_CODERANGE_SET(str, cr);
03676         if (tainted) OBJ_TAINT(str);
03677         if (untrusted) OBJ_UNTRUST(str);
03678 
03679         return str;
03680     }
03681     return Qnil;
03682 }
03683 
03684 
03685 /*
03686  *  call-seq:
03687  *     str.sub(pattern, replacement)         -> new_str
03688  *     str.sub(pattern, hash)                -> new_str
03689  *     str.sub(pattern) {|match| block }     -> new_str
03690  *
03691  *  Returns a copy of <i>str</i> with the <em>first</em> occurrence of
03692  *  <i>pattern</i> substituted for the second argument. The <i>pattern</i> is
03693  *  typically a <code>Regexp</code>; if given as a <code>String</code>, any
03694  *  regular expression metacharacters it contains will be interpreted
03695  *  literally, e.g. <code>'\\\d'</code> will match a backlash followed by 'd',
03696  *  instead of a digit.
03697  *
03698  *  If <i>replacement</i> is a <code>String</code> it will be substituted for
03699  *  the matched text. It may contain back-references to the pattern's capture
03700  *  groups of the form <code>\\\d</code>, where <i>d</i> is a group number, or
03701  *  <code>\\\k<n></code>, where <i>n</i> is a group name. If it is a
03702  *  double-quoted string, both back-references must be preceded by an
03703  *  additional backslash. However, within <i>replacement</i> the special match
03704  *  variables, such as <code>&$</code>, will not refer to the current match.
03705  *
03706  *  If the second argument is a <code>Hash</code>, and the matched text is one
03707  *  of its keys, the corresponding value is the replacement string.
03708  *
03709  *  In the block form, the current match string is passed in as a parameter,
03710  *  and variables such as <code>$1</code>, <code>$2</code>, <code>$`</code>,
03711  *  <code>$&</code>, and <code>$'</code> will be set appropriately. The value
03712  *  returned by the block will be substituted for the match on each call.
03713  *
03714  *  The result inherits any tainting in the original string or any supplied
03715  *  replacement string.
03716  *
03717  *     "hello".sub(/[aeiou]/, '*')                  #=> "h*llo"
03718  *     "hello".sub(/([aeiou])/, '<\1>')             #=> "h<e>llo"
03719  *     "hello".sub(/./) {|s| s.ord.to_s + ' ' }     #=> "104 ello"
03720  *     "hello".sub(/(?<foo>[aeiou])/, '*\k<foo>*')  #=> "h*e*llo"
03721  *     'Is SHELL your preferred shell?'.sub(/[[:upper:]]{2,}/, ENV)
03722  *      #=> "Is /bin/bash your preferred shell?"
03723  */
03724 
03725 static VALUE
03726 rb_str_sub(int argc, VALUE *argv, VALUE str)
03727 {
03728     str = rb_str_dup(str);
03729     rb_str_sub_bang(argc, argv, str);
03730     return str;
03731 }
03732 
03733 static VALUE
03734 str_gsub(int argc, VALUE *argv, VALUE str, int bang)
03735 {
03736     VALUE pat, val, repl, match, dest, hash = Qnil;
03737     struct re_registers *regs;
03738     long beg, n;
03739     long beg0, end0;
03740     long offset, blen, slen, len, last;
03741     int iter = 0;
03742     char *sp, *cp;
03743     int tainted = 0;
03744     rb_encoding *str_enc;
03745 
03746     switch (argc) {
03747       case 1:
03748         RETURN_ENUMERATOR(str, argc, argv);
03749         iter = 1;
03750         break;
03751       case 2:
03752         repl = argv[1];
03753         hash = rb_check_convert_type(argv[1], T_HASH, "Hash", "to_hash");
03754         if (NIL_P(hash)) {
03755             StringValue(repl);
03756         }
03757         if (OBJ_TAINTED(repl)) tainted = 1;
03758         break;
03759       default:
03760         rb_raise(rb_eArgError, "wrong number of arguments (%d for 1..2)", argc);
03761     }
03762 
03763     pat = get_pat(argv[0], 1);
03764     beg = rb_reg_search(pat, str, 0, 0);
03765     if (beg < 0) {
03766         if (bang) return Qnil;  /* no match, no substitution */
03767         return rb_str_dup(str);
03768     }
03769 
03770     offset = 0;
03771     n = 0;
03772     blen = RSTRING_LEN(str) + 30; /* len + margin */
03773     dest = rb_str_buf_new(blen);
03774     sp = RSTRING_PTR(str);
03775     slen = RSTRING_LEN(str);
03776     cp = sp;
03777     str_enc = STR_ENC_GET(str);
03778     rb_enc_associate(dest, str_enc);
03779     ENC_CODERANGE_SET(dest, rb_enc_asciicompat(str_enc) ? ENC_CODERANGE_7BIT : ENC_CODERANGE_VALID);
03780 
03781     do {
03782         n++;
03783         match = rb_backref_get();
03784         regs = RMATCH_REGS(match);
03785         beg0 = BEG(0);
03786         end0 = END(0);
03787         if (iter || !NIL_P(hash)) {
03788             if (iter) {
03789                 val = rb_obj_as_string(rb_yield(rb_reg_nth_match(0, match)));
03790             }
03791             else {
03792                 val = rb_hash_aref(hash, rb_str_subseq(str, BEG(0), END(0) - BEG(0)));
03793                 val = rb_obj_as_string(val);
03794             }
03795             str_mod_check(str, sp, slen);
03796             if (val == dest) {  /* paranoid check [ruby-dev:24827] */
03797                 rb_raise(rb_eRuntimeError, "block should not cheat");
03798             }
03799         }
03800         else {
03801             val = rb_reg_regsub(repl, str, regs, pat);
03802         }
03803 
03804         if (OBJ_TAINTED(val)) tainted = 1;
03805 
03806         len = beg - offset;     /* copy pre-match substr */
03807         if (len) {
03808             rb_enc_str_buf_cat(dest, cp, len, str_enc);
03809         }
03810 
03811         rb_str_buf_append(dest, val);
03812 
03813         last = offset;
03814         offset = end0;
03815         if (beg0 == end0) {
03816             /*
03817              * Always consume at least one character of the input string
03818              * in order to prevent infinite loops.
03819              */
03820             if (RSTRING_LEN(str) <= end0) break;
03821             len = rb_enc_fast_mbclen(RSTRING_PTR(str)+end0, RSTRING_END(str), str_enc);
03822             rb_enc_str_buf_cat(dest, RSTRING_PTR(str)+end0, len, str_enc);
03823             offset = end0 + len;
03824         }
03825         cp = RSTRING_PTR(str) + offset;
03826         if (offset > RSTRING_LEN(str)) break;
03827         beg = rb_reg_search(pat, str, offset, 0);
03828     } while (beg >= 0);
03829     if (RSTRING_LEN(str) > offset) {
03830         rb_enc_str_buf_cat(dest, cp, RSTRING_LEN(str) - offset, str_enc);
03831     }
03832     rb_reg_search(pat, str, last, 0);
03833     if (bang) {
03834         rb_str_shared_replace(str, dest);
03835     }
03836     else {
03837         RBASIC(dest)->klass = rb_obj_class(str);
03838         OBJ_INFECT(dest, str);
03839         str = dest;
03840     }
03841 
03842     if (tainted) OBJ_TAINT(str);
03843     return str;
03844 }
03845 
03846 
03847 /*
03848  *  call-seq:
03849  *     str.gsub!(pattern, replacement)        -> str or nil
03850  *     str.gsub!(pattern) {|match| block }    -> str or nil
03851  *     str.gsub!(pattern)                     -> an_enumerator
03852  *
03853  *  Performs the substitutions of <code>String#gsub</code> in place, returning
03854  *  <i>str</i>, or <code>nil</code> if no substitutions were performed.
03855  *  If no block and no <i>replacement</i> is given, an enumerator is returned instead.
03856  */
03857 
03858 static VALUE
03859 rb_str_gsub_bang(int argc, VALUE *argv, VALUE str)
03860 {
03861     str_modify_keep_cr(str);
03862     return str_gsub(argc, argv, str, 1);
03863 }
03864 
03865 
03866 /*
03867  *  call-seq:
03868  *     str.gsub(pattern, replacement)       -> new_str
03869  *     str.gsub(pattern, hash)              -> new_str
03870  *     str.gsub(pattern) {|match| block }   -> new_str
03871  *     str.gsub(pattern)                    -> enumerator
03872  *
03873  *  Returns a copy of <i>str</i> with the <em>all</em> occurrences of
03874  *  <i>pattern</i> substituted for the second argument. The <i>pattern</i> is
03875  *  typically a <code>Regexp</code>; if given as a <code>String</code>, any
03876  *  regular expression metacharacters it contains will be interpreted
03877  *  literally, e.g. <code>'\\\d'</code> will match a backlash followed by 'd',
03878  *  instead of a digit.
03879  *
03880  *  If <i>replacement</i> is a <code>String</code> it will be substituted for
03881  *  the matched text. It may contain back-references to the pattern's capture
03882  *  groups of the form <code>\\\d</code>, where <i>d</i> is a group number, or
03883  *  <code>\\\k<n></code>, where <i>n</i> is a group name. If it is a
03884  *  double-quoted string, both back-references must be preceded by an
03885  *  additional backslash. However, within <i>replacement</i> the special match
03886  *  variables, such as <code>&$</code>, will not refer to the current match.
03887  *
03888  *  If the second argument is a <code>Hash</code>, and the matched text is one
03889  *  of its keys, the corresponding value is the replacement string.
03890  *
03891  *  In the block form, the current match string is passed in as a parameter,
03892  *  and variables such as <code>$1</code>, <code>$2</code>, <code>$`</code>,
03893  *  <code>$&</code>, and <code>$'</code> will be set appropriately. The value
03894  *  returned by the block will be substituted for the match on each call.
03895  *
03896  *  The result inherits any tainting in the original string or any supplied
03897  *  replacement string.
03898  *
03899  *  When neither a block nor a second argument is supplied, an
03900  *  <code>Enumerator</code> is returned.
03901  *
03902  *     "hello".gsub(/[aeiou]/, '*')                  #=> "h*ll*"
03903  *     "hello".gsub(/([aeiou])/, '<\1>')             #=> "h<e>ll<o>"
03904  *     "hello".gsub(/./) {|s| s.ord.to_s + ' '}      #=> "104 101 108 108 111 "
03905  *     "hello".gsub(/(?<foo>[aeiou])/, '{\k<foo>}')  #=> "h{e}ll{o}"
03906  *     'hello'.gsub(/[eo]/, 'e' => 3, 'o' => '*')    #=> "h3ll*"
03907  */
03908 
03909 static VALUE
03910 rb_str_gsub(int argc, VALUE *argv, VALUE str)
03911 {
03912     return str_gsub(argc, argv, str, 0);
03913 }
03914 
03915 
03916 /*
03917  *  call-seq:
03918  *     str.replace(other_str)   -> str
03919  *
03920  *  Replaces the contents and taintedness of <i>str</i> with the corresponding
03921  *  values in <i>other_str</i>.
03922  *
03923  *     s = "hello"         #=> "hello"
03924  *     s.replace "world"   #=> "world"
03925  */
03926 
03927 VALUE
03928 rb_str_replace(VALUE str, VALUE str2)
03929 {
03930     str_modifiable(str);
03931     if (str == str2) return str;
03932 
03933     StringValue(str2);
03934     str_discard(str);
03935     return str_replace(str, str2);
03936 }
03937 
03938 /*
03939  *  call-seq:
03940  *     string.clear    ->  string
03941  *
03942  *  Makes string empty.
03943  *
03944  *     a = "abcde"
03945  *     a.clear    #=> ""
03946  */
03947 
03948 static VALUE
03949 rb_str_clear(VALUE str)
03950 {
03951     str_discard(str);
03952     STR_SET_EMBED(str);
03953     STR_SET_EMBED_LEN(str, 0);
03954     RSTRING_PTR(str)[0] = 0;
03955     if (rb_enc_asciicompat(STR_ENC_GET(str)))
03956         ENC_CODERANGE_SET(str, ENC_CODERANGE_7BIT);
03957     else
03958         ENC_CODERANGE_SET(str, ENC_CODERANGE_VALID);
03959     return str;
03960 }
03961 
03962 /*
03963  *  call-seq:
03964  *     string.chr    ->  string
03965  *
03966  *  Returns a one-character string at the beginning of the string.
03967  *
03968  *     a = "abcde"
03969  *     a.chr    #=> "a"
03970  */
03971 
03972 static VALUE
03973 rb_str_chr(VALUE str)
03974 {
03975     return rb_str_substr(str, 0, 1);
03976 }
03977 
03978 /*
03979  *  call-seq:
03980  *     str.getbyte(index)          -> 0 .. 255
03981  *
03982  *  returns the <i>index</i>th byte as an integer.
03983  */
03984 static VALUE
03985 rb_str_getbyte(VALUE str, VALUE index)
03986 {
03987     long pos = NUM2LONG(index);
03988 
03989     if (pos < 0)
03990         pos += RSTRING_LEN(str);
03991     if (pos < 0 ||  RSTRING_LEN(str) <= pos)
03992         return Qnil;
03993 
03994     return INT2FIX((unsigned char)RSTRING_PTR(str)[pos]);
03995 }
03996 
03997 /*
03998  *  call-seq:
03999  *     str.setbyte(index, int) -> int
04000  *
04001  *  modifies the <i>index</i>th byte as <i>int</i>.
04002  */
04003 static VALUE
04004 rb_str_setbyte(VALUE str, VALUE index, VALUE value)
04005 {
04006     long pos = NUM2LONG(index);
04007     int byte = NUM2INT(value);
04008 
04009     rb_str_modify(str);
04010 
04011     if (pos < -RSTRING_LEN(str) || RSTRING_LEN(str) <= pos)
04012         rb_raise(rb_eIndexError, "index %ld out of string", pos);
04013     if (pos < 0)
04014         pos += RSTRING_LEN(str);
04015 
04016     RSTRING_PTR(str)[pos] = byte;
04017 
04018     return value;
04019 }
04020 
04021 static VALUE
04022 str_byte_substr(VALUE str, long beg, long len)
04023 {
04024     char *p, *s = RSTRING_PTR(str);
04025     long n = RSTRING_LEN(str);
04026     VALUE str2;
04027 
04028     if (beg > n || len < 0) return Qnil;
04029     if (beg < 0) {
04030         beg += n;
04031         if (beg < 0) return Qnil;
04032     }
04033     if (beg + len > n)
04034         len = n - beg;
04035     if (len <= 0) {
04036         len = 0;
04037         p = 0;
04038     }
04039     else
04040         p = s + beg;
04041 
04042     if (len > RSTRING_EMBED_LEN_MAX && beg + len == n) {
04043         str2 = rb_str_new4(str);
04044         str2 = str_new3(rb_obj_class(str2), str2);
04045         RSTRING(str2)->as.heap.ptr += RSTRING(str2)->as.heap.len - len;
04046         RSTRING(str2)->as.heap.len = len;
04047     }
04048     else {
04049         str2 = rb_str_new5(str, p, len);
04050         rb_enc_cr_str_copy_for_substr(str2, str);
04051         OBJ_INFECT(str2, str);
04052     }
04053 
04054     return str2;
04055 }
04056 
04057 static VALUE
04058 str_byte_aref(VALUE str, VALUE indx)
04059 {
04060     long idx;
04061     switch (TYPE(indx)) {
04062       case T_FIXNUM:
04063         idx = FIX2LONG(indx);
04064 
04065       num_index:
04066         str = str_byte_substr(str, idx, 1);
04067         if (NIL_P(str) || RSTRING_LEN(str) == 0) return Qnil;
04068         return str;
04069 
04070       default:
04071         /* check if indx is Range */
04072         {
04073             long beg, len = RSTRING_LEN(str);
04074 
04075             switch (rb_range_beg_len(indx, &beg, &len, len, 0)) {
04076               case Qfalse:
04077                 break;
04078               case Qnil:
04079                 return Qnil;
04080               default:
04081                 return str_byte_substr(str, beg, len);
04082             }
04083         }
04084         idx = NUM2LONG(indx);
04085         goto num_index;
04086     }
04087     return Qnil;                /* not reached */
04088 }
04089 
04090 /*
04091  *  call-seq:
04092  *     str.byteslice(fixnum)           -> new_str or nil
04093  *     str.byteslice(fixnum, fixnum)   -> new_str or nil
04094  *     str.byteslice(range)            -> new_str or nil
04095  *
04096  *  Byte Reference---If passed a single <code>Fixnum</code>, returns a
04097  *  substring of one byte at that position. If passed two <code>Fixnum</code>
04098  *  objects, returns a substring starting at the offset given by the first, and
04099  *  a length given by the second. If given a <code>Range</code>, a substring containing
04100  *  bytes at offsets given by the range is returned. In all three cases, if
04101  *  an offset is negative, it is counted from the end of <i>str</i>. Returns
04102  *  <code>nil</code> if the initial offset falls outside the string, the length
04103  *  is negative, or the beginning of the range is greater than the end.
04104  *  The encoding of the resulted string keeps original encoding.
04105  *
04106  *     "hello".byteslice(1)     #=> "e"
04107  *     "hello".byteslice(-1)    #=> "o"
04108  *     "hello".byteslice(1, 2)  #=> "el"
04109  *     "\x80\u3042".byteslice(1, 3) #=> "\u3042"
04110  *     "\x03\u3042\xff".byteslice(1..3) #=> "\u3942"
04111  */
04112 
04113 static VALUE
04114 rb_str_byteslice(int argc, VALUE *argv, VALUE str)
04115 {
04116     if (argc == 2) {
04117         return str_byte_substr(str, NUM2LONG(argv[0]), NUM2LONG(argv[1]));
04118     }
04119     if (argc != 1) {
04120         rb_raise(rb_eArgError, "wrong number of arguments (%d for 1..2)", argc);
04121     }
04122     return str_byte_aref(str, argv[0]);
04123 }
04124 
04125 /*
04126  *  call-seq:
04127  *     str.reverse   -> new_str
04128  *
04129  *  Returns a new string with the characters from <i>str</i> in reverse order.
04130  *
04131  *     "stressed".reverse   #=> "desserts"
04132  */
04133 
04134 static VALUE
04135 rb_str_reverse(VALUE str)
04136 {
04137     rb_encoding *enc;
04138     VALUE rev;
04139     char *s, *e, *p;
04140     int single = 1;
04141 
04142     if (RSTRING_LEN(str) <= 1) return rb_str_dup(str);
04143     enc = STR_ENC_GET(str);
04144     rev = rb_str_new5(str, 0, RSTRING_LEN(str));
04145     s = RSTRING_PTR(str); e = RSTRING_END(str);
04146     p = RSTRING_END(rev);
04147 
04148     if (RSTRING_LEN(str) > 1) {
04149         if (single_byte_optimizable(str)) {
04150             while (s < e) {
04151                 *--p = *s++;
04152             }
04153         }
04154         else if (ENC_CODERANGE(str) == ENC_CODERANGE_VALID) {
04155             while (s < e) {
04156                 int clen = rb_enc_fast_mbclen(s, e, enc);
04157 
04158                 if (clen > 1 || (*s & 0x80)) single = 0;
04159                 p -= clen;
04160                 memcpy(p, s, clen);
04161                 s += clen;
04162             }
04163         }
04164         else {
04165             while (s < e) {
04166                 int clen = rb_enc_mbclen(s, e, enc);
04167 
04168                 if (clen > 1 || (*s & 0x80)) single = 0;
04169                 p -= clen;
04170                 memcpy(p, s, clen);
04171                 s += clen;
04172             }
04173         }
04174     }
04175     STR_SET_LEN(rev, RSTRING_LEN(str));
04176     OBJ_INFECT(rev, str);
04177     if (ENC_CODERANGE(str) == ENC_CODERANGE_UNKNOWN) {
04178         if (single) {
04179             ENC_CODERANGE_SET(str, ENC_CODERANGE_7BIT);
04180         }
04181         else {
04182             ENC_CODERANGE_SET(str, ENC_CODERANGE_VALID);
04183         }
04184     }
04185     rb_enc_cr_str_copy_for_substr(rev, str);
04186 
04187     return rev;
04188 }
04189 
04190 
04191 /*
04192  *  call-seq:
04193  *     str.reverse!   -> str
04194  *
04195  *  Reverses <i>str</i> in place.
04196  */
04197 
04198 static VALUE
04199 rb_str_reverse_bang(VALUE str)
04200 {
04201     if (RSTRING_LEN(str) > 1) {
04202         if (single_byte_optimizable(str)) {
04203             char *s, *e, c;
04204 
04205             str_modify_keep_cr(str);
04206             s = RSTRING_PTR(str);
04207             e = RSTRING_END(str) - 1;
04208             while (s < e) {
04209                 c = *s;
04210                 *s++ = *e;
04211                 *e-- = c;
04212             }
04213         }
04214         else {
04215             rb_str_shared_replace(str, rb_str_reverse(str));
04216         }
04217     }
04218     else {
04219         str_modify_keep_cr(str);
04220     }
04221     return str;
04222 }
04223 
04224 
04225 /*
04226  *  call-seq:
04227  *     str.include? other_str   -> true or false
04228  *
04229  *  Returns <code>true</code> if <i>str</i> contains the given string or
04230  *  character.
04231  *
04232  *     "hello".include? "lo"   #=> true
04233  *     "hello".include? "ol"   #=> false
04234  *     "hello".include? ?h     #=> true
04235  */
04236 
04237 static VALUE
04238 rb_str_include(VALUE str, VALUE arg)
04239 {
04240     long i;
04241 
04242     StringValue(arg);
04243     i = rb_str_index(str, arg, 0);
04244 
04245     if (i == -1) return Qfalse;
04246     return Qtrue;
04247 }
04248 
04249 
04250 /*
04251  *  call-seq:
04252  *     str.to_i(base=10)   -> integer
04253  *
04254  *  Returns the result of interpreting leading characters in <i>str</i> as an
04255  *  integer base <i>base</i> (between 2 and 36). Extraneous characters past the
04256  *  end of a valid number are ignored. If there is not a valid number at the
04257  *  start of <i>str</i>, <code>0</code> is returned. This method never raises an
04258  *  exception when <i>base</i> is valid.
04259  *
04260  *     "12345".to_i             #=> 12345
04261  *     "99 red balloons".to_i   #=> 99
04262  *     "0a".to_i                #=> 0
04263  *     "0a".to_i(16)            #=> 10
04264  *     "hello".to_i             #=> 0
04265  *     "1100101".to_i(2)        #=> 101
04266  *     "1100101".to_i(8)        #=> 294977
04267  *     "1100101".to_i(10)       #=> 1100101
04268  *     "1100101".to_i(16)       #=> 17826049
04269  */
04270 
04271 static VALUE
04272 rb_str_to_i(int argc, VALUE *argv, VALUE str)
04273 {
04274     int base;
04275 
04276     if (argc == 0) base = 10;
04277     else {
04278         VALUE b;
04279 
04280         rb_scan_args(argc, argv, "01", &b);
04281         base = NUM2INT(b);
04282     }
04283     if (base < 0) {
04284         rb_raise(rb_eArgError, "invalid radix %d", base);
04285     }
04286     return rb_str_to_inum(str, base, FALSE);
04287 }
04288 
04289 
04290 /*
04291  *  call-seq:
04292  *     str.to_f   -> float
04293  *
04294  *  Returns the result of interpreting leading characters in <i>str</i> as a
04295  *  floating point number. Extraneous characters past the end of a valid number
04296  *  are ignored. If there is not a valid number at the start of <i>str</i>,
04297  *  <code>0.0</code> is returned. This method never raises an exception.
04298  *
04299  *     "123.45e1".to_f        #=> 1234.5
04300  *     "45.67 degrees".to_f   #=> 45.67
04301  *     "thx1138".to_f         #=> 0.0
04302  */
04303 
04304 static VALUE
04305 rb_str_to_f(VALUE str)
04306 {
04307     return DBL2NUM(rb_str_to_dbl(str, FALSE));
04308 }
04309 
04310 
04311 /*
04312  *  call-seq:
04313  *     str.to_s     -> str
04314  *     str.to_str   -> str
04315  *
04316  *  Returns the receiver.
04317  */
04318 
04319 static VALUE
04320 rb_str_to_s(VALUE str)
04321 {
04322     if (rb_obj_class(str) != rb_cString) {
04323         return str_duplicate(rb_cString, str);
04324     }
04325     return str;
04326 }
04327 
04328 #if 0
04329 static void
04330 str_cat_char(VALUE str, unsigned int c, rb_encoding *enc)
04331 {
04332     char s[RUBY_MAX_CHAR_LEN];
04333     int n = rb_enc_codelen(c, enc);
04334 
04335     rb_enc_mbcput(c, s, enc);
04336     rb_enc_str_buf_cat(str, s, n, enc);
04337 }
04338 #endif
04339 
04340 #define CHAR_ESC_LEN 13 /* sizeof(\x{ hex of 32bit unsigned int } \0) */
04341 
04342 int
04343 rb_str_buf_cat_escaped_char(VALUE result, unsigned int c, int unicode_p)
04344 {
04345     char buf[CHAR_ESC_LEN + 1];
04346     int l;
04347 
04348 #if SIZEOF_INT > 4
04349     c &= 0xffffffff;
04350 #endif
04351     if (unicode_p) {
04352         if (c < 0x7F && ISPRINT(c)) {
04353             snprintf(buf, CHAR_ESC_LEN, "%c", c);
04354         }
04355         else if (c < 0x10000) {
04356             snprintf(buf, CHAR_ESC_LEN, "\\u%04X", c);
04357         }
04358         else {
04359             snprintf(buf, CHAR_ESC_LEN, "\\u{%X}", c);
04360         }
04361     }
04362     else {
04363         if (c < 0x100) {
04364             snprintf(buf, CHAR_ESC_LEN, "\\x%02X", c);
04365         }
04366         else {
04367             snprintf(buf, CHAR_ESC_LEN, "\\x{%X}", c);
04368         }
04369     }
04370     l = (int)strlen(buf);       /* CHAR_ESC_LEN cannot exceed INT_MAX */
04371     rb_str_buf_cat(result, buf, l);
04372     return l;
04373 }
04374 
04375 /*
04376  * call-seq:
04377  *   str.inspect   -> string
04378  *
04379  * Returns a printable version of _str_, surrounded by quote marks,
04380  * with special characters escaped.
04381  *
04382  *    str = "hello"
04383  *    str[3] = "\b"
04384  *    str.inspect       #=> "\"hel\\bo\""
04385  */
04386 
04387 VALUE
04388 rb_str_inspect(VALUE str)
04389 {
04390     rb_encoding *enc = STR_ENC_GET(str);
04391     const char *p, *pend, *prev;
04392     char buf[CHAR_ESC_LEN + 1];
04393     VALUE result = rb_str_buf_new(0);
04394     rb_encoding *resenc = rb_default_internal_encoding();
04395     int unicode_p = rb_enc_unicode_p(enc);
04396     int asciicompat = rb_enc_asciicompat(enc);
04397     static rb_encoding *utf16, *utf32;
04398 
04399     if (!utf16) utf16 = rb_enc_find("UTF-16");
04400     if (!utf32) utf32 = rb_enc_find("UTF-32");
04401     if (resenc == NULL) resenc = rb_default_external_encoding();
04402     if (!rb_enc_asciicompat(resenc)) resenc = rb_usascii_encoding();
04403     rb_enc_associate(result, resenc);
04404     str_buf_cat2(result, "\"");
04405 
04406     p = RSTRING_PTR(str); pend = RSTRING_END(str);
04407     prev = p;
04408     if (enc == utf16) {
04409         const unsigned char *q = (const unsigned char *)p;
04410         if (q[0] == 0xFE && q[1] == 0xFF)
04411             enc = rb_enc_find("UTF-16BE");
04412         else if (q[0] == 0xFF && q[1] == 0xFE)
04413             enc = rb_enc_find("UTF-16LE");
04414         else
04415             unicode_p = 0;
04416     }
04417     else if (enc == utf32) {
04418         const unsigned char *q = (const unsigned char *)p;
04419         if (q[0] == 0 && q[1] == 0 && q[2] == 0xFE && q[3] == 0xFF)
04420             enc = rb_enc_find("UTF-32BE");
04421         else if (q[3] == 0 && q[2] == 0 && q[1] == 0xFE && q[0] == 0xFF)
04422             enc = rb_enc_find("UTF-32LE");
04423         else
04424             unicode_p = 0;
04425     }
04426     while (p < pend) {
04427         unsigned int c, cc;
04428         int n;
04429 
04430         n = rb_enc_precise_mbclen(p, pend, enc);
04431         if (!MBCLEN_CHARFOUND_P(n)) {
04432             if (p > prev) str_buf_cat(result, prev, p - prev);
04433             n = rb_enc_mbminlen(enc);
04434             if (pend < p + n)
04435                 n = (int)(pend - p);
04436             while (n--) {
04437                 snprintf(buf, CHAR_ESC_LEN, "\\x%02X", *p & 0377);
04438                 str_buf_cat(result, buf, strlen(buf));
04439                 prev = ++p;
04440             }
04441             continue;
04442         }
04443         n = MBCLEN_CHARFOUND_LEN(n);
04444         c = rb_enc_mbc_to_codepoint(p, pend, enc);
04445         p += n;
04446         if ((asciicompat || unicode_p) &&
04447           (c == '"'|| c == '\\' ||
04448             (c == '#' &&
04449              p < pend &&
04450              MBCLEN_CHARFOUND_P(rb_enc_precise_mbclen(p,pend,enc)) &&
04451              (cc = rb_enc_codepoint(p,pend,enc),
04452               (cc == '$' || cc == '@' || cc == '{'))))) {
04453             if (p - n > prev) str_buf_cat(result, prev, p - n - prev);
04454             str_buf_cat2(result, "\\");
04455             if (asciicompat || enc == resenc) {
04456                 prev = p - n;
04457                 continue;
04458             }
04459         }
04460         switch (c) {
04461           case '\n': cc = 'n'; break;
04462           case '\r': cc = 'r'; break;
04463           case '\t': cc = 't'; break;
04464           case '\f': cc = 'f'; break;
04465           case '\013': cc = 'v'; break;
04466           case '\010': cc = 'b'; break;
04467           case '\007': cc = 'a'; break;
04468           case 033: cc = 'e'; break;
04469           default: cc = 0; break;
04470         }
04471         if (cc) {
04472             if (p - n > prev) str_buf_cat(result, prev, p - n - prev);
04473             buf[0] = '\\';
04474             buf[1] = (char)cc;
04475             str_buf_cat(result, buf, 2);
04476             prev = p;
04477             continue;
04478         }
04479         if ((enc == resenc && rb_enc_isprint(c, enc)) ||
04480             (asciicompat && rb_enc_isascii(c, enc) && ISPRINT(c))) {
04481             continue;
04482         }
04483         else {
04484             if (p - n > prev) str_buf_cat(result, prev, p - n - prev);
04485             rb_str_buf_cat_escaped_char(result, c, unicode_p);
04486             prev = p;
04487             continue;
04488         }
04489     }
04490     if (p > prev) str_buf_cat(result, prev, p - prev);
04491     str_buf_cat2(result, "\"");
04492 
04493     OBJ_INFECT(result, str);
04494     return result;
04495 }
04496 
04497 #define IS_EVSTR(p,e) ((p) < (e) && (*(p) == '$' || *(p) == '@' || *(p) == '{'))
04498 
04499 /*
04500  *  call-seq:
04501  *     str.dump   -> new_str
04502  *
04503  *  Produces a version of <i>str</i> with all nonprinting characters replaced by
04504  *  <code>\nnn</code> notation and all special characters escaped.
04505  */
04506 
04507 VALUE
04508 rb_str_dump(VALUE str)
04509 {
04510     rb_encoding *enc = rb_enc_get(str);
04511     long len;
04512     const char *p, *pend;
04513     char *q, *qend;
04514     VALUE result;
04515     int u8 = (enc == rb_utf8_encoding());
04516 
04517     len = 2;                    /* "" */
04518     p = RSTRING_PTR(str); pend = p + RSTRING_LEN(str);
04519     while (p < pend) {
04520         unsigned char c = *p++;
04521         switch (c) {
04522           case '"':  case '\\':
04523           case '\n': case '\r':
04524           case '\t': case '\f':
04525           case '\013': case '\010': case '\007': case '\033':
04526             len += 2;
04527             break;
04528 
04529           case '#':
04530             len += IS_EVSTR(p, pend) ? 2 : 1;
04531             break;
04532 
04533           default:
04534             if (ISPRINT(c)) {
04535                 len++;
04536             }
04537             else {
04538                 if (u8) {       /* \u{NN} */
04539                     int n = rb_enc_precise_mbclen(p-1, pend, enc);
04540                     if (MBCLEN_CHARFOUND_P(n-1)) {
04541                         unsigned int cc = rb_enc_mbc_to_codepoint(p-1, pend, enc);
04542                         while (cc >>= 4) len++;
04543                         len += 5;
04544                         p += MBCLEN_CHARFOUND_LEN(n)-1;
04545                         break;
04546                     }
04547                 }
04548                 len += 4;       /* \xNN */
04549             }
04550             break;
04551         }
04552     }
04553     if (!rb_enc_asciicompat(enc)) {
04554         len += 19;              /* ".force_encoding('')" */
04555         len += strlen(enc->name);
04556     }
04557 
04558     result = rb_str_new5(str, 0, len);
04559     p = RSTRING_PTR(str); pend = p + RSTRING_LEN(str);
04560     q = RSTRING_PTR(result); qend = q + len + 1;
04561 
04562     *q++ = '"';
04563     while (p < pend) {
04564         unsigned char c = *p++;
04565 
04566         if (c == '"' || c == '\\') {
04567             *q++ = '\\';
04568             *q++ = c;
04569         }
04570         else if (c == '#') {
04571             if (IS_EVSTR(p, pend)) *q++ = '\\';
04572             *q++ = '#';
04573         }
04574         else if (c == '\n') {
04575             *q++ = '\\';
04576             *q++ = 'n';
04577         }
04578         else if (c == '\r') {
04579             *q++ = '\\';
04580             *q++ = 'r';
04581         }
04582         else if (c == '\t') {
04583             *q++ = '\\';
04584             *q++ = 't';
04585         }
04586         else if (c == '\f') {
04587             *q++ = '\\';
04588             *q++ = 'f';
04589         }
04590         else if (c == '\013') {
04591             *q++ = '\\';
04592             *q++ = 'v';
04593         }
04594         else if (c == '\010') {
04595             *q++ = '\\';
04596             *q++ = 'b';
04597         }
04598         else if (c == '\007') {
04599             *q++ = '\\';
04600             *q++ = 'a';
04601         }
04602         else if (c == '\033') {
04603             *q++ = '\\';
04604             *q++ = 'e';
04605         }
04606         else if (ISPRINT(c)) {
04607             *q++ = c;
04608         }
04609         else {
04610             *q++ = '\\';
04611             if (u8) {
04612                 int n = rb_enc_precise_mbclen(p-1, pend, enc) - 1;
04613                 if (MBCLEN_CHARFOUND_P(n)) {
04614                     int cc = rb_enc_mbc_to_codepoint(p-1, pend, enc);
04615                     p += n;
04616                     snprintf(q, qend-q, "u{%x}", cc);
04617                     q += strlen(q);
04618                     continue;
04619                 }
04620             }
04621             snprintf(q, qend-q, "x%02X", c);
04622             q += 3;
04623         }
04624     }
04625     *q++ = '"';
04626     *q = '\0';
04627     if (!rb_enc_asciicompat(enc)) {
04628         snprintf(q, qend-q, ".force_encoding(\"%s\")", enc->name);
04629         enc = rb_ascii8bit_encoding();
04630     }
04631     OBJ_INFECT(result, str);
04632     /* result from dump is ASCII */
04633     rb_enc_associate(result, enc);
04634     ENC_CODERANGE_SET(result, ENC_CODERANGE_7BIT);
04635     return result;
04636 }
04637 
04638 
04639 static void
04640 rb_str_check_dummy_enc(rb_encoding *enc)
04641 {
04642     if (rb_enc_dummy_p(enc)) {
04643         rb_raise(rb_eEncCompatError, "incompatible encoding with this operation: %s",
04644                  rb_enc_name(enc));
04645     }
04646 }
04647 
04648 /*
04649  *  call-seq:
04650  *     str.upcase!   -> str or nil
04651  *
04652  *  Upcases the contents of <i>str</i>, returning <code>nil</code> if no changes
04653  *  were made.
04654  *  Note: case replacement is effective only in ASCII region.
04655  */
04656 
04657 static VALUE
04658 rb_str_upcase_bang(VALUE str)
04659 {
04660     rb_encoding *enc;
04661     char *s, *send;
04662     int modify = 0;
04663     int n;
04664 
04665     str_modify_keep_cr(str);
04666     enc = STR_ENC_GET(str);
04667     rb_str_check_dummy_enc(enc);
04668     s = RSTRING_PTR(str); send = RSTRING_END(str);
04669     if (single_byte_optimizable(str)) {
04670         while (s < send) {
04671             unsigned int c = *(unsigned char*)s;
04672 
04673             if (rb_enc_isascii(c, enc) && 'a' <= c && c <= 'z') {
04674                 *s = 'A' + (c - 'a');
04675                 modify = 1;
04676             }
04677             s++;
04678         }
04679     }
04680     else {
04681         int ascompat = rb_enc_asciicompat(enc);
04682 
04683         while (s < send) {
04684             unsigned int c;
04685 
04686             if (ascompat && (c = *(unsigned char*)s) < 0x80) {
04687                 if (rb_enc_isascii(c, enc) && 'a' <= c && c <= 'z') {
04688                     *s = 'A' + (c - 'a');
04689                     modify = 1;
04690                 }
04691                 s++;
04692             }
04693             else {
04694                 c = rb_enc_codepoint_len(s, send, &n, enc);
04695                 if (rb_enc_islower(c, enc)) {
04696                     /* assuming toupper returns codepoint with same size */
04697                     rb_enc_mbcput(rb_enc_toupper(c, enc), s, enc);
04698                     modify = 1;
04699                 }
04700                 s += n;
04701             }
04702         }
04703     }
04704 
04705     if (modify) return str;
04706     return Qnil;
04707 }
04708 
04709 
04710 /*
04711  *  call-seq:
04712  *     str.upcase   -> new_str
04713  *
04714  *  Returns a copy of <i>str</i> with all lowercase letters replaced with their
04715  *  uppercase counterparts. The operation is locale insensitive---only
04716  *  characters ``a'' to ``z'' are affected.
04717  *  Note: case replacement is effective only in ASCII region.
04718  *
04719  *     "hEllO".upcase   #=> "HELLO"
04720  */
04721 
04722 static VALUE
04723 rb_str_upcase(VALUE str)
04724 {
04725     str = rb_str_dup(str);
04726     rb_str_upcase_bang(str);
04727     return str;
04728 }
04729 
04730 
04731 /*
04732  *  call-seq:
04733  *     str.downcase!   -> str or nil
04734  *
04735  *  Downcases the contents of <i>str</i>, returning <code>nil</code> if no
04736  *  changes were made.
04737  *  Note: case replacement is effective only in ASCII region.
04738  */
04739 
04740 static VALUE
04741 rb_str_downcase_bang(VALUE str)
04742 {
04743     rb_encoding *enc;
04744     char *s, *send;
04745     int modify = 0;
04746 
04747     str_modify_keep_cr(str);
04748     enc = STR_ENC_GET(str);
04749     rb_str_check_dummy_enc(enc);
04750     s = RSTRING_PTR(str); send = RSTRING_END(str);
04751     if (single_byte_optimizable(str)) {
04752         while (s < send) {
04753             unsigned int c = *(unsigned char*)s;
04754 
04755             if (rb_enc_isascii(c, enc) && 'A' <= c && c <= 'Z') {
04756                 *s = 'a' + (c - 'A');
04757                 modify = 1;
04758             }
04759             s++;
04760         }
04761     }
04762     else {
04763         int ascompat = rb_enc_asciicompat(enc);
04764 
04765         while (s < send) {
04766             unsigned int c;
04767             int n;
04768 
04769             if (ascompat && (c = *(unsigned char*)s) < 0x80) {
04770                 if (rb_enc_isascii(c, enc) && 'A' <= c && c <= 'Z') {
04771                     *s = 'a' + (c - 'A');
04772                     modify = 1;
04773                 }
04774                 s++;
04775             }
04776             else {
04777                 c = rb_enc_codepoint_len(s, send, &n, enc);
04778                 if (rb_enc_isupper(c, enc)) {
04779                     /* assuming toupper returns codepoint with same size */
04780                     rb_enc_mbcput(rb_enc_tolower(c, enc), s, enc);
04781                     modify = 1;
04782                 }
04783                 s += n;
04784             }
04785         }
04786     }
04787 
04788     if (modify) return str;
04789     return Qnil;
04790 }
04791 
04792 
04793 /*
04794  *  call-seq:
04795  *     str.downcase   -> new_str
04796  *
04797  *  Returns a copy of <i>str</i> with all uppercase letters replaced with their
04798  *  lowercase counterparts. The operation is locale insensitive---only
04799  *  characters ``A'' to ``Z'' are affected.
04800  *  Note: case replacement is effective only in ASCII region.
04801  *
04802  *     "hEllO".downcase   #=> "hello"
04803  */
04804 
04805 static VALUE
04806 rb_str_downcase(VALUE str)
04807 {
04808     str = rb_str_dup(str);
04809     rb_str_downcase_bang(str);
04810     return str;
04811 }
04812 
04813 
04814 /*
04815  *  call-seq:
04816  *     str.capitalize!   -> str or nil
04817  *
04818  *  Modifies <i>str</i> by converting the first character to uppercase and the
04819  *  remainder to lowercase. Returns <code>nil</code> if no changes are made.
04820  *  Note: case conversion is effective only in ASCII region.
04821  *
04822  *     a = "hello"
04823  *     a.capitalize!   #=> "Hello"
04824  *     a               #=> "Hello"
04825  *     a.capitalize!   #=> nil
04826  */
04827 
04828 static VALUE
04829 rb_str_capitalize_bang(VALUE str)
04830 {
04831     rb_encoding *enc;
04832     char *s, *send;
04833     int modify = 0;
04834     unsigned int c;
04835     int n;
04836 
04837     str_modify_keep_cr(str);
04838     enc = STR_ENC_GET(str);
04839     rb_str_check_dummy_enc(enc);
04840     if (RSTRING_LEN(str) == 0 || !RSTRING_PTR(str)) return Qnil;
04841     s = RSTRING_PTR(str); send = RSTRING_END(str);
04842 
04843     c = rb_enc_codepoint_len(s, send, &n, enc);
04844     if (rb_enc_islower(c, enc)) {
04845         rb_enc_mbcput(rb_enc_toupper(c, enc), s, enc);
04846         modify = 1;
04847     }
04848     s += n;
04849     while (s < send) {
04850         c = rb_enc_codepoint_len(s, send, &n, enc);
04851         if (rb_enc_isupper(c, enc)) {
04852             rb_enc_mbcput(rb_enc_tolower(c, enc), s, enc);
04853             modify = 1;
04854         }
04855         s += n;
04856     }
04857 
04858     if (modify) return str;
04859     return Qnil;
04860 }
04861 
04862 
04863 /*
04864  *  call-seq:
04865  *     str.capitalize   -> new_str
04866  *
04867  *  Returns a copy of <i>str</i> with the first character converted to uppercase
04868  *  and the remainder to lowercase.
04869  *  Note: case conversion is effective only in ASCII region.
04870  *
04871  *     "hello".capitalize    #=> "Hello"
04872  *     "HELLO".capitalize    #=> "Hello"
04873  *     "123ABC".capitalize   #=> "123abc"
04874  */
04875 
04876 static VALUE
04877 rb_str_capitalize(VALUE str)
04878 {
04879     str = rb_str_dup(str);
04880     rb_str_capitalize_bang(str);
04881     return str;
04882 }
04883 
04884 
04885 /*
04886  *  call-seq:
04887  *     str.swapcase!   -> str or nil
04888  *
04889  *  Equivalent to <code>String#swapcase</code>, but modifies the receiver in
04890  *  place, returning <i>str</i>, or <code>nil</code> if no changes were made.
04891  *  Note: case conversion is effective only in ASCII region.
04892  */
04893 
04894 static VALUE
04895 rb_str_swapcase_bang(VALUE str)
04896 {
04897     rb_encoding *enc;
04898     char *s, *send;
04899     int modify = 0;
04900     int n;
04901 
04902     str_modify_keep_cr(str);
04903     enc = STR_ENC_GET(str);
04904     rb_str_check_dummy_enc(enc);
04905     s = RSTRING_PTR(str); send = RSTRING_END(str);
04906     while (s < send) {
04907         unsigned int c = rb_enc_codepoint_len(s, send, &n, enc);
04908 
04909         if (rb_enc_isupper(c, enc)) {
04910             /* assuming toupper returns codepoint with same size */
04911             rb_enc_mbcput(rb_enc_tolower(c, enc), s, enc);
04912             modify = 1;
04913         }
04914         else if (rb_enc_islower(c, enc)) {
04915             /* assuming tolower returns codepoint with same size */
04916             rb_enc_mbcput(rb_enc_toupper(c, enc), s, enc);
04917             modify = 1;
04918         }
04919         s += n;
04920     }
04921 
04922     if (modify) return str;
04923     return Qnil;
04924 }
04925 
04926 
04927 /*
04928  *  call-seq:
04929  *     str.swapcase   -> new_str
04930  *
04931  *  Returns a copy of <i>str</i> with uppercase alphabetic characters converted
04932  *  to lowercase and lowercase characters converted to uppercase.
04933  *  Note: case conversion is effective only in ASCII region.
04934  *
04935  *     "Hello".swapcase          #=> "hELLO"
04936  *     "cYbEr_PuNk11".swapcase   #=> "CyBeR_pUnK11"
04937  */
04938 
04939 static VALUE
04940 rb_str_swapcase(VALUE str)
04941 {
04942     str = rb_str_dup(str);
04943     rb_str_swapcase_bang(str);
04944     return str;
04945 }
04946 
04947 typedef unsigned char *USTR;
04948 
04949 struct tr {
04950     int gen;
04951     unsigned int now, max;
04952     char *p, *pend;
04953 };
04954 
04955 static unsigned int
04956 trnext(struct tr *t, rb_encoding *enc)
04957 {
04958     int n;
04959 
04960     for (;;) {
04961         if (!t->gen) {
04962             if (t->p == t->pend) return -1;
04963             if (t->p < t->pend - 1 && *t->p == '\\') {
04964                 t->p++;
04965             }
04966             t->now = rb_enc_codepoint_len(t->p, t->pend, &n, enc);
04967             t->p += n;
04968             if (t->p < t->pend - 1 && *t->p == '-') {
04969                 t->p++;
04970                 if (t->p < t->pend) {
04971                     unsigned int c = rb_enc_codepoint_len(t->p, t->pend, &n, enc);
04972                     t->p += n;
04973                     if (t->now > c) {
04974                         if (t->now < 0x80 && c < 0x80) {
04975                             rb_raise(rb_eArgError,
04976                                      "invalid range \"%c-%c\" in string transliteration",
04977                                      t->now, c);
04978                         }
04979                         else {
04980                             rb_raise(rb_eArgError, "invalid range in string transliteration");
04981                         }
04982                         continue; /* not reached */
04983                     }
04984                     t->gen = 1;
04985                     t->max = c;
04986                 }
04987             }
04988             return t->now;
04989         }
04990         else if (++t->now < t->max) {
04991             return t->now;
04992         }
04993         else {
04994             t->gen = 0;
04995             return t->max;
04996         }
04997     }
04998 }
04999 
05000 static VALUE rb_str_delete_bang(int,VALUE*,VALUE);
05001 
05002 static VALUE
05003 tr_trans(VALUE str, VALUE src, VALUE repl, int sflag)
05004 {
05005     const unsigned int errc = -1;
05006     unsigned int trans[256];
05007     rb_encoding *enc, *e1, *e2;
05008     struct tr trsrc, trrepl;
05009     int cflag = 0;
05010     unsigned int c, c0, last = 0;
05011     int modify = 0, i, l;
05012     char *s, *send;
05013     VALUE hash = 0;
05014     int singlebyte = single_byte_optimizable(str);
05015     int cr;
05016 
05017 #define CHECK_IF_ASCII(c) \
05018     (void)((cr == ENC_CODERANGE_7BIT && !rb_isascii(c)) ? \
05019            (cr = ENC_CODERANGE_VALID) : 0)
05020 
05021     StringValue(src);
05022     StringValue(repl);
05023     if (RSTRING_LEN(str) == 0 || !RSTRING_PTR(str)) return Qnil;
05024     if (RSTRING_LEN(repl) == 0) {
05025         return rb_str_delete_bang(1, &src, str);
05026     }
05027 
05028     cr = ENC_CODERANGE(str);
05029     e1 = rb_enc_check(str, src);
05030     e2 = rb_enc_check(str, repl);
05031     if (e1 == e2) {
05032         enc = e1;
05033     }
05034     else {
05035         enc = rb_enc_check(src, repl);
05036     }
05037     trsrc.p = RSTRING_PTR(src); trsrc.pend = trsrc.p + RSTRING_LEN(src);
05038     if (RSTRING_LEN(src) > 1 &&
05039         rb_enc_ascget(trsrc.p, trsrc.pend, &l, enc) == '^' &&
05040         trsrc.p + l < trsrc.pend) {
05041         cflag = 1;
05042         trsrc.p += l;
05043     }
05044     trrepl.p = RSTRING_PTR(repl);
05045     trrepl.pend = trrepl.p + RSTRING_LEN(repl);
05046     trsrc.gen = trrepl.gen = 0;
05047     trsrc.now = trrepl.now = 0;
05048     trsrc.max = trrepl.max = 0;
05049 
05050     if (cflag) {
05051         for (i=0; i<256; i++) {
05052             trans[i] = 1;
05053         }
05054         while ((c = trnext(&trsrc, enc)) != errc) {
05055             if (c < 256) {
05056                 trans[c] = errc;
05057             }
05058             else {
05059                 if (!hash) hash = rb_hash_new();
05060                 rb_hash_aset(hash, UINT2NUM(c), Qtrue);
05061             }
05062         }
05063         while ((c = trnext(&trrepl, enc)) != errc)
05064             /* retrieve last replacer */;
05065         last = trrepl.now;
05066         for (i=0; i<256; i++) {
05067             if (trans[i] != errc) {
05068                 trans[i] = last;
05069             }
05070         }
05071     }
05072     else {
05073         unsigned int r;
05074 
05075         for (i=0; i<256; i++) {
05076             trans[i] = errc;
05077         }
05078         while ((c = trnext(&trsrc, enc)) != errc) {
05079             r = trnext(&trrepl, enc);
05080             if (r == errc) r = trrepl.now;
05081             if (c < 256) {
05082                 trans[c] = r;
05083                 if (rb_enc_codelen(r, enc) != 1) singlebyte = 0;
05084             }
05085             else {
05086                 if (!hash) hash = rb_hash_new();
05087                 rb_hash_aset(hash, UINT2NUM(c), UINT2NUM(r));
05088             }
05089         }
05090     }
05091 
05092     if (cr == ENC_CODERANGE_VALID)
05093         cr = ENC_CODERANGE_7BIT;
05094     str_modify_keep_cr(str);
05095     s = RSTRING_PTR(str); send = RSTRING_END(str);
05096     if (sflag) {
05097         int clen, tlen;
05098         long offset, max = RSTRING_LEN(str);
05099         unsigned int save = -1;
05100         char *buf = ALLOC_N(char, max), *t = buf;
05101 
05102         while (s < send) {
05103             int may_modify = 0;
05104 
05105             c0 = c = rb_enc_codepoint_len(s, send, &clen, e1);
05106             tlen = enc == e1 ? clen : rb_enc_codelen(c, enc);
05107 
05108             s += clen;
05109             if (c < 256) {
05110                 c = trans[c];
05111             }
05112             else if (hash) {
05113                 VALUE tmp = rb_hash_lookup(hash, UINT2NUM(c));
05114                 if (NIL_P(tmp)) {
05115                     if (cflag) c = last;
05116                     else c = errc;
05117                 }
05118                 else if (cflag) c = errc;
05119                 else c = NUM2INT(tmp);
05120             }
05121             else {
05122                 c = errc;
05123             }
05124             if (c != (unsigned int)-1) {
05125                 if (save == c) {
05126                     CHECK_IF_ASCII(c);
05127                     continue;
05128                 }
05129                 save = c;
05130                 tlen = rb_enc_codelen(c, enc);
05131                 modify = 1;
05132             }
05133             else {
05134                 save = -1;
05135                 c = c0;
05136                 if (enc != e1) may_modify = 1;
05137             }
05138             while (t - buf + tlen >= max) {
05139                 offset = t - buf;
05140                 max *= 2;
05141                 REALLOC_N(buf, char, max);
05142                 t = buf + offset;
05143             }
05144             rb_enc_mbcput(c, t, enc);
05145             if (may_modify && memcmp(s, t, tlen) != 0) {
05146                 modify = 1;
05147             }
05148             CHECK_IF_ASCII(c);
05149             t += tlen;
05150         }
05151         if (!STR_EMBED_P(str)) {
05152             xfree(RSTRING(str)->as.heap.ptr);
05153         }
05154         *t = '\0';
05155         RSTRING(str)->as.heap.ptr = buf;
05156         RSTRING(str)->as.heap.len = t - buf;
05157         STR_SET_NOEMBED(str);
05158         RSTRING(str)->as.heap.aux.capa = max;
05159     }
05160     else if (rb_enc_mbmaxlen(enc) == 1 || (singlebyte && !hash)) {
05161         while (s < send) {
05162             c = (unsigned char)*s;
05163             if (trans[c] != errc) {
05164                 if (!cflag) {
05165                     c = trans[c];
05166                     *s = c;
05167                     modify = 1;
05168                 }
05169                 else {
05170                     *s = last;
05171                     modify = 1;
05172                 }
05173             }
05174             CHECK_IF_ASCII(c);
05175             s++;
05176         }
05177     }
05178     else {
05179         int clen, tlen, max = (int)(RSTRING_LEN(str) * 1.2);
05180         long offset;
05181         char *buf = ALLOC_N(char, max), *t = buf;
05182 
05183         while (s < send) {
05184             int may_modify = 0;
05185             c0 = c = rb_enc_codepoint_len(s, send, &clen, e1);
05186             tlen = enc == e1 ? clen : rb_enc_codelen(c, enc);
05187 
05188             if (c < 256) {
05189                 c = trans[c];
05190             }
05191             else if (hash) {
05192                 VALUE tmp = rb_hash_lookup(hash, UINT2NUM(c));
05193                 if (NIL_P(tmp)) {
05194                     if (cflag) c = last;
05195                     else c = errc;
05196                 }
05197                 else if (cflag) c = errc;
05198                 else c = NUM2INT(tmp);
05199             }
05200             else {
05201                 c = cflag ? last : errc;
05202             }
05203             if (c != errc) {
05204                 tlen = rb_enc_codelen(c, enc);
05205                 modify = 1;
05206             }
05207             else {
05208                 c = c0;
05209                 if (enc != e1) may_modify = 1;
05210             }
05211             while (t - buf + tlen >= max) {
05212                 offset = t - buf;
05213                 max *= 2;
05214                 REALLOC_N(buf, char, max);
05215                 t = buf + offset;
05216             }
05217             if (s != t) {
05218                 rb_enc_mbcput(c, t, enc);
05219                 if (may_modify && memcmp(s, t, tlen) != 0) {
05220                     modify = 1;
05221                 }
05222             }
05223             CHECK_IF_ASCII(c);
05224             s += clen;
05225             t += tlen;
05226         }
05227         if (!STR_EMBED_P(str)) {
05228             xfree(RSTRING(str)->as.heap.ptr);
05229         }
05230         *t = '\0';
05231         RSTRING(str)->as.heap.ptr = buf;
05232         RSTRING(str)->as.heap.len = t - buf;
05233         STR_SET_NOEMBED(str);
05234         RSTRING(str)->as.heap.aux.capa = max;
05235     }
05236 
05237     if (modify) {
05238         if (cr != ENC_CODERANGE_BROKEN)
05239             ENC_CODERANGE_SET(str, cr);
05240         rb_enc_associate(str, enc);
05241         return str;
05242     }
05243     return Qnil;
05244 }
05245 
05246 
05247 /*
05248  *  call-seq:
05249  *     str.tr!(from_str, to_str)   -> str or nil
05250  *
05251  *  Translates <i>str</i> in place, using the same rules as
05252  *  <code>String#tr</code>. Returns <i>str</i>, or <code>nil</code> if no
05253  *  changes were made.
05254  */
05255 
05256 static VALUE
05257 rb_str_tr_bang(VALUE str, VALUE src, VALUE repl)
05258 {
05259     return tr_trans(str, src, repl, 0);
05260 }
05261 
05262 
05263 /*
05264  *  call-seq:
05265  *     str.tr(from_str, to_str)   => new_str
05266  *
05267  *  Returns a copy of <i>str</i> with the characters in <i>from_str</i>
05268  *  replaced by the corresponding characters in <i>to_str</i>. If
05269  *  <i>to_str</i> is shorter than <i>from_str</i>, it is padded with its last
05270  *  character in order to maintain the correspondence.
05271  *
05272  *     "hello".tr('el', 'ip')      #=> "hippo"
05273  *     "hello".tr('aeiou', '*')    #=> "h*ll*"
05274  *
05275  *  Both strings may use the c1-c2 notation to denote ranges of characters,
05276  *  and <i>from_str</i> may start with a <code>^</code>, which denotes all
05277  *  characters except those listed.
05278  *
05279  *     "hello".tr('a-y', 'b-z')    #=> "ifmmp"
05280  *     "hello".tr('^aeiou', '*')   #=> "*e**o"
05281  */
05282 
05283 static VALUE
05284 rb_str_tr(VALUE str, VALUE src, VALUE repl)
05285 {
05286     str = rb_str_dup(str);
05287     tr_trans(str, src, repl, 0);
05288     return str;
05289 }
05290 
05291 #define TR_TABLE_SIZE 257
05292 static void
05293 tr_setup_table(VALUE str, char stable[TR_TABLE_SIZE], int first,
05294                VALUE *tablep, VALUE *ctablep, rb_encoding *enc)
05295 {
05296     const unsigned int errc = -1;
05297     char buf[256];
05298     struct tr tr;
05299     unsigned int c;
05300     VALUE table = 0, ptable = 0;
05301     int i, l, cflag = 0;
05302 
05303     tr.p = RSTRING_PTR(str); tr.pend = tr.p + RSTRING_LEN(str);
05304     tr.gen = tr.now = tr.max = 0;
05305 
05306     if (RSTRING_LEN(str) > 1 && rb_enc_ascget(tr.p, tr.pend, &l, enc) == '^') {
05307         cflag = 1;
05308         tr.p += l;
05309     }
05310     if (first) {
05311         for (i=0; i<256; i++) {
05312             stable[i] = 1;
05313         }
05314         stable[256] = cflag;
05315     }
05316     else if (stable[256] && !cflag) {
05317         stable[256] = 0;
05318     }
05319     for (i=0; i<256; i++) {
05320         buf[i] = cflag;
05321     }
05322 
05323     while ((c = trnext(&tr, enc)) != errc) {
05324         if (c < 256) {
05325             buf[c & 0xff] = !cflag;
05326         }
05327         else {
05328             VALUE key = UINT2NUM(c);
05329 
05330             if (!table) {
05331                 table = rb_hash_new();
05332                 if (cflag) {
05333                     ptable = *ctablep;
05334                     *ctablep = table;
05335                 }
05336                 else {
05337                     ptable = *tablep;
05338                     *tablep = table;
05339                 }
05340             }
05341             if (!ptable || !NIL_P(rb_hash_aref(ptable, key))) {
05342                 rb_hash_aset(table, key, Qtrue);
05343             }
05344         }
05345     }
05346     for (i=0; i<256; i++) {
05347         stable[i] = stable[i] && buf[i];
05348     }
05349 }
05350 
05351 
05352 static int
05353 tr_find(unsigned int c, char table[TR_TABLE_SIZE], VALUE del, VALUE nodel)
05354 {
05355     if (c < 256) {
05356         return table[c] != 0;
05357     }
05358     else {
05359         VALUE v = UINT2NUM(c);
05360 
05361         if (del) {
05362             if (!NIL_P(rb_hash_lookup(del, v)) &&
05363                     (!nodel || NIL_P(rb_hash_lookup(nodel, v)))) {
05364                 return TRUE;
05365             }
05366         }
05367         else if (nodel && !NIL_P(rb_hash_lookup(nodel, v))) {
05368             return FALSE;
05369         }
05370         return table[256] ? TRUE : FALSE;
05371     }
05372 }
05373 
05374 /*
05375  *  call-seq:
05376  *     str.delete!([other_str]+)   -> str or nil
05377  *
05378  *  Performs a <code>delete</code> operation in place, returning <i>str</i>, or
05379  *  <code>nil</code> if <i>str</i> was not modified.
05380  */
05381 
05382 static VALUE
05383 rb_str_delete_bang(int argc, VALUE *argv, VALUE str)
05384 {
05385     char squeez[TR_TABLE_SIZE];
05386     rb_encoding *enc = 0;
05387     char *s, *send, *t;
05388     VALUE del = 0, nodel = 0;
05389     int modify = 0;
05390     int i, ascompat, cr;
05391 
05392     if (RSTRING_LEN(str) == 0 || !RSTRING_PTR(str)) return Qnil;
05393     if (argc < 1) {
05394         rb_raise(rb_eArgError, "wrong number of arguments (at least 1)");
05395     }
05396     for (i=0; i<argc; i++) {
05397         VALUE s = argv[i];
05398 
05399         StringValue(s);
05400         enc = rb_enc_check(str, s);
05401         tr_setup_table(s, squeez, i==0, &del, &nodel, enc);
05402     }
05403 
05404     str_modify_keep_cr(str);
05405     ascompat = rb_enc_asciicompat(enc);
05406     s = t = RSTRING_PTR(str);
05407     send = RSTRING_END(str);
05408     cr = ascompat ? ENC_CODERANGE_7BIT : ENC_CODERANGE_VALID;
05409     while (s < send) {
05410         unsigned int c;
05411         int clen;
05412 
05413         if (ascompat && (c = *(unsigned char*)s) < 0x80) {
05414             if (squeez[c]) {
05415                 modify = 1;
05416             }
05417             else {
05418                 if (t != s) *t = c;
05419                 t++;
05420             }
05421             s++;
05422         }
05423         else {
05424             c = rb_enc_codepoint_len(s, send, &clen, enc);
05425 
05426             if (tr_find(c, squeez, del, nodel)) {
05427                 modify = 1;
05428             }
05429             else {
05430                 if (t != s) rb_enc_mbcput(c, t, enc);
05431                 t += clen;
05432                 if (cr == ENC_CODERANGE_7BIT) cr = ENC_CODERANGE_VALID;
05433             }
05434             s += clen;
05435         }
05436     }
05437     *t = '\0';
05438     STR_SET_LEN(str, t - RSTRING_PTR(str));
05439     ENC_CODERANGE_SET(str, cr);
05440 
05441     if (modify) return str;
05442     return Qnil;
05443 }
05444 
05445 
05446 /*
05447  *  call-seq:
05448  *     str.delete([other_str]+)   -> new_str
05449  *
05450  *  Returns a copy of <i>str</i> with all characters in the intersection of its
05451  *  arguments deleted. Uses the same rules for building the set of characters as
05452  *  <code>String#count</code>.
05453  *
05454  *     "hello".delete "l","lo"        #=> "heo"
05455  *     "hello".delete "lo"            #=> "he"
05456  *     "hello".delete "aeiou", "^e"   #=> "hell"
05457  *     "hello".delete "ej-m"          #=> "ho"
05458  */
05459 
05460 static VALUE
05461 rb_str_delete(int argc, VALUE *argv, VALUE str)
05462 {
05463     str = rb_str_dup(str);
05464     rb_str_delete_bang(argc, argv, str);
05465     return str;
05466 }
05467 
05468 
05469 /*
05470  *  call-seq:
05471  *     str.squeeze!([other_str]*)   -> str or nil
05472  *
05473  *  Squeezes <i>str</i> in place, returning either <i>str</i>, or
05474  *  <code>nil</code> if no changes were made.
05475  */
05476 
05477 static VALUE
05478 rb_str_squeeze_bang(int argc, VALUE *argv, VALUE str)
05479 {
05480     char squeez[TR_TABLE_SIZE];
05481     rb_encoding *enc = 0;
05482     VALUE del = 0, nodel = 0;
05483     char *s, *send, *t;
05484     int i, modify = 0;
05485     int ascompat, singlebyte = single_byte_optimizable(str);
05486     unsigned int save;
05487 
05488     if (argc == 0) {
05489         enc = STR_ENC_GET(str);
05490     }
05491     else {
05492         for (i=0; i<argc; i++) {
05493             VALUE s = argv[i];
05494 
05495             StringValue(s);
05496             enc = rb_enc_check(str, s);
05497             if (singlebyte && !single_byte_optimizable(s))
05498                 singlebyte = 0;
05499             tr_setup_table(s, squeez, i==0, &del, &nodel, enc);
05500         }
05501     }
05502 
05503     str_modify_keep_cr(str);
05504     s = t = RSTRING_PTR(str);
05505     if (!s || RSTRING_LEN(str) == 0) return Qnil;
05506     send = RSTRING_END(str);
05507     save = -1;
05508     ascompat = rb_enc_asciicompat(enc);
05509 
05510     if (singlebyte) {
05511         while (s < send) {
05512             unsigned int c = *(unsigned char*)s++;
05513             if (c != save || (argc > 0 && !squeez[c])) {
05514                 *t++ = save = c;
05515             }
05516         }
05517     } else {
05518         while (s < send) {
05519             unsigned int c;
05520             int clen;
05521 
05522             if (ascompat && (c = *(unsigned char*)s) < 0x80) {
05523                 if (c != save || (argc > 0 && !squeez[c])) {
05524                     *t++ = save = c;
05525                 }
05526                 s++;
05527             }
05528             else {
05529                 c = rb_enc_codepoint_len(s, send, &clen, enc);
05530 
05531                 if (c != save || (argc > 0 && !tr_find(c, squeez, del, nodel))) {
05532                     if (t != s) rb_enc_mbcput(c, t, enc);
05533                     save = c;
05534                     t += clen;
05535                 }
05536                 s += clen;
05537             }
05538         }
05539     }
05540 
05541     *t = '\0';
05542     if (t - RSTRING_PTR(str) != RSTRING_LEN(str)) {
05543         STR_SET_LEN(str, t - RSTRING_PTR(str));
05544         modify = 1;
05545     }
05546 
05547     if (modify) return str;
05548     return Qnil;
05549 }
05550 
05551 
05552 /*
05553  *  call-seq:
05554  *     str.squeeze([other_str]*)    -> new_str
05555  *
05556  *  Builds a set of characters from the <i>other_str</i> parameter(s) using the
05557  *  procedure described for <code>String#count</code>. Returns a new string
05558  *  where runs of the same character that occur in this set are replaced by a
05559  *  single character. If no arguments are given, all runs of identical
05560  *  characters are replaced by a single character.
05561  *
05562  *     "yellow moon".squeeze                  #=> "yelow mon"
05563  *     "  now   is  the".squeeze(" ")         #=> " now is the"
05564  *     "putters shoot balls".squeeze("m-z")   #=> "puters shot balls"
05565  */
05566 
05567 static VALUE
05568 rb_str_squeeze(int argc, VALUE *argv, VALUE str)
05569 {
05570     str = rb_str_dup(str);
05571     rb_str_squeeze_bang(argc, argv, str);
05572     return str;
05573 }
05574 
05575 
05576 /*
05577  *  call-seq:
05578  *     str.tr_s!(from_str, to_str)   -> str or nil
05579  *
05580  *  Performs <code>String#tr_s</code> processing on <i>str</i> in place,
05581  *  returning <i>str</i>, or <code>nil</code> if no changes were made.
05582  */
05583 
05584 static VALUE
05585 rb_str_tr_s_bang(VALUE str, VALUE src, VALUE repl)
05586 {
05587     return tr_trans(str, src, repl, 1);
05588 }
05589 
05590 
05591 /*
05592  *  call-seq:
05593  *     str.tr_s(from_str, to_str)   -> new_str
05594  *
05595  *  Processes a copy of <i>str</i> as described under <code>String#tr</code>,
05596  *  then removes duplicate characters in regions that were affected by the
05597  *  translation.
05598  *
05599  *     "hello".tr_s('l', 'r')     #=> "hero"
05600  *     "hello".tr_s('el', '*')    #=> "h*o"
05601  *     "hello".tr_s('el', 'hx')   #=> "hhxo"
05602  */
05603 
05604 static VALUE
05605 rb_str_tr_s(VALUE str, VALUE src, VALUE repl)
05606 {
05607     str = rb_str_dup(str);
05608     tr_trans(str, src, repl, 1);
05609     return str;
05610 }
05611 
05612 
05613 /*
05614  *  call-seq:
05615  *     str.count([other_str]+)   -> fixnum
05616  *
05617  *  Each <i>other_str</i> parameter defines a set of characters to count.  The
05618  *  intersection of these sets defines the characters to count in
05619  *  <i>str</i>. Any <i>other_str</i> that starts with a caret (^) is
05620  *  negated. The sequence c1--c2 means all characters between c1 and c2.
05621  *
05622  *     a = "hello world"
05623  *     a.count "lo"            #=> 5
05624  *     a.count "lo", "o"       #=> 2
05625  *     a.count "hello", "^l"   #=> 4
05626  *     a.count "ej-m"          #=> 4
05627  */
05628 
05629 static VALUE
05630 rb_str_count(int argc, VALUE *argv, VALUE str)
05631 {
05632     char table[TR_TABLE_SIZE];
05633     rb_encoding *enc = 0;
05634     VALUE del = 0, nodel = 0;
05635     char *s, *send;
05636     int i;
05637     int ascompat;
05638 
05639     if (argc < 1) {
05640         rb_raise(rb_eArgError, "wrong number of arguments (at least 1)");
05641     }
05642     for (i=0; i<argc; i++) {
05643         VALUE tstr = argv[i];
05644         unsigned char c;
05645 
05646         StringValue(tstr);
05647         enc = rb_enc_check(str, tstr);
05648         if (argc == 1 && RSTRING_LEN(tstr) == 1 && rb_enc_asciicompat(enc) &&
05649             (c = RSTRING_PTR(tstr)[0]) < 0x80 && !is_broken_string(str)) {
05650             int n = 0;
05651 
05652             s = RSTRING_PTR(str);
05653             if (!s || RSTRING_LEN(str) == 0) return INT2FIX(0);
05654             send = RSTRING_END(str);
05655             while (s < send) {
05656                 if (*(unsigned char*)s++ == c) n++;
05657             }
05658             return INT2NUM(n);
05659         }
05660         tr_setup_table(tstr, table, i==0, &del, &nodel, enc);
05661     }
05662 
05663     s = RSTRING_PTR(str);
05664     if (!s || RSTRING_LEN(str) == 0) return INT2FIX(0);
05665     send = RSTRING_END(str);
05666     ascompat = rb_enc_asciicompat(enc);
05667     i = 0;
05668     while (s < send) {
05669         unsigned int c;
05670 
05671         if (ascompat && (c = *(unsigned char*)s) < 0x80) {
05672             if (table[c]) {
05673                 i++;
05674             }
05675             s++;
05676         }
05677         else {
05678             int clen;
05679             c = rb_enc_codepoint_len(s, send, &clen, enc);
05680             if (tr_find(c, table, del, nodel)) {
05681                 i++;
05682             }
05683             s += clen;
05684         }
05685     }
05686 
05687     return INT2NUM(i);
05688 }
05689 
05690 static const char isspacetable[256] = {
05691     0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 0, 0,
05692     0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
05693     1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
05694     0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
05695     0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
05696     0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
05697     0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
05698     0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
05699     0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
05700     0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
05701     0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
05702     0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
05703     0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
05704     0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
05705     0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
05706     0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0
05707 };
05708 
05709 #define ascii_isspace(c) isspacetable[(unsigned char)(c)]
05710 
05711 /*
05712  *  call-seq:
05713  *     str.split(pattern=$;, [limit])   -> anArray
05714  *
05715  *  Divides <i>str</i> into substrings based on a delimiter, returning an array
05716  *  of these substrings.
05717  *
05718  *  If <i>pattern</i> is a <code>String</code>, then its contents are used as
05719  *  the delimiter when splitting <i>str</i>. If <i>pattern</i> is a single
05720  *  space, <i>str</i> is split on whitespace, with leading whitespace and runs
05721  *  of contiguous whitespace characters ignored.
05722  *
05723  *  If <i>pattern</i> is a <code>Regexp</code>, <i>str</i> is divided where the
05724  *  pattern matches. Whenever the pattern matches a zero-length string,
05725  *  <i>str</i> is split into individual characters. If <i>pattern</i> contains
05726  *  groups, the respective matches will be returned in the array as well.
05727  *
05728  *  If <i>pattern</i> is omitted, the value of <code>$;</code> is used.  If
05729  *  <code>$;</code> is <code>nil</code> (which is the default), <i>str</i> is
05730  *  split on whitespace as if ` ' were specified.
05731  *
05732  *  If the <i>limit</i> parameter is omitted, trailing null fields are
05733  *  suppressed. If <i>limit</i> is a positive number, at most that number of
05734  *  fields will be returned (if <i>limit</i> is <code>1</code>, the entire
05735  *  string is returned as the only entry in an array). If negative, there is no
05736  *  limit to the number of fields returned, and trailing null fields are not
05737  *  suppressed.
05738  *
05739  *     " now's  the time".split        #=> ["now's", "the", "time"]
05740  *     " now's  the time".split(' ')   #=> ["now's", "the", "time"]
05741  *     " now's  the time".split(/ /)   #=> ["", "now's", "", "the", "time"]
05742  *     "1, 2.34,56, 7".split(%r{,\s*}) #=> ["1", "2.34", "56", "7"]
05743  *     "hello".split(//)               #=> ["h", "e", "l", "l", "o"]
05744  *     "hello".split(//, 3)            #=> ["h", "e", "llo"]
05745  *     "hi mom".split(%r{\s*})         #=> ["h", "i", "m", "o", "m"]
05746  *
05747  *     "mellow yellow".split("ello")   #=> ["m", "w y", "w"]
05748  *     "1,2,,3,4,,".split(',')         #=> ["1", "2", "", "3", "4"]
05749  *     "1,2,,3,4,,".split(',', 4)      #=> ["1", "2", "", "3,4,,"]
05750  *     "1,2,,3,4,,".split(',', -4)     #=> ["1", "2", "", "3", "4", "", ""]
05751  */
05752 
05753 static VALUE
05754 rb_str_split_m(int argc, VALUE *argv, VALUE str)
05755 {
05756     rb_encoding *enc;
05757     VALUE spat;
05758     VALUE limit;
05759     enum {awk, string, regexp} split_type;
05760     long beg, end, i = 0;
05761     int lim = 0;
05762     VALUE result, tmp;
05763 
05764     if (rb_scan_args(argc, argv, "02", &spat, &limit) == 2) {
05765         lim = NUM2INT(limit);
05766         if (lim <= 0) limit = Qnil;
05767         else if (lim == 1) {
05768             if (RSTRING_LEN(str) == 0)
05769                 return rb_ary_new2(0);
05770             return rb_ary_new3(1, str);
05771         }
05772         i = 1;
05773     }
05774 
05775     enc = STR_ENC_GET(str);
05776     if (NIL_P(spat)) {
05777         if (!NIL_P(rb_fs)) {
05778             spat = rb_fs;
05779             goto fs_set;
05780         }
05781         split_type = awk;
05782     }
05783     else {
05784       fs_set:
05785         if (TYPE(spat) == T_STRING) {
05786             rb_encoding *enc2 = STR_ENC_GET(spat);
05787 
05788             split_type = string;
05789             if (RSTRING_LEN(spat) == 0) {
05790                 /* Special case - split into chars */
05791                 spat = rb_reg_regcomp(spat);
05792                 split_type = regexp;
05793             }
05794             else if (rb_enc_asciicompat(enc2) == 1) {
05795                 if (RSTRING_LEN(spat) == 1 && RSTRING_PTR(spat)[0] == ' '){
05796                     split_type = awk;
05797                 }
05798             }
05799             else {
05800                 int l;
05801                 if (rb_enc_ascget(RSTRING_PTR(spat), RSTRING_END(spat), &l, enc2) == ' ' &&
05802                     RSTRING_LEN(spat) == l) {
05803                     split_type = awk;
05804                 }
05805             }
05806         }
05807         else {
05808             spat = get_pat(spat, 1);
05809             split_type = regexp;
05810         }
05811     }
05812 
05813     result = rb_ary_new();
05814     beg = 0;
05815     if (split_type == awk) {
05816         char *ptr = RSTRING_PTR(str);
05817         char *eptr = RSTRING_END(str);
05818         char *bptr = ptr;
05819         int skip = 1;
05820         unsigned int c;
05821 
05822         end = beg;
05823         if (is_ascii_string(str)) {
05824             while (ptr < eptr) {
05825                 c = (unsigned char)*ptr++;
05826                 if (skip) {
05827                     if (ascii_isspace(c)) {
05828                         beg = ptr - bptr;
05829                     }
05830                     else {
05831                         end = ptr - bptr;
05832                         skip = 0;
05833                         if (!NIL_P(limit) && lim <= i) break;
05834                     }
05835                 }
05836                 else if (ascii_isspace(c)) {
05837                     rb_ary_push(result, rb_str_subseq(str, beg, end-beg));
05838                     skip = 1;
05839                     beg = ptr - bptr;
05840                     if (!NIL_P(limit)) ++i;
05841                 }
05842                 else {
05843                     end = ptr - bptr;
05844                 }
05845             }
05846         }
05847         else {
05848             while (ptr < eptr) {
05849                 int n;
05850 
05851                 c = rb_enc_codepoint_len(ptr, eptr, &n, enc);
05852                 ptr += n;
05853                 if (skip) {
05854                     if (rb_isspace(c)) {
05855                         beg = ptr - bptr;
05856                     }
05857                     else {
05858                         end = ptr - bptr;
05859                         skip = 0;
05860                         if (!NIL_P(limit) && lim <= i) break;
05861                     }
05862                 }
05863                 else if (rb_isspace(c)) {
05864                     rb_ary_push(result, rb_str_subseq(str, beg, end-beg));
05865                     skip = 1;
05866                     beg = ptr - bptr;
05867                     if (!NIL_P(limit)) ++i;
05868                 }
05869                 else {
05870                     end = ptr - bptr;
05871                 }
05872             }
05873         }
05874     }
05875     else if (split_type == string) {
05876         char *ptr = RSTRING_PTR(str);
05877         char *temp = ptr;
05878         char *eptr = RSTRING_END(str);
05879         char *sptr = RSTRING_PTR(spat);
05880         long slen = RSTRING_LEN(spat);
05881 
05882         if (is_broken_string(str)) {
05883             rb_raise(rb_eArgError, "invalid byte sequence in %s", rb_enc_name(STR_ENC_GET(str)));
05884         }
05885         if (is_broken_string(spat)) {
05886             rb_raise(rb_eArgError, "invalid byte sequence in %s", rb_enc_name(STR_ENC_GET(spat)));
05887         }
05888         enc = rb_enc_check(str, spat);
05889         while (ptr < eptr &&
05890                (end = rb_memsearch(sptr, slen, ptr, eptr - ptr, enc)) >= 0) {
05891             /* Check we are at the start of a char */
05892             char *t = rb_enc_right_char_head(ptr, ptr + end, eptr, enc);
05893             if (t != ptr + end) {
05894                 ptr = t;
05895                 continue;
05896             }
05897             rb_ary_push(result, rb_str_subseq(str, ptr - temp, end));
05898             ptr += end + slen;
05899             if (!NIL_P(limit) && lim <= ++i) break;
05900         }
05901         beg = ptr - temp;
05902     }
05903     else {
05904         char *ptr = RSTRING_PTR(str);
05905         long len = RSTRING_LEN(str);
05906         long start = beg;
05907         long idx;
05908         int last_null = 0;
05909         struct re_registers *regs;
05910 
05911         while ((end = rb_reg_search(spat, str, start, 0)) >= 0) {
05912             regs = RMATCH_REGS(rb_backref_get());
05913             if (start == end && BEG(0) == END(0)) {
05914                 if (!ptr) {
05915                     rb_ary_push(result, str_new_empty(str));
05916                     break;
05917                 }
05918                 else if (last_null == 1) {
05919                     rb_ary_push(result, rb_str_subseq(str, beg,
05920                                                       rb_enc_fast_mbclen(ptr+beg,
05921                                                                          ptr+len,
05922                                                                          enc)));
05923                     beg = start;
05924                 }
05925                 else {
05926                     if (ptr+start == ptr+len)
05927                         start++;
05928                     else
05929                         start += rb_enc_fast_mbclen(ptr+start,ptr+len,enc);
05930                     last_null = 1;
05931                     continue;
05932                 }
05933             }
05934             else {
05935                 rb_ary_push(result, rb_str_subseq(str, beg, end-beg));
05936                 beg = start = END(0);
05937             }
05938             last_null = 0;
05939 
05940             for (idx=1; idx < regs->num_regs; idx++) {
05941                 if (BEG(idx) == -1) continue;
05942                 if (BEG(idx) == END(idx))
05943                     tmp = str_new_empty(str);
05944                 else
05945                     tmp = rb_str_subseq(str, BEG(idx), END(idx)-BEG(idx));
05946                 rb_ary_push(result, tmp);
05947             }
05948             if (!NIL_P(limit) && lim <= ++i) break;
05949         }
05950     }
05951     if (RSTRING_LEN(str) > 0 && (!NIL_P(limit) || RSTRING_LEN(str) > beg || lim < 0)) {
05952         if (RSTRING_LEN(str) == beg)
05953             tmp = str_new_empty(str);
05954         else
05955             tmp = rb_str_subseq(str, beg, RSTRING_LEN(str)-beg);
05956         rb_ary_push(result, tmp);
05957     }
05958     if (NIL_P(limit) && lim == 0) {
05959         long len;
05960         while ((len = RARRAY_LEN(result)) > 0 &&
05961                (tmp = RARRAY_PTR(result)[len-1], RSTRING_LEN(tmp) == 0))
05962             rb_ary_pop(result);
05963     }
05964 
05965     return result;
05966 }
05967 
05968 VALUE
05969 rb_str_split(VALUE str, const char *sep0)
05970 {
05971     VALUE sep;
05972 
05973     StringValue(str);
05974     sep = rb_str_new2(sep0);
05975     return rb_str_split_m(1, &sep, str);
05976 }
05977 
05978 
05979 /*
05980  *  call-seq:
05981  *     str.each_line(separator=$/) {|substr| block }   -> str
05982  *     str.each_line(separator=$/)                     -> an_enumerator
05983  *
05984  *     str.lines(separator=$/) {|substr| block }       -> str
05985  *     str.lines(separator=$/)                         -> an_enumerator
05986  *
05987  *  Splits <i>str</i> using the supplied parameter as the record separator
05988  *  (<code>$/</code> by default), passing each substring in turn to the supplied
05989  *  block. If a zero-length record separator is supplied, the string is split
05990  *  into paragraphs delimited by multiple successive newlines.
05991  *
05992  *  If no block is given, an enumerator is returned instead.
05993  *
05994  *     print "Example one\n"
05995  *     "hello\nworld".each_line {|s| p s}
05996  *     print "Example two\n"
05997  *     "hello\nworld".each_line('l') {|s| p s}
05998  *     print "Example three\n"
05999  *     "hello\n\n\nworld".each_line('') {|s| p s}
06000  *
06001  *  <em>produces:</em>
06002  *
06003  *     Example one
06004  *     "hello\n"
06005  *     "world"
06006  *     Example two
06007  *     "hel"
06008  *     "l"
06009  *     "o\nworl"
06010  *     "d"
06011  *     Example three
06012  *     "hello\n\n\n"
06013  *     "world"
06014  */
06015 
06016 static VALUE
06017 rb_str_each_line(int argc, VALUE *argv, VALUE str)
06018 {
06019     rb_encoding *enc;
06020     VALUE rs;
06021     unsigned int newline;
06022     const char *p, *pend, *s, *ptr;
06023     long len, rslen;
06024     VALUE line;
06025     int n;
06026     VALUE orig = str;
06027 
06028     if (argc == 0) {
06029         rs = rb_rs;
06030     }
06031     else {
06032         rb_scan_args(argc, argv, "01", &rs);
06033     }
06034     RETURN_ENUMERATOR(str, argc, argv);
06035     if (NIL_P(rs)) {
06036         rb_yield(str);
06037         return orig;
06038     }
06039     str = rb_str_new4(str);
06040     ptr = p = s = RSTRING_PTR(str);
06041     pend = p + RSTRING_LEN(str);
06042     len = RSTRING_LEN(str);
06043     StringValue(rs);
06044     if (rs == rb_default_rs) {
06045         enc = rb_enc_get(str);
06046         while (p < pend) {
06047             char *p0;
06048 
06049             p = memchr(p, '\n', pend - p);
06050             if (!p) break;
06051             p0 = rb_enc_left_char_head(s, p, pend, enc);
06052             if (!rb_enc_is_newline(p0, pend, enc)) {
06053                 p++;
06054                 continue;
06055             }
06056             p = p0 + rb_enc_mbclen(p0, pend, enc);
06057             line = rb_str_new5(str, s, p - s);
06058             OBJ_INFECT(line, str);
06059             rb_enc_cr_str_copy_for_substr(line, str);
06060             rb_yield(line);
06061             str_mod_check(str, ptr, len);
06062             s = p;
06063         }
06064         goto finish;
06065     }
06066 
06067     enc = rb_enc_check(str, rs);
06068     rslen = RSTRING_LEN(rs);
06069     if (rslen == 0) {
06070         newline = '\n';
06071     }
06072     else {
06073         newline = rb_enc_codepoint(RSTRING_PTR(rs), RSTRING_END(rs), enc);
06074     }
06075 
06076     while (p < pend) {
06077         unsigned int c = rb_enc_codepoint_len(p, pend, &n, enc);
06078 
06079       again:
06080         if (rslen == 0 && c == newline) {
06081             p += n;
06082             if (p < pend && (c = rb_enc_codepoint_len(p, pend, &n, enc)) != newline) {
06083                 goto again;
06084             }
06085             while (p < pend && rb_enc_codepoint(p, pend, enc) == newline) {
06086                 p += n;
06087             }
06088             p -= n;
06089         }
06090         if (c == newline &&
06091             (rslen <= 1 ||
06092              (pend - p >= rslen && memcmp(RSTRING_PTR(rs), p, rslen) == 0))) {
06093             line = rb_str_new5(str, s, p - s + (rslen ? rslen : n));
06094             OBJ_INFECT(line, str);
06095             rb_enc_cr_str_copy_for_substr(line, str);
06096             rb_yield(line);
06097             str_mod_check(str, ptr, len);
06098             s = p + (rslen ? rslen : n);
06099         }
06100         p += n;
06101     }
06102 
06103   finish:
06104     if (s != pend) {
06105         line = rb_str_new5(str, s, pend - s);
06106         OBJ_INFECT(line, str);
06107         rb_enc_cr_str_copy_for_substr(line, str);
06108         rb_yield(line);
06109     }
06110 
06111     return orig;
06112 }
06113 
06114 
06115 /*
06116  *  call-seq:
06117  *     str.bytes {|fixnum| block }        -> str
06118  *     str.bytes                          -> an_enumerator
06119  *
06120  *     str.each_byte {|fixnum| block }    -> str
06121  *     str.each_byte                      -> an_enumerator
06122  *
06123  *  Passes each byte in <i>str</i> to the given block, or returns
06124  *  an enumerator if no block is given.
06125  *
06126  *     "hello".each_byte {|c| print c, ' ' }
06127  *
06128  *  <em>produces:</em>
06129  *
06130  *     104 101 108 108 111
06131  */
06132 
06133 static VALUE
06134 rb_str_each_byte(VALUE str)
06135 {
06136     long i;
06137 
06138     RETURN_ENUMERATOR(str, 0, 0);
06139     for (i=0; i<RSTRING_LEN(str); i++) {
06140         rb_yield(INT2FIX(RSTRING_PTR(str)[i] & 0xff));
06141     }
06142     return str;
06143 }
06144 
06145 
06146 /*
06147  *  call-seq:
06148  *     str.chars {|cstr| block }        -> str
06149  *     str.chars                        -> an_enumerator
06150  *
06151  *     str.each_char {|cstr| block }    -> str
06152  *     str.each_char                    -> an_enumerator
06153  *
06154  *  Passes each character in <i>str</i> to the given block, or returns
06155  *  an enumerator if no block is given.
06156  *
06157  *     "hello".each_char {|c| print c, ' ' }
06158  *
06159  *  <em>produces:</em>
06160  *
06161  *     h e l l o
06162  */
06163 
06164 static VALUE
06165 rb_str_each_char(VALUE str)
06166 {
06167     VALUE orig = str;
06168     long i, len, n;
06169     const char *ptr;
06170     rb_encoding *enc;
06171 
06172     RETURN_ENUMERATOR(str, 0, 0);
06173     str = rb_str_new4(str);
06174     ptr = RSTRING_PTR(str);
06175     len = RSTRING_LEN(str);
06176     enc = rb_enc_get(str);
06177     switch (ENC_CODERANGE(str)) {
06178       case ENC_CODERANGE_VALID:
06179       case ENC_CODERANGE_7BIT:
06180         for (i = 0; i < len; i += n) {
06181             n = rb_enc_fast_mbclen(ptr + i, ptr + len, enc);
06182             rb_yield(rb_str_subseq(str, i, n));
06183         }
06184         break;
06185       default:
06186         for (i = 0; i < len; i += n) {
06187             n = rb_enc_mbclen(ptr + i, ptr + len, enc);
06188             rb_yield(rb_str_subseq(str, i, n));
06189         }
06190     }
06191     return orig;
06192 }
06193 
06194 /*
06195  *  call-seq:
06196  *     str.codepoints {|integer| block }        -> str
06197  *     str.codepoints                           -> an_enumerator
06198  *
06199  *     str.each_codepoint {|integer| block }    -> str
06200  *     str.each_codepoint                       -> an_enumerator
06201  *
06202  *  Passes the <code>Integer</code> ordinal of each character in <i>str</i>,
06203  *  also known as a <i>codepoint</i> when applied to Unicode strings to the
06204  *  given block.
06205  *
06206  *  If no block is given, an enumerator is returned instead.
06207  *
06208  *     "hello\u0639".each_codepoint {|c| print c, ' ' }
06209  *
06210  *  <em>produces:</em>
06211  *
06212  *     104 101 108 108 111 1593
06213  */
06214 
06215 static VALUE
06216 rb_str_each_codepoint(VALUE str)
06217 {
06218     VALUE orig = str;
06219     int n;
06220     unsigned int c;
06221     const char *ptr, *end;
06222     rb_encoding *enc;
06223 
06224     if (single_byte_optimizable(str)) return rb_str_each_byte(str);
06225     RETURN_ENUMERATOR(str, 0, 0);
06226     str = rb_str_new4(str);
06227     ptr = RSTRING_PTR(str);
06228     end = RSTRING_END(str);
06229     enc = STR_ENC_GET(str);
06230     while (ptr < end) {
06231         c = rb_enc_codepoint_len(ptr, end, &n, enc);
06232         rb_yield(UINT2NUM(c));
06233         ptr += n;
06234     }
06235     return orig;
06236 }
06237 
06238 static long
06239 chopped_length(VALUE str)
06240 {
06241     rb_encoding *enc = STR_ENC_GET(str);
06242     const char *p, *p2, *beg, *end;
06243 
06244     beg = RSTRING_PTR(str);
06245     end = beg + RSTRING_LEN(str);
06246     if (beg > end) return 0;
06247     p = rb_enc_prev_char(beg, end, end, enc);
06248     if (!p) return 0;
06249     if (p > beg && rb_enc_ascget(p, end, 0, enc) == '\n') {
06250         p2 = rb_enc_prev_char(beg, p, end, enc);
06251         if (p2 && rb_enc_ascget(p2, end, 0, enc) == '\r') p = p2;
06252     }
06253     return p - beg;
06254 }
06255 
06256 /*
06257  *  call-seq:
06258  *     str.chop!   -> str or nil
06259  *
06260  *  Processes <i>str</i> as for <code>String#chop</code>, returning <i>str</i>,
06261  *  or <code>nil</code> if <i>str</i> is the empty string.  See also
06262  *  <code>String#chomp!</code>.
06263  */
06264 
06265 static VALUE
06266 rb_str_chop_bang(VALUE str)
06267 {
06268     str_modify_keep_cr(str);
06269     if (RSTRING_LEN(str) > 0) {
06270         long len;
06271         len = chopped_length(str);
06272         STR_SET_LEN(str, len);
06273         RSTRING_PTR(str)[len] = '\0';
06274         if (ENC_CODERANGE(str) != ENC_CODERANGE_7BIT) {
06275             ENC_CODERANGE_CLEAR(str);
06276         }
06277         return str;
06278     }
06279     return Qnil;
06280 }
06281 
06282 
06283 /*
06284  *  call-seq:
06285  *     str.chop   -> new_str
06286  *
06287  *  Returns a new <code>String</code> with the last character removed.  If the
06288  *  string ends with <code>\r\n</code>, both characters are removed. Applying
06289  *  <code>chop</code> to an empty string returns an empty
06290  *  string. <code>String#chomp</code> is often a safer alternative, as it leaves
06291  *  the string unchanged if it doesn't end in a record separator.
06292  *
06293  *     "string\r\n".chop   #=> "string"
06294  *     "string\n\r".chop   #=> "string\n"
06295  *     "string\n".chop     #=> "string"
06296  *     "string".chop       #=> "strin"
06297  *     "x".chop.chop       #=> ""
06298  */
06299 
06300 static VALUE
06301 rb_str_chop(VALUE str)
06302 {
06303     VALUE str2 = rb_str_new5(str, RSTRING_PTR(str), chopped_length(str));
06304     rb_enc_cr_str_copy_for_substr(str2, str);
06305     OBJ_INFECT(str2, str);
06306     return str2;
06307 }
06308 
06309 
06310 /*
06311  *  call-seq:
06312  *     str.chomp!(separator=$/)   -> str or nil
06313  *
06314  *  Modifies <i>str</i> in place as described for <code>String#chomp</code>,
06315  *  returning <i>str</i>, or <code>nil</code> if no modifications were made.
06316  */
06317 
06318 static VALUE
06319 rb_str_chomp_bang(int argc, VALUE *argv, VALUE str)
06320 {
06321     rb_encoding *enc;
06322     VALUE rs;
06323     int newline;
06324     char *p, *pp, *e;
06325     long len, rslen;
06326 
06327     str_modify_keep_cr(str);
06328     len = RSTRING_LEN(str);
06329     if (len == 0) return Qnil;
06330     p = RSTRING_PTR(str);
06331     e = p + len;
06332     if (argc == 0) {
06333         rs = rb_rs;
06334         if (rs == rb_default_rs) {
06335           smart_chomp:
06336             enc = rb_enc_get(str);
06337             if (rb_enc_mbminlen(enc) > 1) {
06338                 pp = rb_enc_left_char_head(p, e-rb_enc_mbminlen(enc), e, enc);
06339                 if (rb_enc_is_newline(pp, e, enc)) {
06340                     e = pp;
06341                 }
06342                 pp = e - rb_enc_mbminlen(enc);
06343                 if (pp >= p) {
06344                     pp = rb_enc_left_char_head(p, pp, e, enc);
06345                     if (rb_enc_ascget(pp, e, 0, enc) == '\r') {
06346                         e = pp;
06347                     }
06348                 }
06349                 if (e == RSTRING_END(str)) {
06350                     return Qnil;
06351                 }
06352                 len = e - RSTRING_PTR(str);
06353                 STR_SET_LEN(str, len);
06354             }
06355             else {
06356                 if (RSTRING_PTR(str)[len-1] == '\n') {
06357                     STR_DEC_LEN(str);
06358                     if (RSTRING_LEN(str) > 0 &&
06359                         RSTRING_PTR(str)[RSTRING_LEN(str)-1] == '\r') {
06360                         STR_DEC_LEN(str);
06361                     }
06362                 }
06363                 else if (RSTRING_PTR(str)[len-1] == '\r') {
06364                     STR_DEC_LEN(str);
06365                 }
06366                 else {
06367                     return Qnil;
06368                 }
06369             }
06370             RSTRING_PTR(str)[RSTRING_LEN(str)] = '\0';
06371             return str;
06372         }
06373     }
06374     else {
06375         rb_scan_args(argc, argv, "01", &rs);
06376     }
06377     if (NIL_P(rs)) return Qnil;
06378     StringValue(rs);
06379     rslen = RSTRING_LEN(rs);
06380     if (rslen == 0) {
06381         while (len>0 && p[len-1] == '\n') {
06382             len--;
06383             if (len>0 && p[len-1] == '\r')
06384                 len--;
06385         }
06386         if (len < RSTRING_LEN(str)) {
06387             STR_SET_LEN(str, len);
06388             RSTRING_PTR(str)[len] = '\0';
06389             return str;
06390         }
06391         return Qnil;
06392     }
06393     if (rslen > len) return Qnil;
06394     newline = RSTRING_PTR(rs)[rslen-1];
06395     if (rslen == 1 && newline == '\n')
06396         goto smart_chomp;
06397 
06398     enc = rb_enc_check(str, rs);
06399     if (is_broken_string(rs)) {
06400         return Qnil;
06401     }
06402     pp = e - rslen;
06403     if (p[len-1] == newline &&
06404         (rslen <= 1 ||
06405          memcmp(RSTRING_PTR(rs), pp, rslen) == 0)) {
06406         if (rb_enc_left_char_head(p, pp, e, enc) != pp)
06407             return Qnil;
06408         if (ENC_CODERANGE(str) != ENC_CODERANGE_7BIT) {
06409             ENC_CODERANGE_CLEAR(str);
06410         }
06411         STR_SET_LEN(str, RSTRING_LEN(str) - rslen);
06412         RSTRING_PTR(str)[RSTRING_LEN(str)] = '\0';
06413         return str;
06414     }
06415     return Qnil;
06416 }
06417 
06418 
06419 /*
06420  *  call-seq:
06421  *     str.chomp(separator=$/)   -> new_str
06422  *
06423  *  Returns a new <code>String</code> with the given record separator removed
06424  *  from the end of <i>str</i> (if present). If <code>$/</code> has not been
06425  *  changed from the default Ruby record separator, then <code>chomp</code> also
06426  *  removes carriage return characters (that is it will remove <code>\n</code>,
06427  *  <code>\r</code>, and <code>\r\n</code>).
06428  *
06429  *     "hello".chomp            #=> "hello"
06430  *     "hello\n".chomp          #=> "hello"
06431  *     "hello\r\n".chomp        #=> "hello"
06432  *     "hello\n\r".chomp        #=> "hello\n"
06433  *     "hello\r".chomp          #=> "hello"
06434  *     "hello \n there".chomp   #=> "hello \n there"
06435  *     "hello".chomp("llo")     #=> "he"
06436  */
06437 
06438 static VALUE
06439 rb_str_chomp(int argc, VALUE *argv, VALUE str)
06440 {
06441     str = rb_str_dup(str);
06442     rb_str_chomp_bang(argc, argv, str);
06443     return str;
06444 }
06445 
06446 /*
06447  *  call-seq:
06448  *     str.lstrip!   -> self or nil
06449  *
06450  *  Removes leading whitespace from <i>str</i>, returning <code>nil</code> if no
06451  *  change was made. See also <code>String#rstrip!</code> and
06452  *  <code>String#strip!</code>.
06453  *
06454  *     "  hello  ".lstrip   #=> "hello  "
06455  *     "hello".lstrip!      #=> nil
06456  */
06457 
06458 static VALUE
06459 rb_str_lstrip_bang(VALUE str)
06460 {
06461     rb_encoding *enc;
06462     char *s, *t, *e;
06463 
06464     str_modify_keep_cr(str);
06465     enc = STR_ENC_GET(str);
06466     s = RSTRING_PTR(str);
06467     if (!s || RSTRING_LEN(str) == 0) return Qnil;
06468     e = t = RSTRING_END(str);
06469     /* remove spaces at head */
06470     while (s < e) {
06471         int n;
06472         unsigned int cc = rb_enc_codepoint_len(s, e, &n, enc);
06473 
06474         if (!rb_isspace(cc)) break;
06475         s += n;
06476     }
06477 
06478     if (s > RSTRING_PTR(str)) {
06479         STR_SET_LEN(str, t-s);
06480         memmove(RSTRING_PTR(str), s, RSTRING_LEN(str));
06481         RSTRING_PTR(str)[RSTRING_LEN(str)] = '\0';
06482         return str;
06483     }
06484     return Qnil;
06485 }
06486 
06487 
06488 /*
06489  *  call-seq:
06490  *     str.lstrip   -> new_str
06491  *
06492  *  Returns a copy of <i>str</i> with leading whitespace removed. See also
06493  *  <code>String#rstrip</code> and <code>String#strip</code>.
06494  *
06495  *     "  hello  ".lstrip   #=> "hello  "
06496  *     "hello".lstrip       #=> "hello"
06497  */
06498 
06499 static VALUE
06500 rb_str_lstrip(VALUE str)
06501 {
06502     str = rb_str_dup(str);
06503     rb_str_lstrip_bang(str);
06504     return str;
06505 }
06506 
06507 
06508 /*
06509  *  call-seq:
06510  *     str.rstrip!   -> self or nil
06511  *
06512  *  Removes trailing whitespace from <i>str</i>, returning <code>nil</code> if
06513  *  no change was made. See also <code>String#lstrip!</code> and
06514  *  <code>String#strip!</code>.
06515  *
06516  *     "  hello  ".rstrip   #=> "  hello"
06517  *     "hello".rstrip!      #=> nil
06518  */
06519 
06520 static VALUE
06521 rb_str_rstrip_bang(VALUE str)
06522 {
06523     rb_encoding *enc;
06524     char *s, *t, *e;
06525 
06526     str_modify_keep_cr(str);
06527     enc = STR_ENC_GET(str);
06528     rb_str_check_dummy_enc(enc);
06529     s = RSTRING_PTR(str);
06530     if (!s || RSTRING_LEN(str) == 0) return Qnil;
06531     t = e = RSTRING_END(str);
06532 
06533     /* remove trailing spaces or '\0's */
06534     if (single_byte_optimizable(str)) {
06535         unsigned char c;
06536         while (s < t && ((c = *(t-1)) == '\0' || ascii_isspace(c))) t--;
06537     }
06538     else {
06539         char *tp;
06540 
06541         while ((tp = rb_enc_prev_char(s, t, e, enc)) != NULL) {
06542             unsigned int c = rb_enc_codepoint(tp, e, enc);
06543             if (c && !rb_isspace(c)) break;
06544             t = tp;
06545         }
06546     }
06547     if (t < e) {
06548         long len = t-RSTRING_PTR(str);
06549 
06550         STR_SET_LEN(str, len);
06551         RSTRING_PTR(str)[len] = '\0';
06552         return str;
06553     }
06554     return Qnil;
06555 }
06556 
06557 
06558 /*
06559  *  call-seq:
06560  *     str.rstrip   -> new_str
06561  *
06562  *  Returns a copy of <i>str</i> with trailing whitespace removed. See also
06563  *  <code>String#lstrip</code> and <code>String#strip</code>.
06564  *
06565  *     "  hello  ".rstrip   #=> "  hello"
06566  *     "hello".rstrip       #=> "hello"
06567  */
06568 
06569 static VALUE
06570 rb_str_rstrip(VALUE str)
06571 {
06572     str = rb_str_dup(str);
06573     rb_str_rstrip_bang(str);
06574     return str;
06575 }
06576 
06577 
06578 /*
06579  *  call-seq:
06580  *     str.strip!   -> str or nil
06581  *
06582  *  Removes leading and trailing whitespace from <i>str</i>. Returns
06583  *  <code>nil</code> if <i>str</i> was not altered.
06584  */
06585 
06586 static VALUE
06587 rb_str_strip_bang(VALUE str)
06588 {
06589     VALUE l = rb_str_lstrip_bang(str);
06590     VALUE r = rb_str_rstrip_bang(str);
06591 
06592     if (NIL_P(l) && NIL_P(r)) return Qnil;
06593     return str;
06594 }
06595 
06596 
06597 /*
06598  *  call-seq:
06599  *     str.strip   -> new_str
06600  *
06601  *  Returns a copy of <i>str</i> with leading and trailing whitespace removed.
06602  *
06603  *     "    hello    ".strip   #=> "hello"
06604  *     "\tgoodbye\r\n".strip   #=> "goodbye"
06605  */
06606 
06607 static VALUE
06608 rb_str_strip(VALUE str)
06609 {
06610     str = rb_str_dup(str);
06611     rb_str_strip_bang(str);
06612     return str;
06613 }
06614 
06615 static VALUE
06616 scan_once(VALUE str, VALUE pat, long *start)
06617 {
06618     VALUE result, match;
06619     struct re_registers *regs;
06620     int i;
06621 
06622     if (rb_reg_search(pat, str, *start, 0) >= 0) {
06623         match = rb_backref_get();
06624         regs = RMATCH_REGS(match);
06625         if (BEG(0) == END(0)) {
06626             rb_encoding *enc = STR_ENC_GET(str);
06627             /*
06628              * Always consume at least one character of the input string
06629              */
06630             if (RSTRING_LEN(str) > END(0))
06631                 *start = END(0)+rb_enc_fast_mbclen(RSTRING_PTR(str)+END(0),
06632                                                    RSTRING_END(str), enc);
06633             else
06634                 *start = END(0)+1;
06635         }
06636         else {
06637             *start = END(0);
06638         }
06639         if (regs->num_regs == 1) {
06640             return rb_reg_nth_match(0, match);
06641         }
06642         result = rb_ary_new2(regs->num_regs);
06643         for (i=1; i < regs->num_regs; i++) {
06644             rb_ary_push(result, rb_reg_nth_match(i, match));
06645         }
06646 
06647         return result;
06648     }
06649     return Qnil;
06650 }
06651 
06652 
06653 /*
06654  *  call-seq:
06655  *     str.scan(pattern)                         -> array
06656  *     str.scan(pattern) {|match, ...| block }   -> str
06657  *
06658  *  Both forms iterate through <i>str</i>, matching the pattern (which may be a
06659  *  <code>Regexp</code> or a <code>String</code>). For each match, a result is
06660  *  generated and either added to the result array or passed to the block. If
06661  *  the pattern contains no groups, each individual result consists of the
06662  *  matched string, <code>$&</code>.  If the pattern contains groups, each
06663  *  individual result is itself an array containing one entry per group.
06664  *
06665  *     a = "cruel world"
06666  *     a.scan(/\w+/)        #=> ["cruel", "world"]
06667  *     a.scan(/.../)        #=> ["cru", "el ", "wor"]
06668  *     a.scan(/(...)/)      #=> [["cru"], ["el "], ["wor"]]
06669  *     a.scan(/(..)(..)/)   #=> [["cr", "ue"], ["l ", "wo"]]
06670  *
06671  *  And the block form:
06672  *
06673  *     a.scan(/\w+/) {|w| print "<<#{w}>> " }
06674  *     print "\n"
06675  *     a.scan(/(.)(.)/) {|x,y| print y, x }
06676  *     print "\n"
06677  *
06678  *  <em>produces:</em>
06679  *
06680  *     <<cruel>> <<world>>
06681  *     rceu lowlr
06682  */
06683 
06684 static VALUE
06685 rb_str_scan(VALUE str, VALUE pat)
06686 {
06687     VALUE result;
06688     long start = 0;
06689     long last = -1, prev = 0;
06690     char *p = RSTRING_PTR(str); long len = RSTRING_LEN(str);
06691 
06692     pat = get_pat(pat, 1);
06693     if (!rb_block_given_p()) {
06694         VALUE ary = rb_ary_new();
06695 
06696         while (!NIL_P(result = scan_once(str, pat, &start))) {
06697             last = prev;
06698             prev = start;
06699             rb_ary_push(ary, result);
06700         }
06701         if (last >= 0) rb_reg_search(pat, str, last, 0);
06702         return ary;
06703     }
06704 
06705     while (!NIL_P(result = scan_once(str, pat, &start))) {
06706         last = prev;
06707         prev = start;
06708         rb_yield(result);
06709         str_mod_check(str, p, len);
06710     }
06711     if (last >= 0) rb_reg_search(pat, str, last, 0);
06712     return str;
06713 }
06714 
06715 
06716 /*
06717  *  call-seq:
06718  *     str.hex   -> integer
06719  *
06720  *  Treats leading characters from <i>str</i> as a string of hexadecimal digits
06721  *  (with an optional sign and an optional <code>0x</code>) and returns the
06722  *  corresponding number. Zero is returned on error.
06723  *
06724  *     "0x0a".hex     #=> 10
06725  *     "-1234".hex    #=> -4660
06726  *     "0".hex        #=> 0
06727  *     "wombat".hex   #=> 0
06728  */
06729 
06730 static VALUE
06731 rb_str_hex(VALUE str)
06732 {
06733     rb_encoding *enc = rb_enc_get(str);
06734 
06735     if (!rb_enc_asciicompat(enc)) {
06736         rb_raise(rb_eEncCompatError, "ASCII incompatible encoding: %s", rb_enc_name(enc));
06737     }
06738     return rb_str_to_inum(str, 16, FALSE);
06739 }
06740 
06741 
06742 /*
06743  *  call-seq:
06744  *     str.oct   -> integer
06745  *
06746  *  Treats leading characters of <i>str</i> as a string of octal digits (with an
06747  *  optional sign) and returns the corresponding number.  Returns 0 if the
06748  *  conversion fails.
06749  *
06750  *     "123".oct       #=> 83
06751  *     "-377".oct      #=> -255
06752  *     "bad".oct       #=> 0
06753  *     "0377bad".oct   #=> 255
06754  */
06755 
06756 static VALUE
06757 rb_str_oct(VALUE str)
06758 {
06759     rb_encoding *enc = rb_enc_get(str);
06760 
06761     if (!rb_enc_asciicompat(enc)) {
06762         rb_raise(rb_eEncCompatError, "ASCII incompatible encoding: %s", rb_enc_name(enc));
06763     }
06764     return rb_str_to_inum(str, -8, FALSE);
06765 }
06766 
06767 
06768 /*
06769  *  call-seq:
06770  *     str.crypt(other_str)   -> new_str
06771  *
06772  *  Applies a one-way cryptographic hash to <i>str</i> by invoking the standard
06773  *  library function <code>crypt</code>. The argument is the salt string, which
06774  *  should be two characters long, each character drawn from
06775  *  <code>[a-zA-Z0-9./]</code>.
06776  */
06777 
06778 static VALUE
06779 rb_str_crypt(VALUE str, VALUE salt)
06780 {
06781     extern char *crypt(const char *, const char *);
06782     VALUE result;
06783     const char *s, *saltp;
06784 #ifdef BROKEN_CRYPT
06785     char salt_8bit_clean[3];
06786 #endif
06787 
06788     StringValue(salt);
06789     if (RSTRING_LEN(salt) < 2)
06790         rb_raise(rb_eArgError, "salt too short (need >=2 bytes)");
06791 
06792     s = RSTRING_PTR(str);
06793     if (!s) s = "";
06794     saltp = RSTRING_PTR(salt);
06795 #ifdef BROKEN_CRYPT
06796     if (!ISASCII((unsigned char)saltp[0]) || !ISASCII((unsigned char)saltp[1])) {
06797         salt_8bit_clean[0] = saltp[0] & 0x7f;
06798         salt_8bit_clean[1] = saltp[1] & 0x7f;
06799         salt_8bit_clean[2] = '\0';
06800         saltp = salt_8bit_clean;
06801     }
06802 #endif
06803     result = rb_str_new2(crypt(s, saltp));
06804     OBJ_INFECT(result, str);
06805     OBJ_INFECT(result, salt);
06806     return result;
06807 }
06808 
06809 
06810 /*
06811  *  call-seq:
06812  *     str.intern   -> symbol
06813  *     str.to_sym   -> symbol
06814  *
06815  *  Returns the <code>Symbol</code> corresponding to <i>str</i>, creating the
06816  *  symbol if it did not previously exist. See <code>Symbol#id2name</code>.
06817  *
06818  *     "Koala".intern         #=> :Koala
06819  *     s = 'cat'.to_sym       #=> :cat
06820  *     s == :cat              #=> true
06821  *     s = '@cat'.to_sym      #=> :@cat
06822  *     s == :@cat             #=> true
06823  *
06824  *  This can also be used to create symbols that cannot be represented using the
06825  *  <code>:xxx</code> notation.
06826  *
06827  *     'cat and dog'.to_sym   #=> :"cat and dog"
06828  */
06829 
06830 VALUE
06831 rb_str_intern(VALUE s)
06832 {
06833     VALUE str = RB_GC_GUARD(s);
06834     ID id;
06835 
06836     id = rb_intern_str(str);
06837     return ID2SYM(id);
06838 }
06839 
06840 
06841 /*
06842  *  call-seq:
06843  *     str.ord   -> integer
06844  *
06845  *  Return the <code>Integer</code> ordinal of a one-character string.
06846  *
06847  *     "a".ord         #=> 97
06848  */
06849 
06850 VALUE
06851 rb_str_ord(VALUE s)
06852 {
06853     unsigned int c;
06854 
06855     c = rb_enc_codepoint(RSTRING_PTR(s), RSTRING_END(s), STR_ENC_GET(s));
06856     return UINT2NUM(c);
06857 }
06858 /*
06859  *  call-seq:
06860  *     str.sum(n=16)   -> integer
06861  *
06862  *  Returns a basic <em>n</em>-bit checksum of the characters in <i>str</i>,
06863  *  where <em>n</em> is the optional <code>Fixnum</code> parameter, defaulting
06864  *  to 16. The result is simply the sum of the binary value of each character in
06865  *  <i>str</i> modulo <code>2**n - 1</code>. This is not a particularly good
06866  *  checksum.
06867  */
06868 
06869 static VALUE
06870 rb_str_sum(int argc, VALUE *argv, VALUE str)
06871 {
06872     VALUE vbits;
06873     int bits;
06874     char *ptr, *p, *pend;
06875     long len;
06876     VALUE sum = INT2FIX(0);
06877     unsigned long sum0 = 0;
06878 
06879     if (argc == 0) {
06880         bits = 16;
06881     }
06882     else {
06883         rb_scan_args(argc, argv, "01", &vbits);
06884         bits = NUM2INT(vbits);
06885     }
06886     ptr = p = RSTRING_PTR(str);
06887     len = RSTRING_LEN(str);
06888     pend = p + len;
06889 
06890     while (p < pend) {
06891         if (FIXNUM_MAX - UCHAR_MAX < sum0) {
06892             sum = rb_funcall(sum, '+', 1, LONG2FIX(sum0));
06893             str_mod_check(str, ptr, len);
06894             sum0 = 0;
06895         }
06896         sum0 += (unsigned char)*p;
06897         p++;
06898     }
06899 
06900     if (bits == 0) {
06901         if (sum0) {
06902             sum = rb_funcall(sum, '+', 1, LONG2FIX(sum0));
06903         }
06904     }
06905     else {
06906         if (sum == INT2FIX(0)) {
06907             if (bits < (int)sizeof(long)*CHAR_BIT) {
06908                 sum0 &= (((unsigned long)1)<<bits)-1;
06909             }
06910             sum = LONG2FIX(sum0);
06911         }
06912         else {
06913             VALUE mod;
06914 
06915             if (sum0) {
06916                 sum = rb_funcall(sum, '+', 1, LONG2FIX(sum0));
06917             }
06918 
06919             mod = rb_funcall(INT2FIX(1), rb_intern("<<"), 1, INT2FIX(bits));
06920             mod = rb_funcall(mod, '-', 1, INT2FIX(1));
06921             sum = rb_funcall(sum, '&', 1, mod);
06922         }
06923     }
06924     return sum;
06925 }
06926 
06927 static VALUE
06928 rb_str_justify(int argc, VALUE *argv, VALUE str, char jflag)
06929 {
06930     rb_encoding *enc;
06931     VALUE w;
06932     long width, len, flen = 1, fclen = 1;
06933     VALUE res;
06934     char *p;
06935     const char *f = " ";
06936     long n, size, llen, rlen, llen2 = 0, rlen2 = 0;
06937     volatile VALUE pad;
06938     int singlebyte = 1, cr;
06939 
06940     rb_scan_args(argc, argv, "11", &w, &pad);
06941     enc = STR_ENC_GET(str);
06942     width = NUM2LONG(w);
06943     if (argc == 2) {
06944         StringValue(pad);
06945         enc = rb_enc_check(str, pad);
06946         f = RSTRING_PTR(pad);
06947         flen = RSTRING_LEN(pad);
06948         fclen = str_strlen(pad, enc);
06949         singlebyte = single_byte_optimizable(pad);
06950         if (flen == 0 || fclen == 0) {
06951             rb_raise(rb_eArgError, "zero width padding");
06952         }
06953     }
06954     len = str_strlen(str, enc);
06955     if (width < 0 || len >= width) return rb_str_dup(str);
06956     n = width - len;
06957     llen = (jflag == 'l') ? 0 : ((jflag == 'r') ? n : n/2);
06958     rlen = n - llen;
06959     cr = ENC_CODERANGE(str);
06960     if (flen > 1) {
06961        llen2 = str_offset(f, f + flen, llen % fclen, enc, singlebyte);
06962        rlen2 = str_offset(f, f + flen, rlen % fclen, enc, singlebyte);
06963     }
06964     size = RSTRING_LEN(str);
06965     if ((len = llen / fclen + rlen / fclen) >= LONG_MAX / flen ||
06966        (len *= flen) >= LONG_MAX - llen2 - rlen2 ||
06967        (len += llen2 + rlen2) >= LONG_MAX - size) {
06968        rb_raise(rb_eArgError, "argument too big");
06969     }
06970     len += size;
06971     res = rb_str_new5(str, 0, len);
06972     p = RSTRING_PTR(res);
06973     if (flen <= 1) {
06974        memset(p, *f, llen);
06975        p += llen;
06976     }
06977     else {
06978        while (llen >= fclen) {
06979             memcpy(p,f,flen);
06980             p += flen;
06981             llen -= fclen;
06982         }
06983        if (llen > 0) {
06984            memcpy(p, f, llen2);
06985            p += llen2;
06986         }
06987     }
06988     memcpy(p, RSTRING_PTR(str), size);
06989     p += size;
06990     if (flen <= 1) {
06991        memset(p, *f, rlen);
06992        p += rlen;
06993     }
06994     else {
06995        while (rlen >= fclen) {
06996             memcpy(p,f,flen);
06997             p += flen;
06998             rlen -= fclen;
06999         }
07000        if (rlen > 0) {
07001            memcpy(p, f, rlen2);
07002            p += rlen2;
07003         }
07004     }
07005     *p = '\0';
07006     STR_SET_LEN(res, p-RSTRING_PTR(res));
07007     OBJ_INFECT(res, str);
07008     if (!NIL_P(pad)) OBJ_INFECT(res, pad);
07009     rb_enc_associate(res, enc);
07010     if (argc == 2)
07011         cr = ENC_CODERANGE_AND(cr, ENC_CODERANGE(pad));
07012     if (cr != ENC_CODERANGE_BROKEN)
07013         ENC_CODERANGE_SET(res, cr);
07014     return res;
07015 }
07016 
07017 
07018 /*
07019  *  call-seq:
07020  *     str.ljust(integer, padstr=' ')   -> new_str
07021  *
07022  *  If <i>integer</i> is greater than the length of <i>str</i>, returns a new
07023  *  <code>String</code> of length <i>integer</i> with <i>str</i> left justified
07024  *  and padded with <i>padstr</i>; otherwise, returns <i>str</i>.
07025  *
07026  *     "hello".ljust(4)            #=> "hello"
07027  *     "hello".ljust(20)           #=> "hello               "
07028  *     "hello".ljust(20, '1234')   #=> "hello123412341234123"
07029  */
07030 
07031 static VALUE
07032 rb_str_ljust(int argc, VALUE *argv, VALUE str)
07033 {
07034     return rb_str_justify(argc, argv, str, 'l');
07035 }
07036 
07037 
07038 /*
07039  *  call-seq:
07040  *     str.rjust(integer, padstr=' ')   -> new_str
07041  *
07042  *  If <i>integer</i> is greater than the length of <i>str</i>, returns a new
07043  *  <code>String</code> of length <i>integer</i> with <i>str</i> right justified
07044  *  and padded with <i>padstr</i>; otherwise, returns <i>str</i>.
07045  *
07046  *     "hello".rjust(4)            #=> "hello"
07047  *     "hello".rjust(20)           #=> "               hello"
07048  *     "hello".rjust(20, '1234')   #=> "123412341234123hello"
07049  */
07050 
07051 static VALUE
07052 rb_str_rjust(int argc, VALUE *argv, VALUE str)
07053 {
07054     return rb_str_justify(argc, argv, str, 'r');
07055 }
07056 
07057 
07058 /*
07059  *  call-seq:
07060  *     str.center(integer, padstr)   -> new_str
07061  *
07062  *  If <i>integer</i> is greater than the length of <i>str</i>, returns a new
07063  *  <code>String</code> of length <i>integer</i> with <i>str</i> centered and
07064  *  padded with <i>padstr</i>; otherwise, returns <i>str</i>.
07065  *
07066  *     "hello".center(4)         #=> "hello"
07067  *     "hello".center(20)        #=> "       hello        "
07068  *     "hello".center(20, '123') #=> "1231231hello12312312"
07069  */
07070 
07071 static VALUE
07072 rb_str_center(int argc, VALUE *argv, VALUE str)
07073 {
07074     return rb_str_justify(argc, argv, str, 'c');
07075 }
07076 
07077 /*
07078  *  call-seq:
07079  *     str.partition(sep)              -> [head, sep, tail]
07080  *     str.partition(regexp)           -> [head, match, tail]
07081  *
07082  *  Searches <i>sep</i> or pattern (<i>regexp</i>) in the string
07083  *  and returns the part before it, the match, and the part
07084  *  after it.
07085  *  If it is not found, returns two empty strings and <i>str</i>.
07086  *
07087  *     "hello".partition("l")         #=> ["he", "l", "lo"]
07088  *     "hello".partition("x")         #=> ["hello", "", ""]
07089  *     "hello".partition(/.l/)        #=> ["h", "el", "lo"]
07090  */
07091 
07092 static VALUE
07093 rb_str_partition(VALUE str, VALUE sep)
07094 {
07095     long pos;
07096     int regex = FALSE;
07097 
07098     if (TYPE(sep) == T_REGEXP) {
07099         pos = rb_reg_search(sep, str, 0, 0);
07100         regex = TRUE;
07101     }
07102     else {
07103         VALUE tmp;
07104 
07105         tmp = rb_check_string_type(sep);
07106         if (NIL_P(tmp)) {
07107             rb_raise(rb_eTypeError, "type mismatch: %s given",
07108                      rb_obj_classname(sep));
07109         }
07110         sep = tmp;
07111         pos = rb_str_index(str, sep, 0);
07112     }
07113     if (pos < 0) {
07114       failed:
07115         return rb_ary_new3(3, str, str_new_empty(str), str_new_empty(str));
07116     }
07117     if (regex) {
07118         sep = rb_str_subpat(str, sep, INT2FIX(0));
07119         if (pos == 0 && RSTRING_LEN(sep) == 0) goto failed;
07120     }
07121     return rb_ary_new3(3, rb_str_subseq(str, 0, pos),
07122                           sep,
07123                           rb_str_subseq(str, pos+RSTRING_LEN(sep),
07124                                              RSTRING_LEN(str)-pos-RSTRING_LEN(sep)));
07125 }
07126 
07127 /*
07128  *  call-seq:
07129  *     str.rpartition(sep)             -> [head, sep, tail]
07130  *     str.rpartition(regexp)          -> [head, match, tail]
07131  *
07132  *  Searches <i>sep</i> or pattern (<i>regexp</i>) in the string from the end
07133  *  of the string, and returns the part before it, the match, and the part
07134  *  after it.
07135  *  If it is not found, returns two empty strings and <i>str</i>.
07136  *
07137  *     "hello".rpartition("l")         #=> ["hel", "l", "o"]
07138  *     "hello".rpartition("x")         #=> ["", "", "hello"]
07139  *     "hello".rpartition(/.l/)        #=> ["he", "ll", "o"]
07140  */
07141 
07142 static VALUE
07143 rb_str_rpartition(VALUE str, VALUE sep)
07144 {
07145     long pos = RSTRING_LEN(str);
07146     int regex = FALSE;
07147 
07148     if (TYPE(sep) == T_REGEXP) {
07149         pos = rb_reg_search(sep, str, pos, 1);
07150         regex = TRUE;
07151     }
07152     else {
07153         VALUE tmp;
07154 
07155         tmp = rb_check_string_type(sep);
07156         if (NIL_P(tmp)) {
07157             rb_raise(rb_eTypeError, "type mismatch: %s given",
07158                      rb_obj_classname(sep));
07159         }
07160         sep = tmp;
07161         pos = rb_str_sublen(str, pos);
07162         pos = rb_str_rindex(str, sep, pos);
07163     }
07164     if (pos < 0) {
07165         return rb_ary_new3(3, str_new_empty(str), str_new_empty(str), str);
07166     }
07167     if (regex) {
07168         sep = rb_reg_nth_match(0, rb_backref_get());
07169     }
07170     return rb_ary_new3(3, rb_str_substr(str, 0, pos),
07171                           sep,
07172                           rb_str_substr(str,pos+str_strlen(sep,STR_ENC_GET(sep)),RSTRING_LEN(str)));
07173 }
07174 
07175 /*
07176  *  call-seq:
07177  *     str.start_with?([prefix]+)   -> true or false
07178  *
07179  *  Returns true if <i>str</i> starts with one of the prefixes given.
07180  *
07181  *    p "hello".start_with?("hell")               #=> true
07182  *
07183  *    # returns true if one of the prefixes matches.
07184  *    p "hello".start_with?("heaven", "hell")     #=> true
07185  *    p "hello".start_with?("heaven", "paradise") #=> false
07186  *
07187  *
07188  *
07189  */
07190 
07191 static VALUE
07192 rb_str_start_with(int argc, VALUE *argv, VALUE str)
07193 {
07194     int i;
07195 
07196     for (i=0; i<argc; i++) {
07197         VALUE tmp = rb_check_string_type(argv[i]);
07198         if (NIL_P(tmp)) continue;
07199         rb_enc_check(str, tmp);
07200         if (RSTRING_LEN(str) < RSTRING_LEN(tmp)) continue;
07201         if (memcmp(RSTRING_PTR(str), RSTRING_PTR(tmp), RSTRING_LEN(tmp)) == 0)
07202             return Qtrue;
07203     }
07204     return Qfalse;
07205 }
07206 
07207 /*
07208  *  call-seq:
07209  *     str.end_with?([suffix]+)   -> true or false
07210  *
07211  *  Returns true if <i>str</i> ends with one of the suffixes given.
07212  */
07213 
07214 static VALUE
07215 rb_str_end_with(int argc, VALUE *argv, VALUE str)
07216 {
07217     int i;
07218     char *p, *s, *e;
07219     rb_encoding *enc;
07220 
07221     for (i=0; i<argc; i++) {
07222         VALUE tmp = rb_check_string_type(argv[i]);
07223         if (NIL_P(tmp)) continue;
07224         enc = rb_enc_check(str, tmp);
07225         if (RSTRING_LEN(str) < RSTRING_LEN(tmp)) continue;
07226         p = RSTRING_PTR(str);
07227         e = p + RSTRING_LEN(str);
07228         s = e - RSTRING_LEN(tmp);
07229         if (rb_enc_left_char_head(p, s, e, enc) != s)
07230             continue;
07231         if (memcmp(s, RSTRING_PTR(tmp), RSTRING_LEN(tmp)) == 0)
07232             return Qtrue;
07233     }
07234     return Qfalse;
07235 }
07236 
07237 void
07238 rb_str_setter(VALUE val, ID id, VALUE *var)
07239 {
07240     if (!NIL_P(val) && TYPE(val) != T_STRING) {
07241         rb_raise(rb_eTypeError, "value of %s must be String", rb_id2name(id));
07242     }
07243     *var = val;
07244 }
07245 
07246 
07247 /*
07248  *  call-seq:
07249  *     str.force_encoding(encoding)   -> str
07250  *
07251  *  Changes the encoding to +encoding+ and returns self.
07252  */
07253 
07254 static VALUE
07255 rb_str_force_encoding(VALUE str, VALUE enc)
07256 {
07257     str_modifiable(str);
07258     rb_enc_associate(str, rb_to_encoding(enc));
07259     ENC_CODERANGE_CLEAR(str);
07260     return str;
07261 }
07262 
07263 /*
07264  *  call-seq:
07265  *     str.valid_encoding?  -> true or false
07266  *
07267  *  Returns true for a string which encoded correctly.
07268  *
07269  *    "\xc2\xa1".force_encoding("UTF-8").valid_encoding?  #=> true
07270  *    "\xc2".force_encoding("UTF-8").valid_encoding?      #=> false
07271  *    "\x80".force_encoding("UTF-8").valid_encoding?      #=> false
07272  */
07273 
07274 static VALUE
07275 rb_str_valid_encoding_p(VALUE str)
07276 {
07277     int cr = rb_enc_str_coderange(str);
07278 
07279     return cr == ENC_CODERANGE_BROKEN ? Qfalse : Qtrue;
07280 }
07281 
07282 /*
07283  *  call-seq:
07284  *     str.ascii_only?  -> true or false
07285  *
07286  *  Returns true for a string which has only ASCII characters.
07287  *
07288  *    "abc".force_encoding("UTF-8").ascii_only?          #=> true
07289  *    "abc\u{6666}".force_encoding("UTF-8").ascii_only?  #=> false
07290  */
07291 
07292 static VALUE
07293 rb_str_is_ascii_only_p(VALUE str)
07294 {
07295     int cr = rb_enc_str_coderange(str);
07296 
07297     return cr == ENC_CODERANGE_7BIT ? Qtrue : Qfalse;
07298 }
07299 
07314 VALUE
07315 rb_str_ellipsize(VALUE str, long len)
07316 {
07317     static const char ellipsis[] = "...";
07318     const long ellipsislen = sizeof(ellipsis) - 1;
07319     rb_encoding *const enc = rb_enc_get(str);
07320     const long blen = RSTRING_LEN(str);
07321     const char *const p = RSTRING_PTR(str), *e = p + blen;
07322     VALUE estr, ret = 0;
07323 
07324     if (len < 0) rb_raise(rb_eIndexError, "negative length %ld", len);
07325     if (len * rb_enc_mbminlen(enc) >= blen ||
07326         (e = rb_enc_nth(p, e, len, enc)) - p == blen) {
07327         ret = str;
07328     }
07329     else if (len <= ellipsislen ||
07330              !(e = rb_enc_step_back(p, e, e, len = ellipsislen, enc))) {
07331         if (rb_enc_asciicompat(enc)) {
07332             ret = rb_str_new_with_class(str, ellipsis, len);
07333             rb_enc_associate(ret, enc);
07334         }
07335         else {
07336             estr = rb_usascii_str_new(ellipsis, len);
07337             ret = rb_str_encode(estr, rb_enc_from_encoding(enc), 0, Qnil);
07338         }
07339     }
07340     else if (ret = rb_str_subseq(str, 0, e - p), rb_enc_asciicompat(enc)) {
07341         rb_str_cat(ret, ellipsis, ellipsislen);
07342     }
07343     else {
07344         estr = rb_str_encode(rb_usascii_str_new(ellipsis, ellipsislen),
07345                              rb_enc_from_encoding(enc), 0, Qnil);
07346         rb_str_append(ret, estr);
07347     }
07348     return ret;
07349 }
07350 
07351 /**********************************************************************
07352  * Document-class: Symbol
07353  *
07354  *  <code>Symbol</code> objects represent names and some strings
07355  *  inside the Ruby
07356  *  interpreter. They are generated using the <code>:name</code> and
07357  *  <code>:"string"</code> literals
07358  *  syntax, and by the various <code>to_sym</code> methods. The same
07359  *  <code>Symbol</code> object will be created for a given name or string
07360  *  for the duration of a program's execution, regardless of the context
07361  *  or meaning of that name. Thus if <code>Fred</code> is a constant in
07362  *  one context, a method in another, and a class in a third, the
07363  *  <code>Symbol</code> <code>:Fred</code> will be the same object in
07364  *  all three contexts.
07365  *
07366  *     module One
07367  *       class Fred
07368  *       end
07369  *       $f1 = :Fred
07370  *     end
07371  *     module Two
07372  *       Fred = 1
07373  *       $f2 = :Fred
07374  *     end
07375  *     def Fred()
07376  *     end
07377  *     $f3 = :Fred
07378  *     $f1.object_id   #=> 2514190
07379  *     $f2.object_id   #=> 2514190
07380  *     $f3.object_id   #=> 2514190
07381  *
07382  */
07383 
07384 
07385 /*
07386  *  call-seq:
07387  *     sym == obj   -> true or false
07388  *
07389  *  Equality---If <i>sym</i> and <i>obj</i> are exactly the same
07390  *  symbol, returns <code>true</code>.
07391  */
07392 
07393 static VALUE
07394 sym_equal(VALUE sym1, VALUE sym2)
07395 {
07396     if (sym1 == sym2) return Qtrue;
07397     return Qfalse;
07398 }
07399 
07400 
07401 static int
07402 sym_printable(const char *s, const char *send, rb_encoding *enc)
07403 {
07404     while (s < send) {
07405         int n;
07406         int c = rb_enc_codepoint_len(s, send, &n, enc);
07407 
07408         if (!rb_enc_isprint(c, enc)) return FALSE;
07409         s += n;
07410     }
07411     return TRUE;
07412 }
07413 
07414 /*
07415  *  call-seq:
07416  *     sym.inspect    -> string
07417  *
07418  *  Returns the representation of <i>sym</i> as a symbol literal.
07419  *
07420  *     :fred.inspect   #=> ":fred"
07421  */
07422 
07423 static VALUE
07424 sym_inspect(VALUE sym)
07425 {
07426     VALUE str;
07427     ID id = SYM2ID(sym);
07428     rb_encoding *enc;
07429     const char *ptr;
07430     long len;
07431     char *dest;
07432     rb_encoding *resenc = rb_default_internal_encoding();
07433 
07434     if (resenc == NULL) resenc = rb_default_external_encoding();
07435     sym = rb_id2str(id);
07436     enc = STR_ENC_GET(sym);
07437     ptr = RSTRING_PTR(sym);
07438     len = RSTRING_LEN(sym);
07439     if ((resenc != enc && !rb_str_is_ascii_only_p(sym)) || len != (long)strlen(ptr) ||
07440         !rb_enc_symname_p(ptr, enc) || !sym_printable(ptr, ptr + len, enc)) {
07441         str = rb_str_inspect(sym);
07442         len = RSTRING_LEN(str);
07443         rb_str_resize(str, len + 1);
07444         dest = RSTRING_PTR(str);
07445         memmove(dest + 1, dest, len);
07446         dest[0] = ':';
07447     }
07448     else {
07449         char *dest;
07450         str = rb_enc_str_new(0, len + 1, enc);
07451         dest = RSTRING_PTR(str);
07452         dest[0] = ':';
07453         memcpy(dest + 1, ptr, len);
07454     }
07455     return str;
07456 }
07457 
07458 
07459 /*
07460  *  call-seq:
07461  *     sym.id2name   -> string
07462  *     sym.to_s      -> string
07463  *
07464  *  Returns the name or string corresponding to <i>sym</i>.
07465  *
07466  *     :fred.id2name   #=> "fred"
07467  */
07468 
07469 
07470 VALUE
07471 rb_sym_to_s(VALUE sym)
07472 {
07473     ID id = SYM2ID(sym);
07474 
07475     return str_new3(rb_cString, rb_id2str(id));
07476 }
07477 
07478 
07479 /*
07480  * call-seq:
07481  *   sym.to_sym   -> sym
07482  *   sym.intern   -> sym
07483  *
07484  * In general, <code>to_sym</code> returns the <code>Symbol</code> corresponding
07485  * to an object. As <i>sym</i> is already a symbol, <code>self</code> is returned
07486  * in this case.
07487  */
07488 
07489 static VALUE
07490 sym_to_sym(VALUE sym)
07491 {
07492     return sym;
07493 }
07494 
07495 static VALUE
07496 sym_call(VALUE args, VALUE sym, int argc, VALUE *argv)
07497 {
07498     VALUE obj;
07499 
07500     if (argc < 1) {
07501         rb_raise(rb_eArgError, "no receiver given");
07502     }
07503     obj = argv[0];
07504     return rb_funcall_passing_block(obj, (ID)sym, argc - 1, argv + 1);
07505 }
07506 
07507 /*
07508  * call-seq:
07509  *   sym.to_proc
07510  *
07511  * Returns a _Proc_ object which respond to the given method by _sym_.
07512  *
07513  *   (1..3).collect(&:to_s)  #=> ["1", "2", "3"]
07514  */
07515 
07516 static VALUE
07517 sym_to_proc(VALUE sym)
07518 {
07519     static VALUE sym_proc_cache = Qfalse;
07520     enum {SYM_PROC_CACHE_SIZE = 67};
07521     VALUE proc;
07522     long id, index;
07523     VALUE *aryp;
07524 
07525     if (!sym_proc_cache) {
07526         sym_proc_cache = rb_ary_tmp_new(SYM_PROC_CACHE_SIZE * 2);
07527         rb_gc_register_mark_object(sym_proc_cache);
07528         rb_ary_store(sym_proc_cache, SYM_PROC_CACHE_SIZE*2 - 1, Qnil);
07529     }
07530 
07531     id = SYM2ID(sym);
07532     index = (id % SYM_PROC_CACHE_SIZE) << 1;
07533 
07534     aryp = RARRAY_PTR(sym_proc_cache);
07535     if (aryp[index] == sym) {
07536         return aryp[index + 1];
07537     }
07538     else {
07539         proc = rb_proc_new(sym_call, (VALUE)id);
07540         aryp[index] = sym;
07541         aryp[index + 1] = proc;
07542         return proc;
07543     }
07544 }
07545 
07546 /*
07547  * call-seq:
07548  *
07549  *   sym.succ
07550  *
07551  * Same as <code>sym.to_s.succ.intern</code>.
07552  */
07553 
07554 static VALUE
07555 sym_succ(VALUE sym)
07556 {
07557     return rb_str_intern(rb_str_succ(rb_sym_to_s(sym)));
07558 }
07559 
07560 /*
07561  * call-seq:
07562  *
07563  *   str <=> other       -> -1, 0, +1 or nil
07564  *
07565  * Compares _sym_ with _other_ in string form.
07566  */
07567 
07568 static VALUE
07569 sym_cmp(VALUE sym, VALUE other)
07570 {
07571     if (!SYMBOL_P(other)) {
07572         return Qnil;
07573     }
07574     return rb_str_cmp_m(rb_sym_to_s(sym), rb_sym_to_s(other));
07575 }
07576 
07577 /*
07578  * call-seq:
07579  *
07580  *   sym.casecmp(other)  -> -1, 0, +1 or nil
07581  *
07582  * Case-insensitive version of <code>Symbol#<=></code>.
07583  */
07584 
07585 static VALUE
07586 sym_casecmp(VALUE sym, VALUE other)
07587 {
07588     if (!SYMBOL_P(other)) {
07589         return Qnil;
07590     }
07591     return rb_str_casecmp(rb_sym_to_s(sym), rb_sym_to_s(other));
07592 }
07593 
07594 /*
07595  * call-seq:
07596  *   sym =~ obj   -> fixnum or nil
07597  *
07598  * Returns <code>sym.to_s =~ obj</code>.
07599  */
07600 
07601 static VALUE
07602 sym_match(VALUE sym, VALUE other)
07603 {
07604     return rb_str_match(rb_sym_to_s(sym), other);
07605 }
07606 
07607 /*
07608  * call-seq:
07609  *   sym[idx]      -> char
07610  *   sym[b, n]     -> char
07611  *
07612  * Returns <code>sym.to_s[]</code>.
07613  */
07614 
07615 static VALUE
07616 sym_aref(int argc, VALUE *argv, VALUE sym)
07617 {
07618     return rb_str_aref_m(argc, argv, rb_sym_to_s(sym));
07619 }
07620 
07621 /*
07622  * call-seq:
07623  *   sym.length    -> integer
07624  *
07625  * Same as <code>sym.to_s.length</code>.
07626  */
07627 
07628 static VALUE
07629 sym_length(VALUE sym)
07630 {
07631     return rb_str_length(rb_id2str(SYM2ID(sym)));
07632 }
07633 
07634 /*
07635  * call-seq:
07636  *   sym.empty?   -> true or false
07637  *
07638  * Returns that _sym_ is :"" or not.
07639  */
07640 
07641 static VALUE
07642 sym_empty(VALUE sym)
07643 {
07644     return rb_str_empty(rb_id2str(SYM2ID(sym)));
07645 }
07646 
07647 /*
07648  * call-seq:
07649  *   sym.upcase    -> symbol
07650  *
07651  * Same as <code>sym.to_s.upcase.intern</code>.
07652  */
07653 
07654 static VALUE
07655 sym_upcase(VALUE sym)
07656 {
07657     return rb_str_intern(rb_str_upcase(rb_id2str(SYM2ID(sym))));
07658 }
07659 
07660 /*
07661  * call-seq:
07662  *   sym.downcase  -> symbol
07663  *
07664  * Same as <code>sym.to_s.downcase.intern</code>.
07665  */
07666 
07667 static VALUE
07668 sym_downcase(VALUE sym)
07669 {
07670     return rb_str_intern(rb_str_downcase(rb_id2str(SYM2ID(sym))));
07671 }
07672 
07673 /*
07674  * call-seq:
07675  *   sym.capitalize  -> symbol
07676  *
07677  * Same as <code>sym.to_s.capitalize.intern</code>.
07678  */
07679 
07680 static VALUE
07681 sym_capitalize(VALUE sym)
07682 {
07683     return rb_str_intern(rb_str_capitalize(rb_id2str(SYM2ID(sym))));
07684 }
07685 
07686 /*
07687  * call-seq:
07688  *   sym.swapcase  -> symbol
07689  *
07690  * Same as <code>sym.to_s.swapcase.intern</code>.
07691  */
07692 
07693 static VALUE
07694 sym_swapcase(VALUE sym)
07695 {
07696     return rb_str_intern(rb_str_swapcase(rb_id2str(SYM2ID(sym))));
07697 }
07698 
07699 /*
07700  * call-seq:
07701  *   sym.encoding   -> encoding
07702  *
07703  * Returns the Encoding object that represents the encoding of _sym_.
07704  */
07705 
07706 static VALUE
07707 sym_encoding(VALUE sym)
07708 {
07709     return rb_obj_encoding(rb_id2str(SYM2ID(sym)));
07710 }
07711 
07712 ID
07713 rb_to_id(VALUE name)
07714 {
07715     VALUE tmp;
07716 
07717     switch (TYPE(name)) {
07718       default:
07719         tmp = rb_check_string_type(name);
07720         if (NIL_P(tmp)) {
07721             tmp = rb_inspect(name);
07722             rb_raise(rb_eTypeError, "%s is not a symbol",
07723                      RSTRING_PTR(tmp));
07724         }
07725         name = tmp;
07726         /* fall through */
07727       case T_STRING:
07728         name = rb_str_intern(name);
07729         /* fall through */
07730       case T_SYMBOL:
07731         return SYM2ID(name);
07732     }
07733     return Qnil; /* not reached */
07734 }
07735 
07736 /*
07737  *  A <code>String</code> object holds and manipulates an arbitrary sequence of
07738  *  bytes, typically representing characters. String objects may be created
07739  *  using <code>String::new</code> or as literals.
07740  *
07741  *  Because of aliasing issues, users of strings should be aware of the methods
07742  *  that modify the contents of a <code>String</code> object.  Typically,
07743  *  methods with names ending in ``!'' modify their receiver, while those
07744  *  without a ``!'' return a new <code>String</code>.  However, there are
07745  *  exceptions, such as <code>String#[]=</code>.
07746  *
07747  */
07748 
07749 void
07750 Init_String(void)
07751 {
07752 #undef rb_intern
07753 #define rb_intern(str) rb_intern_const(str)
07754 
07755     rb_cString  = rb_define_class("String", rb_cObject);
07756     rb_include_module(rb_cString, rb_mComparable);
07757     rb_define_alloc_func(rb_cString, str_alloc);
07758     rb_define_singleton_method(rb_cString, "try_convert", rb_str_s_try_convert, 1);
07759     rb_define_method(rb_cString, "initialize", rb_str_init, -1);
07760     rb_define_method(rb_cString, "initialize_copy", rb_str_replace, 1);
07761     rb_define_method(rb_cString, "<=>", rb_str_cmp_m, 1);
07762     rb_define_method(rb_cString, "==", rb_str_equal, 1);
07763     rb_define_method(rb_cString, "===", rb_str_equal, 1);
07764     rb_define_method(rb_cString, "eql?", rb_str_eql, 1);
07765     rb_define_method(rb_cString, "hash", rb_str_hash_m, 0);
07766     rb_define_method(rb_cString, "casecmp", rb_str_casecmp, 1);
07767     rb_define_method(rb_cString, "+", rb_str_plus, 1);
07768     rb_define_method(rb_cString, "*", rb_str_times, 1);
07769     rb_define_method(rb_cString, "%", rb_str_format_m, 1);
07770     rb_define_method(rb_cString, "[]", rb_str_aref_m, -1);
07771     rb_define_method(rb_cString, "[]=", rb_str_aset_m, -1);
07772     rb_define_method(rb_cString, "insert", rb_str_insert, 2);
07773     rb_define_method(rb_cString, "length", rb_str_length, 0);
07774     rb_define_method(rb_cString, "size", rb_str_length, 0);
07775     rb_define_method(rb_cString, "bytesize", rb_str_bytesize, 0);
07776     rb_define_method(rb_cString, "empty?", rb_str_empty, 0);
07777     rb_define_method(rb_cString, "=~", rb_str_match, 1);
07778     rb_define_method(rb_cString, "match", rb_str_match_m, -1);
07779     rb_define_method(rb_cString, "succ", rb_str_succ, 0);
07780     rb_define_method(rb_cString, "succ!", rb_str_succ_bang, 0);
07781     rb_define_method(rb_cString, "next", rb_str_succ, 0);
07782     rb_define_method(rb_cString, "next!", rb_str_succ_bang, 0);
07783     rb_define_method(rb_cString, "upto", rb_str_upto, -1);
07784     rb_define_method(rb_cString, "index", rb_str_index_m, -1);
07785     rb_define_method(rb_cString, "rindex", rb_str_rindex_m, -1);
07786     rb_define_method(rb_cString, "replace", rb_str_replace, 1);
07787     rb_define_method(rb_cString, "clear", rb_str_clear, 0);
07788     rb_define_method(rb_cString, "chr", rb_str_chr, 0);
07789     rb_define_method(rb_cString, "getbyte", rb_str_getbyte, 1);
07790     rb_define_method(rb_cString, "setbyte", rb_str_setbyte, 2);
07791     rb_define_method(rb_cString, "byteslice", rb_str_byteslice, -1);
07792 
07793     rb_define_method(rb_cString, "to_i", rb_str_to_i, -1);
07794     rb_define_method(rb_cString, "to_f", rb_str_to_f, 0);
07795     rb_define_method(rb_cString, "to_s", rb_str_to_s, 0);
07796     rb_define_method(rb_cString, "to_str", rb_str_to_s, 0);
07797     rb_define_method(rb_cString, "inspect", rb_str_inspect, 0);
07798     rb_define_method(rb_cString, "dump", rb_str_dump, 0);
07799 
07800     rb_define_method(rb_cString, "upcase", rb_str_upcase, 0);
07801     rb_define_method(rb_cString, "downcase", rb_str_downcase, 0);
07802     rb_define_method(rb_cString, "capitalize", rb_str_capitalize, 0);
07803     rb_define_method(rb_cString, "swapcase", rb_str_swapcase, 0);
07804 
07805     rb_define_method(rb_cString, "upcase!", rb_str_upcase_bang, 0);
07806     rb_define_method(rb_cString, "downcase!", rb_str_downcase_bang, 0);
07807     rb_define_method(rb_cString, "capitalize!", rb_str_capitalize_bang, 0);
07808     rb_define_method(rb_cString, "swapcase!", rb_str_swapcase_bang, 0);
07809 
07810     rb_define_method(rb_cString, "hex", rb_str_hex, 0);
07811     rb_define_method(rb_cString, "oct", rb_str_oct, 0);
07812     rb_define_method(rb_cString, "split", rb_str_split_m, -1);
07813     rb_define_method(rb_cString, "lines", rb_str_each_line, -1);
07814     rb_define_method(rb_cString, "bytes", rb_str_each_byte, 0);
07815     rb_define_method(rb_cString, "chars", rb_str_each_char, 0);
07816     rb_define_method(rb_cString, "codepoints", rb_str_each_codepoint, 0);
07817     rb_define_method(rb_cString, "reverse", rb_str_reverse, 0);
07818     rb_define_method(rb_cString, "reverse!", rb_str_reverse_bang, 0);
07819     rb_define_method(rb_cString, "concat", rb_str_concat, 1);
07820     rb_define_method(rb_cString, "<<", rb_str_concat, 1);
07821     rb_define_method(rb_cString, "prepend", rb_str_prepend, 1);
07822     rb_define_method(rb_cString, "crypt", rb_str_crypt, 1);
07823     rb_define_method(rb_cString, "intern", rb_str_intern, 0);
07824     rb_define_method(rb_cString, "to_sym", rb_str_intern, 0);
07825     rb_define_method(rb_cString, "ord", rb_str_ord, 0);
07826 
07827     rb_define_method(rb_cString, "include?", rb_str_include, 1);
07828     rb_define_method(rb_cString, "start_with?", rb_str_start_with, -1);
07829     rb_define_method(rb_cString, "end_with?", rb_str_end_with, -1);
07830 
07831     rb_define_method(rb_cString, "scan", rb_str_scan, 1);
07832 
07833     rb_define_method(rb_cString, "ljust", rb_str_ljust, -1);
07834     rb_define_method(rb_cString, "rjust", rb_str_rjust, -1);
07835     rb_define_method(rb_cString, "center", rb_str_center, -1);
07836 
07837     rb_define_method(rb_cString, "sub", rb_str_sub, -1);
07838     rb_define_method(rb_cString, "gsub", rb_str_gsub, -1);
07839     rb_define_method(rb_cString, "chop", rb_str_chop, 0);
07840     rb_define_method(rb_cString, "chomp", rb_str_chomp, -1);
07841     rb_define_method(rb_cString, "strip", rb_str_strip, 0);
07842     rb_define_method(rb_cString, "lstrip", rb_str_lstrip, 0);
07843     rb_define_method(rb_cString, "rstrip", rb_str_rstrip, 0);
07844 
07845     rb_define_method(rb_cString, "sub!", rb_str_sub_bang, -1);
07846     rb_define_method(rb_cString, "gsub!", rb_str_gsub_bang, -1);
07847     rb_define_method(rb_cString, "chop!", rb_str_chop_bang, 0);
07848     rb_define_method(rb_cString, "chomp!", rb_str_chomp_bang, -1);
07849     rb_define_method(rb_cString, "strip!", rb_str_strip_bang, 0);
07850     rb_define_method(rb_cString, "lstrip!", rb_str_lstrip_bang, 0);
07851     rb_define_method(rb_cString, "rstrip!", rb_str_rstrip_bang, 0);
07852 
07853     rb_define_method(rb_cString, "tr", rb_str_tr, 2);
07854     rb_define_method(rb_cString, "tr_s", rb_str_tr_s, 2);
07855     rb_define_method(rb_cString, "delete", rb_str_delete, -1);
07856     rb_define_method(rb_cString, "squeeze", rb_str_squeeze, -1);
07857     rb_define_method(rb_cString, "count", rb_str_count, -1);
07858 
07859     rb_define_method(rb_cString, "tr!", rb_str_tr_bang, 2);
07860     rb_define_method(rb_cString, "tr_s!", rb_str_tr_s_bang, 2);
07861     rb_define_method(rb_cString, "delete!", rb_str_delete_bang, -1);
07862     rb_define_method(rb_cString, "squeeze!", rb_str_squeeze_bang, -1);
07863 
07864     rb_define_method(rb_cString, "each_line", rb_str_each_line, -1);
07865     rb_define_method(rb_cString, "each_byte", rb_str_each_byte, 0);
07866     rb_define_method(rb_cString, "each_char", rb_str_each_char, 0);
07867     rb_define_method(rb_cString, "each_codepoint", rb_str_each_codepoint, 0);
07868 
07869     rb_define_method(rb_cString, "sum", rb_str_sum, -1);
07870 
07871     rb_define_method(rb_cString, "slice", rb_str_aref_m, -1);
07872     rb_define_method(rb_cString, "slice!", rb_str_slice_bang, -1);
07873 
07874     rb_define_method(rb_cString, "partition", rb_str_partition, 1);
07875     rb_define_method(rb_cString, "rpartition", rb_str_rpartition, 1);
07876 
07877     rb_define_method(rb_cString, "encoding", rb_obj_encoding, 0); /* in encoding.c */
07878     rb_define_method(rb_cString, "force_encoding", rb_str_force_encoding, 1);
07879     rb_define_method(rb_cString, "valid_encoding?", rb_str_valid_encoding_p, 0);
07880     rb_define_method(rb_cString, "ascii_only?", rb_str_is_ascii_only_p, 0);
07881 
07882     id_to_s = rb_intern("to_s");
07883 
07884     rb_fs = Qnil;
07885     rb_define_variable("$;", &rb_fs);
07886     rb_define_variable("$-F", &rb_fs);
07887 
07888     rb_cSymbol = rb_define_class("Symbol", rb_cObject);
07889     rb_include_module(rb_cSymbol, rb_mComparable);
07890     rb_undef_alloc_func(rb_cSymbol);
07891     rb_undef_method(CLASS_OF(rb_cSymbol), "new");
07892     rb_define_singleton_method(rb_cSymbol, "all_symbols", rb_sym_all_symbols, 0); /* in parse.y */
07893 
07894     rb_define_method(rb_cSymbol, "==", sym_equal, 1);
07895     rb_define_method(rb_cSymbol, "===", sym_equal, 1);
07896     rb_define_method(rb_cSymbol, "inspect", sym_inspect, 0);
07897     rb_define_method(rb_cSymbol, "to_s", rb_sym_to_s, 0);
07898     rb_define_method(rb_cSymbol, "id2name", rb_sym_to_s, 0);
07899     rb_define_method(rb_cSymbol, "intern", sym_to_sym, 0);
07900     rb_define_method(rb_cSymbol, "to_sym", sym_to_sym, 0);
07901     rb_define_method(rb_cSymbol, "to_proc", sym_to_proc, 0);
07902     rb_define_method(rb_cSymbol, "succ", sym_succ, 0);
07903     rb_define_method(rb_cSymbol, "next", sym_succ, 0);
07904 
07905     rb_define_method(rb_cSymbol, "<=>", sym_cmp, 1);
07906     rb_define_method(rb_cSymbol, "casecmp", sym_casecmp, 1);
07907     rb_define_method(rb_cSymbol, "=~", sym_match, 1);
07908 
07909     rb_define_method(rb_cSymbol, "[]", sym_aref, -1);
07910     rb_define_method(rb_cSymbol, "slice", sym_aref, -1);
07911     rb_define_method(rb_cSymbol, "length", sym_length, 0);
07912     rb_define_method(rb_cSymbol, "size", sym_length, 0);
07913     rb_define_method(rb_cSymbol, "empty?", sym_empty, 0);
07914     rb_define_method(rb_cSymbol, "match", sym_match, 1);
07915 
07916     rb_define_method(rb_cSymbol, "upcase", sym_upcase, 0);
07917     rb_define_method(rb_cSymbol, "downcase", sym_downcase, 0);
07918     rb_define_method(rb_cSymbol, "capitalize", sym_capitalize, 0);
07919     rb_define_method(rb_cSymbol, "swapcase", sym_swapcase, 0);
07920 
07921     rb_define_method(rb_cSymbol, "encoding", sym_encoding, 0);
07922 }
07923