|
Ruby
2.0.0p594(2014-10-27revision48167)
|
00001 /********************************************************************** 00002 00003 string.c - 00004 00005 $Author: usa $ 00006 created at: Mon Aug 9 17:12:58 JST 1993 00007 00008 Copyright (C) 1993-2007 Yukihiro Matsumoto 00009 Copyright (C) 2000 Network Applied Communication Laboratory, Inc. 00010 Copyright (C) 2000 Information-technology Promotion Agency, Japan 00011 00012 **********************************************************************/ 00013 00014 #include "ruby/ruby.h" 00015 #include "ruby/re.h" 00016 #include "ruby/encoding.h" 00017 #include "vm_core.h" 00018 #include "internal.h" 00019 #include "probes.h" 00020 #include <assert.h> 00021 00022 #define BEG(no) (regs->beg[(no)]) 00023 #define END(no) (regs->end[(no)]) 00024 00025 #include <math.h> 00026 #include <ctype.h> 00027 00028 #ifdef HAVE_UNISTD_H 00029 #include <unistd.h> 00030 #endif 00031 00032 #define numberof(array) (int)(sizeof(array) / sizeof((array)[0])) 00033 00034 #undef rb_str_new_cstr 00035 #undef rb_tainted_str_new_cstr 00036 #undef rb_usascii_str_new_cstr 00037 #undef rb_external_str_new_cstr 00038 #undef rb_locale_str_new_cstr 00039 #undef rb_str_new2 00040 #undef rb_str_new3 00041 #undef rb_str_new4 00042 #undef rb_str_new5 00043 #undef rb_tainted_str_new2 00044 #undef rb_usascii_str_new2 00045 #undef rb_str_dup_frozen 00046 #undef rb_str_buf_new_cstr 00047 #undef rb_str_buf_new2 00048 #undef rb_str_buf_cat2 00049 #undef rb_str_cat2 00050 00051 static VALUE rb_str_clear(VALUE str); 00052 00053 VALUE rb_cString; 00054 VALUE rb_cSymbol; 00055 00056 #define RUBY_MAX_CHAR_LEN 16 00057 #define STR_TMPLOCK FL_USER7 00058 #define STR_NOEMBED FL_USER1 00059 #define STR_SHARED FL_USER2 /* = ELTS_SHARED */ 00060 #define STR_ASSOC FL_USER3 00061 #define STR_SHARED_P(s) FL_ALL((s), STR_NOEMBED|ELTS_SHARED) 00062 #define STR_ASSOC_P(s) FL_ALL((s), STR_NOEMBED|STR_ASSOC) 00063 #define STR_NOCAPA (STR_NOEMBED|ELTS_SHARED|STR_ASSOC) 00064 #define STR_NOCAPA_P(s) (FL_TEST((s),STR_NOEMBED) && FL_ANY((s),ELTS_SHARED|STR_ASSOC)) 00065 #define STR_UNSET_NOCAPA(s) do {\ 00066 if (FL_TEST((s),STR_NOEMBED)) FL_UNSET((s),(ELTS_SHARED|STR_ASSOC));\ 00067 } while (0) 00068 00069 00070 #define STR_SET_NOEMBED(str) do {\ 00071 FL_SET((str), STR_NOEMBED);\ 00072 STR_SET_EMBED_LEN((str), 0);\ 00073 } while (0) 00074 #define STR_SET_EMBED(str) FL_UNSET((str), STR_NOEMBED) 00075 #define STR_EMBED_P(str) (!FL_TEST((str), STR_NOEMBED)) 00076 #define STR_SET_EMBED_LEN(str, n) do { \ 00077 long tmp_n = (n);\ 00078 RBASIC(str)->flags &= ~RSTRING_EMBED_LEN_MASK;\ 00079 RBASIC(str)->flags |= (tmp_n) << RSTRING_EMBED_LEN_SHIFT;\ 00080 } while (0) 00081 00082 #define STR_SET_LEN(str, n) do { \ 00083 if (STR_EMBED_P(str)) {\ 00084 STR_SET_EMBED_LEN((str), (n));\ 00085 }\ 00086 else {\ 00087 RSTRING(str)->as.heap.len = (n);\ 00088 }\ 00089 } while (0) 00090 00091 #define STR_DEC_LEN(str) do {\ 00092 if (STR_EMBED_P(str)) {\ 00093 long n = RSTRING_LEN(str);\ 00094 n--;\ 00095 STR_SET_EMBED_LEN((str), n);\ 00096 }\ 00097 else {\ 00098 RSTRING(str)->as.heap.len--;\ 00099 }\ 00100 } while (0) 00101 00102 #define RESIZE_CAPA(str,capacity) do {\ 00103 if (STR_EMBED_P(str)) {\ 00104 if ((capacity) > RSTRING_EMBED_LEN_MAX) {\ 00105 char *tmp = ALLOC_N(char, (capacity)+1);\ 00106 memcpy(tmp, RSTRING_PTR(str), RSTRING_LEN(str));\ 00107 RSTRING(str)->as.heap.ptr = tmp;\ 00108 RSTRING(str)->as.heap.len = RSTRING_LEN(str);\ 00109 STR_SET_NOEMBED(str);\ 00110 RSTRING(str)->as.heap.aux.capa = (capacity);\ 00111 }\ 00112 }\ 00113 else {\ 00114 REALLOC_N(RSTRING(str)->as.heap.ptr, char, (capacity)+1);\ 00115 if (!STR_NOCAPA_P(str))\ 00116 RSTRING(str)->as.heap.aux.capa = (capacity);\ 00117 }\ 00118 } while (0) 00119 00120 #define is_ascii_string(str) (rb_enc_str_coderange(str) == ENC_CODERANGE_7BIT) 00121 #define is_broken_string(str) (rb_enc_str_coderange(str) == ENC_CODERANGE_BROKEN) 00122 00123 #define STR_ENC_GET(str) rb_enc_from_index(ENCODING_GET(str)) 00124 00125 static inline int 00126 single_byte_optimizable(VALUE str) 00127 { 00128 rb_encoding *enc; 00129 00130 /* Conservative. It may be ENC_CODERANGE_UNKNOWN. */ 00131 if (ENC_CODERANGE(str) == ENC_CODERANGE_7BIT) 00132 return 1; 00133 00134 enc = STR_ENC_GET(str); 00135 if (rb_enc_mbmaxlen(enc) == 1) 00136 return 1; 00137 00138 /* Conservative. Possibly single byte. 00139 * "\xa1" in Shift_JIS for example. */ 00140 return 0; 00141 } 00142 00143 VALUE rb_fs; 00144 00145 static inline const char * 00146 search_nonascii(const char *p, const char *e) 00147 { 00148 #if SIZEOF_VALUE == 8 00149 # define NONASCII_MASK 0x8080808080808080ULL 00150 #elif SIZEOF_VALUE == 4 00151 # define NONASCII_MASK 0x80808080UL 00152 #endif 00153 #ifdef NONASCII_MASK 00154 if ((int)sizeof(VALUE) * 2 < e - p) { 00155 const VALUE *s, *t; 00156 const VALUE lowbits = sizeof(VALUE) - 1; 00157 s = (const VALUE*)(~lowbits & ((VALUE)p + lowbits)); 00158 while (p < (const char *)s) { 00159 if (!ISASCII(*p)) 00160 return p; 00161 p++; 00162 } 00163 t = (const VALUE*)(~lowbits & (VALUE)e); 00164 while (s < t) { 00165 if (*s & NONASCII_MASK) { 00166 t = s; 00167 break; 00168 } 00169 s++; 00170 } 00171 p = (const char *)t; 00172 } 00173 #endif 00174 while (p < e) { 00175 if (!ISASCII(*p)) 00176 return p; 00177 p++; 00178 } 00179 return NULL; 00180 } 00181 00182 static int 00183 coderange_scan(const char *p, long len, rb_encoding *enc) 00184 { 00185 const char *e = p + len; 00186 00187 if (rb_enc_to_index(enc) == 0) { 00188 /* enc is ASCII-8BIT. ASCII-8BIT string never be broken. */ 00189 p = search_nonascii(p, e); 00190 return p ? ENC_CODERANGE_VALID : ENC_CODERANGE_7BIT; 00191 } 00192 00193 if (rb_enc_asciicompat(enc)) { 00194 p = search_nonascii(p, e); 00195 if (!p) { 00196 return ENC_CODERANGE_7BIT; 00197 } 00198 while (p < e) { 00199 int ret = rb_enc_precise_mbclen(p, e, enc); 00200 if (!MBCLEN_CHARFOUND_P(ret)) { 00201 return ENC_CODERANGE_BROKEN; 00202 } 00203 p += MBCLEN_CHARFOUND_LEN(ret); 00204 if (p < e) { 00205 p = search_nonascii(p, e); 00206 if (!p) { 00207 return ENC_CODERANGE_VALID; 00208 } 00209 } 00210 } 00211 if (e < p) { 00212 return ENC_CODERANGE_BROKEN; 00213 } 00214 return ENC_CODERANGE_VALID; 00215 } 00216 00217 while (p < e) { 00218 int ret = rb_enc_precise_mbclen(p, e, enc); 00219 00220 if (!MBCLEN_CHARFOUND_P(ret)) { 00221 return ENC_CODERANGE_BROKEN; 00222 } 00223 p += MBCLEN_CHARFOUND_LEN(ret); 00224 } 00225 if (e < p) { 00226 return ENC_CODERANGE_BROKEN; 00227 } 00228 return ENC_CODERANGE_VALID; 00229 } 00230 00231 long 00232 rb_str_coderange_scan_restartable(const char *s, const char *e, rb_encoding *enc, int *cr) 00233 { 00234 const char *p = s; 00235 00236 if (*cr == ENC_CODERANGE_BROKEN) 00237 return e - s; 00238 00239 if (rb_enc_to_index(enc) == 0) { 00240 /* enc is ASCII-8BIT. ASCII-8BIT string never be broken. */ 00241 p = search_nonascii(p, e); 00242 *cr = (!p && *cr != ENC_CODERANGE_VALID) ? ENC_CODERANGE_7BIT : ENC_CODERANGE_VALID; 00243 return e - s; 00244 } 00245 else if (rb_enc_asciicompat(enc)) { 00246 p = search_nonascii(p, e); 00247 if (!p) { 00248 if (*cr != ENC_CODERANGE_VALID) *cr = ENC_CODERANGE_7BIT; 00249 return e - s; 00250 } 00251 while (p < e) { 00252 int ret = rb_enc_precise_mbclen(p, e, enc); 00253 if (!MBCLEN_CHARFOUND_P(ret)) { 00254 *cr = MBCLEN_INVALID_P(ret) ? ENC_CODERANGE_BROKEN: ENC_CODERANGE_UNKNOWN; 00255 return p - s; 00256 } 00257 p += MBCLEN_CHARFOUND_LEN(ret); 00258 if (p < e) { 00259 p = search_nonascii(p, e); 00260 if (!p) { 00261 *cr = ENC_CODERANGE_VALID; 00262 return e - s; 00263 } 00264 } 00265 } 00266 *cr = e < p ? ENC_CODERANGE_BROKEN: ENC_CODERANGE_VALID; 00267 return p - s; 00268 } 00269 else { 00270 while (p < e) { 00271 int ret = rb_enc_precise_mbclen(p, e, enc); 00272 if (!MBCLEN_CHARFOUND_P(ret)) { 00273 *cr = MBCLEN_INVALID_P(ret) ? ENC_CODERANGE_BROKEN: ENC_CODERANGE_UNKNOWN; 00274 return p - s; 00275 } 00276 p += MBCLEN_CHARFOUND_LEN(ret); 00277 } 00278 *cr = e < p ? ENC_CODERANGE_BROKEN: ENC_CODERANGE_VALID; 00279 return p - s; 00280 } 00281 } 00282 00283 static inline void 00284 str_enc_copy(VALUE str1, VALUE str2) 00285 { 00286 rb_enc_set_index(str1, ENCODING_GET(str2)); 00287 } 00288 00289 static void 00290 rb_enc_cr_str_copy_for_substr(VALUE dest, VALUE src) 00291 { 00292 /* this function is designed for copying encoding and coderange 00293 * from src to new string "dest" which is made from the part of src. 00294 */ 00295 str_enc_copy(dest, src); 00296 if (RSTRING_LEN(dest) == 0) { 00297 if (!rb_enc_asciicompat(STR_ENC_GET(src))) 00298 ENC_CODERANGE_SET(dest, ENC_CODERANGE_VALID); 00299 else 00300 ENC_CODERANGE_SET(dest, ENC_CODERANGE_7BIT); 00301 return; 00302 } 00303 switch (ENC_CODERANGE(src)) { 00304 case ENC_CODERANGE_7BIT: 00305 ENC_CODERANGE_SET(dest, ENC_CODERANGE_7BIT); 00306 break; 00307 case ENC_CODERANGE_VALID: 00308 if (!rb_enc_asciicompat(STR_ENC_GET(src)) || 00309 search_nonascii(RSTRING_PTR(dest), RSTRING_END(dest))) 00310 ENC_CODERANGE_SET(dest, ENC_CODERANGE_VALID); 00311 else 00312 ENC_CODERANGE_SET(dest, ENC_CODERANGE_7BIT); 00313 break; 00314 default: 00315 break; 00316 } 00317 } 00318 00319 static void 00320 rb_enc_cr_str_exact_copy(VALUE dest, VALUE src) 00321 { 00322 str_enc_copy(dest, src); 00323 ENC_CODERANGE_SET(dest, ENC_CODERANGE(src)); 00324 } 00325 00326 int 00327 rb_enc_str_coderange(VALUE str) 00328 { 00329 int cr = ENC_CODERANGE(str); 00330 00331 if (cr == ENC_CODERANGE_UNKNOWN) { 00332 rb_encoding *enc = STR_ENC_GET(str); 00333 cr = coderange_scan(RSTRING_PTR(str), RSTRING_LEN(str), enc); 00334 ENC_CODERANGE_SET(str, cr); 00335 } 00336 return cr; 00337 } 00338 00339 int 00340 rb_enc_str_asciionly_p(VALUE str) 00341 { 00342 rb_encoding *enc = STR_ENC_GET(str); 00343 00344 if (!rb_enc_asciicompat(enc)) 00345 return FALSE; 00346 else if (rb_enc_str_coderange(str) == ENC_CODERANGE_7BIT) 00347 return TRUE; 00348 return FALSE; 00349 } 00350 00351 static inline void 00352 str_mod_check(VALUE s, const char *p, long len) 00353 { 00354 if (RSTRING_PTR(s) != p || RSTRING_LEN(s) != len){ 00355 rb_raise(rb_eRuntimeError, "string modified"); 00356 } 00357 } 00358 00359 size_t 00360 rb_str_capacity(VALUE str) 00361 { 00362 if (STR_EMBED_P(str)) { 00363 return RSTRING_EMBED_LEN_MAX; 00364 } 00365 else if (STR_NOCAPA_P(str)) { 00366 return RSTRING(str)->as.heap.len; 00367 } 00368 else { 00369 return RSTRING(str)->as.heap.aux.capa; 00370 } 00371 } 00372 00373 static inline VALUE 00374 str_alloc(VALUE klass) 00375 { 00376 NEWOBJ_OF(str, struct RString, klass, T_STRING); 00377 00378 str->as.heap.ptr = 0; 00379 str->as.heap.len = 0; 00380 str->as.heap.aux.capa = 0; 00381 00382 return (VALUE)str; 00383 } 00384 00385 static inline VALUE 00386 empty_str_alloc(VALUE klass) 00387 { 00388 if (RUBY_DTRACE_STRING_CREATE_ENABLED()) { 00389 RUBY_DTRACE_STRING_CREATE(0, rb_sourcefile(), rb_sourceline()); 00390 } 00391 return str_alloc(klass); 00392 } 00393 00394 static VALUE 00395 str_new(VALUE klass, const char *ptr, long len) 00396 { 00397 VALUE str; 00398 00399 if (len < 0) { 00400 rb_raise(rb_eArgError, "negative string size (or size too big)"); 00401 } 00402 00403 if (RUBY_DTRACE_STRING_CREATE_ENABLED()) { 00404 RUBY_DTRACE_STRING_CREATE(len, rb_sourcefile(), rb_sourceline()); 00405 } 00406 00407 str = str_alloc(klass); 00408 if (len > RSTRING_EMBED_LEN_MAX) { 00409 RSTRING(str)->as.heap.aux.capa = len; 00410 RSTRING(str)->as.heap.ptr = ALLOC_N(char,len+1); 00411 STR_SET_NOEMBED(str); 00412 } 00413 else if (len == 0) { 00414 ENC_CODERANGE_SET(str, ENC_CODERANGE_7BIT); 00415 } 00416 if (ptr) { 00417 memcpy(RSTRING_PTR(str), ptr, len); 00418 } 00419 STR_SET_LEN(str, len); 00420 RSTRING_PTR(str)[len] = '\0'; 00421 return str; 00422 } 00423 00424 VALUE 00425 rb_str_new(const char *ptr, long len) 00426 { 00427 return str_new(rb_cString, ptr, len); 00428 } 00429 00430 VALUE 00431 rb_usascii_str_new(const char *ptr, long len) 00432 { 00433 VALUE str = rb_str_new(ptr, len); 00434 ENCODING_CODERANGE_SET(str, rb_usascii_encindex(), ENC_CODERANGE_7BIT); 00435 return str; 00436 } 00437 00438 VALUE 00439 rb_enc_str_new(const char *ptr, long len, rb_encoding *enc) 00440 { 00441 VALUE str = rb_str_new(ptr, len); 00442 rb_enc_associate(str, enc); 00443 return str; 00444 } 00445 00446 VALUE 00447 rb_str_new_cstr(const char *ptr) 00448 { 00449 if (!ptr) { 00450 rb_raise(rb_eArgError, "NULL pointer given"); 00451 } 00452 return rb_str_new(ptr, strlen(ptr)); 00453 } 00454 00455 RUBY_ALIAS_FUNCTION(rb_str_new2(const char *ptr), rb_str_new_cstr, (ptr)) 00456 #define rb_str_new2 rb_str_new_cstr 00457 00458 VALUE 00459 rb_usascii_str_new_cstr(const char *ptr) 00460 { 00461 VALUE str = rb_str_new2(ptr); 00462 ENCODING_CODERANGE_SET(str, rb_usascii_encindex(), ENC_CODERANGE_7BIT); 00463 return str; 00464 } 00465 00466 RUBY_ALIAS_FUNCTION(rb_usascii_str_new2(const char *ptr), rb_usascii_str_new_cstr, (ptr)) 00467 #define rb_usascii_str_new2 rb_usascii_str_new_cstr 00468 00469 VALUE 00470 rb_tainted_str_new(const char *ptr, long len) 00471 { 00472 VALUE str = rb_str_new(ptr, len); 00473 00474 OBJ_TAINT(str); 00475 return str; 00476 } 00477 00478 VALUE 00479 rb_tainted_str_new_cstr(const char *ptr) 00480 { 00481 VALUE str = rb_str_new2(ptr); 00482 00483 OBJ_TAINT(str); 00484 return str; 00485 } 00486 00487 RUBY_ALIAS_FUNCTION(rb_tainted_str_new2(const char *ptr), rb_tainted_str_new_cstr, (ptr)) 00488 #define rb_tainted_str_new2 rb_tainted_str_new_cstr 00489 00490 VALUE 00491 rb_str_conv_enc_opts(VALUE str, rb_encoding *from, rb_encoding *to, int ecflags, VALUE ecopts) 00492 { 00493 extern VALUE rb_cEncodingConverter; 00494 rb_econv_t *ec; 00495 rb_econv_result_t ret; 00496 long len, olen; 00497 VALUE econv_wrapper; 00498 VALUE newstr; 00499 const unsigned char *start, *sp; 00500 unsigned char *dest, *dp; 00501 size_t converted_output = 0; 00502 00503 if (!to) return str; 00504 if (!from) from = rb_enc_get(str); 00505 if (from == to) return str; 00506 if ((rb_enc_asciicompat(to) && ENC_CODERANGE(str) == ENC_CODERANGE_7BIT) || 00507 to == rb_ascii8bit_encoding()) { 00508 if (STR_ENC_GET(str) != to) { 00509 str = rb_str_dup(str); 00510 rb_enc_associate(str, to); 00511 } 00512 return str; 00513 } 00514 00515 len = RSTRING_LEN(str); 00516 newstr = rb_str_new(0, len); 00517 olen = len; 00518 00519 econv_wrapper = rb_obj_alloc(rb_cEncodingConverter); 00520 RBASIC(econv_wrapper)->klass = 0; 00521 ec = rb_econv_open_opts(from->name, to->name, ecflags, ecopts); 00522 if (!ec) return str; 00523 DATA_PTR(econv_wrapper) = ec; 00524 00525 sp = (unsigned char*)RSTRING_PTR(str); 00526 start = sp; 00527 while ((dest = (unsigned char*)RSTRING_PTR(newstr)), 00528 (dp = dest + converted_output), 00529 (ret = rb_econv_convert(ec, &sp, start + len, &dp, dest + olen, 0)), 00530 ret == econv_destination_buffer_full) { 00531 /* destination buffer short */ 00532 size_t converted_input = sp - start; 00533 size_t rest = len - converted_input; 00534 converted_output = dp - dest; 00535 rb_str_set_len(newstr, converted_output); 00536 if (converted_input && converted_output && 00537 rest < (LONG_MAX / converted_output)) { 00538 rest = (rest * converted_output) / converted_input; 00539 } 00540 else { 00541 rest = olen; 00542 } 00543 olen += rest < 2 ? 2 : rest; 00544 rb_str_resize(newstr, olen); 00545 } 00546 DATA_PTR(econv_wrapper) = 0; 00547 rb_econv_close(ec); 00548 rb_gc_force_recycle(econv_wrapper); 00549 switch (ret) { 00550 case econv_finished: 00551 len = dp - (unsigned char*)RSTRING_PTR(newstr); 00552 rb_str_set_len(newstr, len); 00553 rb_enc_associate(newstr, to); 00554 return newstr; 00555 00556 default: 00557 /* some error, return original */ 00558 return str; 00559 } 00560 } 00561 00562 VALUE 00563 rb_str_conv_enc(VALUE str, rb_encoding *from, rb_encoding *to) 00564 { 00565 return rb_str_conv_enc_opts(str, from, to, 0, Qnil); 00566 } 00567 00568 VALUE 00569 rb_external_str_new_with_enc(const char *ptr, long len, rb_encoding *eenc) 00570 { 00571 VALUE str; 00572 00573 str = rb_tainted_str_new(ptr, len); 00574 if (eenc == rb_usascii_encoding() && 00575 rb_enc_str_coderange(str) != ENC_CODERANGE_7BIT) { 00576 rb_enc_associate(str, rb_ascii8bit_encoding()); 00577 return str; 00578 } 00579 rb_enc_associate(str, eenc); 00580 return rb_str_conv_enc(str, eenc, rb_default_internal_encoding()); 00581 } 00582 00583 VALUE 00584 rb_external_str_new(const char *ptr, long len) 00585 { 00586 return rb_external_str_new_with_enc(ptr, len, rb_default_external_encoding()); 00587 } 00588 00589 VALUE 00590 rb_external_str_new_cstr(const char *ptr) 00591 { 00592 return rb_external_str_new_with_enc(ptr, strlen(ptr), rb_default_external_encoding()); 00593 } 00594 00595 VALUE 00596 rb_locale_str_new(const char *ptr, long len) 00597 { 00598 return rb_external_str_new_with_enc(ptr, len, rb_locale_encoding()); 00599 } 00600 00601 VALUE 00602 rb_locale_str_new_cstr(const char *ptr) 00603 { 00604 return rb_external_str_new_with_enc(ptr, strlen(ptr), rb_locale_encoding()); 00605 } 00606 00607 VALUE 00608 rb_filesystem_str_new(const char *ptr, long len) 00609 { 00610 return rb_external_str_new_with_enc(ptr, len, rb_filesystem_encoding()); 00611 } 00612 00613 VALUE 00614 rb_filesystem_str_new_cstr(const char *ptr) 00615 { 00616 return rb_external_str_new_with_enc(ptr, strlen(ptr), rb_filesystem_encoding()); 00617 } 00618 00619 VALUE 00620 rb_str_export(VALUE str) 00621 { 00622 return rb_str_conv_enc(str, STR_ENC_GET(str), rb_default_external_encoding()); 00623 } 00624 00625 VALUE 00626 rb_str_export_locale(VALUE str) 00627 { 00628 return rb_str_conv_enc(str, STR_ENC_GET(str), rb_locale_encoding()); 00629 } 00630 00631 VALUE 00632 rb_str_export_to_enc(VALUE str, rb_encoding *enc) 00633 { 00634 return rb_str_conv_enc(str, STR_ENC_GET(str), enc); 00635 } 00636 00637 static VALUE 00638 str_replace_shared_without_enc(VALUE str2, VALUE str) 00639 { 00640 if (RSTRING_LEN(str) <= RSTRING_EMBED_LEN_MAX) { 00641 STR_SET_EMBED(str2); 00642 memcpy(RSTRING_PTR(str2), RSTRING_PTR(str), RSTRING_LEN(str)+1); 00643 STR_SET_EMBED_LEN(str2, RSTRING_LEN(str)); 00644 } 00645 else { 00646 str = rb_str_new_frozen(str); 00647 FL_SET(str2, STR_NOEMBED); 00648 RSTRING(str2)->as.heap.len = RSTRING_LEN(str); 00649 RSTRING(str2)->as.heap.ptr = RSTRING_PTR(str); 00650 RSTRING(str2)->as.heap.aux.shared = str; 00651 FL_SET(str2, ELTS_SHARED); 00652 } 00653 return str2; 00654 } 00655 00656 static VALUE 00657 str_replace_shared(VALUE str2, VALUE str) 00658 { 00659 str_replace_shared_without_enc(str2, str); 00660 rb_enc_cr_str_exact_copy(str2, str); 00661 return str2; 00662 } 00663 00664 static VALUE 00665 str_new_shared(VALUE klass, VALUE str) 00666 { 00667 return str_replace_shared(str_alloc(klass), str); 00668 } 00669 00670 static VALUE 00671 str_new3(VALUE klass, VALUE str) 00672 { 00673 return str_new_shared(klass, str); 00674 } 00675 00676 VALUE 00677 rb_str_new_shared(VALUE str) 00678 { 00679 VALUE str2 = str_new3(rb_obj_class(str), str); 00680 00681 OBJ_INFECT(str2, str); 00682 return str2; 00683 } 00684 00685 RUBY_ALIAS_FUNCTION(rb_str_new3(VALUE str), rb_str_new_shared, (str)) 00686 #define rb_str_new3 rb_str_new_shared 00687 00688 static VALUE 00689 str_new4(VALUE klass, VALUE str) 00690 { 00691 VALUE str2; 00692 00693 str2 = str_alloc(klass); 00694 STR_SET_NOEMBED(str2); 00695 RSTRING(str2)->as.heap.len = RSTRING_LEN(str); 00696 RSTRING(str2)->as.heap.ptr = RSTRING_PTR(str); 00697 if (STR_SHARED_P(str)) { 00698 VALUE shared = RSTRING(str)->as.heap.aux.shared; 00699 assert(OBJ_FROZEN(shared)); 00700 FL_SET(str2, ELTS_SHARED); 00701 RSTRING(str2)->as.heap.aux.shared = shared; 00702 } 00703 else { 00704 FL_SET(str, ELTS_SHARED); 00705 RSTRING(str)->as.heap.aux.shared = str2; 00706 } 00707 rb_enc_cr_str_exact_copy(str2, str); 00708 OBJ_INFECT(str2, str); 00709 return str2; 00710 } 00711 00712 VALUE 00713 rb_str_new_frozen(VALUE orig) 00714 { 00715 VALUE klass, str; 00716 00717 if (OBJ_FROZEN(orig)) return orig; 00718 klass = rb_obj_class(orig); 00719 if (STR_SHARED_P(orig) && (str = RSTRING(orig)->as.heap.aux.shared)) { 00720 long ofs; 00721 assert(OBJ_FROZEN(str)); 00722 ofs = RSTRING_LEN(str) - RSTRING_LEN(orig); 00723 if ((ofs > 0) || (klass != RBASIC(str)->klass) || 00724 ((RBASIC(str)->flags ^ RBASIC(orig)->flags) & (FL_TAINT|FL_UNTRUSTED)) || 00725 ENCODING_GET(str) != ENCODING_GET(orig)) { 00726 str = str_new3(klass, str); 00727 RSTRING(str)->as.heap.ptr += ofs; 00728 RSTRING(str)->as.heap.len -= ofs; 00729 rb_enc_cr_str_exact_copy(str, orig); 00730 OBJ_INFECT(str, orig); 00731 } 00732 } 00733 else if (STR_EMBED_P(orig)) { 00734 str = str_new(klass, RSTRING_PTR(orig), RSTRING_LEN(orig)); 00735 rb_enc_cr_str_exact_copy(str, orig); 00736 OBJ_INFECT(str, orig); 00737 } 00738 else if (STR_ASSOC_P(orig)) { 00739 VALUE assoc = RSTRING(orig)->as.heap.aux.shared; 00740 FL_UNSET(orig, STR_ASSOC); 00741 str = str_new4(klass, orig); 00742 FL_SET(str, STR_ASSOC); 00743 RSTRING(str)->as.heap.aux.shared = assoc; 00744 } 00745 else { 00746 str = str_new4(klass, orig); 00747 } 00748 OBJ_FREEZE(str); 00749 return str; 00750 } 00751 00752 RUBY_ALIAS_FUNCTION(rb_str_new4(VALUE orig), rb_str_new_frozen, (orig)) 00753 #define rb_str_new4 rb_str_new_frozen 00754 00755 VALUE 00756 rb_str_new_with_class(VALUE obj, const char *ptr, long len) 00757 { 00758 return str_new(rb_obj_class(obj), ptr, len); 00759 } 00760 00761 RUBY_ALIAS_FUNCTION(rb_str_new5(VALUE obj, const char *ptr, long len), 00762 rb_str_new_with_class, (obj, ptr, len)) 00763 #define rb_str_new5 rb_str_new_with_class 00764 00765 static VALUE 00766 str_new_empty(VALUE str) 00767 { 00768 VALUE v = rb_str_new5(str, 0, 0); 00769 rb_enc_copy(v, str); 00770 OBJ_INFECT(v, str); 00771 return v; 00772 } 00773 00774 #define STR_BUF_MIN_SIZE 128 00775 00776 VALUE 00777 rb_str_buf_new(long capa) 00778 { 00779 VALUE str = str_alloc(rb_cString); 00780 00781 if (capa < STR_BUF_MIN_SIZE) { 00782 capa = STR_BUF_MIN_SIZE; 00783 } 00784 FL_SET(str, STR_NOEMBED); 00785 RSTRING(str)->as.heap.aux.capa = capa; 00786 RSTRING(str)->as.heap.ptr = ALLOC_N(char, capa+1); 00787 RSTRING(str)->as.heap.ptr[0] = '\0'; 00788 00789 return str; 00790 } 00791 00792 VALUE 00793 rb_str_buf_new_cstr(const char *ptr) 00794 { 00795 VALUE str; 00796 long len = strlen(ptr); 00797 00798 str = rb_str_buf_new(len); 00799 rb_str_buf_cat(str, ptr, len); 00800 00801 return str; 00802 } 00803 00804 RUBY_ALIAS_FUNCTION(rb_str_buf_new2(const char *ptr), rb_str_buf_new_cstr, (ptr)) 00805 #define rb_str_buf_new2 rb_str_buf_new_cstr 00806 00807 VALUE 00808 rb_str_tmp_new(long len) 00809 { 00810 return str_new(0, 0, len); 00811 } 00812 00813 void * 00814 rb_alloc_tmp_buffer(volatile VALUE *store, long len) 00815 { 00816 VALUE s = rb_str_tmp_new(len); 00817 *store = s; 00818 return RSTRING_PTR(s); 00819 } 00820 00821 void 00822 rb_free_tmp_buffer(volatile VALUE *store) 00823 { 00824 VALUE s = *store; 00825 *store = 0; 00826 if (s) rb_str_clear(s); 00827 } 00828 00829 void 00830 rb_str_free(VALUE str) 00831 { 00832 if (!STR_EMBED_P(str) && !STR_SHARED_P(str)) { 00833 xfree(RSTRING(str)->as.heap.ptr); 00834 } 00835 } 00836 00837 RUBY_FUNC_EXPORTED size_t 00838 rb_str_memsize(VALUE str) 00839 { 00840 if (!STR_EMBED_P(str) && !STR_SHARED_P(str)) { 00841 return RSTRING(str)->as.heap.aux.capa + 1; /* termlen */ 00842 } 00843 else { 00844 return 0; 00845 } 00846 } 00847 00848 VALUE 00849 rb_str_to_str(VALUE str) 00850 { 00851 return rb_convert_type(str, T_STRING, "String", "to_str"); 00852 } 00853 00854 static inline void str_discard(VALUE str); 00855 00856 void 00857 rb_str_shared_replace(VALUE str, VALUE str2) 00858 { 00859 rb_encoding *enc; 00860 int cr; 00861 if (str == str2) return; 00862 enc = STR_ENC_GET(str2); 00863 cr = ENC_CODERANGE(str2); 00864 str_discard(str); 00865 OBJ_INFECT(str, str2); 00866 if (RSTRING_LEN(str2) <= RSTRING_EMBED_LEN_MAX) { 00867 STR_SET_EMBED(str); 00868 memcpy(RSTRING_PTR(str), RSTRING_PTR(str2), RSTRING_LEN(str2)+1); 00869 STR_SET_EMBED_LEN(str, RSTRING_LEN(str2)); 00870 rb_enc_associate(str, enc); 00871 ENC_CODERANGE_SET(str, cr); 00872 return; 00873 } 00874 STR_SET_NOEMBED(str); 00875 STR_UNSET_NOCAPA(str); 00876 RSTRING(str)->as.heap.ptr = RSTRING_PTR(str2); 00877 RSTRING(str)->as.heap.len = RSTRING_LEN(str2); 00878 if (STR_NOCAPA_P(str2)) { 00879 FL_SET(str, RBASIC(str2)->flags & STR_NOCAPA); 00880 RSTRING(str)->as.heap.aux.shared = RSTRING(str2)->as.heap.aux.shared; 00881 } 00882 else { 00883 RSTRING(str)->as.heap.aux.capa = RSTRING(str2)->as.heap.aux.capa; 00884 } 00885 STR_SET_EMBED(str2); /* abandon str2 */ 00886 RSTRING_PTR(str2)[0] = 0; 00887 STR_SET_EMBED_LEN(str2, 0); 00888 rb_enc_associate(str, enc); 00889 ENC_CODERANGE_SET(str, cr); 00890 } 00891 00892 static ID id_to_s; 00893 00894 VALUE 00895 rb_obj_as_string(VALUE obj) 00896 { 00897 VALUE str; 00898 00899 if (RB_TYPE_P(obj, T_STRING)) { 00900 return obj; 00901 } 00902 str = rb_funcall(obj, id_to_s, 0); 00903 if (!RB_TYPE_P(str, T_STRING)) 00904 return rb_any_to_s(obj); 00905 if (OBJ_TAINTED(obj)) OBJ_TAINT(str); 00906 return str; 00907 } 00908 00909 static VALUE 00910 str_replace(VALUE str, VALUE str2) 00911 { 00912 long len; 00913 00914 len = RSTRING_LEN(str2); 00915 if (STR_ASSOC_P(str2)) { 00916 str2 = rb_str_new4(str2); 00917 } 00918 if (STR_SHARED_P(str2)) { 00919 VALUE shared = RSTRING(str2)->as.heap.aux.shared; 00920 assert(OBJ_FROZEN(shared)); 00921 STR_SET_NOEMBED(str); 00922 RSTRING(str)->as.heap.len = len; 00923 RSTRING(str)->as.heap.ptr = RSTRING_PTR(str2); 00924 FL_SET(str, ELTS_SHARED); 00925 FL_UNSET(str, STR_ASSOC); 00926 RSTRING(str)->as.heap.aux.shared = shared; 00927 } 00928 else { 00929 str_replace_shared(str, str2); 00930 } 00931 00932 OBJ_INFECT(str, str2); 00933 rb_enc_cr_str_exact_copy(str, str2); 00934 return str; 00935 } 00936 00937 static VALUE 00938 str_duplicate(VALUE klass, VALUE str) 00939 { 00940 VALUE dup = str_alloc(klass); 00941 str_replace(dup, str); 00942 return dup; 00943 } 00944 00945 VALUE 00946 rb_str_dup(VALUE str) 00947 { 00948 return str_duplicate(rb_obj_class(str), str); 00949 } 00950 00951 VALUE 00952 rb_str_resurrect(VALUE str) 00953 { 00954 if (RUBY_DTRACE_STRING_CREATE_ENABLED()) { 00955 RUBY_DTRACE_STRING_CREATE(RSTRING_LEN(str), 00956 rb_sourcefile(), rb_sourceline()); 00957 } 00958 return str_replace(str_alloc(rb_cString), str); 00959 } 00960 00961 /* 00962 * call-seq: 00963 * String.new(str="") -> new_str 00964 * 00965 * Returns a new string object containing a copy of <i>str</i>. 00966 */ 00967 00968 static VALUE 00969 rb_str_init(int argc, VALUE *argv, VALUE str) 00970 { 00971 VALUE orig; 00972 00973 if (argc > 0 && rb_scan_args(argc, argv, "01", &orig) == 1) 00974 rb_str_replace(str, orig); 00975 return str; 00976 } 00977 00978 static inline long 00979 enc_strlen(const char *p, const char *e, rb_encoding *enc, int cr) 00980 { 00981 long c; 00982 const char *q; 00983 00984 if (rb_enc_mbmaxlen(enc) == rb_enc_mbminlen(enc)) { 00985 return (e - p + rb_enc_mbminlen(enc) - 1) / rb_enc_mbminlen(enc); 00986 } 00987 else if (rb_enc_asciicompat(enc)) { 00988 c = 0; 00989 if (cr == ENC_CODERANGE_7BIT || cr == ENC_CODERANGE_VALID) { 00990 while (p < e) { 00991 if (ISASCII(*p)) { 00992 q = search_nonascii(p, e); 00993 if (!q) 00994 return c + (e - p); 00995 c += q - p; 00996 p = q; 00997 } 00998 p += rb_enc_fast_mbclen(p, e, enc); 00999 c++; 01000 } 01001 } 01002 else { 01003 while (p < e) { 01004 if (ISASCII(*p)) { 01005 q = search_nonascii(p, e); 01006 if (!q) 01007 return c + (e - p); 01008 c += q - p; 01009 p = q; 01010 } 01011 p += rb_enc_mbclen(p, e, enc); 01012 c++; 01013 } 01014 } 01015 return c; 01016 } 01017 01018 for (c=0; p<e; c++) { 01019 p += rb_enc_mbclen(p, e, enc); 01020 } 01021 return c; 01022 } 01023 01024 long 01025 rb_enc_strlen(const char *p, const char *e, rb_encoding *enc) 01026 { 01027 return enc_strlen(p, e, enc, ENC_CODERANGE_UNKNOWN); 01028 } 01029 01030 long 01031 rb_enc_strlen_cr(const char *p, const char *e, rb_encoding *enc, int *cr) 01032 { 01033 long c; 01034 const char *q; 01035 int ret; 01036 01037 *cr = 0; 01038 if (rb_enc_mbmaxlen(enc) == rb_enc_mbminlen(enc)) { 01039 return (e - p + rb_enc_mbminlen(enc) - 1) / rb_enc_mbminlen(enc); 01040 } 01041 else if (rb_enc_asciicompat(enc)) { 01042 c = 0; 01043 while (p < e) { 01044 if (ISASCII(*p)) { 01045 q = search_nonascii(p, e); 01046 if (!q) { 01047 if (!*cr) *cr = ENC_CODERANGE_7BIT; 01048 return c + (e - p); 01049 } 01050 c += q - p; 01051 p = q; 01052 } 01053 ret = rb_enc_precise_mbclen(p, e, enc); 01054 if (MBCLEN_CHARFOUND_P(ret)) { 01055 *cr |= ENC_CODERANGE_VALID; 01056 p += MBCLEN_CHARFOUND_LEN(ret); 01057 } 01058 else { 01059 *cr = ENC_CODERANGE_BROKEN; 01060 p++; 01061 } 01062 c++; 01063 } 01064 if (!*cr) *cr = ENC_CODERANGE_7BIT; 01065 return c; 01066 } 01067 01068 for (c=0; p<e; c++) { 01069 ret = rb_enc_precise_mbclen(p, e, enc); 01070 if (MBCLEN_CHARFOUND_P(ret)) { 01071 *cr |= ENC_CODERANGE_VALID; 01072 p += MBCLEN_CHARFOUND_LEN(ret); 01073 } 01074 else { 01075 *cr = ENC_CODERANGE_BROKEN; 01076 if (p + rb_enc_mbminlen(enc) <= e) 01077 p += rb_enc_mbminlen(enc); 01078 else 01079 p = e; 01080 } 01081 } 01082 if (!*cr) *cr = ENC_CODERANGE_7BIT; 01083 return c; 01084 } 01085 01086 #ifdef NONASCII_MASK 01087 #define is_utf8_lead_byte(c) (((c)&0xC0) != 0x80) 01088 01089 /* 01090 * UTF-8 leading bytes have either 0xxxxxxx or 11xxxxxx 01091 * bit represention. (see http://en.wikipedia.org/wiki/UTF-8) 01092 * Therefore, following pseudo code can detect UTF-8 leading byte. 01093 * 01094 * if (!(byte & 0x80)) 01095 * byte |= 0x40; // turn on bit6 01096 * return ((byte>>6) & 1); // bit6 represent it's leading byte or not. 01097 * 01098 * This function calculate every bytes in the argument word `s' 01099 * using the above logic concurrently. and gather every bytes result. 01100 */ 01101 static inline VALUE 01102 count_utf8_lead_bytes_with_word(const VALUE *s) 01103 { 01104 VALUE d = *s; 01105 01106 /* Transform into bit0 represent UTF-8 leading or not. */ 01107 d |= ~(d>>1); 01108 d >>= 6; 01109 d &= NONASCII_MASK >> 7; 01110 01111 /* Gather every bytes. */ 01112 d += (d>>8); 01113 d += (d>>16); 01114 #if SIZEOF_VALUE == 8 01115 d += (d>>32); 01116 #endif 01117 return (d&0xF); 01118 } 01119 #endif 01120 01121 static long 01122 str_strlen(VALUE str, rb_encoding *enc) 01123 { 01124 const char *p, *e; 01125 long n; 01126 int cr; 01127 01128 if (single_byte_optimizable(str)) return RSTRING_LEN(str); 01129 if (!enc) enc = STR_ENC_GET(str); 01130 p = RSTRING_PTR(str); 01131 e = RSTRING_END(str); 01132 cr = ENC_CODERANGE(str); 01133 #ifdef NONASCII_MASK 01134 if (ENC_CODERANGE(str) == ENC_CODERANGE_VALID && 01135 enc == rb_utf8_encoding()) { 01136 01137 VALUE len = 0; 01138 if ((int)sizeof(VALUE) * 2 < e - p) { 01139 const VALUE *s, *t; 01140 const VALUE lowbits = sizeof(VALUE) - 1; 01141 s = (const VALUE*)(~lowbits & ((VALUE)p + lowbits)); 01142 t = (const VALUE*)(~lowbits & (VALUE)e); 01143 while (p < (const char *)s) { 01144 if (is_utf8_lead_byte(*p)) len++; 01145 p++; 01146 } 01147 while (s < t) { 01148 len += count_utf8_lead_bytes_with_word(s); 01149 s++; 01150 } 01151 p = (const char *)s; 01152 } 01153 while (p < e) { 01154 if (is_utf8_lead_byte(*p)) len++; 01155 p++; 01156 } 01157 return (long)len; 01158 } 01159 #endif 01160 n = rb_enc_strlen_cr(p, e, enc, &cr); 01161 if (cr) { 01162 ENC_CODERANGE_SET(str, cr); 01163 } 01164 return n; 01165 } 01166 01167 long 01168 rb_str_strlen(VALUE str) 01169 { 01170 return str_strlen(str, STR_ENC_GET(str)); 01171 } 01172 01173 /* 01174 * call-seq: 01175 * str.length -> integer 01176 * str.size -> integer 01177 * 01178 * Returns the character length of <i>str</i>. 01179 */ 01180 01181 VALUE 01182 rb_str_length(VALUE str) 01183 { 01184 long len; 01185 01186 len = str_strlen(str, STR_ENC_GET(str)); 01187 return LONG2NUM(len); 01188 } 01189 01190 /* 01191 * call-seq: 01192 * str.bytesize -> integer 01193 * 01194 * Returns the length of +str+ in bytes. 01195 * 01196 * "\x80\u3042".bytesize #=> 4 01197 * "hello".bytesize #=> 5 01198 */ 01199 01200 static VALUE 01201 rb_str_bytesize(VALUE str) 01202 { 01203 return LONG2NUM(RSTRING_LEN(str)); 01204 } 01205 01206 /* 01207 * call-seq: 01208 * str.empty? -> true or false 01209 * 01210 * Returns <code>true</code> if <i>str</i> has a length of zero. 01211 * 01212 * "hello".empty? #=> false 01213 * " ".empty? #=> false 01214 * "".empty? #=> true 01215 */ 01216 01217 static VALUE 01218 rb_str_empty(VALUE str) 01219 { 01220 if (RSTRING_LEN(str) == 0) 01221 return Qtrue; 01222 return Qfalse; 01223 } 01224 01225 /* 01226 * call-seq: 01227 * str + other_str -> new_str 01228 * 01229 * Concatenation---Returns a new <code>String</code> containing 01230 * <i>other_str</i> concatenated to <i>str</i>. 01231 * 01232 * "Hello from " + self.to_s #=> "Hello from main" 01233 */ 01234 01235 VALUE 01236 rb_str_plus(VALUE str1, VALUE str2) 01237 { 01238 VALUE str3; 01239 rb_encoding *enc; 01240 01241 StringValue(str2); 01242 enc = rb_enc_check(str1, str2); 01243 str3 = rb_str_new(0, RSTRING_LEN(str1)+RSTRING_LEN(str2)); 01244 memcpy(RSTRING_PTR(str3), RSTRING_PTR(str1), RSTRING_LEN(str1)); 01245 memcpy(RSTRING_PTR(str3) + RSTRING_LEN(str1), 01246 RSTRING_PTR(str2), RSTRING_LEN(str2)); 01247 RSTRING_PTR(str3)[RSTRING_LEN(str3)] = '\0'; 01248 01249 if (OBJ_TAINTED(str1) || OBJ_TAINTED(str2)) 01250 OBJ_TAINT(str3); 01251 ENCODING_CODERANGE_SET(str3, rb_enc_to_index(enc), 01252 ENC_CODERANGE_AND(ENC_CODERANGE(str1), ENC_CODERANGE(str2))); 01253 return str3; 01254 } 01255 01256 /* 01257 * call-seq: 01258 * str * integer -> new_str 01259 * 01260 * Copy --- Returns a new String containing +integer+ copies of the receiver. 01261 * +integer+ must be greater than or equal to 0. 01262 * 01263 * "Ho! " * 3 #=> "Ho! Ho! Ho! " 01264 * "Ho! " * 0 #=> "" 01265 */ 01266 01267 VALUE 01268 rb_str_times(VALUE str, VALUE times) 01269 { 01270 VALUE str2; 01271 long n, len; 01272 char *ptr2; 01273 01274 len = NUM2LONG(times); 01275 if (len < 0) { 01276 rb_raise(rb_eArgError, "negative argument"); 01277 } 01278 if (len && LONG_MAX/len < RSTRING_LEN(str)) { 01279 rb_raise(rb_eArgError, "argument too big"); 01280 } 01281 01282 str2 = rb_str_new5(str, 0, len *= RSTRING_LEN(str)); 01283 ptr2 = RSTRING_PTR(str2); 01284 if (len) { 01285 n = RSTRING_LEN(str); 01286 memcpy(ptr2, RSTRING_PTR(str), n); 01287 while (n <= len/2) { 01288 memcpy(ptr2 + n, ptr2, n); 01289 n *= 2; 01290 } 01291 memcpy(ptr2 + n, ptr2, len-n); 01292 } 01293 ptr2[RSTRING_LEN(str2)] = '\0'; 01294 OBJ_INFECT(str2, str); 01295 rb_enc_cr_str_copy_for_substr(str2, str); 01296 01297 return str2; 01298 } 01299 01300 /* 01301 * call-seq: 01302 * str % arg -> new_str 01303 * 01304 * Format---Uses <i>str</i> as a format specification, and returns the result 01305 * of applying it to <i>arg</i>. If the format specification contains more than 01306 * one substitution, then <i>arg</i> must be an <code>Array</code> or <code>Hash</code> 01307 * containing the values to be substituted. See <code>Kernel::sprintf</code> for 01308 * details of the format string. 01309 * 01310 * "%05d" % 123 #=> "00123" 01311 * "%-5s: %08x" % [ "ID", self.object_id ] #=> "ID : 200e14d6" 01312 * "foo = %{foo}" % { :foo => 'bar' } #=> "foo = bar" 01313 */ 01314 01315 static VALUE 01316 rb_str_format_m(VALUE str, VALUE arg) 01317 { 01318 volatile VALUE tmp = rb_check_array_type(arg); 01319 01320 if (!NIL_P(tmp)) { 01321 return rb_str_format(RARRAY_LENINT(tmp), RARRAY_PTR(tmp), str); 01322 } 01323 return rb_str_format(1, &arg, str); 01324 } 01325 01326 static inline void 01327 str_modifiable(VALUE str) 01328 { 01329 if (FL_TEST(str, STR_TMPLOCK)) { 01330 rb_raise(rb_eRuntimeError, "can't modify string; temporarily locked"); 01331 } 01332 rb_check_frozen(str); 01333 if (!OBJ_UNTRUSTED(str) && rb_safe_level() >= 4) 01334 rb_raise(rb_eSecurityError, "Insecure: can't modify string"); 01335 } 01336 01337 static inline int 01338 str_independent(VALUE str) 01339 { 01340 str_modifiable(str); 01341 if (!STR_SHARED_P(str)) return 1; 01342 if (STR_EMBED_P(str)) return 1; 01343 return 0; 01344 } 01345 01346 static void 01347 str_make_independent_expand(VALUE str, long expand) 01348 { 01349 char *ptr; 01350 long len = RSTRING_LEN(str); 01351 long capa = len + expand; 01352 01353 if (len > capa) len = capa; 01354 ptr = ALLOC_N(char, capa + 1); 01355 if (RSTRING_PTR(str)) { 01356 memcpy(ptr, RSTRING_PTR(str), len); 01357 } 01358 STR_SET_NOEMBED(str); 01359 STR_UNSET_NOCAPA(str); 01360 ptr[len] = 0; 01361 RSTRING(str)->as.heap.ptr = ptr; 01362 RSTRING(str)->as.heap.len = len; 01363 RSTRING(str)->as.heap.aux.capa = capa; 01364 } 01365 01366 #define str_make_independent(str) str_make_independent_expand((str), 0L) 01367 01368 void 01369 rb_str_modify(VALUE str) 01370 { 01371 if (!str_independent(str)) 01372 str_make_independent(str); 01373 ENC_CODERANGE_CLEAR(str); 01374 } 01375 01376 void 01377 rb_str_modify_expand(VALUE str, long expand) 01378 { 01379 if (expand < 0) { 01380 rb_raise(rb_eArgError, "negative expanding string size"); 01381 } 01382 if (!str_independent(str)) { 01383 str_make_independent_expand(str, expand); 01384 } 01385 else if (expand > 0) { 01386 long len = RSTRING_LEN(str); 01387 long capa = len + expand; 01388 if (!STR_EMBED_P(str)) { 01389 REALLOC_N(RSTRING(str)->as.heap.ptr, char, capa+1); 01390 STR_UNSET_NOCAPA(str); 01391 RSTRING(str)->as.heap.aux.capa = capa; 01392 } 01393 else if (capa > RSTRING_EMBED_LEN_MAX) { 01394 str_make_independent_expand(str, expand); 01395 } 01396 } 01397 ENC_CODERANGE_CLEAR(str); 01398 } 01399 01400 /* As rb_str_modify(), but don't clear coderange */ 01401 static void 01402 str_modify_keep_cr(VALUE str) 01403 { 01404 if (!str_independent(str)) 01405 str_make_independent(str); 01406 if (ENC_CODERANGE(str) == ENC_CODERANGE_BROKEN) 01407 /* Force re-scan later */ 01408 ENC_CODERANGE_CLEAR(str); 01409 } 01410 01411 static inline void 01412 str_discard(VALUE str) 01413 { 01414 str_modifiable(str); 01415 if (!STR_SHARED_P(str) && !STR_EMBED_P(str)) { 01416 xfree(RSTRING_PTR(str)); 01417 RSTRING(str)->as.heap.ptr = 0; 01418 RSTRING(str)->as.heap.len = 0; 01419 } 01420 } 01421 01422 void 01423 rb_str_associate(VALUE str, VALUE add) 01424 { 01425 /* sanity check */ 01426 rb_check_frozen(str); 01427 if (STR_ASSOC_P(str)) { 01428 /* already associated */ 01429 rb_ary_concat(RSTRING(str)->as.heap.aux.shared, add); 01430 } 01431 else { 01432 if (STR_SHARED_P(str)) { 01433 VALUE assoc = RSTRING(str)->as.heap.aux.shared; 01434 str_make_independent(str); 01435 if (STR_ASSOC_P(assoc)) { 01436 assoc = RSTRING(assoc)->as.heap.aux.shared; 01437 rb_ary_concat(assoc, add); 01438 add = assoc; 01439 } 01440 } 01441 else if (STR_EMBED_P(str)) { 01442 str_make_independent(str); 01443 } 01444 else if (RSTRING(str)->as.heap.aux.capa != RSTRING_LEN(str)) { 01445 RESIZE_CAPA(str, RSTRING_LEN(str)); 01446 } 01447 FL_SET(str, STR_ASSOC); 01448 RBASIC(add)->klass = 0; 01449 RSTRING(str)->as.heap.aux.shared = add; 01450 } 01451 } 01452 01453 VALUE 01454 rb_str_associated(VALUE str) 01455 { 01456 if (STR_SHARED_P(str)) str = RSTRING(str)->as.heap.aux.shared; 01457 if (STR_ASSOC_P(str)) { 01458 return RSTRING(str)->as.heap.aux.shared; 01459 } 01460 return Qfalse; 01461 } 01462 01463 void 01464 rb_must_asciicompat(VALUE str) 01465 { 01466 rb_encoding *enc = rb_enc_get(str); 01467 if (!rb_enc_asciicompat(enc)) { 01468 rb_raise(rb_eEncCompatError, "ASCII incompatible encoding: %s", rb_enc_name(enc)); 01469 } 01470 } 01471 01472 VALUE 01473 rb_string_value(volatile VALUE *ptr) 01474 { 01475 VALUE s = *ptr; 01476 if (!RB_TYPE_P(s, T_STRING)) { 01477 s = rb_str_to_str(s); 01478 *ptr = s; 01479 } 01480 return s; 01481 } 01482 01483 char * 01484 rb_string_value_ptr(volatile VALUE *ptr) 01485 { 01486 VALUE str = rb_string_value(ptr); 01487 return RSTRING_PTR(str); 01488 } 01489 01490 char * 01491 rb_string_value_cstr(volatile VALUE *ptr) 01492 { 01493 VALUE str = rb_string_value(ptr); 01494 char *s = RSTRING_PTR(str); 01495 long len = RSTRING_LEN(str); 01496 01497 if (!s || memchr(s, 0, len)) { 01498 rb_raise(rb_eArgError, "string contains null byte"); 01499 } 01500 if (s[len]) { 01501 rb_str_modify(str); 01502 s = RSTRING_PTR(str); 01503 s[RSTRING_LEN(str)] = 0; 01504 } 01505 return s; 01506 } 01507 01508 VALUE 01509 rb_check_string_type(VALUE str) 01510 { 01511 str = rb_check_convert_type(str, T_STRING, "String", "to_str"); 01512 return str; 01513 } 01514 01515 /* 01516 * call-seq: 01517 * String.try_convert(obj) -> string or nil 01518 * 01519 * Try to convert <i>obj</i> into a String, using to_str method. 01520 * Returns converted string or nil if <i>obj</i> cannot be converted 01521 * for any reason. 01522 * 01523 * String.try_convert("str") #=> "str" 01524 * String.try_convert(/re/) #=> nil 01525 */ 01526 static VALUE 01527 rb_str_s_try_convert(VALUE dummy, VALUE str) 01528 { 01529 return rb_check_string_type(str); 01530 } 01531 01532 static char* 01533 str_nth_len(const char *p, const char *e, long *nthp, rb_encoding *enc) 01534 { 01535 long nth = *nthp; 01536 if (rb_enc_mbmaxlen(enc) == 1) { 01537 p += nth; 01538 } 01539 else if (rb_enc_mbmaxlen(enc) == rb_enc_mbminlen(enc)) { 01540 p += nth * rb_enc_mbmaxlen(enc); 01541 } 01542 else if (rb_enc_asciicompat(enc)) { 01543 const char *p2, *e2; 01544 int n; 01545 01546 while (p < e && 0 < nth) { 01547 e2 = p + nth; 01548 if (e < e2) { 01549 *nthp = nth; 01550 return (char *)e; 01551 } 01552 if (ISASCII(*p)) { 01553 p2 = search_nonascii(p, e2); 01554 if (!p2) { 01555 nth -= e2 - p; 01556 *nthp = nth; 01557 return (char *)e2; 01558 } 01559 nth -= p2 - p; 01560 p = p2; 01561 } 01562 n = rb_enc_mbclen(p, e, enc); 01563 p += n; 01564 nth--; 01565 } 01566 *nthp = nth; 01567 if (nth != 0) { 01568 return (char *)e; 01569 } 01570 return (char *)p; 01571 } 01572 else { 01573 while (p < e && nth--) { 01574 p += rb_enc_mbclen(p, e, enc); 01575 } 01576 } 01577 if (p > e) p = e; 01578 *nthp = nth; 01579 return (char*)p; 01580 } 01581 01582 char* 01583 rb_enc_nth(const char *p, const char *e, long nth, rb_encoding *enc) 01584 { 01585 return str_nth_len(p, e, &nth, enc); 01586 } 01587 01588 static char* 01589 str_nth(const char *p, const char *e, long nth, rb_encoding *enc, int singlebyte) 01590 { 01591 if (singlebyte) 01592 p += nth; 01593 else { 01594 p = str_nth_len(p, e, &nth, enc); 01595 } 01596 if (!p) return 0; 01597 if (p > e) p = e; 01598 return (char *)p; 01599 } 01600 01601 /* char offset to byte offset */ 01602 static long 01603 str_offset(const char *p, const char *e, long nth, rb_encoding *enc, int singlebyte) 01604 { 01605 const char *pp = str_nth(p, e, nth, enc, singlebyte); 01606 if (!pp) return e - p; 01607 return pp - p; 01608 } 01609 01610 long 01611 rb_str_offset(VALUE str, long pos) 01612 { 01613 return str_offset(RSTRING_PTR(str), RSTRING_END(str), pos, 01614 STR_ENC_GET(str), single_byte_optimizable(str)); 01615 } 01616 01617 #ifdef NONASCII_MASK 01618 static char * 01619 str_utf8_nth(const char *p, const char *e, long *nthp) 01620 { 01621 long nth = *nthp; 01622 if ((int)SIZEOF_VALUE * 2 < e - p && (int)SIZEOF_VALUE * 2 < nth) { 01623 const VALUE *s, *t; 01624 const VALUE lowbits = sizeof(VALUE) - 1; 01625 s = (const VALUE*)(~lowbits & ((VALUE)p + lowbits)); 01626 t = (const VALUE*)(~lowbits & (VALUE)e); 01627 while (p < (const char *)s) { 01628 if (is_utf8_lead_byte(*p)) nth--; 01629 p++; 01630 } 01631 do { 01632 nth -= count_utf8_lead_bytes_with_word(s); 01633 s++; 01634 } while (s < t && (int)sizeof(VALUE) <= nth); 01635 p = (char *)s; 01636 } 01637 while (p < e) { 01638 if (is_utf8_lead_byte(*p)) { 01639 if (nth == 0) break; 01640 nth--; 01641 } 01642 p++; 01643 } 01644 *nthp = nth; 01645 return (char *)p; 01646 } 01647 01648 static long 01649 str_utf8_offset(const char *p, const char *e, long nth) 01650 { 01651 const char *pp = str_utf8_nth(p, e, &nth); 01652 return pp - p; 01653 } 01654 #endif 01655 01656 /* byte offset to char offset */ 01657 long 01658 rb_str_sublen(VALUE str, long pos) 01659 { 01660 if (single_byte_optimizable(str) || pos < 0) 01661 return pos; 01662 else { 01663 char *p = RSTRING_PTR(str); 01664 return enc_strlen(p, p + pos, STR_ENC_GET(str), ENC_CODERANGE(str)); 01665 } 01666 } 01667 01668 VALUE 01669 rb_str_subseq(VALUE str, long beg, long len) 01670 { 01671 VALUE str2; 01672 01673 if (RSTRING_LEN(str) == beg + len && 01674 RSTRING_EMBED_LEN_MAX < len) { 01675 str2 = rb_str_new_shared(rb_str_new_frozen(str)); 01676 rb_str_drop_bytes(str2, beg); 01677 } 01678 else { 01679 str2 = rb_str_new5(str, RSTRING_PTR(str)+beg, len); 01680 RB_GC_GUARD(str); 01681 } 01682 01683 rb_enc_cr_str_copy_for_substr(str2, str); 01684 OBJ_INFECT(str2, str); 01685 01686 return str2; 01687 } 01688 01689 static char * 01690 rb_str_subpos(VALUE str, long beg, long *lenp) 01691 { 01692 long len = *lenp; 01693 long slen = -1L; 01694 long blen = RSTRING_LEN(str); 01695 rb_encoding *enc = STR_ENC_GET(str); 01696 char *p, *s = RSTRING_PTR(str), *e = s + blen; 01697 01698 if (len < 0) return 0; 01699 if (!blen) { 01700 len = 0; 01701 } 01702 if (single_byte_optimizable(str)) { 01703 if (beg > blen) return 0; 01704 if (beg < 0) { 01705 beg += blen; 01706 if (beg < 0) return 0; 01707 } 01708 if (beg + len > blen) 01709 len = blen - beg; 01710 if (len < 0) return 0; 01711 p = s + beg; 01712 goto end; 01713 } 01714 if (beg < 0) { 01715 if (len > -beg) len = -beg; 01716 if (-beg * rb_enc_mbmaxlen(enc) < RSTRING_LEN(str) / 8) { 01717 beg = -beg; 01718 while (beg-- > len && (e = rb_enc_prev_char(s, e, e, enc)) != 0); 01719 p = e; 01720 if (!p) return 0; 01721 while (len-- > 0 && (p = rb_enc_prev_char(s, p, e, enc)) != 0); 01722 if (!p) return 0; 01723 len = e - p; 01724 goto end; 01725 } 01726 else { 01727 slen = str_strlen(str, enc); 01728 beg += slen; 01729 if (beg < 0) return 0; 01730 p = s + beg; 01731 if (len == 0) goto end; 01732 } 01733 } 01734 else if (beg > 0 && beg > RSTRING_LEN(str)) { 01735 return 0; 01736 } 01737 if (len == 0) { 01738 if (beg > str_strlen(str, enc)) return 0; 01739 p = s + beg; 01740 } 01741 #ifdef NONASCII_MASK 01742 else if (ENC_CODERANGE(str) == ENC_CODERANGE_VALID && 01743 enc == rb_utf8_encoding()) { 01744 p = str_utf8_nth(s, e, &beg); 01745 if (beg > 0) return 0; 01746 len = str_utf8_offset(p, e, len); 01747 } 01748 #endif 01749 else if (rb_enc_mbmaxlen(enc) == rb_enc_mbminlen(enc)) { 01750 int char_sz = rb_enc_mbmaxlen(enc); 01751 01752 p = s + beg * char_sz; 01753 if (p > e) { 01754 return 0; 01755 } 01756 else if (len * char_sz > e - p) 01757 len = e - p; 01758 else 01759 len *= char_sz; 01760 } 01761 else if ((p = str_nth_len(s, e, &beg, enc)) == e) { 01762 if (beg > 0) return 0; 01763 len = 0; 01764 } 01765 else { 01766 len = str_offset(p, e, len, enc, 0); 01767 } 01768 end: 01769 *lenp = len; 01770 RB_GC_GUARD(str); 01771 return p; 01772 } 01773 01774 VALUE 01775 rb_str_substr(VALUE str, long beg, long len) 01776 { 01777 VALUE str2; 01778 char *p = rb_str_subpos(str, beg, &len); 01779 01780 if (!p) return Qnil; 01781 if (len > RSTRING_EMBED_LEN_MAX && p + len == RSTRING_END(str)) { 01782 str2 = rb_str_new4(str); 01783 str2 = str_new3(rb_obj_class(str2), str2); 01784 RSTRING(str2)->as.heap.ptr += RSTRING(str2)->as.heap.len - len; 01785 RSTRING(str2)->as.heap.len = len; 01786 } 01787 else { 01788 str2 = rb_str_new5(str, p, len); 01789 OBJ_INFECT(str2, str); 01790 RB_GC_GUARD(str); 01791 } 01792 rb_enc_cr_str_copy_for_substr(str2, str); 01793 01794 return str2; 01795 } 01796 01797 VALUE 01798 rb_str_freeze(VALUE str) 01799 { 01800 if (STR_ASSOC_P(str)) { 01801 VALUE ary = RSTRING(str)->as.heap.aux.shared; 01802 OBJ_FREEZE(ary); 01803 } 01804 return rb_obj_freeze(str); 01805 } 01806 01807 RUBY_ALIAS_FUNCTION(rb_str_dup_frozen(VALUE str), rb_str_new_frozen, (str)) 01808 #define rb_str_dup_frozen rb_str_new_frozen 01809 01810 VALUE 01811 rb_str_locktmp(VALUE str) 01812 { 01813 if (FL_TEST(str, STR_TMPLOCK)) { 01814 rb_raise(rb_eRuntimeError, "temporal locking already locked string"); 01815 } 01816 FL_SET(str, STR_TMPLOCK); 01817 return str; 01818 } 01819 01820 VALUE 01821 rb_str_unlocktmp(VALUE str) 01822 { 01823 if (!FL_TEST(str, STR_TMPLOCK)) { 01824 rb_raise(rb_eRuntimeError, "temporal unlocking already unlocked string"); 01825 } 01826 FL_UNSET(str, STR_TMPLOCK); 01827 return str; 01828 } 01829 01830 VALUE 01831 rb_str_locktmp_ensure(VALUE str, VALUE (*func)(VALUE), VALUE arg) 01832 { 01833 rb_str_locktmp(str); 01834 return rb_ensure(func, arg, rb_str_unlocktmp, str); 01835 } 01836 01837 void 01838 rb_str_set_len(VALUE str, long len) 01839 { 01840 long capa; 01841 01842 str_modifiable(str); 01843 if (STR_SHARED_P(str)) { 01844 rb_raise(rb_eRuntimeError, "can't set length of shared string"); 01845 } 01846 if (len > (capa = (long)rb_str_capacity(str))) { 01847 rb_bug("probable buffer overflow: %ld for %ld", len, capa); 01848 } 01849 STR_SET_LEN(str, len); 01850 RSTRING_PTR(str)[len] = '\0'; 01851 } 01852 01853 VALUE 01854 rb_str_resize(VALUE str, long len) 01855 { 01856 long slen; 01857 int independent; 01858 01859 if (len < 0) { 01860 rb_raise(rb_eArgError, "negative string size (or size too big)"); 01861 } 01862 01863 independent = str_independent(str); 01864 ENC_CODERANGE_CLEAR(str); 01865 slen = RSTRING_LEN(str); 01866 { 01867 long capa; 01868 if (STR_EMBED_P(str)) { 01869 if (len == slen) return str; 01870 if (len + 1 <= RSTRING_EMBED_LEN_MAX + 1) { 01871 STR_SET_EMBED_LEN(str, len); 01872 RSTRING(str)->as.ary[len] = '\0'; 01873 return str; 01874 } 01875 str_make_independent_expand(str, len - slen); 01876 STR_SET_NOEMBED(str); 01877 } 01878 else if (len <= RSTRING_EMBED_LEN_MAX) { 01879 char *ptr = RSTRING(str)->as.heap.ptr; 01880 STR_SET_EMBED(str); 01881 if (slen > len) slen = len; 01882 if (slen > 0) MEMCPY(RSTRING(str)->as.ary, ptr, char, slen); 01883 RSTRING(str)->as.ary[len] = '\0'; 01884 STR_SET_EMBED_LEN(str, len); 01885 if (independent) xfree(ptr); 01886 return str; 01887 } 01888 else if (!independent) { 01889 if (len == slen) return str; 01890 str_make_independent_expand(str, len - slen); 01891 } 01892 else if ((capa = RSTRING(str)->as.heap.aux.capa) < len || 01893 (capa - len) > (len < 1024 ? len : 1024)) { 01894 REALLOC_N(RSTRING(str)->as.heap.ptr, char, len+1); 01895 RSTRING(str)->as.heap.aux.capa = len; 01896 } 01897 else if (len == slen) return str; 01898 RSTRING(str)->as.heap.len = len; 01899 RSTRING(str)->as.heap.ptr[len] = '\0'; /* sentinel */ 01900 } 01901 return str; 01902 } 01903 01904 static VALUE 01905 str_buf_cat(VALUE str, const char *ptr, long len) 01906 { 01907 long capa, total, off = -1; 01908 01909 if (ptr >= RSTRING_PTR(str) && ptr <= RSTRING_END(str)) { 01910 off = ptr - RSTRING_PTR(str); 01911 } 01912 rb_str_modify(str); 01913 if (len == 0) return 0; 01914 if (STR_ASSOC_P(str)) { 01915 FL_UNSET(str, STR_ASSOC); 01916 capa = RSTRING(str)->as.heap.aux.capa = RSTRING_LEN(str); 01917 } 01918 else if (STR_EMBED_P(str)) { 01919 capa = RSTRING_EMBED_LEN_MAX; 01920 } 01921 else { 01922 capa = RSTRING(str)->as.heap.aux.capa; 01923 } 01924 if (RSTRING_LEN(str) >= LONG_MAX - len) { 01925 rb_raise(rb_eArgError, "string sizes too big"); 01926 } 01927 total = RSTRING_LEN(str)+len; 01928 if (capa <= total) { 01929 while (total > capa) { 01930 if (capa + 1 >= LONG_MAX / 2) { 01931 capa = (total + 4095) / 4096 * 4096; 01932 break; 01933 } 01934 capa = (capa + 1) * 2; 01935 } 01936 RESIZE_CAPA(str, capa); 01937 } 01938 if (off != -1) { 01939 ptr = RSTRING_PTR(str) + off; 01940 } 01941 memcpy(RSTRING_PTR(str) + RSTRING_LEN(str), ptr, len); 01942 STR_SET_LEN(str, total); 01943 RSTRING_PTR(str)[total] = '\0'; /* sentinel */ 01944 01945 return str; 01946 } 01947 01948 #define str_buf_cat2(str, ptr) str_buf_cat((str), (ptr), strlen(ptr)) 01949 01950 VALUE 01951 rb_str_buf_cat(VALUE str, const char *ptr, long len) 01952 { 01953 if (len == 0) return str; 01954 if (len < 0) { 01955 rb_raise(rb_eArgError, "negative string size (or size too big)"); 01956 } 01957 return str_buf_cat(str, ptr, len); 01958 } 01959 01960 VALUE 01961 rb_str_buf_cat2(VALUE str, const char *ptr) 01962 { 01963 return rb_str_buf_cat(str, ptr, strlen(ptr)); 01964 } 01965 01966 VALUE 01967 rb_str_cat(VALUE str, const char *ptr, long len) 01968 { 01969 if (len < 0) { 01970 rb_raise(rb_eArgError, "negative string size (or size too big)"); 01971 } 01972 if (STR_ASSOC_P(str)) { 01973 char *p; 01974 rb_str_modify_expand(str, len); 01975 p = RSTRING(str)->as.heap.ptr; 01976 memcpy(p + RSTRING(str)->as.heap.len, ptr, len); 01977 len = RSTRING(str)->as.heap.len += len; 01978 p[len] = '\0'; /* sentinel */ 01979 return str; 01980 } 01981 01982 return rb_str_buf_cat(str, ptr, len); 01983 } 01984 01985 VALUE 01986 rb_str_cat2(VALUE str, const char *ptr) 01987 { 01988 return rb_str_cat(str, ptr, strlen(ptr)); 01989 } 01990 01991 static VALUE 01992 rb_enc_cr_str_buf_cat(VALUE str, const char *ptr, long len, 01993 int ptr_encindex, int ptr_cr, int *ptr_cr_ret) 01994 { 01995 int str_encindex = ENCODING_GET(str); 01996 int res_encindex; 01997 int str_cr, res_cr; 01998 01999 str_cr = ENC_CODERANGE(str); 02000 02001 if (str_encindex == ptr_encindex) { 02002 if (str_cr == ENC_CODERANGE_UNKNOWN) 02003 ptr_cr = ENC_CODERANGE_UNKNOWN; 02004 else if (ptr_cr == ENC_CODERANGE_UNKNOWN) { 02005 ptr_cr = coderange_scan(ptr, len, rb_enc_from_index(ptr_encindex)); 02006 } 02007 } 02008 else { 02009 rb_encoding *str_enc = rb_enc_from_index(str_encindex); 02010 rb_encoding *ptr_enc = rb_enc_from_index(ptr_encindex); 02011 if (!rb_enc_asciicompat(str_enc) || !rb_enc_asciicompat(ptr_enc)) { 02012 if (len == 0) 02013 return str; 02014 if (RSTRING_LEN(str) == 0) { 02015 rb_str_buf_cat(str, ptr, len); 02016 ENCODING_CODERANGE_SET(str, ptr_encindex, ptr_cr); 02017 return str; 02018 } 02019 goto incompatible; 02020 } 02021 if (ptr_cr == ENC_CODERANGE_UNKNOWN) { 02022 ptr_cr = coderange_scan(ptr, len, ptr_enc); 02023 } 02024 if (str_cr == ENC_CODERANGE_UNKNOWN) { 02025 if (ENCODING_IS_ASCII8BIT(str) || ptr_cr != ENC_CODERANGE_7BIT) { 02026 str_cr = rb_enc_str_coderange(str); 02027 } 02028 } 02029 } 02030 if (ptr_cr_ret) 02031 *ptr_cr_ret = ptr_cr; 02032 02033 if (str_encindex != ptr_encindex && 02034 str_cr != ENC_CODERANGE_7BIT && 02035 ptr_cr != ENC_CODERANGE_7BIT) { 02036 incompatible: 02037 rb_raise(rb_eEncCompatError, "incompatible character encodings: %s and %s", 02038 rb_enc_name(rb_enc_from_index(str_encindex)), 02039 rb_enc_name(rb_enc_from_index(ptr_encindex))); 02040 } 02041 02042 if (str_cr == ENC_CODERANGE_UNKNOWN) { 02043 res_encindex = str_encindex; 02044 res_cr = ENC_CODERANGE_UNKNOWN; 02045 } 02046 else if (str_cr == ENC_CODERANGE_7BIT) { 02047 if (ptr_cr == ENC_CODERANGE_7BIT) { 02048 res_encindex = str_encindex; 02049 res_cr = ENC_CODERANGE_7BIT; 02050 } 02051 else { 02052 res_encindex = ptr_encindex; 02053 res_cr = ptr_cr; 02054 } 02055 } 02056 else if (str_cr == ENC_CODERANGE_VALID) { 02057 res_encindex = str_encindex; 02058 if (ptr_cr == ENC_CODERANGE_7BIT || ptr_cr == ENC_CODERANGE_VALID) 02059 res_cr = str_cr; 02060 else 02061 res_cr = ptr_cr; 02062 } 02063 else { /* str_cr == ENC_CODERANGE_BROKEN */ 02064 res_encindex = str_encindex; 02065 res_cr = str_cr; 02066 if (0 < len) res_cr = ENC_CODERANGE_UNKNOWN; 02067 } 02068 02069 if (len < 0) { 02070 rb_raise(rb_eArgError, "negative string size (or size too big)"); 02071 } 02072 str_buf_cat(str, ptr, len); 02073 ENCODING_CODERANGE_SET(str, res_encindex, res_cr); 02074 return str; 02075 } 02076 02077 VALUE 02078 rb_enc_str_buf_cat(VALUE str, const char *ptr, long len, rb_encoding *ptr_enc) 02079 { 02080 return rb_enc_cr_str_buf_cat(str, ptr, len, 02081 rb_enc_to_index(ptr_enc), ENC_CODERANGE_UNKNOWN, NULL); 02082 } 02083 02084 VALUE 02085 rb_str_buf_cat_ascii(VALUE str, const char *ptr) 02086 { 02087 /* ptr must reference NUL terminated ASCII string. */ 02088 int encindex = ENCODING_GET(str); 02089 rb_encoding *enc = rb_enc_from_index(encindex); 02090 if (rb_enc_asciicompat(enc)) { 02091 return rb_enc_cr_str_buf_cat(str, ptr, strlen(ptr), 02092 encindex, ENC_CODERANGE_7BIT, 0); 02093 } 02094 else { 02095 char *buf = ALLOCA_N(char, rb_enc_mbmaxlen(enc)); 02096 while (*ptr) { 02097 unsigned int c = (unsigned char)*ptr; 02098 int len = rb_enc_codelen(c, enc); 02099 rb_enc_mbcput(c, buf, enc); 02100 rb_enc_cr_str_buf_cat(str, buf, len, 02101 encindex, ENC_CODERANGE_VALID, 0); 02102 ptr++; 02103 } 02104 return str; 02105 } 02106 } 02107 02108 VALUE 02109 rb_str_buf_append(VALUE str, VALUE str2) 02110 { 02111 int str2_cr; 02112 02113 str2_cr = ENC_CODERANGE(str2); 02114 02115 rb_enc_cr_str_buf_cat(str, RSTRING_PTR(str2), RSTRING_LEN(str2), 02116 ENCODING_GET(str2), str2_cr, &str2_cr); 02117 02118 OBJ_INFECT(str, str2); 02119 ENC_CODERANGE_SET(str2, str2_cr); 02120 02121 return str; 02122 } 02123 02124 VALUE 02125 rb_str_append(VALUE str, VALUE str2) 02126 { 02127 rb_encoding *enc; 02128 int cr, cr2; 02129 long len2; 02130 02131 StringValue(str2); 02132 if ((len2 = RSTRING_LEN(str2)) > 0 && STR_ASSOC_P(str)) { 02133 long len = RSTRING_LEN(str) + len2; 02134 enc = rb_enc_check(str, str2); 02135 cr = ENC_CODERANGE(str); 02136 if ((cr2 = ENC_CODERANGE(str2)) > cr) cr = cr2; 02137 rb_str_modify_expand(str, len2); 02138 memcpy(RSTRING(str)->as.heap.ptr + RSTRING(str)->as.heap.len, 02139 RSTRING_PTR(str2), len2+1); 02140 RSTRING(str)->as.heap.len = len; 02141 rb_enc_associate(str, enc); 02142 ENC_CODERANGE_SET(str, cr); 02143 OBJ_INFECT(str, str2); 02144 return str; 02145 } 02146 return rb_str_buf_append(str, str2); 02147 } 02148 02149 /* 02150 * call-seq: 02151 * str << integer -> str 02152 * str.concat(integer) -> str 02153 * str << obj -> str 02154 * str.concat(obj) -> str 02155 * 02156 * Append---Concatenates the given object to <i>str</i>. If the object is a 02157 * <code>Integer</code>, it is considered as a codepoint, and is converted 02158 * to a character before concatenation. 02159 * 02160 * a = "hello " 02161 * a << "world" #=> "hello world" 02162 * a.concat(33) #=> "hello world!" 02163 */ 02164 02165 VALUE 02166 rb_str_concat(VALUE str1, VALUE str2) 02167 { 02168 unsigned int code; 02169 rb_encoding *enc = STR_ENC_GET(str1); 02170 02171 if (FIXNUM_P(str2) || RB_TYPE_P(str2, T_BIGNUM)) { 02172 if (rb_num_to_uint(str2, &code) == 0) { 02173 } 02174 else if (FIXNUM_P(str2)) { 02175 rb_raise(rb_eRangeError, "%ld out of char range", FIX2LONG(str2)); 02176 } 02177 else { 02178 rb_raise(rb_eRangeError, "bignum out of char range"); 02179 } 02180 } 02181 else { 02182 return rb_str_append(str1, str2); 02183 } 02184 02185 if (enc == rb_usascii_encoding()) { 02186 /* US-ASCII automatically extended to ASCII-8BIT */ 02187 char buf[1]; 02188 buf[0] = (char)code; 02189 if (code > 0xFF) { 02190 rb_raise(rb_eRangeError, "%u out of char range", code); 02191 } 02192 rb_str_cat(str1, buf, 1); 02193 if (code > 127) { 02194 rb_enc_associate(str1, rb_ascii8bit_encoding()); 02195 ENC_CODERANGE_SET(str1, ENC_CODERANGE_VALID); 02196 } 02197 } 02198 else { 02199 long pos = RSTRING_LEN(str1); 02200 int cr = ENC_CODERANGE(str1); 02201 int len; 02202 char *buf; 02203 02204 switch (len = rb_enc_codelen(code, enc)) { 02205 case ONIGERR_INVALID_CODE_POINT_VALUE: 02206 rb_raise(rb_eRangeError, "invalid codepoint 0x%X in %s", code, rb_enc_name(enc)); 02207 break; 02208 case ONIGERR_TOO_BIG_WIDE_CHAR_VALUE: 02209 case 0: 02210 rb_raise(rb_eRangeError, "%u out of char range", code); 02211 break; 02212 } 02213 buf = ALLOCA_N(char, len + 1); 02214 rb_enc_mbcput(code, buf, enc); 02215 if (rb_enc_precise_mbclen(buf, buf + len + 1, enc) != len) { 02216 rb_raise(rb_eRangeError, "invalid codepoint 0x%X in %s", code, rb_enc_name(enc)); 02217 } 02218 rb_str_resize(str1, pos+len); 02219 memcpy(RSTRING_PTR(str1) + pos, buf, len); 02220 if (cr == ENC_CODERANGE_7BIT && code > 127) 02221 cr = ENC_CODERANGE_VALID; 02222 ENC_CODERANGE_SET(str1, cr); 02223 } 02224 return str1; 02225 } 02226 02227 /* 02228 * call-seq: 02229 * str.prepend(other_str) -> str 02230 * 02231 * Prepend---Prepend the given string to <i>str</i>. 02232 * 02233 * a = "world" 02234 * a.prepend("hello ") #=> "hello world" 02235 * a #=> "hello world" 02236 */ 02237 02238 static VALUE 02239 rb_str_prepend(VALUE str, VALUE str2) 02240 { 02241 StringValue(str2); 02242 StringValue(str); 02243 rb_str_update(str, 0L, 0L, str2); 02244 return str; 02245 } 02246 02247 st_index_t 02248 rb_str_hash(VALUE str) 02249 { 02250 int e = ENCODING_GET(str); 02251 if (e && rb_enc_str_coderange(str) == ENC_CODERANGE_7BIT) { 02252 e = 0; 02253 } 02254 return rb_memhash((const void *)RSTRING_PTR(str), RSTRING_LEN(str)) ^ e; 02255 } 02256 02257 int 02258 rb_str_hash_cmp(VALUE str1, VALUE str2) 02259 { 02260 long len; 02261 02262 if (!rb_str_comparable(str1, str2)) return 1; 02263 if (RSTRING_LEN(str1) == (len = RSTRING_LEN(str2)) && 02264 memcmp(RSTRING_PTR(str1), RSTRING_PTR(str2), len) == 0) { 02265 return 0; 02266 } 02267 return 1; 02268 } 02269 02270 /* 02271 * call-seq: 02272 * str.hash -> fixnum 02273 * 02274 * Return a hash based on the string's length and content. 02275 */ 02276 02277 static VALUE 02278 rb_str_hash_m(VALUE str) 02279 { 02280 st_index_t hval = rb_str_hash(str); 02281 return INT2FIX(hval); 02282 } 02283 02284 #define lesser(a,b) (((a)>(b))?(b):(a)) 02285 02286 int 02287 rb_str_comparable(VALUE str1, VALUE str2) 02288 { 02289 int idx1, idx2; 02290 int rc1, rc2; 02291 02292 if (RSTRING_LEN(str1) == 0) return TRUE; 02293 if (RSTRING_LEN(str2) == 0) return TRUE; 02294 idx1 = ENCODING_GET(str1); 02295 idx2 = ENCODING_GET(str2); 02296 if (idx1 == idx2) return TRUE; 02297 rc1 = rb_enc_str_coderange(str1); 02298 rc2 = rb_enc_str_coderange(str2); 02299 if (rc1 == ENC_CODERANGE_7BIT) { 02300 if (rc2 == ENC_CODERANGE_7BIT) return TRUE; 02301 if (rb_enc_asciicompat(rb_enc_from_index(idx2))) 02302 return TRUE; 02303 } 02304 if (rc2 == ENC_CODERANGE_7BIT) { 02305 if (rb_enc_asciicompat(rb_enc_from_index(idx1))) 02306 return TRUE; 02307 } 02308 return FALSE; 02309 } 02310 02311 int 02312 rb_str_cmp(VALUE str1, VALUE str2) 02313 { 02314 long len1, len2; 02315 const char *ptr1, *ptr2; 02316 int retval; 02317 02318 if (str1 == str2) return 0; 02319 RSTRING_GETMEM(str1, ptr1, len1); 02320 RSTRING_GETMEM(str2, ptr2, len2); 02321 if (ptr1 == ptr2 || (retval = memcmp(ptr1, ptr2, lesser(len1, len2))) == 0) { 02322 if (len1 == len2) { 02323 if (!rb_str_comparable(str1, str2)) { 02324 if (ENCODING_GET(str1) > ENCODING_GET(str2)) 02325 return 1; 02326 return -1; 02327 } 02328 return 0; 02329 } 02330 if (len1 > len2) return 1; 02331 return -1; 02332 } 02333 if (retval > 0) return 1; 02334 return -1; 02335 } 02336 02337 /* expect tail call optimization */ 02338 static VALUE 02339 str_eql(const VALUE str1, const VALUE str2) 02340 { 02341 const long len = RSTRING_LEN(str1); 02342 const char *ptr1, *ptr2; 02343 02344 if (len != RSTRING_LEN(str2)) return Qfalse; 02345 if (!rb_str_comparable(str1, str2)) return Qfalse; 02346 if ((ptr1 = RSTRING_PTR(str1)) == (ptr2 = RSTRING_PTR(str2))) 02347 return Qtrue; 02348 if (memcmp(ptr1, ptr2, len) == 0) 02349 return Qtrue; 02350 return Qfalse; 02351 } 02352 02353 /* 02354 * call-seq: 02355 * str == obj -> true or false 02356 * 02357 * Equality---If <i>obj</i> is not a <code>String</code>, returns 02358 * <code>false</code>. Otherwise, returns <code>true</code> if <i>str</i> 02359 * <code><=></code> <i>obj</i> returns zero. 02360 */ 02361 02362 VALUE 02363 rb_str_equal(VALUE str1, VALUE str2) 02364 { 02365 if (str1 == str2) return Qtrue; 02366 if (!RB_TYPE_P(str2, T_STRING)) { 02367 if (!rb_respond_to(str2, rb_intern("to_str"))) { 02368 return Qfalse; 02369 } 02370 return rb_equal(str2, str1); 02371 } 02372 return str_eql(str1, str2); 02373 } 02374 02375 /* 02376 * call-seq: 02377 * str.eql?(other) -> true or false 02378 * 02379 * Two strings are equal if they have the same length and content. 02380 */ 02381 02382 static VALUE 02383 rb_str_eql(VALUE str1, VALUE str2) 02384 { 02385 if (str1 == str2) return Qtrue; 02386 if (!RB_TYPE_P(str2, T_STRING)) return Qfalse; 02387 return str_eql(str1, str2); 02388 } 02389 02390 /* 02391 * call-seq: 02392 * string <=> other_string -> -1, 0, +1 or nil 02393 * 02394 * 02395 * Comparison---Returns -1, 0, +1 or nil depending on whether +string+ is less 02396 * than, equal to, or greater than +other_string+. 02397 * 02398 * +nil+ is returned if the two values are incomparable. 02399 * 02400 * If the strings are of different lengths, and the strings are equal when 02401 * compared up to the shortest length, then the longer string is considered 02402 * greater than the shorter one. 02403 * 02404 * <code><=></code> is the basis for the methods <code><</code>, 02405 * <code><=</code>, <code>></code>, <code>>=</code>, and 02406 * <code>between?</code>, included from module Comparable. The method 02407 * String#== does not use Comparable#==. 02408 * 02409 * "abcdef" <=> "abcde" #=> 1 02410 * "abcdef" <=> "abcdef" #=> 0 02411 * "abcdef" <=> "abcdefg" #=> -1 02412 * "abcdef" <=> "ABCDEF" #=> 1 02413 */ 02414 02415 static VALUE 02416 rb_str_cmp_m(VALUE str1, VALUE str2) 02417 { 02418 int result; 02419 02420 if (!RB_TYPE_P(str2, T_STRING)) { 02421 VALUE tmp = rb_check_funcall(str2, rb_intern("to_str"), 0, 0); 02422 if (RB_TYPE_P(tmp, T_STRING)) { 02423 result = rb_str_cmp(str1, tmp); 02424 } 02425 else { 02426 return rb_invcmp(str1, str2); 02427 } 02428 } 02429 else { 02430 result = rb_str_cmp(str1, str2); 02431 } 02432 return INT2FIX(result); 02433 } 02434 02435 /* 02436 * call-seq: 02437 * str.casecmp(other_str) -> -1, 0, +1 or nil 02438 * 02439 * Case-insensitive version of <code>String#<=></code>. 02440 * 02441 * "abcdef".casecmp("abcde") #=> 1 02442 * "aBcDeF".casecmp("abcdef") #=> 0 02443 * "abcdef".casecmp("abcdefg") #=> -1 02444 * "abcdef".casecmp("ABCDEF") #=> 0 02445 */ 02446 02447 static VALUE 02448 rb_str_casecmp(VALUE str1, VALUE str2) 02449 { 02450 long len; 02451 rb_encoding *enc; 02452 char *p1, *p1end, *p2, *p2end; 02453 02454 StringValue(str2); 02455 enc = rb_enc_compatible(str1, str2); 02456 if (!enc) { 02457 return Qnil; 02458 } 02459 02460 p1 = RSTRING_PTR(str1); p1end = RSTRING_END(str1); 02461 p2 = RSTRING_PTR(str2); p2end = RSTRING_END(str2); 02462 if (single_byte_optimizable(str1) && single_byte_optimizable(str2)) { 02463 while (p1 < p1end && p2 < p2end) { 02464 if (*p1 != *p2) { 02465 unsigned int c1 = TOUPPER(*p1 & 0xff); 02466 unsigned int c2 = TOUPPER(*p2 & 0xff); 02467 if (c1 != c2) 02468 return INT2FIX(c1 < c2 ? -1 : 1); 02469 } 02470 p1++; 02471 p2++; 02472 } 02473 } 02474 else { 02475 while (p1 < p1end && p2 < p2end) { 02476 int l1, c1 = rb_enc_ascget(p1, p1end, &l1, enc); 02477 int l2, c2 = rb_enc_ascget(p2, p2end, &l2, enc); 02478 02479 if (0 <= c1 && 0 <= c2) { 02480 c1 = TOUPPER(c1); 02481 c2 = TOUPPER(c2); 02482 if (c1 != c2) 02483 return INT2FIX(c1 < c2 ? -1 : 1); 02484 } 02485 else { 02486 int r; 02487 l1 = rb_enc_mbclen(p1, p1end, enc); 02488 l2 = rb_enc_mbclen(p2, p2end, enc); 02489 len = l1 < l2 ? l1 : l2; 02490 r = memcmp(p1, p2, len); 02491 if (r != 0) 02492 return INT2FIX(r < 0 ? -1 : 1); 02493 if (l1 != l2) 02494 return INT2FIX(l1 < l2 ? -1 : 1); 02495 } 02496 p1 += l1; 02497 p2 += l2; 02498 } 02499 } 02500 if (RSTRING_LEN(str1) == RSTRING_LEN(str2)) return INT2FIX(0); 02501 if (RSTRING_LEN(str1) > RSTRING_LEN(str2)) return INT2FIX(1); 02502 return INT2FIX(-1); 02503 } 02504 02505 static long 02506 rb_str_index(VALUE str, VALUE sub, long offset) 02507 { 02508 long pos; 02509 char *s, *sptr, *e; 02510 long len, slen; 02511 rb_encoding *enc; 02512 02513 enc = rb_enc_check(str, sub); 02514 if (is_broken_string(sub)) { 02515 return -1; 02516 } 02517 len = str_strlen(str, enc); 02518 slen = str_strlen(sub, enc); 02519 if (offset < 0) { 02520 offset += len; 02521 if (offset < 0) return -1; 02522 } 02523 if (len - offset < slen) return -1; 02524 s = RSTRING_PTR(str); 02525 e = s + RSTRING_LEN(str); 02526 if (offset) { 02527 offset = str_offset(s, RSTRING_END(str), offset, enc, single_byte_optimizable(str)); 02528 s += offset; 02529 } 02530 if (slen == 0) return offset; 02531 /* need proceed one character at a time */ 02532 sptr = RSTRING_PTR(sub); 02533 slen = RSTRING_LEN(sub); 02534 len = RSTRING_LEN(str) - offset; 02535 for (;;) { 02536 char *t; 02537 pos = rb_memsearch(sptr, slen, s, len, enc); 02538 if (pos < 0) return pos; 02539 t = rb_enc_right_char_head(s, s+pos, e, enc); 02540 if (t == s + pos) break; 02541 if ((len -= t - s) <= 0) return -1; 02542 offset += t - s; 02543 s = t; 02544 } 02545 return pos + offset; 02546 } 02547 02548 02549 /* 02550 * call-seq: 02551 * str.index(substring [, offset]) -> fixnum or nil 02552 * str.index(regexp [, offset]) -> fixnum or nil 02553 * 02554 * Returns the index of the first occurrence of the given <i>substring</i> or 02555 * pattern (<i>regexp</i>) in <i>str</i>. Returns <code>nil</code> if not 02556 * found. If the second parameter is present, it specifies the position in the 02557 * string to begin the search. 02558 * 02559 * "hello".index('e') #=> 1 02560 * "hello".index('lo') #=> 3 02561 * "hello".index('a') #=> nil 02562 * "hello".index(?e) #=> 1 02563 * "hello".index(/[aeiou]/, -3) #=> 4 02564 */ 02565 02566 static VALUE 02567 rb_str_index_m(int argc, VALUE *argv, VALUE str) 02568 { 02569 VALUE sub; 02570 VALUE initpos; 02571 long pos; 02572 02573 if (rb_scan_args(argc, argv, "11", &sub, &initpos) == 2) { 02574 pos = NUM2LONG(initpos); 02575 } 02576 else { 02577 pos = 0; 02578 } 02579 if (pos < 0) { 02580 pos += str_strlen(str, STR_ENC_GET(str)); 02581 if (pos < 0) { 02582 if (RB_TYPE_P(sub, T_REGEXP)) { 02583 rb_backref_set(Qnil); 02584 } 02585 return Qnil; 02586 } 02587 } 02588 02589 if (SPECIAL_CONST_P(sub)) goto generic; 02590 switch (BUILTIN_TYPE(sub)) { 02591 case T_REGEXP: 02592 if (pos > str_strlen(str, STR_ENC_GET(str))) 02593 return Qnil; 02594 pos = str_offset(RSTRING_PTR(str), RSTRING_END(str), pos, 02595 rb_enc_check(str, sub), single_byte_optimizable(str)); 02596 02597 pos = rb_reg_search(sub, str, pos, 0); 02598 pos = rb_str_sublen(str, pos); 02599 break; 02600 02601 generic: 02602 default: { 02603 VALUE tmp; 02604 02605 tmp = rb_check_string_type(sub); 02606 if (NIL_P(tmp)) { 02607 rb_raise(rb_eTypeError, "type mismatch: %s given", 02608 rb_obj_classname(sub)); 02609 } 02610 sub = tmp; 02611 } 02612 /* fall through */ 02613 case T_STRING: 02614 pos = rb_str_index(str, sub, pos); 02615 pos = rb_str_sublen(str, pos); 02616 break; 02617 } 02618 02619 if (pos == -1) return Qnil; 02620 return LONG2NUM(pos); 02621 } 02622 02623 static long 02624 rb_str_rindex(VALUE str, VALUE sub, long pos) 02625 { 02626 long len, slen; 02627 char *s, *sbeg, *e, *t; 02628 rb_encoding *enc; 02629 int singlebyte = single_byte_optimizable(str); 02630 02631 enc = rb_enc_check(str, sub); 02632 if (is_broken_string(sub)) { 02633 return -1; 02634 } 02635 len = str_strlen(str, enc); 02636 slen = str_strlen(sub, enc); 02637 /* substring longer than string */ 02638 if (len < slen) return -1; 02639 if (len - pos < slen) { 02640 pos = len - slen; 02641 } 02642 if (len == 0) { 02643 return pos; 02644 } 02645 sbeg = RSTRING_PTR(str); 02646 e = RSTRING_END(str); 02647 t = RSTRING_PTR(sub); 02648 slen = RSTRING_LEN(sub); 02649 s = str_nth(sbeg, e, pos, enc, singlebyte); 02650 while (s) { 02651 if (memcmp(s, t, slen) == 0) { 02652 return pos; 02653 } 02654 if (pos == 0) break; 02655 pos--; 02656 s = rb_enc_prev_char(sbeg, s, e, enc); 02657 } 02658 return -1; 02659 } 02660 02661 02662 /* 02663 * call-seq: 02664 * str.rindex(substring [, fixnum]) -> fixnum or nil 02665 * str.rindex(regexp [, fixnum]) -> fixnum or nil 02666 * 02667 * Returns the index of the last occurrence of the given <i>substring</i> or 02668 * pattern (<i>regexp</i>) in <i>str</i>. Returns <code>nil</code> if not 02669 * found. If the second parameter is present, it specifies the position in the 02670 * string to end the search---characters beyond this point will not be 02671 * considered. 02672 * 02673 * "hello".rindex('e') #=> 1 02674 * "hello".rindex('l') #=> 3 02675 * "hello".rindex('a') #=> nil 02676 * "hello".rindex(?e) #=> 1 02677 * "hello".rindex(/[aeiou]/, -2) #=> 1 02678 */ 02679 02680 static VALUE 02681 rb_str_rindex_m(int argc, VALUE *argv, VALUE str) 02682 { 02683 VALUE sub; 02684 VALUE vpos; 02685 rb_encoding *enc = STR_ENC_GET(str); 02686 long pos, len = str_strlen(str, enc); 02687 02688 if (rb_scan_args(argc, argv, "11", &sub, &vpos) == 2) { 02689 pos = NUM2LONG(vpos); 02690 if (pos < 0) { 02691 pos += len; 02692 if (pos < 0) { 02693 if (RB_TYPE_P(sub, T_REGEXP)) { 02694 rb_backref_set(Qnil); 02695 } 02696 return Qnil; 02697 } 02698 } 02699 if (pos > len) pos = len; 02700 } 02701 else { 02702 pos = len; 02703 } 02704 02705 if (SPECIAL_CONST_P(sub)) goto generic; 02706 switch (BUILTIN_TYPE(sub)) { 02707 case T_REGEXP: 02708 /* enc = rb_get_check(str, sub); */ 02709 pos = str_offset(RSTRING_PTR(str), RSTRING_END(str), pos, 02710 STR_ENC_GET(str), single_byte_optimizable(str)); 02711 02712 if (!RREGEXP(sub)->ptr || RREGEXP_SRC_LEN(sub)) { 02713 pos = rb_reg_search(sub, str, pos, 1); 02714 pos = rb_str_sublen(str, pos); 02715 } 02716 if (pos >= 0) return LONG2NUM(pos); 02717 break; 02718 02719 generic: 02720 default: { 02721 VALUE tmp; 02722 02723 tmp = rb_check_string_type(sub); 02724 if (NIL_P(tmp)) { 02725 rb_raise(rb_eTypeError, "type mismatch: %s given", 02726 rb_obj_classname(sub)); 02727 } 02728 sub = tmp; 02729 } 02730 /* fall through */ 02731 case T_STRING: 02732 pos = rb_str_rindex(str, sub, pos); 02733 if (pos >= 0) return LONG2NUM(pos); 02734 break; 02735 } 02736 return Qnil; 02737 } 02738 02739 /* 02740 * call-seq: 02741 * str =~ obj -> fixnum or nil 02742 * 02743 * Match---If <i>obj</i> is a <code>Regexp</code>, use it as a pattern to match 02744 * against <i>str</i>,and returns the position the match starts, or 02745 * <code>nil</code> if there is no match. Otherwise, invokes 02746 * <i>obj.=~</i>, passing <i>str</i> as an argument. The default 02747 * <code>=~</code> in <code>Object</code> returns <code>nil</code>. 02748 * 02749 * Note: <code>str =~ regexp</code> is not the same as 02750 * <code>regexp =~ str</code>. Strings captured from named capture groups 02751 * are assigned to local variables only in the second case. 02752 * 02753 * "cat o' 9 tails" =~ /\d/ #=> 7 02754 * "cat o' 9 tails" =~ 9 #=> nil 02755 */ 02756 02757 static VALUE 02758 rb_str_match(VALUE x, VALUE y) 02759 { 02760 if (SPECIAL_CONST_P(y)) goto generic; 02761 switch (BUILTIN_TYPE(y)) { 02762 case T_STRING: 02763 rb_raise(rb_eTypeError, "type mismatch: String given"); 02764 02765 case T_REGEXP: 02766 return rb_reg_match(y, x); 02767 02768 generic: 02769 default: 02770 return rb_funcall(y, rb_intern("=~"), 1, x); 02771 } 02772 } 02773 02774 02775 static VALUE get_pat(VALUE, int); 02776 02777 02778 /* 02779 * call-seq: 02780 * str.match(pattern) -> matchdata or nil 02781 * str.match(pattern, pos) -> matchdata or nil 02782 * 02783 * Converts <i>pattern</i> to a <code>Regexp</code> (if it isn't already one), 02784 * then invokes its <code>match</code> method on <i>str</i>. If the second 02785 * parameter is present, it specifies the position in the string to begin the 02786 * search. 02787 * 02788 * 'hello'.match('(.)\1') #=> #<MatchData "ll" 1:"l"> 02789 * 'hello'.match('(.)\1')[0] #=> "ll" 02790 * 'hello'.match(/(.)\1/)[0] #=> "ll" 02791 * 'hello'.match('xx') #=> nil 02792 * 02793 * If a block is given, invoke the block with MatchData if match succeed, so 02794 * that you can write 02795 * 02796 * str.match(pat) {|m| ...} 02797 * 02798 * instead of 02799 * 02800 * if m = str.match(pat) 02801 * ... 02802 * end 02803 * 02804 * The return value is a value from block execution in this case. 02805 */ 02806 02807 static VALUE 02808 rb_str_match_m(int argc, VALUE *argv, VALUE str) 02809 { 02810 VALUE re, result; 02811 if (argc < 1) 02812 rb_check_arity(argc, 1, 2); 02813 re = argv[0]; 02814 argv[0] = str; 02815 result = rb_funcall2(get_pat(re, 0), rb_intern("match"), argc, argv); 02816 if (!NIL_P(result) && rb_block_given_p()) { 02817 return rb_yield(result); 02818 } 02819 return result; 02820 } 02821 02822 enum neighbor_char { 02823 NEIGHBOR_NOT_CHAR, 02824 NEIGHBOR_FOUND, 02825 NEIGHBOR_WRAPPED 02826 }; 02827 02828 static enum neighbor_char 02829 enc_succ_char(char *p, long len, rb_encoding *enc) 02830 { 02831 long i; 02832 int l; 02833 while (1) { 02834 for (i = len-1; 0 <= i && (unsigned char)p[i] == 0xff; i--) 02835 p[i] = '\0'; 02836 if (i < 0) 02837 return NEIGHBOR_WRAPPED; 02838 ++((unsigned char*)p)[i]; 02839 l = rb_enc_precise_mbclen(p, p+len, enc); 02840 if (MBCLEN_CHARFOUND_P(l)) { 02841 l = MBCLEN_CHARFOUND_LEN(l); 02842 if (l == len) { 02843 return NEIGHBOR_FOUND; 02844 } 02845 else { 02846 memset(p+l, 0xff, len-l); 02847 } 02848 } 02849 if (MBCLEN_INVALID_P(l) && i < len-1) { 02850 long len2; 02851 int l2; 02852 for (len2 = len-1; 0 < len2; len2--) { 02853 l2 = rb_enc_precise_mbclen(p, p+len2, enc); 02854 if (!MBCLEN_INVALID_P(l2)) 02855 break; 02856 } 02857 memset(p+len2+1, 0xff, len-(len2+1)); 02858 } 02859 } 02860 } 02861 02862 static enum neighbor_char 02863 enc_pred_char(char *p, long len, rb_encoding *enc) 02864 { 02865 long i; 02866 int l; 02867 while (1) { 02868 for (i = len-1; 0 <= i && (unsigned char)p[i] == 0; i--) 02869 p[i] = '\xff'; 02870 if (i < 0) 02871 return NEIGHBOR_WRAPPED; 02872 --((unsigned char*)p)[i]; 02873 l = rb_enc_precise_mbclen(p, p+len, enc); 02874 if (MBCLEN_CHARFOUND_P(l)) { 02875 l = MBCLEN_CHARFOUND_LEN(l); 02876 if (l == len) { 02877 return NEIGHBOR_FOUND; 02878 } 02879 else { 02880 memset(p+l, 0, len-l); 02881 } 02882 } 02883 if (MBCLEN_INVALID_P(l) && i < len-1) { 02884 long len2; 02885 int l2; 02886 for (len2 = len-1; 0 < len2; len2--) { 02887 l2 = rb_enc_precise_mbclen(p, p+len2, enc); 02888 if (!MBCLEN_INVALID_P(l2)) 02889 break; 02890 } 02891 memset(p+len2+1, 0, len-(len2+1)); 02892 } 02893 } 02894 } 02895 02896 /* 02897 overwrite +p+ by succeeding letter in +enc+ and returns 02898 NEIGHBOR_FOUND or NEIGHBOR_WRAPPED. 02899 When NEIGHBOR_WRAPPED, carried-out letter is stored into carry. 02900 assuming each ranges are successive, and mbclen 02901 never change in each ranges. 02902 NEIGHBOR_NOT_CHAR is returned if invalid character or the range has only one 02903 character. 02904 */ 02905 static enum neighbor_char 02906 enc_succ_alnum_char(char *p, long len, rb_encoding *enc, char *carry) 02907 { 02908 enum neighbor_char ret; 02909 unsigned int c; 02910 int ctype; 02911 int range; 02912 char save[ONIGENC_CODE_TO_MBC_MAXLEN]; 02913 02914 c = rb_enc_mbc_to_codepoint(p, p+len, enc); 02915 if (rb_enc_isctype(c, ONIGENC_CTYPE_DIGIT, enc)) 02916 ctype = ONIGENC_CTYPE_DIGIT; 02917 else if (rb_enc_isctype(c, ONIGENC_CTYPE_ALPHA, enc)) 02918 ctype = ONIGENC_CTYPE_ALPHA; 02919 else 02920 return NEIGHBOR_NOT_CHAR; 02921 02922 MEMCPY(save, p, char, len); 02923 ret = enc_succ_char(p, len, enc); 02924 if (ret == NEIGHBOR_FOUND) { 02925 c = rb_enc_mbc_to_codepoint(p, p+len, enc); 02926 if (rb_enc_isctype(c, ctype, enc)) 02927 return NEIGHBOR_FOUND; 02928 } 02929 MEMCPY(p, save, char, len); 02930 range = 1; 02931 while (1) { 02932 MEMCPY(save, p, char, len); 02933 ret = enc_pred_char(p, len, enc); 02934 if (ret == NEIGHBOR_FOUND) { 02935 c = rb_enc_mbc_to_codepoint(p, p+len, enc); 02936 if (!rb_enc_isctype(c, ctype, enc)) { 02937 MEMCPY(p, save, char, len); 02938 break; 02939 } 02940 } 02941 else { 02942 MEMCPY(p, save, char, len); 02943 break; 02944 } 02945 range++; 02946 } 02947 if (range == 1) { 02948 return NEIGHBOR_NOT_CHAR; 02949 } 02950 02951 if (ctype != ONIGENC_CTYPE_DIGIT) { 02952 MEMCPY(carry, p, char, len); 02953 return NEIGHBOR_WRAPPED; 02954 } 02955 02956 MEMCPY(carry, p, char, len); 02957 enc_succ_char(carry, len, enc); 02958 return NEIGHBOR_WRAPPED; 02959 } 02960 02961 02962 /* 02963 * call-seq: 02964 * str.succ -> new_str 02965 * str.next -> new_str 02966 * 02967 * Returns the successor to <i>str</i>. The successor is calculated by 02968 * incrementing characters starting from the rightmost alphanumeric (or 02969 * the rightmost character if there are no alphanumerics) in the 02970 * string. Incrementing a digit always results in another digit, and 02971 * incrementing a letter results in another letter of the same case. 02972 * Incrementing nonalphanumerics uses the underlying character set's 02973 * collating sequence. 02974 * 02975 * If the increment generates a ``carry,'' the character to the left of 02976 * it is incremented. This process repeats until there is no carry, 02977 * adding an additional character if necessary. 02978 * 02979 * "abcd".succ #=> "abce" 02980 * "THX1138".succ #=> "THX1139" 02981 * "<<koala>>".succ #=> "<<koalb>>" 02982 * "1999zzz".succ #=> "2000aaa" 02983 * "ZZZ9999".succ #=> "AAAA0000" 02984 * "***".succ #=> "**+" 02985 */ 02986 02987 VALUE 02988 rb_str_succ(VALUE orig) 02989 { 02990 rb_encoding *enc; 02991 VALUE str; 02992 char *sbeg, *s, *e, *last_alnum = 0; 02993 int c = -1; 02994 long l; 02995 char carry[ONIGENC_CODE_TO_MBC_MAXLEN] = "\1"; 02996 long carry_pos = 0, carry_len = 1; 02997 enum neighbor_char neighbor = NEIGHBOR_FOUND; 02998 02999 str = rb_str_new5(orig, RSTRING_PTR(orig), RSTRING_LEN(orig)); 03000 rb_enc_cr_str_copy_for_substr(str, orig); 03001 OBJ_INFECT(str, orig); 03002 if (RSTRING_LEN(str) == 0) return str; 03003 03004 enc = STR_ENC_GET(orig); 03005 sbeg = RSTRING_PTR(str); 03006 s = e = sbeg + RSTRING_LEN(str); 03007 03008 while ((s = rb_enc_prev_char(sbeg, s, e, enc)) != 0) { 03009 if (neighbor == NEIGHBOR_NOT_CHAR && last_alnum) { 03010 if (ISALPHA(*last_alnum) ? ISDIGIT(*s) : 03011 ISDIGIT(*last_alnum) ? ISALPHA(*s) : 0) { 03012 s = last_alnum; 03013 break; 03014 } 03015 } 03016 if ((l = rb_enc_precise_mbclen(s, e, enc)) <= 0) continue; 03017 neighbor = enc_succ_alnum_char(s, l, enc, carry); 03018 switch (neighbor) { 03019 case NEIGHBOR_NOT_CHAR: 03020 continue; 03021 case NEIGHBOR_FOUND: 03022 return str; 03023 case NEIGHBOR_WRAPPED: 03024 last_alnum = s; 03025 break; 03026 } 03027 c = 1; 03028 carry_pos = s - sbeg; 03029 carry_len = l; 03030 } 03031 if (c == -1) { /* str contains no alnum */ 03032 s = e; 03033 while ((s = rb_enc_prev_char(sbeg, s, e, enc)) != 0) { 03034 enum neighbor_char neighbor; 03035 if ((l = rb_enc_precise_mbclen(s, e, enc)) <= 0) continue; 03036 neighbor = enc_succ_char(s, l, enc); 03037 if (neighbor == NEIGHBOR_FOUND) 03038 return str; 03039 if (rb_enc_precise_mbclen(s, s+l, enc) != l) { 03040 /* wrapped to \0...\0. search next valid char. */ 03041 enc_succ_char(s, l, enc); 03042 } 03043 if (!rb_enc_asciicompat(enc)) { 03044 MEMCPY(carry, s, char, l); 03045 carry_len = l; 03046 } 03047 carry_pos = s - sbeg; 03048 } 03049 } 03050 RESIZE_CAPA(str, RSTRING_LEN(str) + carry_len); 03051 s = RSTRING_PTR(str) + carry_pos; 03052 memmove(s + carry_len, s, RSTRING_LEN(str) - carry_pos); 03053 memmove(s, carry, carry_len); 03054 STR_SET_LEN(str, RSTRING_LEN(str) + carry_len); 03055 RSTRING_PTR(str)[RSTRING_LEN(str)] = '\0'; 03056 rb_enc_str_coderange(str); 03057 return str; 03058 } 03059 03060 03061 /* 03062 * call-seq: 03063 * str.succ! -> str 03064 * str.next! -> str 03065 * 03066 * Equivalent to <code>String#succ</code>, but modifies the receiver in 03067 * place. 03068 */ 03069 03070 static VALUE 03071 rb_str_succ_bang(VALUE str) 03072 { 03073 rb_str_shared_replace(str, rb_str_succ(str)); 03074 03075 return str; 03076 } 03077 03078 03079 /* 03080 * call-seq: 03081 * str.upto(other_str, exclusive=false) {|s| block } -> str 03082 * str.upto(other_str, exclusive=false) -> an_enumerator 03083 * 03084 * Iterates through successive values, starting at <i>str</i> and 03085 * ending at <i>other_str</i> inclusive, passing each value in turn to 03086 * the block. The <code>String#succ</code> method is used to generate 03087 * each value. If optional second argument exclusive is omitted or is false, 03088 * the last value will be included; otherwise it will be excluded. 03089 * 03090 * If no block is given, an enumerator is returned instead. 03091 * 03092 * "a8".upto("b6") {|s| print s, ' ' } 03093 * for s in "a8".."b6" 03094 * print s, ' ' 03095 * end 03096 * 03097 * <em>produces:</em> 03098 * 03099 * a8 a9 b0 b1 b2 b3 b4 b5 b6 03100 * a8 a9 b0 b1 b2 b3 b4 b5 b6 03101 * 03102 * If <i>str</i> and <i>other_str</i> contains only ascii numeric characters, 03103 * both are recognized as decimal numbers. In addition, the width of 03104 * string (e.g. leading zeros) is handled appropriately. 03105 * 03106 * "9".upto("11").to_a #=> ["9", "10", "11"] 03107 * "25".upto("5").to_a #=> [] 03108 * "07".upto("11").to_a #=> ["07", "08", "09", "10", "11"] 03109 */ 03110 03111 static VALUE 03112 rb_str_upto(int argc, VALUE *argv, VALUE beg) 03113 { 03114 VALUE end, exclusive; 03115 VALUE current, after_end; 03116 ID succ; 03117 int n, excl, ascii; 03118 rb_encoding *enc; 03119 03120 rb_scan_args(argc, argv, "11", &end, &exclusive); 03121 RETURN_ENUMERATOR(beg, argc, argv); 03122 excl = RTEST(exclusive); 03123 CONST_ID(succ, "succ"); 03124 StringValue(end); 03125 enc = rb_enc_check(beg, end); 03126 ascii = (is_ascii_string(beg) && is_ascii_string(end)); 03127 /* single character */ 03128 if (RSTRING_LEN(beg) == 1 && RSTRING_LEN(end) == 1 && ascii) { 03129 char c = RSTRING_PTR(beg)[0]; 03130 char e = RSTRING_PTR(end)[0]; 03131 03132 if (c > e || (excl && c == e)) return beg; 03133 for (;;) { 03134 rb_yield(rb_enc_str_new(&c, 1, enc)); 03135 if (!excl && c == e) break; 03136 c++; 03137 if (excl && c == e) break; 03138 } 03139 return beg; 03140 } 03141 /* both edges are all digits */ 03142 if (ascii && ISDIGIT(RSTRING_PTR(beg)[0]) && ISDIGIT(RSTRING_PTR(end)[0])) { 03143 char *s, *send; 03144 VALUE b, e; 03145 int width; 03146 03147 s = RSTRING_PTR(beg); send = RSTRING_END(beg); 03148 width = rb_long2int(send - s); 03149 while (s < send) { 03150 if (!ISDIGIT(*s)) goto no_digits; 03151 s++; 03152 } 03153 s = RSTRING_PTR(end); send = RSTRING_END(end); 03154 while (s < send) { 03155 if (!ISDIGIT(*s)) goto no_digits; 03156 s++; 03157 } 03158 b = rb_str_to_inum(beg, 10, FALSE); 03159 e = rb_str_to_inum(end, 10, FALSE); 03160 if (FIXNUM_P(b) && FIXNUM_P(e)) { 03161 long bi = FIX2LONG(b); 03162 long ei = FIX2LONG(e); 03163 rb_encoding *usascii = rb_usascii_encoding(); 03164 03165 while (bi <= ei) { 03166 if (excl && bi == ei) break; 03167 rb_yield(rb_enc_sprintf(usascii, "%.*ld", width, bi)); 03168 bi++; 03169 } 03170 } 03171 else { 03172 ID op = excl ? '<' : rb_intern("<="); 03173 VALUE args[2], fmt = rb_obj_freeze(rb_usascii_str_new_cstr("%.*d")); 03174 03175 args[0] = INT2FIX(width); 03176 while (rb_funcall(b, op, 1, e)) { 03177 args[1] = b; 03178 rb_yield(rb_str_format(numberof(args), args, fmt)); 03179 b = rb_funcall(b, succ, 0, 0); 03180 } 03181 } 03182 return beg; 03183 } 03184 /* normal case */ 03185 no_digits: 03186 n = rb_str_cmp(beg, end); 03187 if (n > 0 || (excl && n == 0)) return beg; 03188 03189 after_end = rb_funcall(end, succ, 0, 0); 03190 current = rb_str_dup(beg); 03191 while (!rb_str_equal(current, after_end)) { 03192 VALUE next = Qnil; 03193 if (excl || !rb_str_equal(current, end)) 03194 next = rb_funcall(current, succ, 0, 0); 03195 rb_yield(current); 03196 if (NIL_P(next)) break; 03197 current = next; 03198 StringValue(current); 03199 if (excl && rb_str_equal(current, end)) break; 03200 if (RSTRING_LEN(current) > RSTRING_LEN(end) || RSTRING_LEN(current) == 0) 03201 break; 03202 } 03203 03204 return beg; 03205 } 03206 03207 static VALUE 03208 rb_str_subpat(VALUE str, VALUE re, VALUE backref) 03209 { 03210 if (rb_reg_search(re, str, 0, 0) >= 0) { 03211 VALUE match = rb_backref_get(); 03212 int nth = rb_reg_backref_number(match, backref); 03213 return rb_reg_nth_match(nth, match); 03214 } 03215 return Qnil; 03216 } 03217 03218 static VALUE 03219 rb_str_aref(VALUE str, VALUE indx) 03220 { 03221 long idx; 03222 03223 if (FIXNUM_P(indx)) { 03224 idx = FIX2LONG(indx); 03225 03226 num_index: 03227 str = rb_str_substr(str, idx, 1); 03228 if (!NIL_P(str) && RSTRING_LEN(str) == 0) return Qnil; 03229 return str; 03230 } 03231 03232 if (SPECIAL_CONST_P(indx)) goto generic; 03233 switch (BUILTIN_TYPE(indx)) { 03234 case T_REGEXP: 03235 return rb_str_subpat(str, indx, INT2FIX(0)); 03236 03237 case T_STRING: 03238 if (rb_str_index(str, indx, 0) != -1) 03239 return rb_str_dup(indx); 03240 return Qnil; 03241 03242 generic: 03243 default: 03244 /* check if indx is Range */ 03245 { 03246 long beg, len; 03247 VALUE tmp; 03248 03249 len = str_strlen(str, STR_ENC_GET(str)); 03250 switch (rb_range_beg_len(indx, &beg, &len, len, 0)) { 03251 case Qfalse: 03252 break; 03253 case Qnil: 03254 return Qnil; 03255 default: 03256 tmp = rb_str_substr(str, beg, len); 03257 return tmp; 03258 } 03259 } 03260 idx = NUM2LONG(indx); 03261 goto num_index; 03262 } 03263 03264 UNREACHABLE; 03265 } 03266 03267 03268 /* 03269 * call-seq: 03270 * str[index] -> new_str or nil 03271 * str[start, length] -> new_str or nil 03272 * str[range] -> new_str or nil 03273 * str[regexp] -> new_str or nil 03274 * str[regexp, capture] -> new_str or nil 03275 * str[match_str] -> new_str or nil 03276 * str.slice(index) -> new_str or nil 03277 * str.slice(start, length) -> new_str or nil 03278 * str.slice(range) -> new_str or nil 03279 * str.slice(regexp) -> new_str or nil 03280 * str.slice(regexp, capture) -> new_str or nil 03281 * str.slice(match_str) -> new_str or nil 03282 * 03283 * Element Reference --- If passed a single +index+, returns a substring of 03284 * one character at that index. If passed a +start+ index and a +length+, 03285 * returns a substring containing +length+ characters starting at the 03286 * +index+. If passed a +range+, its beginning and end are interpreted as 03287 * offsets delimiting the substring to be returned. 03288 * 03289 * In these three cases, if an index is negative, it is counted from the end 03290 * of the string. For the +start+ and +range+ cases the starting index 03291 * is just before a character and an index matching the string's size. 03292 * Additionally, an empty string is returned when the starting index for a 03293 * character range is at the end of the string. 03294 * 03295 * Returns +nil+ if the initial index falls outside the string or the length 03296 * is negative. 03297 * 03298 * If a +Regexp+ is supplied, the matching portion of the string is 03299 * returned. If a +capture+ follows the regular expression, which may be a 03300 * capture group index or name, follows the regular expression that component 03301 * of the MatchData is returned instead. 03302 * 03303 * If a +match_str+ is given, that string is returned if it occurs in 03304 * the string. 03305 * 03306 * Returns +nil+ if the regular expression does not match or the match string 03307 * cannot be found. 03308 * 03309 * a = "hello there" 03310 * 03311 * a[1] #=> "e" 03312 * a[2, 3] #=> "llo" 03313 * a[2..3] #=> "ll" 03314 * 03315 * a[-3, 2] #=> "er" 03316 * a[7..-2] #=> "her" 03317 * a[-4..-2] #=> "her" 03318 * a[-2..-4] #=> "" 03319 * 03320 * a[11, 0] #=> "" 03321 * a[11] #=> nil 03322 * a[12, 0] #=> nil 03323 * a[12..-1] #=> nil 03324 * 03325 * a[/[aeiou](.)\1/] #=> "ell" 03326 * a[/[aeiou](.)\1/, 0] #=> "ell" 03327 * a[/[aeiou](.)\1/, 1] #=> "l" 03328 * a[/[aeiou](.)\1/, 2] #=> nil 03329 * 03330 * a[/(?<vowel>[aeiou])(?<non_vowel>[^aeiou])/, "non_vowel"] #=> "l" 03331 * a[/(?<vowel>[aeiou])(?<non_vowel>[^aeiou])/, "vowel"] #=> "e" 03332 * 03333 * a["lo"] #=> "lo" 03334 * a["bye"] #=> nil 03335 */ 03336 03337 static VALUE 03338 rb_str_aref_m(int argc, VALUE *argv, VALUE str) 03339 { 03340 if (argc == 2) { 03341 if (RB_TYPE_P(argv[0], T_REGEXP)) { 03342 return rb_str_subpat(str, argv[0], argv[1]); 03343 } 03344 return rb_str_substr(str, NUM2LONG(argv[0]), NUM2LONG(argv[1])); 03345 } 03346 rb_check_arity(argc, 1, 2); 03347 return rb_str_aref(str, argv[0]); 03348 } 03349 03350 VALUE 03351 rb_str_drop_bytes(VALUE str, long len) 03352 { 03353 char *ptr = RSTRING_PTR(str); 03354 long olen = RSTRING_LEN(str), nlen; 03355 03356 str_modifiable(str); 03357 if (len > olen) len = olen; 03358 nlen = olen - len; 03359 if (nlen <= RSTRING_EMBED_LEN_MAX) { 03360 char *oldptr = ptr; 03361 int fl = (int)(RBASIC(str)->flags & (STR_NOEMBED|ELTS_SHARED)); 03362 STR_SET_EMBED(str); 03363 STR_SET_EMBED_LEN(str, nlen); 03364 ptr = RSTRING(str)->as.ary; 03365 memmove(ptr, oldptr + len, nlen); 03366 if (fl == STR_NOEMBED) xfree(oldptr); 03367 } 03368 else { 03369 if (!STR_SHARED_P(str)) rb_str_new4(str); 03370 ptr = RSTRING(str)->as.heap.ptr += len; 03371 RSTRING(str)->as.heap.len = nlen; 03372 } 03373 ptr[nlen] = 0; 03374 ENC_CODERANGE_CLEAR(str); 03375 return str; 03376 } 03377 03378 static void 03379 rb_str_splice_0(VALUE str, long beg, long len, VALUE val) 03380 { 03381 if (beg == 0 && RSTRING_LEN(val) == 0) { 03382 rb_str_drop_bytes(str, len); 03383 OBJ_INFECT(str, val); 03384 return; 03385 } 03386 03387 rb_str_modify(str); 03388 if (len < RSTRING_LEN(val)) { 03389 /* expand string */ 03390 RESIZE_CAPA(str, RSTRING_LEN(str) + RSTRING_LEN(val) - len + 1); 03391 } 03392 03393 if (RSTRING_LEN(val) != len) { 03394 memmove(RSTRING_PTR(str) + beg + RSTRING_LEN(val), 03395 RSTRING_PTR(str) + beg + len, 03396 RSTRING_LEN(str) - (beg + len)); 03397 } 03398 if (RSTRING_LEN(val) < beg && len < 0) { 03399 MEMZERO(RSTRING_PTR(str) + RSTRING_LEN(str), char, -len); 03400 } 03401 if (RSTRING_LEN(val) > 0) { 03402 memmove(RSTRING_PTR(str)+beg, RSTRING_PTR(val), RSTRING_LEN(val)); 03403 } 03404 STR_SET_LEN(str, RSTRING_LEN(str) + RSTRING_LEN(val) - len); 03405 if (RSTRING_PTR(str)) { 03406 RSTRING_PTR(str)[RSTRING_LEN(str)] = '\0'; 03407 } 03408 OBJ_INFECT(str, val); 03409 } 03410 03411 static void 03412 rb_str_splice(VALUE str, long beg, long len, VALUE val) 03413 { 03414 long slen; 03415 char *p, *e; 03416 rb_encoding *enc; 03417 int singlebyte = single_byte_optimizable(str); 03418 int cr; 03419 03420 if (len < 0) rb_raise(rb_eIndexError, "negative length %ld", len); 03421 03422 StringValue(val); 03423 enc = rb_enc_check(str, val); 03424 slen = str_strlen(str, enc); 03425 03426 if (slen < beg) { 03427 out_of_range: 03428 rb_raise(rb_eIndexError, "index %ld out of string", beg); 03429 } 03430 if (beg < 0) { 03431 if (-beg > slen) { 03432 goto out_of_range; 03433 } 03434 beg += slen; 03435 } 03436 if (slen < len || slen < beg + len) { 03437 len = slen - beg; 03438 } 03439 str_modify_keep_cr(str); 03440 p = str_nth(RSTRING_PTR(str), RSTRING_END(str), beg, enc, singlebyte); 03441 if (!p) p = RSTRING_END(str); 03442 e = str_nth(p, RSTRING_END(str), len, enc, singlebyte); 03443 if (!e) e = RSTRING_END(str); 03444 /* error check */ 03445 beg = p - RSTRING_PTR(str); /* physical position */ 03446 len = e - p; /* physical length */ 03447 rb_str_splice_0(str, beg, len, val); 03448 rb_enc_associate(str, enc); 03449 cr = ENC_CODERANGE_AND(ENC_CODERANGE(str), ENC_CODERANGE(val)); 03450 if (cr != ENC_CODERANGE_BROKEN) 03451 ENC_CODERANGE_SET(str, cr); 03452 } 03453 03454 void 03455 rb_str_update(VALUE str, long beg, long len, VALUE val) 03456 { 03457 rb_str_splice(str, beg, len, val); 03458 } 03459 03460 static void 03461 rb_str_subpat_set(VALUE str, VALUE re, VALUE backref, VALUE val) 03462 { 03463 int nth; 03464 VALUE match; 03465 long start, end, len; 03466 rb_encoding *enc; 03467 struct re_registers *regs; 03468 03469 if (rb_reg_search(re, str, 0, 0) < 0) { 03470 rb_raise(rb_eIndexError, "regexp not matched"); 03471 } 03472 match = rb_backref_get(); 03473 nth = rb_reg_backref_number(match, backref); 03474 regs = RMATCH_REGS(match); 03475 if (nth >= regs->num_regs) { 03476 out_of_range: 03477 rb_raise(rb_eIndexError, "index %d out of regexp", nth); 03478 } 03479 if (nth < 0) { 03480 if (-nth >= regs->num_regs) { 03481 goto out_of_range; 03482 } 03483 nth += regs->num_regs; 03484 } 03485 03486 start = BEG(nth); 03487 if (start == -1) { 03488 rb_raise(rb_eIndexError, "regexp group %d not matched", nth); 03489 } 03490 end = END(nth); 03491 len = end - start; 03492 StringValue(val); 03493 enc = rb_enc_check(str, val); 03494 rb_str_splice_0(str, start, len, val); 03495 rb_enc_associate(str, enc); 03496 } 03497 03498 static VALUE 03499 rb_str_aset(VALUE str, VALUE indx, VALUE val) 03500 { 03501 long idx, beg; 03502 03503 if (FIXNUM_P(indx)) { 03504 idx = FIX2LONG(indx); 03505 num_index: 03506 rb_str_splice(str, idx, 1, val); 03507 return val; 03508 } 03509 03510 if (SPECIAL_CONST_P(indx)) goto generic; 03511 switch (TYPE(indx)) { 03512 case T_REGEXP: 03513 rb_str_subpat_set(str, indx, INT2FIX(0), val); 03514 return val; 03515 03516 case T_STRING: 03517 beg = rb_str_index(str, indx, 0); 03518 if (beg < 0) { 03519 rb_raise(rb_eIndexError, "string not matched"); 03520 } 03521 beg = rb_str_sublen(str, beg); 03522 rb_str_splice(str, beg, str_strlen(indx, 0), val); 03523 return val; 03524 03525 generic: 03526 default: 03527 /* check if indx is Range */ 03528 { 03529 long beg, len; 03530 if (rb_range_beg_len(indx, &beg, &len, str_strlen(str, 0), 2)) { 03531 rb_str_splice(str, beg, len, val); 03532 return val; 03533 } 03534 } 03535 idx = NUM2LONG(indx); 03536 goto num_index; 03537 } 03538 } 03539 03540 /* 03541 * call-seq: 03542 * str[fixnum] = new_str 03543 * str[fixnum, fixnum] = new_str 03544 * str[range] = aString 03545 * str[regexp] = new_str 03546 * str[regexp, fixnum] = new_str 03547 * str[regexp, name] = new_str 03548 * str[other_str] = new_str 03549 * 03550 * Element Assignment---Replaces some or all of the content of <i>str</i>. The 03551 * portion of the string affected is determined using the same criteria as 03552 * <code>String#[]</code>. If the replacement string is not the same length as 03553 * the text it is replacing, the string will be adjusted accordingly. If the 03554 * regular expression or string is used as the index doesn't match a position 03555 * in the string, <code>IndexError</code> is raised. If the regular expression 03556 * form is used, the optional second <code>Fixnum</code> allows you to specify 03557 * which portion of the match to replace (effectively using the 03558 * <code>MatchData</code> indexing rules. The forms that take a 03559 * <code>Fixnum</code> will raise an <code>IndexError</code> if the value is 03560 * out of range; the <code>Range</code> form will raise a 03561 * <code>RangeError</code>, and the <code>Regexp</code> and <code>String</code> 03562 * will raise an <code>IndexError</code> on negative match. 03563 */ 03564 03565 static VALUE 03566 rb_str_aset_m(int argc, VALUE *argv, VALUE str) 03567 { 03568 if (argc == 3) { 03569 if (RB_TYPE_P(argv[0], T_REGEXP)) { 03570 rb_str_subpat_set(str, argv[0], argv[1], argv[2]); 03571 } 03572 else { 03573 rb_str_splice(str, NUM2LONG(argv[0]), NUM2LONG(argv[1]), argv[2]); 03574 } 03575 return argv[2]; 03576 } 03577 rb_check_arity(argc, 2, 3); 03578 return rb_str_aset(str, argv[0], argv[1]); 03579 } 03580 03581 /* 03582 * call-seq: 03583 * str.insert(index, other_str) -> str 03584 * 03585 * Inserts <i>other_str</i> before the character at the given 03586 * <i>index</i>, modifying <i>str</i>. Negative indices count from the 03587 * end of the string, and insert <em>after</em> the given character. 03588 * The intent is insert <i>aString</i> so that it starts at the given 03589 * <i>index</i>. 03590 * 03591 * "abcd".insert(0, 'X') #=> "Xabcd" 03592 * "abcd".insert(3, 'X') #=> "abcXd" 03593 * "abcd".insert(4, 'X') #=> "abcdX" 03594 * "abcd".insert(-3, 'X') #=> "abXcd" 03595 * "abcd".insert(-1, 'X') #=> "abcdX" 03596 */ 03597 03598 static VALUE 03599 rb_str_insert(VALUE str, VALUE idx, VALUE str2) 03600 { 03601 long pos = NUM2LONG(idx); 03602 03603 if (pos == -1) { 03604 return rb_str_append(str, str2); 03605 } 03606 else if (pos < 0) { 03607 pos++; 03608 } 03609 rb_str_splice(str, pos, 0, str2); 03610 return str; 03611 } 03612 03613 03614 /* 03615 * call-seq: 03616 * str.slice!(fixnum) -> fixnum or nil 03617 * str.slice!(fixnum, fixnum) -> new_str or nil 03618 * str.slice!(range) -> new_str or nil 03619 * str.slice!(regexp) -> new_str or nil 03620 * str.slice!(other_str) -> new_str or nil 03621 * 03622 * Deletes the specified portion from <i>str</i>, and returns the portion 03623 * deleted. 03624 * 03625 * string = "this is a string" 03626 * string.slice!(2) #=> "i" 03627 * string.slice!(3..6) #=> " is " 03628 * string.slice!(/s.*t/) #=> "sa st" 03629 * string.slice!("r") #=> "r" 03630 * string #=> "thing" 03631 */ 03632 03633 static VALUE 03634 rb_str_slice_bang(int argc, VALUE *argv, VALUE str) 03635 { 03636 VALUE result; 03637 VALUE buf[3]; 03638 int i; 03639 03640 rb_check_arity(argc, 1, 2); 03641 for (i=0; i<argc; i++) { 03642 buf[i] = argv[i]; 03643 } 03644 str_modify_keep_cr(str); 03645 result = rb_str_aref_m(argc, buf, str); 03646 if (!NIL_P(result)) { 03647 buf[i] = rb_str_new(0,0); 03648 rb_str_aset_m(argc+1, buf, str); 03649 } 03650 return result; 03651 } 03652 03653 static VALUE 03654 get_pat(VALUE pat, int quote) 03655 { 03656 VALUE val; 03657 03658 switch (TYPE(pat)) { 03659 case T_REGEXP: 03660 return pat; 03661 03662 case T_STRING: 03663 break; 03664 03665 default: 03666 val = rb_check_string_type(pat); 03667 if (NIL_P(val)) { 03668 Check_Type(pat, T_REGEXP); 03669 } 03670 pat = val; 03671 } 03672 03673 if (quote) { 03674 pat = rb_reg_quote(pat); 03675 } 03676 03677 return rb_reg_regcomp(pat); 03678 } 03679 03680 03681 /* 03682 * call-seq: 03683 * str.sub!(pattern, replacement) -> str or nil 03684 * str.sub!(pattern) {|match| block } -> str or nil 03685 * 03686 * Performs the same substitution as String#sub in-place. 03687 * 03688 * Returns +str+ if a substitution was performed or +nil+ if no substitution 03689 * was performed. 03690 */ 03691 03692 static VALUE 03693 rb_str_sub_bang(int argc, VALUE *argv, VALUE str) 03694 { 03695 VALUE pat, repl, hash = Qnil; 03696 int iter = 0; 03697 int tainted = 0; 03698 int untrusted = 0; 03699 long plen; 03700 int min_arity = rb_block_given_p() ? 1 : 2; 03701 03702 rb_check_arity(argc, min_arity, 2); 03703 if (argc == 1) { 03704 iter = 1; 03705 } 03706 else { 03707 repl = argv[1]; 03708 hash = rb_check_hash_type(argv[1]); 03709 if (NIL_P(hash)) { 03710 StringValue(repl); 03711 } 03712 if (OBJ_TAINTED(repl)) tainted = 1; 03713 if (OBJ_UNTRUSTED(repl)) untrusted = 1; 03714 } 03715 03716 pat = get_pat(argv[0], 1); 03717 str_modifiable(str); 03718 if (rb_reg_search(pat, str, 0, 0) >= 0) { 03719 rb_encoding *enc; 03720 int cr = ENC_CODERANGE(str); 03721 VALUE match = rb_backref_get(); 03722 struct re_registers *regs = RMATCH_REGS(match); 03723 long beg0 = BEG(0); 03724 long end0 = END(0); 03725 char *p, *rp; 03726 long len, rlen; 03727 03728 if (iter || !NIL_P(hash)) { 03729 p = RSTRING_PTR(str); len = RSTRING_LEN(str); 03730 03731 if (iter) { 03732 repl = rb_obj_as_string(rb_yield(rb_reg_nth_match(0, match))); 03733 } 03734 else { 03735 repl = rb_hash_aref(hash, rb_str_subseq(str, beg0, end0 - beg0)); 03736 repl = rb_obj_as_string(repl); 03737 } 03738 str_mod_check(str, p, len); 03739 rb_check_frozen(str); 03740 } 03741 else { 03742 repl = rb_reg_regsub(repl, str, regs, pat); 03743 } 03744 enc = rb_enc_compatible(str, repl); 03745 if (!enc) { 03746 rb_encoding *str_enc = STR_ENC_GET(str); 03747 p = RSTRING_PTR(str); len = RSTRING_LEN(str); 03748 if (coderange_scan(p, beg0, str_enc) != ENC_CODERANGE_7BIT || 03749 coderange_scan(p+end0, len-end0, str_enc) != ENC_CODERANGE_7BIT) { 03750 rb_raise(rb_eEncCompatError, "incompatible character encodings: %s and %s", 03751 rb_enc_name(str_enc), 03752 rb_enc_name(STR_ENC_GET(repl))); 03753 } 03754 enc = STR_ENC_GET(repl); 03755 } 03756 rb_str_modify(str); 03757 rb_enc_associate(str, enc); 03758 if (OBJ_TAINTED(repl)) tainted = 1; 03759 if (OBJ_UNTRUSTED(repl)) untrusted = 1; 03760 if (ENC_CODERANGE_UNKNOWN < cr && cr < ENC_CODERANGE_BROKEN) { 03761 int cr2 = ENC_CODERANGE(repl); 03762 if (cr2 == ENC_CODERANGE_BROKEN || 03763 (cr == ENC_CODERANGE_VALID && cr2 == ENC_CODERANGE_7BIT)) 03764 cr = ENC_CODERANGE_UNKNOWN; 03765 else 03766 cr = cr2; 03767 } 03768 plen = end0 - beg0; 03769 rp = RSTRING_PTR(repl); rlen = RSTRING_LEN(repl); 03770 len = RSTRING_LEN(str); 03771 if (rlen > plen) { 03772 RESIZE_CAPA(str, len + rlen - plen); 03773 } 03774 p = RSTRING_PTR(str); 03775 if (rlen != plen) { 03776 memmove(p + beg0 + rlen, p + beg0 + plen, len - beg0 - plen); 03777 } 03778 memcpy(p + beg0, rp, rlen); 03779 len += rlen - plen; 03780 STR_SET_LEN(str, len); 03781 RSTRING_PTR(str)[len] = '\0'; 03782 ENC_CODERANGE_SET(str, cr); 03783 if (tainted) OBJ_TAINT(str); 03784 if (untrusted) OBJ_UNTRUST(str); 03785 03786 return str; 03787 } 03788 return Qnil; 03789 } 03790 03791 03792 /* 03793 * call-seq: 03794 * str.sub(pattern, replacement) -> new_str 03795 * str.sub(pattern, hash) -> new_str 03796 * str.sub(pattern) {|match| block } -> new_str 03797 * 03798 * Returns a copy of +str+ with the _first_ occurrence of +pattern+ 03799 * replaced by the second argument. The +pattern+ is typically a Regexp; if 03800 * given as a String, any regular expression metacharacters it contains will 03801 * be interpreted literally, e.g. <code>'\\\d'</code> will match a backlash 03802 * followed by 'd', instead of a digit. 03803 * 03804 * If +replacement+ is a String it will be substituted for the matched text. 03805 * It may contain back-references to the pattern's capture groups of the form 03806 * <code>"\\d"</code>, where <i>d</i> is a group number, or 03807 * <code>"\\k<n>"</code>, where <i>n</i> is a group name. If it is a 03808 * double-quoted string, both back-references must be preceded by an 03809 * additional backslash. However, within +replacement+ the special match 03810 * variables, such as <code>&$</code>, will not refer to the current match. 03811 * 03812 * If the second argument is a Hash, and the matched text is one of its keys, 03813 * the corresponding value is the replacement string. 03814 * 03815 * In the block form, the current match string is passed in as a parameter, 03816 * and variables such as <code>$1</code>, <code>$2</code>, <code>$`</code>, 03817 * <code>$&</code>, and <code>$'</code> will be set appropriately. The value 03818 * returned by the block will be substituted for the match on each call. 03819 * 03820 * The result inherits any tainting in the original string or any supplied 03821 * replacement string. 03822 * 03823 * "hello".sub(/[aeiou]/, '*') #=> "h*llo" 03824 * "hello".sub(/([aeiou])/, '<\1>') #=> "h<e>llo" 03825 * "hello".sub(/./) {|s| s.ord.to_s + ' ' } #=> "104 ello" 03826 * "hello".sub(/(?<foo>[aeiou])/, '*\k<foo>*') #=> "h*e*llo" 03827 * 'Is SHELL your preferred shell?'.sub(/[[:upper:]]{2,}/, ENV) 03828 * #=> "Is /bin/bash your preferred shell?" 03829 */ 03830 03831 static VALUE 03832 rb_str_sub(int argc, VALUE *argv, VALUE str) 03833 { 03834 str = rb_str_dup(str); 03835 rb_str_sub_bang(argc, argv, str); 03836 return str; 03837 } 03838 03839 static VALUE 03840 str_gsub(int argc, VALUE *argv, VALUE str, int bang) 03841 { 03842 VALUE pat, val, repl, match, dest, hash = Qnil; 03843 struct re_registers *regs; 03844 long beg, n; 03845 long beg0, end0; 03846 long offset, blen, slen, len, last; 03847 int iter = 0; 03848 char *sp, *cp; 03849 int tainted = 0; 03850 rb_encoding *str_enc; 03851 03852 switch (argc) { 03853 case 1: 03854 RETURN_ENUMERATOR(str, argc, argv); 03855 iter = 1; 03856 break; 03857 case 2: 03858 repl = argv[1]; 03859 hash = rb_check_hash_type(argv[1]); 03860 if (NIL_P(hash)) { 03861 StringValue(repl); 03862 } 03863 if (OBJ_TAINTED(repl)) tainted = 1; 03864 break; 03865 default: 03866 rb_check_arity(argc, 1, 2); 03867 } 03868 03869 pat = get_pat(argv[0], 1); 03870 beg = rb_reg_search(pat, str, 0, 0); 03871 if (beg < 0) { 03872 if (bang) return Qnil; /* no match, no substitution */ 03873 return rb_str_dup(str); 03874 } 03875 03876 offset = 0; 03877 n = 0; 03878 blen = RSTRING_LEN(str) + 30; /* len + margin */ 03879 dest = rb_str_buf_new(blen); 03880 sp = RSTRING_PTR(str); 03881 slen = RSTRING_LEN(str); 03882 cp = sp; 03883 str_enc = STR_ENC_GET(str); 03884 rb_enc_associate(dest, str_enc); 03885 ENC_CODERANGE_SET(dest, rb_enc_asciicompat(str_enc) ? ENC_CODERANGE_7BIT : ENC_CODERANGE_VALID); 03886 03887 do { 03888 n++; 03889 match = rb_backref_get(); 03890 regs = RMATCH_REGS(match); 03891 beg0 = BEG(0); 03892 end0 = END(0); 03893 if (iter || !NIL_P(hash)) { 03894 if (iter) { 03895 val = rb_obj_as_string(rb_yield(rb_reg_nth_match(0, match))); 03896 } 03897 else { 03898 val = rb_hash_aref(hash, rb_str_subseq(str, BEG(0), END(0) - BEG(0))); 03899 val = rb_obj_as_string(val); 03900 } 03901 str_mod_check(str, sp, slen); 03902 if (val == dest) { /* paranoid check [ruby-dev:24827] */ 03903 rb_raise(rb_eRuntimeError, "block should not cheat"); 03904 } 03905 } 03906 else { 03907 val = rb_reg_regsub(repl, str, regs, pat); 03908 } 03909 03910 if (OBJ_TAINTED(val)) tainted = 1; 03911 03912 len = beg0 - offset; /* copy pre-match substr */ 03913 if (len) { 03914 rb_enc_str_buf_cat(dest, cp, len, str_enc); 03915 } 03916 03917 rb_str_buf_append(dest, val); 03918 03919 last = offset; 03920 offset = end0; 03921 if (beg0 == end0) { 03922 /* 03923 * Always consume at least one character of the input string 03924 * in order to prevent infinite loops. 03925 */ 03926 if (RSTRING_LEN(str) <= end0) break; 03927 len = rb_enc_fast_mbclen(RSTRING_PTR(str)+end0, RSTRING_END(str), str_enc); 03928 rb_enc_str_buf_cat(dest, RSTRING_PTR(str)+end0, len, str_enc); 03929 offset = end0 + len; 03930 } 03931 cp = RSTRING_PTR(str) + offset; 03932 if (offset > RSTRING_LEN(str)) break; 03933 beg = rb_reg_search(pat, str, offset, 0); 03934 } while (beg >= 0); 03935 if (RSTRING_LEN(str) > offset) { 03936 rb_enc_str_buf_cat(dest, cp, RSTRING_LEN(str) - offset, str_enc); 03937 } 03938 rb_reg_search(pat, str, last, 0); 03939 if (bang) { 03940 rb_str_shared_replace(str, dest); 03941 } 03942 else { 03943 RBASIC(dest)->klass = rb_obj_class(str); 03944 OBJ_INFECT(dest, str); 03945 str = dest; 03946 } 03947 03948 if (tainted) OBJ_TAINT(str); 03949 return str; 03950 } 03951 03952 03953 /* 03954 * call-seq: 03955 * str.gsub!(pattern, replacement) -> str or nil 03956 * str.gsub!(pattern) {|match| block } -> str or nil 03957 * str.gsub!(pattern) -> an_enumerator 03958 * 03959 * Performs the substitutions of <code>String#gsub</code> in place, returning 03960 * <i>str</i>, or <code>nil</code> if no substitutions were performed. 03961 * If no block and no <i>replacement</i> is given, an enumerator is returned instead. 03962 */ 03963 03964 static VALUE 03965 rb_str_gsub_bang(int argc, VALUE *argv, VALUE str) 03966 { 03967 str_modify_keep_cr(str); 03968 return str_gsub(argc, argv, str, 1); 03969 } 03970 03971 03972 /* 03973 * call-seq: 03974 * str.gsub(pattern, replacement) -> new_str 03975 * str.gsub(pattern, hash) -> new_str 03976 * str.gsub(pattern) {|match| block } -> new_str 03977 * str.gsub(pattern) -> enumerator 03978 * 03979 * Returns a copy of <i>str</i> with the <em>all</em> occurrences of 03980 * <i>pattern</i> substituted for the second argument. The <i>pattern</i> is 03981 * typically a <code>Regexp</code>; if given as a <code>String</code>, any 03982 * regular expression metacharacters it contains will be interpreted 03983 * literally, e.g. <code>'\\\d'</code> will match a backlash followed by 'd', 03984 * instead of a digit. 03985 * 03986 * If <i>replacement</i> is a <code>String</code> it will be substituted for 03987 * the matched text. It may contain back-references to the pattern's capture 03988 * groups of the form <code>\\\d</code>, where <i>d</i> is a group number, or 03989 * <code>\\\k<n></code>, where <i>n</i> is a group name. If it is a 03990 * double-quoted string, both back-references must be preceded by an 03991 * additional backslash. However, within <i>replacement</i> the special match 03992 * variables, such as <code>$&</code>, will not refer to the current match. 03993 * 03994 * If the second argument is a <code>Hash</code>, and the matched text is one 03995 * of its keys, the corresponding value is the replacement string. 03996 * 03997 * In the block form, the current match string is passed in as a parameter, 03998 * and variables such as <code>$1</code>, <code>$2</code>, <code>$`</code>, 03999 * <code>$&</code>, and <code>$'</code> will be set appropriately. The value 04000 * returned by the block will be substituted for the match on each call. 04001 * 04002 * The result inherits any tainting in the original string or any supplied 04003 * replacement string. 04004 * 04005 * When neither a block nor a second argument is supplied, an 04006 * <code>Enumerator</code> is returned. 04007 * 04008 * "hello".gsub(/[aeiou]/, '*') #=> "h*ll*" 04009 * "hello".gsub(/([aeiou])/, '<\1>') #=> "h<e>ll<o>" 04010 * "hello".gsub(/./) {|s| s.ord.to_s + ' '} #=> "104 101 108 108 111 " 04011 * "hello".gsub(/(?<foo>[aeiou])/, '{\k<foo>}') #=> "h{e}ll{o}" 04012 * 'hello'.gsub(/[eo]/, 'e' => 3, 'o' => '*') #=> "h3ll*" 04013 */ 04014 04015 static VALUE 04016 rb_str_gsub(int argc, VALUE *argv, VALUE str) 04017 { 04018 return str_gsub(argc, argv, str, 0); 04019 } 04020 04021 04022 /* 04023 * call-seq: 04024 * str.replace(other_str) -> str 04025 * 04026 * Replaces the contents and taintedness of <i>str</i> with the corresponding 04027 * values in <i>other_str</i>. 04028 * 04029 * s = "hello" #=> "hello" 04030 * s.replace "world" #=> "world" 04031 */ 04032 04033 VALUE 04034 rb_str_replace(VALUE str, VALUE str2) 04035 { 04036 str_modifiable(str); 04037 if (str == str2) return str; 04038 04039 StringValue(str2); 04040 str_discard(str); 04041 return str_replace(str, str2); 04042 } 04043 04044 /* 04045 * call-seq: 04046 * string.clear -> string 04047 * 04048 * Makes string empty. 04049 * 04050 * a = "abcde" 04051 * a.clear #=> "" 04052 */ 04053 04054 static VALUE 04055 rb_str_clear(VALUE str) 04056 { 04057 str_discard(str); 04058 STR_SET_EMBED(str); 04059 STR_SET_EMBED_LEN(str, 0); 04060 RSTRING_PTR(str)[0] = 0; 04061 if (rb_enc_asciicompat(STR_ENC_GET(str))) 04062 ENC_CODERANGE_SET(str, ENC_CODERANGE_7BIT); 04063 else 04064 ENC_CODERANGE_SET(str, ENC_CODERANGE_VALID); 04065 return str; 04066 } 04067 04068 /* 04069 * call-seq: 04070 * string.chr -> string 04071 * 04072 * Returns a one-character string at the beginning of the string. 04073 * 04074 * a = "abcde" 04075 * a.chr #=> "a" 04076 */ 04077 04078 static VALUE 04079 rb_str_chr(VALUE str) 04080 { 04081 return rb_str_substr(str, 0, 1); 04082 } 04083 04084 /* 04085 * call-seq: 04086 * str.getbyte(index) -> 0 .. 255 04087 * 04088 * returns the <i>index</i>th byte as an integer. 04089 */ 04090 static VALUE 04091 rb_str_getbyte(VALUE str, VALUE index) 04092 { 04093 long pos = NUM2LONG(index); 04094 04095 if (pos < 0) 04096 pos += RSTRING_LEN(str); 04097 if (pos < 0 || RSTRING_LEN(str) <= pos) 04098 return Qnil; 04099 04100 return INT2FIX((unsigned char)RSTRING_PTR(str)[pos]); 04101 } 04102 04103 /* 04104 * call-seq: 04105 * str.setbyte(index, integer) -> integer 04106 * 04107 * modifies the <i>index</i>th byte as <i>integer</i>. 04108 */ 04109 static VALUE 04110 rb_str_setbyte(VALUE str, VALUE index, VALUE value) 04111 { 04112 long pos = NUM2LONG(index); 04113 int byte = NUM2INT(value); 04114 04115 rb_str_modify(str); 04116 04117 if (pos < -RSTRING_LEN(str) || RSTRING_LEN(str) <= pos) 04118 rb_raise(rb_eIndexError, "index %ld out of string", pos); 04119 if (pos < 0) 04120 pos += RSTRING_LEN(str); 04121 04122 RSTRING_PTR(str)[pos] = byte; 04123 04124 return value; 04125 } 04126 04127 static VALUE 04128 str_byte_substr(VALUE str, long beg, long len) 04129 { 04130 char *p, *s = RSTRING_PTR(str); 04131 long n = RSTRING_LEN(str); 04132 VALUE str2; 04133 04134 if (beg > n || len < 0) return Qnil; 04135 if (beg < 0) { 04136 beg += n; 04137 if (beg < 0) return Qnil; 04138 } 04139 if (beg + len > n) 04140 len = n - beg; 04141 if (len <= 0) { 04142 len = 0; 04143 p = 0; 04144 } 04145 else 04146 p = s + beg; 04147 04148 if (len > RSTRING_EMBED_LEN_MAX && beg + len == n) { 04149 str2 = rb_str_new4(str); 04150 str2 = str_new3(rb_obj_class(str2), str2); 04151 RSTRING(str2)->as.heap.ptr += RSTRING(str2)->as.heap.len - len; 04152 RSTRING(str2)->as.heap.len = len; 04153 } 04154 else { 04155 str2 = rb_str_new5(str, p, len); 04156 } 04157 04158 str_enc_copy(str2, str); 04159 04160 if (RSTRING_LEN(str2) == 0) { 04161 if (!rb_enc_asciicompat(STR_ENC_GET(str))) 04162 ENC_CODERANGE_SET(str2, ENC_CODERANGE_VALID); 04163 else 04164 ENC_CODERANGE_SET(str2, ENC_CODERANGE_7BIT); 04165 } 04166 else { 04167 switch (ENC_CODERANGE(str)) { 04168 case ENC_CODERANGE_7BIT: 04169 ENC_CODERANGE_SET(str2, ENC_CODERANGE_7BIT); 04170 break; 04171 default: 04172 ENC_CODERANGE_SET(str2, ENC_CODERANGE_UNKNOWN); 04173 break; 04174 } 04175 } 04176 04177 OBJ_INFECT(str2, str); 04178 04179 return str2; 04180 } 04181 04182 static VALUE 04183 str_byte_aref(VALUE str, VALUE indx) 04184 { 04185 long idx; 04186 switch (TYPE(indx)) { 04187 case T_FIXNUM: 04188 idx = FIX2LONG(indx); 04189 04190 num_index: 04191 str = str_byte_substr(str, idx, 1); 04192 if (NIL_P(str) || RSTRING_LEN(str) == 0) return Qnil; 04193 return str; 04194 04195 default: 04196 /* check if indx is Range */ 04197 { 04198 long beg, len = RSTRING_LEN(str); 04199 04200 switch (rb_range_beg_len(indx, &beg, &len, len, 0)) { 04201 case Qfalse: 04202 break; 04203 case Qnil: 04204 return Qnil; 04205 default: 04206 return str_byte_substr(str, beg, len); 04207 } 04208 } 04209 idx = NUM2LONG(indx); 04210 goto num_index; 04211 } 04212 04213 UNREACHABLE; 04214 } 04215 04216 /* 04217 * call-seq: 04218 * str.byteslice(fixnum) -> new_str or nil 04219 * str.byteslice(fixnum, fixnum) -> new_str or nil 04220 * str.byteslice(range) -> new_str or nil 04221 * 04222 * Byte Reference---If passed a single <code>Fixnum</code>, returns a 04223 * substring of one byte at that position. If passed two <code>Fixnum</code> 04224 * objects, returns a substring starting at the offset given by the first, and 04225 * a length given by the second. If given a <code>Range</code>, a substring containing 04226 * bytes at offsets given by the range is returned. In all three cases, if 04227 * an offset is negative, it is counted from the end of <i>str</i>. Returns 04228 * <code>nil</code> if the initial offset falls outside the string, the length 04229 * is negative, or the beginning of the range is greater than the end. 04230 * The encoding of the resulted string keeps original encoding. 04231 * 04232 * "hello".byteslice(1) #=> "e" 04233 * "hello".byteslice(-1) #=> "o" 04234 * "hello".byteslice(1, 2) #=> "el" 04235 * "\x80\u3042".byteslice(1, 3) #=> "\u3042" 04236 * "\x03\u3042\xff".byteslice(1..3) #=> "\u3042" 04237 */ 04238 04239 static VALUE 04240 rb_str_byteslice(int argc, VALUE *argv, VALUE str) 04241 { 04242 if (argc == 2) { 04243 return str_byte_substr(str, NUM2LONG(argv[0]), NUM2LONG(argv[1])); 04244 } 04245 rb_check_arity(argc, 1, 2); 04246 return str_byte_aref(str, argv[0]); 04247 } 04248 04249 /* 04250 * call-seq: 04251 * str.reverse -> new_str 04252 * 04253 * Returns a new string with the characters from <i>str</i> in reverse order. 04254 * 04255 * "stressed".reverse #=> "desserts" 04256 */ 04257 04258 static VALUE 04259 rb_str_reverse(VALUE str) 04260 { 04261 rb_encoding *enc; 04262 VALUE rev; 04263 char *s, *e, *p; 04264 int single = 1; 04265 04266 if (RSTRING_LEN(str) <= 1) return rb_str_dup(str); 04267 enc = STR_ENC_GET(str); 04268 rev = rb_str_new5(str, 0, RSTRING_LEN(str)); 04269 s = RSTRING_PTR(str); e = RSTRING_END(str); 04270 p = RSTRING_END(rev); 04271 04272 if (RSTRING_LEN(str) > 1) { 04273 if (single_byte_optimizable(str)) { 04274 while (s < e) { 04275 *--p = *s++; 04276 } 04277 } 04278 else if (ENC_CODERANGE(str) == ENC_CODERANGE_VALID) { 04279 while (s < e) { 04280 int clen = rb_enc_fast_mbclen(s, e, enc); 04281 04282 if (clen > 1 || (*s & 0x80)) single = 0; 04283 p -= clen; 04284 memcpy(p, s, clen); 04285 s += clen; 04286 } 04287 } 04288 else { 04289 while (s < e) { 04290 int clen = rb_enc_mbclen(s, e, enc); 04291 04292 if (clen > 1 || (*s & 0x80)) single = 0; 04293 p -= clen; 04294 memcpy(p, s, clen); 04295 s += clen; 04296 } 04297 } 04298 } 04299 STR_SET_LEN(rev, RSTRING_LEN(str)); 04300 OBJ_INFECT(rev, str); 04301 if (ENC_CODERANGE(str) == ENC_CODERANGE_UNKNOWN) { 04302 if (single) { 04303 ENC_CODERANGE_SET(str, ENC_CODERANGE_7BIT); 04304 } 04305 else { 04306 ENC_CODERANGE_SET(str, ENC_CODERANGE_VALID); 04307 } 04308 } 04309 rb_enc_cr_str_copy_for_substr(rev, str); 04310 04311 return rev; 04312 } 04313 04314 04315 /* 04316 * call-seq: 04317 * str.reverse! -> str 04318 * 04319 * Reverses <i>str</i> in place. 04320 */ 04321 04322 static VALUE 04323 rb_str_reverse_bang(VALUE str) 04324 { 04325 if (RSTRING_LEN(str) > 1) { 04326 if (single_byte_optimizable(str)) { 04327 char *s, *e, c; 04328 04329 str_modify_keep_cr(str); 04330 s = RSTRING_PTR(str); 04331 e = RSTRING_END(str) - 1; 04332 while (s < e) { 04333 c = *s; 04334 *s++ = *e; 04335 *e-- = c; 04336 } 04337 } 04338 else { 04339 rb_str_shared_replace(str, rb_str_reverse(str)); 04340 } 04341 } 04342 else { 04343 str_modify_keep_cr(str); 04344 } 04345 return str; 04346 } 04347 04348 04349 /* 04350 * call-seq: 04351 * str.include? other_str -> true or false 04352 * 04353 * Returns <code>true</code> if <i>str</i> contains the given string or 04354 * character. 04355 * 04356 * "hello".include? "lo" #=> true 04357 * "hello".include? "ol" #=> false 04358 * "hello".include? ?h #=> true 04359 */ 04360 04361 static VALUE 04362 rb_str_include(VALUE str, VALUE arg) 04363 { 04364 long i; 04365 04366 StringValue(arg); 04367 i = rb_str_index(str, arg, 0); 04368 04369 if (i == -1) return Qfalse; 04370 return Qtrue; 04371 } 04372 04373 04374 /* 04375 * call-seq: 04376 * str.to_i(base=10) -> integer 04377 * 04378 * Returns the result of interpreting leading characters in <i>str</i> as an 04379 * integer base <i>base</i> (between 2 and 36). Extraneous characters past the 04380 * end of a valid number are ignored. If there is not a valid number at the 04381 * start of <i>str</i>, <code>0</code> is returned. This method never raises an 04382 * exception when <i>base</i> is valid. 04383 * 04384 * "12345".to_i #=> 12345 04385 * "99 red balloons".to_i #=> 99 04386 * "0a".to_i #=> 0 04387 * "0a".to_i(16) #=> 10 04388 * "hello".to_i #=> 0 04389 * "1100101".to_i(2) #=> 101 04390 * "1100101".to_i(8) #=> 294977 04391 * "1100101".to_i(10) #=> 1100101 04392 * "1100101".to_i(16) #=> 17826049 04393 */ 04394 04395 static VALUE 04396 rb_str_to_i(int argc, VALUE *argv, VALUE str) 04397 { 04398 int base; 04399 04400 if (argc == 0) base = 10; 04401 else { 04402 VALUE b; 04403 04404 rb_scan_args(argc, argv, "01", &b); 04405 base = NUM2INT(b); 04406 } 04407 if (base < 0) { 04408 rb_raise(rb_eArgError, "invalid radix %d", base); 04409 } 04410 return rb_str_to_inum(str, base, FALSE); 04411 } 04412 04413 04414 /* 04415 * call-seq: 04416 * str.to_f -> float 04417 * 04418 * Returns the result of interpreting leading characters in <i>str</i> as a 04419 * floating point number. Extraneous characters past the end of a valid number 04420 * are ignored. If there is not a valid number at the start of <i>str</i>, 04421 * <code>0.0</code> is returned. This method never raises an exception. 04422 * 04423 * "123.45e1".to_f #=> 1234.5 04424 * "45.67 degrees".to_f #=> 45.67 04425 * "thx1138".to_f #=> 0.0 04426 */ 04427 04428 static VALUE 04429 rb_str_to_f(VALUE str) 04430 { 04431 return DBL2NUM(rb_str_to_dbl(str, FALSE)); 04432 } 04433 04434 04435 /* 04436 * call-seq: 04437 * str.to_s -> str 04438 * str.to_str -> str 04439 * 04440 * Returns the receiver. 04441 */ 04442 04443 static VALUE 04444 rb_str_to_s(VALUE str) 04445 { 04446 if (rb_obj_class(str) != rb_cString) { 04447 return str_duplicate(rb_cString, str); 04448 } 04449 return str; 04450 } 04451 04452 #if 0 04453 static void 04454 str_cat_char(VALUE str, unsigned int c, rb_encoding *enc) 04455 { 04456 char s[RUBY_MAX_CHAR_LEN]; 04457 int n = rb_enc_codelen(c, enc); 04458 04459 rb_enc_mbcput(c, s, enc); 04460 rb_enc_str_buf_cat(str, s, n, enc); 04461 } 04462 #endif 04463 04464 #define CHAR_ESC_LEN 13 /* sizeof(\x{ hex of 32bit unsigned int } \0) */ 04465 04466 int 04467 rb_str_buf_cat_escaped_char(VALUE result, unsigned int c, int unicode_p) 04468 { 04469 char buf[CHAR_ESC_LEN + 1]; 04470 int l; 04471 04472 #if SIZEOF_INT > 4 04473 c &= 0xffffffff; 04474 #endif 04475 if (unicode_p) { 04476 if (c < 0x7F && ISPRINT(c)) { 04477 snprintf(buf, CHAR_ESC_LEN, "%c", c); 04478 } 04479 else if (c < 0x10000) { 04480 snprintf(buf, CHAR_ESC_LEN, "\\u%04X", c); 04481 } 04482 else { 04483 snprintf(buf, CHAR_ESC_LEN, "\\u{%X}", c); 04484 } 04485 } 04486 else { 04487 if (c < 0x100) { 04488 snprintf(buf, CHAR_ESC_LEN, "\\x%02X", c); 04489 } 04490 else { 04491 snprintf(buf, CHAR_ESC_LEN, "\\x{%X}", c); 04492 } 04493 } 04494 l = (int)strlen(buf); /* CHAR_ESC_LEN cannot exceed INT_MAX */ 04495 rb_str_buf_cat(result, buf, l); 04496 return l; 04497 } 04498 04499 /* 04500 * call-seq: 04501 * str.inspect -> string 04502 * 04503 * Returns a printable version of _str_, surrounded by quote marks, 04504 * with special characters escaped. 04505 * 04506 * str = "hello" 04507 * str[3] = "\b" 04508 * str.inspect #=> "\"hel\\bo\"" 04509 */ 04510 04511 VALUE 04512 rb_str_inspect(VALUE str) 04513 { 04514 rb_encoding *enc = STR_ENC_GET(str); 04515 const char *p, *pend, *prev; 04516 char buf[CHAR_ESC_LEN + 1]; 04517 VALUE result = rb_str_buf_new(0); 04518 rb_encoding *resenc = rb_default_internal_encoding(); 04519 int unicode_p = rb_enc_unicode_p(enc); 04520 int asciicompat = rb_enc_asciicompat(enc); 04521 static rb_encoding *utf16, *utf32; 04522 04523 if (!utf16) utf16 = rb_enc_find("UTF-16"); 04524 if (!utf32) utf32 = rb_enc_find("UTF-32"); 04525 if (resenc == NULL) resenc = rb_default_external_encoding(); 04526 if (!rb_enc_asciicompat(resenc)) resenc = rb_usascii_encoding(); 04527 rb_enc_associate(result, resenc); 04528 str_buf_cat2(result, "\""); 04529 04530 p = RSTRING_PTR(str); pend = RSTRING_END(str); 04531 prev = p; 04532 if (enc == utf16) { 04533 const unsigned char *q = (const unsigned char *)p; 04534 if (q[0] == 0xFE && q[1] == 0xFF) 04535 enc = rb_enc_find("UTF-16BE"); 04536 else if (q[0] == 0xFF && q[1] == 0xFE) 04537 enc = rb_enc_find("UTF-16LE"); 04538 else 04539 unicode_p = 0; 04540 } 04541 else if (enc == utf32) { 04542 const unsigned char *q = (const unsigned char *)p; 04543 if (q[0] == 0 && q[1] == 0 && q[2] == 0xFE && q[3] == 0xFF) 04544 enc = rb_enc_find("UTF-32BE"); 04545 else if (q[3] == 0 && q[2] == 0 && q[1] == 0xFE && q[0] == 0xFF) 04546 enc = rb_enc_find("UTF-32LE"); 04547 else 04548 unicode_p = 0; 04549 } 04550 while (p < pend) { 04551 unsigned int c, cc; 04552 int n; 04553 04554 n = rb_enc_precise_mbclen(p, pend, enc); 04555 if (!MBCLEN_CHARFOUND_P(n)) { 04556 if (p > prev) str_buf_cat(result, prev, p - prev); 04557 n = rb_enc_mbminlen(enc); 04558 if (pend < p + n) 04559 n = (int)(pend - p); 04560 while (n--) { 04561 snprintf(buf, CHAR_ESC_LEN, "\\x%02X", *p & 0377); 04562 str_buf_cat(result, buf, strlen(buf)); 04563 prev = ++p; 04564 } 04565 continue; 04566 } 04567 n = MBCLEN_CHARFOUND_LEN(n); 04568 c = rb_enc_mbc_to_codepoint(p, pend, enc); 04569 p += n; 04570 if ((asciicompat || unicode_p) && 04571 (c == '"'|| c == '\\' || 04572 (c == '#' && 04573 p < pend && 04574 MBCLEN_CHARFOUND_P(rb_enc_precise_mbclen(p,pend,enc)) && 04575 (cc = rb_enc_codepoint(p,pend,enc), 04576 (cc == '$' || cc == '@' || cc == '{'))))) { 04577 if (p - n > prev) str_buf_cat(result, prev, p - n - prev); 04578 str_buf_cat2(result, "\\"); 04579 if (asciicompat || enc == resenc) { 04580 prev = p - n; 04581 continue; 04582 } 04583 } 04584 switch (c) { 04585 case '\n': cc = 'n'; break; 04586 case '\r': cc = 'r'; break; 04587 case '\t': cc = 't'; break; 04588 case '\f': cc = 'f'; break; 04589 case '\013': cc = 'v'; break; 04590 case '\010': cc = 'b'; break; 04591 case '\007': cc = 'a'; break; 04592 case 033: cc = 'e'; break; 04593 default: cc = 0; break; 04594 } 04595 if (cc) { 04596 if (p - n > prev) str_buf_cat(result, prev, p - n - prev); 04597 buf[0] = '\\'; 04598 buf[1] = (char)cc; 04599 str_buf_cat(result, buf, 2); 04600 prev = p; 04601 continue; 04602 } 04603 if ((enc == resenc && rb_enc_isprint(c, enc)) || 04604 (asciicompat && rb_enc_isascii(c, enc) && ISPRINT(c))) { 04605 continue; 04606 } 04607 else { 04608 if (p - n > prev) str_buf_cat(result, prev, p - n - prev); 04609 rb_str_buf_cat_escaped_char(result, c, unicode_p); 04610 prev = p; 04611 continue; 04612 } 04613 } 04614 if (p > prev) str_buf_cat(result, prev, p - prev); 04615 str_buf_cat2(result, "\""); 04616 04617 OBJ_INFECT(result, str); 04618 return result; 04619 } 04620 04621 #define IS_EVSTR(p,e) ((p) < (e) && (*(p) == '$' || *(p) == '@' || *(p) == '{')) 04622 04623 /* 04624 * call-seq: 04625 * str.dump -> new_str 04626 * 04627 * Produces a version of +str+ with all non-printing characters replaced by 04628 * <code>\nnn</code> notation and all special characters escaped. 04629 * 04630 * "hello \n ''".dump #=> "\"hello \\n ''\" 04631 */ 04632 04633 VALUE 04634 rb_str_dump(VALUE str) 04635 { 04636 rb_encoding *enc = rb_enc_get(str); 04637 long len; 04638 const char *p, *pend; 04639 char *q, *qend; 04640 VALUE result; 04641 int u8 = (enc == rb_utf8_encoding()); 04642 04643 len = 2; /* "" */ 04644 p = RSTRING_PTR(str); pend = p + RSTRING_LEN(str); 04645 while (p < pend) { 04646 unsigned char c = *p++; 04647 switch (c) { 04648 case '"': case '\\': 04649 case '\n': case '\r': 04650 case '\t': case '\f': 04651 case '\013': case '\010': case '\007': case '\033': 04652 len += 2; 04653 break; 04654 04655 case '#': 04656 len += IS_EVSTR(p, pend) ? 2 : 1; 04657 break; 04658 04659 default: 04660 if (ISPRINT(c)) { 04661 len++; 04662 } 04663 else { 04664 if (u8) { /* \u{NN} */ 04665 int n = rb_enc_precise_mbclen(p-1, pend, enc); 04666 if (MBCLEN_CHARFOUND_P(n-1)) { 04667 unsigned int cc = rb_enc_mbc_to_codepoint(p-1, pend, enc); 04668 while (cc >>= 4) len++; 04669 len += 5; 04670 p += MBCLEN_CHARFOUND_LEN(n)-1; 04671 break; 04672 } 04673 } 04674 len += 4; /* \xNN */ 04675 } 04676 break; 04677 } 04678 } 04679 if (!rb_enc_asciicompat(enc)) { 04680 len += 19; /* ".force_encoding('')" */ 04681 len += strlen(enc->name); 04682 } 04683 04684 result = rb_str_new5(str, 0, len); 04685 p = RSTRING_PTR(str); pend = p + RSTRING_LEN(str); 04686 q = RSTRING_PTR(result); qend = q + len + 1; 04687 04688 *q++ = '"'; 04689 while (p < pend) { 04690 unsigned char c = *p++; 04691 04692 if (c == '"' || c == '\\') { 04693 *q++ = '\\'; 04694 *q++ = c; 04695 } 04696 else if (c == '#') { 04697 if (IS_EVSTR(p, pend)) *q++ = '\\'; 04698 *q++ = '#'; 04699 } 04700 else if (c == '\n') { 04701 *q++ = '\\'; 04702 *q++ = 'n'; 04703 } 04704 else if (c == '\r') { 04705 *q++ = '\\'; 04706 *q++ = 'r'; 04707 } 04708 else if (c == '\t') { 04709 *q++ = '\\'; 04710 *q++ = 't'; 04711 } 04712 else if (c == '\f') { 04713 *q++ = '\\'; 04714 *q++ = 'f'; 04715 } 04716 else if (c == '\013') { 04717 *q++ = '\\'; 04718 *q++ = 'v'; 04719 } 04720 else if (c == '\010') { 04721 *q++ = '\\'; 04722 *q++ = 'b'; 04723 } 04724 else if (c == '\007') { 04725 *q++ = '\\'; 04726 *q++ = 'a'; 04727 } 04728 else if (c == '\033') { 04729 *q++ = '\\'; 04730 *q++ = 'e'; 04731 } 04732 else if (ISPRINT(c)) { 04733 *q++ = c; 04734 } 04735 else { 04736 *q++ = '\\'; 04737 if (u8) { 04738 int n = rb_enc_precise_mbclen(p-1, pend, enc) - 1; 04739 if (MBCLEN_CHARFOUND_P(n)) { 04740 int cc = rb_enc_mbc_to_codepoint(p-1, pend, enc); 04741 p += n; 04742 snprintf(q, qend-q, "u{%x}", cc); 04743 q += strlen(q); 04744 continue; 04745 } 04746 } 04747 snprintf(q, qend-q, "x%02X", c); 04748 q += 3; 04749 } 04750 } 04751 *q++ = '"'; 04752 *q = '\0'; 04753 if (!rb_enc_asciicompat(enc)) { 04754 snprintf(q, qend-q, ".force_encoding(\"%s\")", enc->name); 04755 enc = rb_ascii8bit_encoding(); 04756 } 04757 OBJ_INFECT(result, str); 04758 /* result from dump is ASCII */ 04759 rb_enc_associate(result, enc); 04760 ENC_CODERANGE_SET(result, ENC_CODERANGE_7BIT); 04761 return result; 04762 } 04763 04764 04765 static void 04766 rb_str_check_dummy_enc(rb_encoding *enc) 04767 { 04768 if (rb_enc_dummy_p(enc)) { 04769 rb_raise(rb_eEncCompatError, "incompatible encoding with this operation: %s", 04770 rb_enc_name(enc)); 04771 } 04772 } 04773 04774 /* 04775 * call-seq: 04776 * str.upcase! -> str or nil 04777 * 04778 * Upcases the contents of <i>str</i>, returning <code>nil</code> if no changes 04779 * were made. 04780 * Note: case replacement is effective only in ASCII region. 04781 */ 04782 04783 static VALUE 04784 rb_str_upcase_bang(VALUE str) 04785 { 04786 rb_encoding *enc; 04787 char *s, *send; 04788 int modify = 0; 04789 int n; 04790 04791 str_modify_keep_cr(str); 04792 enc = STR_ENC_GET(str); 04793 rb_str_check_dummy_enc(enc); 04794 s = RSTRING_PTR(str); send = RSTRING_END(str); 04795 if (single_byte_optimizable(str)) { 04796 while (s < send) { 04797 unsigned int c = *(unsigned char*)s; 04798 04799 if (rb_enc_isascii(c, enc) && 'a' <= c && c <= 'z') { 04800 *s = 'A' + (c - 'a'); 04801 modify = 1; 04802 } 04803 s++; 04804 } 04805 } 04806 else { 04807 int ascompat = rb_enc_asciicompat(enc); 04808 04809 while (s < send) { 04810 unsigned int c; 04811 04812 if (ascompat && (c = *(unsigned char*)s) < 0x80) { 04813 if (rb_enc_isascii(c, enc) && 'a' <= c && c <= 'z') { 04814 *s = 'A' + (c - 'a'); 04815 modify = 1; 04816 } 04817 s++; 04818 } 04819 else { 04820 c = rb_enc_codepoint_len(s, send, &n, enc); 04821 if (rb_enc_islower(c, enc)) { 04822 /* assuming toupper returns codepoint with same size */ 04823 rb_enc_mbcput(rb_enc_toupper(c, enc), s, enc); 04824 modify = 1; 04825 } 04826 s += n; 04827 } 04828 } 04829 } 04830 04831 if (modify) return str; 04832 return Qnil; 04833 } 04834 04835 04836 /* 04837 * call-seq: 04838 * str.upcase -> new_str 04839 * 04840 * Returns a copy of <i>str</i> with all lowercase letters replaced with their 04841 * uppercase counterparts. The operation is locale insensitive---only 04842 * characters ``a'' to ``z'' are affected. 04843 * Note: case replacement is effective only in ASCII region. 04844 * 04845 * "hEllO".upcase #=> "HELLO" 04846 */ 04847 04848 static VALUE 04849 rb_str_upcase(VALUE str) 04850 { 04851 str = rb_str_dup(str); 04852 rb_str_upcase_bang(str); 04853 return str; 04854 } 04855 04856 04857 /* 04858 * call-seq: 04859 * str.downcase! -> str or nil 04860 * 04861 * Downcases the contents of <i>str</i>, returning <code>nil</code> if no 04862 * changes were made. 04863 * Note: case replacement is effective only in ASCII region. 04864 */ 04865 04866 static VALUE 04867 rb_str_downcase_bang(VALUE str) 04868 { 04869 rb_encoding *enc; 04870 char *s, *send; 04871 int modify = 0; 04872 04873 str_modify_keep_cr(str); 04874 enc = STR_ENC_GET(str); 04875 rb_str_check_dummy_enc(enc); 04876 s = RSTRING_PTR(str); send = RSTRING_END(str); 04877 if (single_byte_optimizable(str)) { 04878 while (s < send) { 04879 unsigned int c = *(unsigned char*)s; 04880 04881 if (rb_enc_isascii(c, enc) && 'A' <= c && c <= 'Z') { 04882 *s = 'a' + (c - 'A'); 04883 modify = 1; 04884 } 04885 s++; 04886 } 04887 } 04888 else { 04889 int ascompat = rb_enc_asciicompat(enc); 04890 04891 while (s < send) { 04892 unsigned int c; 04893 int n; 04894 04895 if (ascompat && (c = *(unsigned char*)s) < 0x80) { 04896 if (rb_enc_isascii(c, enc) && 'A' <= c && c <= 'Z') { 04897 *s = 'a' + (c - 'A'); 04898 modify = 1; 04899 } 04900 s++; 04901 } 04902 else { 04903 c = rb_enc_codepoint_len(s, send, &n, enc); 04904 if (rb_enc_isupper(c, enc)) { 04905 /* assuming toupper returns codepoint with same size */ 04906 rb_enc_mbcput(rb_enc_tolower(c, enc), s, enc); 04907 modify = 1; 04908 } 04909 s += n; 04910 } 04911 } 04912 } 04913 04914 if (modify) return str; 04915 return Qnil; 04916 } 04917 04918 04919 /* 04920 * call-seq: 04921 * str.downcase -> new_str 04922 * 04923 * Returns a copy of <i>str</i> with all uppercase letters replaced with their 04924 * lowercase counterparts. The operation is locale insensitive---only 04925 * characters ``A'' to ``Z'' are affected. 04926 * Note: case replacement is effective only in ASCII region. 04927 * 04928 * "hEllO".downcase #=> "hello" 04929 */ 04930 04931 static VALUE 04932 rb_str_downcase(VALUE str) 04933 { 04934 str = rb_str_dup(str); 04935 rb_str_downcase_bang(str); 04936 return str; 04937 } 04938 04939 04940 /* 04941 * call-seq: 04942 * str.capitalize! -> str or nil 04943 * 04944 * Modifies <i>str</i> by converting the first character to uppercase and the 04945 * remainder to lowercase. Returns <code>nil</code> if no changes are made. 04946 * Note: case conversion is effective only in ASCII region. 04947 * 04948 * a = "hello" 04949 * a.capitalize! #=> "Hello" 04950 * a #=> "Hello" 04951 * a.capitalize! #=> nil 04952 */ 04953 04954 static VALUE 04955 rb_str_capitalize_bang(VALUE str) 04956 { 04957 rb_encoding *enc; 04958 char *s, *send; 04959 int modify = 0; 04960 unsigned int c; 04961 int n; 04962 04963 str_modify_keep_cr(str); 04964 enc = STR_ENC_GET(str); 04965 rb_str_check_dummy_enc(enc); 04966 if (RSTRING_LEN(str) == 0 || !RSTRING_PTR(str)) return Qnil; 04967 s = RSTRING_PTR(str); send = RSTRING_END(str); 04968 04969 c = rb_enc_codepoint_len(s, send, &n, enc); 04970 if (rb_enc_islower(c, enc)) { 04971 rb_enc_mbcput(rb_enc_toupper(c, enc), s, enc); 04972 modify = 1; 04973 } 04974 s += n; 04975 while (s < send) { 04976 c = rb_enc_codepoint_len(s, send, &n, enc); 04977 if (rb_enc_isupper(c, enc)) { 04978 rb_enc_mbcput(rb_enc_tolower(c, enc), s, enc); 04979 modify = 1; 04980 } 04981 s += n; 04982 } 04983 04984 if (modify) return str; 04985 return Qnil; 04986 } 04987 04988 04989 /* 04990 * call-seq: 04991 * str.capitalize -> new_str 04992 * 04993 * Returns a copy of <i>str</i> with the first character converted to uppercase 04994 * and the remainder to lowercase. 04995 * Note: case conversion is effective only in ASCII region. 04996 * 04997 * "hello".capitalize #=> "Hello" 04998 * "HELLO".capitalize #=> "Hello" 04999 * "123ABC".capitalize #=> "123abc" 05000 */ 05001 05002 static VALUE 05003 rb_str_capitalize(VALUE str) 05004 { 05005 str = rb_str_dup(str); 05006 rb_str_capitalize_bang(str); 05007 return str; 05008 } 05009 05010 05011 /* 05012 * call-seq: 05013 * str.swapcase! -> str or nil 05014 * 05015 * Equivalent to <code>String#swapcase</code>, but modifies the receiver in 05016 * place, returning <i>str</i>, or <code>nil</code> if no changes were made. 05017 * Note: case conversion is effective only in ASCII region. 05018 */ 05019 05020 static VALUE 05021 rb_str_swapcase_bang(VALUE str) 05022 { 05023 rb_encoding *enc; 05024 char *s, *send; 05025 int modify = 0; 05026 int n; 05027 05028 str_modify_keep_cr(str); 05029 enc = STR_ENC_GET(str); 05030 rb_str_check_dummy_enc(enc); 05031 s = RSTRING_PTR(str); send = RSTRING_END(str); 05032 while (s < send) { 05033 unsigned int c = rb_enc_codepoint_len(s, send, &n, enc); 05034 05035 if (rb_enc_isupper(c, enc)) { 05036 /* assuming toupper returns codepoint with same size */ 05037 rb_enc_mbcput(rb_enc_tolower(c, enc), s, enc); 05038 modify = 1; 05039 } 05040 else if (rb_enc_islower(c, enc)) { 05041 /* assuming tolower returns codepoint with same size */ 05042 rb_enc_mbcput(rb_enc_toupper(c, enc), s, enc); 05043 modify = 1; 05044 } 05045 s += n; 05046 } 05047 05048 if (modify) return str; 05049 return Qnil; 05050 } 05051 05052 05053 /* 05054 * call-seq: 05055 * str.swapcase -> new_str 05056 * 05057 * Returns a copy of <i>str</i> with uppercase alphabetic characters converted 05058 * to lowercase and lowercase characters converted to uppercase. 05059 * Note: case conversion is effective only in ASCII region. 05060 * 05061 * "Hello".swapcase #=> "hELLO" 05062 * "cYbEr_PuNk11".swapcase #=> "CyBeR_pUnK11" 05063 */ 05064 05065 static VALUE 05066 rb_str_swapcase(VALUE str) 05067 { 05068 str = rb_str_dup(str); 05069 rb_str_swapcase_bang(str); 05070 return str; 05071 } 05072 05073 typedef unsigned char *USTR; 05074 05075 struct tr { 05076 int gen; 05077 unsigned int now, max; 05078 char *p, *pend; 05079 }; 05080 05081 static unsigned int 05082 trnext(struct tr *t, rb_encoding *enc) 05083 { 05084 int n; 05085 05086 for (;;) { 05087 if (!t->gen) { 05088 nextpart: 05089 if (t->p == t->pend) return -1; 05090 if (rb_enc_ascget(t->p, t->pend, &n, enc) == '\\' && t->p + n < t->pend) { 05091 t->p += n; 05092 } 05093 t->now = rb_enc_codepoint_len(t->p, t->pend, &n, enc); 05094 t->p += n; 05095 if (rb_enc_ascget(t->p, t->pend, &n, enc) == '-' && t->p + n < t->pend) { 05096 t->p += n; 05097 if (t->p < t->pend) { 05098 unsigned int c = rb_enc_codepoint_len(t->p, t->pend, &n, enc); 05099 t->p += n; 05100 if (t->now > c) { 05101 if (t->now < 0x80 && c < 0x80) { 05102 rb_raise(rb_eArgError, 05103 "invalid range \"%c-%c\" in string transliteration", 05104 t->now, c); 05105 } 05106 else { 05107 rb_raise(rb_eArgError, "invalid range in string transliteration"); 05108 } 05109 continue; /* not reached */ 05110 } 05111 t->gen = 1; 05112 t->max = c; 05113 } 05114 } 05115 return t->now; 05116 } 05117 else { 05118 while (ONIGENC_CODE_TO_MBCLEN(enc, ++t->now) <= 0) { 05119 if (t->now == t->max) { 05120 t->gen = 0; 05121 goto nextpart; 05122 } 05123 } 05124 if (t->now < t->max) { 05125 return t->now; 05126 } 05127 else { 05128 t->gen = 0; 05129 return t->max; 05130 } 05131 } 05132 } 05133 } 05134 05135 static VALUE rb_str_delete_bang(int,VALUE*,VALUE); 05136 05137 static VALUE 05138 tr_trans(VALUE str, VALUE src, VALUE repl, int sflag) 05139 { 05140 const unsigned int errc = -1; 05141 unsigned int trans[256]; 05142 rb_encoding *enc, *e1, *e2; 05143 struct tr trsrc, trrepl; 05144 int cflag = 0; 05145 unsigned int c, c0, last = 0; 05146 int modify = 0, i, l; 05147 char *s, *send; 05148 VALUE hash = 0; 05149 int singlebyte = single_byte_optimizable(str); 05150 int cr; 05151 05152 #define CHECK_IF_ASCII(c) \ 05153 (void)((cr == ENC_CODERANGE_7BIT && !rb_isascii(c)) ? \ 05154 (cr = ENC_CODERANGE_VALID) : 0) 05155 05156 StringValue(src); 05157 StringValue(repl); 05158 if (RSTRING_LEN(str) == 0 || !RSTRING_PTR(str)) return Qnil; 05159 if (RSTRING_LEN(repl) == 0) { 05160 return rb_str_delete_bang(1, &src, str); 05161 } 05162 05163 cr = ENC_CODERANGE(str); 05164 e1 = rb_enc_check(str, src); 05165 e2 = rb_enc_check(str, repl); 05166 if (e1 == e2) { 05167 enc = e1; 05168 } 05169 else { 05170 enc = rb_enc_check(src, repl); 05171 } 05172 trsrc.p = RSTRING_PTR(src); trsrc.pend = trsrc.p + RSTRING_LEN(src); 05173 if (RSTRING_LEN(src) > 1 && 05174 rb_enc_ascget(trsrc.p, trsrc.pend, &l, enc) == '^' && 05175 trsrc.p + l < trsrc.pend) { 05176 cflag = 1; 05177 trsrc.p += l; 05178 } 05179 trrepl.p = RSTRING_PTR(repl); 05180 trrepl.pend = trrepl.p + RSTRING_LEN(repl); 05181 trsrc.gen = trrepl.gen = 0; 05182 trsrc.now = trrepl.now = 0; 05183 trsrc.max = trrepl.max = 0; 05184 05185 if (cflag) { 05186 for (i=0; i<256; i++) { 05187 trans[i] = 1; 05188 } 05189 while ((c = trnext(&trsrc, enc)) != errc) { 05190 if (c < 256) { 05191 trans[c] = errc; 05192 } 05193 else { 05194 if (!hash) hash = rb_hash_new(); 05195 rb_hash_aset(hash, UINT2NUM(c), Qtrue); 05196 } 05197 } 05198 while ((c = trnext(&trrepl, enc)) != errc) 05199 /* retrieve last replacer */; 05200 last = trrepl.now; 05201 for (i=0; i<256; i++) { 05202 if (trans[i] != errc) { 05203 trans[i] = last; 05204 } 05205 } 05206 } 05207 else { 05208 unsigned int r; 05209 05210 for (i=0; i<256; i++) { 05211 trans[i] = errc; 05212 } 05213 while ((c = trnext(&trsrc, enc)) != errc) { 05214 r = trnext(&trrepl, enc); 05215 if (r == errc) r = trrepl.now; 05216 if (c < 256) { 05217 trans[c] = r; 05218 if (rb_enc_codelen(r, enc) != 1) singlebyte = 0; 05219 } 05220 else { 05221 if (!hash) hash = rb_hash_new(); 05222 rb_hash_aset(hash, UINT2NUM(c), UINT2NUM(r)); 05223 } 05224 } 05225 } 05226 05227 if (cr == ENC_CODERANGE_VALID) 05228 cr = ENC_CODERANGE_7BIT; 05229 str_modify_keep_cr(str); 05230 s = RSTRING_PTR(str); send = RSTRING_END(str); 05231 if (sflag) { 05232 int clen, tlen; 05233 long offset, max = RSTRING_LEN(str); 05234 unsigned int save = -1; 05235 char *buf = ALLOC_N(char, max), *t = buf; 05236 05237 while (s < send) { 05238 int may_modify = 0; 05239 05240 c0 = c = rb_enc_codepoint_len(s, send, &clen, e1); 05241 tlen = enc == e1 ? clen : rb_enc_codelen(c, enc); 05242 05243 s += clen; 05244 if (c < 256) { 05245 c = trans[c]; 05246 } 05247 else if (hash) { 05248 VALUE tmp = rb_hash_lookup(hash, UINT2NUM(c)); 05249 if (NIL_P(tmp)) { 05250 if (cflag) c = last; 05251 else c = errc; 05252 } 05253 else if (cflag) c = errc; 05254 else c = NUM2INT(tmp); 05255 } 05256 else { 05257 c = errc; 05258 } 05259 if (c != (unsigned int)-1) { 05260 if (save == c) { 05261 CHECK_IF_ASCII(c); 05262 continue; 05263 } 05264 save = c; 05265 tlen = rb_enc_codelen(c, enc); 05266 modify = 1; 05267 } 05268 else { 05269 save = -1; 05270 c = c0; 05271 if (enc != e1) may_modify = 1; 05272 } 05273 while (t - buf + tlen >= max) { 05274 offset = t - buf; 05275 max *= 2; 05276 REALLOC_N(buf, char, max); 05277 t = buf + offset; 05278 } 05279 rb_enc_mbcput(c, t, enc); 05280 if (may_modify && memcmp(s, t, tlen) != 0) { 05281 modify = 1; 05282 } 05283 CHECK_IF_ASCII(c); 05284 t += tlen; 05285 } 05286 if (!STR_EMBED_P(str)) { 05287 xfree(RSTRING(str)->as.heap.ptr); 05288 } 05289 *t = '\0'; 05290 RSTRING(str)->as.heap.ptr = buf; 05291 RSTRING(str)->as.heap.len = t - buf; 05292 STR_SET_NOEMBED(str); 05293 RSTRING(str)->as.heap.aux.capa = max; 05294 } 05295 else if (rb_enc_mbmaxlen(enc) == 1 || (singlebyte && !hash)) { 05296 while (s < send) { 05297 c = (unsigned char)*s; 05298 if (trans[c] != errc) { 05299 if (!cflag) { 05300 c = trans[c]; 05301 *s = c; 05302 modify = 1; 05303 } 05304 else { 05305 *s = last; 05306 modify = 1; 05307 } 05308 } 05309 CHECK_IF_ASCII(c); 05310 s++; 05311 } 05312 } 05313 else { 05314 int clen, tlen, max = (int)(RSTRING_LEN(str) * 1.2); 05315 long offset; 05316 char *buf = ALLOC_N(char, max), *t = buf; 05317 05318 while (s < send) { 05319 int may_modify = 0; 05320 c0 = c = rb_enc_codepoint_len(s, send, &clen, e1); 05321 tlen = enc == e1 ? clen : rb_enc_codelen(c, enc); 05322 05323 if (c < 256) { 05324 c = trans[c]; 05325 } 05326 else if (hash) { 05327 VALUE tmp = rb_hash_lookup(hash, UINT2NUM(c)); 05328 if (NIL_P(tmp)) { 05329 if (cflag) c = last; 05330 else c = errc; 05331 } 05332 else if (cflag) c = errc; 05333 else c = NUM2INT(tmp); 05334 } 05335 else { 05336 c = cflag ? last : errc; 05337 } 05338 if (c != errc) { 05339 tlen = rb_enc_codelen(c, enc); 05340 modify = 1; 05341 } 05342 else { 05343 c = c0; 05344 if (enc != e1) may_modify = 1; 05345 } 05346 while (t - buf + tlen >= max) { 05347 offset = t - buf; 05348 max *= 2; 05349 REALLOC_N(buf, char, max); 05350 t = buf + offset; 05351 } 05352 if (s != t) { 05353 rb_enc_mbcput(c, t, enc); 05354 if (may_modify && memcmp(s, t, tlen) != 0) { 05355 modify = 1; 05356 } 05357 } 05358 CHECK_IF_ASCII(c); 05359 s += clen; 05360 t += tlen; 05361 } 05362 if (!STR_EMBED_P(str)) { 05363 xfree(RSTRING(str)->as.heap.ptr); 05364 } 05365 *t = '\0'; 05366 RSTRING(str)->as.heap.ptr = buf; 05367 RSTRING(str)->as.heap.len = t - buf; 05368 STR_SET_NOEMBED(str); 05369 RSTRING(str)->as.heap.aux.capa = max; 05370 } 05371 05372 if (modify) { 05373 if (cr != ENC_CODERANGE_BROKEN) 05374 ENC_CODERANGE_SET(str, cr); 05375 rb_enc_associate(str, enc); 05376 return str; 05377 } 05378 return Qnil; 05379 } 05380 05381 05382 /* 05383 * call-seq: 05384 * str.tr!(from_str, to_str) -> str or nil 05385 * 05386 * Translates <i>str</i> in place, using the same rules as 05387 * <code>String#tr</code>. Returns <i>str</i>, or <code>nil</code> if no 05388 * changes were made. 05389 */ 05390 05391 static VALUE 05392 rb_str_tr_bang(VALUE str, VALUE src, VALUE repl) 05393 { 05394 return tr_trans(str, src, repl, 0); 05395 } 05396 05397 05398 /* 05399 * call-seq: 05400 * str.tr(from_str, to_str) => new_str 05401 * 05402 * Returns a copy of +str+ with the characters in +from_str+ replaced by the 05403 * corresponding characters in +to_str+. If +to_str+ is shorter than 05404 * +from_str+, it is padded with its last character in order to maintain the 05405 * correspondence. 05406 * 05407 * "hello".tr('el', 'ip') #=> "hippo" 05408 * "hello".tr('aeiou', '*') #=> "h*ll*" 05409 * "hello".tr('aeiou', 'AA*') #=> "hAll*" 05410 * 05411 * Both strings may use the <code>c1-c2</code> notation to denote ranges of 05412 * characters, and +from_str+ may start with a <code>^</code>, which denotes 05413 * all characters except those listed. 05414 * 05415 * "hello".tr('a-y', 'b-z') #=> "ifmmp" 05416 * "hello".tr('^aeiou', '*') #=> "*e**o" 05417 * 05418 * The backslash character <code></code> can be used to escape 05419 * <code>^</code> or <code>-</code> and is otherwise ignored unless it 05420 * appears at the end of a range or the end of the +from_str+ or +to_str+: 05421 * 05422 * "hello^world".tr("\\^aeiou", "*") #=> "h*ll**w*rld" 05423 * "hello-world".tr("a\\-eo", "*") #=> "h*ll**w*rld" 05424 * 05425 * "hello\r\nworld".tr("\r", "") #=> "hello\nworld" 05426 * "hello\r\nworld".tr("\\r", "") #=> "hello\r\nwold" 05427 * "hello\r\nworld".tr("\\\r", "") #=> "hello\nworld" 05428 * 05429 * "X['\\b']".tr("X\\", "") #=> "['b']" 05430 * "X['\\b']".tr("X-\\]", "") #=> "'b'" 05431 */ 05432 05433 static VALUE 05434 rb_str_tr(VALUE str, VALUE src, VALUE repl) 05435 { 05436 str = rb_str_dup(str); 05437 tr_trans(str, src, repl, 0); 05438 return str; 05439 } 05440 05441 #define TR_TABLE_SIZE 257 05442 static void 05443 tr_setup_table(VALUE str, char stable[TR_TABLE_SIZE], int first, 05444 VALUE *tablep, VALUE *ctablep, rb_encoding *enc) 05445 { 05446 const unsigned int errc = -1; 05447 char buf[256]; 05448 struct tr tr; 05449 unsigned int c; 05450 VALUE table = 0, ptable = 0; 05451 int i, l, cflag = 0; 05452 05453 tr.p = RSTRING_PTR(str); tr.pend = tr.p + RSTRING_LEN(str); 05454 tr.gen = tr.now = tr.max = 0; 05455 05456 if (RSTRING_LEN(str) > 1 && rb_enc_ascget(tr.p, tr.pend, &l, enc) == '^') { 05457 cflag = 1; 05458 tr.p += l; 05459 } 05460 if (first) { 05461 for (i=0; i<256; i++) { 05462 stable[i] = 1; 05463 } 05464 stable[256] = cflag; 05465 } 05466 else if (stable[256] && !cflag) { 05467 stable[256] = 0; 05468 } 05469 for (i=0; i<256; i++) { 05470 buf[i] = cflag; 05471 } 05472 05473 while ((c = trnext(&tr, enc)) != errc) { 05474 if (c < 256) { 05475 buf[c & 0xff] = !cflag; 05476 } 05477 else { 05478 VALUE key = UINT2NUM(c); 05479 05480 if (!table && (first || *tablep || stable[256])) { 05481 if (cflag) { 05482 ptable = *ctablep; 05483 table = ptable ? ptable : rb_hash_new(); 05484 *ctablep = table; 05485 } 05486 else { 05487 table = rb_hash_new(); 05488 ptable = *tablep; 05489 *tablep = table; 05490 } 05491 } 05492 if (table && (!ptable || (cflag ^ !NIL_P(rb_hash_aref(ptable, key))))) { 05493 rb_hash_aset(table, key, Qtrue); 05494 } 05495 } 05496 } 05497 for (i=0; i<256; i++) { 05498 stable[i] = stable[i] && buf[i]; 05499 } 05500 if (!table && !cflag) { 05501 *tablep = 0; 05502 } 05503 } 05504 05505 05506 static int 05507 tr_find(unsigned int c, char table[TR_TABLE_SIZE], VALUE del, VALUE nodel) 05508 { 05509 if (c < 256) { 05510 return table[c] != 0; 05511 } 05512 else { 05513 VALUE v = UINT2NUM(c); 05514 05515 if (del) { 05516 if (!NIL_P(rb_hash_lookup(del, v)) && 05517 (!nodel || NIL_P(rb_hash_lookup(nodel, v)))) { 05518 return TRUE; 05519 } 05520 } 05521 else if (nodel && !NIL_P(rb_hash_lookup(nodel, v))) { 05522 return FALSE; 05523 } 05524 return table[256] ? TRUE : FALSE; 05525 } 05526 } 05527 05528 /* 05529 * call-seq: 05530 * str.delete!([other_str]+) -> str or nil 05531 * 05532 * Performs a <code>delete</code> operation in place, returning <i>str</i>, or 05533 * <code>nil</code> if <i>str</i> was not modified. 05534 */ 05535 05536 static VALUE 05537 rb_str_delete_bang(int argc, VALUE *argv, VALUE str) 05538 { 05539 char squeez[TR_TABLE_SIZE]; 05540 rb_encoding *enc = 0; 05541 char *s, *send, *t; 05542 VALUE del = 0, nodel = 0; 05543 int modify = 0; 05544 int i, ascompat, cr; 05545 05546 if (RSTRING_LEN(str) == 0 || !RSTRING_PTR(str)) return Qnil; 05547 rb_check_arity(argc, 1, UNLIMITED_ARGUMENTS); 05548 for (i=0; i<argc; i++) { 05549 VALUE s = argv[i]; 05550 05551 StringValue(s); 05552 enc = rb_enc_check(str, s); 05553 tr_setup_table(s, squeez, i==0, &del, &nodel, enc); 05554 } 05555 05556 str_modify_keep_cr(str); 05557 ascompat = rb_enc_asciicompat(enc); 05558 s = t = RSTRING_PTR(str); 05559 send = RSTRING_END(str); 05560 cr = ascompat ? ENC_CODERANGE_7BIT : ENC_CODERANGE_VALID; 05561 while (s < send) { 05562 unsigned int c; 05563 int clen; 05564 05565 if (ascompat && (c = *(unsigned char*)s) < 0x80) { 05566 if (squeez[c]) { 05567 modify = 1; 05568 } 05569 else { 05570 if (t != s) *t = c; 05571 t++; 05572 } 05573 s++; 05574 } 05575 else { 05576 c = rb_enc_codepoint_len(s, send, &clen, enc); 05577 05578 if (tr_find(c, squeez, del, nodel)) { 05579 modify = 1; 05580 } 05581 else { 05582 if (t != s) rb_enc_mbcput(c, t, enc); 05583 t += clen; 05584 if (cr == ENC_CODERANGE_7BIT) cr = ENC_CODERANGE_VALID; 05585 } 05586 s += clen; 05587 } 05588 } 05589 *t = '\0'; 05590 STR_SET_LEN(str, t - RSTRING_PTR(str)); 05591 ENC_CODERANGE_SET(str, cr); 05592 05593 if (modify) return str; 05594 return Qnil; 05595 } 05596 05597 05598 /* 05599 * call-seq: 05600 * str.delete([other_str]+) -> new_str 05601 * 05602 * Returns a copy of <i>str</i> with all characters in the intersection of its 05603 * arguments deleted. Uses the same rules for building the set of characters as 05604 * <code>String#count</code>. 05605 * 05606 * "hello".delete "l","lo" #=> "heo" 05607 * "hello".delete "lo" #=> "he" 05608 * "hello".delete "aeiou", "^e" #=> "hell" 05609 * "hello".delete "ej-m" #=> "ho" 05610 */ 05611 05612 static VALUE 05613 rb_str_delete(int argc, VALUE *argv, VALUE str) 05614 { 05615 str = rb_str_dup(str); 05616 rb_str_delete_bang(argc, argv, str); 05617 return str; 05618 } 05619 05620 05621 /* 05622 * call-seq: 05623 * str.squeeze!([other_str]*) -> str or nil 05624 * 05625 * Squeezes <i>str</i> in place, returning either <i>str</i>, or 05626 * <code>nil</code> if no changes were made. 05627 */ 05628 05629 static VALUE 05630 rb_str_squeeze_bang(int argc, VALUE *argv, VALUE str) 05631 { 05632 char squeez[TR_TABLE_SIZE]; 05633 rb_encoding *enc = 0; 05634 VALUE del = 0, nodel = 0; 05635 char *s, *send, *t; 05636 int i, modify = 0; 05637 int ascompat, singlebyte = single_byte_optimizable(str); 05638 unsigned int save; 05639 05640 if (argc == 0) { 05641 enc = STR_ENC_GET(str); 05642 } 05643 else { 05644 for (i=0; i<argc; i++) { 05645 VALUE s = argv[i]; 05646 05647 StringValue(s); 05648 enc = rb_enc_check(str, s); 05649 if (singlebyte && !single_byte_optimizable(s)) 05650 singlebyte = 0; 05651 tr_setup_table(s, squeez, i==0, &del, &nodel, enc); 05652 } 05653 } 05654 05655 str_modify_keep_cr(str); 05656 s = t = RSTRING_PTR(str); 05657 if (!s || RSTRING_LEN(str) == 0) return Qnil; 05658 send = RSTRING_END(str); 05659 save = -1; 05660 ascompat = rb_enc_asciicompat(enc); 05661 05662 if (singlebyte) { 05663 while (s < send) { 05664 unsigned int c = *(unsigned char*)s++; 05665 if (c != save || (argc > 0 && !squeez[c])) { 05666 *t++ = save = c; 05667 } 05668 } 05669 } else { 05670 while (s < send) { 05671 unsigned int c; 05672 int clen; 05673 05674 if (ascompat && (c = *(unsigned char*)s) < 0x80) { 05675 if (c != save || (argc > 0 && !squeez[c])) { 05676 *t++ = save = c; 05677 } 05678 s++; 05679 } 05680 else { 05681 c = rb_enc_codepoint_len(s, send, &clen, enc); 05682 05683 if (c != save || (argc > 0 && !tr_find(c, squeez, del, nodel))) { 05684 if (t != s) rb_enc_mbcput(c, t, enc); 05685 save = c; 05686 t += clen; 05687 } 05688 s += clen; 05689 } 05690 } 05691 } 05692 05693 *t = '\0'; 05694 if (t - RSTRING_PTR(str) != RSTRING_LEN(str)) { 05695 STR_SET_LEN(str, t - RSTRING_PTR(str)); 05696 modify = 1; 05697 } 05698 05699 if (modify) return str; 05700 return Qnil; 05701 } 05702 05703 05704 /* 05705 * call-seq: 05706 * str.squeeze([other_str]*) -> new_str 05707 * 05708 * Builds a set of characters from the <i>other_str</i> parameter(s) using the 05709 * procedure described for <code>String#count</code>. Returns a new string 05710 * where runs of the same character that occur in this set are replaced by a 05711 * single character. If no arguments are given, all runs of identical 05712 * characters are replaced by a single character. 05713 * 05714 * "yellow moon".squeeze #=> "yelow mon" 05715 * " now is the".squeeze(" ") #=> " now is the" 05716 * "putters shoot balls".squeeze("m-z") #=> "puters shot balls" 05717 */ 05718 05719 static VALUE 05720 rb_str_squeeze(int argc, VALUE *argv, VALUE str) 05721 { 05722 str = rb_str_dup(str); 05723 rb_str_squeeze_bang(argc, argv, str); 05724 return str; 05725 } 05726 05727 05728 /* 05729 * call-seq: 05730 * str.tr_s!(from_str, to_str) -> str or nil 05731 * 05732 * Performs <code>String#tr_s</code> processing on <i>str</i> in place, 05733 * returning <i>str</i>, or <code>nil</code> if no changes were made. 05734 */ 05735 05736 static VALUE 05737 rb_str_tr_s_bang(VALUE str, VALUE src, VALUE repl) 05738 { 05739 return tr_trans(str, src, repl, 1); 05740 } 05741 05742 05743 /* 05744 * call-seq: 05745 * str.tr_s(from_str, to_str) -> new_str 05746 * 05747 * Processes a copy of <i>str</i> as described under <code>String#tr</code>, 05748 * then removes duplicate characters in regions that were affected by the 05749 * translation. 05750 * 05751 * "hello".tr_s('l', 'r') #=> "hero" 05752 * "hello".tr_s('el', '*') #=> "h*o" 05753 * "hello".tr_s('el', 'hx') #=> "hhxo" 05754 */ 05755 05756 static VALUE 05757 rb_str_tr_s(VALUE str, VALUE src, VALUE repl) 05758 { 05759 str = rb_str_dup(str); 05760 tr_trans(str, src, repl, 1); 05761 return str; 05762 } 05763 05764 05765 /* 05766 * call-seq: 05767 * str.count([other_str]+) -> fixnum 05768 * 05769 * Each +other_str+ parameter defines a set of characters to count. The 05770 * intersection of these sets defines the characters to count in +str+. Any 05771 * +other_str+ that starts with a caret <code>^</code> is negated. The 05772 * sequence <code>c1-c2</code> means all characters between c1 and c2. The 05773 * backslash character <code></code> can be used to escape <code>^</code> or 05774 * <code>-</code> and is otherwise ignored unless it appears at the end of a 05775 * sequence or the end of a +other_str+. 05776 * 05777 * a = "hello world" 05778 * a.count "lo" #=> 5 05779 * a.count "lo", "o" #=> 2 05780 * a.count "hello", "^l" #=> 4 05781 * a.count "ej-m" #=> 4 05782 * 05783 * "hello^world".count "\\^aeiou" #=> 4 05784 * "hello-world".count "a\\-eo" #=> 4 05785 * 05786 * c = "hello world\\r\\n" 05787 * c.count "\\" #=> 2 05788 * c.count "\\A" #=> 0 05789 * c.count "X-\\w" #=> 3 05790 */ 05791 05792 static VALUE 05793 rb_str_count(int argc, VALUE *argv, VALUE str) 05794 { 05795 char table[TR_TABLE_SIZE]; 05796 rb_encoding *enc = 0; 05797 VALUE del = 0, nodel = 0, tstr; 05798 char *s, *send; 05799 int i; 05800 int ascompat; 05801 05802 rb_check_arity(argc, 1, UNLIMITED_ARGUMENTS); 05803 05804 tstr = argv[0]; 05805 StringValue(tstr); 05806 enc = rb_enc_check(str, tstr); 05807 if (argc == 1) { 05808 const char *ptstr; 05809 if (RSTRING_LEN(tstr) == 1 && rb_enc_asciicompat(enc) && 05810 (ptstr = RSTRING_PTR(tstr), 05811 ONIGENC_IS_ALLOWED_REVERSE_MATCH(enc, (const unsigned char *)ptstr, (const unsigned char *)ptstr+1)) && 05812 !is_broken_string(str)) { 05813 int n = 0; 05814 int clen; 05815 unsigned char c = rb_enc_codepoint_len(ptstr, ptstr+1, &clen, enc); 05816 05817 s = RSTRING_PTR(str); 05818 if (!s || RSTRING_LEN(str) == 0) return INT2FIX(0); 05819 send = RSTRING_END(str); 05820 while (s < send) { 05821 if (*(unsigned char*)s++ == c) n++; 05822 } 05823 return INT2NUM(n); 05824 } 05825 } 05826 05827 tr_setup_table(tstr, table, TRUE, &del, &nodel, enc); 05828 for (i=1; i<argc; i++) { 05829 tstr = argv[i]; 05830 StringValue(tstr); 05831 enc = rb_enc_check(str, tstr); 05832 tr_setup_table(tstr, table, FALSE, &del, &nodel, enc); 05833 } 05834 05835 s = RSTRING_PTR(str); 05836 if (!s || RSTRING_LEN(str) == 0) return INT2FIX(0); 05837 send = RSTRING_END(str); 05838 ascompat = rb_enc_asciicompat(enc); 05839 i = 0; 05840 while (s < send) { 05841 unsigned int c; 05842 05843 if (ascompat && (c = *(unsigned char*)s) < 0x80) { 05844 if (table[c]) { 05845 i++; 05846 } 05847 s++; 05848 } 05849 else { 05850 int clen; 05851 c = rb_enc_codepoint_len(s, send, &clen, enc); 05852 if (tr_find(c, table, del, nodel)) { 05853 i++; 05854 } 05855 s += clen; 05856 } 05857 } 05858 05859 return INT2NUM(i); 05860 } 05861 05862 static const char isspacetable[256] = { 05863 0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 0, 0, 05864 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 05865 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 05866 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 05867 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 05868 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 05869 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 05870 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 05871 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 05872 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 05873 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 05874 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 05875 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 05876 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 05877 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 05878 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0 05879 }; 05880 05881 #define ascii_isspace(c) isspacetable[(unsigned char)(c)] 05882 05883 /* 05884 * call-seq: 05885 * str.split(pattern=$;, [limit]) -> anArray 05886 * 05887 * Divides <i>str</i> into substrings based on a delimiter, returning an array 05888 * of these substrings. 05889 * 05890 * If <i>pattern</i> is a <code>String</code>, then its contents are used as 05891 * the delimiter when splitting <i>str</i>. If <i>pattern</i> is a single 05892 * space, <i>str</i> is split on whitespace, with leading whitespace and runs 05893 * of contiguous whitespace characters ignored. 05894 * 05895 * If <i>pattern</i> is a <code>Regexp</code>, <i>str</i> is divided where the 05896 * pattern matches. Whenever the pattern matches a zero-length string, 05897 * <i>str</i> is split into individual characters. If <i>pattern</i> contains 05898 * groups, the respective matches will be returned in the array as well. 05899 * 05900 * If <i>pattern</i> is omitted, the value of <code>$;</code> is used. If 05901 * <code>$;</code> is <code>nil</code> (which is the default), <i>str</i> is 05902 * split on whitespace as if ` ' were specified. 05903 * 05904 * If the <i>limit</i> parameter is omitted, trailing null fields are 05905 * suppressed. If <i>limit</i> is a positive number, at most that number of 05906 * fields will be returned (if <i>limit</i> is <code>1</code>, the entire 05907 * string is returned as the only entry in an array). If negative, there is no 05908 * limit to the number of fields returned, and trailing null fields are not 05909 * suppressed. 05910 * 05911 * When the input +str+ is empty an empty Array is returned as the string is 05912 * considered to have no fields to split. 05913 * 05914 * " now's the time".split #=> ["now's", "the", "time"] 05915 * " now's the time".split(' ') #=> ["now's", "the", "time"] 05916 * " now's the time".split(/ /) #=> ["", "now's", "", "the", "time"] 05917 * "1, 2.34,56, 7".split(%r{,\s*}) #=> ["1", "2.34", "56", "7"] 05918 * "hello".split(//) #=> ["h", "e", "l", "l", "o"] 05919 * "hello".split(//, 3) #=> ["h", "e", "llo"] 05920 * "hi mom".split(%r{\s*}) #=> ["h", "i", "m", "o", "m"] 05921 * 05922 * "mellow yellow".split("ello") #=> ["m", "w y", "w"] 05923 * "1,2,,3,4,,".split(',') #=> ["1", "2", "", "3", "4"] 05924 * "1,2,,3,4,,".split(',', 4) #=> ["1", "2", "", "3,4,,"] 05925 * "1,2,,3,4,,".split(',', -4) #=> ["1", "2", "", "3", "4", "", ""] 05926 * 05927 * "".split(',', -1) #=> [] 05928 */ 05929 05930 static VALUE 05931 rb_str_split_m(int argc, VALUE *argv, VALUE str) 05932 { 05933 rb_encoding *enc; 05934 VALUE spat; 05935 VALUE limit; 05936 enum {awk, string, regexp} split_type; 05937 long beg, end, i = 0; 05938 int lim = 0; 05939 VALUE result, tmp; 05940 05941 if (rb_scan_args(argc, argv, "02", &spat, &limit) == 2) { 05942 lim = NUM2INT(limit); 05943 if (lim <= 0) limit = Qnil; 05944 else if (lim == 1) { 05945 if (RSTRING_LEN(str) == 0) 05946 return rb_ary_new2(0); 05947 return rb_ary_new3(1, str); 05948 } 05949 i = 1; 05950 } 05951 05952 enc = STR_ENC_GET(str); 05953 if (NIL_P(spat)) { 05954 if (!NIL_P(rb_fs)) { 05955 spat = rb_fs; 05956 goto fs_set; 05957 } 05958 split_type = awk; 05959 } 05960 else { 05961 fs_set: 05962 if (RB_TYPE_P(spat, T_STRING)) { 05963 rb_encoding *enc2 = STR_ENC_GET(spat); 05964 05965 split_type = string; 05966 if (RSTRING_LEN(spat) == 0) { 05967 /* Special case - split into chars */ 05968 spat = rb_reg_regcomp(spat); 05969 split_type = regexp; 05970 } 05971 else if (rb_enc_asciicompat(enc2) == 1) { 05972 if (RSTRING_LEN(spat) == 1 && RSTRING_PTR(spat)[0] == ' '){ 05973 split_type = awk; 05974 } 05975 } 05976 else { 05977 int l; 05978 if (rb_enc_ascget(RSTRING_PTR(spat), RSTRING_END(spat), &l, enc2) == ' ' && 05979 RSTRING_LEN(spat) == l) { 05980 split_type = awk; 05981 } 05982 } 05983 } 05984 else { 05985 spat = get_pat(spat, 1); 05986 split_type = regexp; 05987 } 05988 } 05989 05990 result = rb_ary_new(); 05991 beg = 0; 05992 if (split_type == awk) { 05993 char *ptr = RSTRING_PTR(str); 05994 char *eptr = RSTRING_END(str); 05995 char *bptr = ptr; 05996 int skip = 1; 05997 unsigned int c; 05998 05999 end = beg; 06000 if (is_ascii_string(str)) { 06001 while (ptr < eptr) { 06002 c = (unsigned char)*ptr++; 06003 if (skip) { 06004 if (ascii_isspace(c)) { 06005 beg = ptr - bptr; 06006 } 06007 else { 06008 end = ptr - bptr; 06009 skip = 0; 06010 if (!NIL_P(limit) && lim <= i) break; 06011 } 06012 } 06013 else if (ascii_isspace(c)) { 06014 rb_ary_push(result, rb_str_subseq(str, beg, end-beg)); 06015 skip = 1; 06016 beg = ptr - bptr; 06017 if (!NIL_P(limit)) ++i; 06018 } 06019 else { 06020 end = ptr - bptr; 06021 } 06022 } 06023 } 06024 else { 06025 while (ptr < eptr) { 06026 int n; 06027 06028 c = rb_enc_codepoint_len(ptr, eptr, &n, enc); 06029 ptr += n; 06030 if (skip) { 06031 if (rb_isspace(c)) { 06032 beg = ptr - bptr; 06033 } 06034 else { 06035 end = ptr - bptr; 06036 skip = 0; 06037 if (!NIL_P(limit) && lim <= i) break; 06038 } 06039 } 06040 else if (rb_isspace(c)) { 06041 rb_ary_push(result, rb_str_subseq(str, beg, end-beg)); 06042 skip = 1; 06043 beg = ptr - bptr; 06044 if (!NIL_P(limit)) ++i; 06045 } 06046 else { 06047 end = ptr - bptr; 06048 } 06049 } 06050 } 06051 } 06052 else if (split_type == string) { 06053 char *ptr = RSTRING_PTR(str); 06054 char *temp = ptr; 06055 char *eptr = RSTRING_END(str); 06056 char *sptr = RSTRING_PTR(spat); 06057 long slen = RSTRING_LEN(spat); 06058 06059 if (is_broken_string(str)) { 06060 rb_raise(rb_eArgError, "invalid byte sequence in %s", rb_enc_name(STR_ENC_GET(str))); 06061 } 06062 if (is_broken_string(spat)) { 06063 rb_raise(rb_eArgError, "invalid byte sequence in %s", rb_enc_name(STR_ENC_GET(spat))); 06064 } 06065 enc = rb_enc_check(str, spat); 06066 while (ptr < eptr && 06067 (end = rb_memsearch(sptr, slen, ptr, eptr - ptr, enc)) >= 0) { 06068 /* Check we are at the start of a char */ 06069 char *t = rb_enc_right_char_head(ptr, ptr + end, eptr, enc); 06070 if (t != ptr + end) { 06071 ptr = t; 06072 continue; 06073 } 06074 rb_ary_push(result, rb_str_subseq(str, ptr - temp, end)); 06075 ptr += end + slen; 06076 if (!NIL_P(limit) && lim <= ++i) break; 06077 } 06078 beg = ptr - temp; 06079 } 06080 else { 06081 char *ptr = RSTRING_PTR(str); 06082 long len = RSTRING_LEN(str); 06083 long start = beg; 06084 long idx; 06085 int last_null = 0; 06086 struct re_registers *regs; 06087 06088 while ((end = rb_reg_search(spat, str, start, 0)) >= 0) { 06089 regs = RMATCH_REGS(rb_backref_get()); 06090 if (start == end && BEG(0) == END(0)) { 06091 if (!ptr) { 06092 rb_ary_push(result, str_new_empty(str)); 06093 break; 06094 } 06095 else if (last_null == 1) { 06096 rb_ary_push(result, rb_str_subseq(str, beg, 06097 rb_enc_fast_mbclen(ptr+beg, 06098 ptr+len, 06099 enc))); 06100 beg = start; 06101 } 06102 else { 06103 if (ptr+start == ptr+len) 06104 start++; 06105 else 06106 start += rb_enc_fast_mbclen(ptr+start,ptr+len,enc); 06107 last_null = 1; 06108 continue; 06109 } 06110 } 06111 else { 06112 rb_ary_push(result, rb_str_subseq(str, beg, end-beg)); 06113 beg = start = END(0); 06114 } 06115 last_null = 0; 06116 06117 for (idx=1; idx < regs->num_regs; idx++) { 06118 if (BEG(idx) == -1) continue; 06119 if (BEG(idx) == END(idx)) 06120 tmp = str_new_empty(str); 06121 else 06122 tmp = rb_str_subseq(str, BEG(idx), END(idx)-BEG(idx)); 06123 rb_ary_push(result, tmp); 06124 } 06125 if (!NIL_P(limit) && lim <= ++i) break; 06126 } 06127 } 06128 if (RSTRING_LEN(str) > 0 && (!NIL_P(limit) || RSTRING_LEN(str) > beg || lim < 0)) { 06129 if (RSTRING_LEN(str) == beg) 06130 tmp = str_new_empty(str); 06131 else 06132 tmp = rb_str_subseq(str, beg, RSTRING_LEN(str)-beg); 06133 rb_ary_push(result, tmp); 06134 } 06135 if (NIL_P(limit) && lim == 0) { 06136 long len; 06137 while ((len = RARRAY_LEN(result)) > 0 && 06138 (tmp = RARRAY_PTR(result)[len-1], RSTRING_LEN(tmp) == 0)) 06139 rb_ary_pop(result); 06140 } 06141 06142 return result; 06143 } 06144 06145 VALUE 06146 rb_str_split(VALUE str, const char *sep0) 06147 { 06148 VALUE sep; 06149 06150 StringValue(str); 06151 sep = rb_str_new2(sep0); 06152 return rb_str_split_m(1, &sep, str); 06153 } 06154 06155 06156 static VALUE 06157 rb_str_enumerate_lines(int argc, VALUE *argv, VALUE str, int wantarray) 06158 { 06159 rb_encoding *enc; 06160 VALUE rs; 06161 unsigned int newline; 06162 const char *p, *pend, *s, *ptr; 06163 long len, rslen; 06164 VALUE line; 06165 int n; 06166 VALUE orig = str; 06167 VALUE UNINITIALIZED_VAR(ary); 06168 06169 if (argc == 0) { 06170 rs = rb_rs; 06171 } 06172 else { 06173 rb_scan_args(argc, argv, "01", &rs); 06174 } 06175 06176 if (rb_block_given_p()) { 06177 if (wantarray) { 06178 #if 0 /* next major */ 06179 rb_warn("given block not used"); 06180 ary = rb_ary_new(); 06181 #else 06182 rb_warning("passing a block to String#lines is deprecated"); 06183 wantarray = 0; 06184 #endif 06185 } 06186 } 06187 else { 06188 if (wantarray) 06189 ary = rb_ary_new(); 06190 else 06191 RETURN_ENUMERATOR(str, argc, argv); 06192 } 06193 06194 if (NIL_P(rs)) { 06195 if (wantarray) { 06196 rb_ary_push(ary, str); 06197 return ary; 06198 } 06199 else { 06200 rb_yield(str); 06201 return orig; 06202 } 06203 } 06204 str = rb_str_new4(str); 06205 ptr = p = s = RSTRING_PTR(str); 06206 pend = p + RSTRING_LEN(str); 06207 len = RSTRING_LEN(str); 06208 StringValue(rs); 06209 if (rs == rb_default_rs) { 06210 enc = rb_enc_get(str); 06211 while (p < pend) { 06212 char *p0; 06213 06214 p = memchr(p, '\n', pend - p); 06215 if (!p) break; 06216 p0 = rb_enc_left_char_head(s, p, pend, enc); 06217 if (!rb_enc_is_newline(p0, pend, enc)) { 06218 p++; 06219 continue; 06220 } 06221 p = p0 + rb_enc_mbclen(p0, pend, enc); 06222 line = rb_str_subseq(str, s - ptr, p - s); 06223 if (wantarray) 06224 rb_ary_push(ary, line); 06225 else 06226 rb_yield(line); 06227 str_mod_check(str, ptr, len); 06228 s = p; 06229 } 06230 goto finish; 06231 } 06232 06233 enc = rb_enc_check(str, rs); 06234 rslen = RSTRING_LEN(rs); 06235 if (rslen == 0) { 06236 newline = '\n'; 06237 } 06238 else { 06239 newline = rb_enc_codepoint(RSTRING_PTR(rs), RSTRING_END(rs), enc); 06240 } 06241 06242 while (p < pend) { 06243 unsigned int c = rb_enc_codepoint_len(p, pend, &n, enc); 06244 06245 again: 06246 if (rslen == 0 && c == newline) { 06247 p += n; 06248 if (p < pend && (c = rb_enc_codepoint_len(p, pend, &n, enc)) != newline) { 06249 goto again; 06250 } 06251 while (p < pend && rb_enc_codepoint(p, pend, enc) == newline) { 06252 p += n; 06253 } 06254 p -= n; 06255 } 06256 if (c == newline && 06257 (rslen <= 1 || 06258 (pend - p >= rslen && memcmp(RSTRING_PTR(rs), p, rslen) == 0))) { 06259 const char *pp = p + (rslen ? rslen : n); 06260 line = rb_str_subseq(str, s - ptr, pp - s); 06261 if (wantarray) 06262 rb_ary_push(ary, line); 06263 else 06264 rb_yield(line); 06265 str_mod_check(str, ptr, len); 06266 s = pp; 06267 } 06268 p += n; 06269 } 06270 06271 finish: 06272 if (s != pend) { 06273 line = rb_str_subseq(str, s - ptr, pend - s); 06274 if (wantarray) 06275 rb_ary_push(ary, line); 06276 else 06277 rb_yield(line); 06278 RB_GC_GUARD(str); 06279 } 06280 06281 if (wantarray) 06282 return ary; 06283 else 06284 return orig; 06285 } 06286 06287 /* 06288 * call-seq: 06289 * str.each_line(separator=$/) {|substr| block } -> str 06290 * str.each_line(separator=$/) -> an_enumerator 06291 * 06292 * Splits <i>str</i> using the supplied parameter as the record 06293 * separator (<code>$/</code> by default), passing each substring in 06294 * turn to the supplied block. If a zero-length record separator is 06295 * supplied, the string is split into paragraphs delimited by 06296 * multiple successive newlines. 06297 * 06298 * If no block is given, an enumerator is returned instead. 06299 * 06300 * print "Example one\n" 06301 * "hello\nworld".each_line {|s| p s} 06302 * print "Example two\n" 06303 * "hello\nworld".each_line('l') {|s| p s} 06304 * print "Example three\n" 06305 * "hello\n\n\nworld".each_line('') {|s| p s} 06306 * 06307 * <em>produces:</em> 06308 * 06309 * Example one 06310 * "hello\n" 06311 * "world" 06312 * Example two 06313 * "hel" 06314 * "l" 06315 * "o\nworl" 06316 * "d" 06317 * Example three 06318 * "hello\n\n\n" 06319 * "world" 06320 */ 06321 06322 static VALUE 06323 rb_str_each_line(int argc, VALUE *argv, VALUE str) 06324 { 06325 return rb_str_enumerate_lines(argc, argv, str, 0); 06326 } 06327 06328 /* 06329 * call-seq: 06330 * str.lines(separator=$/) -> an_array 06331 * 06332 * Returns an array of lines in <i>str</i> split using the supplied 06333 * record separator (<code>$/</code> by default). This is a 06334 * shorthand for <code>str.each_line(separator).to_a</code>. 06335 * 06336 * If a block is given, which is a deprecated form, works the same as 06337 * <code>each_line</code>. 06338 */ 06339 06340 static VALUE 06341 rb_str_lines(int argc, VALUE *argv, VALUE str) 06342 { 06343 return rb_str_enumerate_lines(argc, argv, str, 1); 06344 } 06345 06346 static VALUE 06347 rb_str_each_byte_size(VALUE str, VALUE args) 06348 { 06349 return LONG2FIX(RSTRING_LEN(str)); 06350 } 06351 06352 static VALUE 06353 rb_str_enumerate_bytes(VALUE str, int wantarray) 06354 { 06355 long i; 06356 VALUE UNINITIALIZED_VAR(ary); 06357 06358 if (rb_block_given_p()) { 06359 if (wantarray) { 06360 #if 0 /* next major */ 06361 rb_warn("given block not used"); 06362 ary = rb_ary_new(); 06363 #else 06364 rb_warning("passing a block to String#bytes is deprecated"); 06365 wantarray = 0; 06366 #endif 06367 } 06368 } 06369 else { 06370 if (wantarray) 06371 ary = rb_ary_new2(RSTRING_LEN(str)); 06372 else 06373 RETURN_SIZED_ENUMERATOR(str, 0, 0, rb_str_each_byte_size); 06374 } 06375 06376 for (i=0; i<RSTRING_LEN(str); i++) { 06377 if (wantarray) 06378 rb_ary_push(ary, INT2FIX(RSTRING_PTR(str)[i] & 0xff)); 06379 else 06380 rb_yield(INT2FIX(RSTRING_PTR(str)[i] & 0xff)); 06381 } 06382 if (wantarray) 06383 return ary; 06384 else 06385 return str; 06386 } 06387 06388 /* 06389 * call-seq: 06390 * str.each_byte {|fixnum| block } -> str 06391 * str.each_byte -> an_enumerator 06392 * 06393 * Passes each byte in <i>str</i> to the given block, or returns an 06394 * enumerator if no block is given. 06395 * 06396 * "hello".each_byte {|c| print c, ' ' } 06397 * 06398 * <em>produces:</em> 06399 * 06400 * 104 101 108 108 111 06401 */ 06402 06403 static VALUE 06404 rb_str_each_byte(VALUE str) 06405 { 06406 return rb_str_enumerate_bytes(str, 0); 06407 } 06408 06409 /* 06410 * call-seq: 06411 * str.bytes -> an_array 06412 * 06413 * Returns an array of bytes in <i>str</i>. This is a shorthand for 06414 * <code>str.each_byte.to_a</code>. 06415 * 06416 * If a block is given, which is a deprecated form, works the same as 06417 * <code>each_byte</code>. 06418 */ 06419 06420 static VALUE 06421 rb_str_bytes(VALUE str) 06422 { 06423 return rb_str_enumerate_bytes(str, 1); 06424 } 06425 06426 static VALUE 06427 rb_str_each_char_size(VALUE str) 06428 { 06429 long len = RSTRING_LEN(str); 06430 if (!single_byte_optimizable(str)) { 06431 const char *ptr = RSTRING_PTR(str); 06432 rb_encoding *enc = rb_enc_get(str); 06433 const char *end_ptr = ptr + len; 06434 for (len = 0; ptr < end_ptr; ++len) { 06435 ptr += rb_enc_mbclen(ptr, end_ptr, enc); 06436 } 06437 } 06438 return LONG2FIX(len); 06439 } 06440 06441 static VALUE 06442 rb_str_enumerate_chars(VALUE str, int wantarray) 06443 { 06444 VALUE orig = str; 06445 VALUE substr; 06446 long i, len, n; 06447 const char *ptr; 06448 rb_encoding *enc; 06449 VALUE UNINITIALIZED_VAR(ary); 06450 06451 if (rb_block_given_p()) { 06452 if (wantarray) { 06453 #if 0 /* next major */ 06454 rb_warn("given block not used"); 06455 ary = rb_ary_new(); 06456 #else 06457 rb_warning("passing a block to String#chars is deprecated"); 06458 wantarray = 0; 06459 #endif 06460 } 06461 } 06462 else { 06463 if (wantarray) 06464 ary = rb_ary_new(); 06465 else 06466 RETURN_SIZED_ENUMERATOR(str, 0, 0, rb_str_each_char_size); 06467 } 06468 06469 str = rb_str_new4(str); 06470 ptr = RSTRING_PTR(str); 06471 len = RSTRING_LEN(str); 06472 enc = rb_enc_get(str); 06473 switch (ENC_CODERANGE(str)) { 06474 case ENC_CODERANGE_VALID: 06475 case ENC_CODERANGE_7BIT: 06476 for (i = 0; i < len; i += n) { 06477 n = rb_enc_fast_mbclen(ptr + i, ptr + len, enc); 06478 substr = rb_str_subseq(str, i, n); 06479 if (wantarray) 06480 rb_ary_push(ary, substr); 06481 else 06482 rb_yield(substr); 06483 } 06484 break; 06485 default: 06486 for (i = 0; i < len; i += n) { 06487 n = rb_enc_mbclen(ptr + i, ptr + len, enc); 06488 substr = rb_str_subseq(str, i, n); 06489 if (wantarray) 06490 rb_ary_push(ary, substr); 06491 else 06492 rb_yield(substr); 06493 } 06494 } 06495 RB_GC_GUARD(str); 06496 if (wantarray) 06497 return ary; 06498 else 06499 return orig; 06500 } 06501 06502 /* 06503 * call-seq: 06504 * str.each_char {|cstr| block } -> str 06505 * str.each_char -> an_enumerator 06506 * 06507 * Passes each character in <i>str</i> to the given block, or returns 06508 * an enumerator if no block is given. 06509 * 06510 * "hello".each_char {|c| print c, ' ' } 06511 * 06512 * <em>produces:</em> 06513 * 06514 * h e l l o 06515 */ 06516 06517 static VALUE 06518 rb_str_each_char(VALUE str) 06519 { 06520 return rb_str_enumerate_chars(str, 0); 06521 } 06522 06523 /* 06524 * call-seq: 06525 * str.chars -> an_array 06526 * 06527 * Returns an array of characters in <i>str</i>. This is a shorthand 06528 * for <code>str.each_char.to_a</code>. 06529 * 06530 * If a block is given, which is a deprecated form, works the same as 06531 * <code>each_char</code>. 06532 */ 06533 06534 static VALUE 06535 rb_str_chars(VALUE str) 06536 { 06537 return rb_str_enumerate_chars(str, 1); 06538 } 06539 06540 06541 static VALUE 06542 rb_str_enumerate_codepoints(VALUE str, int wantarray) 06543 { 06544 VALUE orig = str; 06545 int n; 06546 unsigned int c; 06547 const char *ptr, *end; 06548 rb_encoding *enc; 06549 VALUE UNINITIALIZED_VAR(ary); 06550 06551 if (single_byte_optimizable(str)) 06552 return rb_str_enumerate_bytes(str, wantarray); 06553 06554 if (rb_block_given_p()) { 06555 if (wantarray) { 06556 #if 0 /* next major */ 06557 rb_warn("given block not used"); 06558 ary = rb_ary_new(); 06559 #else 06560 rb_warning("passing a block to String#codepoints is deprecated"); 06561 wantarray = 0; 06562 #endif 06563 } 06564 } 06565 else { 06566 if (wantarray) 06567 ary = rb_ary_new(); 06568 else 06569 RETURN_SIZED_ENUMERATOR(str, 0, 0, rb_str_each_char_size); 06570 } 06571 06572 str = rb_str_new4(str); 06573 ptr = RSTRING_PTR(str); 06574 end = RSTRING_END(str); 06575 enc = STR_ENC_GET(str); 06576 while (ptr < end) { 06577 c = rb_enc_codepoint_len(ptr, end, &n, enc); 06578 if (wantarray) 06579 rb_ary_push(ary, UINT2NUM(c)); 06580 else 06581 rb_yield(UINT2NUM(c)); 06582 ptr += n; 06583 } 06584 RB_GC_GUARD(str); 06585 if (wantarray) 06586 return ary; 06587 else 06588 return orig; 06589 } 06590 06591 /* 06592 * call-seq: 06593 * str.each_codepoint {|integer| block } -> str 06594 * str.each_codepoint -> an_enumerator 06595 * 06596 * Passes the <code>Integer</code> ordinal of each character in <i>str</i>, 06597 * also known as a <i>codepoint</i> when applied to Unicode strings to the 06598 * given block. 06599 * 06600 * If no block is given, an enumerator is returned instead. 06601 * 06602 * "hello\u0639".each_codepoint {|c| print c, ' ' } 06603 * 06604 * <em>produces:</em> 06605 * 06606 * 104 101 108 108 111 1593 06607 */ 06608 06609 static VALUE 06610 rb_str_each_codepoint(VALUE str) 06611 { 06612 return rb_str_enumerate_codepoints(str, 0); 06613 } 06614 06615 /* 06616 * call-seq: 06617 * str.codepoints -> an_array 06618 * 06619 * Returns an array of the <code>Integer</code> ordinals of the 06620 * characters in <i>str</i>. This is a shorthand for 06621 * <code>str.each_codepoint.to_a</code>. 06622 * 06623 * If a block is given, which is a deprecated form, works the same as 06624 * <code>each_codepoint</code>. 06625 */ 06626 06627 static VALUE 06628 rb_str_codepoints(VALUE str) 06629 { 06630 return rb_str_enumerate_codepoints(str, 1); 06631 } 06632 06633 06634 static long 06635 chopped_length(VALUE str) 06636 { 06637 rb_encoding *enc = STR_ENC_GET(str); 06638 const char *p, *p2, *beg, *end; 06639 06640 beg = RSTRING_PTR(str); 06641 end = beg + RSTRING_LEN(str); 06642 if (beg > end) return 0; 06643 p = rb_enc_prev_char(beg, end, end, enc); 06644 if (!p) return 0; 06645 if (p > beg && rb_enc_ascget(p, end, 0, enc) == '\n') { 06646 p2 = rb_enc_prev_char(beg, p, end, enc); 06647 if (p2 && rb_enc_ascget(p2, end, 0, enc) == '\r') p = p2; 06648 } 06649 return p - beg; 06650 } 06651 06652 /* 06653 * call-seq: 06654 * str.chop! -> str or nil 06655 * 06656 * Processes <i>str</i> as for <code>String#chop</code>, returning <i>str</i>, 06657 * or <code>nil</code> if <i>str</i> is the empty string. See also 06658 * <code>String#chomp!</code>. 06659 */ 06660 06661 static VALUE 06662 rb_str_chop_bang(VALUE str) 06663 { 06664 str_modify_keep_cr(str); 06665 if (RSTRING_LEN(str) > 0) { 06666 long len; 06667 len = chopped_length(str); 06668 STR_SET_LEN(str, len); 06669 RSTRING_PTR(str)[len] = '\0'; 06670 if (ENC_CODERANGE(str) != ENC_CODERANGE_7BIT) { 06671 ENC_CODERANGE_CLEAR(str); 06672 } 06673 return str; 06674 } 06675 return Qnil; 06676 } 06677 06678 06679 /* 06680 * call-seq: 06681 * str.chop -> new_str 06682 * 06683 * Returns a new <code>String</code> with the last character removed. If the 06684 * string ends with <code>\r\n</code>, both characters are removed. Applying 06685 * <code>chop</code> to an empty string returns an empty 06686 * string. <code>String#chomp</code> is often a safer alternative, as it leaves 06687 * the string unchanged if it doesn't end in a record separator. 06688 * 06689 * "string\r\n".chop #=> "string" 06690 * "string\n\r".chop #=> "string\n" 06691 * "string\n".chop #=> "string" 06692 * "string".chop #=> "strin" 06693 * "x".chop.chop #=> "" 06694 */ 06695 06696 static VALUE 06697 rb_str_chop(VALUE str) 06698 { 06699 return rb_str_subseq(str, 0, chopped_length(str)); 06700 } 06701 06702 06703 /* 06704 * call-seq: 06705 * str.chomp!(separator=$/) -> str or nil 06706 * 06707 * Modifies <i>str</i> in place as described for <code>String#chomp</code>, 06708 * returning <i>str</i>, or <code>nil</code> if no modifications were made. 06709 */ 06710 06711 static VALUE 06712 rb_str_chomp_bang(int argc, VALUE *argv, VALUE str) 06713 { 06714 rb_encoding *enc; 06715 VALUE rs; 06716 int newline; 06717 char *p, *pp, *e; 06718 long len, rslen; 06719 06720 str_modify_keep_cr(str); 06721 len = RSTRING_LEN(str); 06722 if (len == 0) return Qnil; 06723 p = RSTRING_PTR(str); 06724 e = p + len; 06725 if (argc == 0) { 06726 rs = rb_rs; 06727 if (rs == rb_default_rs) { 06728 smart_chomp: 06729 enc = rb_enc_get(str); 06730 if (rb_enc_mbminlen(enc) > 1) { 06731 pp = rb_enc_left_char_head(p, e-rb_enc_mbminlen(enc), e, enc); 06732 if (rb_enc_is_newline(pp, e, enc)) { 06733 e = pp; 06734 } 06735 pp = e - rb_enc_mbminlen(enc); 06736 if (pp >= p) { 06737 pp = rb_enc_left_char_head(p, pp, e, enc); 06738 if (rb_enc_ascget(pp, e, 0, enc) == '\r') { 06739 e = pp; 06740 } 06741 } 06742 if (e == RSTRING_END(str)) { 06743 return Qnil; 06744 } 06745 len = e - RSTRING_PTR(str); 06746 STR_SET_LEN(str, len); 06747 } 06748 else { 06749 if (RSTRING_PTR(str)[len-1] == '\n') { 06750 STR_DEC_LEN(str); 06751 if (RSTRING_LEN(str) > 0 && 06752 RSTRING_PTR(str)[RSTRING_LEN(str)-1] == '\r') { 06753 STR_DEC_LEN(str); 06754 } 06755 } 06756 else if (RSTRING_PTR(str)[len-1] == '\r') { 06757 STR_DEC_LEN(str); 06758 } 06759 else { 06760 return Qnil; 06761 } 06762 } 06763 RSTRING_PTR(str)[RSTRING_LEN(str)] = '\0'; 06764 return str; 06765 } 06766 } 06767 else { 06768 rb_scan_args(argc, argv, "01", &rs); 06769 } 06770 if (NIL_P(rs)) return Qnil; 06771 StringValue(rs); 06772 rslen = RSTRING_LEN(rs); 06773 if (rslen == 0) { 06774 while (len>0 && p[len-1] == '\n') { 06775 len--; 06776 if (len>0 && p[len-1] == '\r') 06777 len--; 06778 } 06779 if (len < RSTRING_LEN(str)) { 06780 STR_SET_LEN(str, len); 06781 RSTRING_PTR(str)[len] = '\0'; 06782 return str; 06783 } 06784 return Qnil; 06785 } 06786 if (rslen > len) return Qnil; 06787 newline = RSTRING_PTR(rs)[rslen-1]; 06788 if (rslen == 1 && newline == '\n') 06789 goto smart_chomp; 06790 06791 enc = rb_enc_check(str, rs); 06792 if (is_broken_string(rs)) { 06793 return Qnil; 06794 } 06795 pp = e - rslen; 06796 if (p[len-1] == newline && 06797 (rslen <= 1 || 06798 memcmp(RSTRING_PTR(rs), pp, rslen) == 0)) { 06799 if (rb_enc_left_char_head(p, pp, e, enc) != pp) 06800 return Qnil; 06801 if (ENC_CODERANGE(str) != ENC_CODERANGE_7BIT) { 06802 ENC_CODERANGE_CLEAR(str); 06803 } 06804 STR_SET_LEN(str, RSTRING_LEN(str) - rslen); 06805 RSTRING_PTR(str)[RSTRING_LEN(str)] = '\0'; 06806 return str; 06807 } 06808 return Qnil; 06809 } 06810 06811 06812 /* 06813 * call-seq: 06814 * str.chomp(separator=$/) -> new_str 06815 * 06816 * Returns a new <code>String</code> with the given record separator removed 06817 * from the end of <i>str</i> (if present). If <code>$/</code> has not been 06818 * changed from the default Ruby record separator, then <code>chomp</code> also 06819 * removes carriage return characters (that is it will remove <code>\n</code>, 06820 * <code>\r</code>, and <code>\r\n</code>). 06821 * 06822 * "hello".chomp #=> "hello" 06823 * "hello\n".chomp #=> "hello" 06824 * "hello\r\n".chomp #=> "hello" 06825 * "hello\n\r".chomp #=> "hello\n" 06826 * "hello\r".chomp #=> "hello" 06827 * "hello \n there".chomp #=> "hello \n there" 06828 * "hello".chomp("llo") #=> "he" 06829 */ 06830 06831 static VALUE 06832 rb_str_chomp(int argc, VALUE *argv, VALUE str) 06833 { 06834 str = rb_str_dup(str); 06835 rb_str_chomp_bang(argc, argv, str); 06836 return str; 06837 } 06838 06839 /* 06840 * call-seq: 06841 * str.lstrip! -> self or nil 06842 * 06843 * Removes leading whitespace from <i>str</i>, returning <code>nil</code> if no 06844 * change was made. See also <code>String#rstrip!</code> and 06845 * <code>String#strip!</code>. 06846 * 06847 * " hello ".lstrip #=> "hello " 06848 * "hello".lstrip! #=> nil 06849 */ 06850 06851 static VALUE 06852 rb_str_lstrip_bang(VALUE str) 06853 { 06854 rb_encoding *enc; 06855 char *s, *t, *e; 06856 06857 str_modify_keep_cr(str); 06858 enc = STR_ENC_GET(str); 06859 s = RSTRING_PTR(str); 06860 if (!s || RSTRING_LEN(str) == 0) return Qnil; 06861 e = t = RSTRING_END(str); 06862 /* remove spaces at head */ 06863 while (s < e) { 06864 int n; 06865 unsigned int cc = rb_enc_codepoint_len(s, e, &n, enc); 06866 06867 if (!rb_isspace(cc)) break; 06868 s += n; 06869 } 06870 06871 if (s > RSTRING_PTR(str)) { 06872 STR_SET_LEN(str, t-s); 06873 memmove(RSTRING_PTR(str), s, RSTRING_LEN(str)); 06874 RSTRING_PTR(str)[RSTRING_LEN(str)] = '\0'; 06875 return str; 06876 } 06877 return Qnil; 06878 } 06879 06880 06881 /* 06882 * call-seq: 06883 * str.lstrip -> new_str 06884 * 06885 * Returns a copy of <i>str</i> with leading whitespace removed. See also 06886 * <code>String#rstrip</code> and <code>String#strip</code>. 06887 * 06888 * " hello ".lstrip #=> "hello " 06889 * "hello".lstrip #=> "hello" 06890 */ 06891 06892 static VALUE 06893 rb_str_lstrip(VALUE str) 06894 { 06895 str = rb_str_dup(str); 06896 rb_str_lstrip_bang(str); 06897 return str; 06898 } 06899 06900 06901 /* 06902 * call-seq: 06903 * str.rstrip! -> self or nil 06904 * 06905 * Removes trailing whitespace from <i>str</i>, returning <code>nil</code> if 06906 * no change was made. See also <code>String#lstrip!</code> and 06907 * <code>String#strip!</code>. 06908 * 06909 * " hello ".rstrip #=> " hello" 06910 * "hello".rstrip! #=> nil 06911 */ 06912 06913 static VALUE 06914 rb_str_rstrip_bang(VALUE str) 06915 { 06916 rb_encoding *enc; 06917 char *s, *t, *e; 06918 06919 str_modify_keep_cr(str); 06920 enc = STR_ENC_GET(str); 06921 rb_str_check_dummy_enc(enc); 06922 s = RSTRING_PTR(str); 06923 if (!s || RSTRING_LEN(str) == 0) return Qnil; 06924 t = e = RSTRING_END(str); 06925 06926 /* remove trailing spaces or '\0's */ 06927 if (single_byte_optimizable(str)) { 06928 unsigned char c; 06929 while (s < t && ((c = *(t-1)) == '\0' || ascii_isspace(c))) t--; 06930 } 06931 else { 06932 char *tp; 06933 06934 while ((tp = rb_enc_prev_char(s, t, e, enc)) != NULL) { 06935 unsigned int c = rb_enc_codepoint(tp, e, enc); 06936 if (c && !rb_isspace(c)) break; 06937 t = tp; 06938 } 06939 } 06940 if (t < e) { 06941 long len = t-RSTRING_PTR(str); 06942 06943 STR_SET_LEN(str, len); 06944 RSTRING_PTR(str)[len] = '\0'; 06945 return str; 06946 } 06947 return Qnil; 06948 } 06949 06950 06951 /* 06952 * call-seq: 06953 * str.rstrip -> new_str 06954 * 06955 * Returns a copy of <i>str</i> with trailing whitespace removed. See also 06956 * <code>String#lstrip</code> and <code>String#strip</code>. 06957 * 06958 * " hello ".rstrip #=> " hello" 06959 * "hello".rstrip #=> "hello" 06960 */ 06961 06962 static VALUE 06963 rb_str_rstrip(VALUE str) 06964 { 06965 str = rb_str_dup(str); 06966 rb_str_rstrip_bang(str); 06967 return str; 06968 } 06969 06970 06971 /* 06972 * call-seq: 06973 * str.strip! -> str or nil 06974 * 06975 * Removes leading and trailing whitespace from <i>str</i>. Returns 06976 * <code>nil</code> if <i>str</i> was not altered. 06977 */ 06978 06979 static VALUE 06980 rb_str_strip_bang(VALUE str) 06981 { 06982 VALUE l = rb_str_lstrip_bang(str); 06983 VALUE r = rb_str_rstrip_bang(str); 06984 06985 if (NIL_P(l) && NIL_P(r)) return Qnil; 06986 return str; 06987 } 06988 06989 06990 /* 06991 * call-seq: 06992 * str.strip -> new_str 06993 * 06994 * Returns a copy of <i>str</i> with leading and trailing whitespace removed. 06995 * 06996 * " hello ".strip #=> "hello" 06997 * "\tgoodbye\r\n".strip #=> "goodbye" 06998 */ 06999 07000 static VALUE 07001 rb_str_strip(VALUE str) 07002 { 07003 str = rb_str_dup(str); 07004 rb_str_strip_bang(str); 07005 return str; 07006 } 07007 07008 static VALUE 07009 scan_once(VALUE str, VALUE pat, long *start) 07010 { 07011 VALUE result, match; 07012 struct re_registers *regs; 07013 int i; 07014 07015 if (rb_reg_search(pat, str, *start, 0) >= 0) { 07016 match = rb_backref_get(); 07017 regs = RMATCH_REGS(match); 07018 if (BEG(0) == END(0)) { 07019 rb_encoding *enc = STR_ENC_GET(str); 07020 /* 07021 * Always consume at least one character of the input string 07022 */ 07023 if (RSTRING_LEN(str) > END(0)) 07024 *start = END(0)+rb_enc_fast_mbclen(RSTRING_PTR(str)+END(0), 07025 RSTRING_END(str), enc); 07026 else 07027 *start = END(0)+1; 07028 } 07029 else { 07030 *start = END(0); 07031 } 07032 if (regs->num_regs == 1) { 07033 return rb_reg_nth_match(0, match); 07034 } 07035 result = rb_ary_new2(regs->num_regs); 07036 for (i=1; i < regs->num_regs; i++) { 07037 rb_ary_push(result, rb_reg_nth_match(i, match)); 07038 } 07039 07040 return result; 07041 } 07042 return Qnil; 07043 } 07044 07045 07046 /* 07047 * call-seq: 07048 * str.scan(pattern) -> array 07049 * str.scan(pattern) {|match, ...| block } -> str 07050 * 07051 * Both forms iterate through <i>str</i>, matching the pattern (which may be a 07052 * <code>Regexp</code> or a <code>String</code>). For each match, a result is 07053 * generated and either added to the result array or passed to the block. If 07054 * the pattern contains no groups, each individual result consists of the 07055 * matched string, <code>$&</code>. If the pattern contains groups, each 07056 * individual result is itself an array containing one entry per group. 07057 * 07058 * a = "cruel world" 07059 * a.scan(/\w+/) #=> ["cruel", "world"] 07060 * a.scan(/.../) #=> ["cru", "el ", "wor"] 07061 * a.scan(/(...)/) #=> [["cru"], ["el "], ["wor"]] 07062 * a.scan(/(..)(..)/) #=> [["cr", "ue"], ["l ", "wo"]] 07063 * 07064 * And the block form: 07065 * 07066 * a.scan(/\w+/) {|w| print "<<#{w}>> " } 07067 * print "\n" 07068 * a.scan(/(.)(.)/) {|x,y| print y, x } 07069 * print "\n" 07070 * 07071 * <em>produces:</em> 07072 * 07073 * <<cruel>> <<world>> 07074 * rceu lowlr 07075 */ 07076 07077 static VALUE 07078 rb_str_scan(VALUE str, VALUE pat) 07079 { 07080 VALUE result; 07081 long start = 0; 07082 long last = -1, prev = 0; 07083 char *p = RSTRING_PTR(str); long len = RSTRING_LEN(str); 07084 07085 pat = get_pat(pat, 1); 07086 if (!rb_block_given_p()) { 07087 VALUE ary = rb_ary_new(); 07088 07089 while (!NIL_P(result = scan_once(str, pat, &start))) { 07090 last = prev; 07091 prev = start; 07092 rb_ary_push(ary, result); 07093 } 07094 if (last >= 0) rb_reg_search(pat, str, last, 0); 07095 return ary; 07096 } 07097 07098 while (!NIL_P(result = scan_once(str, pat, &start))) { 07099 last = prev; 07100 prev = start; 07101 rb_yield(result); 07102 str_mod_check(str, p, len); 07103 } 07104 if (last >= 0) rb_reg_search(pat, str, last, 0); 07105 return str; 07106 } 07107 07108 07109 /* 07110 * call-seq: 07111 * str.hex -> integer 07112 * 07113 * Treats leading characters from <i>str</i> as a string of hexadecimal digits 07114 * (with an optional sign and an optional <code>0x</code>) and returns the 07115 * corresponding number. Zero is returned on error. 07116 * 07117 * "0x0a".hex #=> 10 07118 * "-1234".hex #=> -4660 07119 * "0".hex #=> 0 07120 * "wombat".hex #=> 0 07121 */ 07122 07123 static VALUE 07124 rb_str_hex(VALUE str) 07125 { 07126 return rb_str_to_inum(str, 16, FALSE); 07127 } 07128 07129 07130 /* 07131 * call-seq: 07132 * str.oct -> integer 07133 * 07134 * Treats leading characters of <i>str</i> as a string of octal digits (with an 07135 * optional sign) and returns the corresponding number. Returns 0 if the 07136 * conversion fails. 07137 * 07138 * "123".oct #=> 83 07139 * "-377".oct #=> -255 07140 * "bad".oct #=> 0 07141 * "0377bad".oct #=> 255 07142 */ 07143 07144 static VALUE 07145 rb_str_oct(VALUE str) 07146 { 07147 return rb_str_to_inum(str, -8, FALSE); 07148 } 07149 07150 07151 /* 07152 * call-seq: 07153 * str.crypt(salt_str) -> new_str 07154 * 07155 * Applies a one-way cryptographic hash to <i>str</i> by invoking the 07156 * standard library function <code>crypt(3)</code> with the given 07157 * salt string. While the format and the result are system and 07158 * implementation dependent, using a salt matching the regular 07159 * expression <code>\A[a-zA-Z0-9./]{2}</code> should be valid and 07160 * safe on any platform, in which only the first two characters are 07161 * significant. 07162 * 07163 * This method is for use in system specific scripts, so if you want 07164 * a cross-platform hash function consider using Digest or OpenSSL 07165 * instead. 07166 */ 07167 07168 static VALUE 07169 rb_str_crypt(VALUE str, VALUE salt) 07170 { 07171 extern char *crypt(const char *, const char *); 07172 VALUE result; 07173 const char *s, *saltp; 07174 char *res; 07175 #ifdef BROKEN_CRYPT 07176 char salt_8bit_clean[3]; 07177 #endif 07178 07179 StringValue(salt); 07180 if (RSTRING_LEN(salt) < 2) 07181 rb_raise(rb_eArgError, "salt too short (need >=2 bytes)"); 07182 07183 s = RSTRING_PTR(str); 07184 if (!s) s = ""; 07185 saltp = RSTRING_PTR(salt); 07186 #ifdef BROKEN_CRYPT 07187 if (!ISASCII((unsigned char)saltp[0]) || !ISASCII((unsigned char)saltp[1])) { 07188 salt_8bit_clean[0] = saltp[0] & 0x7f; 07189 salt_8bit_clean[1] = saltp[1] & 0x7f; 07190 salt_8bit_clean[2] = '\0'; 07191 saltp = salt_8bit_clean; 07192 } 07193 #endif 07194 res = crypt(s, saltp); 07195 if (!res) { 07196 rb_sys_fail("crypt"); 07197 } 07198 result = rb_str_new2(res); 07199 OBJ_INFECT(result, str); 07200 OBJ_INFECT(result, salt); 07201 return result; 07202 } 07203 07204 07205 /* 07206 * call-seq: 07207 * str.intern -> symbol 07208 * str.to_sym -> symbol 07209 * 07210 * Returns the <code>Symbol</code> corresponding to <i>str</i>, creating the 07211 * symbol if it did not previously exist. See <code>Symbol#id2name</code>. 07212 * 07213 * "Koala".intern #=> :Koala 07214 * s = 'cat'.to_sym #=> :cat 07215 * s == :cat #=> true 07216 * s = '@cat'.to_sym #=> :@cat 07217 * s == :@cat #=> true 07218 * 07219 * This can also be used to create symbols that cannot be represented using the 07220 * <code>:xxx</code> notation. 07221 * 07222 * 'cat and dog'.to_sym #=> :"cat and dog" 07223 */ 07224 07225 VALUE 07226 rb_str_intern(VALUE s) 07227 { 07228 VALUE str = RB_GC_GUARD(s); 07229 ID id; 07230 07231 id = rb_intern_str(str); 07232 return ID2SYM(id); 07233 } 07234 07235 07236 /* 07237 * call-seq: 07238 * str.ord -> integer 07239 * 07240 * Return the <code>Integer</code> ordinal of a one-character string. 07241 * 07242 * "a".ord #=> 97 07243 */ 07244 07245 VALUE 07246 rb_str_ord(VALUE s) 07247 { 07248 unsigned int c; 07249 07250 c = rb_enc_codepoint(RSTRING_PTR(s), RSTRING_END(s), STR_ENC_GET(s)); 07251 return UINT2NUM(c); 07252 } 07253 /* 07254 * call-seq: 07255 * str.sum(n=16) -> integer 07256 * 07257 * Returns a basic <em>n</em>-bit checksum of the characters in <i>str</i>, 07258 * where <em>n</em> is the optional <code>Fixnum</code> parameter, defaulting 07259 * to 16. The result is simply the sum of the binary value of each character in 07260 * <i>str</i> modulo <code>2**n - 1</code>. This is not a particularly good 07261 * checksum. 07262 */ 07263 07264 static VALUE 07265 rb_str_sum(int argc, VALUE *argv, VALUE str) 07266 { 07267 VALUE vbits; 07268 int bits; 07269 char *ptr, *p, *pend; 07270 long len; 07271 VALUE sum = INT2FIX(0); 07272 unsigned long sum0 = 0; 07273 07274 if (argc == 0) { 07275 bits = 16; 07276 } 07277 else { 07278 rb_scan_args(argc, argv, "01", &vbits); 07279 bits = NUM2INT(vbits); 07280 } 07281 ptr = p = RSTRING_PTR(str); 07282 len = RSTRING_LEN(str); 07283 pend = p + len; 07284 07285 while (p < pend) { 07286 if (FIXNUM_MAX - UCHAR_MAX < sum0) { 07287 sum = rb_funcall(sum, '+', 1, LONG2FIX(sum0)); 07288 str_mod_check(str, ptr, len); 07289 sum0 = 0; 07290 } 07291 sum0 += (unsigned char)*p; 07292 p++; 07293 } 07294 07295 if (bits == 0) { 07296 if (sum0) { 07297 sum = rb_funcall(sum, '+', 1, LONG2FIX(sum0)); 07298 } 07299 } 07300 else { 07301 if (sum == INT2FIX(0)) { 07302 if (bits < (int)sizeof(long)*CHAR_BIT) { 07303 sum0 &= (((unsigned long)1)<<bits)-1; 07304 } 07305 sum = LONG2FIX(sum0); 07306 } 07307 else { 07308 VALUE mod; 07309 07310 if (sum0) { 07311 sum = rb_funcall(sum, '+', 1, LONG2FIX(sum0)); 07312 } 07313 07314 mod = rb_funcall(INT2FIX(1), rb_intern("<<"), 1, INT2FIX(bits)); 07315 mod = rb_funcall(mod, '-', 1, INT2FIX(1)); 07316 sum = rb_funcall(sum, '&', 1, mod); 07317 } 07318 } 07319 return sum; 07320 } 07321 07322 static VALUE 07323 rb_str_justify(int argc, VALUE *argv, VALUE str, char jflag) 07324 { 07325 rb_encoding *enc; 07326 VALUE w; 07327 long width, len, flen = 1, fclen = 1; 07328 VALUE res; 07329 char *p; 07330 const char *f = " "; 07331 long n, size, llen, rlen, llen2 = 0, rlen2 = 0; 07332 volatile VALUE pad; 07333 int singlebyte = 1, cr; 07334 07335 rb_scan_args(argc, argv, "11", &w, &pad); 07336 enc = STR_ENC_GET(str); 07337 width = NUM2LONG(w); 07338 if (argc == 2) { 07339 StringValue(pad); 07340 enc = rb_enc_check(str, pad); 07341 f = RSTRING_PTR(pad); 07342 flen = RSTRING_LEN(pad); 07343 fclen = str_strlen(pad, enc); 07344 singlebyte = single_byte_optimizable(pad); 07345 if (flen == 0 || fclen == 0) { 07346 rb_raise(rb_eArgError, "zero width padding"); 07347 } 07348 } 07349 len = str_strlen(str, enc); 07350 if (width < 0 || len >= width) return rb_str_dup(str); 07351 n = width - len; 07352 llen = (jflag == 'l') ? 0 : ((jflag == 'r') ? n : n/2); 07353 rlen = n - llen; 07354 cr = ENC_CODERANGE(str); 07355 if (flen > 1) { 07356 llen2 = str_offset(f, f + flen, llen % fclen, enc, singlebyte); 07357 rlen2 = str_offset(f, f + flen, rlen % fclen, enc, singlebyte); 07358 } 07359 size = RSTRING_LEN(str); 07360 if ((len = llen / fclen + rlen / fclen) >= LONG_MAX / flen || 07361 (len *= flen) >= LONG_MAX - llen2 - rlen2 || 07362 (len += llen2 + rlen2) >= LONG_MAX - size) { 07363 rb_raise(rb_eArgError, "argument too big"); 07364 } 07365 len += size; 07366 res = rb_str_new5(str, 0, len); 07367 p = RSTRING_PTR(res); 07368 if (flen <= 1) { 07369 memset(p, *f, llen); 07370 p += llen; 07371 } 07372 else { 07373 while (llen >= fclen) { 07374 memcpy(p,f,flen); 07375 p += flen; 07376 llen -= fclen; 07377 } 07378 if (llen > 0) { 07379 memcpy(p, f, llen2); 07380 p += llen2; 07381 } 07382 } 07383 memcpy(p, RSTRING_PTR(str), size); 07384 p += size; 07385 if (flen <= 1) { 07386 memset(p, *f, rlen); 07387 p += rlen; 07388 } 07389 else { 07390 while (rlen >= fclen) { 07391 memcpy(p,f,flen); 07392 p += flen; 07393 rlen -= fclen; 07394 } 07395 if (rlen > 0) { 07396 memcpy(p, f, rlen2); 07397 p += rlen2; 07398 } 07399 } 07400 *p = '\0'; 07401 STR_SET_LEN(res, p-RSTRING_PTR(res)); 07402 OBJ_INFECT(res, str); 07403 if (!NIL_P(pad)) OBJ_INFECT(res, pad); 07404 rb_enc_associate(res, enc); 07405 if (argc == 2) 07406 cr = ENC_CODERANGE_AND(cr, ENC_CODERANGE(pad)); 07407 if (cr != ENC_CODERANGE_BROKEN) 07408 ENC_CODERANGE_SET(res, cr); 07409 return res; 07410 } 07411 07412 07413 /* 07414 * call-seq: 07415 * str.ljust(integer, padstr=' ') -> new_str 07416 * 07417 * If <i>integer</i> is greater than the length of <i>str</i>, returns a new 07418 * <code>String</code> of length <i>integer</i> with <i>str</i> left justified 07419 * and padded with <i>padstr</i>; otherwise, returns <i>str</i>. 07420 * 07421 * "hello".ljust(4) #=> "hello" 07422 * "hello".ljust(20) #=> "hello " 07423 * "hello".ljust(20, '1234') #=> "hello123412341234123" 07424 */ 07425 07426 static VALUE 07427 rb_str_ljust(int argc, VALUE *argv, VALUE str) 07428 { 07429 return rb_str_justify(argc, argv, str, 'l'); 07430 } 07431 07432 07433 /* 07434 * call-seq: 07435 * str.rjust(integer, padstr=' ') -> new_str 07436 * 07437 * If <i>integer</i> is greater than the length of <i>str</i>, returns a new 07438 * <code>String</code> of length <i>integer</i> with <i>str</i> right justified 07439 * and padded with <i>padstr</i>; otherwise, returns <i>str</i>. 07440 * 07441 * "hello".rjust(4) #=> "hello" 07442 * "hello".rjust(20) #=> " hello" 07443 * "hello".rjust(20, '1234') #=> "123412341234123hello" 07444 */ 07445 07446 static VALUE 07447 rb_str_rjust(int argc, VALUE *argv, VALUE str) 07448 { 07449 return rb_str_justify(argc, argv, str, 'r'); 07450 } 07451 07452 07453 /* 07454 * call-seq: 07455 * str.center(width, padstr=' ') -> new_str 07456 * 07457 * Centers +str+ in +width+. If +width+ is greater than the length of +str+, 07458 * returns a new String of length +width+ with +str+ centered and padded with 07459 * +padstr+; otherwise, returns +str+. 07460 * 07461 * "hello".center(4) #=> "hello" 07462 * "hello".center(20) #=> " hello " 07463 * "hello".center(20, '123') #=> "1231231hello12312312" 07464 */ 07465 07466 static VALUE 07467 rb_str_center(int argc, VALUE *argv, VALUE str) 07468 { 07469 return rb_str_justify(argc, argv, str, 'c'); 07470 } 07471 07472 /* 07473 * call-seq: 07474 * str.partition(sep) -> [head, sep, tail] 07475 * str.partition(regexp) -> [head, match, tail] 07476 * 07477 * Searches <i>sep</i> or pattern (<i>regexp</i>) in the string 07478 * and returns the part before it, the match, and the part 07479 * after it. 07480 * If it is not found, returns two empty strings and <i>str</i>. 07481 * 07482 * "hello".partition("l") #=> ["he", "l", "lo"] 07483 * "hello".partition("x") #=> ["hello", "", ""] 07484 * "hello".partition(/.l/) #=> ["h", "el", "lo"] 07485 */ 07486 07487 static VALUE 07488 rb_str_partition(VALUE str, VALUE sep) 07489 { 07490 long pos; 07491 int regex = FALSE; 07492 07493 if (RB_TYPE_P(sep, T_REGEXP)) { 07494 pos = rb_reg_search(sep, str, 0, 0); 07495 regex = TRUE; 07496 } 07497 else { 07498 VALUE tmp; 07499 07500 tmp = rb_check_string_type(sep); 07501 if (NIL_P(tmp)) { 07502 rb_raise(rb_eTypeError, "type mismatch: %s given", 07503 rb_obj_classname(sep)); 07504 } 07505 sep = tmp; 07506 pos = rb_str_index(str, sep, 0); 07507 } 07508 if (pos < 0) { 07509 failed: 07510 return rb_ary_new3(3, str, str_new_empty(str), str_new_empty(str)); 07511 } 07512 if (regex) { 07513 sep = rb_str_subpat(str, sep, INT2FIX(0)); 07514 if (pos == 0 && RSTRING_LEN(sep) == 0) goto failed; 07515 } 07516 return rb_ary_new3(3, rb_str_subseq(str, 0, pos), 07517 sep, 07518 rb_str_subseq(str, pos+RSTRING_LEN(sep), 07519 RSTRING_LEN(str)-pos-RSTRING_LEN(sep))); 07520 } 07521 07522 /* 07523 * call-seq: 07524 * str.rpartition(sep) -> [head, sep, tail] 07525 * str.rpartition(regexp) -> [head, match, tail] 07526 * 07527 * Searches <i>sep</i> or pattern (<i>regexp</i>) in the string from the end 07528 * of the string, and returns the part before it, the match, and the part 07529 * after it. 07530 * If it is not found, returns two empty strings and <i>str</i>. 07531 * 07532 * "hello".rpartition("l") #=> ["hel", "l", "o"] 07533 * "hello".rpartition("x") #=> ["", "", "hello"] 07534 * "hello".rpartition(/.l/) #=> ["he", "ll", "o"] 07535 */ 07536 07537 static VALUE 07538 rb_str_rpartition(VALUE str, VALUE sep) 07539 { 07540 long pos = RSTRING_LEN(str); 07541 int regex = FALSE; 07542 07543 if (RB_TYPE_P(sep, T_REGEXP)) { 07544 pos = rb_reg_search(sep, str, pos, 1); 07545 regex = TRUE; 07546 } 07547 else { 07548 VALUE tmp; 07549 07550 tmp = rb_check_string_type(sep); 07551 if (NIL_P(tmp)) { 07552 rb_raise(rb_eTypeError, "type mismatch: %s given", 07553 rb_obj_classname(sep)); 07554 } 07555 sep = tmp; 07556 pos = rb_str_sublen(str, pos); 07557 pos = rb_str_rindex(str, sep, pos); 07558 } 07559 if (pos < 0) { 07560 return rb_ary_new3(3, str_new_empty(str), str_new_empty(str), str); 07561 } 07562 if (regex) { 07563 sep = rb_reg_nth_match(0, rb_backref_get()); 07564 } 07565 return rb_ary_new3(3, rb_str_substr(str, 0, pos), 07566 sep, 07567 rb_str_substr(str,pos+str_strlen(sep,STR_ENC_GET(sep)),RSTRING_LEN(str))); 07568 } 07569 07570 /* 07571 * call-seq: 07572 * str.start_with?([prefixes]+) -> true or false 07573 * 07574 * Returns true if +str+ starts with one of the +prefixes+ given. 07575 * 07576 * "hello".start_with?("hell") #=> true 07577 * 07578 * # returns true if one of the prefixes matches. 07579 * "hello".start_with?("heaven", "hell") #=> true 07580 * "hello".start_with?("heaven", "paradise") #=> false 07581 */ 07582 07583 static VALUE 07584 rb_str_start_with(int argc, VALUE *argv, VALUE str) 07585 { 07586 int i; 07587 07588 for (i=0; i<argc; i++) { 07589 VALUE tmp = argv[i]; 07590 StringValue(tmp); 07591 rb_enc_check(str, tmp); 07592 if (RSTRING_LEN(str) < RSTRING_LEN(tmp)) continue; 07593 if (memcmp(RSTRING_PTR(str), RSTRING_PTR(tmp), RSTRING_LEN(tmp)) == 0) 07594 return Qtrue; 07595 } 07596 return Qfalse; 07597 } 07598 07599 /* 07600 * call-seq: 07601 * str.end_with?([suffixes]+) -> true or false 07602 * 07603 * Returns true if +str+ ends with one of the +suffixes+ given. 07604 */ 07605 07606 static VALUE 07607 rb_str_end_with(int argc, VALUE *argv, VALUE str) 07608 { 07609 int i; 07610 char *p, *s, *e; 07611 rb_encoding *enc; 07612 07613 for (i=0; i<argc; i++) { 07614 VALUE tmp = argv[i]; 07615 StringValue(tmp); 07616 enc = rb_enc_check(str, tmp); 07617 if (RSTRING_LEN(str) < RSTRING_LEN(tmp)) continue; 07618 p = RSTRING_PTR(str); 07619 e = p + RSTRING_LEN(str); 07620 s = e - RSTRING_LEN(tmp); 07621 if (rb_enc_left_char_head(p, s, e, enc) != s) 07622 continue; 07623 if (memcmp(s, RSTRING_PTR(tmp), RSTRING_LEN(tmp)) == 0) 07624 return Qtrue; 07625 } 07626 return Qfalse; 07627 } 07628 07629 void 07630 rb_str_setter(VALUE val, ID id, VALUE *var) 07631 { 07632 if (!NIL_P(val) && !RB_TYPE_P(val, T_STRING)) { 07633 rb_raise(rb_eTypeError, "value of %s must be String", rb_id2name(id)); 07634 } 07635 *var = val; 07636 } 07637 07638 07639 /* 07640 * call-seq: 07641 * str.force_encoding(encoding) -> str 07642 * 07643 * Changes the encoding to +encoding+ and returns self. 07644 */ 07645 07646 static VALUE 07647 rb_str_force_encoding(VALUE str, VALUE enc) 07648 { 07649 str_modifiable(str); 07650 rb_enc_associate(str, rb_to_encoding(enc)); 07651 ENC_CODERANGE_CLEAR(str); 07652 return str; 07653 } 07654 07655 /* 07656 * call-seq: 07657 * str.b -> str 07658 * 07659 * Returns a copied string whose encoding is ASCII-8BIT. 07660 */ 07661 07662 static VALUE 07663 rb_str_b(VALUE str) 07664 { 07665 VALUE str2 = str_alloc(rb_cString); 07666 str_replace_shared_without_enc(str2, str); 07667 OBJ_INFECT(str2, str); 07668 ENC_CODERANGE_SET(str2, ENC_CODERANGE_VALID); 07669 return str2; 07670 } 07671 07672 /* 07673 * call-seq: 07674 * str.valid_encoding? -> true or false 07675 * 07676 * Returns true for a string which encoded correctly. 07677 * 07678 * "\xc2\xa1".force_encoding("UTF-8").valid_encoding? #=> true 07679 * "\xc2".force_encoding("UTF-8").valid_encoding? #=> false 07680 * "\x80".force_encoding("UTF-8").valid_encoding? #=> false 07681 */ 07682 07683 static VALUE 07684 rb_str_valid_encoding_p(VALUE str) 07685 { 07686 int cr = rb_enc_str_coderange(str); 07687 07688 return cr == ENC_CODERANGE_BROKEN ? Qfalse : Qtrue; 07689 } 07690 07691 /* 07692 * call-seq: 07693 * str.ascii_only? -> true or false 07694 * 07695 * Returns true for a string which has only ASCII characters. 07696 * 07697 * "abc".force_encoding("UTF-8").ascii_only? #=> true 07698 * "abc\u{6666}".force_encoding("UTF-8").ascii_only? #=> false 07699 */ 07700 07701 static VALUE 07702 rb_str_is_ascii_only_p(VALUE str) 07703 { 07704 int cr = rb_enc_str_coderange(str); 07705 07706 return cr == ENC_CODERANGE_7BIT ? Qtrue : Qfalse; 07707 } 07708 07723 VALUE 07724 rb_str_ellipsize(VALUE str, long len) 07725 { 07726 static const char ellipsis[] = "..."; 07727 const long ellipsislen = sizeof(ellipsis) - 1; 07728 rb_encoding *const enc = rb_enc_get(str); 07729 const long blen = RSTRING_LEN(str); 07730 const char *const p = RSTRING_PTR(str), *e = p + blen; 07731 VALUE estr, ret = 0; 07732 07733 if (len < 0) rb_raise(rb_eIndexError, "negative length %ld", len); 07734 if (len * rb_enc_mbminlen(enc) >= blen || 07735 (e = rb_enc_nth(p, e, len, enc)) - p == blen) { 07736 ret = str; 07737 } 07738 else if (len <= ellipsislen || 07739 !(e = rb_enc_step_back(p, e, e, len = ellipsislen, enc))) { 07740 if (rb_enc_asciicompat(enc)) { 07741 ret = rb_str_new_with_class(str, ellipsis, len); 07742 rb_enc_associate(ret, enc); 07743 } 07744 else { 07745 estr = rb_usascii_str_new(ellipsis, len); 07746 ret = rb_str_encode(estr, rb_enc_from_encoding(enc), 0, Qnil); 07747 } 07748 } 07749 else if (ret = rb_str_subseq(str, 0, e - p), rb_enc_asciicompat(enc)) { 07750 rb_str_cat(ret, ellipsis, ellipsislen); 07751 } 07752 else { 07753 estr = rb_str_encode(rb_usascii_str_new(ellipsis, ellipsislen), 07754 rb_enc_from_encoding(enc), 0, Qnil); 07755 rb_str_append(ret, estr); 07756 } 07757 return ret; 07758 } 07759 07760 /********************************************************************** 07761 * Document-class: Symbol 07762 * 07763 * <code>Symbol</code> objects represent names and some strings 07764 * inside the Ruby 07765 * interpreter. They are generated using the <code>:name</code> and 07766 * <code>:"string"</code> literals 07767 * syntax, and by the various <code>to_sym</code> methods. The same 07768 * <code>Symbol</code> object will be created for a given name or string 07769 * for the duration of a program's execution, regardless of the context 07770 * or meaning of that name. Thus if <code>Fred</code> is a constant in 07771 * one context, a method in another, and a class in a third, the 07772 * <code>Symbol</code> <code>:Fred</code> will be the same object in 07773 * all three contexts. 07774 * 07775 * module One 07776 * class Fred 07777 * end 07778 * $f1 = :Fred 07779 * end 07780 * module Two 07781 * Fred = 1 07782 * $f2 = :Fred 07783 * end 07784 * def Fred() 07785 * end 07786 * $f3 = :Fred 07787 * $f1.object_id #=> 2514190 07788 * $f2.object_id #=> 2514190 07789 * $f3.object_id #=> 2514190 07790 * 07791 */ 07792 07793 07794 /* 07795 * call-seq: 07796 * sym == obj -> true or false 07797 * 07798 * Equality---If <i>sym</i> and <i>obj</i> are exactly the same 07799 * symbol, returns <code>true</code>. 07800 */ 07801 07802 static VALUE 07803 sym_equal(VALUE sym1, VALUE sym2) 07804 { 07805 if (sym1 == sym2) return Qtrue; 07806 return Qfalse; 07807 } 07808 07809 07810 static int 07811 sym_printable(const char *s, const char *send, rb_encoding *enc) 07812 { 07813 while (s < send) { 07814 int n; 07815 int c = rb_enc_codepoint_len(s, send, &n, enc); 07816 07817 if (!rb_enc_isprint(c, enc)) return FALSE; 07818 s += n; 07819 } 07820 return TRUE; 07821 } 07822 07823 int 07824 rb_str_symname_p(VALUE sym) 07825 { 07826 rb_encoding *enc; 07827 const char *ptr; 07828 long len; 07829 rb_encoding *resenc = rb_default_internal_encoding(); 07830 07831 if (resenc == NULL) resenc = rb_default_external_encoding(); 07832 enc = STR_ENC_GET(sym); 07833 ptr = RSTRING_PTR(sym); 07834 len = RSTRING_LEN(sym); 07835 if ((resenc != enc && !rb_str_is_ascii_only_p(sym)) || len != (long)strlen(ptr) || 07836 !rb_enc_symname_p(ptr, enc) || !sym_printable(ptr, ptr + len, enc)) { 07837 return FALSE; 07838 } 07839 return TRUE; 07840 } 07841 07842 VALUE 07843 rb_str_quote_unprintable(VALUE str) 07844 { 07845 rb_encoding *enc; 07846 const char *ptr; 07847 long len; 07848 rb_encoding *resenc; 07849 07850 Check_Type(str, T_STRING); 07851 resenc = rb_default_internal_encoding(); 07852 if (resenc == NULL) resenc = rb_default_external_encoding(); 07853 enc = STR_ENC_GET(str); 07854 ptr = RSTRING_PTR(str); 07855 len = RSTRING_LEN(str); 07856 if ((resenc != enc && !rb_str_is_ascii_only_p(str)) || 07857 !sym_printable(ptr, ptr + len, enc)) { 07858 return rb_str_inspect(str); 07859 } 07860 return str; 07861 } 07862 07863 VALUE 07864 rb_id_quote_unprintable(ID id) 07865 { 07866 return rb_str_quote_unprintable(rb_id2str(id)); 07867 } 07868 07869 /* 07870 * call-seq: 07871 * sym.inspect -> string 07872 * 07873 * Returns the representation of <i>sym</i> as a symbol literal. 07874 * 07875 * :fred.inspect #=> ":fred" 07876 */ 07877 07878 static VALUE 07879 sym_inspect(VALUE sym) 07880 { 07881 VALUE str; 07882 const char *ptr; 07883 long len; 07884 ID id = SYM2ID(sym); 07885 char *dest; 07886 07887 sym = rb_id2str(id); 07888 if (!rb_str_symname_p(sym)) { 07889 str = rb_str_inspect(sym); 07890 len = RSTRING_LEN(str); 07891 rb_str_resize(str, len + 1); 07892 dest = RSTRING_PTR(str); 07893 memmove(dest + 1, dest, len); 07894 dest[0] = ':'; 07895 } 07896 else { 07897 rb_encoding *enc = STR_ENC_GET(sym); 07898 ptr = RSTRING_PTR(sym); 07899 len = RSTRING_LEN(sym); 07900 str = rb_enc_str_new(0, len + 1, enc); 07901 dest = RSTRING_PTR(str); 07902 dest[0] = ':'; 07903 memcpy(dest + 1, ptr, len); 07904 } 07905 return str; 07906 } 07907 07908 07909 /* 07910 * call-seq: 07911 * sym.id2name -> string 07912 * sym.to_s -> string 07913 * 07914 * Returns the name or string corresponding to <i>sym</i>. 07915 * 07916 * :fred.id2name #=> "fred" 07917 */ 07918 07919 07920 VALUE 07921 rb_sym_to_s(VALUE sym) 07922 { 07923 ID id = SYM2ID(sym); 07924 07925 return str_new3(rb_cString, rb_id2str(id)); 07926 } 07927 07928 07929 /* 07930 * call-seq: 07931 * sym.to_sym -> sym 07932 * sym.intern -> sym 07933 * 07934 * In general, <code>to_sym</code> returns the <code>Symbol</code> corresponding 07935 * to an object. As <i>sym</i> is already a symbol, <code>self</code> is returned 07936 * in this case. 07937 */ 07938 07939 static VALUE 07940 sym_to_sym(VALUE sym) 07941 { 07942 return sym; 07943 } 07944 07945 static VALUE 07946 sym_call(VALUE args, VALUE sym, int argc, VALUE *argv, VALUE passed_proc) 07947 { 07948 VALUE obj; 07949 07950 if (argc < 1) { 07951 rb_raise(rb_eArgError, "no receiver given"); 07952 } 07953 obj = argv[0]; 07954 return rb_funcall_with_block(obj, (ID)sym, argc - 1, argv + 1, passed_proc); 07955 } 07956 07957 /* 07958 * call-seq: 07959 * sym.to_proc 07960 * 07961 * Returns a _Proc_ object which respond to the given method by _sym_. 07962 * 07963 * (1..3).collect(&:to_s) #=> ["1", "2", "3"] 07964 */ 07965 07966 static VALUE 07967 sym_to_proc(VALUE sym) 07968 { 07969 static VALUE sym_proc_cache = Qfalse; 07970 enum {SYM_PROC_CACHE_SIZE = 67}; 07971 VALUE proc; 07972 long id, index; 07973 VALUE *aryp; 07974 07975 if (!sym_proc_cache) { 07976 sym_proc_cache = rb_ary_tmp_new(SYM_PROC_CACHE_SIZE * 2); 07977 rb_gc_register_mark_object(sym_proc_cache); 07978 rb_ary_store(sym_proc_cache, SYM_PROC_CACHE_SIZE*2 - 1, Qnil); 07979 } 07980 07981 id = SYM2ID(sym); 07982 index = (id % SYM_PROC_CACHE_SIZE) << 1; 07983 07984 aryp = RARRAY_PTR(sym_proc_cache); 07985 if (aryp[index] == sym) { 07986 return aryp[index + 1]; 07987 } 07988 else { 07989 proc = rb_proc_new(sym_call, (VALUE)id); 07990 aryp[index] = sym; 07991 aryp[index + 1] = proc; 07992 return proc; 07993 } 07994 } 07995 07996 /* 07997 * call-seq: 07998 * 07999 * sym.succ 08000 * 08001 * Same as <code>sym.to_s.succ.intern</code>. 08002 */ 08003 08004 static VALUE 08005 sym_succ(VALUE sym) 08006 { 08007 return rb_str_intern(rb_str_succ(rb_sym_to_s(sym))); 08008 } 08009 08010 /* 08011 * call-seq: 08012 * 08013 * symbol <=> other_symbol -> -1, 0, +1 or nil 08014 * 08015 * Compares +symbol+ with +other_symbol+ after calling #to_s on each of the 08016 * symbols. Returns -1, 0, +1 or nil depending on whether +symbol+ is less 08017 * than, equal to, or greater than +other_symbol+. 08018 * 08019 * +nil+ is returned if the two values are incomparable. 08020 * 08021 * See String#<=> for more information. 08022 */ 08023 08024 static VALUE 08025 sym_cmp(VALUE sym, VALUE other) 08026 { 08027 if (!SYMBOL_P(other)) { 08028 return Qnil; 08029 } 08030 return rb_str_cmp_m(rb_sym_to_s(sym), rb_sym_to_s(other)); 08031 } 08032 08033 /* 08034 * call-seq: 08035 * 08036 * sym.casecmp(other) -> -1, 0, +1 or nil 08037 * 08038 * Case-insensitive version of <code>Symbol#<=></code>. 08039 */ 08040 08041 static VALUE 08042 sym_casecmp(VALUE sym, VALUE other) 08043 { 08044 if (!SYMBOL_P(other)) { 08045 return Qnil; 08046 } 08047 return rb_str_casecmp(rb_sym_to_s(sym), rb_sym_to_s(other)); 08048 } 08049 08050 /* 08051 * call-seq: 08052 * sym =~ obj -> fixnum or nil 08053 * 08054 * Returns <code>sym.to_s =~ obj</code>. 08055 */ 08056 08057 static VALUE 08058 sym_match(VALUE sym, VALUE other) 08059 { 08060 return rb_str_match(rb_sym_to_s(sym), other); 08061 } 08062 08063 /* 08064 * call-seq: 08065 * sym[idx] -> char 08066 * sym[b, n] -> char 08067 * 08068 * Returns <code>sym.to_s[]</code>. 08069 */ 08070 08071 static VALUE 08072 sym_aref(int argc, VALUE *argv, VALUE sym) 08073 { 08074 return rb_str_aref_m(argc, argv, rb_sym_to_s(sym)); 08075 } 08076 08077 /* 08078 * call-seq: 08079 * sym.length -> integer 08080 * 08081 * Same as <code>sym.to_s.length</code>. 08082 */ 08083 08084 static VALUE 08085 sym_length(VALUE sym) 08086 { 08087 return rb_str_length(rb_id2str(SYM2ID(sym))); 08088 } 08089 08090 /* 08091 * call-seq: 08092 * sym.empty? -> true or false 08093 * 08094 * Returns that _sym_ is :"" or not. 08095 */ 08096 08097 static VALUE 08098 sym_empty(VALUE sym) 08099 { 08100 return rb_str_empty(rb_id2str(SYM2ID(sym))); 08101 } 08102 08103 /* 08104 * call-seq: 08105 * sym.upcase -> symbol 08106 * 08107 * Same as <code>sym.to_s.upcase.intern</code>. 08108 */ 08109 08110 static VALUE 08111 sym_upcase(VALUE sym) 08112 { 08113 return rb_str_intern(rb_str_upcase(rb_id2str(SYM2ID(sym)))); 08114 } 08115 08116 /* 08117 * call-seq: 08118 * sym.downcase -> symbol 08119 * 08120 * Same as <code>sym.to_s.downcase.intern</code>. 08121 */ 08122 08123 static VALUE 08124 sym_downcase(VALUE sym) 08125 { 08126 return rb_str_intern(rb_str_downcase(rb_id2str(SYM2ID(sym)))); 08127 } 08128 08129 /* 08130 * call-seq: 08131 * sym.capitalize -> symbol 08132 * 08133 * Same as <code>sym.to_s.capitalize.intern</code>. 08134 */ 08135 08136 static VALUE 08137 sym_capitalize(VALUE sym) 08138 { 08139 return rb_str_intern(rb_str_capitalize(rb_id2str(SYM2ID(sym)))); 08140 } 08141 08142 /* 08143 * call-seq: 08144 * sym.swapcase -> symbol 08145 * 08146 * Same as <code>sym.to_s.swapcase.intern</code>. 08147 */ 08148 08149 static VALUE 08150 sym_swapcase(VALUE sym) 08151 { 08152 return rb_str_intern(rb_str_swapcase(rb_id2str(SYM2ID(sym)))); 08153 } 08154 08155 /* 08156 * call-seq: 08157 * sym.encoding -> encoding 08158 * 08159 * Returns the Encoding object that represents the encoding of _sym_. 08160 */ 08161 08162 static VALUE 08163 sym_encoding(VALUE sym) 08164 { 08165 return rb_obj_encoding(rb_id2str(SYM2ID(sym))); 08166 } 08167 08168 ID 08169 rb_to_id(VALUE name) 08170 { 08171 VALUE tmp; 08172 08173 switch (TYPE(name)) { 08174 default: 08175 tmp = rb_check_string_type(name); 08176 if (NIL_P(tmp)) { 08177 tmp = rb_inspect(name); 08178 rb_raise(rb_eTypeError, "%s is not a symbol", 08179 RSTRING_PTR(tmp)); 08180 } 08181 name = tmp; 08182 /* fall through */ 08183 case T_STRING: 08184 name = rb_str_intern(name); 08185 /* fall through */ 08186 case T_SYMBOL: 08187 return SYM2ID(name); 08188 } 08189 08190 UNREACHABLE; 08191 } 08192 08193 /* 08194 * A <code>String</code> object holds and manipulates an arbitrary sequence of 08195 * bytes, typically representing characters. String objects may be created 08196 * using <code>String::new</code> or as literals. 08197 * 08198 * Because of aliasing issues, users of strings should be aware of the methods 08199 * that modify the contents of a <code>String</code> object. Typically, 08200 * methods with names ending in ``!'' modify their receiver, while those 08201 * without a ``!'' return a new <code>String</code>. However, there are 08202 * exceptions, such as <code>String#[]=</code>. 08203 * 08204 */ 08205 08206 void 08207 Init_String(void) 08208 { 08209 #undef rb_intern 08210 #define rb_intern(str) rb_intern_const(str) 08211 08212 rb_cString = rb_define_class("String", rb_cObject); 08213 rb_include_module(rb_cString, rb_mComparable); 08214 rb_define_alloc_func(rb_cString, empty_str_alloc); 08215 rb_define_singleton_method(rb_cString, "try_convert", rb_str_s_try_convert, 1); 08216 rb_define_method(rb_cString, "initialize", rb_str_init, -1); 08217 rb_define_method(rb_cString, "initialize_copy", rb_str_replace, 1); 08218 rb_define_method(rb_cString, "<=>", rb_str_cmp_m, 1); 08219 rb_define_method(rb_cString, "==", rb_str_equal, 1); 08220 rb_define_method(rb_cString, "===", rb_str_equal, 1); 08221 rb_define_method(rb_cString, "eql?", rb_str_eql, 1); 08222 rb_define_method(rb_cString, "hash", rb_str_hash_m, 0); 08223 rb_define_method(rb_cString, "casecmp", rb_str_casecmp, 1); 08224 rb_define_method(rb_cString, "+", rb_str_plus, 1); 08225 rb_define_method(rb_cString, "*", rb_str_times, 1); 08226 rb_define_method(rb_cString, "%", rb_str_format_m, 1); 08227 rb_define_method(rb_cString, "[]", rb_str_aref_m, -1); 08228 rb_define_method(rb_cString, "[]=", rb_str_aset_m, -1); 08229 rb_define_method(rb_cString, "insert", rb_str_insert, 2); 08230 rb_define_method(rb_cString, "length", rb_str_length, 0); 08231 rb_define_method(rb_cString, "size", rb_str_length, 0); 08232 rb_define_method(rb_cString, "bytesize", rb_str_bytesize, 0); 08233 rb_define_method(rb_cString, "empty?", rb_str_empty, 0); 08234 rb_define_method(rb_cString, "=~", rb_str_match, 1); 08235 rb_define_method(rb_cString, "match", rb_str_match_m, -1); 08236 rb_define_method(rb_cString, "succ", rb_str_succ, 0); 08237 rb_define_method(rb_cString, "succ!", rb_str_succ_bang, 0); 08238 rb_define_method(rb_cString, "next", rb_str_succ, 0); 08239 rb_define_method(rb_cString, "next!", rb_str_succ_bang, 0); 08240 rb_define_method(rb_cString, "upto", rb_str_upto, -1); 08241 rb_define_method(rb_cString, "index", rb_str_index_m, -1); 08242 rb_define_method(rb_cString, "rindex", rb_str_rindex_m, -1); 08243 rb_define_method(rb_cString, "replace", rb_str_replace, 1); 08244 rb_define_method(rb_cString, "clear", rb_str_clear, 0); 08245 rb_define_method(rb_cString, "chr", rb_str_chr, 0); 08246 rb_define_method(rb_cString, "getbyte", rb_str_getbyte, 1); 08247 rb_define_method(rb_cString, "setbyte", rb_str_setbyte, 2); 08248 rb_define_method(rb_cString, "byteslice", rb_str_byteslice, -1); 08249 08250 rb_define_method(rb_cString, "to_i", rb_str_to_i, -1); 08251 rb_define_method(rb_cString, "to_f", rb_str_to_f, 0); 08252 rb_define_method(rb_cString, "to_s", rb_str_to_s, 0); 08253 rb_define_method(rb_cString, "to_str", rb_str_to_s, 0); 08254 rb_define_method(rb_cString, "inspect", rb_str_inspect, 0); 08255 rb_define_method(rb_cString, "dump", rb_str_dump, 0); 08256 08257 rb_define_method(rb_cString, "upcase", rb_str_upcase, 0); 08258 rb_define_method(rb_cString, "downcase", rb_str_downcase, 0); 08259 rb_define_method(rb_cString, "capitalize", rb_str_capitalize, 0); 08260 rb_define_method(rb_cString, "swapcase", rb_str_swapcase, 0); 08261 08262 rb_define_method(rb_cString, "upcase!", rb_str_upcase_bang, 0); 08263 rb_define_method(rb_cString, "downcase!", rb_str_downcase_bang, 0); 08264 rb_define_method(rb_cString, "capitalize!", rb_str_capitalize_bang, 0); 08265 rb_define_method(rb_cString, "swapcase!", rb_str_swapcase_bang, 0); 08266 08267 rb_define_method(rb_cString, "hex", rb_str_hex, 0); 08268 rb_define_method(rb_cString, "oct", rb_str_oct, 0); 08269 rb_define_method(rb_cString, "split", rb_str_split_m, -1); 08270 rb_define_method(rb_cString, "lines", rb_str_lines, -1); 08271 rb_define_method(rb_cString, "bytes", rb_str_bytes, 0); 08272 rb_define_method(rb_cString, "chars", rb_str_chars, 0); 08273 rb_define_method(rb_cString, "codepoints", rb_str_codepoints, 0); 08274 rb_define_method(rb_cString, "reverse", rb_str_reverse, 0); 08275 rb_define_method(rb_cString, "reverse!", rb_str_reverse_bang, 0); 08276 rb_define_method(rb_cString, "concat", rb_str_concat, 1); 08277 rb_define_method(rb_cString, "<<", rb_str_concat, 1); 08278 rb_define_method(rb_cString, "prepend", rb_str_prepend, 1); 08279 rb_define_method(rb_cString, "crypt", rb_str_crypt, 1); 08280 rb_define_method(rb_cString, "intern", rb_str_intern, 0); 08281 rb_define_method(rb_cString, "to_sym", rb_str_intern, 0); 08282 rb_define_method(rb_cString, "ord", rb_str_ord, 0); 08283 08284 rb_define_method(rb_cString, "include?", rb_str_include, 1); 08285 rb_define_method(rb_cString, "start_with?", rb_str_start_with, -1); 08286 rb_define_method(rb_cString, "end_with?", rb_str_end_with, -1); 08287 08288 rb_define_method(rb_cString, "scan", rb_str_scan, 1); 08289 08290 rb_define_method(rb_cString, "ljust", rb_str_ljust, -1); 08291 rb_define_method(rb_cString, "rjust", rb_str_rjust, -1); 08292 rb_define_method(rb_cString, "center", rb_str_center, -1); 08293 08294 rb_define_method(rb_cString, "sub", rb_str_sub, -1); 08295 rb_define_method(rb_cString, "gsub", rb_str_gsub, -1); 08296 rb_define_method(rb_cString, "chop", rb_str_chop, 0); 08297 rb_define_method(rb_cString, "chomp", rb_str_chomp, -1); 08298 rb_define_method(rb_cString, "strip", rb_str_strip, 0); 08299 rb_define_method(rb_cString, "lstrip", rb_str_lstrip, 0); 08300 rb_define_method(rb_cString, "rstrip", rb_str_rstrip, 0); 08301 08302 rb_define_method(rb_cString, "sub!", rb_str_sub_bang, -1); 08303 rb_define_method(rb_cString, "gsub!", rb_str_gsub_bang, -1); 08304 rb_define_method(rb_cString, "chop!", rb_str_chop_bang, 0); 08305 rb_define_method(rb_cString, "chomp!", rb_str_chomp_bang, -1); 08306 rb_define_method(rb_cString, "strip!", rb_str_strip_bang, 0); 08307 rb_define_method(rb_cString, "lstrip!", rb_str_lstrip_bang, 0); 08308 rb_define_method(rb_cString, "rstrip!", rb_str_rstrip_bang, 0); 08309 08310 rb_define_method(rb_cString, "tr", rb_str_tr, 2); 08311 rb_define_method(rb_cString, "tr_s", rb_str_tr_s, 2); 08312 rb_define_method(rb_cString, "delete", rb_str_delete, -1); 08313 rb_define_method(rb_cString, "squeeze", rb_str_squeeze, -1); 08314 rb_define_method(rb_cString, "count", rb_str_count, -1); 08315 08316 rb_define_method(rb_cString, "tr!", rb_str_tr_bang, 2); 08317 rb_define_method(rb_cString, "tr_s!", rb_str_tr_s_bang, 2); 08318 rb_define_method(rb_cString, "delete!", rb_str_delete_bang, -1); 08319 rb_define_method(rb_cString, "squeeze!", rb_str_squeeze_bang, -1); 08320 08321 rb_define_method(rb_cString, "each_line", rb_str_each_line, -1); 08322 rb_define_method(rb_cString, "each_byte", rb_str_each_byte, 0); 08323 rb_define_method(rb_cString, "each_char", rb_str_each_char, 0); 08324 rb_define_method(rb_cString, "each_codepoint", rb_str_each_codepoint, 0); 08325 08326 rb_define_method(rb_cString, "sum", rb_str_sum, -1); 08327 08328 rb_define_method(rb_cString, "slice", rb_str_aref_m, -1); 08329 rb_define_method(rb_cString, "slice!", rb_str_slice_bang, -1); 08330 08331 rb_define_method(rb_cString, "partition", rb_str_partition, 1); 08332 rb_define_method(rb_cString, "rpartition", rb_str_rpartition, 1); 08333 08334 rb_define_method(rb_cString, "encoding", rb_obj_encoding, 0); /* in encoding.c */ 08335 rb_define_method(rb_cString, "force_encoding", rb_str_force_encoding, 1); 08336 rb_define_method(rb_cString, "b", rb_str_b, 0); 08337 rb_define_method(rb_cString, "valid_encoding?", rb_str_valid_encoding_p, 0); 08338 rb_define_method(rb_cString, "ascii_only?", rb_str_is_ascii_only_p, 0); 08339 08340 id_to_s = rb_intern("to_s"); 08341 08342 rb_fs = Qnil; 08343 rb_define_variable("$;", &rb_fs); 08344 rb_define_variable("$-F", &rb_fs); 08345 08346 rb_cSymbol = rb_define_class("Symbol", rb_cObject); 08347 rb_include_module(rb_cSymbol, rb_mComparable); 08348 rb_undef_alloc_func(rb_cSymbol); 08349 rb_undef_method(CLASS_OF(rb_cSymbol), "new"); 08350 rb_define_singleton_method(rb_cSymbol, "all_symbols", rb_sym_all_symbols, 0); /* in parse.y */ 08351 08352 rb_define_method(rb_cSymbol, "==", sym_equal, 1); 08353 rb_define_method(rb_cSymbol, "===", sym_equal, 1); 08354 rb_define_method(rb_cSymbol, "inspect", sym_inspect, 0); 08355 rb_define_method(rb_cSymbol, "to_s", rb_sym_to_s, 0); 08356 rb_define_method(rb_cSymbol, "id2name", rb_sym_to_s, 0); 08357 rb_define_method(rb_cSymbol, "intern", sym_to_sym, 0); 08358 rb_define_method(rb_cSymbol, "to_sym", sym_to_sym, 0); 08359 rb_define_method(rb_cSymbol, "to_proc", sym_to_proc, 0); 08360 rb_define_method(rb_cSymbol, "succ", sym_succ, 0); 08361 rb_define_method(rb_cSymbol, "next", sym_succ, 0); 08362 08363 rb_define_method(rb_cSymbol, "<=>", sym_cmp, 1); 08364 rb_define_method(rb_cSymbol, "casecmp", sym_casecmp, 1); 08365 rb_define_method(rb_cSymbol, "=~", sym_match, 1); 08366 08367 rb_define_method(rb_cSymbol, "[]", sym_aref, -1); 08368 rb_define_method(rb_cSymbol, "slice", sym_aref, -1); 08369 rb_define_method(rb_cSymbol, "length", sym_length, 0); 08370 rb_define_method(rb_cSymbol, "size", sym_length, 0); 08371 rb_define_method(rb_cSymbol, "empty?", sym_empty, 0); 08372 rb_define_method(rb_cSymbol, "match", sym_match, 1); 08373 08374 rb_define_method(rb_cSymbol, "upcase", sym_upcase, 0); 08375 rb_define_method(rb_cSymbol, "downcase", sym_downcase, 0); 08376 rb_define_method(rb_cSymbol, "capitalize", sym_capitalize, 0); 08377 rb_define_method(rb_cSymbol, "swapcase", sym_swapcase, 0); 08378 08379 rb_define_method(rb_cSymbol, "encoding", sym_encoding, 0); 08380 } 08381
1.7.6.1