|
Ruby
1.9.3p286(2012-10-12revision37165)
|
00001 /********************************************************************** 00002 00003 string.c - 00004 00005 $Author: naruse $ 00006 created at: Mon Aug 9 17:12:58 JST 1993 00007 00008 Copyright (C) 1993-2007 Yukihiro Matsumoto 00009 Copyright (C) 2000 Network Applied Communication Laboratory, Inc. 00010 Copyright (C) 2000 Information-technology Promotion Agency, Japan 00011 00012 **********************************************************************/ 00013 00014 #include "ruby/ruby.h" 00015 #include "ruby/re.h" 00016 #include "ruby/encoding.h" 00017 #include "internal.h" 00018 #include <assert.h> 00019 00020 #define BEG(no) (regs->beg[(no)]) 00021 #define END(no) (regs->end[(no)]) 00022 00023 #include <math.h> 00024 #include <ctype.h> 00025 00026 #ifdef HAVE_UNISTD_H 00027 #include <unistd.h> 00028 #endif 00029 00030 #define numberof(array) (int)(sizeof(array) / sizeof((array)[0])) 00031 00032 #undef rb_str_new_cstr 00033 #undef rb_tainted_str_new_cstr 00034 #undef rb_usascii_str_new_cstr 00035 #undef rb_external_str_new_cstr 00036 #undef rb_locale_str_new_cstr 00037 #undef rb_str_new2 00038 #undef rb_str_new3 00039 #undef rb_str_new4 00040 #undef rb_str_new5 00041 #undef rb_tainted_str_new2 00042 #undef rb_usascii_str_new2 00043 #undef rb_str_dup_frozen 00044 #undef rb_str_buf_new_cstr 00045 #undef rb_str_buf_new2 00046 #undef rb_str_buf_cat2 00047 #undef rb_str_cat2 00048 00049 static VALUE rb_str_clear(VALUE str); 00050 00051 VALUE rb_cString; 00052 VALUE rb_cSymbol; 00053 00054 #define RUBY_MAX_CHAR_LEN 16 00055 #define STR_TMPLOCK FL_USER7 00056 #define STR_NOEMBED FL_USER1 00057 #define STR_SHARED FL_USER2 /* = ELTS_SHARED */ 00058 #define STR_ASSOC FL_USER3 00059 #define STR_SHARED_P(s) FL_ALL((s), STR_NOEMBED|ELTS_SHARED) 00060 #define STR_ASSOC_P(s) FL_ALL((s), STR_NOEMBED|STR_ASSOC) 00061 #define STR_NOCAPA (STR_NOEMBED|ELTS_SHARED|STR_ASSOC) 00062 #define STR_NOCAPA_P(s) (FL_TEST((s),STR_NOEMBED) && FL_ANY((s),ELTS_SHARED|STR_ASSOC)) 00063 #define STR_UNSET_NOCAPA(s) do {\ 00064 if (FL_TEST((s),STR_NOEMBED)) FL_UNSET((s),(ELTS_SHARED|STR_ASSOC));\ 00065 } while (0) 00066 00067 00068 #define STR_SET_NOEMBED(str) do {\ 00069 FL_SET((str), STR_NOEMBED);\ 00070 STR_SET_EMBED_LEN((str), 0);\ 00071 } while (0) 00072 #define STR_SET_EMBED(str) FL_UNSET((str), STR_NOEMBED) 00073 #define STR_EMBED_P(str) (!FL_TEST((str), STR_NOEMBED)) 00074 #define STR_SET_EMBED_LEN(str, n) do { \ 00075 long tmp_n = (n);\ 00076 RBASIC(str)->flags &= ~RSTRING_EMBED_LEN_MASK;\ 00077 RBASIC(str)->flags |= (tmp_n) << RSTRING_EMBED_LEN_SHIFT;\ 00078 } while (0) 00079 00080 #define STR_SET_LEN(str, n) do { \ 00081 if (STR_EMBED_P(str)) {\ 00082 STR_SET_EMBED_LEN((str), (n));\ 00083 }\ 00084 else {\ 00085 RSTRING(str)->as.heap.len = (n);\ 00086 }\ 00087 } while (0) 00088 00089 #define STR_DEC_LEN(str) do {\ 00090 if (STR_EMBED_P(str)) {\ 00091 long n = RSTRING_LEN(str);\ 00092 n--;\ 00093 STR_SET_EMBED_LEN((str), n);\ 00094 }\ 00095 else {\ 00096 RSTRING(str)->as.heap.len--;\ 00097 }\ 00098 } while (0) 00099 00100 #define RESIZE_CAPA(str,capacity) do {\ 00101 if (STR_EMBED_P(str)) {\ 00102 if ((capacity) > RSTRING_EMBED_LEN_MAX) {\ 00103 char *tmp = ALLOC_N(char, (capacity)+1);\ 00104 memcpy(tmp, RSTRING_PTR(str), RSTRING_LEN(str));\ 00105 RSTRING(str)->as.heap.ptr = tmp;\ 00106 RSTRING(str)->as.heap.len = RSTRING_LEN(str);\ 00107 STR_SET_NOEMBED(str);\ 00108 RSTRING(str)->as.heap.aux.capa = (capacity);\ 00109 }\ 00110 }\ 00111 else {\ 00112 REALLOC_N(RSTRING(str)->as.heap.ptr, char, (capacity)+1);\ 00113 if (!STR_NOCAPA_P(str))\ 00114 RSTRING(str)->as.heap.aux.capa = (capacity);\ 00115 }\ 00116 } while (0) 00117 00118 #define is_ascii_string(str) (rb_enc_str_coderange(str) == ENC_CODERANGE_7BIT) 00119 #define is_broken_string(str) (rb_enc_str_coderange(str) == ENC_CODERANGE_BROKEN) 00120 00121 #define STR_ENC_GET(str) rb_enc_from_index(ENCODING_GET(str)) 00122 00123 static inline int 00124 single_byte_optimizable(VALUE str) 00125 { 00126 rb_encoding *enc; 00127 00128 /* Conservative. It may be ENC_CODERANGE_UNKNOWN. */ 00129 if (ENC_CODERANGE(str) == ENC_CODERANGE_7BIT) 00130 return 1; 00131 00132 enc = STR_ENC_GET(str); 00133 if (rb_enc_mbmaxlen(enc) == 1) 00134 return 1; 00135 00136 /* Conservative. Possibly single byte. 00137 * "\xa1" in Shift_JIS for example. */ 00138 return 0; 00139 } 00140 00141 VALUE rb_fs; 00142 00143 static inline const char * 00144 search_nonascii(const char *p, const char *e) 00145 { 00146 #if SIZEOF_VALUE == 8 00147 # define NONASCII_MASK 0x8080808080808080ULL 00148 #elif SIZEOF_VALUE == 4 00149 # define NONASCII_MASK 0x80808080UL 00150 #endif 00151 #ifdef NONASCII_MASK 00152 if ((int)sizeof(VALUE) * 2 < e - p) { 00153 const VALUE *s, *t; 00154 const VALUE lowbits = sizeof(VALUE) - 1; 00155 s = (const VALUE*)(~lowbits & ((VALUE)p + lowbits)); 00156 while (p < (const char *)s) { 00157 if (!ISASCII(*p)) 00158 return p; 00159 p++; 00160 } 00161 t = (const VALUE*)(~lowbits & (VALUE)e); 00162 while (s < t) { 00163 if (*s & NONASCII_MASK) { 00164 t = s; 00165 break; 00166 } 00167 s++; 00168 } 00169 p = (const char *)t; 00170 } 00171 #endif 00172 while (p < e) { 00173 if (!ISASCII(*p)) 00174 return p; 00175 p++; 00176 } 00177 return NULL; 00178 } 00179 00180 static int 00181 coderange_scan(const char *p, long len, rb_encoding *enc) 00182 { 00183 const char *e = p + len; 00184 00185 if (rb_enc_to_index(enc) == 0) { 00186 /* enc is ASCII-8BIT. ASCII-8BIT string never be broken. */ 00187 p = search_nonascii(p, e); 00188 return p ? ENC_CODERANGE_VALID : ENC_CODERANGE_7BIT; 00189 } 00190 00191 if (rb_enc_asciicompat(enc)) { 00192 p = search_nonascii(p, e); 00193 if (!p) { 00194 return ENC_CODERANGE_7BIT; 00195 } 00196 while (p < e) { 00197 int ret = rb_enc_precise_mbclen(p, e, enc); 00198 if (!MBCLEN_CHARFOUND_P(ret)) { 00199 return ENC_CODERANGE_BROKEN; 00200 } 00201 p += MBCLEN_CHARFOUND_LEN(ret); 00202 if (p < e) { 00203 p = search_nonascii(p, e); 00204 if (!p) { 00205 return ENC_CODERANGE_VALID; 00206 } 00207 } 00208 } 00209 if (e < p) { 00210 return ENC_CODERANGE_BROKEN; 00211 } 00212 return ENC_CODERANGE_VALID; 00213 } 00214 00215 while (p < e) { 00216 int ret = rb_enc_precise_mbclen(p, e, enc); 00217 00218 if (!MBCLEN_CHARFOUND_P(ret)) { 00219 return ENC_CODERANGE_BROKEN; 00220 } 00221 p += MBCLEN_CHARFOUND_LEN(ret); 00222 } 00223 if (e < p) { 00224 return ENC_CODERANGE_BROKEN; 00225 } 00226 return ENC_CODERANGE_VALID; 00227 } 00228 00229 long 00230 rb_str_coderange_scan_restartable(const char *s, const char *e, rb_encoding *enc, int *cr) 00231 { 00232 const char *p = s; 00233 00234 if (*cr == ENC_CODERANGE_BROKEN) 00235 return e - s; 00236 00237 if (rb_enc_to_index(enc) == 0) { 00238 /* enc is ASCII-8BIT. ASCII-8BIT string never be broken. */ 00239 p = search_nonascii(p, e); 00240 *cr = (!p && *cr != ENC_CODERANGE_VALID) ? ENC_CODERANGE_7BIT : ENC_CODERANGE_VALID; 00241 return e - s; 00242 } 00243 else if (rb_enc_asciicompat(enc)) { 00244 p = search_nonascii(p, e); 00245 if (!p) { 00246 if (*cr != ENC_CODERANGE_VALID) *cr = ENC_CODERANGE_7BIT; 00247 return e - s; 00248 } 00249 while (p < e) { 00250 int ret = rb_enc_precise_mbclen(p, e, enc); 00251 if (!MBCLEN_CHARFOUND_P(ret)) { 00252 *cr = MBCLEN_INVALID_P(ret) ? ENC_CODERANGE_BROKEN: ENC_CODERANGE_UNKNOWN; 00253 return p - s; 00254 } 00255 p += MBCLEN_CHARFOUND_LEN(ret); 00256 if (p < e) { 00257 p = search_nonascii(p, e); 00258 if (!p) { 00259 *cr = ENC_CODERANGE_VALID; 00260 return e - s; 00261 } 00262 } 00263 } 00264 *cr = e < p ? ENC_CODERANGE_BROKEN: ENC_CODERANGE_VALID; 00265 return p - s; 00266 } 00267 else { 00268 while (p < e) { 00269 int ret = rb_enc_precise_mbclen(p, e, enc); 00270 if (!MBCLEN_CHARFOUND_P(ret)) { 00271 *cr = MBCLEN_INVALID_P(ret) ? ENC_CODERANGE_BROKEN: ENC_CODERANGE_UNKNOWN; 00272 return p - s; 00273 } 00274 p += MBCLEN_CHARFOUND_LEN(ret); 00275 } 00276 *cr = e < p ? ENC_CODERANGE_BROKEN: ENC_CODERANGE_VALID; 00277 return p - s; 00278 } 00279 } 00280 00281 static inline void 00282 str_enc_copy(VALUE str1, VALUE str2) 00283 { 00284 rb_enc_set_index(str1, ENCODING_GET(str2)); 00285 } 00286 00287 static void 00288 rb_enc_cr_str_copy_for_substr(VALUE dest, VALUE src) 00289 { 00290 /* this function is designed for copying encoding and coderange 00291 * from src to new string "dest" which is made from the part of src. 00292 */ 00293 str_enc_copy(dest, src); 00294 switch (ENC_CODERANGE(src)) { 00295 case ENC_CODERANGE_7BIT: 00296 ENC_CODERANGE_SET(dest, ENC_CODERANGE_7BIT); 00297 break; 00298 case ENC_CODERANGE_VALID: 00299 if (!rb_enc_asciicompat(STR_ENC_GET(src)) || 00300 search_nonascii(RSTRING_PTR(dest), RSTRING_END(dest))) 00301 ENC_CODERANGE_SET(dest, ENC_CODERANGE_VALID); 00302 else 00303 ENC_CODERANGE_SET(dest, ENC_CODERANGE_7BIT); 00304 break; 00305 default: 00306 if (RSTRING_LEN(dest) == 0) { 00307 if (!rb_enc_asciicompat(STR_ENC_GET(src))) 00308 ENC_CODERANGE_SET(dest, ENC_CODERANGE_VALID); 00309 else 00310 ENC_CODERANGE_SET(dest, ENC_CODERANGE_7BIT); 00311 } 00312 break; 00313 } 00314 } 00315 00316 static void 00317 rb_enc_cr_str_exact_copy(VALUE dest, VALUE src) 00318 { 00319 str_enc_copy(dest, src); 00320 ENC_CODERANGE_SET(dest, ENC_CODERANGE(src)); 00321 } 00322 00323 int 00324 rb_enc_str_coderange(VALUE str) 00325 { 00326 int cr = ENC_CODERANGE(str); 00327 00328 if (cr == ENC_CODERANGE_UNKNOWN) { 00329 rb_encoding *enc = STR_ENC_GET(str); 00330 cr = coderange_scan(RSTRING_PTR(str), RSTRING_LEN(str), enc); 00331 ENC_CODERANGE_SET(str, cr); 00332 } 00333 return cr; 00334 } 00335 00336 int 00337 rb_enc_str_asciionly_p(VALUE str) 00338 { 00339 rb_encoding *enc = STR_ENC_GET(str); 00340 00341 if (!rb_enc_asciicompat(enc)) 00342 return FALSE; 00343 else if (rb_enc_str_coderange(str) == ENC_CODERANGE_7BIT) 00344 return TRUE; 00345 return FALSE; 00346 } 00347 00348 static inline void 00349 str_mod_check(VALUE s, const char *p, long len) 00350 { 00351 if (RSTRING_PTR(s) != p || RSTRING_LEN(s) != len){ 00352 rb_raise(rb_eRuntimeError, "string modified"); 00353 } 00354 } 00355 00356 size_t 00357 rb_str_capacity(VALUE str) 00358 { 00359 if (STR_EMBED_P(str)) { 00360 return RSTRING_EMBED_LEN_MAX; 00361 } 00362 else if (STR_NOCAPA_P(str)) { 00363 return RSTRING(str)->as.heap.len; 00364 } 00365 else { 00366 return RSTRING(str)->as.heap.aux.capa; 00367 } 00368 } 00369 00370 static inline VALUE 00371 str_alloc(VALUE klass) 00372 { 00373 NEWOBJ(str, struct RString); 00374 OBJSETUP(str, klass, T_STRING); 00375 00376 str->as.heap.ptr = 0; 00377 str->as.heap.len = 0; 00378 str->as.heap.aux.capa = 0; 00379 00380 return (VALUE)str; 00381 } 00382 00383 static VALUE 00384 str_new(VALUE klass, const char *ptr, long len) 00385 { 00386 VALUE str; 00387 00388 if (len < 0) { 00389 rb_raise(rb_eArgError, "negative string size (or size too big)"); 00390 } 00391 00392 str = str_alloc(klass); 00393 if (len > RSTRING_EMBED_LEN_MAX) { 00394 RSTRING(str)->as.heap.aux.capa = len; 00395 RSTRING(str)->as.heap.ptr = ALLOC_N(char,len+1); 00396 STR_SET_NOEMBED(str); 00397 } 00398 else if (len == 0) { 00399 ENC_CODERANGE_SET(str, ENC_CODERANGE_7BIT); 00400 } 00401 if (ptr) { 00402 memcpy(RSTRING_PTR(str), ptr, len); 00403 } 00404 STR_SET_LEN(str, len); 00405 RSTRING_PTR(str)[len] = '\0'; 00406 return str; 00407 } 00408 00409 VALUE 00410 rb_str_new(const char *ptr, long len) 00411 { 00412 return str_new(rb_cString, ptr, len); 00413 } 00414 00415 VALUE 00416 rb_usascii_str_new(const char *ptr, long len) 00417 { 00418 VALUE str = rb_str_new(ptr, len); 00419 ENCODING_CODERANGE_SET(str, rb_usascii_encindex(), ENC_CODERANGE_7BIT); 00420 return str; 00421 } 00422 00423 VALUE 00424 rb_enc_str_new(const char *ptr, long len, rb_encoding *enc) 00425 { 00426 VALUE str = rb_str_new(ptr, len); 00427 rb_enc_associate(str, enc); 00428 return str; 00429 } 00430 00431 VALUE 00432 rb_str_new_cstr(const char *ptr) 00433 { 00434 if (!ptr) { 00435 rb_raise(rb_eArgError, "NULL pointer given"); 00436 } 00437 return rb_str_new(ptr, strlen(ptr)); 00438 } 00439 00440 RUBY_ALIAS_FUNCTION(rb_str_new2(const char *ptr), rb_str_new_cstr, (ptr)) 00441 #define rb_str_new2 rb_str_new_cstr 00442 00443 VALUE 00444 rb_usascii_str_new_cstr(const char *ptr) 00445 { 00446 VALUE str = rb_str_new2(ptr); 00447 ENCODING_CODERANGE_SET(str, rb_usascii_encindex(), ENC_CODERANGE_7BIT); 00448 return str; 00449 } 00450 00451 RUBY_ALIAS_FUNCTION(rb_usascii_str_new2(const char *ptr), rb_usascii_str_new_cstr, (ptr)) 00452 #define rb_usascii_str_new2 rb_usascii_str_new_cstr 00453 00454 VALUE 00455 rb_tainted_str_new(const char *ptr, long len) 00456 { 00457 VALUE str = rb_str_new(ptr, len); 00458 00459 OBJ_TAINT(str); 00460 return str; 00461 } 00462 00463 VALUE 00464 rb_tainted_str_new_cstr(const char *ptr) 00465 { 00466 VALUE str = rb_str_new2(ptr); 00467 00468 OBJ_TAINT(str); 00469 return str; 00470 } 00471 00472 RUBY_ALIAS_FUNCTION(rb_tainted_str_new2(const char *ptr), rb_tainted_str_new_cstr, (ptr)) 00473 #define rb_tainted_str_new2 rb_tainted_str_new_cstr 00474 00475 VALUE 00476 rb_str_conv_enc_opts(VALUE str, rb_encoding *from, rb_encoding *to, int ecflags, VALUE ecopts) 00477 { 00478 rb_econv_t *ec; 00479 rb_econv_result_t ret; 00480 long len; 00481 VALUE newstr; 00482 const unsigned char *sp; 00483 unsigned char *dp; 00484 00485 if (!to) return str; 00486 if (from == to) return str; 00487 if ((rb_enc_asciicompat(to) && ENC_CODERANGE(str) == ENC_CODERANGE_7BIT) || 00488 to == rb_ascii8bit_encoding()) { 00489 if (STR_ENC_GET(str) != to) { 00490 str = rb_str_dup(str); 00491 rb_enc_associate(str, to); 00492 } 00493 return str; 00494 } 00495 00496 len = RSTRING_LEN(str); 00497 newstr = rb_str_new(0, len); 00498 00499 retry: 00500 ec = rb_econv_open_opts(from->name, to->name, ecflags, ecopts); 00501 if (!ec) return str; 00502 00503 sp = (unsigned char*)RSTRING_PTR(str); 00504 dp = (unsigned char*)RSTRING_PTR(newstr); 00505 ret = rb_econv_convert(ec, &sp, (unsigned char*)RSTRING_END(str), 00506 &dp, (unsigned char*)RSTRING_END(newstr), 0); 00507 rb_econv_close(ec); 00508 switch (ret) { 00509 case econv_destination_buffer_full: 00510 /* destination buffer short */ 00511 len = len < 2 ? 2 : len * 2; 00512 rb_str_resize(newstr, len); 00513 goto retry; 00514 00515 case econv_finished: 00516 len = dp - (unsigned char*)RSTRING_PTR(newstr); 00517 rb_str_set_len(newstr, len); 00518 rb_enc_associate(newstr, to); 00519 return newstr; 00520 00521 default: 00522 /* some error, return original */ 00523 return str; 00524 } 00525 } 00526 00527 VALUE 00528 rb_str_conv_enc(VALUE str, rb_encoding *from, rb_encoding *to) 00529 { 00530 return rb_str_conv_enc_opts(str, from, to, 0, Qnil); 00531 } 00532 00533 VALUE 00534 rb_external_str_new_with_enc(const char *ptr, long len, rb_encoding *eenc) 00535 { 00536 VALUE str; 00537 00538 str = rb_tainted_str_new(ptr, len); 00539 if (eenc == rb_usascii_encoding() && 00540 rb_enc_str_coderange(str) != ENC_CODERANGE_7BIT) { 00541 rb_enc_associate(str, rb_ascii8bit_encoding()); 00542 return str; 00543 } 00544 rb_enc_associate(str, eenc); 00545 return rb_str_conv_enc(str, eenc, rb_default_internal_encoding()); 00546 } 00547 00548 VALUE 00549 rb_external_str_new(const char *ptr, long len) 00550 { 00551 return rb_external_str_new_with_enc(ptr, len, rb_default_external_encoding()); 00552 } 00553 00554 VALUE 00555 rb_external_str_new_cstr(const char *ptr) 00556 { 00557 return rb_external_str_new_with_enc(ptr, strlen(ptr), rb_default_external_encoding()); 00558 } 00559 00560 VALUE 00561 rb_locale_str_new(const char *ptr, long len) 00562 { 00563 return rb_external_str_new_with_enc(ptr, len, rb_locale_encoding()); 00564 } 00565 00566 VALUE 00567 rb_locale_str_new_cstr(const char *ptr) 00568 { 00569 return rb_external_str_new_with_enc(ptr, strlen(ptr), rb_locale_encoding()); 00570 } 00571 00572 VALUE 00573 rb_filesystem_str_new(const char *ptr, long len) 00574 { 00575 return rb_external_str_new_with_enc(ptr, len, rb_filesystem_encoding()); 00576 } 00577 00578 VALUE 00579 rb_filesystem_str_new_cstr(const char *ptr) 00580 { 00581 return rb_external_str_new_with_enc(ptr, strlen(ptr), rb_filesystem_encoding()); 00582 } 00583 00584 VALUE 00585 rb_str_export(VALUE str) 00586 { 00587 return rb_str_conv_enc(str, STR_ENC_GET(str), rb_default_external_encoding()); 00588 } 00589 00590 VALUE 00591 rb_str_export_locale(VALUE str) 00592 { 00593 return rb_str_conv_enc(str, STR_ENC_GET(str), rb_locale_encoding()); 00594 } 00595 00596 VALUE 00597 rb_str_export_to_enc(VALUE str, rb_encoding *enc) 00598 { 00599 return rb_str_conv_enc(str, STR_ENC_GET(str), enc); 00600 } 00601 00602 static VALUE 00603 str_replace_shared(VALUE str2, VALUE str) 00604 { 00605 if (RSTRING_LEN(str) <= RSTRING_EMBED_LEN_MAX) { 00606 STR_SET_EMBED(str2); 00607 memcpy(RSTRING_PTR(str2), RSTRING_PTR(str), RSTRING_LEN(str)+1); 00608 STR_SET_EMBED_LEN(str2, RSTRING_LEN(str)); 00609 } 00610 else { 00611 str = rb_str_new_frozen(str); 00612 FL_SET(str2, STR_NOEMBED); 00613 RSTRING(str2)->as.heap.len = RSTRING_LEN(str); 00614 RSTRING(str2)->as.heap.ptr = RSTRING_PTR(str); 00615 RSTRING(str2)->as.heap.aux.shared = str; 00616 FL_SET(str2, ELTS_SHARED); 00617 } 00618 rb_enc_cr_str_exact_copy(str2, str); 00619 00620 return str2; 00621 } 00622 00623 static VALUE 00624 str_new_shared(VALUE klass, VALUE str) 00625 { 00626 return str_replace_shared(str_alloc(klass), str); 00627 } 00628 00629 static VALUE 00630 str_new3(VALUE klass, VALUE str) 00631 { 00632 return str_new_shared(klass, str); 00633 } 00634 00635 VALUE 00636 rb_str_new_shared(VALUE str) 00637 { 00638 VALUE str2 = str_new3(rb_obj_class(str), str); 00639 00640 OBJ_INFECT(str2, str); 00641 return str2; 00642 } 00643 00644 RUBY_ALIAS_FUNCTION(rb_str_new3(VALUE str), rb_str_new_shared, (str)) 00645 #define rb_str_new3 rb_str_new_shared 00646 00647 static VALUE 00648 str_new4(VALUE klass, VALUE str) 00649 { 00650 VALUE str2; 00651 00652 str2 = str_alloc(klass); 00653 STR_SET_NOEMBED(str2); 00654 RSTRING(str2)->as.heap.len = RSTRING_LEN(str); 00655 RSTRING(str2)->as.heap.ptr = RSTRING_PTR(str); 00656 if (STR_SHARED_P(str)) { 00657 VALUE shared = RSTRING(str)->as.heap.aux.shared; 00658 assert(OBJ_FROZEN(shared)); 00659 FL_SET(str2, ELTS_SHARED); 00660 RSTRING(str2)->as.heap.aux.shared = shared; 00661 } 00662 else { 00663 FL_SET(str, ELTS_SHARED); 00664 RSTRING(str)->as.heap.aux.shared = str2; 00665 } 00666 rb_enc_cr_str_exact_copy(str2, str); 00667 OBJ_INFECT(str2, str); 00668 return str2; 00669 } 00670 00671 VALUE 00672 rb_str_new_frozen(VALUE orig) 00673 { 00674 VALUE klass, str; 00675 00676 if (OBJ_FROZEN(orig)) return orig; 00677 klass = rb_obj_class(orig); 00678 if (STR_SHARED_P(orig) && (str = RSTRING(orig)->as.heap.aux.shared)) { 00679 long ofs; 00680 assert(OBJ_FROZEN(str)); 00681 ofs = RSTRING_LEN(str) - RSTRING_LEN(orig); 00682 if ((ofs > 0) || (klass != RBASIC(str)->klass) || 00683 (!OBJ_TAINTED(str) && OBJ_TAINTED(orig)) || 00684 ENCODING_GET(str) != ENCODING_GET(orig)) { 00685 str = str_new3(klass, str); 00686 RSTRING(str)->as.heap.ptr += ofs; 00687 RSTRING(str)->as.heap.len -= ofs; 00688 rb_enc_cr_str_exact_copy(str, orig); 00689 OBJ_INFECT(str, orig); 00690 } 00691 } 00692 else if (STR_EMBED_P(orig)) { 00693 str = str_new(klass, RSTRING_PTR(orig), RSTRING_LEN(orig)); 00694 rb_enc_cr_str_exact_copy(str, orig); 00695 OBJ_INFECT(str, orig); 00696 } 00697 else if (STR_ASSOC_P(orig)) { 00698 VALUE assoc = RSTRING(orig)->as.heap.aux.shared; 00699 FL_UNSET(orig, STR_ASSOC); 00700 str = str_new4(klass, orig); 00701 FL_SET(str, STR_ASSOC); 00702 RSTRING(str)->as.heap.aux.shared = assoc; 00703 } 00704 else { 00705 str = str_new4(klass, orig); 00706 } 00707 OBJ_FREEZE(str); 00708 return str; 00709 } 00710 00711 RUBY_ALIAS_FUNCTION(rb_str_new4(VALUE orig), rb_str_new_frozen, (orig)) 00712 #define rb_str_new4 rb_str_new_frozen 00713 00714 VALUE 00715 rb_str_new_with_class(VALUE obj, const char *ptr, long len) 00716 { 00717 return str_new(rb_obj_class(obj), ptr, len); 00718 } 00719 00720 RUBY_ALIAS_FUNCTION(rb_str_new5(VALUE obj, const char *ptr, long len), 00721 rb_str_new_with_class, (obj, ptr, len)) 00722 #define rb_str_new5 rb_str_new_with_class 00723 00724 static VALUE 00725 str_new_empty(VALUE str) 00726 { 00727 VALUE v = rb_str_new5(str, 0, 0); 00728 rb_enc_copy(v, str); 00729 OBJ_INFECT(v, str); 00730 return v; 00731 } 00732 00733 #define STR_BUF_MIN_SIZE 128 00734 00735 VALUE 00736 rb_str_buf_new(long capa) 00737 { 00738 VALUE str = str_alloc(rb_cString); 00739 00740 if (capa < STR_BUF_MIN_SIZE) { 00741 capa = STR_BUF_MIN_SIZE; 00742 } 00743 FL_SET(str, STR_NOEMBED); 00744 RSTRING(str)->as.heap.aux.capa = capa; 00745 RSTRING(str)->as.heap.ptr = ALLOC_N(char, capa+1); 00746 RSTRING(str)->as.heap.ptr[0] = '\0'; 00747 00748 return str; 00749 } 00750 00751 VALUE 00752 rb_str_buf_new_cstr(const char *ptr) 00753 { 00754 VALUE str; 00755 long len = strlen(ptr); 00756 00757 str = rb_str_buf_new(len); 00758 rb_str_buf_cat(str, ptr, len); 00759 00760 return str; 00761 } 00762 00763 RUBY_ALIAS_FUNCTION(rb_str_buf_new2(const char *ptr), rb_str_buf_new_cstr, (ptr)) 00764 #define rb_str_buf_new2 rb_str_buf_new_cstr 00765 00766 VALUE 00767 rb_str_tmp_new(long len) 00768 { 00769 return str_new(0, 0, len); 00770 } 00771 00772 void * 00773 rb_alloc_tmp_buffer(volatile VALUE *store, long len) 00774 { 00775 VALUE s = rb_str_tmp_new(len); 00776 *store = s; 00777 return RSTRING_PTR(s); 00778 } 00779 00780 void 00781 rb_free_tmp_buffer(volatile VALUE *store) 00782 { 00783 VALUE s = *store; 00784 *store = 0; 00785 if (s) rb_str_clear(s); 00786 } 00787 00788 void 00789 rb_str_free(VALUE str) 00790 { 00791 if (!STR_EMBED_P(str) && !STR_SHARED_P(str)) { 00792 xfree(RSTRING(str)->as.heap.ptr); 00793 } 00794 } 00795 00796 RUBY_FUNC_EXPORTED size_t 00797 rb_str_memsize(VALUE str) 00798 { 00799 if (!STR_EMBED_P(str) && !STR_SHARED_P(str)) { 00800 return RSTRING(str)->as.heap.aux.capa; 00801 } 00802 else { 00803 return 0; 00804 } 00805 } 00806 00807 VALUE 00808 rb_str_to_str(VALUE str) 00809 { 00810 return rb_convert_type(str, T_STRING, "String", "to_str"); 00811 } 00812 00813 static inline void str_discard(VALUE str); 00814 00815 void 00816 rb_str_shared_replace(VALUE str, VALUE str2) 00817 { 00818 rb_encoding *enc; 00819 int cr; 00820 if (str == str2) return; 00821 enc = STR_ENC_GET(str2); 00822 cr = ENC_CODERANGE(str2); 00823 str_discard(str); 00824 OBJ_INFECT(str, str2); 00825 if (RSTRING_LEN(str2) <= RSTRING_EMBED_LEN_MAX) { 00826 STR_SET_EMBED(str); 00827 memcpy(RSTRING_PTR(str), RSTRING_PTR(str2), RSTRING_LEN(str2)+1); 00828 STR_SET_EMBED_LEN(str, RSTRING_LEN(str2)); 00829 rb_enc_associate(str, enc); 00830 ENC_CODERANGE_SET(str, cr); 00831 return; 00832 } 00833 STR_SET_NOEMBED(str); 00834 STR_UNSET_NOCAPA(str); 00835 RSTRING(str)->as.heap.ptr = RSTRING_PTR(str2); 00836 RSTRING(str)->as.heap.len = RSTRING_LEN(str2); 00837 if (STR_NOCAPA_P(str2)) { 00838 FL_SET(str, RBASIC(str2)->flags & STR_NOCAPA); 00839 RSTRING(str)->as.heap.aux.shared = RSTRING(str2)->as.heap.aux.shared; 00840 } 00841 else { 00842 RSTRING(str)->as.heap.aux.capa = RSTRING(str2)->as.heap.aux.capa; 00843 } 00844 STR_SET_EMBED(str2); /* abandon str2 */ 00845 RSTRING_PTR(str2)[0] = 0; 00846 STR_SET_EMBED_LEN(str2, 0); 00847 rb_enc_associate(str, enc); 00848 ENC_CODERANGE_SET(str, cr); 00849 } 00850 00851 static ID id_to_s; 00852 00853 VALUE 00854 rb_obj_as_string(VALUE obj) 00855 { 00856 VALUE str; 00857 00858 if (TYPE(obj) == T_STRING) { 00859 return obj; 00860 } 00861 str = rb_funcall(obj, id_to_s, 0); 00862 if (TYPE(str) != T_STRING) 00863 return rb_any_to_s(obj); 00864 if (OBJ_TAINTED(obj)) OBJ_TAINT(str); 00865 return str; 00866 } 00867 00868 static VALUE 00869 str_replace(VALUE str, VALUE str2) 00870 { 00871 long len; 00872 00873 len = RSTRING_LEN(str2); 00874 if (STR_ASSOC_P(str2)) { 00875 str2 = rb_str_new4(str2); 00876 } 00877 if (STR_SHARED_P(str2)) { 00878 VALUE shared = RSTRING(str2)->as.heap.aux.shared; 00879 assert(OBJ_FROZEN(shared)); 00880 STR_SET_NOEMBED(str); 00881 RSTRING(str)->as.heap.len = len; 00882 RSTRING(str)->as.heap.ptr = RSTRING_PTR(str2); 00883 FL_SET(str, ELTS_SHARED); 00884 FL_UNSET(str, STR_ASSOC); 00885 RSTRING(str)->as.heap.aux.shared = shared; 00886 } 00887 else { 00888 str_replace_shared(str, str2); 00889 } 00890 00891 OBJ_INFECT(str, str2); 00892 rb_enc_cr_str_exact_copy(str, str2); 00893 return str; 00894 } 00895 00896 static VALUE 00897 str_duplicate(VALUE klass, VALUE str) 00898 { 00899 VALUE dup = str_alloc(klass); 00900 str_replace(dup, str); 00901 return dup; 00902 } 00903 00904 VALUE 00905 rb_str_dup(VALUE str) 00906 { 00907 return str_duplicate(rb_obj_class(str), str); 00908 } 00909 00910 VALUE 00911 rb_str_resurrect(VALUE str) 00912 { 00913 return str_replace(str_alloc(rb_cString), str); 00914 } 00915 00916 /* 00917 * call-seq: 00918 * String.new(str="") -> new_str 00919 * 00920 * Returns a new string object containing a copy of <i>str</i>. 00921 */ 00922 00923 static VALUE 00924 rb_str_init(int argc, VALUE *argv, VALUE str) 00925 { 00926 VALUE orig; 00927 00928 if (argc > 0 && rb_scan_args(argc, argv, "01", &orig) == 1) 00929 rb_str_replace(str, orig); 00930 return str; 00931 } 00932 00933 static inline long 00934 enc_strlen(const char *p, const char *e, rb_encoding *enc, int cr) 00935 { 00936 long c; 00937 const char *q; 00938 00939 if (rb_enc_mbmaxlen(enc) == rb_enc_mbminlen(enc)) { 00940 return (e - p + rb_enc_mbminlen(enc) - 1) / rb_enc_mbminlen(enc); 00941 } 00942 else if (rb_enc_asciicompat(enc)) { 00943 c = 0; 00944 if (cr == ENC_CODERANGE_7BIT || cr == ENC_CODERANGE_VALID) { 00945 while (p < e) { 00946 if (ISASCII(*p)) { 00947 q = search_nonascii(p, e); 00948 if (!q) 00949 return c + (e - p); 00950 c += q - p; 00951 p = q; 00952 } 00953 p += rb_enc_fast_mbclen(p, e, enc); 00954 c++; 00955 } 00956 } 00957 else { 00958 while (p < e) { 00959 if (ISASCII(*p)) { 00960 q = search_nonascii(p, e); 00961 if (!q) 00962 return c + (e - p); 00963 c += q - p; 00964 p = q; 00965 } 00966 p += rb_enc_mbclen(p, e, enc); 00967 c++; 00968 } 00969 } 00970 return c; 00971 } 00972 00973 for (c=0; p<e; c++) { 00974 p += rb_enc_mbclen(p, e, enc); 00975 } 00976 return c; 00977 } 00978 00979 long 00980 rb_enc_strlen(const char *p, const char *e, rb_encoding *enc) 00981 { 00982 return enc_strlen(p, e, enc, ENC_CODERANGE_UNKNOWN); 00983 } 00984 00985 long 00986 rb_enc_strlen_cr(const char *p, const char *e, rb_encoding *enc, int *cr) 00987 { 00988 long c; 00989 const char *q; 00990 int ret; 00991 00992 *cr = 0; 00993 if (rb_enc_mbmaxlen(enc) == rb_enc_mbminlen(enc)) { 00994 return (e - p + rb_enc_mbminlen(enc) - 1) / rb_enc_mbminlen(enc); 00995 } 00996 else if (rb_enc_asciicompat(enc)) { 00997 c = 0; 00998 while (p < e) { 00999 if (ISASCII(*p)) { 01000 q = search_nonascii(p, e); 01001 if (!q) { 01002 if (!*cr) *cr = ENC_CODERANGE_7BIT; 01003 return c + (e - p); 01004 } 01005 c += q - p; 01006 p = q; 01007 } 01008 ret = rb_enc_precise_mbclen(p, e, enc); 01009 if (MBCLEN_CHARFOUND_P(ret)) { 01010 *cr |= ENC_CODERANGE_VALID; 01011 p += MBCLEN_CHARFOUND_LEN(ret); 01012 } 01013 else { 01014 *cr = ENC_CODERANGE_BROKEN; 01015 p++; 01016 } 01017 c++; 01018 } 01019 if (!*cr) *cr = ENC_CODERANGE_7BIT; 01020 return c; 01021 } 01022 01023 for (c=0; p<e; c++) { 01024 ret = rb_enc_precise_mbclen(p, e, enc); 01025 if (MBCLEN_CHARFOUND_P(ret)) { 01026 *cr |= ENC_CODERANGE_VALID; 01027 p += MBCLEN_CHARFOUND_LEN(ret); 01028 } 01029 else { 01030 *cr = ENC_CODERANGE_BROKEN; 01031 if (p + rb_enc_mbminlen(enc) <= e) 01032 p += rb_enc_mbminlen(enc); 01033 else 01034 p = e; 01035 } 01036 } 01037 if (!*cr) *cr = ENC_CODERANGE_7BIT; 01038 return c; 01039 } 01040 01041 #ifdef NONASCII_MASK 01042 #define is_utf8_lead_byte(c) (((c)&0xC0) != 0x80) 01043 01044 /* 01045 * UTF-8 leading bytes have either 0xxxxxxx or 11xxxxxx 01046 * bit represention. (see http://en.wikipedia.org/wiki/UTF-8) 01047 * Therefore, following pseudo code can detect UTF-8 leading byte. 01048 * 01049 * if (!(byte & 0x80)) 01050 * byte |= 0x40; // turn on bit6 01051 * return ((byte>>6) & 1); // bit6 represent it's leading byte or not. 01052 * 01053 * This function calculate every bytes in the argument word `s' 01054 * using the above logic concurrently. and gather every bytes result. 01055 */ 01056 static inline VALUE 01057 count_utf8_lead_bytes_with_word(const VALUE *s) 01058 { 01059 VALUE d = *s; 01060 01061 /* Transform into bit0 represent UTF-8 leading or not. */ 01062 d |= ~(d>>1); 01063 d >>= 6; 01064 d &= NONASCII_MASK >> 7; 01065 01066 /* Gather every bytes. */ 01067 d += (d>>8); 01068 d += (d>>16); 01069 #if SIZEOF_VALUE == 8 01070 d += (d>>32); 01071 #endif 01072 return (d&0xF); 01073 } 01074 #endif 01075 01076 static long 01077 str_strlen(VALUE str, rb_encoding *enc) 01078 { 01079 const char *p, *e; 01080 long n; 01081 int cr; 01082 01083 if (single_byte_optimizable(str)) return RSTRING_LEN(str); 01084 if (!enc) enc = STR_ENC_GET(str); 01085 p = RSTRING_PTR(str); 01086 e = RSTRING_END(str); 01087 cr = ENC_CODERANGE(str); 01088 #ifdef NONASCII_MASK 01089 if (ENC_CODERANGE(str) == ENC_CODERANGE_VALID && 01090 enc == rb_utf8_encoding()) { 01091 01092 VALUE len = 0; 01093 if ((int)sizeof(VALUE) * 2 < e - p) { 01094 const VALUE *s, *t; 01095 const VALUE lowbits = sizeof(VALUE) - 1; 01096 s = (const VALUE*)(~lowbits & ((VALUE)p + lowbits)); 01097 t = (const VALUE*)(~lowbits & (VALUE)e); 01098 while (p < (const char *)s) { 01099 if (is_utf8_lead_byte(*p)) len++; 01100 p++; 01101 } 01102 while (s < t) { 01103 len += count_utf8_lead_bytes_with_word(s); 01104 s++; 01105 } 01106 p = (const char *)s; 01107 } 01108 while (p < e) { 01109 if (is_utf8_lead_byte(*p)) len++; 01110 p++; 01111 } 01112 return (long)len; 01113 } 01114 #endif 01115 n = rb_enc_strlen_cr(p, e, enc, &cr); 01116 if (cr) { 01117 ENC_CODERANGE_SET(str, cr); 01118 } 01119 return n; 01120 } 01121 01122 long 01123 rb_str_strlen(VALUE str) 01124 { 01125 return str_strlen(str, STR_ENC_GET(str)); 01126 } 01127 01128 /* 01129 * call-seq: 01130 * str.length -> integer 01131 * str.size -> integer 01132 * 01133 * Returns the character length of <i>str</i>. 01134 */ 01135 01136 VALUE 01137 rb_str_length(VALUE str) 01138 { 01139 long len; 01140 01141 len = str_strlen(str, STR_ENC_GET(str)); 01142 return LONG2NUM(len); 01143 } 01144 01145 /* 01146 * call-seq: 01147 * str.bytesize -> integer 01148 * 01149 * Returns the length of <i>str</i> in bytes. 01150 */ 01151 01152 static VALUE 01153 rb_str_bytesize(VALUE str) 01154 { 01155 return LONG2NUM(RSTRING_LEN(str)); 01156 } 01157 01158 /* 01159 * call-seq: 01160 * str.empty? -> true or false 01161 * 01162 * Returns <code>true</code> if <i>str</i> has a length of zero. 01163 * 01164 * "hello".empty? #=> false 01165 * "".empty? #=> true 01166 */ 01167 01168 static VALUE 01169 rb_str_empty(VALUE str) 01170 { 01171 if (RSTRING_LEN(str) == 0) 01172 return Qtrue; 01173 return Qfalse; 01174 } 01175 01176 /* 01177 * call-seq: 01178 * str + other_str -> new_str 01179 * 01180 * Concatenation---Returns a new <code>String</code> containing 01181 * <i>other_str</i> concatenated to <i>str</i>. 01182 * 01183 * "Hello from " + self.to_s #=> "Hello from main" 01184 */ 01185 01186 VALUE 01187 rb_str_plus(VALUE str1, VALUE str2) 01188 { 01189 VALUE str3; 01190 rb_encoding *enc; 01191 01192 StringValue(str2); 01193 enc = rb_enc_check(str1, str2); 01194 str3 = rb_str_new(0, RSTRING_LEN(str1)+RSTRING_LEN(str2)); 01195 memcpy(RSTRING_PTR(str3), RSTRING_PTR(str1), RSTRING_LEN(str1)); 01196 memcpy(RSTRING_PTR(str3) + RSTRING_LEN(str1), 01197 RSTRING_PTR(str2), RSTRING_LEN(str2)); 01198 RSTRING_PTR(str3)[RSTRING_LEN(str3)] = '\0'; 01199 01200 if (OBJ_TAINTED(str1) || OBJ_TAINTED(str2)) 01201 OBJ_TAINT(str3); 01202 ENCODING_CODERANGE_SET(str3, rb_enc_to_index(enc), 01203 ENC_CODERANGE_AND(ENC_CODERANGE(str1), ENC_CODERANGE(str2))); 01204 return str3; 01205 } 01206 01207 /* 01208 * call-seq: 01209 * str * integer -> new_str 01210 * 01211 * Copy---Returns a new <code>String</code> containing <i>integer</i> copies of 01212 * the receiver. 01213 * 01214 * "Ho! " * 3 #=> "Ho! Ho! Ho! " 01215 */ 01216 01217 VALUE 01218 rb_str_times(VALUE str, VALUE times) 01219 { 01220 VALUE str2; 01221 long n, len; 01222 char *ptr2; 01223 01224 len = NUM2LONG(times); 01225 if (len < 0) { 01226 rb_raise(rb_eArgError, "negative argument"); 01227 } 01228 if (len && LONG_MAX/len < RSTRING_LEN(str)) { 01229 rb_raise(rb_eArgError, "argument too big"); 01230 } 01231 01232 str2 = rb_str_new5(str, 0, len *= RSTRING_LEN(str)); 01233 ptr2 = RSTRING_PTR(str2); 01234 if (len) { 01235 n = RSTRING_LEN(str); 01236 memcpy(ptr2, RSTRING_PTR(str), n); 01237 while (n <= len/2) { 01238 memcpy(ptr2 + n, ptr2, n); 01239 n *= 2; 01240 } 01241 memcpy(ptr2 + n, ptr2, len-n); 01242 } 01243 ptr2[RSTRING_LEN(str2)] = '\0'; 01244 OBJ_INFECT(str2, str); 01245 rb_enc_cr_str_copy_for_substr(str2, str); 01246 01247 return str2; 01248 } 01249 01250 /* 01251 * call-seq: 01252 * str % arg -> new_str 01253 * 01254 * Format---Uses <i>str</i> as a format specification, and returns the result 01255 * of applying it to <i>arg</i>. If the format specification contains more than 01256 * one substitution, then <i>arg</i> must be an <code>Array</code> or <code>Hash</code> 01257 * containing the values to be substituted. See <code>Kernel::sprintf</code> for 01258 * details of the format string. 01259 * 01260 * "%05d" % 123 #=> "00123" 01261 * "%-5s: %08x" % [ "ID", self.object_id ] #=> "ID : 200e14d6" 01262 * "foo = %{foo}" % { :foo => 'bar' } #=> "foo = bar" 01263 */ 01264 01265 static VALUE 01266 rb_str_format_m(VALUE str, VALUE arg) 01267 { 01268 volatile VALUE tmp = rb_check_array_type(arg); 01269 01270 if (!NIL_P(tmp)) { 01271 return rb_str_format(RARRAY_LENINT(tmp), RARRAY_PTR(tmp), str); 01272 } 01273 return rb_str_format(1, &arg, str); 01274 } 01275 01276 static inline void 01277 str_modifiable(VALUE str) 01278 { 01279 if (FL_TEST(str, STR_TMPLOCK)) { 01280 rb_raise(rb_eRuntimeError, "can't modify string; temporarily locked"); 01281 } 01282 rb_check_frozen(str); 01283 if (!OBJ_UNTRUSTED(str) && rb_safe_level() >= 4) 01284 rb_raise(rb_eSecurityError, "Insecure: can't modify string"); 01285 } 01286 01287 static inline int 01288 str_independent(VALUE str) 01289 { 01290 str_modifiable(str); 01291 if (!STR_SHARED_P(str)) return 1; 01292 if (STR_EMBED_P(str)) return 1; 01293 return 0; 01294 } 01295 01296 static void 01297 str_make_independent_expand(VALUE str, long expand) 01298 { 01299 char *ptr; 01300 long len = RSTRING_LEN(str); 01301 long capa = len + expand; 01302 01303 if (len > capa) len = capa; 01304 ptr = ALLOC_N(char, capa + 1); 01305 if (RSTRING_PTR(str)) { 01306 memcpy(ptr, RSTRING_PTR(str), len); 01307 } 01308 STR_SET_NOEMBED(str); 01309 STR_UNSET_NOCAPA(str); 01310 ptr[len] = 0; 01311 RSTRING(str)->as.heap.ptr = ptr; 01312 RSTRING(str)->as.heap.len = len; 01313 RSTRING(str)->as.heap.aux.capa = capa; 01314 } 01315 01316 #define str_make_independent(str) str_make_independent_expand((str), 0L) 01317 01318 void 01319 rb_str_modify(VALUE str) 01320 { 01321 if (!str_independent(str)) 01322 str_make_independent(str); 01323 ENC_CODERANGE_CLEAR(str); 01324 } 01325 01326 void 01327 rb_str_modify_expand(VALUE str, long expand) 01328 { 01329 if (expand < 0) { 01330 rb_raise(rb_eArgError, "negative expanding string size"); 01331 } 01332 if (!str_independent(str)) { 01333 str_make_independent_expand(str, expand); 01334 } 01335 else if (expand > 0) { 01336 long len = RSTRING_LEN(str); 01337 long capa = len + expand; 01338 if (!STR_EMBED_P(str)) { 01339 REALLOC_N(RSTRING(str)->as.heap.ptr, char, capa+1); 01340 RSTRING(str)->as.heap.aux.capa = capa; 01341 } 01342 else if (capa > RSTRING_EMBED_LEN_MAX) { 01343 str_make_independent_expand(str, expand); 01344 } 01345 } 01346 ENC_CODERANGE_CLEAR(str); 01347 } 01348 01349 /* As rb_str_modify(), but don't clear coderange */ 01350 static void 01351 str_modify_keep_cr(VALUE str) 01352 { 01353 if (!str_independent(str)) 01354 str_make_independent(str); 01355 if (ENC_CODERANGE(str) == ENC_CODERANGE_BROKEN) 01356 /* Force re-scan later */ 01357 ENC_CODERANGE_CLEAR(str); 01358 } 01359 01360 static inline void 01361 str_discard(VALUE str) 01362 { 01363 str_modifiable(str); 01364 if (!STR_SHARED_P(str) && !STR_EMBED_P(str)) { 01365 xfree(RSTRING_PTR(str)); 01366 RSTRING(str)->as.heap.ptr = 0; 01367 RSTRING(str)->as.heap.len = 0; 01368 } 01369 } 01370 01371 void 01372 rb_str_associate(VALUE str, VALUE add) 01373 { 01374 /* sanity check */ 01375 rb_check_frozen(str); 01376 if (STR_ASSOC_P(str)) { 01377 /* already associated */ 01378 rb_ary_concat(RSTRING(str)->as.heap.aux.shared, add); 01379 } 01380 else { 01381 if (STR_SHARED_P(str)) { 01382 VALUE assoc = RSTRING(str)->as.heap.aux.shared; 01383 str_make_independent(str); 01384 if (STR_ASSOC_P(assoc)) { 01385 assoc = RSTRING(assoc)->as.heap.aux.shared; 01386 rb_ary_concat(assoc, add); 01387 add = assoc; 01388 } 01389 } 01390 else if (STR_EMBED_P(str)) { 01391 str_make_independent(str); 01392 } 01393 else if (RSTRING(str)->as.heap.aux.capa != RSTRING_LEN(str)) { 01394 RESIZE_CAPA(str, RSTRING_LEN(str)); 01395 } 01396 FL_SET(str, STR_ASSOC); 01397 RBASIC(add)->klass = 0; 01398 RSTRING(str)->as.heap.aux.shared = add; 01399 } 01400 } 01401 01402 VALUE 01403 rb_str_associated(VALUE str) 01404 { 01405 if (STR_SHARED_P(str)) str = RSTRING(str)->as.heap.aux.shared; 01406 if (STR_ASSOC_P(str)) { 01407 return RSTRING(str)->as.heap.aux.shared; 01408 } 01409 return Qfalse; 01410 } 01411 01412 VALUE 01413 rb_string_value(volatile VALUE *ptr) 01414 { 01415 VALUE s = *ptr; 01416 if (TYPE(s) != T_STRING) { 01417 s = rb_str_to_str(s); 01418 *ptr = s; 01419 } 01420 return s; 01421 } 01422 01423 char * 01424 rb_string_value_ptr(volatile VALUE *ptr) 01425 { 01426 VALUE str = rb_string_value(ptr); 01427 return RSTRING_PTR(str); 01428 } 01429 01430 char * 01431 rb_string_value_cstr(volatile VALUE *ptr) 01432 { 01433 VALUE str = rb_string_value(ptr); 01434 char *s = RSTRING_PTR(str); 01435 long len = RSTRING_LEN(str); 01436 01437 if (!s || memchr(s, 0, len)) { 01438 rb_raise(rb_eArgError, "string contains null byte"); 01439 } 01440 if (s[len]) { 01441 rb_str_modify(str); 01442 s = RSTRING_PTR(str); 01443 s[RSTRING_LEN(str)] = 0; 01444 } 01445 return s; 01446 } 01447 01448 VALUE 01449 rb_check_string_type(VALUE str) 01450 { 01451 str = rb_check_convert_type(str, T_STRING, "String", "to_str"); 01452 return str; 01453 } 01454 01455 /* 01456 * call-seq: 01457 * String.try_convert(obj) -> string or nil 01458 * 01459 * Try to convert <i>obj</i> into a String, using to_str method. 01460 * Returns converted string or nil if <i>obj</i> cannot be converted 01461 * for any reason. 01462 * 01463 * String.try_convert("str") #=> "str" 01464 * String.try_convert(/re/) #=> nil 01465 */ 01466 static VALUE 01467 rb_str_s_try_convert(VALUE dummy, VALUE str) 01468 { 01469 return rb_check_string_type(str); 01470 } 01471 01472 static char* 01473 str_nth_len(const char *p, const char *e, long *nthp, rb_encoding *enc) 01474 { 01475 long nth = *nthp; 01476 if (rb_enc_mbmaxlen(enc) == 1) { 01477 p += nth; 01478 } 01479 else if (rb_enc_mbmaxlen(enc) == rb_enc_mbminlen(enc)) { 01480 p += nth * rb_enc_mbmaxlen(enc); 01481 } 01482 else if (rb_enc_asciicompat(enc)) { 01483 const char *p2, *e2; 01484 int n; 01485 01486 while (p < e && 0 < nth) { 01487 e2 = p + nth; 01488 if (e < e2) { 01489 *nthp = nth; 01490 return (char *)e; 01491 } 01492 if (ISASCII(*p)) { 01493 p2 = search_nonascii(p, e2); 01494 if (!p2) { 01495 *nthp = nth; 01496 return (char *)e2; 01497 } 01498 nth -= p2 - p; 01499 p = p2; 01500 } 01501 n = rb_enc_mbclen(p, e, enc); 01502 p += n; 01503 nth--; 01504 } 01505 *nthp = nth; 01506 if (nth != 0) { 01507 return (char *)e; 01508 } 01509 return (char *)p; 01510 } 01511 else { 01512 while (p < e && nth--) { 01513 p += rb_enc_mbclen(p, e, enc); 01514 } 01515 } 01516 if (p > e) p = e; 01517 *nthp = nth; 01518 return (char*)p; 01519 } 01520 01521 char* 01522 rb_enc_nth(const char *p, const char *e, long nth, rb_encoding *enc) 01523 { 01524 return str_nth_len(p, e, &nth, enc); 01525 } 01526 01527 static char* 01528 str_nth(const char *p, const char *e, long nth, rb_encoding *enc, int singlebyte) 01529 { 01530 if (singlebyte) 01531 p += nth; 01532 else { 01533 p = str_nth_len(p, e, &nth, enc); 01534 } 01535 if (!p) return 0; 01536 if (p > e) p = e; 01537 return (char *)p; 01538 } 01539 01540 /* char offset to byte offset */ 01541 static long 01542 str_offset(const char *p, const char *e, long nth, rb_encoding *enc, int singlebyte) 01543 { 01544 const char *pp = str_nth(p, e, nth, enc, singlebyte); 01545 if (!pp) return e - p; 01546 return pp - p; 01547 } 01548 01549 long 01550 rb_str_offset(VALUE str, long pos) 01551 { 01552 return str_offset(RSTRING_PTR(str), RSTRING_END(str), pos, 01553 STR_ENC_GET(str), single_byte_optimizable(str)); 01554 } 01555 01556 #ifdef NONASCII_MASK 01557 static char * 01558 str_utf8_nth(const char *p, const char *e, long *nthp) 01559 { 01560 long nth = *nthp; 01561 if ((int)SIZEOF_VALUE * 2 < e - p && (int)SIZEOF_VALUE * 2 < nth) { 01562 const VALUE *s, *t; 01563 const VALUE lowbits = sizeof(VALUE) - 1; 01564 s = (const VALUE*)(~lowbits & ((VALUE)p + lowbits)); 01565 t = (const VALUE*)(~lowbits & (VALUE)e); 01566 while (p < (const char *)s) { 01567 if (is_utf8_lead_byte(*p)) nth--; 01568 p++; 01569 } 01570 do { 01571 nth -= count_utf8_lead_bytes_with_word(s); 01572 s++; 01573 } while (s < t && (int)sizeof(VALUE) <= nth); 01574 p = (char *)s; 01575 } 01576 while (p < e) { 01577 if (is_utf8_lead_byte(*p)) { 01578 if (nth == 0) break; 01579 nth--; 01580 } 01581 p++; 01582 } 01583 *nthp = nth; 01584 return (char *)p; 01585 } 01586 01587 static long 01588 str_utf8_offset(const char *p, const char *e, long nth) 01589 { 01590 const char *pp = str_utf8_nth(p, e, &nth); 01591 return pp - p; 01592 } 01593 #endif 01594 01595 /* byte offset to char offset */ 01596 long 01597 rb_str_sublen(VALUE str, long pos) 01598 { 01599 if (single_byte_optimizable(str) || pos < 0) 01600 return pos; 01601 else { 01602 char *p = RSTRING_PTR(str); 01603 return enc_strlen(p, p + pos, STR_ENC_GET(str), ENC_CODERANGE(str)); 01604 } 01605 } 01606 01607 VALUE 01608 rb_str_subseq(VALUE str, long beg, long len) 01609 { 01610 VALUE str2; 01611 01612 if (RSTRING_LEN(str) == beg + len && 01613 RSTRING_EMBED_LEN_MAX < len) { 01614 str2 = rb_str_new_shared(rb_str_new_frozen(str)); 01615 rb_str_drop_bytes(str2, beg); 01616 } 01617 else { 01618 str2 = rb_str_new5(str, RSTRING_PTR(str)+beg, len); 01619 } 01620 01621 rb_enc_cr_str_copy_for_substr(str2, str); 01622 OBJ_INFECT(str2, str); 01623 01624 return str2; 01625 } 01626 01627 VALUE 01628 rb_str_substr(VALUE str, long beg, long len) 01629 { 01630 rb_encoding *enc = STR_ENC_GET(str); 01631 VALUE str2; 01632 char *p, *s = RSTRING_PTR(str), *e = s + RSTRING_LEN(str); 01633 01634 if (len < 0) return Qnil; 01635 if (!RSTRING_LEN(str)) { 01636 len = 0; 01637 } 01638 if (single_byte_optimizable(str)) { 01639 if (beg > RSTRING_LEN(str)) return Qnil; 01640 if (beg < 0) { 01641 beg += RSTRING_LEN(str); 01642 if (beg < 0) return Qnil; 01643 } 01644 if (beg + len > RSTRING_LEN(str)) 01645 len = RSTRING_LEN(str) - beg; 01646 if (len <= 0) { 01647 len = 0; 01648 p = 0; 01649 } 01650 else 01651 p = s + beg; 01652 goto sub; 01653 } 01654 if (beg < 0) { 01655 if (len > -beg) len = -beg; 01656 if (-beg * rb_enc_mbmaxlen(enc) < RSTRING_LEN(str) / 8) { 01657 beg = -beg; 01658 while (beg-- > len && (e = rb_enc_prev_char(s, e, e, enc)) != 0); 01659 p = e; 01660 if (!p) return Qnil; 01661 while (len-- > 0 && (p = rb_enc_prev_char(s, p, e, enc)) != 0); 01662 if (!p) return Qnil; 01663 len = e - p; 01664 goto sub; 01665 } 01666 else { 01667 beg += str_strlen(str, enc); 01668 if (beg < 0) return Qnil; 01669 } 01670 } 01671 else if (beg > 0 && beg > RSTRING_LEN(str)) { 01672 return Qnil; 01673 } 01674 if (len == 0) { 01675 if (beg > str_strlen(str, enc)) return Qnil; 01676 p = 0; 01677 } 01678 #ifdef NONASCII_MASK 01679 else if (ENC_CODERANGE(str) == ENC_CODERANGE_VALID && 01680 enc == rb_utf8_encoding()) { 01681 p = str_utf8_nth(s, e, &beg); 01682 if (beg > 0) return Qnil; 01683 len = str_utf8_offset(p, e, len); 01684 } 01685 #endif 01686 else if (rb_enc_mbmaxlen(enc) == rb_enc_mbminlen(enc)) { 01687 int char_sz = rb_enc_mbmaxlen(enc); 01688 01689 p = s + beg * char_sz; 01690 if (p > e) { 01691 return Qnil; 01692 } 01693 else if (len * char_sz > e - p) 01694 len = e - p; 01695 else 01696 len *= char_sz; 01697 } 01698 else if ((p = str_nth_len(s, e, &beg, enc)) == e) { 01699 if (beg > 0) return Qnil; 01700 len = 0; 01701 } 01702 else { 01703 len = str_offset(p, e, len, enc, 0); 01704 } 01705 sub: 01706 if (len > RSTRING_EMBED_LEN_MAX && beg + len == RSTRING_LEN(str)) { 01707 str2 = rb_str_new4(str); 01708 str2 = str_new3(rb_obj_class(str2), str2); 01709 RSTRING(str2)->as.heap.ptr += RSTRING(str2)->as.heap.len - len; 01710 RSTRING(str2)->as.heap.len = len; 01711 } 01712 else { 01713 str2 = rb_str_new5(str, p, len); 01714 rb_enc_cr_str_copy_for_substr(str2, str); 01715 OBJ_INFECT(str2, str); 01716 } 01717 01718 return str2; 01719 } 01720 01721 VALUE 01722 rb_str_freeze(VALUE str) 01723 { 01724 if (STR_ASSOC_P(str)) { 01725 VALUE ary = RSTRING(str)->as.heap.aux.shared; 01726 OBJ_FREEZE(ary); 01727 } 01728 return rb_obj_freeze(str); 01729 } 01730 01731 RUBY_ALIAS_FUNCTION(rb_str_dup_frozen(VALUE str), rb_str_new_frozen, (str)) 01732 #define rb_str_dup_frozen rb_str_new_frozen 01733 01734 VALUE 01735 rb_str_locktmp(VALUE str) 01736 { 01737 if (FL_TEST(str, STR_TMPLOCK)) { 01738 rb_raise(rb_eRuntimeError, "temporal locking already locked string"); 01739 } 01740 FL_SET(str, STR_TMPLOCK); 01741 return str; 01742 } 01743 01744 VALUE 01745 rb_str_unlocktmp(VALUE str) 01746 { 01747 if (!FL_TEST(str, STR_TMPLOCK)) { 01748 rb_raise(rb_eRuntimeError, "temporal unlocking already unlocked string"); 01749 } 01750 FL_UNSET(str, STR_TMPLOCK); 01751 return str; 01752 } 01753 01754 void 01755 rb_str_set_len(VALUE str, long len) 01756 { 01757 long capa; 01758 01759 str_modifiable(str); 01760 if (STR_SHARED_P(str)) { 01761 rb_raise(rb_eRuntimeError, "can't set length of shared string"); 01762 } 01763 if (len > (capa = (long)rb_str_capacity(str))) { 01764 rb_bug("probable buffer overflow: %ld for %ld", len, capa); 01765 } 01766 STR_SET_LEN(str, len); 01767 RSTRING_PTR(str)[len] = '\0'; 01768 } 01769 01770 VALUE 01771 rb_str_resize(VALUE str, long len) 01772 { 01773 long slen; 01774 int independent; 01775 01776 if (len < 0) { 01777 rb_raise(rb_eArgError, "negative string size (or size too big)"); 01778 } 01779 01780 independent = str_independent(str); 01781 ENC_CODERANGE_CLEAR(str); 01782 slen = RSTRING_LEN(str); 01783 if (len != slen) { 01784 if (STR_EMBED_P(str)) { 01785 if (len <= RSTRING_EMBED_LEN_MAX) { 01786 STR_SET_EMBED_LEN(str, len); 01787 RSTRING(str)->as.ary[len] = '\0'; 01788 return str; 01789 } 01790 str_make_independent_expand(str, len - slen); 01791 STR_SET_NOEMBED(str); 01792 } 01793 else if (len <= RSTRING_EMBED_LEN_MAX) { 01794 char *ptr = RSTRING(str)->as.heap.ptr; 01795 STR_SET_EMBED(str); 01796 if (slen > len) slen = len; 01797 if (slen > 0) MEMCPY(RSTRING(str)->as.ary, ptr, char, slen); 01798 RSTRING(str)->as.ary[len] = '\0'; 01799 STR_SET_EMBED_LEN(str, len); 01800 if (independent) xfree(ptr); 01801 return str; 01802 } 01803 else if (!independent) { 01804 str_make_independent_expand(str, len - slen); 01805 } 01806 else if (slen < len || slen - len > 1024) { 01807 REALLOC_N(RSTRING(str)->as.heap.ptr, char, len+1); 01808 } 01809 if (!STR_NOCAPA_P(str)) { 01810 RSTRING(str)->as.heap.aux.capa = len; 01811 } 01812 RSTRING(str)->as.heap.len = len; 01813 RSTRING(str)->as.heap.ptr[len] = '\0'; /* sentinel */ 01814 } 01815 return str; 01816 } 01817 01818 static VALUE 01819 str_buf_cat(VALUE str, const char *ptr, long len) 01820 { 01821 long capa, total, off = -1; 01822 01823 if (ptr >= RSTRING_PTR(str) && ptr <= RSTRING_END(str)) { 01824 off = ptr - RSTRING_PTR(str); 01825 } 01826 rb_str_modify(str); 01827 if (len == 0) return 0; 01828 if (STR_ASSOC_P(str)) { 01829 FL_UNSET(str, STR_ASSOC); 01830 capa = RSTRING(str)->as.heap.aux.capa = RSTRING_LEN(str); 01831 } 01832 else if (STR_EMBED_P(str)) { 01833 capa = RSTRING_EMBED_LEN_MAX; 01834 } 01835 else { 01836 capa = RSTRING(str)->as.heap.aux.capa; 01837 } 01838 if (RSTRING_LEN(str) >= LONG_MAX - len) { 01839 rb_raise(rb_eArgError, "string sizes too big"); 01840 } 01841 total = RSTRING_LEN(str)+len; 01842 if (capa <= total) { 01843 while (total > capa) { 01844 if (capa + 1 >= LONG_MAX / 2) { 01845 capa = (total + 4095) / 4096; 01846 break; 01847 } 01848 capa = (capa + 1) * 2; 01849 } 01850 RESIZE_CAPA(str, capa); 01851 } 01852 if (off != -1) { 01853 ptr = RSTRING_PTR(str) + off; 01854 } 01855 memcpy(RSTRING_PTR(str) + RSTRING_LEN(str), ptr, len); 01856 STR_SET_LEN(str, total); 01857 RSTRING_PTR(str)[total] = '\0'; /* sentinel */ 01858 01859 return str; 01860 } 01861 01862 #define str_buf_cat2(str, ptr) str_buf_cat((str), (ptr), strlen(ptr)) 01863 01864 VALUE 01865 rb_str_buf_cat(VALUE str, const char *ptr, long len) 01866 { 01867 if (len == 0) return str; 01868 if (len < 0) { 01869 rb_raise(rb_eArgError, "negative string size (or size too big)"); 01870 } 01871 return str_buf_cat(str, ptr, len); 01872 } 01873 01874 VALUE 01875 rb_str_buf_cat2(VALUE str, const char *ptr) 01876 { 01877 return rb_str_buf_cat(str, ptr, strlen(ptr)); 01878 } 01879 01880 VALUE 01881 rb_str_cat(VALUE str, const char *ptr, long len) 01882 { 01883 if (len < 0) { 01884 rb_raise(rb_eArgError, "negative string size (or size too big)"); 01885 } 01886 if (STR_ASSOC_P(str)) { 01887 char *p; 01888 rb_str_modify_expand(str, len); 01889 p = RSTRING(str)->as.heap.ptr; 01890 memcpy(p + RSTRING(str)->as.heap.len, ptr, len); 01891 len = RSTRING(str)->as.heap.len += len; 01892 p[len] = '\0'; /* sentinel */ 01893 return str; 01894 } 01895 01896 return rb_str_buf_cat(str, ptr, len); 01897 } 01898 01899 VALUE 01900 rb_str_cat2(VALUE str, const char *ptr) 01901 { 01902 return rb_str_cat(str, ptr, strlen(ptr)); 01903 } 01904 01905 static VALUE 01906 rb_enc_cr_str_buf_cat(VALUE str, const char *ptr, long len, 01907 int ptr_encindex, int ptr_cr, int *ptr_cr_ret) 01908 { 01909 int str_encindex = ENCODING_GET(str); 01910 int res_encindex; 01911 int str_cr, res_cr; 01912 01913 str_cr = ENC_CODERANGE(str); 01914 01915 if (str_encindex == ptr_encindex) { 01916 if (str_cr == ENC_CODERANGE_UNKNOWN) 01917 ptr_cr = ENC_CODERANGE_UNKNOWN; 01918 else if (ptr_cr == ENC_CODERANGE_UNKNOWN) { 01919 ptr_cr = coderange_scan(ptr, len, rb_enc_from_index(ptr_encindex)); 01920 } 01921 } 01922 else { 01923 rb_encoding *str_enc = rb_enc_from_index(str_encindex); 01924 rb_encoding *ptr_enc = rb_enc_from_index(ptr_encindex); 01925 if (!rb_enc_asciicompat(str_enc) || !rb_enc_asciicompat(ptr_enc)) { 01926 if (len == 0) 01927 return str; 01928 if (RSTRING_LEN(str) == 0) { 01929 rb_str_buf_cat(str, ptr, len); 01930 ENCODING_CODERANGE_SET(str, ptr_encindex, ptr_cr); 01931 return str; 01932 } 01933 goto incompatible; 01934 } 01935 if (ptr_cr == ENC_CODERANGE_UNKNOWN) { 01936 ptr_cr = coderange_scan(ptr, len, ptr_enc); 01937 } 01938 if (str_cr == ENC_CODERANGE_UNKNOWN) { 01939 if (ENCODING_IS_ASCII8BIT(str) || ptr_cr != ENC_CODERANGE_7BIT) { 01940 str_cr = rb_enc_str_coderange(str); 01941 } 01942 } 01943 } 01944 if (ptr_cr_ret) 01945 *ptr_cr_ret = ptr_cr; 01946 01947 if (str_encindex != ptr_encindex && 01948 str_cr != ENC_CODERANGE_7BIT && 01949 ptr_cr != ENC_CODERANGE_7BIT) { 01950 incompatible: 01951 rb_raise(rb_eEncCompatError, "incompatible character encodings: %s and %s", 01952 rb_enc_name(rb_enc_from_index(str_encindex)), 01953 rb_enc_name(rb_enc_from_index(ptr_encindex))); 01954 } 01955 01956 if (str_cr == ENC_CODERANGE_UNKNOWN) { 01957 res_encindex = str_encindex; 01958 res_cr = ENC_CODERANGE_UNKNOWN; 01959 } 01960 else if (str_cr == ENC_CODERANGE_7BIT) { 01961 if (ptr_cr == ENC_CODERANGE_7BIT) { 01962 res_encindex = str_encindex; 01963 res_cr = ENC_CODERANGE_7BIT; 01964 } 01965 else { 01966 res_encindex = ptr_encindex; 01967 res_cr = ptr_cr; 01968 } 01969 } 01970 else if (str_cr == ENC_CODERANGE_VALID) { 01971 res_encindex = str_encindex; 01972 if (ptr_cr == ENC_CODERANGE_7BIT || ptr_cr == ENC_CODERANGE_VALID) 01973 res_cr = str_cr; 01974 else 01975 res_cr = ptr_cr; 01976 } 01977 else { /* str_cr == ENC_CODERANGE_BROKEN */ 01978 res_encindex = str_encindex; 01979 res_cr = str_cr; 01980 if (0 < len) res_cr = ENC_CODERANGE_UNKNOWN; 01981 } 01982 01983 if (len < 0) { 01984 rb_raise(rb_eArgError, "negative string size (or size too big)"); 01985 } 01986 str_buf_cat(str, ptr, len); 01987 ENCODING_CODERANGE_SET(str, res_encindex, res_cr); 01988 return str; 01989 } 01990 01991 VALUE 01992 rb_enc_str_buf_cat(VALUE str, const char *ptr, long len, rb_encoding *ptr_enc) 01993 { 01994 return rb_enc_cr_str_buf_cat(str, ptr, len, 01995 rb_enc_to_index(ptr_enc), ENC_CODERANGE_UNKNOWN, NULL); 01996 } 01997 01998 VALUE 01999 rb_str_buf_cat_ascii(VALUE str, const char *ptr) 02000 { 02001 /* ptr must reference NUL terminated ASCII string. */ 02002 int encindex = ENCODING_GET(str); 02003 rb_encoding *enc = rb_enc_from_index(encindex); 02004 if (rb_enc_asciicompat(enc)) { 02005 return rb_enc_cr_str_buf_cat(str, ptr, strlen(ptr), 02006 encindex, ENC_CODERANGE_7BIT, 0); 02007 } 02008 else { 02009 char *buf = ALLOCA_N(char, rb_enc_mbmaxlen(enc)); 02010 while (*ptr) { 02011 unsigned int c = (unsigned char)*ptr; 02012 int len = rb_enc_codelen(c, enc); 02013 rb_enc_mbcput(c, buf, enc); 02014 rb_enc_cr_str_buf_cat(str, buf, len, 02015 encindex, ENC_CODERANGE_VALID, 0); 02016 ptr++; 02017 } 02018 return str; 02019 } 02020 } 02021 02022 VALUE 02023 rb_str_buf_append(VALUE str, VALUE str2) 02024 { 02025 int str2_cr; 02026 02027 str2_cr = ENC_CODERANGE(str2); 02028 02029 rb_enc_cr_str_buf_cat(str, RSTRING_PTR(str2), RSTRING_LEN(str2), 02030 ENCODING_GET(str2), str2_cr, &str2_cr); 02031 02032 OBJ_INFECT(str, str2); 02033 ENC_CODERANGE_SET(str2, str2_cr); 02034 02035 return str; 02036 } 02037 02038 VALUE 02039 rb_str_append(VALUE str, VALUE str2) 02040 { 02041 rb_encoding *enc; 02042 int cr, cr2; 02043 long len2; 02044 02045 StringValue(str2); 02046 if ((len2 = RSTRING_LEN(str2)) > 0 && STR_ASSOC_P(str)) { 02047 long len = RSTRING_LEN(str) + len2; 02048 enc = rb_enc_check(str, str2); 02049 cr = ENC_CODERANGE(str); 02050 if ((cr2 = ENC_CODERANGE(str2)) > cr) cr = cr2; 02051 rb_str_modify_expand(str, len2); 02052 memcpy(RSTRING(str)->as.heap.ptr + RSTRING(str)->as.heap.len, 02053 RSTRING_PTR(str2), len2+1); 02054 RSTRING(str)->as.heap.len = len; 02055 rb_enc_associate(str, enc); 02056 ENC_CODERANGE_SET(str, cr); 02057 OBJ_INFECT(str, str2); 02058 return str; 02059 } 02060 return rb_str_buf_append(str, str2); 02061 } 02062 02063 /* 02064 * call-seq: 02065 * str << integer -> str 02066 * str.concat(integer) -> str 02067 * str << obj -> str 02068 * str.concat(obj) -> str 02069 * 02070 * Append---Concatenates the given object to <i>str</i>. If the object is a 02071 * <code>Integer</code>, it is considered as a codepoint, and is converted 02072 * to a character before concatenation. 02073 * 02074 * a = "hello " 02075 * a << "world" #=> "hello world" 02076 * a.concat(33) #=> "hello world!" 02077 */ 02078 02079 VALUE 02080 rb_str_concat(VALUE str1, VALUE str2) 02081 { 02082 unsigned int code; 02083 rb_encoding *enc = STR_ENC_GET(str1); 02084 02085 if (FIXNUM_P(str2) || TYPE(str2) == T_BIGNUM) { 02086 if (rb_num_to_uint(str2, &code) == 0) { 02087 } 02088 else if (FIXNUM_P(str2)) { 02089 rb_raise(rb_eRangeError, "%ld out of char range", FIX2LONG(str2)); 02090 } 02091 else { 02092 rb_raise(rb_eRangeError, "bignum out of char range"); 02093 } 02094 } 02095 else { 02096 return rb_str_append(str1, str2); 02097 } 02098 02099 if (enc == rb_usascii_encoding()) { 02100 /* US-ASCII automatically extended to ASCII-8BIT */ 02101 char buf[1] = {(char)code}; 02102 if (code > 0xFF) { 02103 rb_raise(rb_eRangeError, "%u out of char range", code); 02104 } 02105 rb_str_cat(str1, buf, 1); 02106 if (code > 127) { 02107 rb_enc_associate(str1, rb_ascii8bit_encoding()); 02108 ENC_CODERANGE_SET(str1, ENC_CODERANGE_VALID); 02109 } 02110 } 02111 else { 02112 long pos = RSTRING_LEN(str1); 02113 int cr = ENC_CODERANGE(str1); 02114 int len; 02115 char *buf; 02116 02117 switch (len = rb_enc_codelen(code, enc)) { 02118 case ONIGERR_INVALID_CODE_POINT_VALUE: 02119 rb_raise(rb_eRangeError, "invalid codepoint 0x%X in %s", code, rb_enc_name(enc)); 02120 break; 02121 case ONIGERR_TOO_BIG_WIDE_CHAR_VALUE: 02122 case 0: 02123 rb_raise(rb_eRangeError, "%u out of char range", code); 02124 break; 02125 } 02126 buf = ALLOCA_N(char, len + 1); 02127 rb_enc_mbcput(code, buf, enc); 02128 if (rb_enc_precise_mbclen(buf, buf + len + 1, enc) != len) { 02129 rb_raise(rb_eRangeError, "invalid codepoint 0x%X in %s", code, rb_enc_name(enc)); 02130 } 02131 rb_str_resize(str1, pos+len); 02132 strncpy(RSTRING_PTR(str1) + pos, buf, len); 02133 if (cr == ENC_CODERANGE_7BIT && code > 127) 02134 cr = ENC_CODERANGE_VALID; 02135 ENC_CODERANGE_SET(str1, cr); 02136 } 02137 return str1; 02138 } 02139 02140 /* 02141 * call-seq: 02142 * str.prepend(other_str) -> str 02143 * 02144 * Prepend---Prepend the given string to <i>str</i>. 02145 * 02146 * a = "world" 02147 * a.prepend("hello ") #=> "hello world" 02148 * a #=> "hello world" 02149 */ 02150 02151 static VALUE 02152 rb_str_prepend(VALUE str, VALUE str2) 02153 { 02154 StringValue(str2); 02155 StringValue(str); 02156 rb_str_update(str, 0L, 0L, str2); 02157 return str; 02158 } 02159 02160 st_index_t 02161 rb_memhash(const void *ptr, long len) 02162 { 02163 return st_hash(ptr, len, rb_hash_start((st_index_t)len)); 02164 } 02165 02166 st_index_t 02167 rb_str_hash(VALUE str) 02168 { 02169 int e = ENCODING_GET(str); 02170 if (e && rb_enc_str_coderange(str) == ENC_CODERANGE_7BIT) { 02171 e = 0; 02172 } 02173 return rb_memhash((const void *)RSTRING_PTR(str), RSTRING_LEN(str)) ^ e; 02174 } 02175 02176 int 02177 rb_str_hash_cmp(VALUE str1, VALUE str2) 02178 { 02179 long len; 02180 02181 if (!rb_str_comparable(str1, str2)) return 1; 02182 if (RSTRING_LEN(str1) == (len = RSTRING_LEN(str2)) && 02183 memcmp(RSTRING_PTR(str1), RSTRING_PTR(str2), len) == 0) { 02184 return 0; 02185 } 02186 return 1; 02187 } 02188 02189 /* 02190 * call-seq: 02191 * str.hash -> fixnum 02192 * 02193 * Return a hash based on the string's length and content. 02194 */ 02195 02196 static VALUE 02197 rb_str_hash_m(VALUE str) 02198 { 02199 st_index_t hval = rb_str_hash(str); 02200 return INT2FIX(hval); 02201 } 02202 02203 #define lesser(a,b) (((a)>(b))?(b):(a)) 02204 02205 int 02206 rb_str_comparable(VALUE str1, VALUE str2) 02207 { 02208 int idx1, idx2; 02209 int rc1, rc2; 02210 02211 if (RSTRING_LEN(str1) == 0) return TRUE; 02212 if (RSTRING_LEN(str2) == 0) return TRUE; 02213 idx1 = ENCODING_GET(str1); 02214 idx2 = ENCODING_GET(str2); 02215 if (idx1 == idx2) return TRUE; 02216 rc1 = rb_enc_str_coderange(str1); 02217 rc2 = rb_enc_str_coderange(str2); 02218 if (rc1 == ENC_CODERANGE_7BIT) { 02219 if (rc2 == ENC_CODERANGE_7BIT) return TRUE; 02220 if (rb_enc_asciicompat(rb_enc_from_index(idx2))) 02221 return TRUE; 02222 } 02223 if (rc2 == ENC_CODERANGE_7BIT) { 02224 if (rb_enc_asciicompat(rb_enc_from_index(idx1))) 02225 return TRUE; 02226 } 02227 return FALSE; 02228 } 02229 02230 int 02231 rb_str_cmp(VALUE str1, VALUE str2) 02232 { 02233 long len1, len2; 02234 const char *ptr1, *ptr2; 02235 int retval; 02236 02237 if (str1 == str2) return 0; 02238 RSTRING_GETMEM(str1, ptr1, len1); 02239 RSTRING_GETMEM(str2, ptr2, len2); 02240 if (ptr1 == ptr2 || (retval = memcmp(ptr1, ptr2, lesser(len1, len2))) == 0) { 02241 if (len1 == len2) { 02242 if (!rb_str_comparable(str1, str2)) { 02243 if (ENCODING_GET(str1) > ENCODING_GET(str2)) 02244 return 1; 02245 return -1; 02246 } 02247 return 0; 02248 } 02249 if (len1 > len2) return 1; 02250 return -1; 02251 } 02252 if (retval > 0) return 1; 02253 return -1; 02254 } 02255 02256 /* expect tail call optimization */ 02257 static VALUE 02258 str_eql(const VALUE str1, const VALUE str2) 02259 { 02260 const long len = RSTRING_LEN(str1); 02261 const char *ptr1, *ptr2; 02262 02263 if (len != RSTRING_LEN(str2)) return Qfalse; 02264 if (!rb_str_comparable(str1, str2)) return Qfalse; 02265 if ((ptr1 = RSTRING_PTR(str1)) == (ptr2 = RSTRING_PTR(str2))) 02266 return Qtrue; 02267 if (memcmp(ptr1, ptr2, len) == 0) 02268 return Qtrue; 02269 return Qfalse; 02270 } 02271 /* 02272 * call-seq: 02273 * str == obj -> true or false 02274 * 02275 * Equality---If <i>obj</i> is not a <code>String</code>, returns 02276 * <code>false</code>. Otherwise, returns <code>true</code> if <i>str</i> 02277 * <code><=></code> <i>obj</i> returns zero. 02278 */ 02279 02280 VALUE 02281 rb_str_equal(VALUE str1, VALUE str2) 02282 { 02283 if (str1 == str2) return Qtrue; 02284 if (TYPE(str2) != T_STRING) { 02285 if (!rb_respond_to(str2, rb_intern("to_str"))) { 02286 return Qfalse; 02287 } 02288 return rb_equal(str2, str1); 02289 } 02290 return str_eql(str1, str2); 02291 } 02292 02293 /* 02294 * call-seq: 02295 * str.eql?(other) -> true or false 02296 * 02297 * Two strings are equal if they have the same length and content. 02298 */ 02299 02300 static VALUE 02301 rb_str_eql(VALUE str1, VALUE str2) 02302 { 02303 if (str1 == str2) return Qtrue; 02304 if (TYPE(str2) != T_STRING) return Qfalse; 02305 return str_eql(str1, str2); 02306 } 02307 02308 /* 02309 * call-seq: 02310 * str <=> other_str -> -1, 0, +1 or nil 02311 * 02312 * Comparison---Returns -1 if <i>other_str</i> is greater than, 0 if 02313 * <i>other_str</i> is equal to, and +1 if <i>other_str</i> is less than 02314 * <i>str</i>. If the strings are of different lengths, and the strings are 02315 * equal when compared up to the shortest length, then the longer string is 02316 * considered greater than the shorter one. In older versions of Ruby, setting 02317 * <code>$=</code> allowed case-insensitive comparisons; this is now deprecated 02318 * in favor of using <code>String#casecmp</code>. 02319 * 02320 * <code><=></code> is the basis for the methods <code><</code>, 02321 * <code><=</code>, <code>></code>, <code>>=</code>, and <code>between?</code>, 02322 * included from module <code>Comparable</code>. The method 02323 * <code>String#==</code> does not use <code>Comparable#==</code>. 02324 * 02325 * "abcdef" <=> "abcde" #=> 1 02326 * "abcdef" <=> "abcdef" #=> 0 02327 * "abcdef" <=> "abcdefg" #=> -1 02328 * "abcdef" <=> "ABCDEF" #=> 1 02329 */ 02330 02331 static VALUE 02332 rb_str_cmp_m(VALUE str1, VALUE str2) 02333 { 02334 long result; 02335 02336 if (TYPE(str2) != T_STRING) { 02337 if (!rb_respond_to(str2, rb_intern("to_str"))) { 02338 return Qnil; 02339 } 02340 else if (!rb_respond_to(str2, rb_intern("<=>"))) { 02341 return Qnil; 02342 } 02343 else { 02344 VALUE tmp = rb_funcall(str2, rb_intern("<=>"), 1, str1); 02345 02346 if (NIL_P(tmp)) return Qnil; 02347 if (!FIXNUM_P(tmp)) { 02348 return rb_funcall(LONG2FIX(0), '-', 1, tmp); 02349 } 02350 result = -FIX2LONG(tmp); 02351 } 02352 } 02353 else { 02354 result = rb_str_cmp(str1, str2); 02355 } 02356 return LONG2NUM(result); 02357 } 02358 02359 /* 02360 * call-seq: 02361 * str.casecmp(other_str) -> -1, 0, +1 or nil 02362 * 02363 * Case-insensitive version of <code>String#<=></code>. 02364 * 02365 * "abcdef".casecmp("abcde") #=> 1 02366 * "aBcDeF".casecmp("abcdef") #=> 0 02367 * "abcdef".casecmp("abcdefg") #=> -1 02368 * "abcdef".casecmp("ABCDEF") #=> 0 02369 */ 02370 02371 static VALUE 02372 rb_str_casecmp(VALUE str1, VALUE str2) 02373 { 02374 long len; 02375 rb_encoding *enc; 02376 char *p1, *p1end, *p2, *p2end; 02377 02378 StringValue(str2); 02379 enc = rb_enc_compatible(str1, str2); 02380 if (!enc) { 02381 return Qnil; 02382 } 02383 02384 p1 = RSTRING_PTR(str1); p1end = RSTRING_END(str1); 02385 p2 = RSTRING_PTR(str2); p2end = RSTRING_END(str2); 02386 if (single_byte_optimizable(str1) && single_byte_optimizable(str2)) { 02387 while (p1 < p1end && p2 < p2end) { 02388 if (*p1 != *p2) { 02389 unsigned int c1 = TOUPPER(*p1 & 0xff); 02390 unsigned int c2 = TOUPPER(*p2 & 0xff); 02391 if (c1 != c2) 02392 return INT2FIX(c1 < c2 ? -1 : 1); 02393 } 02394 p1++; 02395 p2++; 02396 } 02397 } 02398 else { 02399 while (p1 < p1end && p2 < p2end) { 02400 int l1, c1 = rb_enc_ascget(p1, p1end, &l1, enc); 02401 int l2, c2 = rb_enc_ascget(p2, p2end, &l2, enc); 02402 02403 if (0 <= c1 && 0 <= c2) { 02404 c1 = TOUPPER(c1); 02405 c2 = TOUPPER(c2); 02406 if (c1 != c2) 02407 return INT2FIX(c1 < c2 ? -1 : 1); 02408 } 02409 else { 02410 int r; 02411 l1 = rb_enc_mbclen(p1, p1end, enc); 02412 l2 = rb_enc_mbclen(p2, p2end, enc); 02413 len = l1 < l2 ? l1 : l2; 02414 r = memcmp(p1, p2, len); 02415 if (r != 0) 02416 return INT2FIX(r < 0 ? -1 : 1); 02417 if (l1 != l2) 02418 return INT2FIX(l1 < l2 ? -1 : 1); 02419 } 02420 p1 += l1; 02421 p2 += l2; 02422 } 02423 } 02424 if (RSTRING_LEN(str1) == RSTRING_LEN(str2)) return INT2FIX(0); 02425 if (RSTRING_LEN(str1) > RSTRING_LEN(str2)) return INT2FIX(1); 02426 return INT2FIX(-1); 02427 } 02428 02429 static long 02430 rb_str_index(VALUE str, VALUE sub, long offset) 02431 { 02432 long pos; 02433 char *s, *sptr, *e; 02434 long len, slen; 02435 rb_encoding *enc; 02436 02437 enc = rb_enc_check(str, sub); 02438 if (is_broken_string(sub)) { 02439 return -1; 02440 } 02441 len = str_strlen(str, enc); 02442 slen = str_strlen(sub, enc); 02443 if (offset < 0) { 02444 offset += len; 02445 if (offset < 0) return -1; 02446 } 02447 if (len - offset < slen) return -1; 02448 s = RSTRING_PTR(str); 02449 e = s + RSTRING_LEN(str); 02450 if (offset) { 02451 offset = str_offset(s, RSTRING_END(str), offset, enc, single_byte_optimizable(str)); 02452 s += offset; 02453 } 02454 if (slen == 0) return offset; 02455 /* need proceed one character at a time */ 02456 sptr = RSTRING_PTR(sub); 02457 slen = RSTRING_LEN(sub); 02458 len = RSTRING_LEN(str) - offset; 02459 for (;;) { 02460 char *t; 02461 pos = rb_memsearch(sptr, slen, s, len, enc); 02462 if (pos < 0) return pos; 02463 t = rb_enc_right_char_head(s, s+pos, e, enc); 02464 if (t == s + pos) break; 02465 if ((len -= t - s) <= 0) return -1; 02466 offset += t - s; 02467 s = t; 02468 } 02469 return pos + offset; 02470 } 02471 02472 02473 /* 02474 * call-seq: 02475 * str.index(substring [, offset]) -> fixnum or nil 02476 * str.index(regexp [, offset]) -> fixnum or nil 02477 * 02478 * Returns the index of the first occurrence of the given <i>substring</i> or 02479 * pattern (<i>regexp</i>) in <i>str</i>. Returns <code>nil</code> if not 02480 * found. If the second parameter is present, it specifies the position in the 02481 * string to begin the search. 02482 * 02483 * "hello".index('e') #=> 1 02484 * "hello".index('lo') #=> 3 02485 * "hello".index('a') #=> nil 02486 * "hello".index(?e) #=> 1 02487 * "hello".index(/[aeiou]/, -3) #=> 4 02488 */ 02489 02490 static VALUE 02491 rb_str_index_m(int argc, VALUE *argv, VALUE str) 02492 { 02493 VALUE sub; 02494 VALUE initpos; 02495 long pos; 02496 02497 if (rb_scan_args(argc, argv, "11", &sub, &initpos) == 2) { 02498 pos = NUM2LONG(initpos); 02499 } 02500 else { 02501 pos = 0; 02502 } 02503 if (pos < 0) { 02504 pos += str_strlen(str, STR_ENC_GET(str)); 02505 if (pos < 0) { 02506 if (TYPE(sub) == T_REGEXP) { 02507 rb_backref_set(Qnil); 02508 } 02509 return Qnil; 02510 } 02511 } 02512 02513 switch (TYPE(sub)) { 02514 case T_REGEXP: 02515 if (pos > str_strlen(str, STR_ENC_GET(str))) 02516 return Qnil; 02517 pos = str_offset(RSTRING_PTR(str), RSTRING_END(str), pos, 02518 rb_enc_check(str, sub), single_byte_optimizable(str)); 02519 02520 pos = rb_reg_search(sub, str, pos, 0); 02521 pos = rb_str_sublen(str, pos); 02522 break; 02523 02524 default: { 02525 VALUE tmp; 02526 02527 tmp = rb_check_string_type(sub); 02528 if (NIL_P(tmp)) { 02529 rb_raise(rb_eTypeError, "type mismatch: %s given", 02530 rb_obj_classname(sub)); 02531 } 02532 sub = tmp; 02533 } 02534 /* fall through */ 02535 case T_STRING: 02536 pos = rb_str_index(str, sub, pos); 02537 pos = rb_str_sublen(str, pos); 02538 break; 02539 } 02540 02541 if (pos == -1) return Qnil; 02542 return LONG2NUM(pos); 02543 } 02544 02545 static long 02546 rb_str_rindex(VALUE str, VALUE sub, long pos) 02547 { 02548 long len, slen; 02549 char *s, *sbeg, *e, *t; 02550 rb_encoding *enc; 02551 int singlebyte = single_byte_optimizable(str); 02552 02553 enc = rb_enc_check(str, sub); 02554 if (is_broken_string(sub)) { 02555 return -1; 02556 } 02557 len = str_strlen(str, enc); 02558 slen = str_strlen(sub, enc); 02559 /* substring longer than string */ 02560 if (len < slen) return -1; 02561 if (len - pos < slen) { 02562 pos = len - slen; 02563 } 02564 if (len == 0) { 02565 return pos; 02566 } 02567 sbeg = RSTRING_PTR(str); 02568 e = RSTRING_END(str); 02569 t = RSTRING_PTR(sub); 02570 slen = RSTRING_LEN(sub); 02571 s = str_nth(sbeg, e, pos, enc, singlebyte); 02572 while (s) { 02573 if (memcmp(s, t, slen) == 0) { 02574 return pos; 02575 } 02576 if (pos == 0) break; 02577 pos--; 02578 s = rb_enc_prev_char(sbeg, s, e, enc); 02579 } 02580 return -1; 02581 } 02582 02583 02584 /* 02585 * call-seq: 02586 * str.rindex(substring [, fixnum]) -> fixnum or nil 02587 * str.rindex(regexp [, fixnum]) -> fixnum or nil 02588 * 02589 * Returns the index of the last occurrence of the given <i>substring</i> or 02590 * pattern (<i>regexp</i>) in <i>str</i>. Returns <code>nil</code> if not 02591 * found. If the second parameter is present, it specifies the position in the 02592 * string to end the search---characters beyond this point will not be 02593 * considered. 02594 * 02595 * "hello".rindex('e') #=> 1 02596 * "hello".rindex('l') #=> 3 02597 * "hello".rindex('a') #=> nil 02598 * "hello".rindex(?e) #=> 1 02599 * "hello".rindex(/[aeiou]/, -2) #=> 1 02600 */ 02601 02602 static VALUE 02603 rb_str_rindex_m(int argc, VALUE *argv, VALUE str) 02604 { 02605 VALUE sub; 02606 VALUE vpos; 02607 rb_encoding *enc = STR_ENC_GET(str); 02608 long pos, len = str_strlen(str, enc); 02609 02610 if (rb_scan_args(argc, argv, "11", &sub, &vpos) == 2) { 02611 pos = NUM2LONG(vpos); 02612 if (pos < 0) { 02613 pos += len; 02614 if (pos < 0) { 02615 if (TYPE(sub) == T_REGEXP) { 02616 rb_backref_set(Qnil); 02617 } 02618 return Qnil; 02619 } 02620 } 02621 if (pos > len) pos = len; 02622 } 02623 else { 02624 pos = len; 02625 } 02626 02627 switch (TYPE(sub)) { 02628 case T_REGEXP: 02629 /* enc = rb_get_check(str, sub); */ 02630 pos = str_offset(RSTRING_PTR(str), RSTRING_END(str), pos, 02631 STR_ENC_GET(str), single_byte_optimizable(str)); 02632 02633 if (!RREGEXP(sub)->ptr || RREGEXP_SRC_LEN(sub)) { 02634 pos = rb_reg_search(sub, str, pos, 1); 02635 pos = rb_str_sublen(str, pos); 02636 } 02637 if (pos >= 0) return LONG2NUM(pos); 02638 break; 02639 02640 default: { 02641 VALUE tmp; 02642 02643 tmp = rb_check_string_type(sub); 02644 if (NIL_P(tmp)) { 02645 rb_raise(rb_eTypeError, "type mismatch: %s given", 02646 rb_obj_classname(sub)); 02647 } 02648 sub = tmp; 02649 } 02650 /* fall through */ 02651 case T_STRING: 02652 pos = rb_str_rindex(str, sub, pos); 02653 if (pos >= 0) return LONG2NUM(pos); 02654 break; 02655 } 02656 return Qnil; 02657 } 02658 02659 /* 02660 * call-seq: 02661 * str =~ obj -> fixnum or nil 02662 * 02663 * Match---If <i>obj</i> is a <code>Regexp</code>, use it as a pattern to match 02664 * against <i>str</i>,and returns the position the match starts, or 02665 * <code>nil</code> if there is no match. Otherwise, invokes 02666 * <i>obj.=~</i>, passing <i>str</i> as an argument. The default 02667 * <code>=~</code> in <code>Object</code> returns <code>nil</code>. 02668 * 02669 * "cat o' 9 tails" =~ /\d/ #=> 7 02670 * "cat o' 9 tails" =~ 9 #=> nil 02671 */ 02672 02673 static VALUE 02674 rb_str_match(VALUE x, VALUE y) 02675 { 02676 switch (TYPE(y)) { 02677 case T_STRING: 02678 rb_raise(rb_eTypeError, "type mismatch: String given"); 02679 02680 case T_REGEXP: 02681 return rb_reg_match(y, x); 02682 02683 default: 02684 return rb_funcall(y, rb_intern("=~"), 1, x); 02685 } 02686 } 02687 02688 02689 static VALUE get_pat(VALUE, int); 02690 02691 02692 /* 02693 * call-seq: 02694 * str.match(pattern) -> matchdata or nil 02695 * str.match(pattern, pos) -> matchdata or nil 02696 * 02697 * Converts <i>pattern</i> to a <code>Regexp</code> (if it isn't already one), 02698 * then invokes its <code>match</code> method on <i>str</i>. If the second 02699 * parameter is present, it specifies the position in the string to begin the 02700 * search. 02701 * 02702 * 'hello'.match('(.)\1') #=> #<MatchData "ll" 1:"l"> 02703 * 'hello'.match('(.)\1')[0] #=> "ll" 02704 * 'hello'.match(/(.)\1/)[0] #=> "ll" 02705 * 'hello'.match('xx') #=> nil 02706 * 02707 * If a block is given, invoke the block with MatchData if match succeed, so 02708 * that you can write 02709 * 02710 * str.match(pat) {|m| ...} 02711 * 02712 * instead of 02713 * 02714 * if m = str.match(pat) 02715 * ... 02716 * end 02717 * 02718 * The return value is a value from block execution in this case. 02719 */ 02720 02721 static VALUE 02722 rb_str_match_m(int argc, VALUE *argv, VALUE str) 02723 { 02724 VALUE re, result; 02725 if (argc < 1) 02726 rb_raise(rb_eArgError, "wrong number of arguments (%d for 1..2)", argc); 02727 re = argv[0]; 02728 argv[0] = str; 02729 result = rb_funcall2(get_pat(re, 0), rb_intern("match"), argc, argv); 02730 if (!NIL_P(result) && rb_block_given_p()) { 02731 return rb_yield(result); 02732 } 02733 return result; 02734 } 02735 02736 enum neighbor_char { 02737 NEIGHBOR_NOT_CHAR, 02738 NEIGHBOR_FOUND, 02739 NEIGHBOR_WRAPPED 02740 }; 02741 02742 static enum neighbor_char 02743 enc_succ_char(char *p, long len, rb_encoding *enc) 02744 { 02745 long i; 02746 int l; 02747 while (1) { 02748 for (i = len-1; 0 <= i && (unsigned char)p[i] == 0xff; i--) 02749 p[i] = '\0'; 02750 if (i < 0) 02751 return NEIGHBOR_WRAPPED; 02752 ++((unsigned char*)p)[i]; 02753 l = rb_enc_precise_mbclen(p, p+len, enc); 02754 if (MBCLEN_CHARFOUND_P(l)) { 02755 l = MBCLEN_CHARFOUND_LEN(l); 02756 if (l == len) { 02757 return NEIGHBOR_FOUND; 02758 } 02759 else { 02760 memset(p+l, 0xff, len-l); 02761 } 02762 } 02763 if (MBCLEN_INVALID_P(l) && i < len-1) { 02764 long len2; 02765 int l2; 02766 for (len2 = len-1; 0 < len2; len2--) { 02767 l2 = rb_enc_precise_mbclen(p, p+len2, enc); 02768 if (!MBCLEN_INVALID_P(l2)) 02769 break; 02770 } 02771 memset(p+len2+1, 0xff, len-(len2+1)); 02772 } 02773 } 02774 } 02775 02776 static enum neighbor_char 02777 enc_pred_char(char *p, long len, rb_encoding *enc) 02778 { 02779 long i; 02780 int l; 02781 while (1) { 02782 for (i = len-1; 0 <= i && (unsigned char)p[i] == 0; i--) 02783 p[i] = '\xff'; 02784 if (i < 0) 02785 return NEIGHBOR_WRAPPED; 02786 --((unsigned char*)p)[i]; 02787 l = rb_enc_precise_mbclen(p, p+len, enc); 02788 if (MBCLEN_CHARFOUND_P(l)) { 02789 l = MBCLEN_CHARFOUND_LEN(l); 02790 if (l == len) { 02791 return NEIGHBOR_FOUND; 02792 } 02793 else { 02794 memset(p+l, 0, len-l); 02795 } 02796 } 02797 if (MBCLEN_INVALID_P(l) && i < len-1) { 02798 long len2; 02799 int l2; 02800 for (len2 = len-1; 0 < len2; len2--) { 02801 l2 = rb_enc_precise_mbclen(p, p+len2, enc); 02802 if (!MBCLEN_INVALID_P(l2)) 02803 break; 02804 } 02805 memset(p+len2+1, 0, len-(len2+1)); 02806 } 02807 } 02808 } 02809 02810 /* 02811 overwrite +p+ by succeeding letter in +enc+ and returns 02812 NEIGHBOR_FOUND or NEIGHBOR_WRAPPED. 02813 When NEIGHBOR_WRAPPED, carried-out letter is stored into carry. 02814 assuming each ranges are successive, and mbclen 02815 never change in each ranges. 02816 NEIGHBOR_NOT_CHAR is returned if invalid character or the range has only one 02817 character. 02818 */ 02819 static enum neighbor_char 02820 enc_succ_alnum_char(char *p, long len, rb_encoding *enc, char *carry) 02821 { 02822 enum neighbor_char ret; 02823 unsigned int c; 02824 int ctype; 02825 int range; 02826 char save[ONIGENC_CODE_TO_MBC_MAXLEN]; 02827 02828 c = rb_enc_mbc_to_codepoint(p, p+len, enc); 02829 if (rb_enc_isctype(c, ONIGENC_CTYPE_DIGIT, enc)) 02830 ctype = ONIGENC_CTYPE_DIGIT; 02831 else if (rb_enc_isctype(c, ONIGENC_CTYPE_ALPHA, enc)) 02832 ctype = ONIGENC_CTYPE_ALPHA; 02833 else 02834 return NEIGHBOR_NOT_CHAR; 02835 02836 MEMCPY(save, p, char, len); 02837 ret = enc_succ_char(p, len, enc); 02838 if (ret == NEIGHBOR_FOUND) { 02839 c = rb_enc_mbc_to_codepoint(p, p+len, enc); 02840 if (rb_enc_isctype(c, ctype, enc)) 02841 return NEIGHBOR_FOUND; 02842 } 02843 MEMCPY(p, save, char, len); 02844 range = 1; 02845 while (1) { 02846 MEMCPY(save, p, char, len); 02847 ret = enc_pred_char(p, len, enc); 02848 if (ret == NEIGHBOR_FOUND) { 02849 c = rb_enc_mbc_to_codepoint(p, p+len, enc); 02850 if (!rb_enc_isctype(c, ctype, enc)) { 02851 MEMCPY(p, save, char, len); 02852 break; 02853 } 02854 } 02855 else { 02856 MEMCPY(p, save, char, len); 02857 break; 02858 } 02859 range++; 02860 } 02861 if (range == 1) { 02862 return NEIGHBOR_NOT_CHAR; 02863 } 02864 02865 if (ctype != ONIGENC_CTYPE_DIGIT) { 02866 MEMCPY(carry, p, char, len); 02867 return NEIGHBOR_WRAPPED; 02868 } 02869 02870 MEMCPY(carry, p, char, len); 02871 enc_succ_char(carry, len, enc); 02872 return NEIGHBOR_WRAPPED; 02873 } 02874 02875 02876 /* 02877 * call-seq: 02878 * str.succ -> new_str 02879 * str.next -> new_str 02880 * 02881 * Returns the successor to <i>str</i>. The successor is calculated by 02882 * incrementing characters starting from the rightmost alphanumeric (or 02883 * the rightmost character if there are no alphanumerics) in the 02884 * string. Incrementing a digit always results in another digit, and 02885 * incrementing a letter results in another letter of the same case. 02886 * Incrementing nonalphanumerics uses the underlying character set's 02887 * collating sequence. 02888 * 02889 * If the increment generates a ``carry,'' the character to the left of 02890 * it is incremented. This process repeats until there is no carry, 02891 * adding an additional character if necessary. 02892 * 02893 * "abcd".succ #=> "abce" 02894 * "THX1138".succ #=> "THX1139" 02895 * "<<koala>>".succ #=> "<<koalb>>" 02896 * "1999zzz".succ #=> "2000aaa" 02897 * "ZZZ9999".succ #=> "AAAA0000" 02898 * "***".succ #=> "**+" 02899 */ 02900 02901 VALUE 02902 rb_str_succ(VALUE orig) 02903 { 02904 rb_encoding *enc; 02905 VALUE str; 02906 char *sbeg, *s, *e, *last_alnum = 0; 02907 int c = -1; 02908 long l; 02909 char carry[ONIGENC_CODE_TO_MBC_MAXLEN] = "\1"; 02910 long carry_pos = 0, carry_len = 1; 02911 enum neighbor_char neighbor = NEIGHBOR_FOUND; 02912 02913 str = rb_str_new5(orig, RSTRING_PTR(orig), RSTRING_LEN(orig)); 02914 rb_enc_cr_str_copy_for_substr(str, orig); 02915 OBJ_INFECT(str, orig); 02916 if (RSTRING_LEN(str) == 0) return str; 02917 02918 enc = STR_ENC_GET(orig); 02919 sbeg = RSTRING_PTR(str); 02920 s = e = sbeg + RSTRING_LEN(str); 02921 02922 while ((s = rb_enc_prev_char(sbeg, s, e, enc)) != 0) { 02923 if (neighbor == NEIGHBOR_NOT_CHAR && last_alnum) { 02924 if (ISALPHA(*last_alnum) ? ISDIGIT(*s) : 02925 ISDIGIT(*last_alnum) ? ISALPHA(*s) : 0) { 02926 s = last_alnum; 02927 break; 02928 } 02929 } 02930 if ((l = rb_enc_precise_mbclen(s, e, enc)) <= 0) continue; 02931 neighbor = enc_succ_alnum_char(s, l, enc, carry); 02932 switch (neighbor) { 02933 case NEIGHBOR_NOT_CHAR: 02934 continue; 02935 case NEIGHBOR_FOUND: 02936 return str; 02937 case NEIGHBOR_WRAPPED: 02938 last_alnum = s; 02939 break; 02940 } 02941 c = 1; 02942 carry_pos = s - sbeg; 02943 carry_len = l; 02944 } 02945 if (c == -1) { /* str contains no alnum */ 02946 s = e; 02947 while ((s = rb_enc_prev_char(sbeg, s, e, enc)) != 0) { 02948 enum neighbor_char neighbor; 02949 if ((l = rb_enc_precise_mbclen(s, e, enc)) <= 0) continue; 02950 neighbor = enc_succ_char(s, l, enc); 02951 if (neighbor == NEIGHBOR_FOUND) 02952 return str; 02953 if (rb_enc_precise_mbclen(s, s+l, enc) != l) { 02954 /* wrapped to \0...\0. search next valid char. */ 02955 enc_succ_char(s, l, enc); 02956 } 02957 if (!rb_enc_asciicompat(enc)) { 02958 MEMCPY(carry, s, char, l); 02959 carry_len = l; 02960 } 02961 carry_pos = s - sbeg; 02962 } 02963 } 02964 RESIZE_CAPA(str, RSTRING_LEN(str) + carry_len); 02965 s = RSTRING_PTR(str) + carry_pos; 02966 memmove(s + carry_len, s, RSTRING_LEN(str) - carry_pos); 02967 memmove(s, carry, carry_len); 02968 STR_SET_LEN(str, RSTRING_LEN(str) + carry_len); 02969 RSTRING_PTR(str)[RSTRING_LEN(str)] = '\0'; 02970 rb_enc_str_coderange(str); 02971 return str; 02972 } 02973 02974 02975 /* 02976 * call-seq: 02977 * str.succ! -> str 02978 * str.next! -> str 02979 * 02980 * Equivalent to <code>String#succ</code>, but modifies the receiver in 02981 * place. 02982 */ 02983 02984 static VALUE 02985 rb_str_succ_bang(VALUE str) 02986 { 02987 rb_str_shared_replace(str, rb_str_succ(str)); 02988 02989 return str; 02990 } 02991 02992 02993 /* 02994 * call-seq: 02995 * str.upto(other_str, exclusive=false) {|s| block } -> str 02996 * str.upto(other_str, exclusive=false) -> an_enumerator 02997 * 02998 * Iterates through successive values, starting at <i>str</i> and 02999 * ending at <i>other_str</i> inclusive, passing each value in turn to 03000 * the block. The <code>String#succ</code> method is used to generate 03001 * each value. If optional second argument exclusive is omitted or is false, 03002 * the last value will be included; otherwise it will be excluded. 03003 * 03004 * If no block is given, an enumerator is returned instead. 03005 * 03006 * "a8".upto("b6") {|s| print s, ' ' } 03007 * for s in "a8".."b6" 03008 * print s, ' ' 03009 * end 03010 * 03011 * <em>produces:</em> 03012 * 03013 * a8 a9 b0 b1 b2 b3 b4 b5 b6 03014 * a8 a9 b0 b1 b2 b3 b4 b5 b6 03015 * 03016 * If <i>str</i> and <i>other_str</i> contains only ascii numeric characters, 03017 * both are recognized as decimal numbers. In addition, the width of 03018 * string (e.g. leading zeros) is handled appropriately. 03019 * 03020 * "9".upto("11").to_a #=> ["9", "10", "11"] 03021 * "25".upto("5").to_a #=> [] 03022 * "07".upto("11").to_a #=> ["07", "08", "09", "10", "11"] 03023 */ 03024 03025 static VALUE 03026 rb_str_upto(int argc, VALUE *argv, VALUE beg) 03027 { 03028 VALUE end, exclusive; 03029 VALUE current, after_end; 03030 ID succ; 03031 int n, excl, ascii; 03032 rb_encoding *enc; 03033 03034 rb_scan_args(argc, argv, "11", &end, &exclusive); 03035 RETURN_ENUMERATOR(beg, argc, argv); 03036 excl = RTEST(exclusive); 03037 CONST_ID(succ, "succ"); 03038 StringValue(end); 03039 enc = rb_enc_check(beg, end); 03040 ascii = (is_ascii_string(beg) && is_ascii_string(end)); 03041 /* single character */ 03042 if (RSTRING_LEN(beg) == 1 && RSTRING_LEN(end) == 1 && ascii) { 03043 char c = RSTRING_PTR(beg)[0]; 03044 char e = RSTRING_PTR(end)[0]; 03045 03046 if (c > e || (excl && c == e)) return beg; 03047 for (;;) { 03048 rb_yield(rb_enc_str_new(&c, 1, enc)); 03049 if (!excl && c == e) break; 03050 c++; 03051 if (excl && c == e) break; 03052 } 03053 return beg; 03054 } 03055 /* both edges are all digits */ 03056 if (ascii && ISDIGIT(RSTRING_PTR(beg)[0]) && ISDIGIT(RSTRING_PTR(end)[0])) { 03057 char *s, *send; 03058 VALUE b, e; 03059 int width; 03060 03061 s = RSTRING_PTR(beg); send = RSTRING_END(beg); 03062 width = rb_long2int(send - s); 03063 while (s < send) { 03064 if (!ISDIGIT(*s)) goto no_digits; 03065 s++; 03066 } 03067 s = RSTRING_PTR(end); send = RSTRING_END(end); 03068 while (s < send) { 03069 if (!ISDIGIT(*s)) goto no_digits; 03070 s++; 03071 } 03072 b = rb_str_to_inum(beg, 10, FALSE); 03073 e = rb_str_to_inum(end, 10, FALSE); 03074 if (FIXNUM_P(b) && FIXNUM_P(e)) { 03075 long bi = FIX2LONG(b); 03076 long ei = FIX2LONG(e); 03077 rb_encoding *usascii = rb_usascii_encoding(); 03078 03079 while (bi <= ei) { 03080 if (excl && bi == ei) break; 03081 rb_yield(rb_enc_sprintf(usascii, "%.*ld", width, bi)); 03082 bi++; 03083 } 03084 } 03085 else { 03086 ID op = excl ? '<' : rb_intern("<="); 03087 VALUE args[2], fmt = rb_obj_freeze(rb_usascii_str_new_cstr("%.*d")); 03088 03089 args[0] = INT2FIX(width); 03090 while (rb_funcall(b, op, 1, e)) { 03091 args[1] = b; 03092 rb_yield(rb_str_format(numberof(args), args, fmt)); 03093 b = rb_funcall(b, succ, 0, 0); 03094 } 03095 } 03096 return beg; 03097 } 03098 /* normal case */ 03099 no_digits: 03100 n = rb_str_cmp(beg, end); 03101 if (n > 0 || (excl && n == 0)) return beg; 03102 03103 after_end = rb_funcall(end, succ, 0, 0); 03104 current = rb_str_dup(beg); 03105 while (!rb_str_equal(current, after_end)) { 03106 VALUE next = Qnil; 03107 if (excl || !rb_str_equal(current, end)) 03108 next = rb_funcall(current, succ, 0, 0); 03109 rb_yield(current); 03110 if (NIL_P(next)) break; 03111 current = next; 03112 StringValue(current); 03113 if (excl && rb_str_equal(current, end)) break; 03114 if (RSTRING_LEN(current) > RSTRING_LEN(end) || RSTRING_LEN(current) == 0) 03115 break; 03116 } 03117 03118 return beg; 03119 } 03120 03121 static VALUE 03122 rb_str_subpat(VALUE str, VALUE re, VALUE backref) 03123 { 03124 if (rb_reg_search(re, str, 0, 0) >= 0) { 03125 VALUE match = rb_backref_get(); 03126 int nth = rb_reg_backref_number(match, backref); 03127 return rb_reg_nth_match(nth, match); 03128 } 03129 return Qnil; 03130 } 03131 03132 static VALUE 03133 rb_str_aref(VALUE str, VALUE indx) 03134 { 03135 long idx; 03136 03137 switch (TYPE(indx)) { 03138 case T_FIXNUM: 03139 idx = FIX2LONG(indx); 03140 03141 num_index: 03142 str = rb_str_substr(str, idx, 1); 03143 if (!NIL_P(str) && RSTRING_LEN(str) == 0) return Qnil; 03144 return str; 03145 03146 case T_REGEXP: 03147 return rb_str_subpat(str, indx, INT2FIX(0)); 03148 03149 case T_STRING: 03150 if (rb_str_index(str, indx, 0) != -1) 03151 return rb_str_dup(indx); 03152 return Qnil; 03153 03154 default: 03155 /* check if indx is Range */ 03156 { 03157 long beg, len; 03158 VALUE tmp; 03159 03160 len = str_strlen(str, STR_ENC_GET(str)); 03161 switch (rb_range_beg_len(indx, &beg, &len, len, 0)) { 03162 case Qfalse: 03163 break; 03164 case Qnil: 03165 return Qnil; 03166 default: 03167 tmp = rb_str_substr(str, beg, len); 03168 return tmp; 03169 } 03170 } 03171 idx = NUM2LONG(indx); 03172 goto num_index; 03173 } 03174 return Qnil; /* not reached */ 03175 } 03176 03177 03178 /* 03179 * call-seq: 03180 * str[fixnum] -> new_str or nil 03181 * str[fixnum, fixnum] -> new_str or nil 03182 * str[range] -> new_str or nil 03183 * str[regexp] -> new_str or nil 03184 * str[regexp, fixnum] -> new_str or nil 03185 * str[other_str] -> new_str or nil 03186 * str.slice(fixnum) -> new_str or nil 03187 * str.slice(fixnum, fixnum) -> new_str or nil 03188 * str.slice(range) -> new_str or nil 03189 * str.slice(regexp) -> new_str or nil 03190 * str.slice(regexp, fixnum) -> new_str or nil 03191 * str.slice(regexp, capname) -> new_str or nil 03192 * str.slice(other_str) -> new_str or nil 03193 * 03194 * Element Reference---If passed a single <code>Fixnum</code>, returns a 03195 * substring of one character at that position. If passed two <code>Fixnum</code> 03196 * objects, returns a substring starting at the offset given by the first, and 03197 * with a length given by the second. If passed a range, its beginning and end 03198 * are interpreted as offsets delimiting the substring to be returned. In all 03199 * three cases, if an offset is negative, it is counted from the end of <i>str</i>. 03200 * Returns <code>nil</code> if the initial offset falls outside the string or 03201 * the length is negative. 03202 * 03203 * If a <code>Regexp</code> is supplied, the matching portion of <i>str</i> is 03204 * returned. If a numeric or name parameter follows the regular expression, that 03205 * component of the <code>MatchData</code> is returned instead. If a 03206 * <code>String</code> is given, that string is returned if it occurs in 03207 * <i>str</i>. In both cases, <code>nil</code> is returned if there is no 03208 * match. 03209 * 03210 * a = "hello there" 03211 * a[1] #=> "e" 03212 * a[2, 3] #=> "llo" 03213 * a[2..3] #=> "ll" 03214 * a[-3, 2] #=> "er" 03215 * a[7..-2] #=> "her" 03216 * a[-4..-2] #=> "her" 03217 * a[-2..-4] #=> "" 03218 * a[12..-1] #=> nil 03219 * a[/[aeiou](.)\1/] #=> "ell" 03220 * a[/[aeiou](.)\1/, 0] #=> "ell" 03221 * a[/[aeiou](.)\1/, 1] #=> "l" 03222 * a[/[aeiou](.)\1/, 2] #=> nil 03223 * a["lo"] #=> "lo" 03224 * a["bye"] #=> nil 03225 */ 03226 03227 static VALUE 03228 rb_str_aref_m(int argc, VALUE *argv, VALUE str) 03229 { 03230 if (argc == 2) { 03231 if (TYPE(argv[0]) == T_REGEXP) { 03232 return rb_str_subpat(str, argv[0], argv[1]); 03233 } 03234 return rb_str_substr(str, NUM2LONG(argv[0]), NUM2LONG(argv[1])); 03235 } 03236 if (argc != 1) { 03237 rb_raise(rb_eArgError, "wrong number of arguments (%d for 1..2)", argc); 03238 } 03239 return rb_str_aref(str, argv[0]); 03240 } 03241 03242 VALUE 03243 rb_str_drop_bytes(VALUE str, long len) 03244 { 03245 char *ptr = RSTRING_PTR(str); 03246 long olen = RSTRING_LEN(str), nlen; 03247 03248 str_modifiable(str); 03249 if (len > olen) len = olen; 03250 nlen = olen - len; 03251 if (nlen <= RSTRING_EMBED_LEN_MAX) { 03252 char *oldptr = ptr; 03253 int fl = (int)(RBASIC(str)->flags & (STR_NOEMBED|ELTS_SHARED)); 03254 STR_SET_EMBED(str); 03255 STR_SET_EMBED_LEN(str, nlen); 03256 ptr = RSTRING(str)->as.ary; 03257 memmove(ptr, oldptr + len, nlen); 03258 if (fl == STR_NOEMBED) xfree(oldptr); 03259 } 03260 else { 03261 if (!STR_SHARED_P(str)) rb_str_new4(str); 03262 ptr = RSTRING(str)->as.heap.ptr += len; 03263 RSTRING(str)->as.heap.len = nlen; 03264 } 03265 ptr[nlen] = 0; 03266 ENC_CODERANGE_CLEAR(str); 03267 return str; 03268 } 03269 03270 static void 03271 rb_str_splice_0(VALUE str, long beg, long len, VALUE val) 03272 { 03273 if (beg == 0 && RSTRING_LEN(val) == 0) { 03274 rb_str_drop_bytes(str, len); 03275 OBJ_INFECT(str, val); 03276 return; 03277 } 03278 03279 rb_str_modify(str); 03280 if (len < RSTRING_LEN(val)) { 03281 /* expand string */ 03282 RESIZE_CAPA(str, RSTRING_LEN(str) + RSTRING_LEN(val) - len + 1); 03283 } 03284 03285 if (RSTRING_LEN(val) != len) { 03286 memmove(RSTRING_PTR(str) + beg + RSTRING_LEN(val), 03287 RSTRING_PTR(str) + beg + len, 03288 RSTRING_LEN(str) - (beg + len)); 03289 } 03290 if (RSTRING_LEN(val) < beg && len < 0) { 03291 MEMZERO(RSTRING_PTR(str) + RSTRING_LEN(str), char, -len); 03292 } 03293 if (RSTRING_LEN(val) > 0) { 03294 memmove(RSTRING_PTR(str)+beg, RSTRING_PTR(val), RSTRING_LEN(val)); 03295 } 03296 STR_SET_LEN(str, RSTRING_LEN(str) + RSTRING_LEN(val) - len); 03297 if (RSTRING_PTR(str)) { 03298 RSTRING_PTR(str)[RSTRING_LEN(str)] = '\0'; 03299 } 03300 OBJ_INFECT(str, val); 03301 } 03302 03303 static void 03304 rb_str_splice(VALUE str, long beg, long len, VALUE val) 03305 { 03306 long slen; 03307 char *p, *e; 03308 rb_encoding *enc; 03309 int singlebyte = single_byte_optimizable(str); 03310 int cr; 03311 03312 if (len < 0) rb_raise(rb_eIndexError, "negative length %ld", len); 03313 03314 StringValue(val); 03315 enc = rb_enc_check(str, val); 03316 slen = str_strlen(str, enc); 03317 03318 if (slen < beg) { 03319 out_of_range: 03320 rb_raise(rb_eIndexError, "index %ld out of string", beg); 03321 } 03322 if (beg < 0) { 03323 if (-beg > slen) { 03324 goto out_of_range; 03325 } 03326 beg += slen; 03327 } 03328 if (slen < len || slen < beg + len) { 03329 len = slen - beg; 03330 } 03331 str_modify_keep_cr(str); 03332 p = str_nth(RSTRING_PTR(str), RSTRING_END(str), beg, enc, singlebyte); 03333 if (!p) p = RSTRING_END(str); 03334 e = str_nth(p, RSTRING_END(str), len, enc, singlebyte); 03335 if (!e) e = RSTRING_END(str); 03336 /* error check */ 03337 beg = p - RSTRING_PTR(str); /* physical position */ 03338 len = e - p; /* physical length */ 03339 rb_str_splice_0(str, beg, len, val); 03340 rb_enc_associate(str, enc); 03341 cr = ENC_CODERANGE_AND(ENC_CODERANGE(str), ENC_CODERANGE(val)); 03342 if (cr != ENC_CODERANGE_BROKEN) 03343 ENC_CODERANGE_SET(str, cr); 03344 } 03345 03346 void 03347 rb_str_update(VALUE str, long beg, long len, VALUE val) 03348 { 03349 rb_str_splice(str, beg, len, val); 03350 } 03351 03352 static void 03353 rb_str_subpat_set(VALUE str, VALUE re, VALUE backref, VALUE val) 03354 { 03355 int nth; 03356 VALUE match; 03357 long start, end, len; 03358 rb_encoding *enc; 03359 struct re_registers *regs; 03360 03361 if (rb_reg_search(re, str, 0, 0) < 0) { 03362 rb_raise(rb_eIndexError, "regexp not matched"); 03363 } 03364 match = rb_backref_get(); 03365 nth = rb_reg_backref_number(match, backref); 03366 regs = RMATCH_REGS(match); 03367 if (nth >= regs->num_regs) { 03368 out_of_range: 03369 rb_raise(rb_eIndexError, "index %d out of regexp", nth); 03370 } 03371 if (nth < 0) { 03372 if (-nth >= regs->num_regs) { 03373 goto out_of_range; 03374 } 03375 nth += regs->num_regs; 03376 } 03377 03378 start = BEG(nth); 03379 if (start == -1) { 03380 rb_raise(rb_eIndexError, "regexp group %d not matched", nth); 03381 } 03382 end = END(nth); 03383 len = end - start; 03384 StringValue(val); 03385 enc = rb_enc_check(str, val); 03386 rb_str_splice_0(str, start, len, val); 03387 rb_enc_associate(str, enc); 03388 } 03389 03390 static VALUE 03391 rb_str_aset(VALUE str, VALUE indx, VALUE val) 03392 { 03393 long idx, beg; 03394 03395 switch (TYPE(indx)) { 03396 case T_FIXNUM: 03397 idx = FIX2LONG(indx); 03398 num_index: 03399 rb_str_splice(str, idx, 1, val); 03400 return val; 03401 03402 case T_REGEXP: 03403 rb_str_subpat_set(str, indx, INT2FIX(0), val); 03404 return val; 03405 03406 case T_STRING: 03407 beg = rb_str_index(str, indx, 0); 03408 if (beg < 0) { 03409 rb_raise(rb_eIndexError, "string not matched"); 03410 } 03411 beg = rb_str_sublen(str, beg); 03412 rb_str_splice(str, beg, str_strlen(indx, 0), val); 03413 return val; 03414 03415 default: 03416 /* check if indx is Range */ 03417 { 03418 long beg, len; 03419 if (rb_range_beg_len(indx, &beg, &len, str_strlen(str, 0), 2)) { 03420 rb_str_splice(str, beg, len, val); 03421 return val; 03422 } 03423 } 03424 idx = NUM2LONG(indx); 03425 goto num_index; 03426 } 03427 } 03428 03429 /* 03430 * call-seq: 03431 * str[fixnum] = new_str 03432 * str[fixnum, fixnum] = new_str 03433 * str[range] = aString 03434 * str[regexp] = new_str 03435 * str[regexp, fixnum] = new_str 03436 * str[regexp, name] = new_str 03437 * str[other_str] = new_str 03438 * 03439 * Element Assignment---Replaces some or all of the content of <i>str</i>. The 03440 * portion of the string affected is determined using the same criteria as 03441 * <code>String#[]</code>. If the replacement string is not the same length as 03442 * the text it is replacing, the string will be adjusted accordingly. If the 03443 * regular expression or string is used as the index doesn't match a position 03444 * in the string, <code>IndexError</code> is raised. If the regular expression 03445 * form is used, the optional second <code>Fixnum</code> allows you to specify 03446 * which portion of the match to replace (effectively using the 03447 * <code>MatchData</code> indexing rules. The forms that take a 03448 * <code>Fixnum</code> will raise an <code>IndexError</code> if the value is 03449 * out of range; the <code>Range</code> form will raise a 03450 * <code>RangeError</code>, and the <code>Regexp</code> and <code>String</code> 03451 * forms will silently ignore the assignment. 03452 */ 03453 03454 static VALUE 03455 rb_str_aset_m(int argc, VALUE *argv, VALUE str) 03456 { 03457 if (argc == 3) { 03458 if (TYPE(argv[0]) == T_REGEXP) { 03459 rb_str_subpat_set(str, argv[0], argv[1], argv[2]); 03460 } 03461 else { 03462 rb_str_splice(str, NUM2LONG(argv[0]), NUM2LONG(argv[1]), argv[2]); 03463 } 03464 return argv[2]; 03465 } 03466 if (argc != 2) { 03467 rb_raise(rb_eArgError, "wrong number of arguments (%d for 2..3)", argc); 03468 } 03469 return rb_str_aset(str, argv[0], argv[1]); 03470 } 03471 03472 /* 03473 * call-seq: 03474 * str.insert(index, other_str) -> str 03475 * 03476 * Inserts <i>other_str</i> before the character at the given 03477 * <i>index</i>, modifying <i>str</i>. Negative indices count from the 03478 * end of the string, and insert <em>after</em> the given character. 03479 * The intent is insert <i>aString</i> so that it starts at the given 03480 * <i>index</i>. 03481 * 03482 * "abcd".insert(0, 'X') #=> "Xabcd" 03483 * "abcd".insert(3, 'X') #=> "abcXd" 03484 * "abcd".insert(4, 'X') #=> "abcdX" 03485 * "abcd".insert(-3, 'X') #=> "abXcd" 03486 * "abcd".insert(-1, 'X') #=> "abcdX" 03487 */ 03488 03489 static VALUE 03490 rb_str_insert(VALUE str, VALUE idx, VALUE str2) 03491 { 03492 long pos = NUM2LONG(idx); 03493 03494 if (pos == -1) { 03495 return rb_str_append(str, str2); 03496 } 03497 else if (pos < 0) { 03498 pos++; 03499 } 03500 rb_str_splice(str, pos, 0, str2); 03501 return str; 03502 } 03503 03504 03505 /* 03506 * call-seq: 03507 * str.slice!(fixnum) -> fixnum or nil 03508 * str.slice!(fixnum, fixnum) -> new_str or nil 03509 * str.slice!(range) -> new_str or nil 03510 * str.slice!(regexp) -> new_str or nil 03511 * str.slice!(other_str) -> new_str or nil 03512 * 03513 * Deletes the specified portion from <i>str</i>, and returns the portion 03514 * deleted. 03515 * 03516 * string = "this is a string" 03517 * string.slice!(2) #=> "i" 03518 * string.slice!(3..6) #=> " is " 03519 * string.slice!(/s.*t/) #=> "sa st" 03520 * string.slice!("r") #=> "r" 03521 * string #=> "thing" 03522 */ 03523 03524 static VALUE 03525 rb_str_slice_bang(int argc, VALUE *argv, VALUE str) 03526 { 03527 VALUE result; 03528 VALUE buf[3]; 03529 int i; 03530 03531 if (argc < 1 || 2 < argc) { 03532 rb_raise(rb_eArgError, "wrong number of arguments (%d for 1..2)", argc); 03533 } 03534 for (i=0; i<argc; i++) { 03535 buf[i] = argv[i]; 03536 } 03537 str_modify_keep_cr(str); 03538 result = rb_str_aref_m(argc, buf, str); 03539 if (!NIL_P(result)) { 03540 buf[i] = rb_str_new(0,0); 03541 rb_str_aset_m(argc+1, buf, str); 03542 } 03543 return result; 03544 } 03545 03546 static VALUE 03547 get_pat(VALUE pat, int quote) 03548 { 03549 VALUE val; 03550 03551 switch (TYPE(pat)) { 03552 case T_REGEXP: 03553 return pat; 03554 03555 case T_STRING: 03556 break; 03557 03558 default: 03559 val = rb_check_string_type(pat); 03560 if (NIL_P(val)) { 03561 Check_Type(pat, T_REGEXP); 03562 } 03563 pat = val; 03564 } 03565 03566 if (quote) { 03567 pat = rb_reg_quote(pat); 03568 } 03569 03570 return rb_reg_regcomp(pat); 03571 } 03572 03573 03574 /* 03575 * call-seq: 03576 * str.sub!(pattern, replacement) -> str or nil 03577 * str.sub!(pattern) {|match| block } -> str or nil 03578 * 03579 * Performs the substitutions of <code>String#sub</code> in place, 03580 * returning <i>str</i>, or <code>nil</code> if no substitutions were 03581 * performed. 03582 */ 03583 03584 static VALUE 03585 rb_str_sub_bang(int argc, VALUE *argv, VALUE str) 03586 { 03587 VALUE pat, repl, hash = Qnil; 03588 int iter = 0; 03589 int tainted = 0; 03590 int untrusted = 0; 03591 long plen; 03592 03593 if (argc == 1 && rb_block_given_p()) { 03594 iter = 1; 03595 } 03596 else if (argc == 2) { 03597 repl = argv[1]; 03598 hash = rb_check_convert_type(argv[1], T_HASH, "Hash", "to_hash"); 03599 if (NIL_P(hash)) { 03600 StringValue(repl); 03601 } 03602 if (OBJ_TAINTED(repl)) tainted = 1; 03603 if (OBJ_UNTRUSTED(repl)) untrusted = 1; 03604 } 03605 else { 03606 rb_raise(rb_eArgError, "wrong number of arguments (%d for 1..2)", argc); 03607 } 03608 03609 pat = get_pat(argv[0], 1); 03610 str_modifiable(str); 03611 if (rb_reg_search(pat, str, 0, 0) >= 0) { 03612 rb_encoding *enc; 03613 int cr = ENC_CODERANGE(str); 03614 VALUE match = rb_backref_get(); 03615 struct re_registers *regs = RMATCH_REGS(match); 03616 long beg0 = BEG(0); 03617 long end0 = END(0); 03618 char *p, *rp; 03619 long len, rlen; 03620 03621 if (iter || !NIL_P(hash)) { 03622 p = RSTRING_PTR(str); len = RSTRING_LEN(str); 03623 03624 if (iter) { 03625 repl = rb_obj_as_string(rb_yield(rb_reg_nth_match(0, match))); 03626 } 03627 else { 03628 repl = rb_hash_aref(hash, rb_str_subseq(str, beg0, end0 - beg0)); 03629 repl = rb_obj_as_string(repl); 03630 } 03631 str_mod_check(str, p, len); 03632 rb_check_frozen(str); 03633 } 03634 else { 03635 repl = rb_reg_regsub(repl, str, regs, pat); 03636 } 03637 enc = rb_enc_compatible(str, repl); 03638 if (!enc) { 03639 rb_encoding *str_enc = STR_ENC_GET(str); 03640 p = RSTRING_PTR(str); len = RSTRING_LEN(str); 03641 if (coderange_scan(p, beg0, str_enc) != ENC_CODERANGE_7BIT || 03642 coderange_scan(p+end0, len-end0, str_enc) != ENC_CODERANGE_7BIT) { 03643 rb_raise(rb_eEncCompatError, "incompatible character encodings: %s and %s", 03644 rb_enc_name(str_enc), 03645 rb_enc_name(STR_ENC_GET(repl))); 03646 } 03647 enc = STR_ENC_GET(repl); 03648 } 03649 rb_str_modify(str); 03650 rb_enc_associate(str, enc); 03651 if (OBJ_TAINTED(repl)) tainted = 1; 03652 if (OBJ_UNTRUSTED(repl)) untrusted = 1; 03653 if (ENC_CODERANGE_UNKNOWN < cr && cr < ENC_CODERANGE_BROKEN) { 03654 int cr2 = ENC_CODERANGE(repl); 03655 if (cr2 == ENC_CODERANGE_BROKEN || 03656 (cr == ENC_CODERANGE_VALID && cr2 == ENC_CODERANGE_7BIT)) 03657 cr = ENC_CODERANGE_UNKNOWN; 03658 else 03659 cr = cr2; 03660 } 03661 plen = end0 - beg0; 03662 rp = RSTRING_PTR(repl); rlen = RSTRING_LEN(repl); 03663 len = RSTRING_LEN(str); 03664 if (rlen > plen) { 03665 RESIZE_CAPA(str, len + rlen - plen); 03666 } 03667 p = RSTRING_PTR(str); 03668 if (rlen != plen) { 03669 memmove(p + beg0 + rlen, p + beg0 + plen, len - beg0 - plen); 03670 } 03671 memcpy(p + beg0, rp, rlen); 03672 len += rlen - plen; 03673 STR_SET_LEN(str, len); 03674 RSTRING_PTR(str)[len] = '\0'; 03675 ENC_CODERANGE_SET(str, cr); 03676 if (tainted) OBJ_TAINT(str); 03677 if (untrusted) OBJ_UNTRUST(str); 03678 03679 return str; 03680 } 03681 return Qnil; 03682 } 03683 03684 03685 /* 03686 * call-seq: 03687 * str.sub(pattern, replacement) -> new_str 03688 * str.sub(pattern, hash) -> new_str 03689 * str.sub(pattern) {|match| block } -> new_str 03690 * 03691 * Returns a copy of <i>str</i> with the <em>first</em> occurrence of 03692 * <i>pattern</i> substituted for the second argument. The <i>pattern</i> is 03693 * typically a <code>Regexp</code>; if given as a <code>String</code>, any 03694 * regular expression metacharacters it contains will be interpreted 03695 * literally, e.g. <code>'\\\d'</code> will match a backlash followed by 'd', 03696 * instead of a digit. 03697 * 03698 * If <i>replacement</i> is a <code>String</code> it will be substituted for 03699 * the matched text. It may contain back-references to the pattern's capture 03700 * groups of the form <code>\\\d</code>, where <i>d</i> is a group number, or 03701 * <code>\\\k<n></code>, where <i>n</i> is a group name. If it is a 03702 * double-quoted string, both back-references must be preceded by an 03703 * additional backslash. However, within <i>replacement</i> the special match 03704 * variables, such as <code>&$</code>, will not refer to the current match. 03705 * 03706 * If the second argument is a <code>Hash</code>, and the matched text is one 03707 * of its keys, the corresponding value is the replacement string. 03708 * 03709 * In the block form, the current match string is passed in as a parameter, 03710 * and variables such as <code>$1</code>, <code>$2</code>, <code>$`</code>, 03711 * <code>$&</code>, and <code>$'</code> will be set appropriately. The value 03712 * returned by the block will be substituted for the match on each call. 03713 * 03714 * The result inherits any tainting in the original string or any supplied 03715 * replacement string. 03716 * 03717 * "hello".sub(/[aeiou]/, '*') #=> "h*llo" 03718 * "hello".sub(/([aeiou])/, '<\1>') #=> "h<e>llo" 03719 * "hello".sub(/./) {|s| s.ord.to_s + ' ' } #=> "104 ello" 03720 * "hello".sub(/(?<foo>[aeiou])/, '*\k<foo>*') #=> "h*e*llo" 03721 * 'Is SHELL your preferred shell?'.sub(/[[:upper:]]{2,}/, ENV) 03722 * #=> "Is /bin/bash your preferred shell?" 03723 */ 03724 03725 static VALUE 03726 rb_str_sub(int argc, VALUE *argv, VALUE str) 03727 { 03728 str = rb_str_dup(str); 03729 rb_str_sub_bang(argc, argv, str); 03730 return str; 03731 } 03732 03733 static VALUE 03734 str_gsub(int argc, VALUE *argv, VALUE str, int bang) 03735 { 03736 VALUE pat, val, repl, match, dest, hash = Qnil; 03737 struct re_registers *regs; 03738 long beg, n; 03739 long beg0, end0; 03740 long offset, blen, slen, len, last; 03741 int iter = 0; 03742 char *sp, *cp; 03743 int tainted = 0; 03744 rb_encoding *str_enc; 03745 03746 switch (argc) { 03747 case 1: 03748 RETURN_ENUMERATOR(str, argc, argv); 03749 iter = 1; 03750 break; 03751 case 2: 03752 repl = argv[1]; 03753 hash = rb_check_convert_type(argv[1], T_HASH, "Hash", "to_hash"); 03754 if (NIL_P(hash)) { 03755 StringValue(repl); 03756 } 03757 if (OBJ_TAINTED(repl)) tainted = 1; 03758 break; 03759 default: 03760 rb_raise(rb_eArgError, "wrong number of arguments (%d for 1..2)", argc); 03761 } 03762 03763 pat = get_pat(argv[0], 1); 03764 beg = rb_reg_search(pat, str, 0, 0); 03765 if (beg < 0) { 03766 if (bang) return Qnil; /* no match, no substitution */ 03767 return rb_str_dup(str); 03768 } 03769 03770 offset = 0; 03771 n = 0; 03772 blen = RSTRING_LEN(str) + 30; /* len + margin */ 03773 dest = rb_str_buf_new(blen); 03774 sp = RSTRING_PTR(str); 03775 slen = RSTRING_LEN(str); 03776 cp = sp; 03777 str_enc = STR_ENC_GET(str); 03778 rb_enc_associate(dest, str_enc); 03779 ENC_CODERANGE_SET(dest, rb_enc_asciicompat(str_enc) ? ENC_CODERANGE_7BIT : ENC_CODERANGE_VALID); 03780 03781 do { 03782 n++; 03783 match = rb_backref_get(); 03784 regs = RMATCH_REGS(match); 03785 beg0 = BEG(0); 03786 end0 = END(0); 03787 if (iter || !NIL_P(hash)) { 03788 if (iter) { 03789 val = rb_obj_as_string(rb_yield(rb_reg_nth_match(0, match))); 03790 } 03791 else { 03792 val = rb_hash_aref(hash, rb_str_subseq(str, BEG(0), END(0) - BEG(0))); 03793 val = rb_obj_as_string(val); 03794 } 03795 str_mod_check(str, sp, slen); 03796 if (val == dest) { /* paranoid check [ruby-dev:24827] */ 03797 rb_raise(rb_eRuntimeError, "block should not cheat"); 03798 } 03799 } 03800 else { 03801 val = rb_reg_regsub(repl, str, regs, pat); 03802 } 03803 03804 if (OBJ_TAINTED(val)) tainted = 1; 03805 03806 len = beg - offset; /* copy pre-match substr */ 03807 if (len) { 03808 rb_enc_str_buf_cat(dest, cp, len, str_enc); 03809 } 03810 03811 rb_str_buf_append(dest, val); 03812 03813 last = offset; 03814 offset = end0; 03815 if (beg0 == end0) { 03816 /* 03817 * Always consume at least one character of the input string 03818 * in order to prevent infinite loops. 03819 */ 03820 if (RSTRING_LEN(str) <= end0) break; 03821 len = rb_enc_fast_mbclen(RSTRING_PTR(str)+end0, RSTRING_END(str), str_enc); 03822 rb_enc_str_buf_cat(dest, RSTRING_PTR(str)+end0, len, str_enc); 03823 offset = end0 + len; 03824 } 03825 cp = RSTRING_PTR(str) + offset; 03826 if (offset > RSTRING_LEN(str)) break; 03827 beg = rb_reg_search(pat, str, offset, 0); 03828 } while (beg >= 0); 03829 if (RSTRING_LEN(str) > offset) { 03830 rb_enc_str_buf_cat(dest, cp, RSTRING_LEN(str) - offset, str_enc); 03831 } 03832 rb_reg_search(pat, str, last, 0); 03833 if (bang) { 03834 rb_str_shared_replace(str, dest); 03835 } 03836 else { 03837 RBASIC(dest)->klass = rb_obj_class(str); 03838 OBJ_INFECT(dest, str); 03839 str = dest; 03840 } 03841 03842 if (tainted) OBJ_TAINT(str); 03843 return str; 03844 } 03845 03846 03847 /* 03848 * call-seq: 03849 * str.gsub!(pattern, replacement) -> str or nil 03850 * str.gsub!(pattern) {|match| block } -> str or nil 03851 * str.gsub!(pattern) -> an_enumerator 03852 * 03853 * Performs the substitutions of <code>String#gsub</code> in place, returning 03854 * <i>str</i>, or <code>nil</code> if no substitutions were performed. 03855 * If no block and no <i>replacement</i> is given, an enumerator is returned instead. 03856 */ 03857 03858 static VALUE 03859 rb_str_gsub_bang(int argc, VALUE *argv, VALUE str) 03860 { 03861 str_modify_keep_cr(str); 03862 return str_gsub(argc, argv, str, 1); 03863 } 03864 03865 03866 /* 03867 * call-seq: 03868 * str.gsub(pattern, replacement) -> new_str 03869 * str.gsub(pattern, hash) -> new_str 03870 * str.gsub(pattern) {|match| block } -> new_str 03871 * str.gsub(pattern) -> enumerator 03872 * 03873 * Returns a copy of <i>str</i> with the <em>all</em> occurrences of 03874 * <i>pattern</i> substituted for the second argument. The <i>pattern</i> is 03875 * typically a <code>Regexp</code>; if given as a <code>String</code>, any 03876 * regular expression metacharacters it contains will be interpreted 03877 * literally, e.g. <code>'\\\d'</code> will match a backlash followed by 'd', 03878 * instead of a digit. 03879 * 03880 * If <i>replacement</i> is a <code>String</code> it will be substituted for 03881 * the matched text. It may contain back-references to the pattern's capture 03882 * groups of the form <code>\\\d</code>, where <i>d</i> is a group number, or 03883 * <code>\\\k<n></code>, where <i>n</i> is a group name. If it is a 03884 * double-quoted string, both back-references must be preceded by an 03885 * additional backslash. However, within <i>replacement</i> the special match 03886 * variables, such as <code>&$</code>, will not refer to the current match. 03887 * 03888 * If the second argument is a <code>Hash</code>, and the matched text is one 03889 * of its keys, the corresponding value is the replacement string. 03890 * 03891 * In the block form, the current match string is passed in as a parameter, 03892 * and variables such as <code>$1</code>, <code>$2</code>, <code>$`</code>, 03893 * <code>$&</code>, and <code>$'</code> will be set appropriately. The value 03894 * returned by the block will be substituted for the match on each call. 03895 * 03896 * The result inherits any tainting in the original string or any supplied 03897 * replacement string. 03898 * 03899 * When neither a block nor a second argument is supplied, an 03900 * <code>Enumerator</code> is returned. 03901 * 03902 * "hello".gsub(/[aeiou]/, '*') #=> "h*ll*" 03903 * "hello".gsub(/([aeiou])/, '<\1>') #=> "h<e>ll<o>" 03904 * "hello".gsub(/./) {|s| s.ord.to_s + ' '} #=> "104 101 108 108 111 " 03905 * "hello".gsub(/(?<foo>[aeiou])/, '{\k<foo>}') #=> "h{e}ll{o}" 03906 * 'hello'.gsub(/[eo]/, 'e' => 3, 'o' => '*') #=> "h3ll*" 03907 */ 03908 03909 static VALUE 03910 rb_str_gsub(int argc, VALUE *argv, VALUE str) 03911 { 03912 return str_gsub(argc, argv, str, 0); 03913 } 03914 03915 03916 /* 03917 * call-seq: 03918 * str.replace(other_str) -> str 03919 * 03920 * Replaces the contents and taintedness of <i>str</i> with the corresponding 03921 * values in <i>other_str</i>. 03922 * 03923 * s = "hello" #=> "hello" 03924 * s.replace "world" #=> "world" 03925 */ 03926 03927 VALUE 03928 rb_str_replace(VALUE str, VALUE str2) 03929 { 03930 str_modifiable(str); 03931 if (str == str2) return str; 03932 03933 StringValue(str2); 03934 str_discard(str); 03935 return str_replace(str, str2); 03936 } 03937 03938 /* 03939 * call-seq: 03940 * string.clear -> string 03941 * 03942 * Makes string empty. 03943 * 03944 * a = "abcde" 03945 * a.clear #=> "" 03946 */ 03947 03948 static VALUE 03949 rb_str_clear(VALUE str) 03950 { 03951 str_discard(str); 03952 STR_SET_EMBED(str); 03953 STR_SET_EMBED_LEN(str, 0); 03954 RSTRING_PTR(str)[0] = 0; 03955 if (rb_enc_asciicompat(STR_ENC_GET(str))) 03956 ENC_CODERANGE_SET(str, ENC_CODERANGE_7BIT); 03957 else 03958 ENC_CODERANGE_SET(str, ENC_CODERANGE_VALID); 03959 return str; 03960 } 03961 03962 /* 03963 * call-seq: 03964 * string.chr -> string 03965 * 03966 * Returns a one-character string at the beginning of the string. 03967 * 03968 * a = "abcde" 03969 * a.chr #=> "a" 03970 */ 03971 03972 static VALUE 03973 rb_str_chr(VALUE str) 03974 { 03975 return rb_str_substr(str, 0, 1); 03976 } 03977 03978 /* 03979 * call-seq: 03980 * str.getbyte(index) -> 0 .. 255 03981 * 03982 * returns the <i>index</i>th byte as an integer. 03983 */ 03984 static VALUE 03985 rb_str_getbyte(VALUE str, VALUE index) 03986 { 03987 long pos = NUM2LONG(index); 03988 03989 if (pos < 0) 03990 pos += RSTRING_LEN(str); 03991 if (pos < 0 || RSTRING_LEN(str) <= pos) 03992 return Qnil; 03993 03994 return INT2FIX((unsigned char)RSTRING_PTR(str)[pos]); 03995 } 03996 03997 /* 03998 * call-seq: 03999 * str.setbyte(index, int) -> int 04000 * 04001 * modifies the <i>index</i>th byte as <i>int</i>. 04002 */ 04003 static VALUE 04004 rb_str_setbyte(VALUE str, VALUE index, VALUE value) 04005 { 04006 long pos = NUM2LONG(index); 04007 int byte = NUM2INT(value); 04008 04009 rb_str_modify(str); 04010 04011 if (pos < -RSTRING_LEN(str) || RSTRING_LEN(str) <= pos) 04012 rb_raise(rb_eIndexError, "index %ld out of string", pos); 04013 if (pos < 0) 04014 pos += RSTRING_LEN(str); 04015 04016 RSTRING_PTR(str)[pos] = byte; 04017 04018 return value; 04019 } 04020 04021 static VALUE 04022 str_byte_substr(VALUE str, long beg, long len) 04023 { 04024 char *p, *s = RSTRING_PTR(str); 04025 long n = RSTRING_LEN(str); 04026 VALUE str2; 04027 04028 if (beg > n || len < 0) return Qnil; 04029 if (beg < 0) { 04030 beg += n; 04031 if (beg < 0) return Qnil; 04032 } 04033 if (beg + len > n) 04034 len = n - beg; 04035 if (len <= 0) { 04036 len = 0; 04037 p = 0; 04038 } 04039 else 04040 p = s + beg; 04041 04042 if (len > RSTRING_EMBED_LEN_MAX && beg + len == n) { 04043 str2 = rb_str_new4(str); 04044 str2 = str_new3(rb_obj_class(str2), str2); 04045 RSTRING(str2)->as.heap.ptr += RSTRING(str2)->as.heap.len - len; 04046 RSTRING(str2)->as.heap.len = len; 04047 } 04048 else { 04049 str2 = rb_str_new5(str, p, len); 04050 rb_enc_cr_str_copy_for_substr(str2, str); 04051 OBJ_INFECT(str2, str); 04052 } 04053 04054 return str2; 04055 } 04056 04057 static VALUE 04058 str_byte_aref(VALUE str, VALUE indx) 04059 { 04060 long idx; 04061 switch (TYPE(indx)) { 04062 case T_FIXNUM: 04063 idx = FIX2LONG(indx); 04064 04065 num_index: 04066 str = str_byte_substr(str, idx, 1); 04067 if (NIL_P(str) || RSTRING_LEN(str) == 0) return Qnil; 04068 return str; 04069 04070 default: 04071 /* check if indx is Range */ 04072 { 04073 long beg, len = RSTRING_LEN(str); 04074 04075 switch (rb_range_beg_len(indx, &beg, &len, len, 0)) { 04076 case Qfalse: 04077 break; 04078 case Qnil: 04079 return Qnil; 04080 default: 04081 return str_byte_substr(str, beg, len); 04082 } 04083 } 04084 idx = NUM2LONG(indx); 04085 goto num_index; 04086 } 04087 return Qnil; /* not reached */ 04088 } 04089 04090 /* 04091 * call-seq: 04092 * str.byteslice(fixnum) -> new_str or nil 04093 * str.byteslice(fixnum, fixnum) -> new_str or nil 04094 * str.byteslice(range) -> new_str or nil 04095 * 04096 * Byte Reference---If passed a single <code>Fixnum</code>, returns a 04097 * substring of one byte at that position. If passed two <code>Fixnum</code> 04098 * objects, returns a substring starting at the offset given by the first, and 04099 * a length given by the second. If given a <code>Range</code>, a substring containing 04100 * bytes at offsets given by the range is returned. In all three cases, if 04101 * an offset is negative, it is counted from the end of <i>str</i>. Returns 04102 * <code>nil</code> if the initial offset falls outside the string, the length 04103 * is negative, or the beginning of the range is greater than the end. 04104 * The encoding of the resulted string keeps original encoding. 04105 * 04106 * "hello".byteslice(1) #=> "e" 04107 * "hello".byteslice(-1) #=> "o" 04108 * "hello".byteslice(1, 2) #=> "el" 04109 * "\x80\u3042".byteslice(1, 3) #=> "\u3042" 04110 * "\x03\u3042\xff".byteslice(1..3) #=> "\u3942" 04111 */ 04112 04113 static VALUE 04114 rb_str_byteslice(int argc, VALUE *argv, VALUE str) 04115 { 04116 if (argc == 2) { 04117 return str_byte_substr(str, NUM2LONG(argv[0]), NUM2LONG(argv[1])); 04118 } 04119 if (argc != 1) { 04120 rb_raise(rb_eArgError, "wrong number of arguments (%d for 1..2)", argc); 04121 } 04122 return str_byte_aref(str, argv[0]); 04123 } 04124 04125 /* 04126 * call-seq: 04127 * str.reverse -> new_str 04128 * 04129 * Returns a new string with the characters from <i>str</i> in reverse order. 04130 * 04131 * "stressed".reverse #=> "desserts" 04132 */ 04133 04134 static VALUE 04135 rb_str_reverse(VALUE str) 04136 { 04137 rb_encoding *enc; 04138 VALUE rev; 04139 char *s, *e, *p; 04140 int single = 1; 04141 04142 if (RSTRING_LEN(str) <= 1) return rb_str_dup(str); 04143 enc = STR_ENC_GET(str); 04144 rev = rb_str_new5(str, 0, RSTRING_LEN(str)); 04145 s = RSTRING_PTR(str); e = RSTRING_END(str); 04146 p = RSTRING_END(rev); 04147 04148 if (RSTRING_LEN(str) > 1) { 04149 if (single_byte_optimizable(str)) { 04150 while (s < e) { 04151 *--p = *s++; 04152 } 04153 } 04154 else if (ENC_CODERANGE(str) == ENC_CODERANGE_VALID) { 04155 while (s < e) { 04156 int clen = rb_enc_fast_mbclen(s, e, enc); 04157 04158 if (clen > 1 || (*s & 0x80)) single = 0; 04159 p -= clen; 04160 memcpy(p, s, clen); 04161 s += clen; 04162 } 04163 } 04164 else { 04165 while (s < e) { 04166 int clen = rb_enc_mbclen(s, e, enc); 04167 04168 if (clen > 1 || (*s & 0x80)) single = 0; 04169 p -= clen; 04170 memcpy(p, s, clen); 04171 s += clen; 04172 } 04173 } 04174 } 04175 STR_SET_LEN(rev, RSTRING_LEN(str)); 04176 OBJ_INFECT(rev, str); 04177 if (ENC_CODERANGE(str) == ENC_CODERANGE_UNKNOWN) { 04178 if (single) { 04179 ENC_CODERANGE_SET(str, ENC_CODERANGE_7BIT); 04180 } 04181 else { 04182 ENC_CODERANGE_SET(str, ENC_CODERANGE_VALID); 04183 } 04184 } 04185 rb_enc_cr_str_copy_for_substr(rev, str); 04186 04187 return rev; 04188 } 04189 04190 04191 /* 04192 * call-seq: 04193 * str.reverse! -> str 04194 * 04195 * Reverses <i>str</i> in place. 04196 */ 04197 04198 static VALUE 04199 rb_str_reverse_bang(VALUE str) 04200 { 04201 if (RSTRING_LEN(str) > 1) { 04202 if (single_byte_optimizable(str)) { 04203 char *s, *e, c; 04204 04205 str_modify_keep_cr(str); 04206 s = RSTRING_PTR(str); 04207 e = RSTRING_END(str) - 1; 04208 while (s < e) { 04209 c = *s; 04210 *s++ = *e; 04211 *e-- = c; 04212 } 04213 } 04214 else { 04215 rb_str_shared_replace(str, rb_str_reverse(str)); 04216 } 04217 } 04218 else { 04219 str_modify_keep_cr(str); 04220 } 04221 return str; 04222 } 04223 04224 04225 /* 04226 * call-seq: 04227 * str.include? other_str -> true or false 04228 * 04229 * Returns <code>true</code> if <i>str</i> contains the given string or 04230 * character. 04231 * 04232 * "hello".include? "lo" #=> true 04233 * "hello".include? "ol" #=> false 04234 * "hello".include? ?h #=> true 04235 */ 04236 04237 static VALUE 04238 rb_str_include(VALUE str, VALUE arg) 04239 { 04240 long i; 04241 04242 StringValue(arg); 04243 i = rb_str_index(str, arg, 0); 04244 04245 if (i == -1) return Qfalse; 04246 return Qtrue; 04247 } 04248 04249 04250 /* 04251 * call-seq: 04252 * str.to_i(base=10) -> integer 04253 * 04254 * Returns the result of interpreting leading characters in <i>str</i> as an 04255 * integer base <i>base</i> (between 2 and 36). Extraneous characters past the 04256 * end of a valid number are ignored. If there is not a valid number at the 04257 * start of <i>str</i>, <code>0</code> is returned. This method never raises an 04258 * exception when <i>base</i> is valid. 04259 * 04260 * "12345".to_i #=> 12345 04261 * "99 red balloons".to_i #=> 99 04262 * "0a".to_i #=> 0 04263 * "0a".to_i(16) #=> 10 04264 * "hello".to_i #=> 0 04265 * "1100101".to_i(2) #=> 101 04266 * "1100101".to_i(8) #=> 294977 04267 * "1100101".to_i(10) #=> 1100101 04268 * "1100101".to_i(16) #=> 17826049 04269 */ 04270 04271 static VALUE 04272 rb_str_to_i(int argc, VALUE *argv, VALUE str) 04273 { 04274 int base; 04275 04276 if (argc == 0) base = 10; 04277 else { 04278 VALUE b; 04279 04280 rb_scan_args(argc, argv, "01", &b); 04281 base = NUM2INT(b); 04282 } 04283 if (base < 0) { 04284 rb_raise(rb_eArgError, "invalid radix %d", base); 04285 } 04286 return rb_str_to_inum(str, base, FALSE); 04287 } 04288 04289 04290 /* 04291 * call-seq: 04292 * str.to_f -> float 04293 * 04294 * Returns the result of interpreting leading characters in <i>str</i> as a 04295 * floating point number. Extraneous characters past the end of a valid number 04296 * are ignored. If there is not a valid number at the start of <i>str</i>, 04297 * <code>0.0</code> is returned. This method never raises an exception. 04298 * 04299 * "123.45e1".to_f #=> 1234.5 04300 * "45.67 degrees".to_f #=> 45.67 04301 * "thx1138".to_f #=> 0.0 04302 */ 04303 04304 static VALUE 04305 rb_str_to_f(VALUE str) 04306 { 04307 return DBL2NUM(rb_str_to_dbl(str, FALSE)); 04308 } 04309 04310 04311 /* 04312 * call-seq: 04313 * str.to_s -> str 04314 * str.to_str -> str 04315 * 04316 * Returns the receiver. 04317 */ 04318 04319 static VALUE 04320 rb_str_to_s(VALUE str) 04321 { 04322 if (rb_obj_class(str) != rb_cString) { 04323 return str_duplicate(rb_cString, str); 04324 } 04325 return str; 04326 } 04327 04328 #if 0 04329 static void 04330 str_cat_char(VALUE str, unsigned int c, rb_encoding *enc) 04331 { 04332 char s[RUBY_MAX_CHAR_LEN]; 04333 int n = rb_enc_codelen(c, enc); 04334 04335 rb_enc_mbcput(c, s, enc); 04336 rb_enc_str_buf_cat(str, s, n, enc); 04337 } 04338 #endif 04339 04340 #define CHAR_ESC_LEN 13 /* sizeof(\x{ hex of 32bit unsigned int } \0) */ 04341 04342 int 04343 rb_str_buf_cat_escaped_char(VALUE result, unsigned int c, int unicode_p) 04344 { 04345 char buf[CHAR_ESC_LEN + 1]; 04346 int l; 04347 04348 #if SIZEOF_INT > 4 04349 c &= 0xffffffff; 04350 #endif 04351 if (unicode_p) { 04352 if (c < 0x7F && ISPRINT(c)) { 04353 snprintf(buf, CHAR_ESC_LEN, "%c", c); 04354 } 04355 else if (c < 0x10000) { 04356 snprintf(buf, CHAR_ESC_LEN, "\\u%04X", c); 04357 } 04358 else { 04359 snprintf(buf, CHAR_ESC_LEN, "\\u{%X}", c); 04360 } 04361 } 04362 else { 04363 if (c < 0x100) { 04364 snprintf(buf, CHAR_ESC_LEN, "\\x%02X", c); 04365 } 04366 else { 04367 snprintf(buf, CHAR_ESC_LEN, "\\x{%X}", c); 04368 } 04369 } 04370 l = (int)strlen(buf); /* CHAR_ESC_LEN cannot exceed INT_MAX */ 04371 rb_str_buf_cat(result, buf, l); 04372 return l; 04373 } 04374 04375 /* 04376 * call-seq: 04377 * str.inspect -> string 04378 * 04379 * Returns a printable version of _str_, surrounded by quote marks, 04380 * with special characters escaped. 04381 * 04382 * str = "hello" 04383 * str[3] = "\b" 04384 * str.inspect #=> "\"hel\\bo\"" 04385 */ 04386 04387 VALUE 04388 rb_str_inspect(VALUE str) 04389 { 04390 rb_encoding *enc = STR_ENC_GET(str); 04391 const char *p, *pend, *prev; 04392 char buf[CHAR_ESC_LEN + 1]; 04393 VALUE result = rb_str_buf_new(0); 04394 rb_encoding *resenc = rb_default_internal_encoding(); 04395 int unicode_p = rb_enc_unicode_p(enc); 04396 int asciicompat = rb_enc_asciicompat(enc); 04397 static rb_encoding *utf16, *utf32; 04398 04399 if (!utf16) utf16 = rb_enc_find("UTF-16"); 04400 if (!utf32) utf32 = rb_enc_find("UTF-32"); 04401 if (resenc == NULL) resenc = rb_default_external_encoding(); 04402 if (!rb_enc_asciicompat(resenc)) resenc = rb_usascii_encoding(); 04403 rb_enc_associate(result, resenc); 04404 str_buf_cat2(result, "\""); 04405 04406 p = RSTRING_PTR(str); pend = RSTRING_END(str); 04407 prev = p; 04408 if (enc == utf16) { 04409 const unsigned char *q = (const unsigned char *)p; 04410 if (q[0] == 0xFE && q[1] == 0xFF) 04411 enc = rb_enc_find("UTF-16BE"); 04412 else if (q[0] == 0xFF && q[1] == 0xFE) 04413 enc = rb_enc_find("UTF-16LE"); 04414 else 04415 unicode_p = 0; 04416 } 04417 else if (enc == utf32) { 04418 const unsigned char *q = (const unsigned char *)p; 04419 if (q[0] == 0 && q[1] == 0 && q[2] == 0xFE && q[3] == 0xFF) 04420 enc = rb_enc_find("UTF-32BE"); 04421 else if (q[3] == 0 && q[2] == 0 && q[1] == 0xFE && q[0] == 0xFF) 04422 enc = rb_enc_find("UTF-32LE"); 04423 else 04424 unicode_p = 0; 04425 } 04426 while (p < pend) { 04427 unsigned int c, cc; 04428 int n; 04429 04430 n = rb_enc_precise_mbclen(p, pend, enc); 04431 if (!MBCLEN_CHARFOUND_P(n)) { 04432 if (p > prev) str_buf_cat(result, prev, p - prev); 04433 n = rb_enc_mbminlen(enc); 04434 if (pend < p + n) 04435 n = (int)(pend - p); 04436 while (n--) { 04437 snprintf(buf, CHAR_ESC_LEN, "\\x%02X", *p & 0377); 04438 str_buf_cat(result, buf, strlen(buf)); 04439 prev = ++p; 04440 } 04441 continue; 04442 } 04443 n = MBCLEN_CHARFOUND_LEN(n); 04444 c = rb_enc_mbc_to_codepoint(p, pend, enc); 04445 p += n; 04446 if ((asciicompat || unicode_p) && 04447 (c == '"'|| c == '\\' || 04448 (c == '#' && 04449 p < pend && 04450 MBCLEN_CHARFOUND_P(rb_enc_precise_mbclen(p,pend,enc)) && 04451 (cc = rb_enc_codepoint(p,pend,enc), 04452 (cc == '$' || cc == '@' || cc == '{'))))) { 04453 if (p - n > prev) str_buf_cat(result, prev, p - n - prev); 04454 str_buf_cat2(result, "\\"); 04455 if (asciicompat || enc == resenc) { 04456 prev = p - n; 04457 continue; 04458 } 04459 } 04460 switch (c) { 04461 case '\n': cc = 'n'; break; 04462 case '\r': cc = 'r'; break; 04463 case '\t': cc = 't'; break; 04464 case '\f': cc = 'f'; break; 04465 case '\013': cc = 'v'; break; 04466 case '\010': cc = 'b'; break; 04467 case '\007': cc = 'a'; break; 04468 case 033: cc = 'e'; break; 04469 default: cc = 0; break; 04470 } 04471 if (cc) { 04472 if (p - n > prev) str_buf_cat(result, prev, p - n - prev); 04473 buf[0] = '\\'; 04474 buf[1] = (char)cc; 04475 str_buf_cat(result, buf, 2); 04476 prev = p; 04477 continue; 04478 } 04479 if ((enc == resenc && rb_enc_isprint(c, enc)) || 04480 (asciicompat && rb_enc_isascii(c, enc) && ISPRINT(c))) { 04481 continue; 04482 } 04483 else { 04484 if (p - n > prev) str_buf_cat(result, prev, p - n - prev); 04485 rb_str_buf_cat_escaped_char(result, c, unicode_p); 04486 prev = p; 04487 continue; 04488 } 04489 } 04490 if (p > prev) str_buf_cat(result, prev, p - prev); 04491 str_buf_cat2(result, "\""); 04492 04493 OBJ_INFECT(result, str); 04494 return result; 04495 } 04496 04497 #define IS_EVSTR(p,e) ((p) < (e) && (*(p) == '$' || *(p) == '@' || *(p) == '{')) 04498 04499 /* 04500 * call-seq: 04501 * str.dump -> new_str 04502 * 04503 * Produces a version of <i>str</i> with all nonprinting characters replaced by 04504 * <code>\nnn</code> notation and all special characters escaped. 04505 */ 04506 04507 VALUE 04508 rb_str_dump(VALUE str) 04509 { 04510 rb_encoding *enc = rb_enc_get(str); 04511 long len; 04512 const char *p, *pend; 04513 char *q, *qend; 04514 VALUE result; 04515 int u8 = (enc == rb_utf8_encoding()); 04516 04517 len = 2; /* "" */ 04518 p = RSTRING_PTR(str); pend = p + RSTRING_LEN(str); 04519 while (p < pend) { 04520 unsigned char c = *p++; 04521 switch (c) { 04522 case '"': case '\\': 04523 case '\n': case '\r': 04524 case '\t': case '\f': 04525 case '\013': case '\010': case '\007': case '\033': 04526 len += 2; 04527 break; 04528 04529 case '#': 04530 len += IS_EVSTR(p, pend) ? 2 : 1; 04531 break; 04532 04533 default: 04534 if (ISPRINT(c)) { 04535 len++; 04536 } 04537 else { 04538 if (u8) { /* \u{NN} */ 04539 int n = rb_enc_precise_mbclen(p-1, pend, enc); 04540 if (MBCLEN_CHARFOUND_P(n-1)) { 04541 unsigned int cc = rb_enc_mbc_to_codepoint(p-1, pend, enc); 04542 while (cc >>= 4) len++; 04543 len += 5; 04544 p += MBCLEN_CHARFOUND_LEN(n)-1; 04545 break; 04546 } 04547 } 04548 len += 4; /* \xNN */ 04549 } 04550 break; 04551 } 04552 } 04553 if (!rb_enc_asciicompat(enc)) { 04554 len += 19; /* ".force_encoding('')" */ 04555 len += strlen(enc->name); 04556 } 04557 04558 result = rb_str_new5(str, 0, len); 04559 p = RSTRING_PTR(str); pend = p + RSTRING_LEN(str); 04560 q = RSTRING_PTR(result); qend = q + len + 1; 04561 04562 *q++ = '"'; 04563 while (p < pend) { 04564 unsigned char c = *p++; 04565 04566 if (c == '"' || c == '\\') { 04567 *q++ = '\\'; 04568 *q++ = c; 04569 } 04570 else if (c == '#') { 04571 if (IS_EVSTR(p, pend)) *q++ = '\\'; 04572 *q++ = '#'; 04573 } 04574 else if (c == '\n') { 04575 *q++ = '\\'; 04576 *q++ = 'n'; 04577 } 04578 else if (c == '\r') { 04579 *q++ = '\\'; 04580 *q++ = 'r'; 04581 } 04582 else if (c == '\t') { 04583 *q++ = '\\'; 04584 *q++ = 't'; 04585 } 04586 else if (c == '\f') { 04587 *q++ = '\\'; 04588 *q++ = 'f'; 04589 } 04590 else if (c == '\013') { 04591 *q++ = '\\'; 04592 *q++ = 'v'; 04593 } 04594 else if (c == '\010') { 04595 *q++ = '\\'; 04596 *q++ = 'b'; 04597 } 04598 else if (c == '\007') { 04599 *q++ = '\\'; 04600 *q++ = 'a'; 04601 } 04602 else if (c == '\033') { 04603 *q++ = '\\'; 04604 *q++ = 'e'; 04605 } 04606 else if (ISPRINT(c)) { 04607 *q++ = c; 04608 } 04609 else { 04610 *q++ = '\\'; 04611 if (u8) { 04612 int n = rb_enc_precise_mbclen(p-1, pend, enc) - 1; 04613 if (MBCLEN_CHARFOUND_P(n)) { 04614 int cc = rb_enc_mbc_to_codepoint(p-1, pend, enc); 04615 p += n; 04616 snprintf(q, qend-q, "u{%x}", cc); 04617 q += strlen(q); 04618 continue; 04619 } 04620 } 04621 snprintf(q, qend-q, "x%02X", c); 04622 q += 3; 04623 } 04624 } 04625 *q++ = '"'; 04626 *q = '\0'; 04627 if (!rb_enc_asciicompat(enc)) { 04628 snprintf(q, qend-q, ".force_encoding(\"%s\")", enc->name); 04629 enc = rb_ascii8bit_encoding(); 04630 } 04631 OBJ_INFECT(result, str); 04632 /* result from dump is ASCII */ 04633 rb_enc_associate(result, enc); 04634 ENC_CODERANGE_SET(result, ENC_CODERANGE_7BIT); 04635 return result; 04636 } 04637 04638 04639 static void 04640 rb_str_check_dummy_enc(rb_encoding *enc) 04641 { 04642 if (rb_enc_dummy_p(enc)) { 04643 rb_raise(rb_eEncCompatError, "incompatible encoding with this operation: %s", 04644 rb_enc_name(enc)); 04645 } 04646 } 04647 04648 /* 04649 * call-seq: 04650 * str.upcase! -> str or nil 04651 * 04652 * Upcases the contents of <i>str</i>, returning <code>nil</code> if no changes 04653 * were made. 04654 * Note: case replacement is effective only in ASCII region. 04655 */ 04656 04657 static VALUE 04658 rb_str_upcase_bang(VALUE str) 04659 { 04660 rb_encoding *enc; 04661 char *s, *send; 04662 int modify = 0; 04663 int n; 04664 04665 str_modify_keep_cr(str); 04666 enc = STR_ENC_GET(str); 04667 rb_str_check_dummy_enc(enc); 04668 s = RSTRING_PTR(str); send = RSTRING_END(str); 04669 if (single_byte_optimizable(str)) { 04670 while (s < send) { 04671 unsigned int c = *(unsigned char*)s; 04672 04673 if (rb_enc_isascii(c, enc) && 'a' <= c && c <= 'z') { 04674 *s = 'A' + (c - 'a'); 04675 modify = 1; 04676 } 04677 s++; 04678 } 04679 } 04680 else { 04681 int ascompat = rb_enc_asciicompat(enc); 04682 04683 while (s < send) { 04684 unsigned int c; 04685 04686 if (ascompat && (c = *(unsigned char*)s) < 0x80) { 04687 if (rb_enc_isascii(c, enc) && 'a' <= c && c <= 'z') { 04688 *s = 'A' + (c - 'a'); 04689 modify = 1; 04690 } 04691 s++; 04692 } 04693 else { 04694 c = rb_enc_codepoint_len(s, send, &n, enc); 04695 if (rb_enc_islower(c, enc)) { 04696 /* assuming toupper returns codepoint with same size */ 04697 rb_enc_mbcput(rb_enc_toupper(c, enc), s, enc); 04698 modify = 1; 04699 } 04700 s += n; 04701 } 04702 } 04703 } 04704 04705 if (modify) return str; 04706 return Qnil; 04707 } 04708 04709 04710 /* 04711 * call-seq: 04712 * str.upcase -> new_str 04713 * 04714 * Returns a copy of <i>str</i> with all lowercase letters replaced with their 04715 * uppercase counterparts. The operation is locale insensitive---only 04716 * characters ``a'' to ``z'' are affected. 04717 * Note: case replacement is effective only in ASCII region. 04718 * 04719 * "hEllO".upcase #=> "HELLO" 04720 */ 04721 04722 static VALUE 04723 rb_str_upcase(VALUE str) 04724 { 04725 str = rb_str_dup(str); 04726 rb_str_upcase_bang(str); 04727 return str; 04728 } 04729 04730 04731 /* 04732 * call-seq: 04733 * str.downcase! -> str or nil 04734 * 04735 * Downcases the contents of <i>str</i>, returning <code>nil</code> if no 04736 * changes were made. 04737 * Note: case replacement is effective only in ASCII region. 04738 */ 04739 04740 static VALUE 04741 rb_str_downcase_bang(VALUE str) 04742 { 04743 rb_encoding *enc; 04744 char *s, *send; 04745 int modify = 0; 04746 04747 str_modify_keep_cr(str); 04748 enc = STR_ENC_GET(str); 04749 rb_str_check_dummy_enc(enc); 04750 s = RSTRING_PTR(str); send = RSTRING_END(str); 04751 if (single_byte_optimizable(str)) { 04752 while (s < send) { 04753 unsigned int c = *(unsigned char*)s; 04754 04755 if (rb_enc_isascii(c, enc) && 'A' <= c && c <= 'Z') { 04756 *s = 'a' + (c - 'A'); 04757 modify = 1; 04758 } 04759 s++; 04760 } 04761 } 04762 else { 04763 int ascompat = rb_enc_asciicompat(enc); 04764 04765 while (s < send) { 04766 unsigned int c; 04767 int n; 04768 04769 if (ascompat && (c = *(unsigned char*)s) < 0x80) { 04770 if (rb_enc_isascii(c, enc) && 'A' <= c && c <= 'Z') { 04771 *s = 'a' + (c - 'A'); 04772 modify = 1; 04773 } 04774 s++; 04775 } 04776 else { 04777 c = rb_enc_codepoint_len(s, send, &n, enc); 04778 if (rb_enc_isupper(c, enc)) { 04779 /* assuming toupper returns codepoint with same size */ 04780 rb_enc_mbcput(rb_enc_tolower(c, enc), s, enc); 04781 modify = 1; 04782 } 04783 s += n; 04784 } 04785 } 04786 } 04787 04788 if (modify) return str; 04789 return Qnil; 04790 } 04791 04792 04793 /* 04794 * call-seq: 04795 * str.downcase -> new_str 04796 * 04797 * Returns a copy of <i>str</i> with all uppercase letters replaced with their 04798 * lowercase counterparts. The operation is locale insensitive---only 04799 * characters ``A'' to ``Z'' are affected. 04800 * Note: case replacement is effective only in ASCII region. 04801 * 04802 * "hEllO".downcase #=> "hello" 04803 */ 04804 04805 static VALUE 04806 rb_str_downcase(VALUE str) 04807 { 04808 str = rb_str_dup(str); 04809 rb_str_downcase_bang(str); 04810 return str; 04811 } 04812 04813 04814 /* 04815 * call-seq: 04816 * str.capitalize! -> str or nil 04817 * 04818 * Modifies <i>str</i> by converting the first character to uppercase and the 04819 * remainder to lowercase. Returns <code>nil</code> if no changes are made. 04820 * Note: case conversion is effective only in ASCII region. 04821 * 04822 * a = "hello" 04823 * a.capitalize! #=> "Hello" 04824 * a #=> "Hello" 04825 * a.capitalize! #=> nil 04826 */ 04827 04828 static VALUE 04829 rb_str_capitalize_bang(VALUE str) 04830 { 04831 rb_encoding *enc; 04832 char *s, *send; 04833 int modify = 0; 04834 unsigned int c; 04835 int n; 04836 04837 str_modify_keep_cr(str); 04838 enc = STR_ENC_GET(str); 04839 rb_str_check_dummy_enc(enc); 04840 if (RSTRING_LEN(str) == 0 || !RSTRING_PTR(str)) return Qnil; 04841 s = RSTRING_PTR(str); send = RSTRING_END(str); 04842 04843 c = rb_enc_codepoint_len(s, send, &n, enc); 04844 if (rb_enc_islower(c, enc)) { 04845 rb_enc_mbcput(rb_enc_toupper(c, enc), s, enc); 04846 modify = 1; 04847 } 04848 s += n; 04849 while (s < send) { 04850 c = rb_enc_codepoint_len(s, send, &n, enc); 04851 if (rb_enc_isupper(c, enc)) { 04852 rb_enc_mbcput(rb_enc_tolower(c, enc), s, enc); 04853 modify = 1; 04854 } 04855 s += n; 04856 } 04857 04858 if (modify) return str; 04859 return Qnil; 04860 } 04861 04862 04863 /* 04864 * call-seq: 04865 * str.capitalize -> new_str 04866 * 04867 * Returns a copy of <i>str</i> with the first character converted to uppercase 04868 * and the remainder to lowercase. 04869 * Note: case conversion is effective only in ASCII region. 04870 * 04871 * "hello".capitalize #=> "Hello" 04872 * "HELLO".capitalize #=> "Hello" 04873 * "123ABC".capitalize #=> "123abc" 04874 */ 04875 04876 static VALUE 04877 rb_str_capitalize(VALUE str) 04878 { 04879 str = rb_str_dup(str); 04880 rb_str_capitalize_bang(str); 04881 return str; 04882 } 04883 04884 04885 /* 04886 * call-seq: 04887 * str.swapcase! -> str or nil 04888 * 04889 * Equivalent to <code>String#swapcase</code>, but modifies the receiver in 04890 * place, returning <i>str</i>, or <code>nil</code> if no changes were made. 04891 * Note: case conversion is effective only in ASCII region. 04892 */ 04893 04894 static VALUE 04895 rb_str_swapcase_bang(VALUE str) 04896 { 04897 rb_encoding *enc; 04898 char *s, *send; 04899 int modify = 0; 04900 int n; 04901 04902 str_modify_keep_cr(str); 04903 enc = STR_ENC_GET(str); 04904 rb_str_check_dummy_enc(enc); 04905 s = RSTRING_PTR(str); send = RSTRING_END(str); 04906 while (s < send) { 04907 unsigned int c = rb_enc_codepoint_len(s, send, &n, enc); 04908 04909 if (rb_enc_isupper(c, enc)) { 04910 /* assuming toupper returns codepoint with same size */ 04911 rb_enc_mbcput(rb_enc_tolower(c, enc), s, enc); 04912 modify = 1; 04913 } 04914 else if (rb_enc_islower(c, enc)) { 04915 /* assuming tolower returns codepoint with same size */ 04916 rb_enc_mbcput(rb_enc_toupper(c, enc), s, enc); 04917 modify = 1; 04918 } 04919 s += n; 04920 } 04921 04922 if (modify) return str; 04923 return Qnil; 04924 } 04925 04926 04927 /* 04928 * call-seq: 04929 * str.swapcase -> new_str 04930 * 04931 * Returns a copy of <i>str</i> with uppercase alphabetic characters converted 04932 * to lowercase and lowercase characters converted to uppercase. 04933 * Note: case conversion is effective only in ASCII region. 04934 * 04935 * "Hello".swapcase #=> "hELLO" 04936 * "cYbEr_PuNk11".swapcase #=> "CyBeR_pUnK11" 04937 */ 04938 04939 static VALUE 04940 rb_str_swapcase(VALUE str) 04941 { 04942 str = rb_str_dup(str); 04943 rb_str_swapcase_bang(str); 04944 return str; 04945 } 04946 04947 typedef unsigned char *USTR; 04948 04949 struct tr { 04950 int gen; 04951 unsigned int now, max; 04952 char *p, *pend; 04953 }; 04954 04955 static unsigned int 04956 trnext(struct tr *t, rb_encoding *enc) 04957 { 04958 int n; 04959 04960 for (;;) { 04961 if (!t->gen) { 04962 if (t->p == t->pend) return -1; 04963 if (t->p < t->pend - 1 && *t->p == '\\') { 04964 t->p++; 04965 } 04966 t->now = rb_enc_codepoint_len(t->p, t->pend, &n, enc); 04967 t->p += n; 04968 if (t->p < t->pend - 1 && *t->p == '-') { 04969 t->p++; 04970 if (t->p < t->pend) { 04971 unsigned int c = rb_enc_codepoint_len(t->p, t->pend, &n, enc); 04972 t->p += n; 04973 if (t->now > c) { 04974 if (t->now < 0x80 && c < 0x80) { 04975 rb_raise(rb_eArgError, 04976 "invalid range \"%c-%c\" in string transliteration", 04977 t->now, c); 04978 } 04979 else { 04980 rb_raise(rb_eArgError, "invalid range in string transliteration"); 04981 } 04982 continue; /* not reached */ 04983 } 04984 t->gen = 1; 04985 t->max = c; 04986 } 04987 } 04988 return t->now; 04989 } 04990 else if (++t->now < t->max) { 04991 return t->now; 04992 } 04993 else { 04994 t->gen = 0; 04995 return t->max; 04996 } 04997 } 04998 } 04999 05000 static VALUE rb_str_delete_bang(int,VALUE*,VALUE); 05001 05002 static VALUE 05003 tr_trans(VALUE str, VALUE src, VALUE repl, int sflag) 05004 { 05005 const unsigned int errc = -1; 05006 unsigned int trans[256]; 05007 rb_encoding *enc, *e1, *e2; 05008 struct tr trsrc, trrepl; 05009 int cflag = 0; 05010 unsigned int c, c0, last = 0; 05011 int modify = 0, i, l; 05012 char *s, *send; 05013 VALUE hash = 0; 05014 int singlebyte = single_byte_optimizable(str); 05015 int cr; 05016 05017 #define CHECK_IF_ASCII(c) \ 05018 (void)((cr == ENC_CODERANGE_7BIT && !rb_isascii(c)) ? \ 05019 (cr = ENC_CODERANGE_VALID) : 0) 05020 05021 StringValue(src); 05022 StringValue(repl); 05023 if (RSTRING_LEN(str) == 0 || !RSTRING_PTR(str)) return Qnil; 05024 if (RSTRING_LEN(repl) == 0) { 05025 return rb_str_delete_bang(1, &src, str); 05026 } 05027 05028 cr = ENC_CODERANGE(str); 05029 e1 = rb_enc_check(str, src); 05030 e2 = rb_enc_check(str, repl); 05031 if (e1 == e2) { 05032 enc = e1; 05033 } 05034 else { 05035 enc = rb_enc_check(src, repl); 05036 } 05037 trsrc.p = RSTRING_PTR(src); trsrc.pend = trsrc.p + RSTRING_LEN(src); 05038 if (RSTRING_LEN(src) > 1 && 05039 rb_enc_ascget(trsrc.p, trsrc.pend, &l, enc) == '^' && 05040 trsrc.p + l < trsrc.pend) { 05041 cflag = 1; 05042 trsrc.p += l; 05043 } 05044 trrepl.p = RSTRING_PTR(repl); 05045 trrepl.pend = trrepl.p + RSTRING_LEN(repl); 05046 trsrc.gen = trrepl.gen = 0; 05047 trsrc.now = trrepl.now = 0; 05048 trsrc.max = trrepl.max = 0; 05049 05050 if (cflag) { 05051 for (i=0; i<256; i++) { 05052 trans[i] = 1; 05053 } 05054 while ((c = trnext(&trsrc, enc)) != errc) { 05055 if (c < 256) { 05056 trans[c] = errc; 05057 } 05058 else { 05059 if (!hash) hash = rb_hash_new(); 05060 rb_hash_aset(hash, UINT2NUM(c), Qtrue); 05061 } 05062 } 05063 while ((c = trnext(&trrepl, enc)) != errc) 05064 /* retrieve last replacer */; 05065 last = trrepl.now; 05066 for (i=0; i<256; i++) { 05067 if (trans[i] != errc) { 05068 trans[i] = last; 05069 } 05070 } 05071 } 05072 else { 05073 unsigned int r; 05074 05075 for (i=0; i<256; i++) { 05076 trans[i] = errc; 05077 } 05078 while ((c = trnext(&trsrc, enc)) != errc) { 05079 r = trnext(&trrepl, enc); 05080 if (r == errc) r = trrepl.now; 05081 if (c < 256) { 05082 trans[c] = r; 05083 if (rb_enc_codelen(r, enc) != 1) singlebyte = 0; 05084 } 05085 else { 05086 if (!hash) hash = rb_hash_new(); 05087 rb_hash_aset(hash, UINT2NUM(c), UINT2NUM(r)); 05088 } 05089 } 05090 } 05091 05092 if (cr == ENC_CODERANGE_VALID) 05093 cr = ENC_CODERANGE_7BIT; 05094 str_modify_keep_cr(str); 05095 s = RSTRING_PTR(str); send = RSTRING_END(str); 05096 if (sflag) { 05097 int clen, tlen; 05098 long offset, max = RSTRING_LEN(str); 05099 unsigned int save = -1; 05100 char *buf = ALLOC_N(char, max), *t = buf; 05101 05102 while (s < send) { 05103 int may_modify = 0; 05104 05105 c0 = c = rb_enc_codepoint_len(s, send, &clen, e1); 05106 tlen = enc == e1 ? clen : rb_enc_codelen(c, enc); 05107 05108 s += clen; 05109 if (c < 256) { 05110 c = trans[c]; 05111 } 05112 else if (hash) { 05113 VALUE tmp = rb_hash_lookup(hash, UINT2NUM(c)); 05114 if (NIL_P(tmp)) { 05115 if (cflag) c = last; 05116 else c = errc; 05117 } 05118 else if (cflag) c = errc; 05119 else c = NUM2INT(tmp); 05120 } 05121 else { 05122 c = errc; 05123 } 05124 if (c != (unsigned int)-1) { 05125 if (save == c) { 05126 CHECK_IF_ASCII(c); 05127 continue; 05128 } 05129 save = c; 05130 tlen = rb_enc_codelen(c, enc); 05131 modify = 1; 05132 } 05133 else { 05134 save = -1; 05135 c = c0; 05136 if (enc != e1) may_modify = 1; 05137 } 05138 while (t - buf + tlen >= max) { 05139 offset = t - buf; 05140 max *= 2; 05141 REALLOC_N(buf, char, max); 05142 t = buf + offset; 05143 } 05144 rb_enc_mbcput(c, t, enc); 05145 if (may_modify && memcmp(s, t, tlen) != 0) { 05146 modify = 1; 05147 } 05148 CHECK_IF_ASCII(c); 05149 t += tlen; 05150 } 05151 if (!STR_EMBED_P(str)) { 05152 xfree(RSTRING(str)->as.heap.ptr); 05153 } 05154 *t = '\0'; 05155 RSTRING(str)->as.heap.ptr = buf; 05156 RSTRING(str)->as.heap.len = t - buf; 05157 STR_SET_NOEMBED(str); 05158 RSTRING(str)->as.heap.aux.capa = max; 05159 } 05160 else if (rb_enc_mbmaxlen(enc) == 1 || (singlebyte && !hash)) { 05161 while (s < send) { 05162 c = (unsigned char)*s; 05163 if (trans[c] != errc) { 05164 if (!cflag) { 05165 c = trans[c]; 05166 *s = c; 05167 modify = 1; 05168 } 05169 else { 05170 *s = last; 05171 modify = 1; 05172 } 05173 } 05174 CHECK_IF_ASCII(c); 05175 s++; 05176 } 05177 } 05178 else { 05179 int clen, tlen, max = (int)(RSTRING_LEN(str) * 1.2); 05180 long offset; 05181 char *buf = ALLOC_N(char, max), *t = buf; 05182 05183 while (s < send) { 05184 int may_modify = 0; 05185 c0 = c = rb_enc_codepoint_len(s, send, &clen, e1); 05186 tlen = enc == e1 ? clen : rb_enc_codelen(c, enc); 05187 05188 if (c < 256) { 05189 c = trans[c]; 05190 } 05191 else if (hash) { 05192 VALUE tmp = rb_hash_lookup(hash, UINT2NUM(c)); 05193 if (NIL_P(tmp)) { 05194 if (cflag) c = last; 05195 else c = errc; 05196 } 05197 else if (cflag) c = errc; 05198 else c = NUM2INT(tmp); 05199 } 05200 else { 05201 c = cflag ? last : errc; 05202 } 05203 if (c != errc) { 05204 tlen = rb_enc_codelen(c, enc); 05205 modify = 1; 05206 } 05207 else { 05208 c = c0; 05209 if (enc != e1) may_modify = 1; 05210 } 05211 while (t - buf + tlen >= max) { 05212 offset = t - buf; 05213 max *= 2; 05214 REALLOC_N(buf, char, max); 05215 t = buf + offset; 05216 } 05217 if (s != t) { 05218 rb_enc_mbcput(c, t, enc); 05219 if (may_modify && memcmp(s, t, tlen) != 0) { 05220 modify = 1; 05221 } 05222 } 05223 CHECK_IF_ASCII(c); 05224 s += clen; 05225 t += tlen; 05226 } 05227 if (!STR_EMBED_P(str)) { 05228 xfree(RSTRING(str)->as.heap.ptr); 05229 } 05230 *t = '\0'; 05231 RSTRING(str)->as.heap.ptr = buf; 05232 RSTRING(str)->as.heap.len = t - buf; 05233 STR_SET_NOEMBED(str); 05234 RSTRING(str)->as.heap.aux.capa = max; 05235 } 05236 05237 if (modify) { 05238 if (cr != ENC_CODERANGE_BROKEN) 05239 ENC_CODERANGE_SET(str, cr); 05240 rb_enc_associate(str, enc); 05241 return str; 05242 } 05243 return Qnil; 05244 } 05245 05246 05247 /* 05248 * call-seq: 05249 * str.tr!(from_str, to_str) -> str or nil 05250 * 05251 * Translates <i>str</i> in place, using the same rules as 05252 * <code>String#tr</code>. Returns <i>str</i>, or <code>nil</code> if no 05253 * changes were made. 05254 */ 05255 05256 static VALUE 05257 rb_str_tr_bang(VALUE str, VALUE src, VALUE repl) 05258 { 05259 return tr_trans(str, src, repl, 0); 05260 } 05261 05262 05263 /* 05264 * call-seq: 05265 * str.tr(from_str, to_str) => new_str 05266 * 05267 * Returns a copy of <i>str</i> with the characters in <i>from_str</i> 05268 * replaced by the corresponding characters in <i>to_str</i>. If 05269 * <i>to_str</i> is shorter than <i>from_str</i>, it is padded with its last 05270 * character in order to maintain the correspondence. 05271 * 05272 * "hello".tr('el', 'ip') #=> "hippo" 05273 * "hello".tr('aeiou', '*') #=> "h*ll*" 05274 * 05275 * Both strings may use the c1-c2 notation to denote ranges of characters, 05276 * and <i>from_str</i> may start with a <code>^</code>, which denotes all 05277 * characters except those listed. 05278 * 05279 * "hello".tr('a-y', 'b-z') #=> "ifmmp" 05280 * "hello".tr('^aeiou', '*') #=> "*e**o" 05281 */ 05282 05283 static VALUE 05284 rb_str_tr(VALUE str, VALUE src, VALUE repl) 05285 { 05286 str = rb_str_dup(str); 05287 tr_trans(str, src, repl, 0); 05288 return str; 05289 } 05290 05291 #define TR_TABLE_SIZE 257 05292 static void 05293 tr_setup_table(VALUE str, char stable[TR_TABLE_SIZE], int first, 05294 VALUE *tablep, VALUE *ctablep, rb_encoding *enc) 05295 { 05296 const unsigned int errc = -1; 05297 char buf[256]; 05298 struct tr tr; 05299 unsigned int c; 05300 VALUE table = 0, ptable = 0; 05301 int i, l, cflag = 0; 05302 05303 tr.p = RSTRING_PTR(str); tr.pend = tr.p + RSTRING_LEN(str); 05304 tr.gen = tr.now = tr.max = 0; 05305 05306 if (RSTRING_LEN(str) > 1 && rb_enc_ascget(tr.p, tr.pend, &l, enc) == '^') { 05307 cflag = 1; 05308 tr.p += l; 05309 } 05310 if (first) { 05311 for (i=0; i<256; i++) { 05312 stable[i] = 1; 05313 } 05314 stable[256] = cflag; 05315 } 05316 else if (stable[256] && !cflag) { 05317 stable[256] = 0; 05318 } 05319 for (i=0; i<256; i++) { 05320 buf[i] = cflag; 05321 } 05322 05323 while ((c = trnext(&tr, enc)) != errc) { 05324 if (c < 256) { 05325 buf[c & 0xff] = !cflag; 05326 } 05327 else { 05328 VALUE key = UINT2NUM(c); 05329 05330 if (!table) { 05331 table = rb_hash_new(); 05332 if (cflag) { 05333 ptable = *ctablep; 05334 *ctablep = table; 05335 } 05336 else { 05337 ptable = *tablep; 05338 *tablep = table; 05339 } 05340 } 05341 if (!ptable || !NIL_P(rb_hash_aref(ptable, key))) { 05342 rb_hash_aset(table, key, Qtrue); 05343 } 05344 } 05345 } 05346 for (i=0; i<256; i++) { 05347 stable[i] = stable[i] && buf[i]; 05348 } 05349 } 05350 05351 05352 static int 05353 tr_find(unsigned int c, char table[TR_TABLE_SIZE], VALUE del, VALUE nodel) 05354 { 05355 if (c < 256) { 05356 return table[c] != 0; 05357 } 05358 else { 05359 VALUE v = UINT2NUM(c); 05360 05361 if (del) { 05362 if (!NIL_P(rb_hash_lookup(del, v)) && 05363 (!nodel || NIL_P(rb_hash_lookup(nodel, v)))) { 05364 return TRUE; 05365 } 05366 } 05367 else if (nodel && !NIL_P(rb_hash_lookup(nodel, v))) { 05368 return FALSE; 05369 } 05370 return table[256] ? TRUE : FALSE; 05371 } 05372 } 05373 05374 /* 05375 * call-seq: 05376 * str.delete!([other_str]+) -> str or nil 05377 * 05378 * Performs a <code>delete</code> operation in place, returning <i>str</i>, or 05379 * <code>nil</code> if <i>str</i> was not modified. 05380 */ 05381 05382 static VALUE 05383 rb_str_delete_bang(int argc, VALUE *argv, VALUE str) 05384 { 05385 char squeez[TR_TABLE_SIZE]; 05386 rb_encoding *enc = 0; 05387 char *s, *send, *t; 05388 VALUE del = 0, nodel = 0; 05389 int modify = 0; 05390 int i, ascompat, cr; 05391 05392 if (RSTRING_LEN(str) == 0 || !RSTRING_PTR(str)) return Qnil; 05393 if (argc < 1) { 05394 rb_raise(rb_eArgError, "wrong number of arguments (at least 1)"); 05395 } 05396 for (i=0; i<argc; i++) { 05397 VALUE s = argv[i]; 05398 05399 StringValue(s); 05400 enc = rb_enc_check(str, s); 05401 tr_setup_table(s, squeez, i==0, &del, &nodel, enc); 05402 } 05403 05404 str_modify_keep_cr(str); 05405 ascompat = rb_enc_asciicompat(enc); 05406 s = t = RSTRING_PTR(str); 05407 send = RSTRING_END(str); 05408 cr = ascompat ? ENC_CODERANGE_7BIT : ENC_CODERANGE_VALID; 05409 while (s < send) { 05410 unsigned int c; 05411 int clen; 05412 05413 if (ascompat && (c = *(unsigned char*)s) < 0x80) { 05414 if (squeez[c]) { 05415 modify = 1; 05416 } 05417 else { 05418 if (t != s) *t = c; 05419 t++; 05420 } 05421 s++; 05422 } 05423 else { 05424 c = rb_enc_codepoint_len(s, send, &clen, enc); 05425 05426 if (tr_find(c, squeez, del, nodel)) { 05427 modify = 1; 05428 } 05429 else { 05430 if (t != s) rb_enc_mbcput(c, t, enc); 05431 t += clen; 05432 if (cr == ENC_CODERANGE_7BIT) cr = ENC_CODERANGE_VALID; 05433 } 05434 s += clen; 05435 } 05436 } 05437 *t = '\0'; 05438 STR_SET_LEN(str, t - RSTRING_PTR(str)); 05439 ENC_CODERANGE_SET(str, cr); 05440 05441 if (modify) return str; 05442 return Qnil; 05443 } 05444 05445 05446 /* 05447 * call-seq: 05448 * str.delete([other_str]+) -> new_str 05449 * 05450 * Returns a copy of <i>str</i> with all characters in the intersection of its 05451 * arguments deleted. Uses the same rules for building the set of characters as 05452 * <code>String#count</code>. 05453 * 05454 * "hello".delete "l","lo" #=> "heo" 05455 * "hello".delete "lo" #=> "he" 05456 * "hello".delete "aeiou", "^e" #=> "hell" 05457 * "hello".delete "ej-m" #=> "ho" 05458 */ 05459 05460 static VALUE 05461 rb_str_delete(int argc, VALUE *argv, VALUE str) 05462 { 05463 str = rb_str_dup(str); 05464 rb_str_delete_bang(argc, argv, str); 05465 return str; 05466 } 05467 05468 05469 /* 05470 * call-seq: 05471 * str.squeeze!([other_str]*) -> str or nil 05472 * 05473 * Squeezes <i>str</i> in place, returning either <i>str</i>, or 05474 * <code>nil</code> if no changes were made. 05475 */ 05476 05477 static VALUE 05478 rb_str_squeeze_bang(int argc, VALUE *argv, VALUE str) 05479 { 05480 char squeez[TR_TABLE_SIZE]; 05481 rb_encoding *enc = 0; 05482 VALUE del = 0, nodel = 0; 05483 char *s, *send, *t; 05484 int i, modify = 0; 05485 int ascompat, singlebyte = single_byte_optimizable(str); 05486 unsigned int save; 05487 05488 if (argc == 0) { 05489 enc = STR_ENC_GET(str); 05490 } 05491 else { 05492 for (i=0; i<argc; i++) { 05493 VALUE s = argv[i]; 05494 05495 StringValue(s); 05496 enc = rb_enc_check(str, s); 05497 if (singlebyte && !single_byte_optimizable(s)) 05498 singlebyte = 0; 05499 tr_setup_table(s, squeez, i==0, &del, &nodel, enc); 05500 } 05501 } 05502 05503 str_modify_keep_cr(str); 05504 s = t = RSTRING_PTR(str); 05505 if (!s || RSTRING_LEN(str) == 0) return Qnil; 05506 send = RSTRING_END(str); 05507 save = -1; 05508 ascompat = rb_enc_asciicompat(enc); 05509 05510 if (singlebyte) { 05511 while (s < send) { 05512 unsigned int c = *(unsigned char*)s++; 05513 if (c != save || (argc > 0 && !squeez[c])) { 05514 *t++ = save = c; 05515 } 05516 } 05517 } else { 05518 while (s < send) { 05519 unsigned int c; 05520 int clen; 05521 05522 if (ascompat && (c = *(unsigned char*)s) < 0x80) { 05523 if (c != save || (argc > 0 && !squeez[c])) { 05524 *t++ = save = c; 05525 } 05526 s++; 05527 } 05528 else { 05529 c = rb_enc_codepoint_len(s, send, &clen, enc); 05530 05531 if (c != save || (argc > 0 && !tr_find(c, squeez, del, nodel))) { 05532 if (t != s) rb_enc_mbcput(c, t, enc); 05533 save = c; 05534 t += clen; 05535 } 05536 s += clen; 05537 } 05538 } 05539 } 05540 05541 *t = '\0'; 05542 if (t - RSTRING_PTR(str) != RSTRING_LEN(str)) { 05543 STR_SET_LEN(str, t - RSTRING_PTR(str)); 05544 modify = 1; 05545 } 05546 05547 if (modify) return str; 05548 return Qnil; 05549 } 05550 05551 05552 /* 05553 * call-seq: 05554 * str.squeeze([other_str]*) -> new_str 05555 * 05556 * Builds a set of characters from the <i>other_str</i> parameter(s) using the 05557 * procedure described for <code>String#count</code>. Returns a new string 05558 * where runs of the same character that occur in this set are replaced by a 05559 * single character. If no arguments are given, all runs of identical 05560 * characters are replaced by a single character. 05561 * 05562 * "yellow moon".squeeze #=> "yelow mon" 05563 * " now is the".squeeze(" ") #=> " now is the" 05564 * "putters shoot balls".squeeze("m-z") #=> "puters shot balls" 05565 */ 05566 05567 static VALUE 05568 rb_str_squeeze(int argc, VALUE *argv, VALUE str) 05569 { 05570 str = rb_str_dup(str); 05571 rb_str_squeeze_bang(argc, argv, str); 05572 return str; 05573 } 05574 05575 05576 /* 05577 * call-seq: 05578 * str.tr_s!(from_str, to_str) -> str or nil 05579 * 05580 * Performs <code>String#tr_s</code> processing on <i>str</i> in place, 05581 * returning <i>str</i>, or <code>nil</code> if no changes were made. 05582 */ 05583 05584 static VALUE 05585 rb_str_tr_s_bang(VALUE str, VALUE src, VALUE repl) 05586 { 05587 return tr_trans(str, src, repl, 1); 05588 } 05589 05590 05591 /* 05592 * call-seq: 05593 * str.tr_s(from_str, to_str) -> new_str 05594 * 05595 * Processes a copy of <i>str</i> as described under <code>String#tr</code>, 05596 * then removes duplicate characters in regions that were affected by the 05597 * translation. 05598 * 05599 * "hello".tr_s('l', 'r') #=> "hero" 05600 * "hello".tr_s('el', '*') #=> "h*o" 05601 * "hello".tr_s('el', 'hx') #=> "hhxo" 05602 */ 05603 05604 static VALUE 05605 rb_str_tr_s(VALUE str, VALUE src, VALUE repl) 05606 { 05607 str = rb_str_dup(str); 05608 tr_trans(str, src, repl, 1); 05609 return str; 05610 } 05611 05612 05613 /* 05614 * call-seq: 05615 * str.count([other_str]+) -> fixnum 05616 * 05617 * Each <i>other_str</i> parameter defines a set of characters to count. The 05618 * intersection of these sets defines the characters to count in 05619 * <i>str</i>. Any <i>other_str</i> that starts with a caret (^) is 05620 * negated. The sequence c1--c2 means all characters between c1 and c2. 05621 * 05622 * a = "hello world" 05623 * a.count "lo" #=> 5 05624 * a.count "lo", "o" #=> 2 05625 * a.count "hello", "^l" #=> 4 05626 * a.count "ej-m" #=> 4 05627 */ 05628 05629 static VALUE 05630 rb_str_count(int argc, VALUE *argv, VALUE str) 05631 { 05632 char table[TR_TABLE_SIZE]; 05633 rb_encoding *enc = 0; 05634 VALUE del = 0, nodel = 0; 05635 char *s, *send; 05636 int i; 05637 int ascompat; 05638 05639 if (argc < 1) { 05640 rb_raise(rb_eArgError, "wrong number of arguments (at least 1)"); 05641 } 05642 for (i=0; i<argc; i++) { 05643 VALUE tstr = argv[i]; 05644 unsigned char c; 05645 05646 StringValue(tstr); 05647 enc = rb_enc_check(str, tstr); 05648 if (argc == 1 && RSTRING_LEN(tstr) == 1 && rb_enc_asciicompat(enc) && 05649 (c = RSTRING_PTR(tstr)[0]) < 0x80 && !is_broken_string(str)) { 05650 int n = 0; 05651 05652 s = RSTRING_PTR(str); 05653 if (!s || RSTRING_LEN(str) == 0) return INT2FIX(0); 05654 send = RSTRING_END(str); 05655 while (s < send) { 05656 if (*(unsigned char*)s++ == c) n++; 05657 } 05658 return INT2NUM(n); 05659 } 05660 tr_setup_table(tstr, table, i==0, &del, &nodel, enc); 05661 } 05662 05663 s = RSTRING_PTR(str); 05664 if (!s || RSTRING_LEN(str) == 0) return INT2FIX(0); 05665 send = RSTRING_END(str); 05666 ascompat = rb_enc_asciicompat(enc); 05667 i = 0; 05668 while (s < send) { 05669 unsigned int c; 05670 05671 if (ascompat && (c = *(unsigned char*)s) < 0x80) { 05672 if (table[c]) { 05673 i++; 05674 } 05675 s++; 05676 } 05677 else { 05678 int clen; 05679 c = rb_enc_codepoint_len(s, send, &clen, enc); 05680 if (tr_find(c, table, del, nodel)) { 05681 i++; 05682 } 05683 s += clen; 05684 } 05685 } 05686 05687 return INT2NUM(i); 05688 } 05689 05690 static const char isspacetable[256] = { 05691 0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 0, 0, 05692 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 05693 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 05694 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 05695 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 05696 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 05697 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 05698 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 05699 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 05700 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 05701 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 05702 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 05703 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 05704 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 05705 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 05706 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0 05707 }; 05708 05709 #define ascii_isspace(c) isspacetable[(unsigned char)(c)] 05710 05711 /* 05712 * call-seq: 05713 * str.split(pattern=$;, [limit]) -> anArray 05714 * 05715 * Divides <i>str</i> into substrings based on a delimiter, returning an array 05716 * of these substrings. 05717 * 05718 * If <i>pattern</i> is a <code>String</code>, then its contents are used as 05719 * the delimiter when splitting <i>str</i>. If <i>pattern</i> is a single 05720 * space, <i>str</i> is split on whitespace, with leading whitespace and runs 05721 * of contiguous whitespace characters ignored. 05722 * 05723 * If <i>pattern</i> is a <code>Regexp</code>, <i>str</i> is divided where the 05724 * pattern matches. Whenever the pattern matches a zero-length string, 05725 * <i>str</i> is split into individual characters. If <i>pattern</i> contains 05726 * groups, the respective matches will be returned in the array as well. 05727 * 05728 * If <i>pattern</i> is omitted, the value of <code>$;</code> is used. If 05729 * <code>$;</code> is <code>nil</code> (which is the default), <i>str</i> is 05730 * split on whitespace as if ` ' were specified. 05731 * 05732 * If the <i>limit</i> parameter is omitted, trailing null fields are 05733 * suppressed. If <i>limit</i> is a positive number, at most that number of 05734 * fields will be returned (if <i>limit</i> is <code>1</code>, the entire 05735 * string is returned as the only entry in an array). If negative, there is no 05736 * limit to the number of fields returned, and trailing null fields are not 05737 * suppressed. 05738 * 05739 * " now's the time".split #=> ["now's", "the", "time"] 05740 * " now's the time".split(' ') #=> ["now's", "the", "time"] 05741 * " now's the time".split(/ /) #=> ["", "now's", "", "the", "time"] 05742 * "1, 2.34,56, 7".split(%r{,\s*}) #=> ["1", "2.34", "56", "7"] 05743 * "hello".split(//) #=> ["h", "e", "l", "l", "o"] 05744 * "hello".split(//, 3) #=> ["h", "e", "llo"] 05745 * "hi mom".split(%r{\s*}) #=> ["h", "i", "m", "o", "m"] 05746 * 05747 * "mellow yellow".split("ello") #=> ["m", "w y", "w"] 05748 * "1,2,,3,4,,".split(',') #=> ["1", "2", "", "3", "4"] 05749 * "1,2,,3,4,,".split(',', 4) #=> ["1", "2", "", "3,4,,"] 05750 * "1,2,,3,4,,".split(',', -4) #=> ["1", "2", "", "3", "4", "", ""] 05751 */ 05752 05753 static VALUE 05754 rb_str_split_m(int argc, VALUE *argv, VALUE str) 05755 { 05756 rb_encoding *enc; 05757 VALUE spat; 05758 VALUE limit; 05759 enum {awk, string, regexp} split_type; 05760 long beg, end, i = 0; 05761 int lim = 0; 05762 VALUE result, tmp; 05763 05764 if (rb_scan_args(argc, argv, "02", &spat, &limit) == 2) { 05765 lim = NUM2INT(limit); 05766 if (lim <= 0) limit = Qnil; 05767 else if (lim == 1) { 05768 if (RSTRING_LEN(str) == 0) 05769 return rb_ary_new2(0); 05770 return rb_ary_new3(1, str); 05771 } 05772 i = 1; 05773 } 05774 05775 enc = STR_ENC_GET(str); 05776 if (NIL_P(spat)) { 05777 if (!NIL_P(rb_fs)) { 05778 spat = rb_fs; 05779 goto fs_set; 05780 } 05781 split_type = awk; 05782 } 05783 else { 05784 fs_set: 05785 if (TYPE(spat) == T_STRING) { 05786 rb_encoding *enc2 = STR_ENC_GET(spat); 05787 05788 split_type = string; 05789 if (RSTRING_LEN(spat) == 0) { 05790 /* Special case - split into chars */ 05791 spat = rb_reg_regcomp(spat); 05792 split_type = regexp; 05793 } 05794 else if (rb_enc_asciicompat(enc2) == 1) { 05795 if (RSTRING_LEN(spat) == 1 && RSTRING_PTR(spat)[0] == ' '){ 05796 split_type = awk; 05797 } 05798 } 05799 else { 05800 int l; 05801 if (rb_enc_ascget(RSTRING_PTR(spat), RSTRING_END(spat), &l, enc2) == ' ' && 05802 RSTRING_LEN(spat) == l) { 05803 split_type = awk; 05804 } 05805 } 05806 } 05807 else { 05808 spat = get_pat(spat, 1); 05809 split_type = regexp; 05810 } 05811 } 05812 05813 result = rb_ary_new(); 05814 beg = 0; 05815 if (split_type == awk) { 05816 char *ptr = RSTRING_PTR(str); 05817 char *eptr = RSTRING_END(str); 05818 char *bptr = ptr; 05819 int skip = 1; 05820 unsigned int c; 05821 05822 end = beg; 05823 if (is_ascii_string(str)) { 05824 while (ptr < eptr) { 05825 c = (unsigned char)*ptr++; 05826 if (skip) { 05827 if (ascii_isspace(c)) { 05828 beg = ptr - bptr; 05829 } 05830 else { 05831 end = ptr - bptr; 05832 skip = 0; 05833 if (!NIL_P(limit) && lim <= i) break; 05834 } 05835 } 05836 else if (ascii_isspace(c)) { 05837 rb_ary_push(result, rb_str_subseq(str, beg, end-beg)); 05838 skip = 1; 05839 beg = ptr - bptr; 05840 if (!NIL_P(limit)) ++i; 05841 } 05842 else { 05843 end = ptr - bptr; 05844 } 05845 } 05846 } 05847 else { 05848 while (ptr < eptr) { 05849 int n; 05850 05851 c = rb_enc_codepoint_len(ptr, eptr, &n, enc); 05852 ptr += n; 05853 if (skip) { 05854 if (rb_isspace(c)) { 05855 beg = ptr - bptr; 05856 } 05857 else { 05858 end = ptr - bptr; 05859 skip = 0; 05860 if (!NIL_P(limit) && lim <= i) break; 05861 } 05862 } 05863 else if (rb_isspace(c)) { 05864 rb_ary_push(result, rb_str_subseq(str, beg, end-beg)); 05865 skip = 1; 05866 beg = ptr - bptr; 05867 if (!NIL_P(limit)) ++i; 05868 } 05869 else { 05870 end = ptr - bptr; 05871 } 05872 } 05873 } 05874 } 05875 else if (split_type == string) { 05876 char *ptr = RSTRING_PTR(str); 05877 char *temp = ptr; 05878 char *eptr = RSTRING_END(str); 05879 char *sptr = RSTRING_PTR(spat); 05880 long slen = RSTRING_LEN(spat); 05881 05882 if (is_broken_string(str)) { 05883 rb_raise(rb_eArgError, "invalid byte sequence in %s", rb_enc_name(STR_ENC_GET(str))); 05884 } 05885 if (is_broken_string(spat)) { 05886 rb_raise(rb_eArgError, "invalid byte sequence in %s", rb_enc_name(STR_ENC_GET(spat))); 05887 } 05888 enc = rb_enc_check(str, spat); 05889 while (ptr < eptr && 05890 (end = rb_memsearch(sptr, slen, ptr, eptr - ptr, enc)) >= 0) { 05891 /* Check we are at the start of a char */ 05892 char *t = rb_enc_right_char_head(ptr, ptr + end, eptr, enc); 05893 if (t != ptr + end) { 05894 ptr = t; 05895 continue; 05896 } 05897 rb_ary_push(result, rb_str_subseq(str, ptr - temp, end)); 05898 ptr += end + slen; 05899 if (!NIL_P(limit) && lim <= ++i) break; 05900 } 05901 beg = ptr - temp; 05902 } 05903 else { 05904 char *ptr = RSTRING_PTR(str); 05905 long len = RSTRING_LEN(str); 05906 long start = beg; 05907 long idx; 05908 int last_null = 0; 05909 struct re_registers *regs; 05910 05911 while ((end = rb_reg_search(spat, str, start, 0)) >= 0) { 05912 regs = RMATCH_REGS(rb_backref_get()); 05913 if (start == end && BEG(0) == END(0)) { 05914 if (!ptr) { 05915 rb_ary_push(result, str_new_empty(str)); 05916 break; 05917 } 05918 else if (last_null == 1) { 05919 rb_ary_push(result, rb_str_subseq(str, beg, 05920 rb_enc_fast_mbclen(ptr+beg, 05921 ptr+len, 05922 enc))); 05923 beg = start; 05924 } 05925 else { 05926 if (ptr+start == ptr+len) 05927 start++; 05928 else 05929 start += rb_enc_fast_mbclen(ptr+start,ptr+len,enc); 05930 last_null = 1; 05931 continue; 05932 } 05933 } 05934 else { 05935 rb_ary_push(result, rb_str_subseq(str, beg, end-beg)); 05936 beg = start = END(0); 05937 } 05938 last_null = 0; 05939 05940 for (idx=1; idx < regs->num_regs; idx++) { 05941 if (BEG(idx) == -1) continue; 05942 if (BEG(idx) == END(idx)) 05943 tmp = str_new_empty(str); 05944 else 05945 tmp = rb_str_subseq(str, BEG(idx), END(idx)-BEG(idx)); 05946 rb_ary_push(result, tmp); 05947 } 05948 if (!NIL_P(limit) && lim <= ++i) break; 05949 } 05950 } 05951 if (RSTRING_LEN(str) > 0 && (!NIL_P(limit) || RSTRING_LEN(str) > beg || lim < 0)) { 05952 if (RSTRING_LEN(str) == beg) 05953 tmp = str_new_empty(str); 05954 else 05955 tmp = rb_str_subseq(str, beg, RSTRING_LEN(str)-beg); 05956 rb_ary_push(result, tmp); 05957 } 05958 if (NIL_P(limit) && lim == 0) { 05959 long len; 05960 while ((len = RARRAY_LEN(result)) > 0 && 05961 (tmp = RARRAY_PTR(result)[len-1], RSTRING_LEN(tmp) == 0)) 05962 rb_ary_pop(result); 05963 } 05964 05965 return result; 05966 } 05967 05968 VALUE 05969 rb_str_split(VALUE str, const char *sep0) 05970 { 05971 VALUE sep; 05972 05973 StringValue(str); 05974 sep = rb_str_new2(sep0); 05975 return rb_str_split_m(1, &sep, str); 05976 } 05977 05978 05979 /* 05980 * call-seq: 05981 * str.each_line(separator=$/) {|substr| block } -> str 05982 * str.each_line(separator=$/) -> an_enumerator 05983 * 05984 * str.lines(separator=$/) {|substr| block } -> str 05985 * str.lines(separator=$/) -> an_enumerator 05986 * 05987 * Splits <i>str</i> using the supplied parameter as the record separator 05988 * (<code>$/</code> by default), passing each substring in turn to the supplied 05989 * block. If a zero-length record separator is supplied, the string is split 05990 * into paragraphs delimited by multiple successive newlines. 05991 * 05992 * If no block is given, an enumerator is returned instead. 05993 * 05994 * print "Example one\n" 05995 * "hello\nworld".each_line {|s| p s} 05996 * print "Example two\n" 05997 * "hello\nworld".each_line('l') {|s| p s} 05998 * print "Example three\n" 05999 * "hello\n\n\nworld".each_line('') {|s| p s} 06000 * 06001 * <em>produces:</em> 06002 * 06003 * Example one 06004 * "hello\n" 06005 * "world" 06006 * Example two 06007 * "hel" 06008 * "l" 06009 * "o\nworl" 06010 * "d" 06011 * Example three 06012 * "hello\n\n\n" 06013 * "world" 06014 */ 06015 06016 static VALUE 06017 rb_str_each_line(int argc, VALUE *argv, VALUE str) 06018 { 06019 rb_encoding *enc; 06020 VALUE rs; 06021 unsigned int newline; 06022 const char *p, *pend, *s, *ptr; 06023 long len, rslen; 06024 VALUE line; 06025 int n; 06026 VALUE orig = str; 06027 06028 if (argc == 0) { 06029 rs = rb_rs; 06030 } 06031 else { 06032 rb_scan_args(argc, argv, "01", &rs); 06033 } 06034 RETURN_ENUMERATOR(str, argc, argv); 06035 if (NIL_P(rs)) { 06036 rb_yield(str); 06037 return orig; 06038 } 06039 str = rb_str_new4(str); 06040 ptr = p = s = RSTRING_PTR(str); 06041 pend = p + RSTRING_LEN(str); 06042 len = RSTRING_LEN(str); 06043 StringValue(rs); 06044 if (rs == rb_default_rs) { 06045 enc = rb_enc_get(str); 06046 while (p < pend) { 06047 char *p0; 06048 06049 p = memchr(p, '\n', pend - p); 06050 if (!p) break; 06051 p0 = rb_enc_left_char_head(s, p, pend, enc); 06052 if (!rb_enc_is_newline(p0, pend, enc)) { 06053 p++; 06054 continue; 06055 } 06056 p = p0 + rb_enc_mbclen(p0, pend, enc); 06057 line = rb_str_new5(str, s, p - s); 06058 OBJ_INFECT(line, str); 06059 rb_enc_cr_str_copy_for_substr(line, str); 06060 rb_yield(line); 06061 str_mod_check(str, ptr, len); 06062 s = p; 06063 } 06064 goto finish; 06065 } 06066 06067 enc = rb_enc_check(str, rs); 06068 rslen = RSTRING_LEN(rs); 06069 if (rslen == 0) { 06070 newline = '\n'; 06071 } 06072 else { 06073 newline = rb_enc_codepoint(RSTRING_PTR(rs), RSTRING_END(rs), enc); 06074 } 06075 06076 while (p < pend) { 06077 unsigned int c = rb_enc_codepoint_len(p, pend, &n, enc); 06078 06079 again: 06080 if (rslen == 0 && c == newline) { 06081 p += n; 06082 if (p < pend && (c = rb_enc_codepoint_len(p, pend, &n, enc)) != newline) { 06083 goto again; 06084 } 06085 while (p < pend && rb_enc_codepoint(p, pend, enc) == newline) { 06086 p += n; 06087 } 06088 p -= n; 06089 } 06090 if (c == newline && 06091 (rslen <= 1 || 06092 (pend - p >= rslen && memcmp(RSTRING_PTR(rs), p, rslen) == 0))) { 06093 line = rb_str_new5(str, s, p - s + (rslen ? rslen : n)); 06094 OBJ_INFECT(line, str); 06095 rb_enc_cr_str_copy_for_substr(line, str); 06096 rb_yield(line); 06097 str_mod_check(str, ptr, len); 06098 s = p + (rslen ? rslen : n); 06099 } 06100 p += n; 06101 } 06102 06103 finish: 06104 if (s != pend) { 06105 line = rb_str_new5(str, s, pend - s); 06106 OBJ_INFECT(line, str); 06107 rb_enc_cr_str_copy_for_substr(line, str); 06108 rb_yield(line); 06109 } 06110 06111 return orig; 06112 } 06113 06114 06115 /* 06116 * call-seq: 06117 * str.bytes {|fixnum| block } -> str 06118 * str.bytes -> an_enumerator 06119 * 06120 * str.each_byte {|fixnum| block } -> str 06121 * str.each_byte -> an_enumerator 06122 * 06123 * Passes each byte in <i>str</i> to the given block, or returns 06124 * an enumerator if no block is given. 06125 * 06126 * "hello".each_byte {|c| print c, ' ' } 06127 * 06128 * <em>produces:</em> 06129 * 06130 * 104 101 108 108 111 06131 */ 06132 06133 static VALUE 06134 rb_str_each_byte(VALUE str) 06135 { 06136 long i; 06137 06138 RETURN_ENUMERATOR(str, 0, 0); 06139 for (i=0; i<RSTRING_LEN(str); i++) { 06140 rb_yield(INT2FIX(RSTRING_PTR(str)[i] & 0xff)); 06141 } 06142 return str; 06143 } 06144 06145 06146 /* 06147 * call-seq: 06148 * str.chars {|cstr| block } -> str 06149 * str.chars -> an_enumerator 06150 * 06151 * str.each_char {|cstr| block } -> str 06152 * str.each_char -> an_enumerator 06153 * 06154 * Passes each character in <i>str</i> to the given block, or returns 06155 * an enumerator if no block is given. 06156 * 06157 * "hello".each_char {|c| print c, ' ' } 06158 * 06159 * <em>produces:</em> 06160 * 06161 * h e l l o 06162 */ 06163 06164 static VALUE 06165 rb_str_each_char(VALUE str) 06166 { 06167 VALUE orig = str; 06168 long i, len, n; 06169 const char *ptr; 06170 rb_encoding *enc; 06171 06172 RETURN_ENUMERATOR(str, 0, 0); 06173 str = rb_str_new4(str); 06174 ptr = RSTRING_PTR(str); 06175 len = RSTRING_LEN(str); 06176 enc = rb_enc_get(str); 06177 switch (ENC_CODERANGE(str)) { 06178 case ENC_CODERANGE_VALID: 06179 case ENC_CODERANGE_7BIT: 06180 for (i = 0; i < len; i += n) { 06181 n = rb_enc_fast_mbclen(ptr + i, ptr + len, enc); 06182 rb_yield(rb_str_subseq(str, i, n)); 06183 } 06184 break; 06185 default: 06186 for (i = 0; i < len; i += n) { 06187 n = rb_enc_mbclen(ptr + i, ptr + len, enc); 06188 rb_yield(rb_str_subseq(str, i, n)); 06189 } 06190 } 06191 return orig; 06192 } 06193 06194 /* 06195 * call-seq: 06196 * str.codepoints {|integer| block } -> str 06197 * str.codepoints -> an_enumerator 06198 * 06199 * str.each_codepoint {|integer| block } -> str 06200 * str.each_codepoint -> an_enumerator 06201 * 06202 * Passes the <code>Integer</code> ordinal of each character in <i>str</i>, 06203 * also known as a <i>codepoint</i> when applied to Unicode strings to the 06204 * given block. 06205 * 06206 * If no block is given, an enumerator is returned instead. 06207 * 06208 * "hello\u0639".each_codepoint {|c| print c, ' ' } 06209 * 06210 * <em>produces:</em> 06211 * 06212 * 104 101 108 108 111 1593 06213 */ 06214 06215 static VALUE 06216 rb_str_each_codepoint(VALUE str) 06217 { 06218 VALUE orig = str; 06219 int n; 06220 unsigned int c; 06221 const char *ptr, *end; 06222 rb_encoding *enc; 06223 06224 if (single_byte_optimizable(str)) return rb_str_each_byte(str); 06225 RETURN_ENUMERATOR(str, 0, 0); 06226 str = rb_str_new4(str); 06227 ptr = RSTRING_PTR(str); 06228 end = RSTRING_END(str); 06229 enc = STR_ENC_GET(str); 06230 while (ptr < end) { 06231 c = rb_enc_codepoint_len(ptr, end, &n, enc); 06232 rb_yield(UINT2NUM(c)); 06233 ptr += n; 06234 } 06235 return orig; 06236 } 06237 06238 static long 06239 chopped_length(VALUE str) 06240 { 06241 rb_encoding *enc = STR_ENC_GET(str); 06242 const char *p, *p2, *beg, *end; 06243 06244 beg = RSTRING_PTR(str); 06245 end = beg + RSTRING_LEN(str); 06246 if (beg > end) return 0; 06247 p = rb_enc_prev_char(beg, end, end, enc); 06248 if (!p) return 0; 06249 if (p > beg && rb_enc_ascget(p, end, 0, enc) == '\n') { 06250 p2 = rb_enc_prev_char(beg, p, end, enc); 06251 if (p2 && rb_enc_ascget(p2, end, 0, enc) == '\r') p = p2; 06252 } 06253 return p - beg; 06254 } 06255 06256 /* 06257 * call-seq: 06258 * str.chop! -> str or nil 06259 * 06260 * Processes <i>str</i> as for <code>String#chop</code>, returning <i>str</i>, 06261 * or <code>nil</code> if <i>str</i> is the empty string. See also 06262 * <code>String#chomp!</code>. 06263 */ 06264 06265 static VALUE 06266 rb_str_chop_bang(VALUE str) 06267 { 06268 str_modify_keep_cr(str); 06269 if (RSTRING_LEN(str) > 0) { 06270 long len; 06271 len = chopped_length(str); 06272 STR_SET_LEN(str, len); 06273 RSTRING_PTR(str)[len] = '\0'; 06274 if (ENC_CODERANGE(str) != ENC_CODERANGE_7BIT) { 06275 ENC_CODERANGE_CLEAR(str); 06276 } 06277 return str; 06278 } 06279 return Qnil; 06280 } 06281 06282 06283 /* 06284 * call-seq: 06285 * str.chop -> new_str 06286 * 06287 * Returns a new <code>String</code> with the last character removed. If the 06288 * string ends with <code>\r\n</code>, both characters are removed. Applying 06289 * <code>chop</code> to an empty string returns an empty 06290 * string. <code>String#chomp</code> is often a safer alternative, as it leaves 06291 * the string unchanged if it doesn't end in a record separator. 06292 * 06293 * "string\r\n".chop #=> "string" 06294 * "string\n\r".chop #=> "string\n" 06295 * "string\n".chop #=> "string" 06296 * "string".chop #=> "strin" 06297 * "x".chop.chop #=> "" 06298 */ 06299 06300 static VALUE 06301 rb_str_chop(VALUE str) 06302 { 06303 VALUE str2 = rb_str_new5(str, RSTRING_PTR(str), chopped_length(str)); 06304 rb_enc_cr_str_copy_for_substr(str2, str); 06305 OBJ_INFECT(str2, str); 06306 return str2; 06307 } 06308 06309 06310 /* 06311 * call-seq: 06312 * str.chomp!(separator=$/) -> str or nil 06313 * 06314 * Modifies <i>str</i> in place as described for <code>String#chomp</code>, 06315 * returning <i>str</i>, or <code>nil</code> if no modifications were made. 06316 */ 06317 06318 static VALUE 06319 rb_str_chomp_bang(int argc, VALUE *argv, VALUE str) 06320 { 06321 rb_encoding *enc; 06322 VALUE rs; 06323 int newline; 06324 char *p, *pp, *e; 06325 long len, rslen; 06326 06327 str_modify_keep_cr(str); 06328 len = RSTRING_LEN(str); 06329 if (len == 0) return Qnil; 06330 p = RSTRING_PTR(str); 06331 e = p + len; 06332 if (argc == 0) { 06333 rs = rb_rs; 06334 if (rs == rb_default_rs) { 06335 smart_chomp: 06336 enc = rb_enc_get(str); 06337 if (rb_enc_mbminlen(enc) > 1) { 06338 pp = rb_enc_left_char_head(p, e-rb_enc_mbminlen(enc), e, enc); 06339 if (rb_enc_is_newline(pp, e, enc)) { 06340 e = pp; 06341 } 06342 pp = e - rb_enc_mbminlen(enc); 06343 if (pp >= p) { 06344 pp = rb_enc_left_char_head(p, pp, e, enc); 06345 if (rb_enc_ascget(pp, e, 0, enc) == '\r') { 06346 e = pp; 06347 } 06348 } 06349 if (e == RSTRING_END(str)) { 06350 return Qnil; 06351 } 06352 len = e - RSTRING_PTR(str); 06353 STR_SET_LEN(str, len); 06354 } 06355 else { 06356 if (RSTRING_PTR(str)[len-1] == '\n') { 06357 STR_DEC_LEN(str); 06358 if (RSTRING_LEN(str) > 0 && 06359 RSTRING_PTR(str)[RSTRING_LEN(str)-1] == '\r') { 06360 STR_DEC_LEN(str); 06361 } 06362 } 06363 else if (RSTRING_PTR(str)[len-1] == '\r') { 06364 STR_DEC_LEN(str); 06365 } 06366 else { 06367 return Qnil; 06368 } 06369 } 06370 RSTRING_PTR(str)[RSTRING_LEN(str)] = '\0'; 06371 return str; 06372 } 06373 } 06374 else { 06375 rb_scan_args(argc, argv, "01", &rs); 06376 } 06377 if (NIL_P(rs)) return Qnil; 06378 StringValue(rs); 06379 rslen = RSTRING_LEN(rs); 06380 if (rslen == 0) { 06381 while (len>0 && p[len-1] == '\n') { 06382 len--; 06383 if (len>0 && p[len-1] == '\r') 06384 len--; 06385 } 06386 if (len < RSTRING_LEN(str)) { 06387 STR_SET_LEN(str, len); 06388 RSTRING_PTR(str)[len] = '\0'; 06389 return str; 06390 } 06391 return Qnil; 06392 } 06393 if (rslen > len) return Qnil; 06394 newline = RSTRING_PTR(rs)[rslen-1]; 06395 if (rslen == 1 && newline == '\n') 06396 goto smart_chomp; 06397 06398 enc = rb_enc_check(str, rs); 06399 if (is_broken_string(rs)) { 06400 return Qnil; 06401 } 06402 pp = e - rslen; 06403 if (p[len-1] == newline && 06404 (rslen <= 1 || 06405 memcmp(RSTRING_PTR(rs), pp, rslen) == 0)) { 06406 if (rb_enc_left_char_head(p, pp, e, enc) != pp) 06407 return Qnil; 06408 if (ENC_CODERANGE(str) != ENC_CODERANGE_7BIT) { 06409 ENC_CODERANGE_CLEAR(str); 06410 } 06411 STR_SET_LEN(str, RSTRING_LEN(str) - rslen); 06412 RSTRING_PTR(str)[RSTRING_LEN(str)] = '\0'; 06413 return str; 06414 } 06415 return Qnil; 06416 } 06417 06418 06419 /* 06420 * call-seq: 06421 * str.chomp(separator=$/) -> new_str 06422 * 06423 * Returns a new <code>String</code> with the given record separator removed 06424 * from the end of <i>str</i> (if present). If <code>$/</code> has not been 06425 * changed from the default Ruby record separator, then <code>chomp</code> also 06426 * removes carriage return characters (that is it will remove <code>\n</code>, 06427 * <code>\r</code>, and <code>\r\n</code>). 06428 * 06429 * "hello".chomp #=> "hello" 06430 * "hello\n".chomp #=> "hello" 06431 * "hello\r\n".chomp #=> "hello" 06432 * "hello\n\r".chomp #=> "hello\n" 06433 * "hello\r".chomp #=> "hello" 06434 * "hello \n there".chomp #=> "hello \n there" 06435 * "hello".chomp("llo") #=> "he" 06436 */ 06437 06438 static VALUE 06439 rb_str_chomp(int argc, VALUE *argv, VALUE str) 06440 { 06441 str = rb_str_dup(str); 06442 rb_str_chomp_bang(argc, argv, str); 06443 return str; 06444 } 06445 06446 /* 06447 * call-seq: 06448 * str.lstrip! -> self or nil 06449 * 06450 * Removes leading whitespace from <i>str</i>, returning <code>nil</code> if no 06451 * change was made. See also <code>String#rstrip!</code> and 06452 * <code>String#strip!</code>. 06453 * 06454 * " hello ".lstrip #=> "hello " 06455 * "hello".lstrip! #=> nil 06456 */ 06457 06458 static VALUE 06459 rb_str_lstrip_bang(VALUE str) 06460 { 06461 rb_encoding *enc; 06462 char *s, *t, *e; 06463 06464 str_modify_keep_cr(str); 06465 enc = STR_ENC_GET(str); 06466 s = RSTRING_PTR(str); 06467 if (!s || RSTRING_LEN(str) == 0) return Qnil; 06468 e = t = RSTRING_END(str); 06469 /* remove spaces at head */ 06470 while (s < e) { 06471 int n; 06472 unsigned int cc = rb_enc_codepoint_len(s, e, &n, enc); 06473 06474 if (!rb_isspace(cc)) break; 06475 s += n; 06476 } 06477 06478 if (s > RSTRING_PTR(str)) { 06479 STR_SET_LEN(str, t-s); 06480 memmove(RSTRING_PTR(str), s, RSTRING_LEN(str)); 06481 RSTRING_PTR(str)[RSTRING_LEN(str)] = '\0'; 06482 return str; 06483 } 06484 return Qnil; 06485 } 06486 06487 06488 /* 06489 * call-seq: 06490 * str.lstrip -> new_str 06491 * 06492 * Returns a copy of <i>str</i> with leading whitespace removed. See also 06493 * <code>String#rstrip</code> and <code>String#strip</code>. 06494 * 06495 * " hello ".lstrip #=> "hello " 06496 * "hello".lstrip #=> "hello" 06497 */ 06498 06499 static VALUE 06500 rb_str_lstrip(VALUE str) 06501 { 06502 str = rb_str_dup(str); 06503 rb_str_lstrip_bang(str); 06504 return str; 06505 } 06506 06507 06508 /* 06509 * call-seq: 06510 * str.rstrip! -> self or nil 06511 * 06512 * Removes trailing whitespace from <i>str</i>, returning <code>nil</code> if 06513 * no change was made. See also <code>String#lstrip!</code> and 06514 * <code>String#strip!</code>. 06515 * 06516 * " hello ".rstrip #=> " hello" 06517 * "hello".rstrip! #=> nil 06518 */ 06519 06520 static VALUE 06521 rb_str_rstrip_bang(VALUE str) 06522 { 06523 rb_encoding *enc; 06524 char *s, *t, *e; 06525 06526 str_modify_keep_cr(str); 06527 enc = STR_ENC_GET(str); 06528 rb_str_check_dummy_enc(enc); 06529 s = RSTRING_PTR(str); 06530 if (!s || RSTRING_LEN(str) == 0) return Qnil; 06531 t = e = RSTRING_END(str); 06532 06533 /* remove trailing spaces or '\0's */ 06534 if (single_byte_optimizable(str)) { 06535 unsigned char c; 06536 while (s < t && ((c = *(t-1)) == '\0' || ascii_isspace(c))) t--; 06537 } 06538 else { 06539 char *tp; 06540 06541 while ((tp = rb_enc_prev_char(s, t, e, enc)) != NULL) { 06542 unsigned int c = rb_enc_codepoint(tp, e, enc); 06543 if (c && !rb_isspace(c)) break; 06544 t = tp; 06545 } 06546 } 06547 if (t < e) { 06548 long len = t-RSTRING_PTR(str); 06549 06550 STR_SET_LEN(str, len); 06551 RSTRING_PTR(str)[len] = '\0'; 06552 return str; 06553 } 06554 return Qnil; 06555 } 06556 06557 06558 /* 06559 * call-seq: 06560 * str.rstrip -> new_str 06561 * 06562 * Returns a copy of <i>str</i> with trailing whitespace removed. See also 06563 * <code>String#lstrip</code> and <code>String#strip</code>. 06564 * 06565 * " hello ".rstrip #=> " hello" 06566 * "hello".rstrip #=> "hello" 06567 */ 06568 06569 static VALUE 06570 rb_str_rstrip(VALUE str) 06571 { 06572 str = rb_str_dup(str); 06573 rb_str_rstrip_bang(str); 06574 return str; 06575 } 06576 06577 06578 /* 06579 * call-seq: 06580 * str.strip! -> str or nil 06581 * 06582 * Removes leading and trailing whitespace from <i>str</i>. Returns 06583 * <code>nil</code> if <i>str</i> was not altered. 06584 */ 06585 06586 static VALUE 06587 rb_str_strip_bang(VALUE str) 06588 { 06589 VALUE l = rb_str_lstrip_bang(str); 06590 VALUE r = rb_str_rstrip_bang(str); 06591 06592 if (NIL_P(l) && NIL_P(r)) return Qnil; 06593 return str; 06594 } 06595 06596 06597 /* 06598 * call-seq: 06599 * str.strip -> new_str 06600 * 06601 * Returns a copy of <i>str</i> with leading and trailing whitespace removed. 06602 * 06603 * " hello ".strip #=> "hello" 06604 * "\tgoodbye\r\n".strip #=> "goodbye" 06605 */ 06606 06607 static VALUE 06608 rb_str_strip(VALUE str) 06609 { 06610 str = rb_str_dup(str); 06611 rb_str_strip_bang(str); 06612 return str; 06613 } 06614 06615 static VALUE 06616 scan_once(VALUE str, VALUE pat, long *start) 06617 { 06618 VALUE result, match; 06619 struct re_registers *regs; 06620 int i; 06621 06622 if (rb_reg_search(pat, str, *start, 0) >= 0) { 06623 match = rb_backref_get(); 06624 regs = RMATCH_REGS(match); 06625 if (BEG(0) == END(0)) { 06626 rb_encoding *enc = STR_ENC_GET(str); 06627 /* 06628 * Always consume at least one character of the input string 06629 */ 06630 if (RSTRING_LEN(str) > END(0)) 06631 *start = END(0)+rb_enc_fast_mbclen(RSTRING_PTR(str)+END(0), 06632 RSTRING_END(str), enc); 06633 else 06634 *start = END(0)+1; 06635 } 06636 else { 06637 *start = END(0); 06638 } 06639 if (regs->num_regs == 1) { 06640 return rb_reg_nth_match(0, match); 06641 } 06642 result = rb_ary_new2(regs->num_regs); 06643 for (i=1; i < regs->num_regs; i++) { 06644 rb_ary_push(result, rb_reg_nth_match(i, match)); 06645 } 06646 06647 return result; 06648 } 06649 return Qnil; 06650 } 06651 06652 06653 /* 06654 * call-seq: 06655 * str.scan(pattern) -> array 06656 * str.scan(pattern) {|match, ...| block } -> str 06657 * 06658 * Both forms iterate through <i>str</i>, matching the pattern (which may be a 06659 * <code>Regexp</code> or a <code>String</code>). For each match, a result is 06660 * generated and either added to the result array or passed to the block. If 06661 * the pattern contains no groups, each individual result consists of the 06662 * matched string, <code>$&</code>. If the pattern contains groups, each 06663 * individual result is itself an array containing one entry per group. 06664 * 06665 * a = "cruel world" 06666 * a.scan(/\w+/) #=> ["cruel", "world"] 06667 * a.scan(/.../) #=> ["cru", "el ", "wor"] 06668 * a.scan(/(...)/) #=> [["cru"], ["el "], ["wor"]] 06669 * a.scan(/(..)(..)/) #=> [["cr", "ue"], ["l ", "wo"]] 06670 * 06671 * And the block form: 06672 * 06673 * a.scan(/\w+/) {|w| print "<<#{w}>> " } 06674 * print "\n" 06675 * a.scan(/(.)(.)/) {|x,y| print y, x } 06676 * print "\n" 06677 * 06678 * <em>produces:</em> 06679 * 06680 * <<cruel>> <<world>> 06681 * rceu lowlr 06682 */ 06683 06684 static VALUE 06685 rb_str_scan(VALUE str, VALUE pat) 06686 { 06687 VALUE result; 06688 long start = 0; 06689 long last = -1, prev = 0; 06690 char *p = RSTRING_PTR(str); long len = RSTRING_LEN(str); 06691 06692 pat = get_pat(pat, 1); 06693 if (!rb_block_given_p()) { 06694 VALUE ary = rb_ary_new(); 06695 06696 while (!NIL_P(result = scan_once(str, pat, &start))) { 06697 last = prev; 06698 prev = start; 06699 rb_ary_push(ary, result); 06700 } 06701 if (last >= 0) rb_reg_search(pat, str, last, 0); 06702 return ary; 06703 } 06704 06705 while (!NIL_P(result = scan_once(str, pat, &start))) { 06706 last = prev; 06707 prev = start; 06708 rb_yield(result); 06709 str_mod_check(str, p, len); 06710 } 06711 if (last >= 0) rb_reg_search(pat, str, last, 0); 06712 return str; 06713 } 06714 06715 06716 /* 06717 * call-seq: 06718 * str.hex -> integer 06719 * 06720 * Treats leading characters from <i>str</i> as a string of hexadecimal digits 06721 * (with an optional sign and an optional <code>0x</code>) and returns the 06722 * corresponding number. Zero is returned on error. 06723 * 06724 * "0x0a".hex #=> 10 06725 * "-1234".hex #=> -4660 06726 * "0".hex #=> 0 06727 * "wombat".hex #=> 0 06728 */ 06729 06730 static VALUE 06731 rb_str_hex(VALUE str) 06732 { 06733 rb_encoding *enc = rb_enc_get(str); 06734 06735 if (!rb_enc_asciicompat(enc)) { 06736 rb_raise(rb_eEncCompatError, "ASCII incompatible encoding: %s", rb_enc_name(enc)); 06737 } 06738 return rb_str_to_inum(str, 16, FALSE); 06739 } 06740 06741 06742 /* 06743 * call-seq: 06744 * str.oct -> integer 06745 * 06746 * Treats leading characters of <i>str</i> as a string of octal digits (with an 06747 * optional sign) and returns the corresponding number. Returns 0 if the 06748 * conversion fails. 06749 * 06750 * "123".oct #=> 83 06751 * "-377".oct #=> -255 06752 * "bad".oct #=> 0 06753 * "0377bad".oct #=> 255 06754 */ 06755 06756 static VALUE 06757 rb_str_oct(VALUE str) 06758 { 06759 rb_encoding *enc = rb_enc_get(str); 06760 06761 if (!rb_enc_asciicompat(enc)) { 06762 rb_raise(rb_eEncCompatError, "ASCII incompatible encoding: %s", rb_enc_name(enc)); 06763 } 06764 return rb_str_to_inum(str, -8, FALSE); 06765 } 06766 06767 06768 /* 06769 * call-seq: 06770 * str.crypt(other_str) -> new_str 06771 * 06772 * Applies a one-way cryptographic hash to <i>str</i> by invoking the standard 06773 * library function <code>crypt</code>. The argument is the salt string, which 06774 * should be two characters long, each character drawn from 06775 * <code>[a-zA-Z0-9./]</code>. 06776 */ 06777 06778 static VALUE 06779 rb_str_crypt(VALUE str, VALUE salt) 06780 { 06781 extern char *crypt(const char *, const char *); 06782 VALUE result; 06783 const char *s, *saltp; 06784 #ifdef BROKEN_CRYPT 06785 char salt_8bit_clean[3]; 06786 #endif 06787 06788 StringValue(salt); 06789 if (RSTRING_LEN(salt) < 2) 06790 rb_raise(rb_eArgError, "salt too short (need >=2 bytes)"); 06791 06792 s = RSTRING_PTR(str); 06793 if (!s) s = ""; 06794 saltp = RSTRING_PTR(salt); 06795 #ifdef BROKEN_CRYPT 06796 if (!ISASCII((unsigned char)saltp[0]) || !ISASCII((unsigned char)saltp[1])) { 06797 salt_8bit_clean[0] = saltp[0] & 0x7f; 06798 salt_8bit_clean[1] = saltp[1] & 0x7f; 06799 salt_8bit_clean[2] = '\0'; 06800 saltp = salt_8bit_clean; 06801 } 06802 #endif 06803 result = rb_str_new2(crypt(s, saltp)); 06804 OBJ_INFECT(result, str); 06805 OBJ_INFECT(result, salt); 06806 return result; 06807 } 06808 06809 06810 /* 06811 * call-seq: 06812 * str.intern -> symbol 06813 * str.to_sym -> symbol 06814 * 06815 * Returns the <code>Symbol</code> corresponding to <i>str</i>, creating the 06816 * symbol if it did not previously exist. See <code>Symbol#id2name</code>. 06817 * 06818 * "Koala".intern #=> :Koala 06819 * s = 'cat'.to_sym #=> :cat 06820 * s == :cat #=> true 06821 * s = '@cat'.to_sym #=> :@cat 06822 * s == :@cat #=> true 06823 * 06824 * This can also be used to create symbols that cannot be represented using the 06825 * <code>:xxx</code> notation. 06826 * 06827 * 'cat and dog'.to_sym #=> :"cat and dog" 06828 */ 06829 06830 VALUE 06831 rb_str_intern(VALUE s) 06832 { 06833 VALUE str = RB_GC_GUARD(s); 06834 ID id; 06835 06836 id = rb_intern_str(str); 06837 return ID2SYM(id); 06838 } 06839 06840 06841 /* 06842 * call-seq: 06843 * str.ord -> integer 06844 * 06845 * Return the <code>Integer</code> ordinal of a one-character string. 06846 * 06847 * "a".ord #=> 97 06848 */ 06849 06850 VALUE 06851 rb_str_ord(VALUE s) 06852 { 06853 unsigned int c; 06854 06855 c = rb_enc_codepoint(RSTRING_PTR(s), RSTRING_END(s), STR_ENC_GET(s)); 06856 return UINT2NUM(c); 06857 } 06858 /* 06859 * call-seq: 06860 * str.sum(n=16) -> integer 06861 * 06862 * Returns a basic <em>n</em>-bit checksum of the characters in <i>str</i>, 06863 * where <em>n</em> is the optional <code>Fixnum</code> parameter, defaulting 06864 * to 16. The result is simply the sum of the binary value of each character in 06865 * <i>str</i> modulo <code>2**n - 1</code>. This is not a particularly good 06866 * checksum. 06867 */ 06868 06869 static VALUE 06870 rb_str_sum(int argc, VALUE *argv, VALUE str) 06871 { 06872 VALUE vbits; 06873 int bits; 06874 char *ptr, *p, *pend; 06875 long len; 06876 VALUE sum = INT2FIX(0); 06877 unsigned long sum0 = 0; 06878 06879 if (argc == 0) { 06880 bits = 16; 06881 } 06882 else { 06883 rb_scan_args(argc, argv, "01", &vbits); 06884 bits = NUM2INT(vbits); 06885 } 06886 ptr = p = RSTRING_PTR(str); 06887 len = RSTRING_LEN(str); 06888 pend = p + len; 06889 06890 while (p < pend) { 06891 if (FIXNUM_MAX - UCHAR_MAX < sum0) { 06892 sum = rb_funcall(sum, '+', 1, LONG2FIX(sum0)); 06893 str_mod_check(str, ptr, len); 06894 sum0 = 0; 06895 } 06896 sum0 += (unsigned char)*p; 06897 p++; 06898 } 06899 06900 if (bits == 0) { 06901 if (sum0) { 06902 sum = rb_funcall(sum, '+', 1, LONG2FIX(sum0)); 06903 } 06904 } 06905 else { 06906 if (sum == INT2FIX(0)) { 06907 if (bits < (int)sizeof(long)*CHAR_BIT) { 06908 sum0 &= (((unsigned long)1)<<bits)-1; 06909 } 06910 sum = LONG2FIX(sum0); 06911 } 06912 else { 06913 VALUE mod; 06914 06915 if (sum0) { 06916 sum = rb_funcall(sum, '+', 1, LONG2FIX(sum0)); 06917 } 06918 06919 mod = rb_funcall(INT2FIX(1), rb_intern("<<"), 1, INT2FIX(bits)); 06920 mod = rb_funcall(mod, '-', 1, INT2FIX(1)); 06921 sum = rb_funcall(sum, '&', 1, mod); 06922 } 06923 } 06924 return sum; 06925 } 06926 06927 static VALUE 06928 rb_str_justify(int argc, VALUE *argv, VALUE str, char jflag) 06929 { 06930 rb_encoding *enc; 06931 VALUE w; 06932 long width, len, flen = 1, fclen = 1; 06933 VALUE res; 06934 char *p; 06935 const char *f = " "; 06936 long n, size, llen, rlen, llen2 = 0, rlen2 = 0; 06937 volatile VALUE pad; 06938 int singlebyte = 1, cr; 06939 06940 rb_scan_args(argc, argv, "11", &w, &pad); 06941 enc = STR_ENC_GET(str); 06942 width = NUM2LONG(w); 06943 if (argc == 2) { 06944 StringValue(pad); 06945 enc = rb_enc_check(str, pad); 06946 f = RSTRING_PTR(pad); 06947 flen = RSTRING_LEN(pad); 06948 fclen = str_strlen(pad, enc); 06949 singlebyte = single_byte_optimizable(pad); 06950 if (flen == 0 || fclen == 0) { 06951 rb_raise(rb_eArgError, "zero width padding"); 06952 } 06953 } 06954 len = str_strlen(str, enc); 06955 if (width < 0 || len >= width) return rb_str_dup(str); 06956 n = width - len; 06957 llen = (jflag == 'l') ? 0 : ((jflag == 'r') ? n : n/2); 06958 rlen = n - llen; 06959 cr = ENC_CODERANGE(str); 06960 if (flen > 1) { 06961 llen2 = str_offset(f, f + flen, llen % fclen, enc, singlebyte); 06962 rlen2 = str_offset(f, f + flen, rlen % fclen, enc, singlebyte); 06963 } 06964 size = RSTRING_LEN(str); 06965 if ((len = llen / fclen + rlen / fclen) >= LONG_MAX / flen || 06966 (len *= flen) >= LONG_MAX - llen2 - rlen2 || 06967 (len += llen2 + rlen2) >= LONG_MAX - size) { 06968 rb_raise(rb_eArgError, "argument too big"); 06969 } 06970 len += size; 06971 res = rb_str_new5(str, 0, len); 06972 p = RSTRING_PTR(res); 06973 if (flen <= 1) { 06974 memset(p, *f, llen); 06975 p += llen; 06976 } 06977 else { 06978 while (llen >= fclen) { 06979 memcpy(p,f,flen); 06980 p += flen; 06981 llen -= fclen; 06982 } 06983 if (llen > 0) { 06984 memcpy(p, f, llen2); 06985 p += llen2; 06986 } 06987 } 06988 memcpy(p, RSTRING_PTR(str), size); 06989 p += size; 06990 if (flen <= 1) { 06991 memset(p, *f, rlen); 06992 p += rlen; 06993 } 06994 else { 06995 while (rlen >= fclen) { 06996 memcpy(p,f,flen); 06997 p += flen; 06998 rlen -= fclen; 06999 } 07000 if (rlen > 0) { 07001 memcpy(p, f, rlen2); 07002 p += rlen2; 07003 } 07004 } 07005 *p = '\0'; 07006 STR_SET_LEN(res, p-RSTRING_PTR(res)); 07007 OBJ_INFECT(res, str); 07008 if (!NIL_P(pad)) OBJ_INFECT(res, pad); 07009 rb_enc_associate(res, enc); 07010 if (argc == 2) 07011 cr = ENC_CODERANGE_AND(cr, ENC_CODERANGE(pad)); 07012 if (cr != ENC_CODERANGE_BROKEN) 07013 ENC_CODERANGE_SET(res, cr); 07014 return res; 07015 } 07016 07017 07018 /* 07019 * call-seq: 07020 * str.ljust(integer, padstr=' ') -> new_str 07021 * 07022 * If <i>integer</i> is greater than the length of <i>str</i>, returns a new 07023 * <code>String</code> of length <i>integer</i> with <i>str</i> left justified 07024 * and padded with <i>padstr</i>; otherwise, returns <i>str</i>. 07025 * 07026 * "hello".ljust(4) #=> "hello" 07027 * "hello".ljust(20) #=> "hello " 07028 * "hello".ljust(20, '1234') #=> "hello123412341234123" 07029 */ 07030 07031 static VALUE 07032 rb_str_ljust(int argc, VALUE *argv, VALUE str) 07033 { 07034 return rb_str_justify(argc, argv, str, 'l'); 07035 } 07036 07037 07038 /* 07039 * call-seq: 07040 * str.rjust(integer, padstr=' ') -> new_str 07041 * 07042 * If <i>integer</i> is greater than the length of <i>str</i>, returns a new 07043 * <code>String</code> of length <i>integer</i> with <i>str</i> right justified 07044 * and padded with <i>padstr</i>; otherwise, returns <i>str</i>. 07045 * 07046 * "hello".rjust(4) #=> "hello" 07047 * "hello".rjust(20) #=> " hello" 07048 * "hello".rjust(20, '1234') #=> "123412341234123hello" 07049 */ 07050 07051 static VALUE 07052 rb_str_rjust(int argc, VALUE *argv, VALUE str) 07053 { 07054 return rb_str_justify(argc, argv, str, 'r'); 07055 } 07056 07057 07058 /* 07059 * call-seq: 07060 * str.center(integer, padstr) -> new_str 07061 * 07062 * If <i>integer</i> is greater than the length of <i>str</i>, returns a new 07063 * <code>String</code> of length <i>integer</i> with <i>str</i> centered and 07064 * padded with <i>padstr</i>; otherwise, returns <i>str</i>. 07065 * 07066 * "hello".center(4) #=> "hello" 07067 * "hello".center(20) #=> " hello " 07068 * "hello".center(20, '123') #=> "1231231hello12312312" 07069 */ 07070 07071 static VALUE 07072 rb_str_center(int argc, VALUE *argv, VALUE str) 07073 { 07074 return rb_str_justify(argc, argv, str, 'c'); 07075 } 07076 07077 /* 07078 * call-seq: 07079 * str.partition(sep) -> [head, sep, tail] 07080 * str.partition(regexp) -> [head, match, tail] 07081 * 07082 * Searches <i>sep</i> or pattern (<i>regexp</i>) in the string 07083 * and returns the part before it, the match, and the part 07084 * after it. 07085 * If it is not found, returns two empty strings and <i>str</i>. 07086 * 07087 * "hello".partition("l") #=> ["he", "l", "lo"] 07088 * "hello".partition("x") #=> ["hello", "", ""] 07089 * "hello".partition(/.l/) #=> ["h", "el", "lo"] 07090 */ 07091 07092 static VALUE 07093 rb_str_partition(VALUE str, VALUE sep) 07094 { 07095 long pos; 07096 int regex = FALSE; 07097 07098 if (TYPE(sep) == T_REGEXP) { 07099 pos = rb_reg_search(sep, str, 0, 0); 07100 regex = TRUE; 07101 } 07102 else { 07103 VALUE tmp; 07104 07105 tmp = rb_check_string_type(sep); 07106 if (NIL_P(tmp)) { 07107 rb_raise(rb_eTypeError, "type mismatch: %s given", 07108 rb_obj_classname(sep)); 07109 } 07110 sep = tmp; 07111 pos = rb_str_index(str, sep, 0); 07112 } 07113 if (pos < 0) { 07114 failed: 07115 return rb_ary_new3(3, str, str_new_empty(str), str_new_empty(str)); 07116 } 07117 if (regex) { 07118 sep = rb_str_subpat(str, sep, INT2FIX(0)); 07119 if (pos == 0 && RSTRING_LEN(sep) == 0) goto failed; 07120 } 07121 return rb_ary_new3(3, rb_str_subseq(str, 0, pos), 07122 sep, 07123 rb_str_subseq(str, pos+RSTRING_LEN(sep), 07124 RSTRING_LEN(str)-pos-RSTRING_LEN(sep))); 07125 } 07126 07127 /* 07128 * call-seq: 07129 * str.rpartition(sep) -> [head, sep, tail] 07130 * str.rpartition(regexp) -> [head, match, tail] 07131 * 07132 * Searches <i>sep</i> or pattern (<i>regexp</i>) in the string from the end 07133 * of the string, and returns the part before it, the match, and the part 07134 * after it. 07135 * If it is not found, returns two empty strings and <i>str</i>. 07136 * 07137 * "hello".rpartition("l") #=> ["hel", "l", "o"] 07138 * "hello".rpartition("x") #=> ["", "", "hello"] 07139 * "hello".rpartition(/.l/) #=> ["he", "ll", "o"] 07140 */ 07141 07142 static VALUE 07143 rb_str_rpartition(VALUE str, VALUE sep) 07144 { 07145 long pos = RSTRING_LEN(str); 07146 int regex = FALSE; 07147 07148 if (TYPE(sep) == T_REGEXP) { 07149 pos = rb_reg_search(sep, str, pos, 1); 07150 regex = TRUE; 07151 } 07152 else { 07153 VALUE tmp; 07154 07155 tmp = rb_check_string_type(sep); 07156 if (NIL_P(tmp)) { 07157 rb_raise(rb_eTypeError, "type mismatch: %s given", 07158 rb_obj_classname(sep)); 07159 } 07160 sep = tmp; 07161 pos = rb_str_sublen(str, pos); 07162 pos = rb_str_rindex(str, sep, pos); 07163 } 07164 if (pos < 0) { 07165 return rb_ary_new3(3, str_new_empty(str), str_new_empty(str), str); 07166 } 07167 if (regex) { 07168 sep = rb_reg_nth_match(0, rb_backref_get()); 07169 } 07170 return rb_ary_new3(3, rb_str_substr(str, 0, pos), 07171 sep, 07172 rb_str_substr(str,pos+str_strlen(sep,STR_ENC_GET(sep)),RSTRING_LEN(str))); 07173 } 07174 07175 /* 07176 * call-seq: 07177 * str.start_with?([prefix]+) -> true or false 07178 * 07179 * Returns true if <i>str</i> starts with one of the prefixes given. 07180 * 07181 * p "hello".start_with?("hell") #=> true 07182 * 07183 * # returns true if one of the prefixes matches. 07184 * p "hello".start_with?("heaven", "hell") #=> true 07185 * p "hello".start_with?("heaven", "paradise") #=> false 07186 * 07187 * 07188 * 07189 */ 07190 07191 static VALUE 07192 rb_str_start_with(int argc, VALUE *argv, VALUE str) 07193 { 07194 int i; 07195 07196 for (i=0; i<argc; i++) { 07197 VALUE tmp = rb_check_string_type(argv[i]); 07198 if (NIL_P(tmp)) continue; 07199 rb_enc_check(str, tmp); 07200 if (RSTRING_LEN(str) < RSTRING_LEN(tmp)) continue; 07201 if (memcmp(RSTRING_PTR(str), RSTRING_PTR(tmp), RSTRING_LEN(tmp)) == 0) 07202 return Qtrue; 07203 } 07204 return Qfalse; 07205 } 07206 07207 /* 07208 * call-seq: 07209 * str.end_with?([suffix]+) -> true or false 07210 * 07211 * Returns true if <i>str</i> ends with one of the suffixes given. 07212 */ 07213 07214 static VALUE 07215 rb_str_end_with(int argc, VALUE *argv, VALUE str) 07216 { 07217 int i; 07218 char *p, *s, *e; 07219 rb_encoding *enc; 07220 07221 for (i=0; i<argc; i++) { 07222 VALUE tmp = rb_check_string_type(argv[i]); 07223 if (NIL_P(tmp)) continue; 07224 enc = rb_enc_check(str, tmp); 07225 if (RSTRING_LEN(str) < RSTRING_LEN(tmp)) continue; 07226 p = RSTRING_PTR(str); 07227 e = p + RSTRING_LEN(str); 07228 s = e - RSTRING_LEN(tmp); 07229 if (rb_enc_left_char_head(p, s, e, enc) != s) 07230 continue; 07231 if (memcmp(s, RSTRING_PTR(tmp), RSTRING_LEN(tmp)) == 0) 07232 return Qtrue; 07233 } 07234 return Qfalse; 07235 } 07236 07237 void 07238 rb_str_setter(VALUE val, ID id, VALUE *var) 07239 { 07240 if (!NIL_P(val) && TYPE(val) != T_STRING) { 07241 rb_raise(rb_eTypeError, "value of %s must be String", rb_id2name(id)); 07242 } 07243 *var = val; 07244 } 07245 07246 07247 /* 07248 * call-seq: 07249 * str.force_encoding(encoding) -> str 07250 * 07251 * Changes the encoding to +encoding+ and returns self. 07252 */ 07253 07254 static VALUE 07255 rb_str_force_encoding(VALUE str, VALUE enc) 07256 { 07257 str_modifiable(str); 07258 rb_enc_associate(str, rb_to_encoding(enc)); 07259 ENC_CODERANGE_CLEAR(str); 07260 return str; 07261 } 07262 07263 /* 07264 * call-seq: 07265 * str.valid_encoding? -> true or false 07266 * 07267 * Returns true for a string which encoded correctly. 07268 * 07269 * "\xc2\xa1".force_encoding("UTF-8").valid_encoding? #=> true 07270 * "\xc2".force_encoding("UTF-8").valid_encoding? #=> false 07271 * "\x80".force_encoding("UTF-8").valid_encoding? #=> false 07272 */ 07273 07274 static VALUE 07275 rb_str_valid_encoding_p(VALUE str) 07276 { 07277 int cr = rb_enc_str_coderange(str); 07278 07279 return cr == ENC_CODERANGE_BROKEN ? Qfalse : Qtrue; 07280 } 07281 07282 /* 07283 * call-seq: 07284 * str.ascii_only? -> true or false 07285 * 07286 * Returns true for a string which has only ASCII characters. 07287 * 07288 * "abc".force_encoding("UTF-8").ascii_only? #=> true 07289 * "abc\u{6666}".force_encoding("UTF-8").ascii_only? #=> false 07290 */ 07291 07292 static VALUE 07293 rb_str_is_ascii_only_p(VALUE str) 07294 { 07295 int cr = rb_enc_str_coderange(str); 07296 07297 return cr == ENC_CODERANGE_7BIT ? Qtrue : Qfalse; 07298 } 07299 07314 VALUE 07315 rb_str_ellipsize(VALUE str, long len) 07316 { 07317 static const char ellipsis[] = "..."; 07318 const long ellipsislen = sizeof(ellipsis) - 1; 07319 rb_encoding *const enc = rb_enc_get(str); 07320 const long blen = RSTRING_LEN(str); 07321 const char *const p = RSTRING_PTR(str), *e = p + blen; 07322 VALUE estr, ret = 0; 07323 07324 if (len < 0) rb_raise(rb_eIndexError, "negative length %ld", len); 07325 if (len * rb_enc_mbminlen(enc) >= blen || 07326 (e = rb_enc_nth(p, e, len, enc)) - p == blen) { 07327 ret = str; 07328 } 07329 else if (len <= ellipsislen || 07330 !(e = rb_enc_step_back(p, e, e, len = ellipsislen, enc))) { 07331 if (rb_enc_asciicompat(enc)) { 07332 ret = rb_str_new_with_class(str, ellipsis, len); 07333 rb_enc_associate(ret, enc); 07334 } 07335 else { 07336 estr = rb_usascii_str_new(ellipsis, len); 07337 ret = rb_str_encode(estr, rb_enc_from_encoding(enc), 0, Qnil); 07338 } 07339 } 07340 else if (ret = rb_str_subseq(str, 0, e - p), rb_enc_asciicompat(enc)) { 07341 rb_str_cat(ret, ellipsis, ellipsislen); 07342 } 07343 else { 07344 estr = rb_str_encode(rb_usascii_str_new(ellipsis, ellipsislen), 07345 rb_enc_from_encoding(enc), 0, Qnil); 07346 rb_str_append(ret, estr); 07347 } 07348 return ret; 07349 } 07350 07351 /********************************************************************** 07352 * Document-class: Symbol 07353 * 07354 * <code>Symbol</code> objects represent names and some strings 07355 * inside the Ruby 07356 * interpreter. They are generated using the <code>:name</code> and 07357 * <code>:"string"</code> literals 07358 * syntax, and by the various <code>to_sym</code> methods. The same 07359 * <code>Symbol</code> object will be created for a given name or string 07360 * for the duration of a program's execution, regardless of the context 07361 * or meaning of that name. Thus if <code>Fred</code> is a constant in 07362 * one context, a method in another, and a class in a third, the 07363 * <code>Symbol</code> <code>:Fred</code> will be the same object in 07364 * all three contexts. 07365 * 07366 * module One 07367 * class Fred 07368 * end 07369 * $f1 = :Fred 07370 * end 07371 * module Two 07372 * Fred = 1 07373 * $f2 = :Fred 07374 * end 07375 * def Fred() 07376 * end 07377 * $f3 = :Fred 07378 * $f1.object_id #=> 2514190 07379 * $f2.object_id #=> 2514190 07380 * $f3.object_id #=> 2514190 07381 * 07382 */ 07383 07384 07385 /* 07386 * call-seq: 07387 * sym == obj -> true or false 07388 * 07389 * Equality---If <i>sym</i> and <i>obj</i> are exactly the same 07390 * symbol, returns <code>true</code>. 07391 */ 07392 07393 static VALUE 07394 sym_equal(VALUE sym1, VALUE sym2) 07395 { 07396 if (sym1 == sym2) return Qtrue; 07397 return Qfalse; 07398 } 07399 07400 07401 static int 07402 sym_printable(const char *s, const char *send, rb_encoding *enc) 07403 { 07404 while (s < send) { 07405 int n; 07406 int c = rb_enc_codepoint_len(s, send, &n, enc); 07407 07408 if (!rb_enc_isprint(c, enc)) return FALSE; 07409 s += n; 07410 } 07411 return TRUE; 07412 } 07413 07414 /* 07415 * call-seq: 07416 * sym.inspect -> string 07417 * 07418 * Returns the representation of <i>sym</i> as a symbol literal. 07419 * 07420 * :fred.inspect #=> ":fred" 07421 */ 07422 07423 static VALUE 07424 sym_inspect(VALUE sym) 07425 { 07426 VALUE str; 07427 ID id = SYM2ID(sym); 07428 rb_encoding *enc; 07429 const char *ptr; 07430 long len; 07431 char *dest; 07432 rb_encoding *resenc = rb_default_internal_encoding(); 07433 07434 if (resenc == NULL) resenc = rb_default_external_encoding(); 07435 sym = rb_id2str(id); 07436 enc = STR_ENC_GET(sym); 07437 ptr = RSTRING_PTR(sym); 07438 len = RSTRING_LEN(sym); 07439 if ((resenc != enc && !rb_str_is_ascii_only_p(sym)) || len != (long)strlen(ptr) || 07440 !rb_enc_symname_p(ptr, enc) || !sym_printable(ptr, ptr + len, enc)) { 07441 str = rb_str_inspect(sym); 07442 len = RSTRING_LEN(str); 07443 rb_str_resize(str, len + 1); 07444 dest = RSTRING_PTR(str); 07445 memmove(dest + 1, dest, len); 07446 dest[0] = ':'; 07447 } 07448 else { 07449 char *dest; 07450 str = rb_enc_str_new(0, len + 1, enc); 07451 dest = RSTRING_PTR(str); 07452 dest[0] = ':'; 07453 memcpy(dest + 1, ptr, len); 07454 } 07455 return str; 07456 } 07457 07458 07459 /* 07460 * call-seq: 07461 * sym.id2name -> string 07462 * sym.to_s -> string 07463 * 07464 * Returns the name or string corresponding to <i>sym</i>. 07465 * 07466 * :fred.id2name #=> "fred" 07467 */ 07468 07469 07470 VALUE 07471 rb_sym_to_s(VALUE sym) 07472 { 07473 ID id = SYM2ID(sym); 07474 07475 return str_new3(rb_cString, rb_id2str(id)); 07476 } 07477 07478 07479 /* 07480 * call-seq: 07481 * sym.to_sym -> sym 07482 * sym.intern -> sym 07483 * 07484 * In general, <code>to_sym</code> returns the <code>Symbol</code> corresponding 07485 * to an object. As <i>sym</i> is already a symbol, <code>self</code> is returned 07486 * in this case. 07487 */ 07488 07489 static VALUE 07490 sym_to_sym(VALUE sym) 07491 { 07492 return sym; 07493 } 07494 07495 static VALUE 07496 sym_call(VALUE args, VALUE sym, int argc, VALUE *argv) 07497 { 07498 VALUE obj; 07499 07500 if (argc < 1) { 07501 rb_raise(rb_eArgError, "no receiver given"); 07502 } 07503 obj = argv[0]; 07504 return rb_funcall_passing_block(obj, (ID)sym, argc - 1, argv + 1); 07505 } 07506 07507 /* 07508 * call-seq: 07509 * sym.to_proc 07510 * 07511 * Returns a _Proc_ object which respond to the given method by _sym_. 07512 * 07513 * (1..3).collect(&:to_s) #=> ["1", "2", "3"] 07514 */ 07515 07516 static VALUE 07517 sym_to_proc(VALUE sym) 07518 { 07519 static VALUE sym_proc_cache = Qfalse; 07520 enum {SYM_PROC_CACHE_SIZE = 67}; 07521 VALUE proc; 07522 long id, index; 07523 VALUE *aryp; 07524 07525 if (!sym_proc_cache) { 07526 sym_proc_cache = rb_ary_tmp_new(SYM_PROC_CACHE_SIZE * 2); 07527 rb_gc_register_mark_object(sym_proc_cache); 07528 rb_ary_store(sym_proc_cache, SYM_PROC_CACHE_SIZE*2 - 1, Qnil); 07529 } 07530 07531 id = SYM2ID(sym); 07532 index = (id % SYM_PROC_CACHE_SIZE) << 1; 07533 07534 aryp = RARRAY_PTR(sym_proc_cache); 07535 if (aryp[index] == sym) { 07536 return aryp[index + 1]; 07537 } 07538 else { 07539 proc = rb_proc_new(sym_call, (VALUE)id); 07540 aryp[index] = sym; 07541 aryp[index + 1] = proc; 07542 return proc; 07543 } 07544 } 07545 07546 /* 07547 * call-seq: 07548 * 07549 * sym.succ 07550 * 07551 * Same as <code>sym.to_s.succ.intern</code>. 07552 */ 07553 07554 static VALUE 07555 sym_succ(VALUE sym) 07556 { 07557 return rb_str_intern(rb_str_succ(rb_sym_to_s(sym))); 07558 } 07559 07560 /* 07561 * call-seq: 07562 * 07563 * str <=> other -> -1, 0, +1 or nil 07564 * 07565 * Compares _sym_ with _other_ in string form. 07566 */ 07567 07568 static VALUE 07569 sym_cmp(VALUE sym, VALUE other) 07570 { 07571 if (!SYMBOL_P(other)) { 07572 return Qnil; 07573 } 07574 return rb_str_cmp_m(rb_sym_to_s(sym), rb_sym_to_s(other)); 07575 } 07576 07577 /* 07578 * call-seq: 07579 * 07580 * sym.casecmp(other) -> -1, 0, +1 or nil 07581 * 07582 * Case-insensitive version of <code>Symbol#<=></code>. 07583 */ 07584 07585 static VALUE 07586 sym_casecmp(VALUE sym, VALUE other) 07587 { 07588 if (!SYMBOL_P(other)) { 07589 return Qnil; 07590 } 07591 return rb_str_casecmp(rb_sym_to_s(sym), rb_sym_to_s(other)); 07592 } 07593 07594 /* 07595 * call-seq: 07596 * sym =~ obj -> fixnum or nil 07597 * 07598 * Returns <code>sym.to_s =~ obj</code>. 07599 */ 07600 07601 static VALUE 07602 sym_match(VALUE sym, VALUE other) 07603 { 07604 return rb_str_match(rb_sym_to_s(sym), other); 07605 } 07606 07607 /* 07608 * call-seq: 07609 * sym[idx] -> char 07610 * sym[b, n] -> char 07611 * 07612 * Returns <code>sym.to_s[]</code>. 07613 */ 07614 07615 static VALUE 07616 sym_aref(int argc, VALUE *argv, VALUE sym) 07617 { 07618 return rb_str_aref_m(argc, argv, rb_sym_to_s(sym)); 07619 } 07620 07621 /* 07622 * call-seq: 07623 * sym.length -> integer 07624 * 07625 * Same as <code>sym.to_s.length</code>. 07626 */ 07627 07628 static VALUE 07629 sym_length(VALUE sym) 07630 { 07631 return rb_str_length(rb_id2str(SYM2ID(sym))); 07632 } 07633 07634 /* 07635 * call-seq: 07636 * sym.empty? -> true or false 07637 * 07638 * Returns that _sym_ is :"" or not. 07639 */ 07640 07641 static VALUE 07642 sym_empty(VALUE sym) 07643 { 07644 return rb_str_empty(rb_id2str(SYM2ID(sym))); 07645 } 07646 07647 /* 07648 * call-seq: 07649 * sym.upcase -> symbol 07650 * 07651 * Same as <code>sym.to_s.upcase.intern</code>. 07652 */ 07653 07654 static VALUE 07655 sym_upcase(VALUE sym) 07656 { 07657 return rb_str_intern(rb_str_upcase(rb_id2str(SYM2ID(sym)))); 07658 } 07659 07660 /* 07661 * call-seq: 07662 * sym.downcase -> symbol 07663 * 07664 * Same as <code>sym.to_s.downcase.intern</code>. 07665 */ 07666 07667 static VALUE 07668 sym_downcase(VALUE sym) 07669 { 07670 return rb_str_intern(rb_str_downcase(rb_id2str(SYM2ID(sym)))); 07671 } 07672 07673 /* 07674 * call-seq: 07675 * sym.capitalize -> symbol 07676 * 07677 * Same as <code>sym.to_s.capitalize.intern</code>. 07678 */ 07679 07680 static VALUE 07681 sym_capitalize(VALUE sym) 07682 { 07683 return rb_str_intern(rb_str_capitalize(rb_id2str(SYM2ID(sym)))); 07684 } 07685 07686 /* 07687 * call-seq: 07688 * sym.swapcase -> symbol 07689 * 07690 * Same as <code>sym.to_s.swapcase.intern</code>. 07691 */ 07692 07693 static VALUE 07694 sym_swapcase(VALUE sym) 07695 { 07696 return rb_str_intern(rb_str_swapcase(rb_id2str(SYM2ID(sym)))); 07697 } 07698 07699 /* 07700 * call-seq: 07701 * sym.encoding -> encoding 07702 * 07703 * Returns the Encoding object that represents the encoding of _sym_. 07704 */ 07705 07706 static VALUE 07707 sym_encoding(VALUE sym) 07708 { 07709 return rb_obj_encoding(rb_id2str(SYM2ID(sym))); 07710 } 07711 07712 ID 07713 rb_to_id(VALUE name) 07714 { 07715 VALUE tmp; 07716 07717 switch (TYPE(name)) { 07718 default: 07719 tmp = rb_check_string_type(name); 07720 if (NIL_P(tmp)) { 07721 tmp = rb_inspect(name); 07722 rb_raise(rb_eTypeError, "%s is not a symbol", 07723 RSTRING_PTR(tmp)); 07724 } 07725 name = tmp; 07726 /* fall through */ 07727 case T_STRING: 07728 name = rb_str_intern(name); 07729 /* fall through */ 07730 case T_SYMBOL: 07731 return SYM2ID(name); 07732 } 07733 return Qnil; /* not reached */ 07734 } 07735 07736 /* 07737 * A <code>String</code> object holds and manipulates an arbitrary sequence of 07738 * bytes, typically representing characters. String objects may be created 07739 * using <code>String::new</code> or as literals. 07740 * 07741 * Because of aliasing issues, users of strings should be aware of the methods 07742 * that modify the contents of a <code>String</code> object. Typically, 07743 * methods with names ending in ``!'' modify their receiver, while those 07744 * without a ``!'' return a new <code>String</code>. However, there are 07745 * exceptions, such as <code>String#[]=</code>. 07746 * 07747 */ 07748 07749 void 07750 Init_String(void) 07751 { 07752 #undef rb_intern 07753 #define rb_intern(str) rb_intern_const(str) 07754 07755 rb_cString = rb_define_class("String", rb_cObject); 07756 rb_include_module(rb_cString, rb_mComparable); 07757 rb_define_alloc_func(rb_cString, str_alloc); 07758 rb_define_singleton_method(rb_cString, "try_convert", rb_str_s_try_convert, 1); 07759 rb_define_method(rb_cString, "initialize", rb_str_init, -1); 07760 rb_define_method(rb_cString, "initialize_copy", rb_str_replace, 1); 07761 rb_define_method(rb_cString, "<=>", rb_str_cmp_m, 1); 07762 rb_define_method(rb_cString, "==", rb_str_equal, 1); 07763 rb_define_method(rb_cString, "===", rb_str_equal, 1); 07764 rb_define_method(rb_cString, "eql?", rb_str_eql, 1); 07765 rb_define_method(rb_cString, "hash", rb_str_hash_m, 0); 07766 rb_define_method(rb_cString, "casecmp", rb_str_casecmp, 1); 07767 rb_define_method(rb_cString, "+", rb_str_plus, 1); 07768 rb_define_method(rb_cString, "*", rb_str_times, 1); 07769 rb_define_method(rb_cString, "%", rb_str_format_m, 1); 07770 rb_define_method(rb_cString, "[]", rb_str_aref_m, -1); 07771 rb_define_method(rb_cString, "[]=", rb_str_aset_m, -1); 07772 rb_define_method(rb_cString, "insert", rb_str_insert, 2); 07773 rb_define_method(rb_cString, "length", rb_str_length, 0); 07774 rb_define_method(rb_cString, "size", rb_str_length, 0); 07775 rb_define_method(rb_cString, "bytesize", rb_str_bytesize, 0); 07776 rb_define_method(rb_cString, "empty?", rb_str_empty, 0); 07777 rb_define_method(rb_cString, "=~", rb_str_match, 1); 07778 rb_define_method(rb_cString, "match", rb_str_match_m, -1); 07779 rb_define_method(rb_cString, "succ", rb_str_succ, 0); 07780 rb_define_method(rb_cString, "succ!", rb_str_succ_bang, 0); 07781 rb_define_method(rb_cString, "next", rb_str_succ, 0); 07782 rb_define_method(rb_cString, "next!", rb_str_succ_bang, 0); 07783 rb_define_method(rb_cString, "upto", rb_str_upto, -1); 07784 rb_define_method(rb_cString, "index", rb_str_index_m, -1); 07785 rb_define_method(rb_cString, "rindex", rb_str_rindex_m, -1); 07786 rb_define_method(rb_cString, "replace", rb_str_replace, 1); 07787 rb_define_method(rb_cString, "clear", rb_str_clear, 0); 07788 rb_define_method(rb_cString, "chr", rb_str_chr, 0); 07789 rb_define_method(rb_cString, "getbyte", rb_str_getbyte, 1); 07790 rb_define_method(rb_cString, "setbyte", rb_str_setbyte, 2); 07791 rb_define_method(rb_cString, "byteslice", rb_str_byteslice, -1); 07792 07793 rb_define_method(rb_cString, "to_i", rb_str_to_i, -1); 07794 rb_define_method(rb_cString, "to_f", rb_str_to_f, 0); 07795 rb_define_method(rb_cString, "to_s", rb_str_to_s, 0); 07796 rb_define_method(rb_cString, "to_str", rb_str_to_s, 0); 07797 rb_define_method(rb_cString, "inspect", rb_str_inspect, 0); 07798 rb_define_method(rb_cString, "dump", rb_str_dump, 0); 07799 07800 rb_define_method(rb_cString, "upcase", rb_str_upcase, 0); 07801 rb_define_method(rb_cString, "downcase", rb_str_downcase, 0); 07802 rb_define_method(rb_cString, "capitalize", rb_str_capitalize, 0); 07803 rb_define_method(rb_cString, "swapcase", rb_str_swapcase, 0); 07804 07805 rb_define_method(rb_cString, "upcase!", rb_str_upcase_bang, 0); 07806 rb_define_method(rb_cString, "downcase!", rb_str_downcase_bang, 0); 07807 rb_define_method(rb_cString, "capitalize!", rb_str_capitalize_bang, 0); 07808 rb_define_method(rb_cString, "swapcase!", rb_str_swapcase_bang, 0); 07809 07810 rb_define_method(rb_cString, "hex", rb_str_hex, 0); 07811 rb_define_method(rb_cString, "oct", rb_str_oct, 0); 07812 rb_define_method(rb_cString, "split", rb_str_split_m, -1); 07813 rb_define_method(rb_cString, "lines", rb_str_each_line, -1); 07814 rb_define_method(rb_cString, "bytes", rb_str_each_byte, 0); 07815 rb_define_method(rb_cString, "chars", rb_str_each_char, 0); 07816 rb_define_method(rb_cString, "codepoints", rb_str_each_codepoint, 0); 07817 rb_define_method(rb_cString, "reverse", rb_str_reverse, 0); 07818 rb_define_method(rb_cString, "reverse!", rb_str_reverse_bang, 0); 07819 rb_define_method(rb_cString, "concat", rb_str_concat, 1); 07820 rb_define_method(rb_cString, "<<", rb_str_concat, 1); 07821 rb_define_method(rb_cString, "prepend", rb_str_prepend, 1); 07822 rb_define_method(rb_cString, "crypt", rb_str_crypt, 1); 07823 rb_define_method(rb_cString, "intern", rb_str_intern, 0); 07824 rb_define_method(rb_cString, "to_sym", rb_str_intern, 0); 07825 rb_define_method(rb_cString, "ord", rb_str_ord, 0); 07826 07827 rb_define_method(rb_cString, "include?", rb_str_include, 1); 07828 rb_define_method(rb_cString, "start_with?", rb_str_start_with, -1); 07829 rb_define_method(rb_cString, "end_with?", rb_str_end_with, -1); 07830 07831 rb_define_method(rb_cString, "scan", rb_str_scan, 1); 07832 07833 rb_define_method(rb_cString, "ljust", rb_str_ljust, -1); 07834 rb_define_method(rb_cString, "rjust", rb_str_rjust, -1); 07835 rb_define_method(rb_cString, "center", rb_str_center, -1); 07836 07837 rb_define_method(rb_cString, "sub", rb_str_sub, -1); 07838 rb_define_method(rb_cString, "gsub", rb_str_gsub, -1); 07839 rb_define_method(rb_cString, "chop", rb_str_chop, 0); 07840 rb_define_method(rb_cString, "chomp", rb_str_chomp, -1); 07841 rb_define_method(rb_cString, "strip", rb_str_strip, 0); 07842 rb_define_method(rb_cString, "lstrip", rb_str_lstrip, 0); 07843 rb_define_method(rb_cString, "rstrip", rb_str_rstrip, 0); 07844 07845 rb_define_method(rb_cString, "sub!", rb_str_sub_bang, -1); 07846 rb_define_method(rb_cString, "gsub!", rb_str_gsub_bang, -1); 07847 rb_define_method(rb_cString, "chop!", rb_str_chop_bang, 0); 07848 rb_define_method(rb_cString, "chomp!", rb_str_chomp_bang, -1); 07849 rb_define_method(rb_cString, "strip!", rb_str_strip_bang, 0); 07850 rb_define_method(rb_cString, "lstrip!", rb_str_lstrip_bang, 0); 07851 rb_define_method(rb_cString, "rstrip!", rb_str_rstrip_bang, 0); 07852 07853 rb_define_method(rb_cString, "tr", rb_str_tr, 2); 07854 rb_define_method(rb_cString, "tr_s", rb_str_tr_s, 2); 07855 rb_define_method(rb_cString, "delete", rb_str_delete, -1); 07856 rb_define_method(rb_cString, "squeeze", rb_str_squeeze, -1); 07857 rb_define_method(rb_cString, "count", rb_str_count, -1); 07858 07859 rb_define_method(rb_cString, "tr!", rb_str_tr_bang, 2); 07860 rb_define_method(rb_cString, "tr_s!", rb_str_tr_s_bang, 2); 07861 rb_define_method(rb_cString, "delete!", rb_str_delete_bang, -1); 07862 rb_define_method(rb_cString, "squeeze!", rb_str_squeeze_bang, -1); 07863 07864 rb_define_method(rb_cString, "each_line", rb_str_each_line, -1); 07865 rb_define_method(rb_cString, "each_byte", rb_str_each_byte, 0); 07866 rb_define_method(rb_cString, "each_char", rb_str_each_char, 0); 07867 rb_define_method(rb_cString, "each_codepoint", rb_str_each_codepoint, 0); 07868 07869 rb_define_method(rb_cString, "sum", rb_str_sum, -1); 07870 07871 rb_define_method(rb_cString, "slice", rb_str_aref_m, -1); 07872 rb_define_method(rb_cString, "slice!", rb_str_slice_bang, -1); 07873 07874 rb_define_method(rb_cString, "partition", rb_str_partition, 1); 07875 rb_define_method(rb_cString, "rpartition", rb_str_rpartition, 1); 07876 07877 rb_define_method(rb_cString, "encoding", rb_obj_encoding, 0); /* in encoding.c */ 07878 rb_define_method(rb_cString, "force_encoding", rb_str_force_encoding, 1); 07879 rb_define_method(rb_cString, "valid_encoding?", rb_str_valid_encoding_p, 0); 07880 rb_define_method(rb_cString, "ascii_only?", rb_str_is_ascii_only_p, 0); 07881 07882 id_to_s = rb_intern("to_s"); 07883 07884 rb_fs = Qnil; 07885 rb_define_variable("$;", &rb_fs); 07886 rb_define_variable("$-F", &rb_fs); 07887 07888 rb_cSymbol = rb_define_class("Symbol", rb_cObject); 07889 rb_include_module(rb_cSymbol, rb_mComparable); 07890 rb_undef_alloc_func(rb_cSymbol); 07891 rb_undef_method(CLASS_OF(rb_cSymbol), "new"); 07892 rb_define_singleton_method(rb_cSymbol, "all_symbols", rb_sym_all_symbols, 0); /* in parse.y */ 07893 07894 rb_define_method(rb_cSymbol, "==", sym_equal, 1); 07895 rb_define_method(rb_cSymbol, "===", sym_equal, 1); 07896 rb_define_method(rb_cSymbol, "inspect", sym_inspect, 0); 07897 rb_define_method(rb_cSymbol, "to_s", rb_sym_to_s, 0); 07898 rb_define_method(rb_cSymbol, "id2name", rb_sym_to_s, 0); 07899 rb_define_method(rb_cSymbol, "intern", sym_to_sym, 0); 07900 rb_define_method(rb_cSymbol, "to_sym", sym_to_sym, 0); 07901 rb_define_method(rb_cSymbol, "to_proc", sym_to_proc, 0); 07902 rb_define_method(rb_cSymbol, "succ", sym_succ, 0); 07903 rb_define_method(rb_cSymbol, "next", sym_succ, 0); 07904 07905 rb_define_method(rb_cSymbol, "<=>", sym_cmp, 1); 07906 rb_define_method(rb_cSymbol, "casecmp", sym_casecmp, 1); 07907 rb_define_method(rb_cSymbol, "=~", sym_match, 1); 07908 07909 rb_define_method(rb_cSymbol, "[]", sym_aref, -1); 07910 rb_define_method(rb_cSymbol, "slice", sym_aref, -1); 07911 rb_define_method(rb_cSymbol, "length", sym_length, 0); 07912 rb_define_method(rb_cSymbol, "size", sym_length, 0); 07913 rb_define_method(rb_cSymbol, "empty?", sym_empty, 0); 07914 rb_define_method(rb_cSymbol, "match", sym_match, 1); 07915 07916 rb_define_method(rb_cSymbol, "upcase", sym_upcase, 0); 07917 rb_define_method(rb_cSymbol, "downcase", sym_downcase, 0); 07918 rb_define_method(rb_cSymbol, "capitalize", sym_capitalize, 0); 07919 rb_define_method(rb_cSymbol, "swapcase", sym_swapcase, 0); 07920 07921 rb_define_method(rb_cSymbol, "encoding", sym_encoding, 0); 07922 } 07923
1.7.6.1