| @@ -47,8 +47,13 @@ | ||
| 47 | 47 | * - mb_strripos - Finds position of last occurrence of a string within another, case insensitive |
| 48 | 48 | * - mb_strstr - Finds first occurrence of a string within another |
| 49 | 49 | * - mb_strwidth - Return width of string |
| 50 | 50 | * - mb_substr_count - Count the number of substring occurrences |
| 51 | + * - mb_ucfirst - Make a string's first character uppercase | |
| 52 | + * - mb_lcfirst - Make a string's first character lowercase | |
| 53 | + * - mb_trim - Strip whitespace (or other characters) from the beginning and end of a string | |
| 54 | + * - mb_ltrim - Strip whitespace (or other characters) from the beginning of a string | |
| 55 | + * - mb_rtrim - Strip whitespace (or other characters) from the end of a string | |
| 51 | 56 | * |
| 52 | 57 | * Not implemented: |
| 53 | 58 | * - mb_convert_kana - Convert "kana" one from another ("zen-kaku", "han-kaku" and more) |
| 54 | 59 | * - mb_ereg_* - Regular expression with multibyte support |
| @@ -68,9 +73,9 @@ | ||
| 68 | 73 | final class Mbstring |
| 69 | 74 | { |
| 70 | 75 | public const MB_CASE_FOLD = \PHP_INT_MAX; |
| 71 | 76 | |
| 72 | - private const CASE_FOLD = [ | |
| 77 | + private const SIMPLE_CASE_FOLD = [ | |
| 73 | 78 | ['µ', 'ſ', "\xCD\x85", 'ς', "\xCF\x90", "\xCF\x91", "\xCF\x95", "\xCF\x96", "\xCF\xB0", "\xCF\xB1", "\xCF\xB5", "\xE1\xBA\x9B", "\xE1\xBE\xBE"], |
| 74 | 79 | ['μ', 's', 'ι', 'σ', 'β', 'θ', 'φ', 'π', 'κ', 'ρ', 'ε', "\xE1\xB9\xA1", 'ι'], |
| 75 | 80 | ]; |
| 76 | 81 | |
| @@ -76,11 +81,21 @@ | ||
| 76 | 81 | |
| 77 | 82 | private static $encodingList = ['ASCII', 'UTF-8']; |
| 78 | 83 | private static $language = 'neutral'; |
| 79 | 84 | private static $internalEncoding = 'UTF-8'; |
| 85 | + private static $iconvSupportsIgnore; | |
| 80 | 86 | |
| 81 | 87 | public static function mb_convert_encoding($s, $toEncoding, $fromEncoding = null) |
| 82 | 88 | { |
| 89 | + if (\is_array($s)) { | |
| 90 | + $r = []; | |
| 91 | + foreach ($s as $str) { | |
| 92 | + $r[] = self::mb_convert_encoding($str, $toEncoding, $fromEncoding); | |
| 93 | + } | |
| 94 | + | |
| 95 | + return $r; | |
| 96 | + } | |
| 97 | + | |
| 83 | 98 | if (\is_array($fromEncoding) || (null !== $fromEncoding && false !== strpos($fromEncoding, ','))) { |
| 84 | 99 | $fromEncoding = self::mb_detect_encoding($s, $fromEncoding); |
| 85 | 100 | } else { |
| 86 | 101 | $fromEncoding = self::getEncoding($fromEncoding); |
| @@ -101,9 +116,9 @@ | ||
| 101 | 116 | if ('HTML-ENTITIES' === $fromEncoding || 'HTML' === $fromEncoding) { |
| 102 | 117 | $fromEncoding = 'Windows-1252'; |
| 103 | 118 | } |
| 104 | 119 | if ('UTF-8' !== $fromEncoding) { |
| 105 | - $s = iconv($fromEncoding, 'UTF-8//IGNORE', $s); | |
| 120 | + $s = self::iconv($fromEncoding, 'UTF-8', $s); | |
| 106 | 121 | } |
| 107 | 122 | |
| 108 | 123 | return preg_replace_callback('/[\x80-\xFF]+/', [__CLASS__, 'html_encoding_callback'], $s); |
| 109 | 124 | } |
| @@ -108,19 +123,48 @@ | ||
| 108 | 123 | return preg_replace_callback('/[\x80-\xFF]+/', [__CLASS__, 'html_encoding_callback'], $s); |
| 109 | 124 | } |
| 110 | 125 | |
| 111 | 126 | if ('HTML-ENTITIES' === $fromEncoding) { |
| 112 | - $s = html_entity_decode($s, \ENT_COMPAT, 'UTF-8'); | |
| 127 | + $decodeControlChars = static function ($m) { | |
| 128 | + $code = '' !== ($m[2] ?? '') ? hexdec($m[2]) : (int) $m[1]; | |
| 129 | + | |
| 130 | + if ($code < 32 || 127 === $code) { | |
| 131 | + return \chr($code); | |
| 132 | + } | |
| 133 | + if (128 <= $code && $code <= 159) { | |
| 134 | + return "\xC2".\chr(0x80 | ($code & 0x3F)); | |
| 135 | + } | |
| 136 | + | |
| 137 | + return $m[0]; | |
| 138 | + }; | |
| 139 | + | |
| 140 | + if (\PHP_VERSION_ID >= 70400) { | |
| 141 | + $s = html_entity_decode($s, \ENT_QUOTES, 'UTF-8'); | |
| 142 | + // html_entity_decode() leaves numeric entities for C0/C1 control | |
| 143 | + // characters as-is (HTML spec), but mb_convert_encoding() decodes | |
| 144 | + // them. Catch what html_entity_decode() missed. | |
| 145 | + if (false !== strpos($s, '&#')) { | |
| 146 | + $s = preg_replace_callback('/&#(?:0*([0-9]++)|[xX]0*([0-9a-fA-F]++));/', $decodeControlChars, $s); | |
| 147 | + } | |
| 148 | + } else { | |
| 149 | + // PHP < 7.4: html_entity_decode() truncates strings at NUL bytes, | |
| 150 | + // so decode the control character entities first then call | |
| 151 | + // html_entity_decode() on each NUL-delimited chunk independently. | |
| 152 | + $s = preg_replace_callback('/&#(?:0*([0-9]++)|[xX]0*([0-9a-fA-F]++));/', $decodeControlChars, $s); | |
| 153 | + $s = implode("\0", array_map(static function ($chunk) { | |
| 154 | + return html_entity_decode($chunk, \ENT_QUOTES, 'UTF-8'); | |
| 155 | + }, explode("\0", $s))); | |
| 156 | + } | |
| 113 | 157 | $fromEncoding = 'UTF-8'; |
| 114 | 158 | } |
| 115 | 159 | |
| 116 | - return iconv($fromEncoding, $toEncoding.'//IGNORE', $s); | |
| 160 | + return self::iconv($fromEncoding, $toEncoding, $s); | |
| 117 | 161 | } |
| 118 | 162 | |
| 119 | 163 | public static function mb_convert_variables($toEncoding, $fromEncoding, &...$vars) |
| 120 | 164 | { |
| 121 | 165 | $ok = true; |
| 122 | - array_walk_recursive($vars, function (&$v) use (&$ok, $toEncoding, $fromEncoding) { | |
| 166 | + array_walk_recursive($vars, static function (&$v) use (&$ok, $toEncoding, $fromEncoding) { | |
| 123 | 167 | if (false === $v = self::mb_convert_encoding($v, $toEncoding, $fromEncoding)) { |
| 124 | 168 | $ok = false; |
| 125 | 169 | } |
| 126 | 170 | }); |
| @@ -165,12 +209,12 @@ | ||
| 165 | 209 | |
| 166 | 210 | if ('UTF-8' === $encoding) { |
| 167 | 211 | $encoding = null; |
| 168 | 212 | if (!preg_match('//u', $s)) { |
| 169 | - $s = @iconv('UTF-8', 'UTF-8//IGNORE', $s); | |
| 213 | + $s = @self::iconv('UTF-8', 'UTF-8', $s); | |
| 170 | 214 | } |
| 171 | 215 | } else { |
| 172 | - $s = iconv($encoding, 'UTF-8//IGNORE', $s); | |
| 216 | + $s = self::iconv($encoding, 'UTF-8', $s); | |
| 173 | 217 | } |
| 174 | 218 | |
| 175 | 219 | $cnt = floor(\count($convmap) / 4) * 4; |
| 176 | 220 | |
| @@ -179,9 +223,9 @@ | ||
| 179 | 223 | $convmap[$i] += $convmap[$i + 2]; |
| 180 | 224 | $convmap[$i + 1] += $convmap[$i + 2]; |
| 181 | 225 | } |
| 182 | 226 | |
| 183 | - $s = preg_replace_callback('/&#(?:0*([0-9]+)|x0*([0-9a-fA-F]+))(?!&);?/', function (array $m) use ($cnt, $convmap) { | |
| 227 | + $s = preg_replace_callback('/&#(?:0*([0-9]+)|x0*([0-9a-fA-F]+))'.(\PHP_VERSION_ID >= 80200 ? '' : '(?!&)').';?/', static function (array $m) use ($cnt, $convmap) { | |
| 184 | 228 | $c = isset($m[2]) ? (int) hexdec($m[2]) : $m[1]; |
| 185 | 229 | for ($i = 0; $i < $cnt; $i += 4) { |
| 186 | 230 | if ($c >= $convmap[$i] && $c <= $convmap[$i + 1]) { |
| 187 | 231 | return self::mb_chr($c - $convmap[$i + 2]); |
| @@ -194,9 +238,9 @@ | ||
| 194 | 238 | if (null === $encoding) { |
| 195 | 239 | return $s; |
| 196 | 240 | } |
| 197 | 241 | |
| 198 | - return iconv('UTF-8', $encoding.'//IGNORE', $s); | |
| 242 | + return self::iconv('UTF-8', $encoding, $s); | |
| 199 | 243 | } |
| 200 | 244 | |
| 201 | 245 | public static function mb_encode_numericentity($s, $convmap, $encoding = null, $is_hex = false) |
| 202 | 246 | { |
| @@ -231,12 +275,12 @@ | ||
| 231 | 275 | |
| 232 | 276 | if ('UTF-8' === $encoding) { |
| 233 | 277 | $encoding = null; |
| 234 | 278 | if (!preg_match('//u', $s)) { |
| 235 | - $s = @iconv('UTF-8', 'UTF-8//IGNORE', $s); | |
| 279 | + $s = @self::iconv('UTF-8', 'UTF-8', $s); | |
| 236 | 280 | } |
| 237 | 281 | } else { |
| 238 | - $s = iconv($encoding, 'UTF-8//IGNORE', $s); | |
| 282 | + $s = self::iconv($encoding, 'UTF-8', $s); | |
| 239 | 283 | } |
| 240 | 284 | |
| 241 | 285 | static $ulenMask = ["\xC0" => 2, "\xD0" => 2, "\xE0" => 3, "\xF0" => 4]; |
| 242 | 286 | |
| @@ -253,9 +297,9 @@ | ||
| 253 | 297 | |
| 254 | 298 | for ($j = 0; $j < $cnt; $j += 4) { |
| 255 | 299 | if ($c >= $convmap[$j] && $c <= $convmap[$j + 1]) { |
| 256 | 300 | $cOffset = ($c + $convmap[$j + 2]) & $convmap[$j + 3]; |
| 257 | - $result .= $is_hex ? sprintf('&#x%X;', $cOffset) : '&#'.$cOffset.';'; | |
| 301 | + $result .= $is_hex ? \sprintf('&#x%X;', $cOffset) : '&#'.$cOffset.';'; | |
| 258 | 302 | continue 2; |
| 259 | 303 | } |
| 260 | 304 | } |
| 261 | 305 | $result .= $uchr; |
| @@ -264,9 +308,9 @@ | ||
| 264 | 308 | if (null === $encoding) { |
| 265 | 309 | return $result; |
| 266 | 310 | } |
| 267 | 311 | |
| 268 | - return iconv('UTF-8', $encoding.'//IGNORE', $result); | |
| 312 | + return self::iconv('UTF-8', $encoding, $result); | |
| 269 | 313 | } |
| 270 | 314 | |
| 271 | 315 | public static function mb_convert_case($s, $mode, $encoding = null) |
| 272 | 316 | { |
| @@ -279,12 +323,12 @@ | ||
| 279 | 323 | |
| 280 | 324 | if ('UTF-8' === $encoding) { |
| 281 | 325 | $encoding = null; |
| 282 | 326 | if (!preg_match('//u', $s)) { |
| 283 | - $s = @iconv('UTF-8', 'UTF-8//IGNORE', $s); | |
| 327 | + $s = @self::iconv('UTF-8', 'UTF-8', $s); | |
| 284 | 328 | } |
| 285 | 329 | } else { |
| 286 | - $s = iconv($encoding, 'UTF-8//IGNORE', $s); | |
| 330 | + $s = self::iconv($encoding, 'UTF-8', $s); | |
| 287 | 331 | } |
| 288 | 332 | |
| 289 | 333 | if (\MB_CASE_TITLE == $mode) { |
| 290 | 334 | static $titleRegexp = null; |
| @@ -300,9 +344,13 @@ | ||
| 300 | 344 | } |
| 301 | 345 | $map = $upper; |
| 302 | 346 | } else { |
| 303 | 347 | if (self::MB_CASE_FOLD === $mode) { |
| 304 | - $s = str_replace(self::CASE_FOLD[0], self::CASE_FOLD[1], $s); | |
| 348 | + static $caseFolding = null; | |
| 349 | + if (null === $caseFolding) { | |
| 350 | + $caseFolding = self::getData('caseFolding'); | |
| 351 | + } | |
| 352 | + $s = strtr($s, $caseFolding); | |
| 305 | 353 | } |
| 306 | 354 | |
| 307 | 355 | static $lower = null; |
| 308 | 356 | if (null === $lower) { |
| @@ -342,9 +390,9 @@ | ||
| 342 | 390 | if (null === $encoding) { |
| 343 | 391 | return $s; |
| 344 | 392 | } |
| 345 | 393 | |
| 346 | - return iconv('UTF-8', $encoding.'//IGNORE', $s); | |
| 394 | + return self::iconv('UTF-8', $encoding, $s); | |
| 347 | 395 | } |
| 348 | 396 | |
| 349 | 397 | public static function mb_internal_encoding($encoding = null) |
| 350 | 398 | { |
| @@ -363,9 +411,9 @@ | ||
| 363 | 411 | if (80000 > \PHP_VERSION_ID) { |
| 364 | 412 | return false; |
| 365 | 413 | } |
| 366 | 414 | |
| 367 | - throw new \ValueError(sprintf('Argument #1 ($encoding) must be a valid encoding, "%s" given', $encoding)); | |
| 415 | + throw new \ValueError(\sprintf('Argument #1 ($encoding) must be a valid encoding, "%s" given', $encoding)); | |
| 368 | 416 | } |
| 369 | 417 | |
| 370 | 418 | public static function mb_language($lang = null) |
| 371 | 419 | { |
| @@ -384,9 +432,9 @@ | ||
| 384 | 432 | if (80000 > \PHP_VERSION_ID) { |
| 385 | 433 | return false; |
| 386 | 434 | } |
| 387 | 435 | |
| 388 | - throw new \ValueError(sprintf('Argument #1 ($language) must be a valid language, "%s" given', $lang)); | |
| 436 | + throw new \ValueError(\sprintf('Argument #1 ($language) must be a valid language, "%s" given', $lang)); | |
| 389 | 437 | } |
| 390 | 438 | |
| 391 | 439 | public static function mb_list_encodings() |
| 392 | 440 | { |
| @@ -412,9 +460,22 @@ | ||
| 412 | 460 | } |
| 413 | 461 | $encoding = self::$internalEncoding; |
| 414 | 462 | } |
| 415 | 463 | |
| 416 | - return self::mb_detect_encoding($var, [$encoding]) || false !== @iconv($encoding, $encoding, $var); | |
| 464 | + if (!\is_array($var)) { | |
| 465 | + return self::mb_detect_encoding($var, [$encoding]) || false !== @iconv($encoding, $encoding, $var); | |
| 466 | + } | |
| 467 | + | |
| 468 | + foreach ($var as $key => $value) { | |
| 469 | + if (!self::mb_check_encoding($key, $encoding)) { | |
| 470 | + return false; | |
| 471 | + } | |
| 472 | + if (!self::mb_check_encoding($value, $encoding)) { | |
| 473 | + return false; | |
| 474 | + } | |
| 475 | + } | |
| 476 | + | |
| 477 | + return true; | |
| 417 | 478 | } |
| 418 | 479 | |
| 419 | 480 | public static function mb_detect_encoding($str, $encodingList = null, $strict = false) |
| 420 | 481 | { |
| @@ -487,9 +548,17 @@ | ||
| 487 | 548 | if ('CP850' === $encoding || 'ASCII' === $encoding) { |
| 488 | 549 | return \strlen($s); |
| 489 | 550 | } |
| 490 | 551 | |
| 491 | - return @iconv_strlen($s, $encoding); | |
| 552 | + if (false !== $len = @iconv_strlen($s, $encoding)) { | |
| 553 | + return $len; | |
| 554 | + } | |
| 555 | + | |
| 556 | + if ('UTF-8' !== $encoding) { | |
| 557 | + return $len; | |
| 558 | + } | |
| 559 | + | |
| 560 | + return preg_match_all('/[\x00-\x7F]|[\xC0-\xDF][\x80-\xBF]?|[\xE0-\xEF][\x80-\xBF]{0,2}|[\xF0-\xF7][\x80-\xBF]{0,3}|[\xF8-\xFB][\x80-\xBF]{0,4}|[\xFC-\xFD][\x80-\xBF]{0,5}|[\x80-\xBF\xFE\xFF]/s', $s); | |
| 492 | 561 | } |
| 493 | 562 | |
| 494 | 563 | public static function mb_strpos($haystack, $needle, $offset = 0, $encoding = null) |
| 495 | 564 | { |
| @@ -637,10 +706,12 @@ | ||
| 637 | 706 | } |
| 638 | 707 | |
| 639 | 708 | public static function mb_stripos($haystack, $needle, $offset = 0, $encoding = null) |
| 640 | 709 | { |
| 641 | - $haystack = self::mb_convert_case($haystack, self::MB_CASE_FOLD, $encoding); | |
| 642 | - $needle = self::mb_convert_case($needle, self::MB_CASE_FOLD, $encoding); | |
| 710 | + [$haystack, $needle] = str_replace(self::SIMPLE_CASE_FOLD[0], self::SIMPLE_CASE_FOLD[1], [ | |
| 711 | + self::mb_convert_case($haystack, \MB_CASE_LOWER, $encoding), | |
| 712 | + self::mb_convert_case($needle, \MB_CASE_LOWER, $encoding), | |
| 713 | + ]); | |
| 643 | 714 | |
| 644 | 715 | return self::mb_strpos($haystack, $needle, $offset, $encoding); |
| 645 | 716 | } |
| 646 | 717 | |
| @@ -673,11 +744,14 @@ | ||
| 673 | 744 | } |
| 674 | 745 | |
| 675 | 746 | public static function mb_strripos($haystack, $needle, $offset = 0, $encoding = null) |
| 676 | 747 | { |
| 677 | - $haystack = self::mb_convert_case($haystack, self::MB_CASE_FOLD, $encoding); | |
| 678 | - $needle = self::mb_convert_case($needle, self::MB_CASE_FOLD, $encoding); | |
| 748 | + $haystack = self::mb_convert_case($haystack, \MB_CASE_LOWER, $encoding); | |
| 749 | + $needle = self::mb_convert_case($needle, \MB_CASE_LOWER, $encoding); | |
| 679 | 750 | |
| 751 | + $haystack = str_replace(self::SIMPLE_CASE_FOLD[0], self::SIMPLE_CASE_FOLD[1], $haystack); | |
| 752 | + $needle = str_replace(self::SIMPLE_CASE_FOLD[0], self::SIMPLE_CASE_FOLD[1], $needle); | |
| 753 | + | |
| 680 | 754 | return self::mb_strrpos($haystack, $needle, $offset, $encoding); |
| 681 | 755 | } |
| 682 | 756 | |
| 683 | 757 | public static function mb_strstr($haystack, $needle, $part = false, $encoding = null) |
| @@ -736,9 +810,9 @@ | ||
| 736 | 810 | { |
| 737 | 811 | $encoding = self::getEncoding($encoding); |
| 738 | 812 | |
| 739 | 813 | if ('UTF-8' !== $encoding) { |
| 740 | - $s = iconv($encoding, 'UTF-8//IGNORE', $s); | |
| 814 | + $s = self::iconv($encoding, 'UTF-8', $s); | |
| 741 | 815 | } |
| 742 | 816 | |
| 743 | 817 | $s = preg_replace('/[\x{1100}-\x{115F}\x{2329}\x{232A}\x{2E80}-\x{303E}\x{3040}-\x{A4CF}\x{AC00}-\x{D7A3}\x{F900}-\x{FAFF}\x{FE10}-\x{FE19}\x{FE30}-\x{FE6F}\x{FF00}-\x{FF60}\x{FFE0}-\x{FFE6}\x{20000}-\x{2FFFD}\x{30000}-\x{3FFFD}]/u', '', $s, -1, $wide); |
| 744 | 818 | |
| @@ -797,8 +871,116 @@ | ||
| 797 | 871 | |
| 798 | 872 | return $code; |
| 799 | 873 | } |
| 800 | 874 | |
| 875 | + /** @return string|false */ | |
| 876 | + public static function mb_scrub(?string $string, ?string $encoding = null): string | |
| 877 | + { | |
| 878 | + if (null === $encoding) { | |
| 879 | + $encoding = self::mb_internal_encoding(); | |
| 880 | + } elseif (!self::assertEncoding($encoding, 'mb_scrub(): Argument #2 ($encoding) must be a valid encoding, "%s" given')) { | |
| 881 | + return false; | |
| 882 | + } | |
| 883 | + | |
| 884 | + return self::mb_convert_encoding((string) $string, $encoding, $encoding); | |
| 885 | + } | |
| 886 | + | |
| 887 | + /** @return string|false */ | |
| 888 | + public static function mb_str_pad(string $string, int $length, string $pad_string = ' ', int $pad_type = \STR_PAD_RIGHT, ?string $encoding = null) | |
| 889 | + { | |
| 890 | + if (null === $encoding) { | |
| 891 | + $encoding = self::mb_internal_encoding(); | |
| 892 | + } elseif (!self::assertEncoding($encoding, 'mb_str_pad(): Argument #5 ($encoding) must be a valid encoding, "%s" given')) { | |
| 893 | + return false; | |
| 894 | + } | |
| 895 | + | |
| 896 | + if (self::mb_strlen($pad_string, $encoding) <= 0) { | |
| 897 | + if (\PHP_VERSION_ID < 80000) { | |
| 898 | + trigger_error('mb_str_pad(): Argument #3 ($pad_string) must be a non-empty string', \E_USER_WARNING); | |
| 899 | + | |
| 900 | + return false; | |
| 901 | + } | |
| 902 | + | |
| 903 | + throw new \ValueError('mb_str_pad(): Argument #3 ($pad_string) must be a non-empty string'); | |
| 904 | + } | |
| 905 | + | |
| 906 | + if (!\in_array($pad_type, [\STR_PAD_RIGHT, \STR_PAD_LEFT, \STR_PAD_BOTH], true)) { | |
| 907 | + if (\PHP_VERSION_ID < 80000) { | |
| 908 | + trigger_error('mb_str_pad(): Argument #4 ($pad_type) must be STR_PAD_LEFT, STR_PAD_RIGHT, or STR_PAD_BOTH', \E_USER_WARNING); | |
| 909 | + | |
| 910 | + return false; | |
| 911 | + } | |
| 912 | + | |
| 913 | + throw new \ValueError('mb_str_pad(): Argument #4 ($pad_type) must be STR_PAD_LEFT, STR_PAD_RIGHT, or STR_PAD_BOTH'); | |
| 914 | + } | |
| 915 | + | |
| 916 | + $paddingRequired = $length - self::mb_strlen($string, $encoding); | |
| 917 | + | |
| 918 | + if ($paddingRequired < 1) { | |
| 919 | + return $string; | |
| 920 | + } | |
| 921 | + | |
| 922 | + switch ($pad_type) { | |
| 923 | + case \STR_PAD_LEFT: | |
| 924 | + return self::mb_substr(str_repeat($pad_string, $paddingRequired), 0, $paddingRequired, $encoding).$string; | |
| 925 | + case \STR_PAD_RIGHT: | |
| 926 | + return $string.self::mb_substr(str_repeat($pad_string, $paddingRequired), 0, $paddingRequired, $encoding); | |
| 927 | + default: | |
| 928 | + $leftPaddingLength = floor($paddingRequired / 2); | |
| 929 | + $rightPaddingLength = $paddingRequired - $leftPaddingLength; | |
| 930 | + | |
| 931 | + return self::mb_substr(str_repeat($pad_string, $leftPaddingLength), 0, $leftPaddingLength, $encoding).$string.self::mb_substr(str_repeat($pad_string, $rightPaddingLength), 0, $rightPaddingLength, $encoding); | |
| 932 | + } | |
| 933 | + } | |
| 934 | + | |
| 935 | + /** @return string|false */ | |
| 936 | + public static function mb_ucfirst(string $string, ?string $encoding = null) | |
| 937 | + { | |
| 938 | + if (null === $encoding) { | |
| 939 | + $encoding = self::mb_internal_encoding(); | |
| 940 | + } elseif (!self::assertEncoding($encoding, 'mb_ucfirst(): Argument #2 ($encoding) must be a valid encoding, "%s" given')) { | |
| 941 | + return false; | |
| 942 | + } | |
| 943 | + | |
| 944 | + $firstChar = mb_substr($string, 0, 1, $encoding); | |
| 945 | + $firstChar = mb_convert_case($firstChar, \MB_CASE_TITLE, $encoding); | |
| 946 | + | |
| 947 | + return $firstChar.mb_substr($string, 1, null, $encoding); | |
| 948 | + } | |
| 949 | + | |
| 950 | + /** @return string|false */ | |
| 951 | + public static function mb_lcfirst(string $string, ?string $encoding = null) | |
| 952 | + { | |
| 953 | + if (null === $encoding) { | |
| 954 | + $encoding = self::mb_internal_encoding(); | |
| 955 | + } elseif (!self::assertEncoding($encoding, 'mb_lcfirst(): Argument #2 ($encoding) must be a valid encoding, "%s" given')) { | |
| 956 | + return false; | |
| 957 | + } | |
| 958 | + | |
| 959 | + $firstChar = mb_substr($string, 0, 1, $encoding); | |
| 960 | + $firstChar = mb_convert_case($firstChar, \MB_CASE_LOWER, $encoding); | |
| 961 | + | |
| 962 | + return $firstChar.mb_substr($string, 1, null, $encoding); | |
| 963 | + } | |
| 964 | + | |
| 965 | + /** @return string|false */ | |
| 966 | + public static function mb_trim(string $string, ?string $characters = null, ?string $encoding = null) | |
| 967 | + { | |
| 968 | + return self::mb_internal_trim('{^[%s]+|[%1$s]+$}Du', $string, $characters, $encoding, __FUNCTION__); | |
| 969 | + } | |
| 970 | + | |
| 971 | + /** @return string|false */ | |
| 972 | + public static function mb_ltrim(string $string, ?string $characters = null, ?string $encoding = null) | |
| 973 | + { | |
| 974 | + return self::mb_internal_trim('{^[%s]+}Du', $string, $characters, $encoding, __FUNCTION__); | |
| 975 | + } | |
| 976 | + | |
| 977 | + /** @return string|false */ | |
| 978 | + public static function mb_rtrim(string $string, ?string $characters = null, ?string $encoding = null) | |
| 979 | + { | |
| 980 | + return self::mb_internal_trim('{[%s]+$}Du', $string, $characters, $encoding, __FUNCTION__); | |
| 981 | + } | |
| 982 | + | |
| 801 | 983 | private static function getSubpart($pos, $part, $haystack, $encoding) |
| 802 | 984 | { |
| 803 | 985 | if (false === $pos) { |
| 804 | 986 | return false; |
| @@ -868,7 +1050,89 @@ | ||
| 868 | 1050 | if ('UTF8' === $encoding) { |
| 869 | 1051 | return 'UTF-8'; |
| 870 | 1052 | } |
| 871 | 1053 | |
| 1054 | + if ('UTF-32' === $encoding) { | |
| 1055 | + return 'UTF-32BE'; | |
| 1056 | + } | |
| 1057 | + | |
| 1058 | + if ('UTF-16' === $encoding) { | |
| 1059 | + return 'UTF-16BE'; | |
| 1060 | + } | |
| 1061 | + | |
| 872 | 1062 | return $encoding; |
| 1063 | + } | |
| 1064 | + | |
| 1065 | + private static function iconv($fromEncoding, $toEncoding, $s) | |
| 1066 | + { | |
| 1067 | + if (null === self::$iconvSupportsIgnore) { | |
| 1068 | + self::$iconvSupportsIgnore = false !== @iconv('UTF-8', 'UTF-8//IGNORE', ''); | |
| 1069 | + } | |
| 1070 | + | |
| 1071 | + return self::$iconvSupportsIgnore | |
| 1072 | + ? iconv($fromEncoding, $toEncoding.'//IGNORE', $s) | |
| 1073 | + : iconv($fromEncoding, $toEncoding, $s); | |
| 1074 | + } | |
| 1075 | + | |
| 1076 | + /** @return string|false */ | |
| 1077 | + private static function mb_internal_trim(string $regex, string $string, ?string $characters, ?string $encoding, string $function) | |
| 1078 | + { | |
| 1079 | + if (null === $encoding) { | |
| 1080 | + $encoding = self::mb_internal_encoding(); | |
| 1081 | + } elseif (!self::assertEncoding($encoding, $function.'(): Argument #3 ($encoding) must be a valid encoding, "%s" given')) { | |
| 1082 | + return false; | |
| 1083 | + } | |
| 1084 | + | |
| 1085 | + if ('' === $characters) { | |
| 1086 | + return null === $encoding ? $string : self::mb_convert_encoding($string, $encoding); | |
| 1087 | + } | |
| 1088 | + | |
| 1089 | + if ('UTF-8' === $encoding) { | |
| 1090 | + $encoding = null; | |
| 1091 | + if (!preg_match('//u', $string)) { | |
| 1092 | + $string = @self::iconv('UTF-8', 'UTF-8', $string); | |
| 1093 | + } | |
| 1094 | + if (null !== $characters && !preg_match('//u', $characters)) { | |
| 1095 | + $characters = @self::iconv('UTF-8', 'UTF-8', $characters); | |
| 1096 | + } | |
| 1097 | + } else { | |
| 1098 | + $string = self::iconv($encoding, 'UTF-8', $string); | |
| 1099 | + | |
| 1100 | + if (null !== $characters) { | |
| 1101 | + $characters = self::iconv($encoding, 'UTF-8', $characters); | |
| 1102 | + } | |
| 1103 | + } | |
| 1104 | + | |
| 1105 | + if (null === $characters) { | |
| 1106 | + $characters = "\\0 \f\n\r\t\v\u{00A0}\u{1680}\u{2000}\u{2001}\u{2002}\u{2003}\u{2004}\u{2005}\u{2006}\u{2007}\u{2008}\u{2009}\u{200A}\u{2028}\u{2029}\u{202F}\u{205F}\u{3000}\u{0085}\u{180E}"; | |
| 1107 | + } else { | |
| 1108 | + $characters = preg_quote($characters); | |
| 1109 | + } | |
| 1110 | + | |
| 1111 | + $string = preg_replace(\sprintf($regex, $characters), '', $string); | |
| 1112 | + | |
| 1113 | + if (null === $encoding) { | |
| 1114 | + return $string; | |
| 1115 | + } | |
| 1116 | + | |
| 1117 | + return self::iconv('UTF-8', $encoding, $string); | |
| 1118 | + } | |
| 1119 | + | |
| 1120 | + private static function assertEncoding(string $encoding, string $errorFormat): bool | |
| 1121 | + { | |
| 1122 | + try { | |
| 1123 | + $validEncoding = @self::mb_check_encoding('', $encoding); | |
| 1124 | + } catch (\ValueError $e) { | |
| 1125 | + throw new \ValueError(\sprintf($errorFormat, $encoding)); | |
| 1126 | + } | |
| 1127 | + | |
| 1128 | + if (!$validEncoding) { | |
| 1129 | + if (80000 > \PHP_VERSION_ID) { | |
| 1130 | + trigger_error(\sprintf($errorFormat, $encoding), \E_USER_WARNING); | |
| 1131 | + } else { | |
| 1132 | + throw new \ValueError(\sprintf($errorFormat, $encoding)); | |
| 1133 | + } | |
| 1134 | + } | |
| 1135 | + | |
| 1136 | + return $validEncoding; | |
| 873 | 1137 | } |
| 874 | 1138 | } |