PluginProbe
BeyondWords – AI audio for publishers / 7.2.0
BeyondWords – AI audio for publishers v7.2.0
7.2.0 7.1.0 trunk 4.0.0 4.0.1 4.0.2 4.0.3 4.0.4 4.0.5 4.0.6 4.1.0 4.1.1 4.1.2 4.2.0 4.2.1 4.2.2 4.2.3 4.2.4 4.3.0 4.4.0 4.5.0 4.5.1 4.6.0 4.6.1 4.6.2 All 44 releases
speechkit / vendor / symfony / polyfill-mbstring / Mbstring.php

Mbstring.php in BeyondWords – AI audio for publishers 7.2.0, at vendor/symfony/polyfill-mbstring/Mbstring.php

1,139 lines 38.9 KB
No matching file
Up and down to move Enter to open Esc to close
Raw Download Zip
1 <?php
2
3 /*
4 * This file is part of the Symfony package.
5 *
6 * (c) Fabien Potencier <[email protected]>
7 *
8 * For the full copyright and license information, please view the LICENSE
9 * file that was distributed with this source code.
10 */
11
12 namespace Symfony\Polyfill\Mbstring;
13
14 /**
15 * Partial mbstring implementation in PHP, iconv based, UTF-8 centric.
16 *
17 * Implemented:
18 * - mb_chr - Returns a specific character from its Unicode code point
19 * - mb_convert_encoding - Convert character encoding
20 * - mb_convert_variables - Convert character code in variable(s)
21 * - mb_decode_mimeheader - Decode string in MIME header field
22 * - mb_encode_mimeheader - Encode string for MIME header XXX NATIVE IMPLEMENTATION IS REALLY BUGGED
23 * - mb_decode_numericentity - Decode HTML numeric string reference to character
24 * - mb_encode_numericentity - Encode character to HTML numeric string reference
25 * - mb_convert_case - Perform case folding on a string
26 * - mb_detect_encoding - Detect character encoding
27 * - mb_get_info - Get internal settings of mbstring
28 * - mb_http_input - Detect HTTP input character encoding
29 * - mb_http_output - Set/Get HTTP output character encoding
30 * - mb_internal_encoding - Set/Get internal character encoding
31 * - mb_list_encodings - Returns an array of all supported encodings
32 * - mb_ord - Returns the Unicode code point of a character
33 * - mb_output_handler - Callback function converts character encoding in output buffer
34 * - mb_scrub - Replaces ill-formed byte sequences with substitute characters
35 * - mb_strlen - Get string length
36 * - mb_strpos - Find position of first occurrence of string in a string
37 * - mb_strrpos - Find position of last occurrence of a string in a string
38 * - mb_str_split - Convert a string to an array
39 * - mb_strtolower - Make a string lowercase
40 * - mb_strtoupper - Make a string uppercase
41 * - mb_substitute_character - Set/Get substitution character
42 * - mb_substr - Get part of string
43 * - mb_stripos - Finds position of first occurrence of a string within another, case insensitive
44 * - mb_stristr - Finds first occurrence of a string within another, case insensitive
45 * - mb_strrchr - Finds the last occurrence of a character in a string within another
46 * - mb_strrichr - Finds the last occurrence of a character in a string within another, case insensitive
47 * - mb_strripos - Finds position of last occurrence of a string within another, case insensitive
48 * - mb_strstr - Finds first occurrence of a string within another
49 * - mb_strwidth - Return width of string
50 * - mb_substr_count - Count the number of substring occurrences
51 * - mb_ucfirst - Make a string's first character uppercase
52 * - mb_lcfirst - Make a string's first character lowercase
53 * - mb_trim - Strip whitespace (or other characters) from the beginning and end of a string
54 * - mb_ltrim - Strip whitespace (or other characters) from the beginning of a string
55 * - mb_rtrim - Strip whitespace (or other characters) from the end of a string
56 *
57 * Not implemented:
58 * - mb_convert_kana - Convert "kana" one from another ("zen-kaku", "han-kaku" and more)
59 * - mb_ereg_* - Regular expression with multibyte support
60 * - mb_parse_str - Parse GET/POST/COOKIE data and set global variable
61 * - mb_preferred_mime_name - Get MIME charset string
62 * - mb_regex_encoding - Returns current encoding for multibyte regex as string
63 * - mb_regex_set_options - Set/Get the default options for mbregex functions
64 * - mb_send_mail - Send encoded mail
65 * - mb_split - Split multibyte string using regular expression
66 * - mb_strcut - Get part of string
67 * - mb_strimwidth - Get truncated string with specified width
68 *
69 * @author Nicolas Grekas <[email protected]>
70 *
71 * @internal
72 */
73 final class Mbstring
74 {
75 public const MB_CASE_FOLD = \PHP_INT_MAX;
76
77 private const SIMPLE_CASE_FOLD = [
78 ['µ', 'ſ', "\xCD\x85", 'ς', "\xCF\x90", "\xCF\x91", "\xCF\x95", "\xCF\x96", "\xCF\xB0", "\xCF\xB1", "\xCF\xB5", "\xE1\xBA\x9B", "\xE1\xBE\xBE"],
79 ['μ', 's', 'ι', 'σ', 'β', 'θ', 'φ', 'π', 'κ', 'ρ', 'ε', "\xE1\xB9\xA1", 'ι'],
80 ];
81
82 private static $encodingList = ['ASCII', 'UTF-8'];
83 private static $language = 'neutral';
84 private static $internalEncoding = 'UTF-8';
85 private static $iconvSupportsIgnore;
86
87 public static function mb_convert_encoding($s, $toEncoding, $fromEncoding = null)
88 {
89 if (\is_array($s)) {
90 $r = [];
91 foreach ($s as $str) {
92 $r[] = self::mb_convert_encoding($str, $toEncoding, $fromEncoding);
93 }
94
95 return $r;
96 }
97
98 if (\is_array($fromEncoding) || (null !== $fromEncoding && false !== strpos($fromEncoding, ','))) {
99 $fromEncoding = self::mb_detect_encoding($s, $fromEncoding);
100 } else {
101 $fromEncoding = self::getEncoding($fromEncoding);
102 }
103
104 $toEncoding = self::getEncoding($toEncoding);
105
106 if ('BASE64' === $fromEncoding) {
107 $s = base64_decode($s);
108 $fromEncoding = $toEncoding;
109 }
110
111 if ('BASE64' === $toEncoding) {
112 return base64_encode($s);
113 }
114
115 if ('HTML-ENTITIES' === $toEncoding || 'HTML' === $toEncoding) {
116 if ('HTML-ENTITIES' === $fromEncoding || 'HTML' === $fromEncoding) {
117 $fromEncoding = 'Windows-1252';
118 }
119 if ('UTF-8' !== $fromEncoding) {
120 $s = self::iconv($fromEncoding, 'UTF-8', $s);
121 }
122
123 return preg_replace_callback('/[\x80-\xFF]+/', [__CLASS__, 'html_encoding_callback'], $s);
124 }
125
126 if ('HTML-ENTITIES' === $fromEncoding) {
127 $decodeControlChars = static function ($m) {
128 $code = '' !== ($m[2] ?? '') ? hexdec($m[2]) : (int) $m[1];
129
130 if ($code < 32 || 127 === $code) {
131 return \chr($code);
132 }
133 if (128 <= $code && $code <= 159) {
134 return "\xC2".\chr(0x80 | ($code & 0x3F));
135 }
136
137 return $m[0];
138 };
139
140 if (\PHP_VERSION_ID >= 70400) {
141 $s = html_entity_decode($s, \ENT_QUOTES, 'UTF-8');
142 // html_entity_decode() leaves numeric entities for C0/C1 control
143 // characters as-is (HTML spec), but mb_convert_encoding() decodes
144 // them. Catch what html_entity_decode() missed.
145 if (false !== strpos($s, '&#')) {
146 $s = preg_replace_callback('/&#(?:0*([0-9]++)|[xX]0*([0-9a-fA-F]++));/', $decodeControlChars, $s);
147 }
148 } else {
149 // PHP < 7.4: html_entity_decode() truncates strings at NUL bytes,
150 // so decode the control character entities first then call
151 // html_entity_decode() on each NUL-delimited chunk independently.
152 $s = preg_replace_callback('/&#(?:0*([0-9]++)|[xX]0*([0-9a-fA-F]++));/', $decodeControlChars, $s);
153 $s = implode("\0", array_map(static function ($chunk) {
154 return html_entity_decode($chunk, \ENT_QUOTES, 'UTF-8');
155 }, explode("\0", $s)));
156 }
157 $fromEncoding = 'UTF-8';
158 }
159
160 return self::iconv($fromEncoding, $toEncoding, $s);
161 }
162
163 public static function mb_convert_variables($toEncoding, $fromEncoding, &...$vars)
164 {
165 $ok = true;
166 array_walk_recursive($vars, static function (&$v) use (&$ok, $toEncoding, $fromEncoding) {
167 if (false === $v = self::mb_convert_encoding($v, $toEncoding, $fromEncoding)) {
168 $ok = false;
169 }
170 });
171
172 return $ok ? $fromEncoding : false;
173 }
174
175 public static function mb_decode_mimeheader($s)
176 {
177 return iconv_mime_decode($s, 2, self::$internalEncoding);
178 }
179
180 public static function mb_encode_mimeheader($s, $charset = null, $transferEncoding = null, $linefeed = null, $indent = null)
181 {
182 trigger_error('mb_encode_mimeheader() is bugged. Please use iconv_mime_encode() instead', \E_USER_WARNING);
183 }
184
185 public static function mb_decode_numericentity($s, $convmap, $encoding = null)
186 {
187 if (null !== $s && !\is_scalar($s) && !(\is_object($s) && method_exists($s, '__toString'))) {
188 trigger_error('mb_decode_numericentity() expects parameter 1 to be string, '.\gettype($s).' given', \E_USER_WARNING);
189
190 return null;
191 }
192
193 if (!\is_array($convmap) || (80000 > \PHP_VERSION_ID && !$convmap)) {
194 return false;
195 }
196
197 if (null !== $encoding && !\is_scalar($encoding)) {
198 trigger_error('mb_decode_numericentity() expects parameter 3 to be string, '.\gettype($s).' given', \E_USER_WARNING);
199
200 return ''; // Instead of null (cf. mb_encode_numericentity).
201 }
202
203 $s = (string) $s;
204 if ('' === $s) {
205 return '';
206 }
207
208 $encoding = self::getEncoding($encoding);
209
210 if ('UTF-8' === $encoding) {
211 $encoding = null;
212 if (!preg_match('//u', $s)) {
213 $s = @self::iconv('UTF-8', 'UTF-8', $s);
214 }
215 } else {
216 $s = self::iconv($encoding, 'UTF-8', $s);
217 }
218
219 $cnt = floor(\count($convmap) / 4) * 4;
220
221 for ($i = 0; $i < $cnt; $i += 4) {
222 // collector_decode_htmlnumericentity ignores $convmap[$i + 3]
223 $convmap[$i] += $convmap[$i + 2];
224 $convmap[$i + 1] += $convmap[$i + 2];
225 }
226
227 $s = preg_replace_callback('/&#(?:0*([0-9]+)|x0*([0-9a-fA-F]+))'.(\PHP_VERSION_ID >= 80200 ? '' : '(?!&)').';?/', static function (array $m) use ($cnt, $convmap) {
228 $c = isset($m[2]) ? (int) hexdec($m[2]) : $m[1];
229 for ($i = 0; $i < $cnt; $i += 4) {
230 if ($c >= $convmap[$i] && $c <= $convmap[$i + 1]) {
231 return self::mb_chr($c - $convmap[$i + 2]);
232 }
233 }
234
235 return $m[0];
236 }, $s);
237
238 if (null === $encoding) {
239 return $s;
240 }
241
242 return self::iconv('UTF-8', $encoding, $s);
243 }
244
245 public static function mb_encode_numericentity($s, $convmap, $encoding = null, $is_hex = false)
246 {
247 if (null !== $s && !\is_scalar($s) && !(\is_object($s) && method_exists($s, '__toString'))) {
248 trigger_error('mb_encode_numericentity() expects parameter 1 to be string, '.\gettype($s).' given', \E_USER_WARNING);
249
250 return null;
251 }
252
253 if (!\is_array($convmap) || (80000 > \PHP_VERSION_ID && !$convmap)) {
254 return false;
255 }
256
257 if (null !== $encoding && !\is_scalar($encoding)) {
258 trigger_error('mb_encode_numericentity() expects parameter 3 to be string, '.\gettype($s).' given', \E_USER_WARNING);
259
260 return null; // Instead of '' (cf. mb_decode_numericentity).
261 }
262
263 if (null !== $is_hex && !\is_scalar($is_hex)) {
264 trigger_error('mb_encode_numericentity() expects parameter 4 to be boolean, '.\gettype($s).' given', \E_USER_WARNING);
265
266 return null;
267 }
268
269 $s = (string) $s;
270 if ('' === $s) {
271 return '';
272 }
273
274 $encoding = self::getEncoding($encoding);
275
276 if ('UTF-8' === $encoding) {
277 $encoding = null;
278 if (!preg_match('//u', $s)) {
279 $s = @self::iconv('UTF-8', 'UTF-8', $s);
280 }
281 } else {
282 $s = self::iconv($encoding, 'UTF-8', $s);
283 }
284
285 static $ulenMask = ["\xC0" => 2, "\xD0" => 2, "\xE0" => 3, "\xF0" => 4];
286
287 $cnt = floor(\count($convmap) / 4) * 4;
288 $i = 0;
289 $len = \strlen($s);
290 $result = '';
291
292 while ($i < $len) {
293 $ulen = $s[$i] < "\x80" ? 1 : $ulenMask[$s[$i] & "\xF0"];
294 $uchr = substr($s, $i, $ulen);
295 $i += $ulen;
296 $c = self::mb_ord($uchr);
297
298 for ($j = 0; $j < $cnt; $j += 4) {
299 if ($c >= $convmap[$j] && $c <= $convmap[$j + 1]) {
300 $cOffset = ($c + $convmap[$j + 2]) & $convmap[$j + 3];
301 $result .= $is_hex ? \sprintf('&#x%X;', $cOffset) : '&#'.$cOffset.';';
302 continue 2;
303 }
304 }
305 $result .= $uchr;
306 }
307
308 if (null === $encoding) {
309 return $result;
310 }
311
312 return self::iconv('UTF-8', $encoding, $result);
313 }
314
315 public static function mb_convert_case($s, $mode, $encoding = null)
316 {
317 $s = (string) $s;
318 if ('' === $s) {
319 return '';
320 }
321
322 $encoding = self::getEncoding($encoding);
323
324 if ('UTF-8' === $encoding) {
325 $encoding = null;
326 if (!preg_match('//u', $s)) {
327 $s = @self::iconv('UTF-8', 'UTF-8', $s);
328 }
329 } else {
330 $s = self::iconv($encoding, 'UTF-8', $s);
331 }
332
333 if (\MB_CASE_TITLE == $mode) {
334 static $titleRegexp = null;
335 if (null === $titleRegexp) {
336 $titleRegexp = self::getData('titleCaseRegexp');
337 }
338 $s = preg_replace_callback($titleRegexp, [__CLASS__, 'title_case'], $s);
339 } else {
340 if (\MB_CASE_UPPER == $mode) {
341 static $upper = null;
342 if (null === $upper) {
343 $upper = self::getData('upperCase');
344 }
345 $map = $upper;
346 } else {
347 if (self::MB_CASE_FOLD === $mode) {
348 static $caseFolding = null;
349 if (null === $caseFolding) {
350 $caseFolding = self::getData('caseFolding');
351 }
352 $s = strtr($s, $caseFolding);
353 }
354
355 static $lower = null;
356 if (null === $lower) {
357 $lower = self::getData('lowerCase');
358 }
359 $map = $lower;
360 }
361
362 static $ulenMask = ["\xC0" => 2, "\xD0" => 2, "\xE0" => 3, "\xF0" => 4];
363
364 $i = 0;
365 $len = \strlen($s);
366
367 while ($i < $len) {
368 $ulen = $s[$i] < "\x80" ? 1 : $ulenMask[$s[$i] & "\xF0"];
369 $uchr = substr($s, $i, $ulen);
370 $i += $ulen;
371
372 if (isset($map[$uchr])) {
373 $uchr = $map[$uchr];
374 $nlen = \strlen($uchr);
375
376 if ($nlen == $ulen) {
377 $nlen = $i;
378 do {
379 $s[--$nlen] = $uchr[--$ulen];
380 } while ($ulen);
381 } else {
382 $s = substr_replace($s, $uchr, $i - $ulen, $ulen);
383 $len += $nlen - $ulen;
384 $i += $nlen - $ulen;
385 }
386 }
387 }
388 }
389
390 if (null === $encoding) {
391 return $s;
392 }
393
394 return self::iconv('UTF-8', $encoding, $s);
395 }
396
397 public static function mb_internal_encoding($encoding = null)
398 {
399 if (null === $encoding) {
400 return self::$internalEncoding;
401 }
402
403 $normalizedEncoding = self::getEncoding($encoding);
404
405 if ('UTF-8' === $normalizedEncoding || false !== @iconv($normalizedEncoding, $normalizedEncoding, ' ')) {
406 self::$internalEncoding = $normalizedEncoding;
407
408 return true;
409 }
410
411 if (80000 > \PHP_VERSION_ID) {
412 return false;
413 }
414
415 throw new \ValueError(\sprintf('Argument #1 ($encoding) must be a valid encoding, "%s" given', $encoding));
416 }
417
418 public static function mb_language($lang = null)
419 {
420 if (null === $lang) {
421 return self::$language;
422 }
423
424 switch ($normalizedLang = strtolower($lang)) {
425 case 'uni':
426 case 'neutral':
427 self::$language = $normalizedLang;
428
429 return true;
430 }
431
432 if (80000 > \PHP_VERSION_ID) {
433 return false;
434 }
435
436 throw new \ValueError(\sprintf('Argument #1 ($language) must be a valid language, "%s" given', $lang));
437 }
438
439 public static function mb_list_encodings()
440 {
441 return ['UTF-8'];
442 }
443
444 public static function mb_encoding_aliases($encoding)
445 {
446 switch (strtoupper($encoding)) {
447 case 'UTF8':
448 case 'UTF-8':
449 return ['utf8'];
450 }
451
452 return false;
453 }
454
455 public static function mb_check_encoding($var = null, $encoding = null)
456 {
457 if (null === $encoding) {
458 if (null === $var) {
459 return false;
460 }
461 $encoding = self::$internalEncoding;
462 }
463
464 if (!\is_array($var)) {
465 return self::mb_detect_encoding($var, [$encoding]) || false !== @iconv($encoding, $encoding, $var);
466 }
467
468 foreach ($var as $key => $value) {
469 if (!self::mb_check_encoding($key, $encoding)) {
470 return false;
471 }
472 if (!self::mb_check_encoding($value, $encoding)) {
473 return false;
474 }
475 }
476
477 return true;
478 }
479
480 public static function mb_detect_encoding($str, $encodingList = null, $strict = false)
481 {
482 if (null === $encodingList) {
483 $encodingList = self::$encodingList;
484 } else {
485 if (!\is_array($encodingList)) {
486 $encodingList = array_map('trim', explode(',', $encodingList));
487 }
488 $encodingList = array_map('strtoupper', $encodingList);
489 }
490
491 foreach ($encodingList as $enc) {
492 switch ($enc) {
493 case 'ASCII':
494 if (!preg_match('/[\x80-\xFF]/', $str)) {
495 return $enc;
496 }
497 break;
498
499 case 'UTF8':
500 case 'UTF-8':
501 if (preg_match('//u', $str)) {
502 return 'UTF-8';
503 }
504 break;
505
506 default:
507 if (0 === strncmp($enc, 'ISO-8859-', 9)) {
508 return $enc;
509 }
510 }
511 }
512
513 return false;
514 }
515
516 public static function mb_detect_order($encodingList = null)
517 {
518 if (null === $encodingList) {
519 return self::$encodingList;
520 }
521
522 if (!\is_array($encodingList)) {
523 $encodingList = array_map('trim', explode(',', $encodingList));
524 }
525 $encodingList = array_map('strtoupper', $encodingList);
526
527 foreach ($encodingList as $enc) {
528 switch ($enc) {
529 default:
530 if (strncmp($enc, 'ISO-8859-', 9)) {
531 return false;
532 }
533 // no break
534 case 'ASCII':
535 case 'UTF8':
536 case 'UTF-8':
537 }
538 }
539
540 self::$encodingList = $encodingList;
541
542 return true;
543 }
544
545 public static function mb_strlen($s, $encoding = null)
546 {
547 $encoding = self::getEncoding($encoding);
548 if ('CP850' === $encoding || 'ASCII' === $encoding) {
549 return \strlen($s);
550 }
551
552 if (false !== $len = @iconv_strlen($s, $encoding)) {
553 return $len;
554 }
555
556 if ('UTF-8' !== $encoding) {
557 return $len;
558 }
559
560 return preg_match_all('/[\x00-\x7F]|[\xC0-\xDF][\x80-\xBF]?|[\xE0-\xEF][\x80-\xBF]{0,2}|[\xF0-\xF7][\x80-\xBF]{0,3}|[\xF8-\xFB][\x80-\xBF]{0,4}|[\xFC-\xFD][\x80-\xBF]{0,5}|[\x80-\xBF\xFE\xFF]/s', $s);
561 }
562
563 public static function mb_strpos($haystack, $needle, $offset = 0, $encoding = null)
564 {
565 $encoding = self::getEncoding($encoding);
566 if ('CP850' === $encoding || 'ASCII' === $encoding) {
567 return strpos($haystack, $needle, $offset);
568 }
569
570 $needle = (string) $needle;
571 if ('' === $needle) {
572 if (80000 > \PHP_VERSION_ID) {
573 trigger_error(__METHOD__.': Empty delimiter', \E_USER_WARNING);
574
575 return false;
576 }
577
578 return 0;
579 }
580
581 return iconv_strpos($haystack, $needle, $offset, $encoding);
582 }
583
584 public static function mb_strrpos($haystack, $needle, $offset = 0, $encoding = null)
585 {
586 $encoding = self::getEncoding($encoding);
587 if ('CP850' === $encoding || 'ASCII' === $encoding) {
588 return strrpos($haystack, $needle, $offset);
589 }
590
591 if ($offset != (int) $offset) {
592 $offset = 0;
593 } elseif ($offset = (int) $offset) {
594 if ($offset < 0) {
595 if (0 > $offset += self::mb_strlen($needle)) {
596 $haystack = self::mb_substr($haystack, 0, $offset, $encoding);
597 }
598 $offset = 0;
599 } else {
600 $haystack = self::mb_substr($haystack, $offset, 2147483647, $encoding);
601 }
602 }
603
604 $pos = '' !== $needle || 80000 > \PHP_VERSION_ID
605 ? iconv_strrpos($haystack, $needle, $encoding)
606 : self::mb_strlen($haystack, $encoding);
607
608 return false !== $pos ? $offset + $pos : false;
609 }
610
611 public static function mb_str_split($string, $split_length = 1, $encoding = null)
612 {
613 if (null !== $string && !\is_scalar($string) && !(\is_object($string) && method_exists($string, '__toString'))) {
614 trigger_error('mb_str_split() expects parameter 1 to be string, '.\gettype($string).' given', \E_USER_WARNING);
615
616 return null;
617 }
618
619 if (1 > $split_length = (int) $split_length) {
620 if (80000 > \PHP_VERSION_ID) {
621 trigger_error('The length of each segment must be greater than zero', \E_USER_WARNING);
622
623 return false;
624 }
625
626 throw new \ValueError('Argument #2 ($length) must be greater than 0');
627 }
628
629 if (null === $encoding) {
630 $encoding = mb_internal_encoding();
631 }
632
633 if ('UTF-8' === $encoding = self::getEncoding($encoding)) {
634 $rx = '/(';
635 while (65535 < $split_length) {
636 $rx .= '.{65535}';
637 $split_length -= 65535;
638 }
639 $rx .= '.{'.$split_length.'})/us';
640
641 return preg_split($rx, $string, -1, \PREG_SPLIT_DELIM_CAPTURE | \PREG_SPLIT_NO_EMPTY);
642 }
643
644 $result = [];
645 $length = mb_strlen($string, $encoding);
646
647 for ($i = 0; $i < $length; $i += $split_length) {
648 $result[] = mb_substr($string, $i, $split_length, $encoding);
649 }
650
651 return $result;
652 }
653
654 public static function mb_strtolower($s, $encoding = null)
655 {
656 return self::mb_convert_case($s, \MB_CASE_LOWER, $encoding);
657 }
658
659 public static function mb_strtoupper($s, $encoding = null)
660 {
661 return self::mb_convert_case($s, \MB_CASE_UPPER, $encoding);
662 }
663
664 public static function mb_substitute_character($c = null)
665 {
666 if (null === $c) {
667 return 'none';
668 }
669 if (0 === strcasecmp($c, 'none')) {
670 return true;
671 }
672 if (80000 > \PHP_VERSION_ID) {
673 return false;
674 }
675 if (\is_int($c) || 'long' === $c || 'entity' === $c) {
676 return false;
677 }
678
679 throw new \ValueError('Argument #1 ($substitute_character) must be "none", "long", "entity" or a valid codepoint');
680 }
681
682 public static function mb_substr($s, $start, $length = null, $encoding = null)
683 {
684 $encoding = self::getEncoding($encoding);
685 if ('CP850' === $encoding || 'ASCII' === $encoding) {
686 return (string) substr($s, $start, null === $length ? 2147483647 : $length);
687 }
688
689 if ($start < 0) {
690 $start = iconv_strlen($s, $encoding) + $start;
691 if ($start < 0) {
692 $start = 0;
693 }
694 }
695
696 if (null === $length) {
697 $length = 2147483647;
698 } elseif ($length < 0) {
699 $length = iconv_strlen($s, $encoding) + $length - $start;
700 if ($length < 0) {
701 return '';
702 }
703 }
704
705 return (string) iconv_substr($s, $start, $length, $encoding);
706 }
707
708 public static function mb_stripos($haystack, $needle, $offset = 0, $encoding = null)
709 {
710 [$haystack, $needle] = str_replace(self::SIMPLE_CASE_FOLD[0], self::SIMPLE_CASE_FOLD[1], [
711 self::mb_convert_case($haystack, \MB_CASE_LOWER, $encoding),
712 self::mb_convert_case($needle, \MB_CASE_LOWER, $encoding),
713 ]);
714
715 return self::mb_strpos($haystack, $needle, $offset, $encoding);
716 }
717
718 public static function mb_stristr($haystack, $needle, $part = false, $encoding = null)
719 {
720 $pos = self::mb_stripos($haystack, $needle, 0, $encoding);
721
722 return self::getSubpart($pos, $part, $haystack, $encoding);
723 }
724
725 public static function mb_strrchr($haystack, $needle, $part = false, $encoding = null)
726 {
727 $encoding = self::getEncoding($encoding);
728 if ('CP850' === $encoding || 'ASCII' === $encoding) {
729 $pos = strrpos($haystack, $needle);
730 } else {
731 $needle = self::mb_substr($needle, 0, 1, $encoding);
732 $pos = iconv_strrpos($haystack, $needle, $encoding);
733 }
734
735 return self::getSubpart($pos, $part, $haystack, $encoding);
736 }
737
738 public static function mb_strrichr($haystack, $needle, $part = false, $encoding = null)
739 {
740 $needle = self::mb_substr($needle, 0, 1, $encoding);
741 $pos = self::mb_strripos($haystack, $needle, $encoding);
742
743 return self::getSubpart($pos, $part, $haystack, $encoding);
744 }
745
746 public static function mb_strripos($haystack, $needle, $offset = 0, $encoding = null)
747 {
748 $haystack = self::mb_convert_case($haystack, \MB_CASE_LOWER, $encoding);
749 $needle = self::mb_convert_case($needle, \MB_CASE_LOWER, $encoding);
750
751 $haystack = str_replace(self::SIMPLE_CASE_FOLD[0], self::SIMPLE_CASE_FOLD[1], $haystack);
752 $needle = str_replace(self::SIMPLE_CASE_FOLD[0], self::SIMPLE_CASE_FOLD[1], $needle);
753
754 return self::mb_strrpos($haystack, $needle, $offset, $encoding);
755 }
756
757 public static function mb_strstr($haystack, $needle, $part = false, $encoding = null)
758 {
759 $pos = strpos($haystack, $needle);
760 if (false === $pos) {
761 return false;
762 }
763 if ($part) {
764 return substr($haystack, 0, $pos);
765 }
766
767 return substr($haystack, $pos);
768 }
769
770 public static function mb_get_info($type = 'all')
771 {
772 $info = [
773 'internal_encoding' => self::$internalEncoding,
774 'http_output' => 'pass',
775 'http_output_conv_mimetypes' => '^(text/|application/xhtml\+xml)',
776 'func_overload' => 0,
777 'func_overload_list' => 'no overload',
778 'mail_charset' => 'UTF-8',
779 'mail_header_encoding' => 'BASE64',
780 'mail_body_encoding' => 'BASE64',
781 'illegal_chars' => 0,
782 'encoding_translation' => 'Off',
783 'language' => self::$language,
784 'detect_order' => self::$encodingList,
785 'substitute_character' => 'none',
786 'strict_detection' => 'Off',
787 ];
788
789 if ('all' === $type) {
790 return $info;
791 }
792 if (isset($info[$type])) {
793 return $info[$type];
794 }
795
796 return false;
797 }
798
799 public static function mb_http_input($type = '')
800 {
801 return false;
802 }
803
804 public static function mb_http_output($encoding = null)
805 {
806 return null !== $encoding ? 'pass' === $encoding : 'pass';
807 }
808
809 public static function mb_strwidth($s, $encoding = null)
810 {
811 $encoding = self::getEncoding($encoding);
812
813 if ('UTF-8' !== $encoding) {
814 $s = self::iconv($encoding, 'UTF-8', $s);
815 }
816
817 $s = preg_replace('/[\x{1100}-\x{115F}\x{2329}\x{232A}\x{2E80}-\x{303E}\x{3040}-\x{A4CF}\x{AC00}-\x{D7A3}\x{F900}-\x{FAFF}\x{FE10}-\x{FE19}\x{FE30}-\x{FE6F}\x{FF00}-\x{FF60}\x{FFE0}-\x{FFE6}\x{20000}-\x{2FFFD}\x{30000}-\x{3FFFD}]/u', '', $s, -1, $wide);
818
819 return ($wide << 1) + iconv_strlen($s, 'UTF-8');
820 }
821
822 public static function mb_substr_count($haystack, $needle, $encoding = null)
823 {
824 return substr_count($haystack, $needle);
825 }
826
827 public static function mb_output_handler($contents, $status)
828 {
829 return $contents;
830 }
831
832 public static function mb_chr($code, $encoding = null)
833 {
834 if (0x80 > $code %= 0x200000) {
835 $s = \chr($code);
836 } elseif (0x800 > $code) {
837 $s = \chr(0xC0 | $code >> 6).\chr(0x80 | $code & 0x3F);
838 } elseif (0x10000 > $code) {
839 $s = \chr(0xE0 | $code >> 12).\chr(0x80 | $code >> 6 & 0x3F).\chr(0x80 | $code & 0x3F);
840 } else {
841 $s = \chr(0xF0 | $code >> 18).\chr(0x80 | $code >> 12 & 0x3F).\chr(0x80 | $code >> 6 & 0x3F).\chr(0x80 | $code & 0x3F);
842 }
843
844 if ('UTF-8' !== $encoding = self::getEncoding($encoding)) {
845 $s = mb_convert_encoding($s, $encoding, 'UTF-8');
846 }
847
848 return $s;
849 }
850
851 public static function mb_ord($s, $encoding = null)
852 {
853 if ('UTF-8' !== $encoding = self::getEncoding($encoding)) {
854 $s = mb_convert_encoding($s, 'UTF-8', $encoding);
855 }
856
857 if (1 === \strlen($s)) {
858 return \ord($s);
859 }
860
861 $code = ($s = unpack('C*', substr($s, 0, 4))) ? $s[1] : 0;
862 if (0xF0 <= $code) {
863 return (($code - 0xF0) << 18) + (($s[2] - 0x80) << 12) + (($s[3] - 0x80) << 6) + $s[4] - 0x80;
864 }
865 if (0xE0 <= $code) {
866 return (($code - 0xE0) << 12) + (($s[2] - 0x80) << 6) + $s[3] - 0x80;
867 }
868 if (0xC0 <= $code) {
869 return (($code - 0xC0) << 6) + $s[2] - 0x80;
870 }
871
872 return $code;
873 }
874
875 /** @return string|false */
876 public static function mb_scrub(?string $string, ?string $encoding = null): string
877 {
878 if (null === $encoding) {
879 $encoding = self::mb_internal_encoding();
880 } elseif (!self::assertEncoding($encoding, 'mb_scrub(): Argument #2 ($encoding) must be a valid encoding, "%s" given')) {
881 return false;
882 }
883
884 return self::mb_convert_encoding((string) $string, $encoding, $encoding);
885 }
886
887 /** @return string|false */
888 public static function mb_str_pad(string $string, int $length, string $pad_string = ' ', int $pad_type = \STR_PAD_RIGHT, ?string $encoding = null)
889 {
890 if (null === $encoding) {
891 $encoding = self::mb_internal_encoding();
892 } elseif (!self::assertEncoding($encoding, 'mb_str_pad(): Argument #5 ($encoding) must be a valid encoding, "%s" given')) {
893 return false;
894 }
895
896 if (self::mb_strlen($pad_string, $encoding) <= 0) {
897 if (\PHP_VERSION_ID < 80000) {
898 trigger_error('mb_str_pad(): Argument #3 ($pad_string) must be a non-empty string', \E_USER_WARNING);
899
900 return false;
901 }
902
903 throw new \ValueError('mb_str_pad(): Argument #3 ($pad_string) must be a non-empty string');
904 }
905
906 if (!\in_array($pad_type, [\STR_PAD_RIGHT, \STR_PAD_LEFT, \STR_PAD_BOTH], true)) {
907 if (\PHP_VERSION_ID < 80000) {
908 trigger_error('mb_str_pad(): Argument #4 ($pad_type) must be STR_PAD_LEFT, STR_PAD_RIGHT, or STR_PAD_BOTH', \E_USER_WARNING);
909
910 return false;
911 }
912
913 throw new \ValueError('mb_str_pad(): Argument #4 ($pad_type) must be STR_PAD_LEFT, STR_PAD_RIGHT, or STR_PAD_BOTH');
914 }
915
916 $paddingRequired = $length - self::mb_strlen($string, $encoding);
917
918 if ($paddingRequired < 1) {
919 return $string;
920 }
921
922 switch ($pad_type) {
923 case \STR_PAD_LEFT:
924 return self::mb_substr(str_repeat($pad_string, $paddingRequired), 0, $paddingRequired, $encoding).$string;
925 case \STR_PAD_RIGHT:
926 return $string.self::mb_substr(str_repeat($pad_string, $paddingRequired), 0, $paddingRequired, $encoding);
927 default:
928 $leftPaddingLength = floor($paddingRequired / 2);
929 $rightPaddingLength = $paddingRequired - $leftPaddingLength;
930
931 return self::mb_substr(str_repeat($pad_string, $leftPaddingLength), 0, $leftPaddingLength, $encoding).$string.self::mb_substr(str_repeat($pad_string, $rightPaddingLength), 0, $rightPaddingLength, $encoding);
932 }
933 }
934
935 /** @return string|false */
936 public static function mb_ucfirst(string $string, ?string $encoding = null)
937 {
938 if (null === $encoding) {
939 $encoding = self::mb_internal_encoding();
940 } elseif (!self::assertEncoding($encoding, 'mb_ucfirst(): Argument #2 ($encoding) must be a valid encoding, "%s" given')) {
941 return false;
942 }
943
944 $firstChar = mb_substr($string, 0, 1, $encoding);
945 $firstChar = mb_convert_case($firstChar, \MB_CASE_TITLE, $encoding);
946
947 return $firstChar.mb_substr($string, 1, null, $encoding);
948 }
949
950 /** @return string|false */
951 public static function mb_lcfirst(string $string, ?string $encoding = null)
952 {
953 if (null === $encoding) {
954 $encoding = self::mb_internal_encoding();
955 } elseif (!self::assertEncoding($encoding, 'mb_lcfirst(): Argument #2 ($encoding) must be a valid encoding, "%s" given')) {
956 return false;
957 }
958
959 $firstChar = mb_substr($string, 0, 1, $encoding);
960 $firstChar = mb_convert_case($firstChar, \MB_CASE_LOWER, $encoding);
961
962 return $firstChar.mb_substr($string, 1, null, $encoding);
963 }
964
965 /** @return string|false */
966 public static function mb_trim(string $string, ?string $characters = null, ?string $encoding = null)
967 {
968 return self::mb_internal_trim('{^[%s]+|[%1$s]+$}Du', $string, $characters, $encoding, __FUNCTION__);
969 }
970
971 /** @return string|false */
972 public static function mb_ltrim(string $string, ?string $characters = null, ?string $encoding = null)
973 {
974 return self::mb_internal_trim('{^[%s]+}Du', $string, $characters, $encoding, __FUNCTION__);
975 }
976
977 /** @return string|false */
978 public static function mb_rtrim(string $string, ?string $characters = null, ?string $encoding = null)
979 {
980 return self::mb_internal_trim('{[%s]+$}Du', $string, $characters, $encoding, __FUNCTION__);
981 }
982
983 private static function getSubpart($pos, $part, $haystack, $encoding)
984 {
985 if (false === $pos) {
986 return false;
987 }
988 if ($part) {
989 return self::mb_substr($haystack, 0, $pos, $encoding);
990 }
991
992 return self::mb_substr($haystack, $pos, null, $encoding);
993 }
994
995 private static function html_encoding_callback(array $m)
996 {
997 $i = 1;
998 $entities = '';
999 $m = unpack('C*', htmlentities($m[0], \ENT_COMPAT, 'UTF-8'));
1000
1001 while (isset($m[$i])) {
1002 if (0x80 > $m[$i]) {
1003 $entities .= \chr($m[$i++]);
1004 continue;
1005 }
1006 if (0xF0 <= $m[$i]) {
1007 $c = (($m[$i++] - 0xF0) << 18) + (($m[$i++] - 0x80) << 12) + (($m[$i++] - 0x80) << 6) + $m[$i++] - 0x80;
1008 } elseif (0xE0 <= $m[$i]) {
1009 $c = (($m[$i++] - 0xE0) << 12) + (($m[$i++] - 0x80) << 6) + $m[$i++] - 0x80;
1010 } else {
1011 $c = (($m[$i++] - 0xC0) << 6) + $m[$i++] - 0x80;
1012 }
1013
1014 $entities .= '&#'.$c.';';
1015 }
1016
1017 return $entities;
1018 }
1019
1020 private static function title_case(array $s)
1021 {
1022 return self::mb_convert_case($s[1], \MB_CASE_UPPER, 'UTF-8').self::mb_convert_case($s[2], \MB_CASE_LOWER, 'UTF-8');
1023 }
1024
1025 private static function getData($file)
1026 {
1027 if (file_exists($file = __DIR__.'/Resources/unidata/'.$file.'.php')) {
1028 return require $file;
1029 }
1030
1031 return false;
1032 }
1033
1034 private static function getEncoding($encoding)
1035 {
1036 if (null === $encoding) {
1037 return self::$internalEncoding;
1038 }
1039
1040 if ('UTF-8' === $encoding) {
1041 return 'UTF-8';
1042 }
1043
1044 $encoding = strtoupper($encoding);
1045
1046 if ('8BIT' === $encoding || 'BINARY' === $encoding) {
1047 return 'CP850';
1048 }
1049
1050 if ('UTF8' === $encoding) {
1051 return 'UTF-8';
1052 }
1053
1054 if ('UTF-32' === $encoding) {
1055 return 'UTF-32BE';
1056 }
1057
1058 if ('UTF-16' === $encoding) {
1059 return 'UTF-16BE';
1060 }
1061
1062 return $encoding;
1063 }
1064
1065 private static function iconv($fromEncoding, $toEncoding, $s)
1066 {
1067 if (null === self::$iconvSupportsIgnore) {
1068 self::$iconvSupportsIgnore = false !== @iconv('UTF-8', 'UTF-8//IGNORE', '');
1069 }
1070
1071 return self::$iconvSupportsIgnore
1072 ? iconv($fromEncoding, $toEncoding.'//IGNORE', $s)
1073 : iconv($fromEncoding, $toEncoding, $s);
1074 }
1075
1076 /** @return string|false */
1077 private static function mb_internal_trim(string $regex, string $string, ?string $characters, ?string $encoding, string $function)
1078 {
1079 if (null === $encoding) {
1080 $encoding = self::mb_internal_encoding();
1081 } elseif (!self::assertEncoding($encoding, $function.'(): Argument #3 ($encoding) must be a valid encoding, "%s" given')) {
1082 return false;
1083 }
1084
1085 if ('' === $characters) {
1086 return null === $encoding ? $string : self::mb_convert_encoding($string, $encoding);
1087 }
1088
1089 if ('UTF-8' === $encoding) {
1090 $encoding = null;
1091 if (!preg_match('//u', $string)) {
1092 $string = @self::iconv('UTF-8', 'UTF-8', $string);
1093 }
1094 if (null !== $characters && !preg_match('//u', $characters)) {
1095 $characters = @self::iconv('UTF-8', 'UTF-8', $characters);
1096 }
1097 } else {
1098 $string = self::iconv($encoding, 'UTF-8', $string);
1099
1100 if (null !== $characters) {
1101 $characters = self::iconv($encoding, 'UTF-8', $characters);
1102 }
1103 }
1104
1105 if (null === $characters) {
1106 $characters = "\\0 \f\n\r\t\v\u{00A0}\u{1680}\u{2000}\u{2001}\u{2002}\u{2003}\u{2004}\u{2005}\u{2006}\u{2007}\u{2008}\u{2009}\u{200A}\u{2028}\u{2029}\u{202F}\u{205F}\u{3000}\u{0085}\u{180E}";
1107 } else {
1108 $characters = preg_quote($characters);
1109 }
1110
1111 $string = preg_replace(\sprintf($regex, $characters), '', $string);
1112
1113 if (null === $encoding) {
1114 return $string;
1115 }
1116
1117 return self::iconv('UTF-8', $encoding, $string);
1118 }
1119
1120 private static function assertEncoding(string $encoding, string $errorFormat): bool
1121 {
1122 try {
1123 $validEncoding = @self::mb_check_encoding('', $encoding);
1124 } catch (\ValueError $e) {
1125 throw new \ValueError(\sprintf($errorFormat, $encoding));
1126 }
1127
1128 if (!$validEncoding) {
1129 if (80000 > \PHP_VERSION_ID) {
1130 trigger_error(\sprintf($errorFormat, $encoding), \E_USER_WARNING);
1131 } else {
1132 throw new \ValueError(\sprintf($errorFormat, $encoding));
1133 }
1134 }
1135
1136 return $validEncoding;
1137 }
1138 }
1139