PluginProbe
BeyondWords – AI audio for publishers / 7.2.0
BeyondWords – AI audio for publishers v7.2.0
7.2.0 7.1.0 trunk 4.0.0 4.0.1 4.0.2 4.0.3 4.0.4 4.0.5 4.0.6 4.1.0 4.1.1 4.1.2 4.2.0 4.2.1 4.2.2 4.2.3 4.2.4 4.3.0 4.4.0 4.5.0 4.5.1 4.6.0 4.6.1 4.6.2 All 44 releases
← All changes | vendor/symfony/polyfill-mbstring/Mbstring.php +230 -39 4.5.0 → 7.2.0 View file →
@@ -47,8 +47,13 @@
47 47 * - mb_strripos - Finds position of last occurrence of a string within another, case insensitive
48 48 * - mb_strstr - Finds first occurrence of a string within another
49 49 * - mb_strwidth - Return width of string
50 50 * - mb_substr_count - Count the number of substring occurrences
51 + * - mb_ucfirst - Make a string's first character uppercase
52 + * - mb_lcfirst - Make a string's first character lowercase
53 + * - mb_trim - Strip whitespace (or other characters) from the beginning and end of a string
54 + * - mb_ltrim - Strip whitespace (or other characters) from the beginning of a string
55 + * - mb_rtrim - Strip whitespace (or other characters) from the end of a string
51 56 *
52 57 * Not implemented:
53 58 * - mb_convert_kana - Convert "kana" one from another ("zen-kaku", "han-kaku" and more)
54 59 * - mb_ereg_* - Regular expression with multibyte support
@@ -76,11 +81,21 @@
76 81
77 82 private static $encodingList = ['ASCII', 'UTF-8'];
78 83 private static $language = 'neutral';
79 84 private static $internalEncoding = 'UTF-8';
85 + private static $iconvSupportsIgnore;
80 86
81 87 public static function mb_convert_encoding($s, $toEncoding, $fromEncoding = null)
82 88 {
89 + if (\is_array($s)) {
90 + $r = [];
91 + foreach ($s as $str) {
92 + $r[] = self::mb_convert_encoding($str, $toEncoding, $fromEncoding);
93 + }
94 +
95 + return $r;
96 + }
97 +
83 98 if (\is_array($fromEncoding) || (null !== $fromEncoding && false !== strpos($fromEncoding, ','))) {
84 99 $fromEncoding = self::mb_detect_encoding($s, $fromEncoding);
85 100 } else {
86 101 $fromEncoding = self::getEncoding($fromEncoding);
@@ -101,9 +116,9 @@
101 116 if ('HTML-ENTITIES' === $fromEncoding || 'HTML' === $fromEncoding) {
102 117 $fromEncoding = 'Windows-1252';
103 118 }
104 119 if ('UTF-8' !== $fromEncoding) {
105 - $s = iconv($fromEncoding, 'UTF-8//IGNORE', $s);
120 + $s = self::iconv($fromEncoding, 'UTF-8', $s);
106 121 }
107 122
108 123 return preg_replace_callback('/[\x80-\xFF]+/', [__CLASS__, 'html_encoding_callback'], $s);
109 124 }
@@ -108,19 +123,48 @@
108 123 return preg_replace_callback('/[\x80-\xFF]+/', [__CLASS__, 'html_encoding_callback'], $s);
109 124 }
110 125
111 126 if ('HTML-ENTITIES' === $fromEncoding) {
112 - $s = html_entity_decode($s, \ENT_COMPAT, 'UTF-8');
127 + $decodeControlChars = static function ($m) {
128 + $code = '' !== ($m[2] ?? '') ? hexdec($m[2]) : (int) $m[1];
129 +
130 + if ($code < 32 || 127 === $code) {
131 + return \chr($code);
132 + }
133 + if (128 <= $code && $code <= 159) {
134 + return "\xC2".\chr(0x80 | ($code & 0x3F));
135 + }
136 +
137 + return $m[0];
138 + };
139 +
140 + if (\PHP_VERSION_ID >= 70400) {
141 + $s = html_entity_decode($s, \ENT_QUOTES, 'UTF-8');
142 + // html_entity_decode() leaves numeric entities for C0/C1 control
143 + // characters as-is (HTML spec), but mb_convert_encoding() decodes
144 + // them. Catch what html_entity_decode() missed.
145 + if (false !== strpos($s, '&#')) {
146 + $s = preg_replace_callback('/&#(?:0*([0-9]++)|[xX]0*([0-9a-fA-F]++));/', $decodeControlChars, $s);
147 + }
148 + } else {
149 + // PHP < 7.4: html_entity_decode() truncates strings at NUL bytes,
150 + // so decode the control character entities first then call
151 + // html_entity_decode() on each NUL-delimited chunk independently.
152 + $s = preg_replace_callback('/&#(?:0*([0-9]++)|[xX]0*([0-9a-fA-F]++));/', $decodeControlChars, $s);
153 + $s = implode("\0", array_map(static function ($chunk) {
154 + return html_entity_decode($chunk, \ENT_QUOTES, 'UTF-8');
155 + }, explode("\0", $s)));
156 + }
113 157 $fromEncoding = 'UTF-8';
114 158 }
115 159
116 - return iconv($fromEncoding, $toEncoding.'//IGNORE', $s);
160 + return self::iconv($fromEncoding, $toEncoding, $s);
117 161 }
118 162
119 163 public static function mb_convert_variables($toEncoding, $fromEncoding, &...$vars)
120 164 {
121 165 $ok = true;
122 - array_walk_recursive($vars, function (&$v) use (&$ok, $toEncoding, $fromEncoding) {
166 + array_walk_recursive($vars, static function (&$v) use (&$ok, $toEncoding, $fromEncoding) {
123 167 if (false === $v = self::mb_convert_encoding($v, $toEncoding, $fromEncoding)) {
124 168 $ok = false;
125 169 }
126 170 });
@@ -165,12 +209,12 @@
165 209
166 210 if ('UTF-8' === $encoding) {
167 211 $encoding = null;
168 212 if (!preg_match('//u', $s)) {
169 - $s = @iconv('UTF-8', 'UTF-8//IGNORE', $s);
213 + $s = @self::iconv('UTF-8', 'UTF-8', $s);
170 214 }
171 215 } else {
172 - $s = iconv($encoding, 'UTF-8//IGNORE', $s);
216 + $s = self::iconv($encoding, 'UTF-8', $s);
173 217 }
174 218
175 219 $cnt = floor(\count($convmap) / 4) * 4;
176 220
@@ -179,9 +223,9 @@
179 223 $convmap[$i] += $convmap[$i + 2];
180 224 $convmap[$i + 1] += $convmap[$i + 2];
181 225 }
182 226
183 - $s = preg_replace_callback('/&#(?:0*([0-9]+)|x0*([0-9a-fA-F]+))(?!&);?/', function (array $m) use ($cnt, $convmap) {
227 + $s = preg_replace_callback('/&#(?:0*([0-9]+)|x0*([0-9a-fA-F]+))'.(\PHP_VERSION_ID >= 80200 ? '' : '(?!&)').';?/', static function (array $m) use ($cnt, $convmap) {
184 228 $c = isset($m[2]) ? (int) hexdec($m[2]) : $m[1];
185 229 for ($i = 0; $i < $cnt; $i += 4) {
186 230 if ($c >= $convmap[$i] && $c <= $convmap[$i + 1]) {
187 231 return self::mb_chr($c - $convmap[$i + 2]);
@@ -194,9 +238,9 @@
194 238 if (null === $encoding) {
195 239 return $s;
196 240 }
197 241
198 - return iconv('UTF-8', $encoding.'//IGNORE', $s);
242 + return self::iconv('UTF-8', $encoding, $s);
199 243 }
200 244
201 245 public static function mb_encode_numericentity($s, $convmap, $encoding = null, $is_hex = false)
202 246 {
@@ -231,12 +275,12 @@
231 275
232 276 if ('UTF-8' === $encoding) {
233 277 $encoding = null;
234 278 if (!preg_match('//u', $s)) {
235 - $s = @iconv('UTF-8', 'UTF-8//IGNORE', $s);
279 + $s = @self::iconv('UTF-8', 'UTF-8', $s);
236 280 }
237 281 } else {
238 - $s = iconv($encoding, 'UTF-8//IGNORE', $s);
282 + $s = self::iconv($encoding, 'UTF-8', $s);
239 283 }
240 284
241 285 static $ulenMask = ["\xC0" => 2, "\xD0" => 2, "\xE0" => 3, "\xF0" => 4];
242 286
@@ -253,9 +297,9 @@
253 297
254 298 for ($j = 0; $j < $cnt; $j += 4) {
255 299 if ($c >= $convmap[$j] && $c <= $convmap[$j + 1]) {
256 300 $cOffset = ($c + $convmap[$j + 2]) & $convmap[$j + 3];
257 - $result .= $is_hex ? sprintf('&#x%X;', $cOffset) : '&#'.$cOffset.';';
301 + $result .= $is_hex ? \sprintf('&#x%X;', $cOffset) : '&#'.$cOffset.';';
258 302 continue 2;
259 303 }
260 304 }
261 305 $result .= $uchr;
@@ -264,9 +308,9 @@
264 308 if (null === $encoding) {
265 309 return $result;
266 310 }
267 311
268 - return iconv('UTF-8', $encoding.'//IGNORE', $result);
312 + return self::iconv('UTF-8', $encoding, $result);
269 313 }
270 314
271 315 public static function mb_convert_case($s, $mode, $encoding = null)
272 316 {
@@ -279,12 +323,12 @@
279 323
280 324 if ('UTF-8' === $encoding) {
281 325 $encoding = null;
282 326 if (!preg_match('//u', $s)) {
283 - $s = @iconv('UTF-8', 'UTF-8//IGNORE', $s);
327 + $s = @self::iconv('UTF-8', 'UTF-8', $s);
284 328 }
285 329 } else {
286 - $s = iconv($encoding, 'UTF-8//IGNORE', $s);
330 + $s = self::iconv($encoding, 'UTF-8', $s);
287 331 }
288 332
289 333 if (\MB_CASE_TITLE == $mode) {
290 334 static $titleRegexp = null;
@@ -346,9 +390,9 @@
346 390 if (null === $encoding) {
347 391 return $s;
348 392 }
349 393
350 - return iconv('UTF-8', $encoding.'//IGNORE', $s);
394 + return self::iconv('UTF-8', $encoding, $s);
351 395 }
352 396
353 397 public static function mb_internal_encoding($encoding = null)
354 398 {
@@ -367,9 +411,9 @@
367 411 if (80000 > \PHP_VERSION_ID) {
368 412 return false;
369 413 }
370 414
371 - throw new \ValueError(sprintf('Argument #1 ($encoding) must be a valid encoding, "%s" given', $encoding));
415 + throw new \ValueError(\sprintf('Argument #1 ($encoding) must be a valid encoding, "%s" given', $encoding));
372 416 }
373 417
374 418 public static function mb_language($lang = null)
375 419 {
@@ -388,9 +432,9 @@
388 432 if (80000 > \PHP_VERSION_ID) {
389 433 return false;
390 434 }
391 435
392 - throw new \ValueError(sprintf('Argument #1 ($language) must be a valid language, "%s" given', $lang));
436 + throw new \ValueError(\sprintf('Argument #1 ($language) must be a valid language, "%s" given', $lang));
393 437 }
394 438
395 439 public static function mb_list_encodings()
396 440 {
@@ -409,14 +453,8 @@
409 453 }
410 454
411 455 public static function mb_check_encoding($var = null, $encoding = null)
412 456 {
413 - if (PHP_VERSION_ID < 70200 && \is_array($var)) {
414 - trigger_error('mb_check_encoding() expects parameter 1 to be string, array given', \E_USER_WARNING);
415 -
416 - return null;
417 - }
418 -
419 457 if (null === $encoding) {
420 458 if (null === $var) {
421 459 return false;
422 460 }
@@ -436,9 +474,8 @@
436 474 }
437 475 }
438 476
439 477 return true;
440 -
441 478 }
442 479
443 480 public static function mb_detect_encoding($str, $encodingList = null, $strict = false)
444 481 {
@@ -511,9 +548,17 @@
511 548 if ('CP850' === $encoding || 'ASCII' === $encoding) {
512 549 return \strlen($s);
513 550 }
514 551
515 - return @iconv_strlen($s, $encoding);
552 + if (false !== $len = @iconv_strlen($s, $encoding)) {
553 + return $len;
554 + }
555 +
556 + if ('UTF-8' !== $encoding) {
557 + return $len;
558 + }
559 +
560 + return preg_match_all('/[\x00-\x7F]|[\xC0-\xDF][\x80-\xBF]?|[\xE0-\xEF][\x80-\xBF]{0,2}|[\xF0-\xF7][\x80-\xBF]{0,3}|[\xF8-\xFB][\x80-\xBF]{0,4}|[\xFC-\xFD][\x80-\xBF]{0,5}|[\x80-\xBF\xFE\xFF]/s', $s);
516 561 }
517 562
518 563 public static function mb_strpos($haystack, $needle, $offset = 0, $encoding = null)
519 564 {
@@ -765,9 +810,9 @@
765 810 {
766 811 $encoding = self::getEncoding($encoding);
767 812
768 813 if ('UTF-8' !== $encoding) {
769 - $s = iconv($encoding, 'UTF-8//IGNORE', $s);
814 + $s = self::iconv($encoding, 'UTF-8', $s);
770 815 }
771 816
772 817 $s = preg_replace('/[\x{1100}-\x{115F}\x{2329}\x{232A}\x{2E80}-\x{303E}\x{3040}-\x{A4CF}\x{AC00}-\x{D7A3}\x{F900}-\x{FAFF}\x{FE10}-\x{FE19}\x{FE30}-\x{FE6F}\x{FF00}-\x{FF60}\x{FFE0}-\x{FFE6}\x{20000}-\x{2FFFD}\x{30000}-\x{3FFFD}]/u', '', $s, -1, $wide);
773 818
@@ -826,31 +871,47 @@
826 871
827 872 return $code;
828 873 }
829 874
830 - public static function mb_str_pad(string $string, int $length, string $pad_string = ' ', int $pad_type = \STR_PAD_RIGHT, string $encoding = null): string
875 + /** @return string|false */
876 + public static function mb_scrub(?string $string, ?string $encoding = null): string
831 877 {
832 - if (!\in_array($pad_type, [\STR_PAD_RIGHT, \STR_PAD_LEFT, \STR_PAD_BOTH], true)) {
833 - throw new \ValueError('mb_str_pad(): Argument #4 ($pad_type) must be STR_PAD_LEFT, STR_PAD_RIGHT, or STR_PAD_BOTH');
878 + if (null === $encoding) {
879 + $encoding = self::mb_internal_encoding();
880 + } elseif (!self::assertEncoding($encoding, 'mb_scrub(): Argument #2 ($encoding) must be a valid encoding, "%s" given')) {
881 + return false;
834 882 }
835 883
884 + return self::mb_convert_encoding((string) $string, $encoding, $encoding);
885 + }
886 +
887 + /** @return string|false */
888 + public static function mb_str_pad(string $string, int $length, string $pad_string = ' ', int $pad_type = \STR_PAD_RIGHT, ?string $encoding = null)
889 + {
836 890 if (null === $encoding) {
837 891 $encoding = self::mb_internal_encoding();
892 + } elseif (!self::assertEncoding($encoding, 'mb_str_pad(): Argument #5 ($encoding) must be a valid encoding, "%s" given')) {
893 + return false;
838 894 }
839 895
840 - try {
841 - $validEncoding = @self::mb_check_encoding('', $encoding);
842 - } catch (\ValueError $e) {
843 - throw new \ValueError(sprintf('mb_str_pad(): Argument #5 ($encoding) must be a valid encoding, "%s" given', $encoding));
896 + if (self::mb_strlen($pad_string, $encoding) <= 0) {
897 + if (\PHP_VERSION_ID < 80000) {
898 + trigger_error('mb_str_pad(): Argument #3 ($pad_string) must be a non-empty string', \E_USER_WARNING);
899 +
900 + return false;
901 + }
902 +
903 + throw new \ValueError('mb_str_pad(): Argument #3 ($pad_string) must be a non-empty string');
844 904 }
845 905
846 - // BC for PHP 7.3 and lower
847 - if (!$validEncoding) {
848 - throw new \ValueError(sprintf('mb_str_pad(): Argument #5 ($encoding) must be a valid encoding, "%s" given', $encoding));
849 - }
906 + if (!\in_array($pad_type, [\STR_PAD_RIGHT, \STR_PAD_LEFT, \STR_PAD_BOTH], true)) {
907 + if (\PHP_VERSION_ID < 80000) {
908 + trigger_error('mb_str_pad(): Argument #4 ($pad_type) must be STR_PAD_LEFT, STR_PAD_RIGHT, or STR_PAD_BOTH', \E_USER_WARNING);
850 909
851 - if (self::mb_strlen($pad_string, $encoding) <= 0) {
852 - throw new \ValueError('mb_str_pad(): Argument #3 ($pad_string) must be a non-empty string');
910 + return false;
911 + }
912 +
913 + throw new \ValueError('mb_str_pad(): Argument #4 ($pad_type) must be STR_PAD_LEFT, STR_PAD_RIGHT, or STR_PAD_BOTH');
853 914 }
854 915
855 916 $paddingRequired = $length - self::mb_strlen($string, $encoding);
856 917
@@ -870,8 +931,56 @@
870 931 return self::mb_substr(str_repeat($pad_string, $leftPaddingLength), 0, $leftPaddingLength, $encoding).$string.self::mb_substr(str_repeat($pad_string, $rightPaddingLength), 0, $rightPaddingLength, $encoding);
871 932 }
872 933 }
873 934
935 + /** @return string|false */
936 + public static function mb_ucfirst(string $string, ?string $encoding = null)
937 + {
938 + if (null === $encoding) {
939 + $encoding = self::mb_internal_encoding();
940 + } elseif (!self::assertEncoding($encoding, 'mb_ucfirst(): Argument #2 ($encoding) must be a valid encoding, "%s" given')) {
941 + return false;
942 + }
943 +
944 + $firstChar = mb_substr($string, 0, 1, $encoding);
945 + $firstChar = mb_convert_case($firstChar, \MB_CASE_TITLE, $encoding);
946 +
947 + return $firstChar.mb_substr($string, 1, null, $encoding);
948 + }
949 +
950 + /** @return string|false */
951 + public static function mb_lcfirst(string $string, ?string $encoding = null)
952 + {
953 + if (null === $encoding) {
954 + $encoding = self::mb_internal_encoding();
955 + } elseif (!self::assertEncoding($encoding, 'mb_lcfirst(): Argument #2 ($encoding) must be a valid encoding, "%s" given')) {
956 + return false;
957 + }
958 +
959 + $firstChar = mb_substr($string, 0, 1, $encoding);
960 + $firstChar = mb_convert_case($firstChar, \MB_CASE_LOWER, $encoding);
961 +
962 + return $firstChar.mb_substr($string, 1, null, $encoding);
963 + }
964 +
965 + /** @return string|false */
966 + public static function mb_trim(string $string, ?string $characters = null, ?string $encoding = null)
967 + {
968 + return self::mb_internal_trim('{^[%s]+|[%1$s]+$}Du', $string, $characters, $encoding, __FUNCTION__);
969 + }
970 +
971 + /** @return string|false */
972 + public static function mb_ltrim(string $string, ?string $characters = null, ?string $encoding = null)
973 + {
974 + return self::mb_internal_trim('{^[%s]+}Du', $string, $characters, $encoding, __FUNCTION__);
975 + }
976 +
977 + /** @return string|false */
978 + public static function mb_rtrim(string $string, ?string $characters = null, ?string $encoding = null)
979 + {
980 + return self::mb_internal_trim('{[%s]+$}Du', $string, $characters, $encoding, __FUNCTION__);
981 + }
982 +
874 983 private static function getSubpart($pos, $part, $haystack, $encoding)
875 984 {
876 985 if (false === $pos) {
877 986 return false;
@@ -941,7 +1050,89 @@
941 1050 if ('UTF8' === $encoding) {
942 1051 return 'UTF-8';
943 1052 }
944 1053
1054 + if ('UTF-32' === $encoding) {
1055 + return 'UTF-32BE';
1056 + }
1057 +
1058 + if ('UTF-16' === $encoding) {
1059 + return 'UTF-16BE';
1060 + }
1061 +
945 1062 return $encoding;
1063 + }
1064 +
1065 + private static function iconv($fromEncoding, $toEncoding, $s)
1066 + {
1067 + if (null === self::$iconvSupportsIgnore) {
1068 + self::$iconvSupportsIgnore = false !== @iconv('UTF-8', 'UTF-8//IGNORE', '');
1069 + }
1070 +
1071 + return self::$iconvSupportsIgnore
1072 + ? iconv($fromEncoding, $toEncoding.'//IGNORE', $s)
1073 + : iconv($fromEncoding, $toEncoding, $s);
1074 + }
1075 +
1076 + /** @return string|false */
1077 + private static function mb_internal_trim(string $regex, string $string, ?string $characters, ?string $encoding, string $function)
1078 + {
1079 + if (null === $encoding) {
1080 + $encoding = self::mb_internal_encoding();
1081 + } elseif (!self::assertEncoding($encoding, $function.'(): Argument #3 ($encoding) must be a valid encoding, "%s" given')) {
1082 + return false;
1083 + }
1084 +
1085 + if ('' === $characters) {
1086 + return null === $encoding ? $string : self::mb_convert_encoding($string, $encoding);
1087 + }
1088 +
1089 + if ('UTF-8' === $encoding) {
1090 + $encoding = null;
1091 + if (!preg_match('//u', $string)) {
1092 + $string = @self::iconv('UTF-8', 'UTF-8', $string);
1093 + }
1094 + if (null !== $characters && !preg_match('//u', $characters)) {
1095 + $characters = @self::iconv('UTF-8', 'UTF-8', $characters);
1096 + }
1097 + } else {
1098 + $string = self::iconv($encoding, 'UTF-8', $string);
1099 +
1100 + if (null !== $characters) {
1101 + $characters = self::iconv($encoding, 'UTF-8', $characters);
1102 + }
1103 + }
1104 +
1105 + if (null === $characters) {
1106 + $characters = "\\0 \f\n\r\t\v\u{00A0}\u{1680}\u{2000}\u{2001}\u{2002}\u{2003}\u{2004}\u{2005}\u{2006}\u{2007}\u{2008}\u{2009}\u{200A}\u{2028}\u{2029}\u{202F}\u{205F}\u{3000}\u{0085}\u{180E}";
1107 + } else {
1108 + $characters = preg_quote($characters);
1109 + }
1110 +
1111 + $string = preg_replace(\sprintf($regex, $characters), '', $string);
1112 +
1113 + if (null === $encoding) {
1114 + return $string;
1115 + }
1116 +
1117 + return self::iconv('UTF-8', $encoding, $string);
1118 + }
1119 +
1120 + private static function assertEncoding(string $encoding, string $errorFormat): bool
1121 + {
1122 + try {
1123 + $validEncoding = @self::mb_check_encoding('', $encoding);
1124 + } catch (\ValueError $e) {
1125 + throw new \ValueError(\sprintf($errorFormat, $encoding));
1126 + }
1127 +
1128 + if (!$validEncoding) {
1129 + if (80000 > \PHP_VERSION_ID) {
1130 + trigger_error(\sprintf($errorFormat, $encoding), \E_USER_WARNING);
1131 + } else {
1132 + throw new \ValueError(\sprintf($errorFormat, $encoding));
1133 + }
1134 + }
1135 +
1136 + return $validEncoding;
946 1137 }
947 1138 }