PluginProbe
WindPress – Tailwind CSS integration for WordPress / 3.0.13
WindPress – Tailwind CSS integration for WordPress v3.0.13
3.2.89 3.2.88 3.2.87 3.2.86 3.2.85 3.2.84 3.2.83 3.2.82 3.2.81 trunk 3.0.0 3.0.1 3.0.10 3.0.11 3.0.12 3.0.13 3.0.14 3.0.15 3.0.16 3.0.17 3.0.2 3.0.3 3.0.4 3.0.5 3.0.6 All 143 releases
windpress / vendor / masterminds / html5 / src / HTML5 / Parser / UTF8Utils.php

UTF8Utils.php in WindPress – Tailwind CSS integration for WordPress 3.0.13, at vendor/masterminds/html5/src/HTML5/Parser/UTF8Utils.php

160 lines 7.1 KB
No matching file
Up and down to move Enter to open Esc to close
Raw Download Zip
1 <?php
2
3 namespace WindPressDeps\Masterminds\HTML5\Parser;
4
5 /*
6 Portions based on code from html5lib files with the following copyright:
7
8 Copyright 2009 Geoffrey Sneddon <http://gsnedders.com/>
9
10 Permission is hereby granted, free of charge, to any person obtaining a
11 copy of this software and associated documentation files (the
12 "Software"), to deal in the Software without restriction, including
13 without limitation the rights to use, copy, modify, merge, publish,
14 distribute, sublicense, and/or sell copies of the Software, and to
15 permit persons to whom the Software is furnished to do so, subject to
16 the following conditions:
17
18 The above copyright notice and this permission notice shall be included
19 in all copies or substantial portions of the Software.
20
21 THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS
22 OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
23 MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.
24 IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY
25 CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT,
26 TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE
27 SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
28 */
29 use WindPressDeps\Masterminds\HTML5\Exception;
30 class UTF8Utils
31 {
32 /**
33 * The Unicode replacement character.
34 */
35 const FFFD = "";
36 /**
37 * Count the number of characters in a string.
38 * UTF-8 aware. This will try (in order) iconv, MB, and finally a custom counter.
39 *
40 * @param string $string
41 *
42 * @return int
43 */
44 public static function countChars($string)
45 {
46 // Get the length for the string we need.
47 if (\function_exists('mb_strlen')) {
48 return \mb_strlen($string, 'utf-8');
49 }
50 if (\function_exists('iconv_strlen')) {
51 return \iconv_strlen($string, 'utf-8');
52 }
53 $count = \count_chars($string);
54 // 0x80 = 0x7F - 0 + 1 (one added to get inclusive range)
55 // 0x33 = 0xF4 - 0x2C + 1 (one added to get inclusive range)
56 return \array_sum(\array_slice($count, 0, 0x80)) + \array_sum(\array_slice($count, 0xc2, 0x33));
57 }
58 /**
59 * Convert data from the given encoding to UTF-8.
60 *
61 * This has not yet been tested with charactersets other than UTF-8.
62 * It should work with ISO-8859-1/-13 and standard Latin Win charsets.
63 *
64 * @param string $data The data to convert
65 * @param string $encoding A valid encoding. Examples: http://www.php.net/manual/en/mbstring.supported-encodings.php
66 *
67 * @return string
68 */
69 public static function convertToUTF8($data, $encoding = 'UTF-8')
70 {
71 /*
72 * From the HTML5 spec: Given an encoding, the bytes in the input stream must be converted
73 * to Unicode characters for the tokeniser, as described by the rules for that encoding,
74 * except that the leading U+FEFF BYTE ORDER MARK character, if any, must not be stripped
75 * by the encoding layer (it is stripped by the rule below). Bytes or sequences of bytes
76 * in the original byte stream that could not be converted to Unicode characters must be
77 * converted to U+FFFD REPLACEMENT CHARACTER code points.
78 */
79 // mb_convert_encoding is chosen over iconv because of a bug. The best
80 // details for the bug are on http://us1.php.net/manual/en/function.iconv.php#108643
81 // which contains links to the actual but reports as well as work around
82 // details.
83 if (\function_exists('mb_convert_encoding')) {
84 // mb library has the following behaviors:
85 // - UTF-16 surrogates result in false.
86 // - Overlongs and outside Plane 16 result in empty strings.
87 // Before we run mb_convert_encoding we need to tell it what to do with
88 // characters it does not know. This could be different than the parent
89 // application executing this library so we store the value, change it
90 // to our needs, and then change it back when we are done. This feels
91 // a little excessive and it would be great if there was a better way.
92 $save = \mb_substitute_character();
93 \mb_substitute_character('none');
94 $data = \mb_convert_encoding($data, 'UTF-8', $encoding);
95 \mb_substitute_character($save);
96 } elseif (\function_exists('iconv') && 'auto' !== $encoding) {
97 // fprintf(STDOUT, "iconv found\n");
98 // iconv has the following behaviors:
99 // - Overlong representations are ignored.
100 // - Beyond Plane 16 is replaced with a lower char.
101 // - Incomplete sequences generate a warning.
102 $data = @\iconv($encoding, 'UTF-8//IGNORE', $data);
103 } else {
104 throw new Exception('Not implemented, please install mbstring or iconv');
105 }
106 /*
107 * One leading U+FEFF BYTE ORDER MARK character must be ignored if any are present.
108 */
109 if ("" === \substr($data, 0, 3)) {
110 $data = \substr($data, 3);
111 }
112 return $data;
113 }
114 /**
115 * Checks for Unicode code points that are not valid in a document.
116 *
117 * @param string $data A string to analyze
118 *
119 * @return array An array of (string) error messages produced by the scanning
120 */
121 public static function checkForIllegalCodepoints($data)
122 {
123 // Vestigal error handling.
124 $errors = array();
125 /*
126 * All U+0000 null characters in the input must be replaced by U+FFFD REPLACEMENT CHARACTERs.
127 * Any occurrences of such characters is a parse error.
128 */
129 for ($i = 0, $count = \substr_count($data, "\x00"); $i < $count; ++$i) {
130 $errors[] = 'null-character';
131 }
132 /*
133 * Any occurrences of any characters in the ranges U+0001 to U+0008, U+000B, U+000E to U+001F, U+007F
134 * to U+009F, U+D800 to U+DFFF , U+FDD0 to U+FDEF, and characters U+FFFE, U+FFFF, U+1FFFE, U+1FFFF,
135 * U+2FFFE, U+2FFFF, U+3FFFE, U+3FFFF, U+4FFFE, U+4FFFF, U+5FFFE, U+5FFFF, U+6FFFE, U+6FFFF, U+7FFFE,
136 * U+7FFFF, U+8FFFE, U+8FFFF, U+9FFFE, U+9FFFF, U+AFFFE, U+AFFFF, U+BFFFE, U+BFFFF, U+CFFFE, U+CFFFF,
137 * U+DFFFE, U+DFFFF, U+EFFFE, U+EFFFF, U+FFFFE, U+FFFFF, U+10FFFE, and U+10FFFF are parse errors.
138 * (These are all control characters or permanently undefined Unicode characters.)
139 */
140 // Check PCRE is loaded.
141 $count = \preg_match_all('/(?:
142 [\\x01-\\x08\\x0B\\x0E-\\x1F\\x7F] # U+0001 to U+0008, U+000B, U+000E to U+001F and U+007F
143 |
144 \\xC2[\\x80-\\x9F] # U+0080 to U+009F
145 |
146 \\xED(?:\\xA0[\\x80-\\xFF]|[\\xA1-\\xBE][\\x00-\\xFF]|\\xBF[\\x00-\\xBF]) # U+D800 to U+DFFFF
147 |
148 \\xEF\\xB7[\\x90-\\xAF] # U+FDD0 to U+FDEF
149 |
150 \\xEF\\xBF[\\xBE\\xBF] # U+FFFE and U+FFFF
151 |
152 [\\xF0-\\xF4][\\x8F-\\xBF]\\xBF[\\xBE\\xBF] # U+nFFFE and U+nFFFF (1 <= n <= 10_{16})
153 )/x', $data, $matches);
154 for ($i = 0; $i < $count; ++$i) {
155 $errors[] = 'invalid-codepoint';
156 }
157 return $errors;
158 }
159 }
160