| 1 |
<?php |
| 2 |
|
| 3 |
declare (strict_types=1); |
| 4 |
namespace WCPOS\Vendor\GuzzleHttp\Psr7; |
| 5 |
|
| 6 |
/** |
| 7 |
* Syntax predicates and canonicalization helpers for the URI grammar defined |
| 8 |
* by RFC 3986. |
| 9 |
*/ |
| 10 |
final class Rfc3986 |
| 11 |
{ |
| 12 |
private function __construct() |
| 13 |
{ |
| 14 |
} |
| 15 |
/** |
| 16 |
* Sub-delims for use in a regex. |
| 17 |
* |
| 18 |
* @see https://datatracker.ietf.org/doc/html/rfc3986#section-2.2 |
| 19 |
* |
| 20 |
* @internal |
| 21 |
*/ |
| 22 |
public const CHAR_SUB_DELIMS = '!\\$&\'\\(\\)\\*\\+,;='; |
| 23 |
/** |
| 24 |
* Unreserved characters for use in a regex. |
| 25 |
* |
| 26 |
* @see https://datatracker.ietf.org/doc/html/rfc3986#section-2.3 |
| 27 |
* |
| 28 |
* @internal |
| 29 |
*/ |
| 30 |
public const CHAR_UNRESERVED = 'a-zA-Z0-9_\\-\\.~'; |
| 31 |
/** |
| 32 |
* The two hex digits of a percent-encoded octet (the "3A" in "%3A"), for use in a regex. |
| 33 |
* |
| 34 |
* @see https://datatracker.ietf.org/doc/html/rfc3986#section-2.1 |
| 35 |
* |
| 36 |
* @internal |
| 37 |
*/ |
| 38 |
public const HEX_OCTET = '[A-Fa-f0-9]{2}'; |
| 39 |
/** |
| 40 |
* Whether the string is a valid URI scheme. |
| 41 |
* |
| 42 |
* Per RFC 3986 a scheme must start with a letter, followed by letters, |
| 43 |
* digits, `+`, `-`, or `.`. The empty string is also accepted, since a URI |
| 44 |
* reference may omit the scheme. |
| 45 |
* |
| 46 |
* @see https://datatracker.ietf.org/doc/html/rfc3986#section-3.1 |
| 47 |
*/ |
| 48 |
public static function isValidScheme(string $scheme) : bool |
| 49 |
{ |
| 50 |
return $scheme === '' || \preg_match('/^[A-Za-z][A-Za-z0-9.+-]*$/D', $scheme) === 1; |
| 51 |
} |
| 52 |
/** |
| 53 |
* Whether the string is a valid URI host. |
| 54 |
* |
| 55 |
* Per RFC 3986 the host is `IP-literal / IPv4address / reg-name`. An empty |
| 56 |
* host is accepted, since the authority (and thus the host) may be empty. |
| 57 |
* Bracketed values are validated as IPv6 / IPvFuture literals; any other |
| 58 |
* value is rejected if it contains control characters, whitespace, an |
| 59 |
* authority or path delimiter (`/ ? # @ \`), an embedded colon denoting |
| 60 |
* a port, a malformed percent-sequence, or a percent-encoded octet that |
| 61 |
* decodes to one of those rejected bytes, to a bracket, or to `%` itself. |
| 62 |
* |
| 63 |
* @see https://datatracker.ietf.org/doc/html/rfc3986#section-3.2.2 |
| 64 |
*/ |
| 65 |
public static function isValidHost(string $host) : bool |
| 66 |
{ |
| 67 |
if ($host === '') { |
| 68 |
return \true; |
| 69 |
} |
| 70 |
$invalidHost = \preg_match('/[\\x00-\\x20\\x7F\\/\\?#@\\\\]/', $host); |
| 71 |
if ($invalidHost === \false) { |
| 72 |
return \false; |
| 73 |
} |
| 74 |
if ($invalidHost === 1) { |
| 75 |
return \false; |
| 76 |
} |
| 77 |
if (\str_contains($host, '[') || \str_contains($host, ']')) { |
| 78 |
return self::isValidIpLiteralHost($host); |
| 79 |
} |
| 80 |
if (\str_contains($host, ':')) { |
| 81 |
return \false; |
| 82 |
} |
| 83 |
return !\str_contains($host, '%') || self::hasValidHostPercentEncoding($host); |
| 84 |
} |
| 85 |
/** |
| 86 |
* Whether the string is a valid port number (0-65535). |
| 87 |
* |
| 88 |
* RFC 3986 defines the port as `*DIGIT`, which also permits an empty port |
| 89 |
* and has no upper bound. This applies the stricter policy used throughout |
| 90 |
* the library instead: the value must be a non-empty run of digits (leading |
| 91 |
* zeros are accepted and normalized) that resolves to 0-65535. |
| 92 |
* |
| 93 |
* @see https://datatracker.ietf.org/doc/html/rfc3986#section-3.2.3 |
| 94 |
*/ |
| 95 |
public static function isValidPort(string $port) : bool |
| 96 |
{ |
| 97 |
if ($port === '' || !\ctype_digit($port)) { |
| 98 |
return \false; |
| 99 |
} |
| 100 |
$normalized = \ltrim($port, '0'); |
| 101 |
if ($normalized === '') { |
| 102 |
return \true; |
| 103 |
} |
| 104 |
return \strlen($normalized) <= 5 && (int) $normalized <= 0xffff; |
| 105 |
} |
| 106 |
/** |
| 107 |
* Returns the RFC 5952 canonical form of a valid IPv6 address. |
| 108 |
* |
| 109 |
* The address must be a valid textual IPv6 address without brackets and |
| 110 |
* without a zone identifier, such as the inside of an IP-literal accepted |
| 111 |
* by `isValidHost()`. Canonicalization lowercases the hexadecimal fields, |
| 112 |
* suppresses leading zeros, and collapses the longest run of two or more |
| 113 |
* zero fields (the leftmost on a tie) with `::`. Embedded dotted-decimal |
| 114 |
* notation follows the rendering policy of BIND-derived `inet_ntop()` |
| 115 |
* implementations and curl: exactly the IPv4-mapped (`::ffff:0:0/96`) and |
| 116 |
* deprecated IPv4-compatible (`::/96`) layouts use it, while other |
| 117 |
* embedded-IPv4 forms, including translated (NAT64) well-known prefixes |
| 118 |
* such as `64:ff9b::/96` (RFC 6052), serialize in pure hexadecimal fields. |
| 119 |
* |
| 120 |
* Validation is strict and platform-independent: the address is checked |
| 121 |
* against the RFC 3986 `IPv6address` grammar with PHP's |
| 122 |
* `FILTER_VALIDATE_IP` filter and parsed in pure PHP, so spellings that |
| 123 |
* only some platform parsers accept, such as the zero-padded dotted octets |
| 124 |
* in `::ffff:192.168.001.001`, are rejected everywhere. |
| 125 |
* |
| 126 |
* @throws \InvalidArgumentException If the address cannot be parsed. |
| 127 |
* |
| 128 |
* @see https://datatracker.ietf.org/doc/html/rfc5952#section-4 |
| 129 |
*/ |
| 130 |
public static function canonicalizeIpv6(string $address) : string |
| 131 |
{ |
| 132 |
$canonical = self::tryCanonicalizeIpv6($address); |
| 133 |
if ($canonical === null) { |
| 134 |
throw new \InvalidArgumentException('Invalid IPv6 address'); |
| 135 |
} |
| 136 |
return $canonical; |
| 137 |
} |
| 138 |
/** |
| 139 |
* Returns the RFC 5952 canonical form of a valid IPv6 address, or null |
| 140 |
* when the address cannot be parsed. |
| 141 |
* |
| 142 |
* @internal |
| 143 |
*/ |
| 144 |
public static function tryCanonicalizeIpv6(string $address) : ?string |
| 145 |
{ |
| 146 |
// Platform parsers disagree on which spellings are valid: Apple libc |
| 147 |
// and OpenBSD inet_pton() accept zero-padded dotted octets such as |
| 148 |
// "::ffff:192.168.001.001", and macOS additionally accepts and silently |
| 149 |
// strips zone IDs ("fe80::1%eth0"), while glibc, musl, and PHP's own |
| 150 |
// filter reject both. Origin classification built on this helper must |
| 151 |
// fail closed and must not vary by operating system, so the address is |
| 152 |
// validated with the platform-independent FILTER_VALIDATE_IP filter and |
| 153 |
// parsed in pure PHP; no OS parser is consulted. |
| 154 |
if (\filter_var($address, \FILTER_VALIDATE_IP, \FILTER_FLAG_IPV6) === \false) { |
| 155 |
return null; |
| 156 |
} |
| 157 |
$words = self::parseIpv6Words($address); |
| 158 |
if ($words === null) { |
| 159 |
return null; |
| 160 |
} |
| 161 |
// Find the longest run of two or more zero fields; ties keep the |
| 162 |
// leftmost run per RFC 5952 section 4.2.3. |
| 163 |
$bestStart = 0; |
| 164 |
$bestLen = 0; |
| 165 |
$start = -1; |
| 166 |
foreach ($words as $i => $word) { |
| 167 |
if ($word !== 0) { |
| 168 |
$start = -1; |
| 169 |
continue; |
| 170 |
} |
| 171 |
if ($start === -1) { |
| 172 |
$start = $i; |
| 173 |
} |
| 174 |
if ($i - $start + 1 > $bestLen) { |
| 175 |
$bestStart = $start; |
| 176 |
$bestLen = $i - $start + 1; |
| 177 |
} |
| 178 |
} |
| 179 |
if ($bestLen < 2) { |
| 180 |
$bestLen = 0; |
| 181 |
} |
| 182 |
// RFC 5952 section 5: embedded IPv4 notation for IPv4-mapped |
| 183 |
// (::ffff:0:0/96) and IPv4-compatible (::/96) addresses, the same |
| 184 |
// condition BIND-derived inet_ntop() and curl use. bestStart must be |
| 185 |
// zero: a five or six field zero run elsewhere is not an IPv4 prefix. |
| 186 |
$mixed = $bestStart === 0 && ($bestLen === 6 || $bestLen === 5 && $words[5] === 0xffff); |
| 187 |
$groups = []; |
| 188 |
for ($i = 0, $n = $mixed ? 6 : 8; $i < $n; ++$i) { |
| 189 |
$groups[] = \dechex($words[$i]); |
| 190 |
} |
| 191 |
if ($mixed) { |
| 192 |
$groups[] = \sprintf('%d.%d.%d.%d', $words[6] >> 8, $words[6] & 0xff, $words[7] >> 8, $words[7] & 0xff); |
| 193 |
} |
| 194 |
if ($bestLen === 0) { |
| 195 |
return \implode(':', $groups); |
| 196 |
} |
| 197 |
return \implode(':', \array_slice($groups, 0, $bestStart)) . '::' . \implode(':', \array_slice($groups, $bestStart + $bestLen)); |
| 198 |
} |
| 199 |
private static function hasValidHostPercentEncoding(string $host) : bool |
| 200 |
{ |
| 201 |
// Mirror of the raw reg-name policy above for percent-encoded octets: |
| 202 |
// reject malformed sequences (RFC 3986 requires "%" HEXDIG HEXDIG) and |
| 203 |
// octets that decode to bytes the raw grammar rejects - C0 controls, |
| 204 |
// SP, DEL, the delimiters / ? # @ \ [ ], the port colon, and % itself. |
| 205 |
// Octets decoding to any other byte (unreserved, sub-delims, and |
| 206 |
// non-ASCII UTF-8 data) remain accepted. |
| 207 |
$invalidEncoding = \preg_match('/%(?!' . self::HEX_OCTET . ')|%(?:[01][0-9A-Fa-f]|2[035F]|3[AF]|40|5[BCD]|7F)/i', $host); |
| 208 |
return $invalidEncoding === 0; |
| 209 |
} |
| 210 |
private static function isValidIpLiteralHost(string $host) : bool |
| 211 |
{ |
| 212 |
if (!\str_starts_with($host, '[') || !\str_ends_with($host, ']')) { |
| 213 |
return \false; |
| 214 |
} |
| 215 |
$address = \substr($host, 1, -1); |
| 216 |
if (\filter_var($address, \FILTER_VALIDATE_IP, \FILTER_FLAG_IPV6) !== \false) { |
| 217 |
return \true; |
| 218 |
} |
| 219 |
// RFC 6874 IPv6 zone identifiers are intentionally not supported here. |
| 220 |
// Bracketed hosts are validated as IPv6 or IPvFuture only. |
| 221 |
return \preg_match('/^v[0-9a-f]+\\.[' . self::CHAR_UNRESERVED . self::CHAR_SUB_DELIMS . ':]+$/iD', $address) === 1; |
| 222 |
} |
| 223 |
/** |
| 224 |
* Parses a textual IPv6 address into its eight 16-bit words, or null |
| 225 |
* when the text is not a structurally valid address. |
| 226 |
* |
| 227 |
* The grammar enforced here is the RFC 3986 `IPv6address` rule: one to four |
| 228 |
* hexadecimal digits per field, at most one `::` eliding one or more zero |
| 229 |
* fields, and an optional dotted-decimal tail of four octets (0-255, no |
| 230 |
* leading zeros) as the final 32 bits. FILTER_VALIDATE_IP accepts exactly |
| 231 |
* this grammar, so the filter guard in tryCanonicalizeIpv6() and this |
| 232 |
* parser always agree and the null paths here can only fail closed. |
| 233 |
* |
| 234 |
* @return list<int>|null |
| 235 |
*/ |
| 236 |
private static function parseIpv6Words(string $address) : ?array |
| 237 |
{ |
| 238 |
// A dotted-decimal tail is only valid as the final 32 bits, after the |
| 239 |
// final colon. Rewrite it into its two hexadecimal fields so the |
| 240 |
// remainder of the parse handles hexadecimal fields only. |
| 241 |
$dot = \strpos($address, '.'); |
| 242 |
if ($dot !== \false) { |
| 243 |
$colon = \strrpos($address, ':'); |
| 244 |
if ($colon === \false || $colon > $dot) { |
| 245 |
return null; |
| 246 |
} |
| 247 |
$octets = \explode('.', \substr($address, $colon + 1)); |
| 248 |
if (\count($octets) !== 4) { |
| 249 |
return null; |
| 250 |
} |
| 251 |
$bytes = []; |
| 252 |
foreach ($octets as $octet) { |
| 253 |
if ($octet === '' || \strlen($octet) > 3 || !\ctype_digit($octet)) { |
| 254 |
return null; |
| 255 |
} |
| 256 |
if ($octet[0] === '0' && $octet !== '0') { |
| 257 |
return null; |
| 258 |
} |
| 259 |
$byte = (int) $octet; |
| 260 |
if ($byte > 255) { |
| 261 |
return null; |
| 262 |
} |
| 263 |
$bytes[] = $byte; |
| 264 |
} |
| 265 |
$address = \substr($address, 0, $colon + 1) . \dechex($bytes[0] << 8 | $bytes[1]) . ':' . \dechex($bytes[2] << 8 | $bytes[3]); |
| 266 |
} |
| 267 |
$halves = \explode('::', $address); |
| 268 |
if (\count($halves) > 2) { |
| 269 |
return null; |
| 270 |
} |
| 271 |
$head = self::parseHexFields($halves[0]); |
| 272 |
if ($head === null) { |
| 273 |
return null; |
| 274 |
} |
| 275 |
if (\count($halves) === 1) { |
| 276 |
return \count($head) === 8 ? $head : null; |
| 277 |
} |
| 278 |
$tail = self::parseHexFields($halves[1]); |
| 279 |
if ($tail === null) { |
| 280 |
return null; |
| 281 |
} |
| 282 |
// The "::" must elide at least one zero field. |
| 283 |
$elided = 8 - \count($head) - \count($tail); |
| 284 |
if ($elided < 1) { |
| 285 |
return null; |
| 286 |
} |
| 287 |
return \array_merge($head, \array_fill(0, $elided, 0), $tail); |
| 288 |
} |
| 289 |
/** |
| 290 |
* Parses a colon-separated run of 16-bit hexadecimal fields, or null |
| 291 |
* when a field is empty, longer than four digits, or not hexadecimal. |
| 292 |
* |
| 293 |
* @return list<int>|null |
| 294 |
*/ |
| 295 |
private static function parseHexFields(string $fields) : ?array |
| 296 |
{ |
| 297 |
if ($fields === '') { |
| 298 |
return []; |
| 299 |
} |
| 300 |
$words = []; |
| 301 |
foreach (\explode(':', $fields) as $field) { |
| 302 |
if ($field === '' || \strlen($field) > 4 || !\ctype_xdigit($field)) { |
| 303 |
return null; |
| 304 |
} |
| 305 |
$words[] = \intval($field, 16); |
| 306 |
} |
| 307 |
return $words; |
| 308 |
} |
| 309 |
} |
| 310 |
|