← All changes
|
jetpack_vendor/automattic/jetpack-waf/src/class-waf-transforms.php
+71
-10
12.2.3
→
16.3
View file →
| @@ -10,16 +10,35 @@ | ||
| 10 | 10 | /** |
| 11 | 11 | * Waf_Transforms class |
| 12 | 12 | */ |
| 13 | 13 | class Waf_Transforms { |
| 14 | + | |
| 14 | 15 | /** |
| 15 | - * Decode a Base64-encoded string. | |
| 16 | + * Decode a Base64-encoded string. This runs the decode without strict mode, to match Modsecurity's 'base64DecodeExt' transform function. | |
| 16 | 17 | * |
| 17 | 18 | * @param string $value value to be decoded. |
| 18 | 19 | * @return string |
| 19 | 20 | */ |
| 21 | + public function base64_decode_ext( $value ) { | |
| 22 | + return base64_decode( $value ); | |
| 23 | + } | |
| 24 | + | |
| 25 | + /** | |
| 26 | + * Characters to match when trimming a string. | |
| 27 | + * Emulates `std::isspace` used by ModSecurity. | |
| 28 | + * | |
| 29 | + * @see https://en.cppreference.com/w/cpp/string/byte/isspace | |
| 30 | + */ | |
| 31 | + const TRIM_CHARS = " \n\r\t\v\f"; | |
| 32 | + | |
| 33 | + /** | |
| 34 | + * Decode a Base64-encoded string. This runs the decode with strict mode, to match Modsecurity's 'base64Decode' transform function. | |
| 35 | + * | |
| 36 | + * @param string $value value to be decoded. | |
| 37 | + * @return string | |
| 38 | + */ | |
| 20 | 39 | public function base64_decode( $value ) { |
| 21 | - return base64_decode( $value ); | |
| 40 | + return base64_decode( $value, true ); | |
| 22 | 41 | } |
| 23 | 42 | |
| 24 | 43 | /** |
| 25 | 44 | * Remove all characters that might escape a command line command |
| @@ -306,9 +325,9 @@ | ||
| 306 | 325 | * @param string $value value to be trimmed. |
| 307 | 326 | * @return string |
| 308 | 327 | */ |
| 309 | 328 | public function trim_left( $value ) { |
| 310 | - return ltrim( $value ); | |
| 329 | + return ltrim( $value, self::TRIM_CHARS ); | |
| 311 | 330 | } |
| 312 | 331 | |
| 313 | 332 | /** |
| 314 | 333 | * Remove whitespace from the right side of the input string. |
| @@ -316,9 +335,9 @@ | ||
| 316 | 335 | * @param string $value value to be trimmed. |
| 317 | 336 | * @return string |
| 318 | 337 | */ |
| 319 | 338 | public function trim_right( $value ) { |
| 320 | - return rtrim( $value ); | |
| 339 | + return rtrim( $value, self::TRIM_CHARS ); | |
| 321 | 340 | } |
| 322 | 341 | |
| 323 | 342 | /** |
| 324 | 343 | * Remove whitespace from both sides of the input string. |
| @@ -326,17 +345,59 @@ | ||
| 326 | 345 | * @param string $value value to be trimmed. |
| 327 | 346 | * @return string |
| 328 | 347 | */ |
| 329 | 348 | public function trim( $value ) { |
| 330 | - return trim( $value ); | |
| 349 | + return trim( $value, self::TRIM_CHARS ); | |
| 331 | 350 | } |
| 332 | 351 | |
| 333 | 352 | /** |
| 334 | - * Convert utf-8 characters to unicode characters | |
| 353 | + * Convert UTF-8 characters to Unicode characters. | |
| 335 | 354 | * |
| 336 | - * @param string $value value to be encoded. | |
| 337 | - * @return string | |
| 355 | + * This function iterates through each character of the input string, checks the ASCII value, | |
| 356 | + * and converts it to its corresponding Unicode representation. It handles characters that are | |
| 357 | + * represented with 1 to 4 bytes in UTF-8. | |
| 358 | + * | |
| 359 | + * @param string $str The string value to be encoded from UTF-8 to Unicode. | |
| 360 | + * @return string The converted string with Unicode representation. | |
| 338 | 361 | */ |
| 339 | - public function utf8_to_unicode( $value ) { | |
| 340 | - return preg_replace( '/\\\u(?=[a-f0-9]{4})/', '%u', substr( json_encode( $value ), 1, -1 ) ); | |
| 362 | + public function utf8_to_unicode( $str ) { | |
| 363 | + $unicodeStr = ''; | |
| 364 | + $strLen = strlen( $str ); | |
| 365 | + $i = 0; | |
| 366 | + | |
| 367 | + // Iterate through each character of the input string. | |
| 368 | + while ( $i < $strLen ) { | |
| 369 | + // Get the ASCII value of the current character. | |
| 370 | + $value = ord( $str[ $i ] ); | |
| 371 | + | |
| 372 | + if ( $value < 128 ) { | |
| 373 | + // If the character is in the ASCII range (0-127), directly add it to the Unicode string. | |
| 374 | + $unicodeStr .= chr( $value ); | |
| 375 | + ++$i; | |
| 376 | + } else { | |
| 377 | + // For characters outside the ASCII range, determine the number of bytes in the UTF-8 representation. | |
| 378 | + $unicodeValue = ''; | |
| 379 | + if ( $value >= 192 && $value <= 223 ) { | |
| 380 | + // For characters represented with 2 bytes in UTF-8. | |
| 381 | + $unicodeValue = ( ord( $str[ $i ] ) & 0x1F ) << 6 | ( ord( $str[ $i + 1 ] ) & 0x3F ); | |
| 382 | + $i += 2; | |
| 383 | + } elseif ( $value >= 224 && $value <= 239 ) { | |
| 384 | + // For characters represented with 3 bytes in UTF-8. | |
| 385 | + $unicodeValue = ( ord( $str[ $i ] ) & 0x0F ) << 12 | ( ord( $str[ $i + 1 ] ) & 0x3F ) << 6 | ( ord( $str[ $i + 2 ] ) & 0x3F ); | |
| 386 | + $i += 3; | |
| 387 | + } elseif ( $value >= 240 && $value <= 247 ) { | |
| 388 | + // For characters represented with 4 bytes in UTF-8. | |
| 389 | + $unicodeValue = ( ord( $str[ $i ] ) & 0x07 ) << 18 | ( ord( $str[ $i + 1 ] ) & 0x3F ) << 12 | ( ord( $str[ $i + 2 ] ) & 0x3F ) << 6 | ( ord( $str[ $i + 3 ] ) & 0x3F ); | |
| 390 | + $i += 4; | |
| 391 | + } else { | |
| 392 | + // If the sequence does not match any known UTF-8 pattern, skip to the next character. | |
| 393 | + ++$i; | |
| 394 | + continue; | |
| 395 | + } | |
| 396 | + // Convert the Unicode value to a formatted string and append it to the Unicode string. | |
| 397 | + $unicodeStr .= sprintf( '%%u%04X', $unicodeValue ); | |
| 398 | + } | |
| 399 | + } | |
| 400 | + | |
| 401 | + return strtolower( $unicodeStr ); | |
| 341 | 402 | } |
| 342 | 403 | } |