| @@ -11,22 +11,67 @@ | ||
| 11 | 11 | */ |
| 12 | 12 | class SlugHelper |
| 13 | 13 | { |
| 14 | 14 | /** |
| 15 | - * Generate a slug from a string | |
| 16 | - * Uses WordPress sanitize_title() function | |
| 17 | - * | |
| 15 | + * Generate a slug from a string. | |
| 16 | + * | |
| 17 | + * Unicode-aware so Cyrillic / CJK / Devanagari / accented Latin survive. | |
| 18 | + * We avoid WordPress's sanitize_title() because for non-Latin input it | |
| 19 | + * percent-encodes the result (e.g. "моя-семья" → "%d0%bc..."). That | |
| 20 | + * percent-encoded form then gets stripped to "-" by `sanitize_text_field()` | |
| 21 | + * further down the BaseRepository write path, so the DB stores just a | |
| 22 | + * dash. Doing the normalisation here in raw Unicode lets the slug pass | |
| 23 | + * through `sanitize_text_field()` unchanged. | |
| 24 | + * | |
| 18 | 25 | * @param string $text The text to convert to a slug |
| 19 | 26 | * @return string A URL-friendly slug |
| 20 | 27 | */ |
| 21 | 28 | public static function generate(string $text): string |
| 22 | 29 | { |
| 23 | - if (empty($text)) { | |
| 30 | + if ($text === '') { | |
| 24 | 31 | return ''; |
| 25 | 32 | } |
| 26 | 33 | |
| 27 | - // Use WordPress's sanitize_title function for consistency | |
| 28 | - return sanitize_title($text); | |
| 34 | + // Decode any percent-encoded UTF-8 first. This matches WordPress core | |
| 35 | + // behaviour (see WP::parse_request / get_page_by_path) — when our | |
| 36 | + // route matchers receive a slug captured from a rewrite rule, it | |
| 37 | + // arrives in its URL-encoded form (e.g. `%D0%BC%D0%BE%D1%8F-...` | |
| 38 | + // for "моя-..."). Without this, the regex below would treat the `%` | |
| 39 | + // as invalid, strip everything, and produce a meaningless byte | |
| 40 | + // string. rawurldecode (instead of urldecode) preserves a literal | |
| 41 | + // `+` — a `+` is not URL-encoding for space in path segments, so we | |
| 42 | + // shouldn't accidentally turn it into one. | |
| 43 | + if (strpos($text, '%') !== false) { | |
| 44 | + $decoded = rawurldecode($text); | |
| 45 | + if ($decoded !== false && $decoded !== '') { | |
| 46 | + $text = $decoded; | |
| 47 | + } | |
| 48 | + } | |
| 49 | + | |
| 50 | + // Lowercase. mb_strtolower handles Cyrillic / Greek / accented Latin | |
| 51 | + // correctly (strtolower is byte-wise and would mangle multibyte). | |
| 52 | + if (function_exists('mb_strtolower')) { | |
| 53 | + $text = mb_strtolower($text, 'UTF-8'); | |
| 54 | + } else { | |
| 55 | + $text = strtolower($text); | |
| 56 | + } | |
| 57 | + | |
| 58 | + // Strip HTML and decode entities so & etc. don't bleed through. | |
| 59 | + $text = wp_strip_all_tags($text); | |
| 60 | + $text = html_entity_decode($text, ENT_QUOTES | ENT_HTML5, 'UTF-8'); | |
| 61 | + | |
| 62 | + // Replace whitespace / underscore runs with a single hyphen. | |
| 63 | + $text = preg_replace('/[\s_]+/u', '-', $text) ?? ''; | |
| 64 | + | |
| 65 | + // Keep letters (any script), digits, and hyphens. \pL + \pN with the | |
| 66 | + // `u` flag matches Unicode letter and number categories. | |
| 67 | + $text = preg_replace('/[^\pL\pN-]+/u', '', $text) ?? ''; | |
| 68 | + | |
| 69 | + // Collapse multiple consecutive hyphens. | |
| 70 | + $text = preg_replace('/-+/u', '-', $text) ?? ''; | |
| 71 | + | |
| 72 | + // Trim leading / trailing hyphens. | |
| 73 | + return trim($text, '-'); | |
| 29 | 74 | } |
| 30 | 75 | |
| 31 | 76 | /** |
| 32 | 77 | * Generate a unique slug by appending a number if needed |