| 1 |
<?php |
| 2 |
/** |
| 3 |
* Shared length handling for rendered SEO descriptions. |
| 4 |
* |
| 5 |
* @package ThinkRank |
| 6 |
* @subpackage Core |
| 7 |
* @since 2.7.0 |
| 8 |
*/ |
| 9 |
|
| 10 |
declare(strict_types=1); |
| 11 |
|
| 12 |
namespace ThinkRank\Core; |
| 13 |
|
| 14 |
if (!defined('ABSPATH')) { |
| 15 |
exit; |
| 16 |
} |
| 17 |
|
| 18 |
/** |
| 19 |
* Trims SEO text to the length search engines display. |
| 20 |
* |
| 21 |
* Every path that renders a description must measure and cut through here |
| 22 |
* rather than carrying its own copy, because the obvious spelling of this is |
| 23 |
* wrong in two independent ways and both are invisible in English: |
| 24 |
* |
| 25 |
* - `strlen()` counts BYTES. UTF-8 Thai and CJK run three bytes per character, |
| 26 |
* so a 54-character description already trips a 160-"character" gate it is |
| 27 |
* nowhere near. |
| 28 |
* - `wp_trim_words()` does not always count words. WordPress reads the unit |
| 29 |
* from a per-locale gettext string, and `th`, `ja` and `zh_*` set it to |
| 30 |
* `characters_excluding_spaces` — so `wp_trim_words($text, 25)` keeps 25 |
| 31 |
* words in English and 25 *characters* in Thai. |
| 32 |
* |
| 33 |
* Together they cut Thai meta descriptions to roughly 25 characters while the |
| 34 |
* same stored value rendered in full in og:description, which reached the page |
| 35 |
* by a different path (#687). |
| 36 |
* |
| 37 |
* @since 2.7.0 |
| 38 |
*/ |
| 39 |
class Seo_Text { |
| 40 |
|
| 41 |
/** |
| 42 |
* Characters search engines display for a meta description. |
| 43 |
* |
| 44 |
* @since 2.7.0 |
| 45 |
* @var int |
| 46 |
*/ |
| 47 |
public const MAX_LENGTH = 160; |
| 48 |
|
| 49 |
/** |
| 50 |
* Characters search engines display for a title. |
| 51 |
* |
| 52 |
* @since 2.7.0 |
| 53 |
* @var int |
| 54 |
*/ |
| 55 |
public const TITLE_MAX_LENGTH = 60; |
| 56 |
|
| 57 |
/** |
| 58 |
* Whether this locale's word-count unit is actually words. |
| 59 |
* |
| 60 |
* WordPress reads the unit from a per-locale gettext string, and `th`, |
| 61 |
* `ja` and `zh_*` set it to `characters_excluding_spaces`. Anything that |
| 62 |
* passes a *word* cap to wp_trim_words() therefore has to ask first, or it |
| 63 |
* silently becomes a character cap roughly six times tighter (#687). |
| 64 |
* |
| 65 |
* wp_get_word_count_type() only exists from WP 6.2; the plugin supports |
| 66 |
* 6.0, so fall back to the same gettext string core reads. |
| 67 |
* |
| 68 |
* @since 2.7.0 |
| 69 |
* @return bool |
| 70 |
*/ |
| 71 |
public static function locale_counts_words(): bool { |
| 72 |
if (function_exists('wp_get_word_count_type')) { |
| 73 |
return 'words' === wp_get_word_count_type(); |
| 74 |
} |
| 75 |
|
| 76 |
// Core's own string, in core's text domain — this is the value |
| 77 |
// WP_Locale::get_word_count_type() returns from 6.2 onward. Reading it |
| 78 |
// from 'thinkrank' would look up a translation we do not ship and |
| 79 |
// always answer 'words', quietly disabling the check. |
| 80 |
// phpcs:ignore WordPress.WP.I18n.TextDomainMismatch -- deliberately core's string. |
| 81 |
return 'words' === _x('words', 'Word count type. Do not translate!', 'default'); |
| 82 |
} |
| 83 |
|
| 84 |
/** |
| 85 |
* The text a reader sees for a resolved title or description. |
| 86 |
* |
| 87 |
* Pattern_Resolver hands back what the page will print, and that is still |
| 88 |
* HTML: get_the_title() runs wptexturize, so "Foo & Bar" arrives as |
| 89 |
* `Foo & Bar`, while the same words typed into the SEO title field |
| 90 |
* arrive as `Foo & Bar` or a bare `&`. All three render as one `<title>`. |
| 91 |
* Anything that compares, measures or displays the value as text has to |
| 92 |
* look at it after the browser would have decoded it, or it groups two |
| 93 |
* identical titles apart, counts `&` as six characters, and shows the |
| 94 |
* entity to the user. |
| 95 |
* |
| 96 |
* @since 2.10.0 |
| 97 |
* |
| 98 |
* @param string $value Resolved title or description. |
| 99 |
* @return string The same value with HTML entities decoded. |
| 100 |
*/ |
| 101 |
public static function as_displayed(string $value): string { |
| 102 |
return html_entity_decode($value, ENT_QUOTES | ENT_HTML5, 'UTF-8'); |
| 103 |
} |
| 104 |
|
| 105 |
/** |
| 106 |
* Trim a description to a character budget, multibyte-safe. |
| 107 |
* |
| 108 |
* Cuts on a word boundary when one is available inside the budget, so the |
| 109 |
* result does not end mid-word; falls back to a hard character cut for |
| 110 |
* scripts that do not use spaces (CJK, Thai), where a word-boundary search |
| 111 |
* would find nothing and return the string untouched. |
| 112 |
* |
| 113 |
* Lifted from Author_Archives_Manager, which has carried the only correct |
| 114 |
* copy since 2.2.0 while four other paths kept the broken pattern. |
| 115 |
* |
| 116 |
* @since 2.7.0 |
| 117 |
* |
| 118 |
* @param string $description Description text. |
| 119 |
* @param int $limit Maximum length in characters, ellipsis included. |
| 120 |
* @return string |
| 121 |
*/ |
| 122 |
public static function trim_to_length(string $description, int $limit = self::MAX_LENGTH): string { |
| 123 |
if ($limit <= 0) { |
| 124 |
return ''; |
| 125 |
} |
| 126 |
|
| 127 |
if (mb_strlen($description) <= $limit) { |
| 128 |
return $description; |
| 129 |
} |
| 130 |
|
| 131 |
// Reserve one character for the ellipsis. |
| 132 |
$budget = $limit - 1; |
| 133 |
$cut = mb_substr($description, 0, $budget); |
| 134 |
$last_gap = mb_strrpos($cut, ' '); |
| 135 |
|
| 136 |
// Only honour a word boundary that is not absurdly early — otherwise a |
| 137 |
// long unbroken token would collapse the description to a few chars. |
| 138 |
if (false !== $last_gap && $last_gap > (int) ($budget * 0.6)) { |
| 139 |
$cut = mb_substr($cut, 0, $last_gap); |
| 140 |
} |
| 141 |
|
| 142 |
return rtrim($cut) . '…'; |
| 143 |
} |
| 144 |
|
| 145 |
/** |
| 146 |
* Make text fit to appear in JSON-LD. |
| 147 |
* |
| 148 |
* JSON-LD is not HTML, so an HTML entity in it is not decoded by anything |
| 149 |
* downstream: `&` was published to answer engines literally. And the |
| 150 |
* excerpt path appends core's trimming marker, so descriptions arrived |
| 151 |
* ending in `[…]`, a truncation artefact presented as the page's own |
| 152 |
* summary. |
| 153 |
* |
| 154 |
* Lives here rather than in the schema class that introduced it (#766) |
| 155 |
* because two producers build description nodes: the automatic |
| 156 |
* Global_SEO_Schema_Output and the Schema Manager's Schema_Builder. Only |
| 157 |
* the first normalised, so a deployed node, which outranks the automatic |
| 158 |
* one, published the raw entity again. |
| 159 |
* |
| 160 |
* @since 2.10.0 |
| 161 |
* |
| 162 |
* @param string $text Raw text. |
| 163 |
* @return string |
| 164 |
*/ |
| 165 |
public static function normalize_schema_text(string $text): string { |
| 166 |
if ('' === trim($text)) { |
| 167 |
return ''; |
| 168 |
} |
| 169 |
|
| 170 |
$text = self::decode_schema_entities(wp_strip_all_tags($text)); |
| 171 |
|
| 172 |
// Core's excerpt marker, in both its entity and literal forms, with or |
| 173 |
// without the surrounding brackets it is normally wrapped in. |
| 174 |
$text = (string) preg_replace( |
| 175 |
'/\s*(\[\s*(\x{2026}|\.\.\.)\s*\]|\x{2026})\s*$/u', |
| 176 |
'', |
| 177 |
$text |
| 178 |
); |
| 179 |
|
| 180 |
return trim((string) preg_replace('/\s+/u', ' ', $text)); |
| 181 |
} |
| 182 |
|
| 183 |
/** |
| 184 |
* Decode HTML entities in text bound for JSON-LD, and nothing else. |
| 185 |
* |
| 186 |
* The narrow half of normalize_schema_text(), for values that must keep |
| 187 |
* their exact shape otherwise: a stored snapshot already truncated with an |
| 188 |
* ellipsis would lose it to the excerpt-marker strip. |
| 189 |
* |
| 190 |
* @since 2.10.0 |
| 191 |
* |
| 192 |
* @param string $text Text that may carry HTML entities. |
| 193 |
* @return string |
| 194 |
*/ |
| 195 |
public static function decode_schema_entities(string $text): string { |
| 196 |
// Twice: a description that has been through an escaping pass already |
| 197 |
// (core stores `&amp;` for a literal `&` in some paths) would |
| 198 |
// otherwise still carry an entity after one decode. Decoding an |
| 199 |
// already-plain string is a no-op, so this is safe to repeat. |
| 200 |
$text = html_entity_decode($text, ENT_QUOTES | ENT_HTML5, 'UTF-8'); |
| 201 |
|
| 202 |
return html_entity_decode($text, ENT_QUOTES | ENT_HTML5, 'UTF-8'); |
| 203 |
} |
| 204 |
|
| 205 |
/** |
| 206 |
* Apply a WORD cap that stays a word cap. |
| 207 |
* |
| 208 |
* wp_trim_words() reads its unit from the locale, so `$words` silently |
| 209 |
* becomes a *character* cap on th/ja/zh_* — roughly six times tighter than |
| 210 |
* intended. Callers that mean "about N words" go through here: word-counting |
| 211 |
* locales keep wp_trim_words() byte for byte, and the rest get a character |
| 212 |
* budget instead of a mangled one (#687). |
| 213 |
* |
| 214 |
* @since 2.7.0 |
| 215 |
* |
| 216 |
* @param string $text Text to trim. |
| 217 |
* @param int $words Word cap, honoured only where words are the unit. |
| 218 |
* @param string $more Appended by wp_trim_words() when it trims. |
| 219 |
* @param int $fallback Character budget used where words are not the unit. |
| 220 |
* @return string |
| 221 |
*/ |
| 222 |
public static function trim_words( |
| 223 |
string $text, |
| 224 |
int $words, |
| 225 |
string $more = '...', |
| 226 |
int $fallback = self::MAX_LENGTH |
| 227 |
): string { |
| 228 |
if (!self::locale_counts_words()) { |
| 229 |
return self::trim_to_length($text, $fallback); |
| 230 |
} |
| 231 |
|
| 232 |
return wp_trim_words($text, $words, $more); |
| 233 |
} |
| 234 |
} |
| 235 |
|