`. * Anything that compares, measures or displays the value as text has to * look at it after the browser would have decoded it, or it groups two * identical titles apart, counts `&` as six characters, and shows the * entity to the user. * * @since 2.10.0 * * @param string $value Resolved title or description. * @return string The same value with HTML entities decoded. */ public static function as_displayed(string $value): string { return html_entity_decode($value, ENT_QUOTES | ENT_HTML5, 'UTF-8'); } /** * Trim a description to a character budget, multibyte-safe. * * Cuts on a word boundary when one is available inside the budget, so the * result does not end mid-word; falls back to a hard character cut for * scripts that do not use spaces (CJK, Thai), where a word-boundary search * would find nothing and return the string untouched. * * Lifted from Author_Archives_Manager, which has carried the only correct * copy since 2.2.0 while four other paths kept the broken pattern. * * @since 2.7.0 * * @param string $description Description text. * @param int $limit Maximum length in characters, ellipsis included. * @return string */ public static function trim_to_length(string $description, int $limit = self::MAX_LENGTH): string { if ($limit <= 0) { return ''; } if (mb_strlen($description) <= $limit) { return $description; } // Reserve one character for the ellipsis. $budget = $limit - 1; $cut = mb_substr($description, 0, $budget); $last_gap = mb_strrpos($cut, ' '); // Only honour a word boundary that is not absurdly early — otherwise a // long unbroken token would collapse the description to a few chars. if (false !== $last_gap && $last_gap > (int) ($budget * 0.6)) { $cut = mb_substr($cut, 0, $last_gap); } return rtrim($cut) . '…'; } /** * Make text fit to appear in JSON-LD. * * JSON-LD is not HTML, so an HTML entity in it is not decoded by anything * downstream: `&` was published to answer engines literally. And the * excerpt path appends core's trimming marker, so descriptions arrived * ending in `[…]`, a truncation artefact presented as the page's own * summary. * * Lives here rather than in the schema class that introduced it (#766) * because two producers build description nodes: the automatic * Global_SEO_Schema_Output and the Schema Manager's Schema_Builder. Only * the first normalised, so a deployed node, which outranks the automatic * one, published the raw entity again. * * @since 2.10.0 * * @param string $text Raw text. * @return string */ public static function normalize_schema_text(string $text): string { if ('' === trim($text)) { return ''; } $text = self::decode_schema_entities(wp_strip_all_tags($text)); // Core's excerpt marker, in both its entity and literal forms, with or // without the surrounding brackets it is normally wrapped in. $text = (string) preg_replace( '/\s*(\[\s*(\x{2026}|\.\.\.)\s*\]|\x{2026})\s*$/u', '', $text ); return trim((string) preg_replace('/\s+/u', ' ', $text)); } /** * Decode HTML entities in text bound for JSON-LD, and nothing else. * * The narrow half of normalize_schema_text(), for values that must keep * their exact shape otherwise: a stored snapshot already truncated with an * ellipsis would lose it to the excerpt-marker strip. * * @since 2.10.0 * * @param string $text Text that may carry HTML entities. * @return string */ public static function decode_schema_entities(string $text): string { // Twice: a description that has been through an escaping pass already // (core stores `&amp;` for a literal `&` in some paths) would // otherwise still carry an entity after one decode. Decoding an // already-plain string is a no-op, so this is safe to repeat. $text = html_entity_decode($text, ENT_QUOTES | ENT_HTML5, 'UTF-8'); return html_entity_decode($text, ENT_QUOTES | ENT_HTML5, 'UTF-8'); } /** * Apply a WORD cap that stays a word cap. * * wp_trim_words() reads its unit from the locale, so `$words` silently * becomes a *character* cap on th/ja/zh_* — roughly six times tighter than * intended. Callers that mean "about N words" go through here: word-counting * locales keep wp_trim_words() byte for byte, and the rest get a character * budget instead of a mangled one (#687). * * @since 2.7.0 * * @param string $text Text to trim. * @param int $words Word cap, honoured only where words are the unit. * @param string $more Appended by wp_trim_words() when it trims. * @param int $fallback Character budget used where words are not the unit. * @return string */ public static function trim_words( string $text, int $words, string $more = '...', int $fallback = self::MAX_LENGTH ): string { if (!self::locale_counts_words()) { return self::trim_to_length($text, $fallback); } return wp_trim_words($text, $words, $more); } }