PluginProbe
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO / 2.10.0
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO v2.10.0
2.10.0 2.9.0 2.8.0 2.7.0 2.6.0 2.5.0 2.4.0 2.3.0 2.2.0 2.1.1 2.1.0 2.0.2 2.0.1 2.0.0 1.32.0 1.31.0 1.30.0 1.29.0 1.28.0 1.27.0 1.26.0 1.25.0 trunk 1.0.0 1.0.1 All 51 releases
thinkrank / includes / core / class-seo-text.php

class-seo-text.php in ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO 2.10.0, at includes/core/class-seo-text.php

235 lines 8.6 KB
No matching file
Up and down to move Enter to open Esc to close
Raw Download Zip
1 <?php
2 /**
3 * Shared length handling for rendered SEO descriptions.
4 *
5 * @package ThinkRank
6 * @subpackage Core
7 * @since 2.7.0
8 */
9
10 declare(strict_types=1);
11
12 namespace ThinkRank\Core;
13
14 if (!defined('ABSPATH')) {
15 exit;
16 }
17
18 /**
19 * Trims SEO text to the length search engines display.
20 *
21 * Every path that renders a description must measure and cut through here
22 * rather than carrying its own copy, because the obvious spelling of this is
23 * wrong in two independent ways and both are invisible in English:
24 *
25 * - `strlen()` counts BYTES. UTF-8 Thai and CJK run three bytes per character,
26 * so a 54-character description already trips a 160-"character" gate it is
27 * nowhere near.
28 * - `wp_trim_words()` does not always count words. WordPress reads the unit
29 * from a per-locale gettext string, and `th`, `ja` and `zh_*` set it to
30 * `characters_excluding_spaces` — so `wp_trim_words($text, 25)` keeps 25
31 * words in English and 25 *characters* in Thai.
32 *
33 * Together they cut Thai meta descriptions to roughly 25 characters while the
34 * same stored value rendered in full in og:description, which reached the page
35 * by a different path (#687).
36 *
37 * @since 2.7.0
38 */
39 class Seo_Text {
40
41 /**
42 * Characters search engines display for a meta description.
43 *
44 * @since 2.7.0
45 * @var int
46 */
47 public const MAX_LENGTH = 160;
48
49 /**
50 * Characters search engines display for a title.
51 *
52 * @since 2.7.0
53 * @var int
54 */
55 public const TITLE_MAX_LENGTH = 60;
56
57 /**
58 * Whether this locale's word-count unit is actually words.
59 *
60 * WordPress reads the unit from a per-locale gettext string, and `th`,
61 * `ja` and `zh_*` set it to `characters_excluding_spaces`. Anything that
62 * passes a *word* cap to wp_trim_words() therefore has to ask first, or it
63 * silently becomes a character cap roughly six times tighter (#687).
64 *
65 * wp_get_word_count_type() only exists from WP 6.2; the plugin supports
66 * 6.0, so fall back to the same gettext string core reads.
67 *
68 * @since 2.7.0
69 * @return bool
70 */
71 public static function locale_counts_words(): bool {
72 if (function_exists('wp_get_word_count_type')) {
73 return 'words' === wp_get_word_count_type();
74 }
75
76 // Core's own string, in core's text domain — this is the value
77 // WP_Locale::get_word_count_type() returns from 6.2 onward. Reading it
78 // from 'thinkrank' would look up a translation we do not ship and
79 // always answer 'words', quietly disabling the check.
80 // phpcs:ignore WordPress.WP.I18n.TextDomainMismatch -- deliberately core's string.
81 return 'words' === _x('words', 'Word count type. Do not translate!', 'default');
82 }
83
84 /**
85 * The text a reader sees for a resolved title or description.
86 *
87 * Pattern_Resolver hands back what the page will print, and that is still
88 * HTML: get_the_title() runs wptexturize, so "Foo & Bar" arrives as
89 * `Foo &#038; Bar`, while the same words typed into the SEO title field
90 * arrive as `Foo &amp; Bar` or a bare `&`. All three render as one `<title>`.
91 * Anything that compares, measures or displays the value as text has to
92 * look at it after the browser would have decoded it, or it groups two
93 * identical titles apart, counts `&#038;` as six characters, and shows the
94 * entity to the user.
95 *
96 * @since 2.10.0
97 *
98 * @param string $value Resolved title or description.
99 * @return string The same value with HTML entities decoded.
100 */
101 public static function as_displayed(string $value): string {
102 return html_entity_decode($value, ENT_QUOTES | ENT_HTML5, 'UTF-8');
103 }
104
105 /**
106 * Trim a description to a character budget, multibyte-safe.
107 *
108 * Cuts on a word boundary when one is available inside the budget, so the
109 * result does not end mid-word; falls back to a hard character cut for
110 * scripts that do not use spaces (CJK, Thai), where a word-boundary search
111 * would find nothing and return the string untouched.
112 *
113 * Lifted from Author_Archives_Manager, which has carried the only correct
114 * copy since 2.2.0 while four other paths kept the broken pattern.
115 *
116 * @since 2.7.0
117 *
118 * @param string $description Description text.
119 * @param int $limit Maximum length in characters, ellipsis included.
120 * @return string
121 */
122 public static function trim_to_length(string $description, int $limit = self::MAX_LENGTH): string {
123 if ($limit <= 0) {
124 return '';
125 }
126
127 if (mb_strlen($description) <= $limit) {
128 return $description;
129 }
130
131 // Reserve one character for the ellipsis.
132 $budget = $limit - 1;
133 $cut = mb_substr($description, 0, $budget);
134 $last_gap = mb_strrpos($cut, ' ');
135
136 // Only honour a word boundary that is not absurdly early — otherwise a
137 // long unbroken token would collapse the description to a few chars.
138 if (false !== $last_gap && $last_gap > (int) ($budget * 0.6)) {
139 $cut = mb_substr($cut, 0, $last_gap);
140 }
141
142 return rtrim($cut) . '…';
143 }
144
145 /**
146 * Make text fit to appear in JSON-LD.
147 *
148 * JSON-LD is not HTML, so an HTML entity in it is not decoded by anything
149 * downstream: `&amp;` was published to answer engines literally. And the
150 * excerpt path appends core's trimming marker, so descriptions arrived
151 * ending in `[…]`, a truncation artefact presented as the page's own
152 * summary.
153 *
154 * Lives here rather than in the schema class that introduced it (#766)
155 * because two producers build description nodes: the automatic
156 * Global_SEO_Schema_Output and the Schema Manager's Schema_Builder. Only
157 * the first normalised, so a deployed node, which outranks the automatic
158 * one, published the raw entity again.
159 *
160 * @since 2.10.0
161 *
162 * @param string $text Raw text.
163 * @return string
164 */
165 public static function normalize_schema_text(string $text): string {
166 if ('' === trim($text)) {
167 return '';
168 }
169
170 $text = self::decode_schema_entities(wp_strip_all_tags($text));
171
172 // Core's excerpt marker, in both its entity and literal forms, with or
173 // without the surrounding brackets it is normally wrapped in.
174 $text = (string) preg_replace(
175 '/\s*(\[\s*(\x{2026}|\.\.\.)\s*\]|\x{2026})\s*$/u',
176 '',
177 $text
178 );
179
180 return trim((string) preg_replace('/\s+/u', ' ', $text));
181 }
182
183 /**
184 * Decode HTML entities in text bound for JSON-LD, and nothing else.
185 *
186 * The narrow half of normalize_schema_text(), for values that must keep
187 * their exact shape otherwise: a stored snapshot already truncated with an
188 * ellipsis would lose it to the excerpt-marker strip.
189 *
190 * @since 2.10.0
191 *
192 * @param string $text Text that may carry HTML entities.
193 * @return string
194 */
195 public static function decode_schema_entities(string $text): string {
196 // Twice: a description that has been through an escaping pass already
197 // (core stores `&amp;amp;` for a literal `&amp;` in some paths) would
198 // otherwise still carry an entity after one decode. Decoding an
199 // already-plain string is a no-op, so this is safe to repeat.
200 $text = html_entity_decode($text, ENT_QUOTES | ENT_HTML5, 'UTF-8');
201
202 return html_entity_decode($text, ENT_QUOTES | ENT_HTML5, 'UTF-8');
203 }
204
205 /**
206 * Apply a WORD cap that stays a word cap.
207 *
208 * wp_trim_words() reads its unit from the locale, so `$words` silently
209 * becomes a *character* cap on th/ja/zh_* — roughly six times tighter than
210 * intended. Callers that mean "about N words" go through here: word-counting
211 * locales keep wp_trim_words() byte for byte, and the rest get a character
212 * budget instead of a mangled one (#687).
213 *
214 * @since 2.7.0
215 *
216 * @param string $text Text to trim.
217 * @param int $words Word cap, honoured only where words are the unit.
218 * @param string $more Appended by wp_trim_words() when it trims.
219 * @param int $fallback Character budget used where words are not the unit.
220 * @return string
221 */
222 public static function trim_words(
223 string $text,
224 int $words,
225 string $more = '...',
226 int $fallback = self::MAX_LENGTH
227 ): string {
228 if (!self::locale_counts_words()) {
229 return self::trim_to_length($text, $fallback);
230 }
231
232 return wp_trim_words($text, $words, $more);
233 }
234 }
235