PluginProbe
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO / 2.11.0
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO v2.11.0
2.11.0 2.10.0 2.9.0 2.8.0 2.7.0 2.6.0 2.5.0 2.4.0 2.3.0 2.2.0 2.1.1 2.1.0 2.0.2 2.0.1 2.0.0 1.32.0 1.31.0 1.30.0 1.29.0 1.28.0 1.27.0 1.26.0 1.25.0 trunk 1.0.0 All 52 releases
← All changes | includes/seo/class-pattern-resolver.php +426 -15 1.25.0 → 2.11.0 View file →
@@ -42,8 +42,18 @@
42 42 */
43 43 private const DEFAULT_DESCRIPTION = '%excerpt%';
44 44
45 45 /**
46 + * Post meta key holding the per-post SEO title.
47 + */
48 + private const META_TITLE = '_thinkrank_seo_title';
49 +
50 + /**
51 + * Post meta key holding the per-post meta description.
52 + */
53 + private const META_DESCRIPTION = '_thinkrank_meta_description';
54 +
55 + /**
46 56 * Resolve the SEO title pattern for a post.
47 57 *
48 58 * @param int $post_id Post ID.
49 59 * @return string Resolved title, or '' when it resolves to nothing.
@@ -72,8 +82,159 @@
72 82 return self::process($value, self::placeholders_for($post_id));
73 83 }
74 84
75 85 /**
86 + * Sanitize a variable-tag template for storage.
87 + *
88 + * The write-side counterpart of resolve_value(): every template that
89 + * reaches this class has to survive the trip into the database first.
90 + *
91 + * sanitize_text_field() cannot be used for that. Core's
92 + * _sanitize_text_fields() strips percent-encoded characters, looping
93 + * `preg_replace( '/%[a-f0-9]{2}/i', ... )` until nothing matches, so any
94 + * token whose first two characters are hex digits is eaten on save:
95 + * %date% is stored as "te%" and %category% as "tegory%" (#521). They are
96 + * the only two tags in the language that collide, which is why the
97 + * corruption looked arbitrary — %title%, %sitename%, %sep%, %excerpt%,
98 + * %modified% and %author% all pass through core untouched.
99 + *
100 + * This mirrors what core does either side of that percent loop — invalid
101 + * UTF-8 dropped, tags stripped, control characters removed, whitespace
102 + * collapsed — and simply omits the loop itself.
103 + *
104 + * @since 2.1.1
105 + *
106 + * @param string $value Raw template as submitted.
107 + * @param bool $keep_newlines Preserve newlines, as sanitize_textarea_field() does.
108 + * @return string Sanitized template with its %tokens% intact.
109 + */
110 + public static function sanitize_template(string $value, bool $keep_newlines = false): string {
111 + $filtered = wp_check_invalid_utf8($value);
112 +
113 + if (strpos($filtered, '<') !== false) {
114 + $filtered = wp_pre_kses_less_than($filtered);
115 + // Tags out, the text between them kept.
116 + $filtered = wp_strip_all_tags($filtered, false);
117 + $filtered = str_replace("<\n", "&lt;\n", $filtered);
118 + }
119 +
120 + // C0 controls and DEL, less the tab/newline/carriage-return handled below.
121 + $filtered = (string) preg_replace('/[\x00-\x08\x0B\x0C\x0E-\x1F\x7F]/', '', $filtered);
122 +
123 + if (!$keep_newlines) {
124 + $filtered = (string) preg_replace('/[\r\n\t ]+/', ' ', $filtered);
125 + }
126 +
127 + return trim($filtered);
128 + }
129 +
130 + /**
131 + * Sanitize a variable-tag template that may span multiple lines.
132 + *
133 + * The sanitize_textarea_field() counterpart of sanitize_template().
134 + *
135 + * @since 2.1.1
136 + *
137 + * @param string $value Raw template as submitted.
138 + * @return string Sanitized template with its %tokens% and newlines intact.
139 + */
140 + public static function sanitize_template_textarea(string $value): string {
141 + return self::sanitize_template($value, true);
142 + }
143 +
144 + /**
145 + * Derive a description from raw post content.
146 + *
147 + * wp_strip_all_tags() removes HTML but not shortcodes, so a page built with
148 + * them published its shortcode source as the description — `[woocommerce_cart]`
149 + * as the meta description, og:description and twitter:description of the
150 + * cart page. Core's own wp_trim_excerpt() runs strip_shortcodes() and
151 + * excerpt_remove_blocks() first; this path did neither, which is why the
152 + * two disagreed about the same post (#387).
153 + *
154 + * @since 2.0.1
155 + *
156 + * @param string $content Raw post content.
157 + * @param int $words Word cap.
158 + * @return string Derived description, or '' when nothing survives.
159 + */
160 + public static function derive_excerpt(string $content, int $words = 25): string {
161 + if ('' === trim($content)) {
162 + return '';
163 + }
164 +
165 + $text = excerpt_remove_blocks($content);
166 + $text = strip_shortcodes($text);
167 + $text = wp_strip_all_tags($text);
168 +
169 + // $words is a WORD cap, but wp_trim_words() counts CHARACTERS on
170 + // th/ja/zh_*, where it would cut to ~25 characters instead of ~25
171 + // words — about six times too short (#687). trim_words() keeps the
172 + // word cap where words are the unit and falls back to a character
173 + // budget where they are not.
174 + return trim(\ThinkRank\Core\Seo_Text::trim_words($text, $words));
175 + }
176 +
177 + /**
178 + * Resolve any variable-tag string against a term's values.
179 + *
180 + * The term counterpart of resolve_value(). Term SEO fields reach the
181 + * frontend from three writers — the term UI, the abilities API and the
182 + * Yoast/RankMath/AIOSEO/SEOPress importer — and the importers already
183 + * substitute their own term tokens (%%term_title%%, %term%) with the term
184 + * name at export time, so what lands here is either literal text or
185 + * ThinkRank's own tags.
186 + *
187 + * @since 2.0.1
188 + *
189 + * @param string $value Raw string, possibly containing variable tags.
190 + * @param int $term_id Term ID.
191 + * @return string Resolved string.
192 + */
193 + public static function resolve_term_value(string $value, int $term_id): string {
194 + if (strpos($value, '%') === false) {
195 + return $value;
196 + }
197 + return self::process($value, self::placeholders_for_term($term_id));
198 + }
199 +
200 + /**
201 + * Token => value map for a term.
202 + *
203 + * The post-only tokens resolve to an empty string rather than being left
204 + * unreplaced: they have no meaning on an archive, and process() collapses
205 + * the separators an empty token leaves behind. A raw "%author%" in the
206 + * rendered title would be worse than nothing.
207 + *
208 + * @since 2.0.1
209 + *
210 + * @param int $term_id Term ID.
211 + * @return array<string,string> Placeholder map.
212 + */
213 + private static function placeholders_for_term(int $term_id): array {
214 + $term = get_term($term_id);
215 +
216 + $name = ($term && !is_wp_error($term)) ? $term->name : '';
217 + $description = ($term && !is_wp_error($term)) ? (string) $term->description : '';
218 +
219 + return [
220 + '%title%' => $name,
221 + '%term%' => $name,
222 + '%sitename%' => get_bloginfo('name'),
223 + '%sep%' => self::separator(),
224 + // Same locale trap as derive_excerpt(): a word cap here is a
225 + // ~25-character cap on th/ja/zh_* (#687).
226 + '%excerpt%' => $description !== ''
227 + ? self::derive_excerpt($description)
228 + : '',
229 + '%date%' => '',
230 + '%modified%' => '',
231 + '%author%' => '',
232 + '%category%' => '',
233 + ];
234 + }
235 +
236 + /**
76 237 * Token => value map for a post, keyed WITHOUT the surrounding percents
77 238 * (e.g. 'title' => 'My Post'). Used by the editor for live client-side
78 239 * preview of a pattern as the user types.
79 240 *
@@ -99,13 +260,74 @@
99 260 public static function description(int $post_id): string {
100 261 $template = self::template_for($post_id, 'description', self::DEFAULT_DESCRIPTION);
101 262 $description = self::resolve_value($template, $post_id);
102 263
103 - if (strlen($description) > 160) {
104 - $description = wp_trim_words($description, 25, '...');
264 + // Measure and cut in CHARACTERS. strlen() counts bytes, so a Thai or
265 + // CJK description tripped this limit at a third of its length, and
266 + // wp_trim_words() then cut by a unit the locale chooses — 25 words in
267 + // English, 25 characters in Thai (#687).
268 + $description = \ThinkRank\Core\Seo_Text::trim_to_length($description);
269 +
270 + return $description;
271 + }
272 +
273 + /**
274 + * Effective SEO title for a post: the per-post custom value (with any
275 + * variable tags resolved) when set, otherwise the rendered Global/Bulk
276 + * title pattern. This is the value the frontend actually outputs.
277 + *
278 + * Scoring MUST use this rather than the raw `_thinkrank_seo_title` meta —
279 + * an empty meta means "inherit the global pattern", not "no title", so the
280 + * raw value would make an inherited-title post score as if it had none.
281 + *
282 + * @param int $post_id Post ID.
283 + * @return string Effective title.
284 + */
285 + public static function effective_title(int $post_id): string {
286 + return self::effective_value(
287 + (string) get_post_meta($post_id, self::META_TITLE, true),
288 + $post_id,
289 + 'title'
290 + );
291 + }
292 +
293 + /**
294 + * Effective meta description for a post: the per-post custom value (with any
295 + * variable tags resolved) when set, otherwise the rendered Global/Bulk
296 + * description pattern. Counterpart to {@see self::effective_title()}.
297 + *
298 + * @param int $post_id Post ID.
299 + * @return string Effective description.
300 + */
301 + public static function effective_description(int $post_id): string {
302 + return self::effective_value(
303 + (string) get_post_meta($post_id, self::META_DESCRIPTION, true),
304 + $post_id,
305 + 'description'
306 + );
307 + }
308 +
309 + /**
310 + * Resolve a raw per-post field to its effective value.
311 + *
312 + * When the raw value is non-empty its variable tags are resolved; when it is
313 + * empty the field falls back to the rendered Global/Bulk pattern. Exposed so
314 + * callers that already hold a raw value (e.g. the SEO score endpoint scoring
315 + * unsaved editor input) can route through the same fallback logic.
316 + *
317 + * @param string $raw Raw per-post field value (may hold variable tags).
318 + * @param int $post_id Post ID.
319 + * @param string $field Which pattern to fall back to: 'title' or 'description'.
320 + * @return string Effective value.
321 + */
322 + public static function effective_value(string $raw, int $post_id, string $field): string {
323 + if ($raw !== '') {
324 + return self::resolve_value($raw, $post_id);
105 325 }
106 326
107 - return $description;
327 + return 'description' === $field
328 + ? self::description($post_id)
329 + : self::title($post_id);
108 330 }
109 331
110 332 /**
111 333 * Build the full set of pattern previews for the post editor.
@@ -135,17 +357,17 @@
135 357 * Get the configured pattern for a post type, falling back to a default.
136 358 *
137 359 * @param int $post_id Post ID.
138 360 * @param string $key Setting key ('title' or 'description').
139 - * @param string $default Default pattern.
361 + * @param string $fallback Default pattern.
140 362 * @return string Pattern template.
141 363 */
142 - private static function template_for(int $post_id, string $key, string $default): string {
364 + private static function template_for(int $post_id, string $key, string $fallback): string {
143 365 $post_type = get_post_type($post_id) ?: 'post';
144 366 $all = get_option(self::OPTION_NAME, []);
145 367 $template = $all[$post_type][$key] ?? '';
146 368
147 - return is_string($template) && $template !== '' ? $template : $default;
369 + return is_string($template) && $template !== '' ? $template : $fallback;
148 370 }
149 371
150 372 /**
151 373 * Build placeholder values for an explicit post (no loop dependency).
@@ -159,9 +381,9 @@
159 381 $excerpt = '';
160 382 if ($post) {
161 383 $excerpt = !empty($post->post_excerpt)
162 384 ? $post->post_excerpt
163 - : wp_trim_words(wp_strip_all_tags($post->post_content), 25, '...');
385 + : self::derive_excerpt(Builder_Content::visible_content($post));
164 386 }
165 387
166 388 $author_id = (int) get_post_field('post_author', $post_id);
167 389
@@ -170,21 +392,148 @@
170 392 $categories = get_the_category($post_id);
171 393 $category = !empty($categories) ? $categories[0]->name : '';
172 394 }
173 395
396 + return array_merge(
397 + [
398 + '%title%' => get_the_title($post_id),
399 + '%sitename%' => get_bloginfo('name'),
400 + '%sep%' => self::separator(),
401 + '%excerpt%' => $excerpt,
402 + '%date%' => get_the_date('', $post_id),
403 + '%modified%' => get_the_modified_date('', $post_id),
404 + '%author%' => $author_id ? get_the_author_meta('display_name', $author_id) : '',
405 + '%category%' => $category,
406 + ],
407 + self::product_placeholders($post_id)
408 + );
409 + }
410 +
411 + /**
412 + * Tokens that only mean anything on a WooCommerce product.
413 + *
414 + * Rank Math and Yoast WooCommerce SEO both let a product title or
415 + * description carry the price, the SKU and the stock status, and both of
416 + * our converters dropped every token they did not recognise — so
417 + * "Buy %title% for %wc_price%" imported as "Buy %title% for" and the
418 + * customer in support #171748 found four variables where they had had a
419 + * stock one (#715).
420 + *
421 + * Always present, never conditional on the post type. A token that exists
422 + * on a product and is an unknown token everywhere else would resolve on
423 + * one page and leak literally on another; resolving to an empty string off
424 + * a product is the same answer every other token gives when it has nothing
425 + * to say, and process() then collapses the separator it leaves behind.
426 + *
427 + * @since 2.10.1
428 + *
429 + * @param int $post_id Post ID.
430 + * @return array<string,string>
431 + */
432 + private static function product_placeholders(int $post_id): array {
433 + $empty = [
434 + '%price%' => '',
435 + '%sale_price%' => '',
436 + '%sku%' => '',
437 + '%stock_status%' => '',
438 + '%short_description%' => '',
439 + '%brand%' => '',
440 + ];
441 +
442 + if (!function_exists('wc_get_product') || 'product' !== get_post_type($post_id)) {
443 + return $empty;
444 + }
445 +
446 + $product = wc_get_product($post_id);
447 + if (!$product) {
448 + return $empty;
449 + }
450 +
451 + // Read through wc_price()/get_price_html() rather than formatting the
452 + // number here: currency symbol, position, decimals and the "from X"
453 + // form on a variable product are all store settings, and a second
454 + // formatter is how a template starts disagreeing with the price shown
455 + // three lines below it on the same page.
456 + $price = (string) $product->get_price();
457 + $sale_price = (string) $product->get_sale_price();
458 +
459 + $stock_status = (string) $product->get_stock_status();
460 + $stock_labels = [
461 + 'instock' => __('In stock', 'thinkrank'),
462 + 'outofstock' => __('Out of stock', 'thinkrank'),
463 + 'onbackorder' => __('On backorder', 'thinkrank'),
464 + ];
465 +
174 466 return [
175 - '%title%' => get_the_title($post_id),
176 - '%sitename%' => get_bloginfo('name'),
177 - '%sep%' => self::separator(),
178 - '%excerpt%' => $excerpt,
179 - '%date%' => get_the_date('', $post_id),
180 - '%modified%' => get_the_modified_date('', $post_id),
181 - '%author%' => $author_id ? get_the_author_meta('display_name', $author_id) : '',
182 - '%category%' => $category,
467 + '%price%' => self::formatted_price($price),
468 + '%sale_price%' => self::formatted_price($sale_price),
469 + '%sku%' => (string) $product->get_sku(),
470 + '%stock_status%' => $stock_labels[$stock_status] ?? $stock_status,
471 + '%short_description%' => self::derive_excerpt((string) $product->get_short_description()),
472 + '%brand%' => self::product_brand($post_id),
183 473 ];
184 474 }
185 475
186 476 /**
477 + * A price as text, in the store's own currency format.
478 + *
479 + * wc_price() returns markup, and stripping the tags off it leaves the
480 + * entities behind: a title read "11.05&#2547;&nbsp;" on the first live
481 + * run. Entities are decoded and the non-breaking space collapsed, because
482 + * this value ends up inside a `<title>` and a meta description, where
483 + * markup has no meaning and an entity is just noise a reader sees.
484 + *
485 + * @since 2.10.1
486 + *
487 + * @param string $price Raw price, or '' when the product has none.
488 + * @return string
489 + */
490 + private static function formatted_price(string $price): string {
491 + if ('' === $price) {
492 + return '';
493 + }
494 +
495 + $text = wp_strip_all_tags((string) wc_price((float) $price));
496 + $text = html_entity_decode($text, ENT_QUOTES | ENT_HTML5, 'UTF-8');
497 +
498 + // \xC2\xA0 is the non-breaking space wc_price() puts between the
499 + // amount and the symbol; a literal one in a title is invisible to a
500 + // reader and awkward for everything else.
501 + $text = str_replace("\xC2\xA0", ' ', $text);
502 +
503 + return trim((string) preg_replace('/\s+/u', ' ', $text));
504 + }
505 +
506 + /**
507 + * A product's brand, from whichever taxonomy the store uses for one.
508 + *
509 + * WooCommerce core added `product_brand` in 9.4; before that every brand
510 + * plugin shipped its own taxonomy, and a store that migrated from one of
511 + * them still has the old terms. Asking each in turn costs one cached term
512 + * lookup and means the token is not empty on the stores most likely to
513 + * have used a brand token in the plugin they are leaving.
514 + *
515 + * @since 2.10.1
516 + *
517 + * @param int $post_id Product ID.
518 + * @return string
519 + */
520 + private static function product_brand(int $post_id): string {
521 + foreach (['product_brand', 'pwb-brand', 'yith_product_brand', 'berocket_brand'] as $taxonomy) {
522 + if (!taxonomy_exists($taxonomy)) {
523 + continue;
524 + }
525 +
526 + $terms = get_the_terms($post_id, $taxonomy);
527 + if (is_array($terms) && !empty($terms)) {
528 + return (string) $terms[0]->name;
529 + }
530 + }
531 +
532 + return '';
533 + }
534 +
535 + /**
187 536 * Active title separator symbol.
188 537 *
189 538 * @return string Separator.
190 539 */
@@ -202,8 +551,17 @@
202 551 * @param array<string,string> $placeholders Placeholder map.
203 552 * @return string Resolved string.
204 553 */
205 554 private static function process(string $template, array $placeholders): string {
555 + // Both forms go BEFORE substitution, and the strip is decided against
556 + // the placeholder map rather than by what is left over afterwards.
557 + // Running it on the substituted string scanned the resolved VALUES too,
558 + // so a post whose own title read "Using %name% placeholders in
559 + // %%mustache%% templates" published "Using placeholders in %%
560 + // templates" — its title edited, and a stray double percent where the
561 + // inner token had been eaten out of the middle of one.
562 + $template = self::strip_unresolved_tokens($template, $placeholders);
563 +
206 564 $value = str_replace(array_keys($placeholders), array_values($placeholders), $template);
207 565
208 566 // Collapse whitespace.
209 567 $value = preg_replace('/\s+/', ' ', $value);
@@ -219,6 +577,59 @@
219 577 );
220 578
221 579 // Strip leading/trailing separators and whitespace.
222 580 return trim($value, " \t\n\r\0\x0B" . $separator);
581 + }
582 +
583 + /**
584 + * Remove any token the placeholder map did not resolve.
585 + *
586 + * The last line of defence, and the reason it exists is that every layer
587 + * above it is a list someone has to remember to extend. A converter that
588 + * misses a token, a template typed by hand, a value written straight into
589 + * postmeta by an importer we have not met: each one ends with `%%title%%`
590 + * or `%some_token%` rendering literally in a `<title>` on a live site, and
591 + * that is precisely what was reported on 14 September (#715).
592 + *
593 + * Both syntaxes, because a migrated site carries both: Yoast's `%%x%%`
594 + * (which includes Rank Math tokens Yoast's own importer wrapped in double
595 + * percent signs without translating them) and the single-percent form
596 + * ThinkRank and Rank Math share. `%%x%%` is matched as a whole so the pass
597 + * cannot eat the inner `%x%` and leave a stray percent sign at each end.
598 + *
599 + * Applied to the TEMPLATE, and a token is kept only when the placeholder
600 + * map has it. Stripping whatever still looked like a token after
601 + * substitution read the resolved values as well, and content is not a
602 + * template: a post titled "Using %name% placeholders in %%mustache%%
603 + * templates" published "Using placeholders in %% templates", and an excerpt
604 + * of "Learn %name% and %fabric% placeholders." published "Learn and
605 + * placeholders." Deciding against the map also means a token this resolver
606 + * knows about is never at risk, whatever a value happens to contain.
607 + *
608 + * A bare percent is left alone: "50% off" is ordinary copy, and a guard
609 + * that ate it would be a worse bug than the one it prevents.
610 + *
611 + * @since 2.10.1
612 + *
613 + * @param string $template Raw template.
614 + * @param array<string, string> $placeholders Tokens this resolver can resolve.
615 + * @return string
616 + */
617 + private static function strip_unresolved_tokens(string $template, array $placeholders): string {
618 + if (false === strpos($template, '%')) {
619 + return $template;
620 + }
621 +
622 + $stripped = preg_replace_callback(
623 + '/%%[a-z0-9_-]+%%|%[a-z0-9_-]+%/i',
624 + static function (array $found) use ($placeholders): string {
625 + return array_key_exists($found[0], $placeholders) ? $found[0] : '';
626 + },
627 + $template
628 + );
629 +
630 + // preg_replace_callback() answers null on content that is not valid
631 + // UTF-8 rather than throwing, and returning null here would blank a
632 + // title outright.
633 + return null === $stripped ? $template : (string) $stripped;
223 634 }
224 635 }