| @@ -82,8 +82,67 @@ | ||
| 82 | 82 | return self::process($value, self::placeholders_for($post_id)); |
| 83 | 83 | } |
| 84 | 84 | |
| 85 | 85 | /** |
| 86 | + * Sanitize a variable-tag template for storage. | |
| 87 | + * | |
| 88 | + * The write-side counterpart of resolve_value(): every template that | |
| 89 | + * reaches this class has to survive the trip into the database first. | |
| 90 | + * | |
| 91 | + * sanitize_text_field() cannot be used for that. Core's | |
| 92 | + * _sanitize_text_fields() strips percent-encoded characters, looping | |
| 93 | + * `preg_replace( '/%[a-f0-9]{2}/i', ... )` until nothing matches, so any | |
| 94 | + * token whose first two characters are hex digits is eaten on save: | |
| 95 | + * %date% is stored as "te%" and %category% as "tegory%" (#521). They are | |
| 96 | + * the only two tags in the language that collide, which is why the | |
| 97 | + * corruption looked arbitrary — %title%, %sitename%, %sep%, %excerpt%, | |
| 98 | + * %modified% and %author% all pass through core untouched. | |
| 99 | + * | |
| 100 | + * This mirrors what core does either side of that percent loop — invalid | |
| 101 | + * UTF-8 dropped, tags stripped, control characters removed, whitespace | |
| 102 | + * collapsed — and simply omits the loop itself. | |
| 103 | + * | |
| 104 | + * @since 2.1.1 | |
| 105 | + * | |
| 106 | + * @param string $value Raw template as submitted. | |
| 107 | + * @param bool $keep_newlines Preserve newlines, as sanitize_textarea_field() does. | |
| 108 | + * @return string Sanitized template with its %tokens% intact. | |
| 109 | + */ | |
| 110 | + public static function sanitize_template(string $value, bool $keep_newlines = false): string { | |
| 111 | + $filtered = wp_check_invalid_utf8($value); | |
| 112 | + | |
| 113 | + if (strpos($filtered, '<') !== false) { | |
| 114 | + $filtered = wp_pre_kses_less_than($filtered); | |
| 115 | + // Tags out, the text between them kept. | |
| 116 | + $filtered = wp_strip_all_tags($filtered, false); | |
| 117 | + $filtered = str_replace("<\n", "<\n", $filtered); | |
| 118 | + } | |
| 119 | + | |
| 120 | + // C0 controls and DEL, less the tab/newline/carriage-return handled below. | |
| 121 | + $filtered = (string) preg_replace('/[\x00-\x08\x0B\x0C\x0E-\x1F\x7F]/', '', $filtered); | |
| 122 | + | |
| 123 | + if (!$keep_newlines) { | |
| 124 | + $filtered = (string) preg_replace('/[\r\n\t ]+/', ' ', $filtered); | |
| 125 | + } | |
| 126 | + | |
| 127 | + return trim($filtered); | |
| 128 | + } | |
| 129 | + | |
| 130 | + /** | |
| 131 | + * Sanitize a variable-tag template that may span multiple lines. | |
| 132 | + * | |
| 133 | + * The sanitize_textarea_field() counterpart of sanitize_template(). | |
| 134 | + * | |
| 135 | + * @since 2.1.1 | |
| 136 | + * | |
| 137 | + * @param string $value Raw template as submitted. | |
| 138 | + * @return string Sanitized template with its %tokens% and newlines intact. | |
| 139 | + */ | |
| 140 | + public static function sanitize_template_textarea(string $value): string { | |
| 141 | + return self::sanitize_template($value, true); | |
| 142 | + } | |
| 143 | + | |
| 144 | + /** | |
| 86 | 145 | * Derive a description from raw post content. |
| 87 | 146 | * |
| 88 | 147 | * wp_strip_all_tags() removes HTML but not shortcodes, so a page built with |
| 89 | 148 | * them published its shortcode source as the description — `[woocommerce_cart]` |
| @@ -106,9 +165,14 @@ | ||
| 106 | 165 | $text = excerpt_remove_blocks($content); |
| 107 | 166 | $text = strip_shortcodes($text); |
| 108 | 167 | $text = wp_strip_all_tags($text); |
| 109 | 168 | |
| 110 | - return trim(wp_trim_words($text, $words, '...')); | |
| 169 | + // $words is a WORD cap, but wp_trim_words() counts CHARACTERS on | |
| 170 | + // th/ja/zh_*, where it would cut to ~25 characters instead of ~25 | |
| 171 | + // words — about six times too short (#687). trim_words() keeps the | |
| 172 | + // word cap where words are the unit and falls back to a character | |
| 173 | + // budget where they are not. | |
| 174 | + return trim(\ThinkRank\Core\Seo_Text::trim_words($text, $words)); | |
| 111 | 175 | } |
| 112 | 176 | |
| 113 | 177 | /** |
| 114 | 178 | * Resolve any variable-tag string against a term's values. |
| @@ -156,10 +220,12 @@ | ||
| 156 | 220 | '%title%' => $name, |
| 157 | 221 | '%term%' => $name, |
| 158 | 222 | '%sitename%' => get_bloginfo('name'), |
| 159 | 223 | '%sep%' => self::separator(), |
| 224 | + // Same locale trap as derive_excerpt(): a word cap here is a | |
| 225 | + // ~25-character cap on th/ja/zh_* (#687). | |
| 160 | 226 | '%excerpt%' => $description !== '' |
| 161 | - ? wp_trim_words(wp_strip_all_tags($description), 25, '...') | |
| 227 | + ? self::derive_excerpt($description) | |
| 162 | 228 | : '', |
| 163 | 229 | '%date%' => '', |
| 164 | 230 | '%modified%' => '', |
| 165 | 231 | '%author%' => '', |
| @@ -194,11 +260,13 @@ | ||
| 194 | 260 | public static function description(int $post_id): string { |
| 195 | 261 | $template = self::template_for($post_id, 'description', self::DEFAULT_DESCRIPTION); |
| 196 | 262 | $description = self::resolve_value($template, $post_id); |
| 197 | 263 | |
| 198 | - if (strlen($description) > 160) { | |
| 199 | - $description = wp_trim_words($description, 25, '...'); | |
| 200 | - } | |
| 264 | + // Measure and cut in CHARACTERS. strlen() counts bytes, so a Thai or | |
| 265 | + // CJK description tripped this limit at a third of its length, and | |
| 266 | + // wp_trim_words() then cut by a unit the locale chooses — 25 words in | |
| 267 | + // English, 25 characters in Thai (#687). | |
| 268 | + $description = \ThinkRank\Core\Seo_Text::trim_to_length($description); | |
| 201 | 269 | |
| 202 | 270 | return $description; |
| 203 | 271 | } |
| 204 | 272 | |
| @@ -313,9 +381,9 @@ | ||
| 313 | 381 | $excerpt = ''; |
| 314 | 382 | if ($post) { |
| 315 | 383 | $excerpt = !empty($post->post_excerpt) |
| 316 | 384 | ? $post->post_excerpt |
| 317 | - : self::derive_excerpt((string) $post->post_content); | |
| 385 | + : self::derive_excerpt(Builder_Content::visible_content($post)); | |
| 318 | 386 | } |
| 319 | 387 | |
| 320 | 388 | $author_id = (int) get_post_field('post_author', $post_id); |
| 321 | 389 | |
| @@ -324,21 +392,148 @@ | ||
| 324 | 392 | $categories = get_the_category($post_id); |
| 325 | 393 | $category = !empty($categories) ? $categories[0]->name : ''; |
| 326 | 394 | } |
| 327 | 395 | |
| 396 | + return array_merge( | |
| 397 | + [ | |
| 398 | + '%title%' => get_the_title($post_id), | |
| 399 | + '%sitename%' => get_bloginfo('name'), | |
| 400 | + '%sep%' => self::separator(), | |
| 401 | + '%excerpt%' => $excerpt, | |
| 402 | + '%date%' => get_the_date('', $post_id), | |
| 403 | + '%modified%' => get_the_modified_date('', $post_id), | |
| 404 | + '%author%' => $author_id ? get_the_author_meta('display_name', $author_id) : '', | |
| 405 | + '%category%' => $category, | |
| 406 | + ], | |
| 407 | + self::product_placeholders($post_id) | |
| 408 | + ); | |
| 409 | + } | |
| 410 | + | |
| 411 | + /** | |
| 412 | + * Tokens that only mean anything on a WooCommerce product. | |
| 413 | + * | |
| 414 | + * Rank Math and Yoast WooCommerce SEO both let a product title or | |
| 415 | + * description carry the price, the SKU and the stock status, and both of | |
| 416 | + * our converters dropped every token they did not recognise — so | |
| 417 | + * "Buy %title% for %wc_price%" imported as "Buy %title% for" and the | |
| 418 | + * customer in support #171748 found four variables where they had had a | |
| 419 | + * stock one (#715). | |
| 420 | + * | |
| 421 | + * Always present, never conditional on the post type. A token that exists | |
| 422 | + * on a product and is an unknown token everywhere else would resolve on | |
| 423 | + * one page and leak literally on another; resolving to an empty string off | |
| 424 | + * a product is the same answer every other token gives when it has nothing | |
| 425 | + * to say, and process() then collapses the separator it leaves behind. | |
| 426 | + * | |
| 427 | + * @since 2.10.1 | |
| 428 | + * | |
| 429 | + * @param int $post_id Post ID. | |
| 430 | + * @return array<string,string> | |
| 431 | + */ | |
| 432 | + private static function product_placeholders(int $post_id): array { | |
| 433 | + $empty = [ | |
| 434 | + '%price%' => '', | |
| 435 | + '%sale_price%' => '', | |
| 436 | + '%sku%' => '', | |
| 437 | + '%stock_status%' => '', | |
| 438 | + '%short_description%' => '', | |
| 439 | + '%brand%' => '', | |
| 440 | + ]; | |
| 441 | + | |
| 442 | + if (!function_exists('wc_get_product') || 'product' !== get_post_type($post_id)) { | |
| 443 | + return $empty; | |
| 444 | + } | |
| 445 | + | |
| 446 | + $product = wc_get_product($post_id); | |
| 447 | + if (!$product) { | |
| 448 | + return $empty; | |
| 449 | + } | |
| 450 | + | |
| 451 | + // Read through wc_price()/get_price_html() rather than formatting the | |
| 452 | + // number here: currency symbol, position, decimals and the "from X" | |
| 453 | + // form on a variable product are all store settings, and a second | |
| 454 | + // formatter is how a template starts disagreeing with the price shown | |
| 455 | + // three lines below it on the same page. | |
| 456 | + $price = (string) $product->get_price(); | |
| 457 | + $sale_price = (string) $product->get_sale_price(); | |
| 458 | + | |
| 459 | + $stock_status = (string) $product->get_stock_status(); | |
| 460 | + $stock_labels = [ | |
| 461 | + 'instock' => __('In stock', 'thinkrank'), | |
| 462 | + 'outofstock' => __('Out of stock', 'thinkrank'), | |
| 463 | + 'onbackorder' => __('On backorder', 'thinkrank'), | |
| 464 | + ]; | |
| 465 | + | |
| 328 | 466 | return [ |
| 329 | - '%title%' => get_the_title($post_id), | |
| 330 | - '%sitename%' => get_bloginfo('name'), | |
| 331 | - '%sep%' => self::separator(), | |
| 332 | - '%excerpt%' => $excerpt, | |
| 333 | - '%date%' => get_the_date('', $post_id), | |
| 334 | - '%modified%' => get_the_modified_date('', $post_id), | |
| 335 | - '%author%' => $author_id ? get_the_author_meta('display_name', $author_id) : '', | |
| 336 | - '%category%' => $category, | |
| 467 | + '%price%' => self::formatted_price($price), | |
| 468 | + '%sale_price%' => self::formatted_price($sale_price), | |
| 469 | + '%sku%' => (string) $product->get_sku(), | |
| 470 | + '%stock_status%' => $stock_labels[$stock_status] ?? $stock_status, | |
| 471 | + '%short_description%' => self::derive_excerpt((string) $product->get_short_description()), | |
| 472 | + '%brand%' => self::product_brand($post_id), | |
| 337 | 473 | ]; |
| 338 | 474 | } |
| 339 | 475 | |
| 340 | 476 | /** |
| 477 | + * A price as text, in the store's own currency format. | |
| 478 | + * | |
| 479 | + * wc_price() returns markup, and stripping the tags off it leaves the | |
| 480 | + * entities behind: a title read "11.05৳ " on the first live | |
| 481 | + * run. Entities are decoded and the non-breaking space collapsed, because | |
| 482 | + * this value ends up inside a `<title>` and a meta description, where | |
| 483 | + * markup has no meaning and an entity is just noise a reader sees. | |
| 484 | + * | |
| 485 | + * @since 2.10.1 | |
| 486 | + * | |
| 487 | + * @param string $price Raw price, or '' when the product has none. | |
| 488 | + * @return string | |
| 489 | + */ | |
| 490 | + private static function formatted_price(string $price): string { | |
| 491 | + if ('' === $price) { | |
| 492 | + return ''; | |
| 493 | + } | |
| 494 | + | |
| 495 | + $text = wp_strip_all_tags((string) wc_price((float) $price)); | |
| 496 | + $text = html_entity_decode($text, ENT_QUOTES | ENT_HTML5, 'UTF-8'); | |
| 497 | + | |
| 498 | + // \xC2\xA0 is the non-breaking space wc_price() puts between the | |
| 499 | + // amount and the symbol; a literal one in a title is invisible to a | |
| 500 | + // reader and awkward for everything else. | |
| 501 | + $text = str_replace("\xC2\xA0", ' ', $text); | |
| 502 | + | |
| 503 | + return trim((string) preg_replace('/\s+/u', ' ', $text)); | |
| 504 | + } | |
| 505 | + | |
| 506 | + /** | |
| 507 | + * A product's brand, from whichever taxonomy the store uses for one. | |
| 508 | + * | |
| 509 | + * WooCommerce core added `product_brand` in 9.4; before that every brand | |
| 510 | + * plugin shipped its own taxonomy, and a store that migrated from one of | |
| 511 | + * them still has the old terms. Asking each in turn costs one cached term | |
| 512 | + * lookup and means the token is not empty on the stores most likely to | |
| 513 | + * have used a brand token in the plugin they are leaving. | |
| 514 | + * | |
| 515 | + * @since 2.10.1 | |
| 516 | + * | |
| 517 | + * @param int $post_id Product ID. | |
| 518 | + * @return string | |
| 519 | + */ | |
| 520 | + private static function product_brand(int $post_id): string { | |
| 521 | + foreach (['product_brand', 'pwb-brand', 'yith_product_brand', 'berocket_brand'] as $taxonomy) { | |
| 522 | + if (!taxonomy_exists($taxonomy)) { | |
| 523 | + continue; | |
| 524 | + } | |
| 525 | + | |
| 526 | + $terms = get_the_terms($post_id, $taxonomy); | |
| 527 | + if (is_array($terms) && !empty($terms)) { | |
| 528 | + return (string) $terms[0]->name; | |
| 529 | + } | |
| 530 | + } | |
| 531 | + | |
| 532 | + return ''; | |
| 533 | + } | |
| 534 | + | |
| 535 | + /** | |
| 341 | 536 | * Active title separator symbol. |
| 342 | 537 | * |
| 343 | 538 | * @return string Separator. |
| 344 | 539 | */ |
| @@ -356,8 +551,17 @@ | ||
| 356 | 551 | * @param array<string,string> $placeholders Placeholder map. |
| 357 | 552 | * @return string Resolved string. |
| 358 | 553 | */ |
| 359 | 554 | private static function process(string $template, array $placeholders): string { |
| 555 | + // Both forms go BEFORE substitution, and the strip is decided against | |
| 556 | + // the placeholder map rather than by what is left over afterwards. | |
| 557 | + // Running it on the substituted string scanned the resolved VALUES too, | |
| 558 | + // so a post whose own title read "Using %name% placeholders in | |
| 559 | + // %%mustache%% templates" published "Using placeholders in %% | |
| 560 | + // templates" — its title edited, and a stray double percent where the | |
| 561 | + // inner token had been eaten out of the middle of one. | |
| 562 | + $template = self::strip_unresolved_tokens($template, $placeholders); | |
| 563 | + | |
| 360 | 564 | $value = str_replace(array_keys($placeholders), array_values($placeholders), $template); |
| 361 | 565 | |
| 362 | 566 | // Collapse whitespace. |
| 363 | 567 | $value = preg_replace('/\s+/', ' ', $value); |
| @@ -373,6 +577,59 @@ | ||
| 373 | 577 | ); |
| 374 | 578 | |
| 375 | 579 | // Strip leading/trailing separators and whitespace. |
| 376 | 580 | return trim($value, " \t\n\r\0\x0B" . $separator); |
| 581 | + } | |
| 582 | + | |
| 583 | + /** | |
| 584 | + * Remove any token the placeholder map did not resolve. | |
| 585 | + * | |
| 586 | + * The last line of defence, and the reason it exists is that every layer | |
| 587 | + * above it is a list someone has to remember to extend. A converter that | |
| 588 | + * misses a token, a template typed by hand, a value written straight into | |
| 589 | + * postmeta by an importer we have not met: each one ends with `%%title%%` | |
| 590 | + * or `%some_token%` rendering literally in a `<title>` on a live site, and | |
| 591 | + * that is precisely what was reported on 14 September (#715). | |
| 592 | + * | |
| 593 | + * Both syntaxes, because a migrated site carries both: Yoast's `%%x%%` | |
| 594 | + * (which includes Rank Math tokens Yoast's own importer wrapped in double | |
| 595 | + * percent signs without translating them) and the single-percent form | |
| 596 | + * ThinkRank and Rank Math share. `%%x%%` is matched as a whole so the pass | |
| 597 | + * cannot eat the inner `%x%` and leave a stray percent sign at each end. | |
| 598 | + * | |
| 599 | + * Applied to the TEMPLATE, and a token is kept only when the placeholder | |
| 600 | + * map has it. Stripping whatever still looked like a token after | |
| 601 | + * substitution read the resolved values as well, and content is not a | |
| 602 | + * template: a post titled "Using %name% placeholders in %%mustache%% | |
| 603 | + * templates" published "Using placeholders in %% templates", and an excerpt | |
| 604 | + * of "Learn %name% and %fabric% placeholders." published "Learn and | |
| 605 | + * placeholders." Deciding against the map also means a token this resolver | |
| 606 | + * knows about is never at risk, whatever a value happens to contain. | |
| 607 | + * | |
| 608 | + * A bare percent is left alone: "50% off" is ordinary copy, and a guard | |
| 609 | + * that ate it would be a worse bug than the one it prevents. | |
| 610 | + * | |
| 611 | + * @since 2.10.1 | |
| 612 | + * | |
| 613 | + * @param string $template Raw template. | |
| 614 | + * @param array<string, string> $placeholders Tokens this resolver can resolve. | |
| 615 | + * @return string | |
| 616 | + */ | |
| 617 | + private static function strip_unresolved_tokens(string $template, array $placeholders): string { | |
| 618 | + if (false === strpos($template, '%')) { | |
| 619 | + return $template; | |
| 620 | + } | |
| 621 | + | |
| 622 | + $stripped = preg_replace_callback( | |
| 623 | + '/%%[a-z0-9_-]+%%|%[a-z0-9_-]+%/i', | |
| 624 | + static function (array $found) use ($placeholders): string { | |
| 625 | + return array_key_exists($found[0], $placeholders) ? $found[0] : ''; | |
| 626 | + }, | |
| 627 | + $template | |
| 628 | + ); | |
| 629 | + | |
| 630 | + // preg_replace_callback() answers null on content that is not valid | |
| 631 | + // UTF-8 rather than throwing, and returning null here would blank a | |
| 632 | + // title outright. | |
| 633 | + return null === $stripped ? $template : (string) $stripped; | |
| 377 | 634 | } |
| 378 | 635 | } |