| @@ -40,8 +40,36 @@ | ||
| 40 | 40 | */ |
| 41 | 41 | class Site_Identity_Manager extends Abstract_SEO_Manager { |
| 42 | 42 | |
| 43 | 43 | /** |
| 44 | + * The per-context title formats as ThinkRank ships them. | |
| 45 | + * | |
| 46 | + * These are not in get_default_settings(): the admin screen seeds them on | |
| 47 | + * first save, so on a real install they are stored values, indistinguishable | |
| 48 | + * from a template the user typed. The migration needs to tell those two | |
| 49 | + * apart — it may overwrite a shipped default with an imported template, and | |
| 50 | + * must never overwrite a choice the user made — so this is the record of | |
| 51 | + * what "untouched" looks like. | |
| 52 | + * | |
| 53 | + * Keep in step with getDefaultSettings() in | |
| 54 | + * src/admin/components/essential-seo/SiteIdentityTab.js. SiteIdentityTitleFormatDefaultsTest | |
| 55 | + * fails when the two drift. | |
| 56 | + * | |
| 57 | + * @since 2.8.0 | |
| 58 | + * @var array<string, string> | |
| 59 | + */ | |
| 60 | + public const TITLE_FORMAT_DEFAULTS = [ | |
| 61 | + 'homepage_title' => '%site_title% %sep% %site_description%', | |
| 62 | + 'post_title' => '%post_title% %sep% %site_title%', | |
| 63 | + 'page_title' => '%page_title% %sep% %site_title%', | |
| 64 | + 'category_title' => '%category_title% %sep% %site_title%', | |
| 65 | + 'tag_title' => '%tag_title% %sep% %site_title%', | |
| 66 | + 'author_title' => '%author_name% %sep% %site_title%', | |
| 67 | + 'search_title' => 'Search Results for "%search_term%" %sep% %site_title%', | |
| 68 | + 'archive_title' => '%archive_title% %sep% %site_title%', | |
| 69 | + ]; | |
| 70 | + | |
| 71 | + /** | |
| 44 | 72 | * WordPress filesystem instance |
| 45 | 73 | * |
| 46 | 74 | * @since 1.0.0 |
| 47 | 75 | * @var \WP_Filesystem_Base|null |
| @@ -290,8 +318,24 @@ | ||
| 290 | 318 | * @var bool |
| 291 | 319 | */ |
| 292 | 320 | private static bool $icon_sizes_listener_registered = false; |
| 293 | 321 | |
| 322 | + /** | |
| 323 | + * Whether the robots.txt resync listener is registered for this request. | |
| 324 | + * | |
| 325 | + * @since 2.14.0 | |
| 326 | + * @var bool | |
| 327 | + */ | |
| 328 | + private static bool $robots_sync_listener_registered = false; | |
| 329 | + | |
| 330 | + /** | |
| 331 | + * Flag set when a plugin change may have altered the sitemap set. | |
| 332 | + * | |
| 333 | + * @since 2.14.0 | |
| 334 | + * @var string | |
| 335 | + */ | |
| 336 | + public const ROBOTS_RESYNC_OPTION = 'thinkrank_robots_txt_resync_pending'; | |
| 337 | + | |
| 294 | 338 | public function __construct() { |
| 295 | 339 | parent::__construct('site_identity'); |
| 296 | 340 | |
| 297 | 341 | if (!self::$icon_sizes_listener_registered) { |
| @@ -300,11 +344,155 @@ | ||
| 300 | 344 | // Admin only: resizing is not front-end work, and admin traffic is |
| 301 | 345 | // enough to run a one-time backfill promptly. |
| 302 | 346 | add_action('admin_init', [self::class, 'maybe_backfill_icon_sizes']); |
| 303 | 347 | } |
| 348 | + | |
| 349 | + if (!self::$robots_sync_listener_registered) { | |
| 350 | + self::$robots_sync_listener_registered = true; | |
| 351 | + | |
| 352 | + // A physical robots.txt bypasses PHP entirely, so composing the | |
| 353 | + // Sitemap block at render time fixes the served output only on | |
| 354 | + // sites with no file. Activating or deactivating a sitemap | |
| 355 | + // contributor changes the set, and until #835 nothing rewrote the | |
| 356 | + // file: the deactivated plugin's sitemap stayed advertised, serving | |
| 357 | + // HTML to anything that followed it. | |
| 358 | + add_action('activated_plugin', [self::class, 'flag_robots_txt_resync']); | |
| 359 | + add_action('deactivated_plugin', [self::class, 'flag_robots_txt_resync']); | |
| 360 | + add_action('init', [self::class, 'maybe_resync_robots_txt'], 99); | |
| 361 | + } | |
| 304 | 362 | } |
| 305 | 363 | |
| 306 | 364 | /** |
| 365 | + * Note that the set of sitemap contributors may have changed. | |
| 366 | + * | |
| 367 | + * Deliberately unconditional about which plugin: a contributor is anything | |
| 368 | + * hooking `thinkrank_additional_sitemaps`, which is resolved at runtime and | |
| 369 | + * cannot be inspected for a plugin that is on its way out. | |
| 370 | + * | |
| 371 | + * The rewrite is not done here. `deactivated_plugin` fires inside the | |
| 372 | + * request that deactivated it, while that plugin's filters are still | |
| 373 | + * attached, so rendering now still sees the sitemap that is going away — | |
| 374 | + * measured, not assumed: the first version of this fix wrote the | |
| 375 | + * deactivated plugin's sitemap straight back into the file. The next | |
| 376 | + * request has the real plugin set loaded, so the work waits for it. | |
| 377 | + * | |
| 378 | + * @since 2.14.0 | |
| 379 | + * @return void | |
| 380 | + */ | |
| 381 | + public static function flag_robots_txt_resync(): void { | |
| 382 | + if (!file_exists(ABSPATH . 'robots.txt')) { | |
| 383 | + return; | |
| 384 | + } | |
| 385 | + | |
| 386 | + update_option(self::ROBOTS_RESYNC_OPTION, 1, false); | |
| 387 | + } | |
| 388 | + | |
| 389 | + /** | |
| 390 | + * Rewrite the physical robots.txt once, on the request after a change. | |
| 391 | + * | |
| 392 | + * @since 2.14.0 | |
| 393 | + * @return void | |
| 394 | + */ | |
| 395 | + public static function maybe_resync_robots_txt(): void { | |
| 396 | + if (!get_option(self::ROBOTS_RESYNC_OPTION)) { | |
| 397 | + return; | |
| 398 | + } | |
| 399 | + | |
| 400 | + // Cleared first, so a render that fatals cannot retry on every request | |
| 401 | + // for the rest of the site's life. | |
| 402 | + delete_option(self::ROBOTS_RESYNC_OPTION); | |
| 403 | + | |
| 404 | + if (!file_exists(ABSPATH . 'robots.txt')) { | |
| 405 | + return; | |
| 406 | + } | |
| 407 | + | |
| 408 | + (new self())->sync_robots_txt_file(); | |
| 409 | + } | |
| 410 | + | |
| 411 | + /** | |
| 412 | + * Save settings, then refresh what a new canonical scheme invalidates. | |
| 413 | + * | |
| 414 | + * The static sitemap files are written with the scheme in force when they | |
| 415 | + * were built, and nothing else rebuilds them until a post or term changes. | |
| 416 | + * So a change of scheme left every `<loc>` on the old one while canonical | |
| 417 | + * and og:url had already moved (#736). Every writer (the settings route, | |
| 418 | + * the robots route, the MCP abilities, an import) lands here. | |
| 419 | + * | |
| 420 | + * @since 2.7.0 | |
| 421 | + * | |
| 422 | + * @param string $context_type Context type. | |
| 423 | + * @param int|null $context_id Context ID. | |
| 424 | + * @param array $settings Settings to save. | |
| 425 | + * @return bool | |
| 426 | + */ | |
| 427 | + public function save_settings(string $context_type, ?int $context_id, array $settings): bool { | |
| 428 | + if (!self::touches_canonical_scheme($context_type, $context_id, $settings)) { | |
| 429 | + return parent::save_settings($context_type, $context_id, $settings); | |
| 430 | + } | |
| 431 | + | |
| 432 | + $before = Url_Scheme::preference(); | |
| 433 | + $saved = parent::save_settings($context_type, $context_id, $settings); | |
| 434 | + | |
| 435 | + if ($saved) { | |
| 436 | + $this->on_canonical_scheme_saved($before); | |
| 437 | + } | |
| 438 | + | |
| 439 | + return $saved; | |
| 440 | + } | |
| 441 | + | |
| 442 | + /** | |
| 443 | + * Whether a save can change the site-wide canonical scheme. | |
| 444 | + * | |
| 445 | + * @since 2.7.0 | |
| 446 | + * | |
| 447 | + * @param string $context_type Context type. | |
| 448 | + * @param int|null $context_id Context ID. | |
| 449 | + * @param array $settings Settings being saved. | |
| 450 | + * @return bool | |
| 451 | + */ | |
| 452 | + public static function touches_canonical_scheme(string $context_type, ?int $context_id, array $settings): bool { | |
| 453 | + return 'site' === sanitize_key($context_type) | |
| 454 | + && empty($context_id) | |
| 455 | + && array_key_exists('canonical_scheme', $settings); | |
| 456 | + } | |
| 457 | + | |
| 458 | + /** | |
| 459 | + * Rebuild the static sitemaps when the effective scheme changed. | |
| 460 | + * | |
| 461 | + * Compares the effective preference, filter included, so a site whose | |
| 462 | + * scheme is pinned by `thinkrank_canonical_scheme` does not rebuild on a | |
| 463 | + * stored value that changes nothing it publishes. | |
| 464 | + * | |
| 465 | + * @since 2.7.0 | |
| 466 | + * | |
| 467 | + * @param string $before Effective scheme before the save. | |
| 468 | + * @return void | |
| 469 | + */ | |
| 470 | + protected function on_canonical_scheme_saved(string $before): void { | |
| 471 | + // The preference is cached for the request; the save just changed it. | |
| 472 | + Url_Scheme::reset(); | |
| 473 | + | |
| 474 | + if (Url_Scheme::preference() === $before) { | |
| 475 | + return; | |
| 476 | + } | |
| 477 | + | |
| 478 | + $this->schedule_sitemap_rebuild(); | |
| 479 | + } | |
| 480 | + | |
| 481 | + /** | |
| 482 | + * Queue a settings-driven sitemap rebuild. | |
| 483 | + * | |
| 484 | + * Debounced and run after the response, like any other settings change | |
| 485 | + * that alters what the sitemap publishes. | |
| 486 | + * | |
| 487 | + * @since 2.7.0 | |
| 488 | + * @return void | |
| 489 | + */ | |
| 490 | + protected function schedule_sitemap_rebuild(): void { | |
| 491 | + (new Sitemap_Generator(false))->schedule_regeneration(); | |
| 492 | + } | |
| 493 | + | |
| 494 | + /** | |
| 307 | 495 | * Build the icon derivatives for a newly chosen favicon. |
| 308 | 496 | * |
| 309 | 497 | * Runs on save, which is the only moment the choice changes and the only |
| 310 | 498 | * place image work belongs — resolving a size on the front end must stay a |
| @@ -397,11 +585,12 @@ | ||
| 397 | 585 | * Attachment ID behind a configured icon URL, or 0 when it is not ours. |
| 398 | 586 | * |
| 399 | 587 | * attachment_url_to_postid() matches _wp_attached_file, which holds the |
| 400 | 588 | * ORIGINAL upload path, so the URL of a generated derivative |
| 401 | - * (`logo-512.png`) returns 0 — and that is exactly what the media picker | |
| 402 | - * hands back when the user chooses a size. Strip the dimension suffix and | |
| 403 | - * try the original once. | |
| 589 | + * (`logo-512x512.png`) returns 0 — and that is exactly what the media | |
| 590 | + * picker hands back when the user chooses a size. Attachment_Lookup falls | |
| 591 | + * back to the original behind it; the fallback started here and moved | |
| 592 | + * there when every other image lookup turned out to need it (#847). | |
| 404 | 593 | * |
| 405 | 594 | * Shared with SEO_Manager's site-icon filter so both sides of the feature |
| 406 | 595 | * agree on which attachment a configured URL means. |
| 407 | 596 | * |
| @@ -410,21 +599,9 @@ | ||
| 410 | 599 | * @param string $url Configured icon URL. |
| 411 | 600 | * @return int Attachment ID, or 0. |
| 412 | 601 | */ |
| 413 | 602 | public static function icon_attachment_id(string $url): int { |
| 414 | - $attachment_id = (int) attachment_url_to_postid($url); | |
| 415 | - | |
| 416 | - if ($attachment_id) { | |
| 417 | - return $attachment_id; | |
| 418 | - } | |
| 419 | - | |
| 420 | - $original = preg_replace('/-\d+x\d+(?=\.[a-zA-Z0-9]+$)/', '', $url); | |
| 421 | - | |
| 422 | - if (is_string($original) && $original !== $url) { | |
| 423 | - return (int) attachment_url_to_postid($original); | |
| 424 | - } | |
| 425 | - | |
| 426 | - return 0; | |
| 603 | + return Attachment_Lookup::id_from_url($url); | |
| 427 | 604 | } |
| 428 | 605 | |
| 429 | 606 | /** |
| 430 | 607 | * Which ICON_SIZES derivatives this attachment still needs. |
| @@ -709,8 +886,28 @@ | ||
| 709 | 886 | $body = ($custom !== '' && !$fully_blocked) |
| 710 | 887 | ? $this->strip_robots_header($custom) |
| 711 | 888 | : trim($this->generate_robots_txt()['content']); |
| 712 | 889 | |
| 890 | + // The per-agent AI directives are machine-owned, so they are composed | |
| 891 | + // here rather than stored: the textarea holds the user's body, with | |
| 892 | + // the fenced block stripped out of every read and re-applied on every | |
| 893 | + // render. A site-wide block already disallows everyone, so adding the | |
| 894 | + // per-agent group there would be noise restating the same refusal. | |
| 895 | + // Composed here rather than read from storage, for the same reason as | |
| 896 | + // the AI block below: the set of sitemaps an install publishes is a | |
| 897 | + // runtime fact. `robots_txt_content` is a snapshot of it taken at the | |
| 898 | + // last save, and nothing invalidated that snapshot, so deactivating a | |
| 899 | + // sitemap provider left its URL advertised and serving HTML (#835). | |
| 900 | + // Composing it on every render means the advertisement agrees with what | |
| 901 | + // the install publishes, in both directions, with no cache to expire. | |
| 902 | + if (!$fully_blocked) { | |
| 903 | + $body = $this->apply_sitemap_block($body); | |
| 904 | + } | |
| 905 | + | |
| 906 | + if (!$fully_blocked) { | |
| 907 | + $body = $this->apply_ai_crawler_block($body, $settings); | |
| 908 | + } | |
| 909 | + | |
| 713 | 910 | if ($body === '') { |
| 714 | 911 | return ''; |
| 715 | 912 | } |
| 716 | 913 | |
| @@ -717,8 +914,87 @@ | ||
| 717 | 914 | return $this->robots_txt_header() . $body . "\n"; |
| 718 | 915 | } |
| 719 | 916 | |
| 720 | 917 | /** |
| 918 | + * Replace the generated Sitemap block with the one this install publishes. | |
| 919 | + * | |
| 920 | + * @since 2.14.0 | |
| 921 | + * @param string $body Robots.txt body, without the header. | |
| 922 | + * @return string | |
| 923 | + */ | |
| 924 | + private function apply_sitemap_block(string $body): string { | |
| 925 | + $stripped = $this->strip_generated_sitemap_block($body); | |
| 926 | + $urls = $this->get_sitemap_urls_for_robots(); | |
| 927 | + | |
| 928 | + if (empty($urls)) { | |
| 929 | + return $stripped; | |
| 930 | + } | |
| 931 | + | |
| 932 | + $block = ''; | |
| 933 | + foreach ($urls as $url) { | |
| 934 | + $block .= 'Sitemap: ' . $url . "\n"; | |
| 935 | + } | |
| 936 | + | |
| 937 | + if ('' === trim($stripped)) { | |
| 938 | + return trim($block); | |
| 939 | + } | |
| 940 | + | |
| 941 | + // The grammar build_robots_txt_content() writes: one blank line before | |
| 942 | + // the block, none inside it. A blank line terminates a record in the | |
| 943 | + // robots.txt grammar, so a line between every directive is invalid. | |
| 944 | + return rtrim($stripped) . "\n\n" . trim($block); | |
| 945 | + } | |
| 946 | + | |
| 947 | + /** | |
| 948 | + * Remove the plugin-written Sitemap block from a stored body. | |
| 949 | + * | |
| 950 | + * Only the trailing run of `Sitemap:` lines is removed, which is the exact | |
| 951 | + * shape `build_robots_txt_content()` writes: a blank line, then nothing but | |
| 952 | + * `Sitemap:` lines to the end of the body. A `Sitemap:` line anywhere else | |
| 953 | + * was typed by the site owner and is left exactly where they put it, which | |
| 954 | + * is why this cannot simply strip every matching line. | |
| 955 | + * | |
| 956 | + * @since 2.14.0 | |
| 957 | + * @param string $body Robots.txt body. | |
| 958 | + * @return string | |
| 959 | + */ | |
| 960 | + private function strip_generated_sitemap_block(string $body): string { | |
| 961 | + $lines = preg_split('/\R/', $body); | |
| 962 | + | |
| 963 | + if (!is_array($lines)) { | |
| 964 | + return $body; | |
| 965 | + } | |
| 966 | + | |
| 967 | + $cut = count($lines); | |
| 968 | + | |
| 969 | + // Walk back over the trailing block: sitemap lines, and the blank lines | |
| 970 | + // that separate or pad it. Anything else ends the block. | |
| 971 | + for ($i = count($lines) - 1; $i >= 0; $i--) { | |
| 972 | + $line = trim($lines[$i]); | |
| 973 | + | |
| 974 | + if ('' === $line) { | |
| 975 | + $cut = $i; | |
| 976 | + continue; | |
| 977 | + } | |
| 978 | + | |
| 979 | + if (0 === stripos($line, 'sitemap:')) { | |
| 980 | + $cut = $i; | |
| 981 | + continue; | |
| 982 | + } | |
| 983 | + | |
| 984 | + break; | |
| 985 | + } | |
| 986 | + | |
| 987 | + if ($cut >= count($lines)) { | |
| 988 | + return $body; | |
| 989 | + } | |
| 990 | + | |
| 991 | + // Nothing but sitemap lines in the whole body means there is no owner | |
| 992 | + // content to keep. | |
| 993 | + return rtrim(implode("\n", array_slice($lines, 0, $cut))); | |
| 994 | + } | |
| 995 | + | |
| 996 | + /** | |
| 721 | 997 | * Resolve the robots.txt actually served to crawlers, with its origin. |
| 722 | 998 | * |
| 723 | 999 | * Lets an API/MCP consumer see the effective output without crawling the |
| 724 | 1000 | * URL. Mirrors serving precedence: a physical robots.txt in the web root is |
| @@ -786,9 +1062,12 @@ | ||
| 786 | 1062 | $settings = $this->get_settings('site'); |
| 787 | 1063 | |
| 788 | 1064 | // Compare bodies, not raw strings: the auto-generated header carries a |
| 789 | 1065 | // regeneration timestamp that always differs and means nothing here. |
| 790 | - $served = $this->strip_robots_header($effective['content']); | |
| 1066 | + // The AI crawler block is composed at render time on both sides, so it | |
| 1067 | + // is identical by construction and comparing it would only ever report | |
| 1068 | + // a false drift the admin cannot act on. | |
| 1069 | + $served = $this->strip_ai_crawler_block($this->strip_robots_header($effective['content'])); | |
| 791 | 1070 | |
| 792 | 1071 | // Measure against the body the editor is displaying — get_served_robots_body() |
| 793 | 1072 | // — not against render_robots_txt(). Two things made the old comparison |
| 794 | 1073 | // report "in sync" while the screen showed rules no crawler receives: |
| @@ -1166,17 +1445,19 @@ | ||
| 1166 | 1445 | $home_text = $settings['breadcrumb_home_text'] ?? 'Home'; |
| 1167 | 1446 | if (empty($home_text)) { |
| 1168 | 1447 | $optimization['warnings'][] = 'Empty home text reduces accessibility for screen readers'; |
| 1169 | 1448 | $optimization['score'] -= 15; |
| 1170 | - } elseif (strlen($home_text) > 20) { | |
| 1171 | - $optimization['suggestions'][] = 'Keep home text concise (current: ' . strlen($home_text) . ' chars)'; | |
| 1449 | + } elseif (mb_strlen($home_text) > 20) { | |
| 1450 | + // mb_strlen: this number is shown to the user as "chars" (#687). | |
| 1451 | + $optimization['suggestions'][] = 'Keep home text concise (current: ' . mb_strlen($home_text) . ' chars)'; | |
| 1172 | 1452 | $optimization['score'] -= 5; |
| 1173 | 1453 | } |
| 1174 | 1454 | |
| 1175 | 1455 | // Check prefix usage |
| 1176 | 1456 | $prefix = $settings['breadcrumb_prefix'] ?? ''; |
| 1177 | - if (!empty($prefix) && strlen($prefix) > 50) { | |
| 1178 | - $optimization['suggestions'][] = 'Breadcrumb prefix is quite long (' . strlen($prefix) . ' chars) - consider shortening'; | |
| 1457 | + if (!empty($prefix) && mb_strlen($prefix) > 50) { | |
| 1458 | + // mb_strlen: this number is shown to the user as "chars" (#687). | |
| 1459 | + $optimization['suggestions'][] = 'Breadcrumb prefix is quite long (' . mb_strlen($prefix) . ' chars) - consider shortening'; | |
| 1179 | 1460 | $optimization['score'] -= 5; |
| 1180 | 1461 | } |
| 1181 | 1462 | |
| 1182 | 1463 | // Current page display |
| @@ -1307,13 +1588,16 @@ | ||
| 1307 | 1588 | } |
| 1308 | 1589 | |
| 1309 | 1590 | // Additional logo analysis for local images |
| 1310 | 1591 | if (!empty($logo_url) && filter_var($logo_url, FILTER_VALIDATE_URL)) { |
| 1311 | - $attachment_id = attachment_url_to_postid($logo_url); | |
| 1592 | + $attachment_id = Attachment_Lookup::id_from_url($logo_url); | |
| 1312 | 1593 | if ($attachment_id) { |
| 1313 | 1594 | $image_meta = wp_get_attachment_metadata($attachment_id); |
| 1314 | - $width = isset($image_meta['width']) ? (int) $image_meta['width'] : 0; | |
| 1315 | - $height = isset($image_meta['height']) ? (int) $image_meta['height'] : 0; | |
| 1595 | + // The configured file's own size — a logo picked at a generated | |
| 1596 | + // size is not as large as the upload behind it. | |
| 1597 | + $logo_file = Attachment_Lookup::describe($attachment_id, $logo_url); | |
| 1598 | + $width = $logo_file['width']; | |
| 1599 | + $height = $logo_file['height']; | |
| 1316 | 1600 | |
| 1317 | 1601 | // SVG logos store 0x0 metadata — no dimension/ratio analysis |
| 1318 | 1602 | // is possible (and dividing by 0 is fatal). |
| 1319 | 1603 | if ($image_meta && $width > 0 && $height > 0) { |
| @@ -1416,11 +1700,12 @@ | ||
| 1416 | 1700 | $optimization['score'] -= 15; |
| 1417 | 1701 | } |
| 1418 | 1702 | } |
| 1419 | 1703 | |
| 1420 | - // Business type validation | |
| 1421 | - if (empty($settings['business_type'])) { | |
| 1422 | - $optimization['suggestions'][] = 'Select a specific business type for better schema markup'; | |
| 1704 | + // Business type validation (shared rule, one message — #622). | |
| 1705 | + $business_type = $this->business_type_status($settings); | |
| 1706 | + if ('suggestion' === $business_type['status']) { | |
| 1707 | + $optimization['suggestions'][] = $business_type['message']; | |
| 1423 | 1708 | $optimization['score'] -= 5; |
| 1424 | 1709 | } |
| 1425 | 1710 | |
| 1426 | 1711 | // Email validation |
| @@ -1950,24 +2235,17 @@ | ||
| 1950 | 2235 | 'icon' => '✗' |
| 1951 | 2236 | ]; |
| 1952 | 2237 | } |
| 1953 | 2238 | |
| 1954 | - // Business Type validation | |
| 1955 | - if (!empty($settings['business_type']) && $settings['business_type'] !== 'LocalBusiness') { | |
| 1956 | - $field_details[] = [ | |
| 1957 | - 'field' => 'business_type', | |
| 1958 | - 'label' => 'Business type is selected for proper schema markup.', | |
| 1959 | - 'status' => 'valid', | |
| 1960 | - 'icon' => '✓' | |
| 1961 | - ]; | |
| 1962 | - } else { | |
| 1963 | - $field_details[] = [ | |
| 1964 | - 'field' => 'business_type', | |
| 1965 | - 'label' => 'Specific business type selection recommended for better schema markup.', | |
| 1966 | - 'status' => 'suggestion', | |
| 1967 | - 'icon' => '⚠' | |
| 1968 | - ]; | |
| 1969 | - } | |
| 2239 | + // Business Type validation — see business_type_status() for why there | |
| 2240 | + // is exactly one rule here now (#622). | |
| 2241 | + $business_type = $this->business_type_status($settings); | |
| 2242 | + $field_details[] = [ | |
| 2243 | + 'field' => 'business_type', | |
| 2244 | + 'label' => $business_type['message'], | |
| 2245 | + 'status' => $business_type['status'], | |
| 2246 | + 'icon' => 'valid' === $business_type['status'] ? '✓' : '⚠', | |
| 2247 | + ]; | |
| 1970 | 2248 | |
| 1971 | 2249 | // Address validation (NAP consistency) |
| 1972 | 2250 | $address_fields = ['business_address', 'business_city', 'business_state', 'business_country']; |
| 1973 | 2251 | $address_complete = true; |
| @@ -2651,11 +2929,15 @@ | ||
| 2651 | 2929 | } else { |
| 2652 | 2930 | $validation['suggestions'][] = 'Add business hours to improve local search visibility'; |
| 2653 | 2931 | } |
| 2654 | 2932 | |
| 2655 | - // Validate business type | |
| 2656 | - if (empty($settings['business_type'])) { | |
| 2657 | - $validation['suggestions'][] = 'Select a specific business type for better schema markup'; | |
| 2933 | + // Business type, through the shared rule (#622). This is the only place | |
| 2934 | + // it is reported on the generic path: validate_settings() with no tab | |
| 2935 | + // context attaches basic-info field details, not business-info ones, so | |
| 2936 | + // without this the setting would go unreported there entirely. | |
| 2937 | + $business_type = $this->business_type_status($settings); | |
| 2938 | + if ('suggestion' === $business_type['status']) { | |
| 2939 | + $validation['suggestions'][] = $business_type['message']; | |
| 2658 | 2940 | } |
| 2659 | 2941 | |
| 2660 | 2942 | return $validation; |
| 2661 | 2943 | } |
| @@ -2747,13 +3029,54 @@ | ||
| 2747 | 3029 | * @since 2.0.1 |
| 2748 | 3030 | * |
| 2749 | 3031 | * @return string[] |
| 2750 | 3032 | */ |
| 3033 | + /** | |
| 3034 | + * The stored alternate name(s), shaped for schema output. | |
| 3035 | + * | |
| 3036 | + * schema.org and Google both allow `alternateName` to carry one value or | |
| 3037 | + * several, and the store already round-trips either shape, so this accepts | |
| 3038 | + * both and normalises: null when there is nothing to publish, a bare string | |
| 3039 | + * for one name, a list for more. Emitting a one-element array would be | |
| 3040 | + * valid but noisier than it needs to be. | |
| 3041 | + * | |
| 3042 | + * Shared because both WebSite producers need it and must agree — a property | |
| 3043 | + * added to one and not the other is how #688 happened. | |
| 3044 | + * | |
| 3045 | + * @since 2.7.0 | |
| 3046 | + * | |
| 3047 | + * @param mixed $value Stored alternate_name value. | |
| 3048 | + * @return string|string[]|null | |
| 3049 | + */ | |
| 3050 | + public static function alternate_name_for_schema($value) { | |
| 3051 | + $names = []; | |
| 3052 | + | |
| 3053 | + foreach ((array) $value as $name) { | |
| 3054 | + if (!is_scalar($name)) { | |
| 3055 | + continue; | |
| 3056 | + } | |
| 3057 | + | |
| 3058 | + $name = trim((string) $name); | |
| 3059 | + | |
| 3060 | + if ('' !== $name && !in_array($name, $names, true)) { | |
| 3061 | + $names[] = $name; | |
| 3062 | + } | |
| 3063 | + } | |
| 3064 | + | |
| 3065 | + if (empty($names)) { | |
| 3066 | + return null; | |
| 3067 | + } | |
| 3068 | + | |
| 3069 | + return 1 === count($names) ? $names[0] : $names; | |
| 3070 | + } | |
| 3071 | + | |
| 2751 | 3072 | protected function additional_setting_keys(): array { |
| 2752 | 3073 | return [ |
| 2753 | 3074 | // Title formats, one per context. |
| 2754 | 3075 | 'homepage_title', 'post_title', 'page_title', 'category_title', |
| 2755 | 3076 | 'tag_title', 'author_title', 'search_title', 'archive_title', |
| 3077 | + // The blog-index homepage's meta description (#897). | |
| 3078 | + 'homepage_description', | |
| 2756 | 3079 | // Breadcrumbs. |
| 2757 | 3080 | 'breadcrumb_prefix', 'show_current_page', 'breadcrumb_use_seo_title', |
| 2758 | 3081 | // Identity, as written by the setup wizard and the importers. |
| 2759 | 3082 | 'alternate_name', 'identity_type', 'represents', |
| @@ -2762,8 +3085,10 @@ | ||
| 2762 | 3085 | // Schema toggles that live on this screen. |
| 2763 | 3086 | 'organization_schema', 'knowledge_graph', |
| 2764 | 3087 | // Robots rules composed by the Robots.txt panel. |
| 2765 | 3088 | 'custom_robots_rules', |
| 3089 | + // Per-agent AI crawler allow/block map (#657). | |
| 3090 | + 'ai_crawler_rules', | |
| 2766 | 3091 | // Hero section. |
| 2767 | 3092 | 'hero_title', 'hero_subtitle', 'hero_cta_text', 'hero_cta_url', |
| 2768 | 3093 | 'hero_background_image', |
| 2769 | 3094 | // Local SEO / business details. |
| @@ -2775,8 +3100,118 @@ | ||
| 2775 | 3100 | ]; |
| 2776 | 3101 | } |
| 2777 | 3102 | |
| 2778 | 3103 | /** |
| 3104 | + * Sanitize settings, normalising the AI crawler rule map. | |
| 3105 | + * | |
| 3106 | + * The generic array sanitizer keeps the shape but says nothing about the | |
| 3107 | + * values: a payload could store `ai_crawler_rules[gptbot] = "maybe"`, or a | |
| 3108 | + * slug no crawler answers to, and both would round-trip through every | |
| 3109 | + * later response. Normalising here rather than in the REST handler puts it | |
| 3110 | + * on the one path every writer shares — the settings route, the robots | |
| 3111 | + * route and the MCP abilities all land in save_settings() (#657). | |
| 3112 | + * | |
| 3113 | + * @since 2.5.0 | |
| 3114 | + * | |
| 3115 | + * @param array $settings Settings to sanitize. | |
| 3116 | + * @param string $context_type Context type. | |
| 3117 | + * @return array Sanitized settings. | |
| 3118 | + */ | |
| 3119 | + protected function sanitize_settings(array $settings, string $context_type = 'site'): array { | |
| 3120 | + $sanitized = parent::sanitize_settings($settings, $context_type); | |
| 3121 | + | |
| 3122 | + if (array_key_exists('ai_crawler_rules', $sanitized)) { | |
| 3123 | + $sanitized['ai_crawler_rules'] = AI_Crawlers::normalize_rules($sanitized['ai_crawler_rules']); | |
| 3124 | + } | |
| 3125 | + | |
| 3126 | + // Same reasoning one key up, for the scheme override (#638). Anything | |
| 3127 | + // that is not one of the three modes means "follow WordPress", and is | |
| 3128 | + // stored as that rather than kept verbatim — otherwise get-site-identity | |
| 3129 | + // -settings would report a scheme the site does not actually publish. | |
| 3130 | + if (array_key_exists('canonical_scheme', $sanitized)) { | |
| 3131 | + $sanitized['canonical_scheme'] = in_array($sanitized['canonical_scheme'], Url_Scheme::MODES, true) | |
| 3132 | + ? $sanitized['canonical_scheme'] | |
| 3133 | + : Url_Scheme::AUTOMATIC; | |
| 3134 | + } | |
| 3135 | + | |
| 3136 | + // Same reasoning again for the business type. It goes straight into | |
| 3137 | + // LocalBusiness schema, so a type that is not in the schema.org | |
| 3138 | + // vocabulary is invalid structured data — and storing it verbatim would | |
| 3139 | + // have get-site-identity-settings report a type the site cannot | |
| 3140 | + // actually publish. An empty value keeps meaning "not set"; anything | |
| 3141 | + // else unrecognised falls back to the general-purpose root (#623). | |
| 3142 | + if (array_key_exists('business_type', $sanitized)) { | |
| 3143 | + $type = (string) $sanitized['business_type']; | |
| 3144 | + | |
| 3145 | + if ('' !== $type && !\ThinkRank\Config\Local_Business_Types_Config::is_valid($type)) { | |
| 3146 | + $type = \ThinkRank\Config\Local_Business_Types_Config::ROOT; | |
| 3147 | + } | |
| 3148 | + | |
| 3149 | + $sanitized['business_type'] = $type; | |
| 3150 | + } | |
| 3151 | + | |
| 3152 | + return $sanitized; | |
| 3153 | + } | |
| 3154 | + | |
| 3155 | + /** | |
| 3156 | + * schema.org's general-purpose LocalBusiness type. | |
| 3157 | + * | |
| 3158 | + * The default, the first option in the control, and a valid answer in its | |
| 3159 | + * own right — which is the whole point of #622. | |
| 3160 | + * | |
| 3161 | + * @since 2.10.0 | |
| 3162 | + * @var string | |
| 3163 | + */ | |
| 3164 | + private const GENERAL_BUSINESS_TYPE = 'LocalBusiness'; | |
| 3165 | + | |
| 3166 | + /** | |
| 3167 | + * The one rule for whether a business type needs the user's attention. | |
| 3168 | + * | |
| 3169 | + * There were three, with two wordings and two different conditions. Two | |
| 3170 | + * fired when the value was empty; the third fired when it WAS | |
| 3171 | + * `LocalBusiness` — which is the default, the first option in the control | |
| 3172 | + * and a perfectly valid schema.org type. So the warning appeared out of the | |
| 3173 | + * box for every site, could not be cleared without choosing a type that | |
| 3174 | + * might be inaccurate, and on an empty value it appeared three times in two | |
| 3175 | + * different phrasings, which is why it was reported as showing twice (#622). | |
| 3176 | + * | |
| 3177 | + * The rule now: a type is expected, and any type in the vocabulary is a | |
| 3178 | + * correct answer. Only an unset value is worth prompting about. | |
| 3179 | + * `LocalBusiness` is the general-purpose answer and is accepted as one — | |
| 3180 | + * with a note that a more specific type sharpens the schema, phrased as the | |
| 3181 | + * guidance it is rather than as a fault the user has to clear. | |
| 3182 | + * | |
| 3183 | + * @since 2.10.0 | |
| 3184 | + * | |
| 3185 | + * @param array $settings Site identity settings. | |
| 3186 | + * @return array{status:string,message:string} `valid` or `suggestion`. | |
| 3187 | + */ | |
| 3188 | + private function business_type_status(array $settings): array { | |
| 3189 | + $type = trim((string) ($settings['business_type'] ?? '')); | |
| 3190 | + | |
| 3191 | + if ('' === $type) { | |
| 3192 | + return [ | |
| 3193 | + 'status' => 'suggestion', | |
| 3194 | + 'message' => __('Select a business type so your local schema describes the right kind of business.', 'thinkrank'), | |
| 3195 | + ]; | |
| 3196 | + } | |
| 3197 | + | |
| 3198 | + // The literal rather than a constant from the expanded type list (#623): | |
| 3199 | + // that lands on its own branch, and this fix must not wait on it. | |
| 3200 | + if (self::GENERAL_BUSINESS_TYPE === $type) { | |
| 3201 | + return [ | |
| 3202 | + 'status' => 'valid', | |
| 3203 | + 'message' => __('Business type is set to Local Business. A more specific type sharpens your schema, if one fits.', 'thinkrank'), | |
| 3204 | + ]; | |
| 3205 | + } | |
| 3206 | + | |
| 3207 | + return [ | |
| 3208 | + 'status' => 'valid', | |
| 3209 | + 'message' => __('Business type is selected for proper schema markup.', 'thinkrank'), | |
| 3210 | + ]; | |
| 3211 | + } | |
| 3212 | + | |
| 3213 | + /** | |
| 2779 | 3214 | * Get default settings for a context type (implements interface) |
| 2780 | 3215 | * |
| 2781 | 3216 | * @since 1.0.0 |
| 2782 | 3217 | * |
| @@ -2796,9 +3231,32 @@ | ||
| 2796 | 3231 | 'breadcrumb_home_text' => 'Home', |
| 2797 | 3232 | 'breadcrumb_separator' => '>', |
| 2798 | 3233 | 'robots_txt_enabled' => true, |
| 2799 | 3234 | 'allow_search_engines' => true, |
| 3235 | + // Answer 404 when a content selector in the URL resolved to | |
| 3236 | + // nothing (#634). On by default, unlike the other new settings | |
| 3237 | + // here: it changes no URL a visitor or a correct crawler uses, only | |
| 3238 | + // ones where WordPress resolved nothing and served the blog listing | |
| 3239 | + // at 200 anyway. | |
| 3240 | + 'query_protection' => true, | |
| 3241 | + | |
| 3242 | + // Feed controls (#635). All three off, so an upgrade changes | |
| 3243 | + // nothing about what an existing site already sends its | |
| 3244 | + // subscribers; a brand-new install is seeded with the signature and | |
| 3245 | + // the noindex on, in Activator::seed_feed_defaults(). | |
| 3246 | + 'feed_excerpt_only' => false, | |
| 3247 | + 'feed_source_link' => false, | |
| 3248 | + 'feed_noindex' => false, | |
| 3249 | + | |
| 3250 | + // The scheme self-referential URLs go out with (#638). 'automatic' | |
| 3251 | + // means substitute nothing and follow WordPress, which is what | |
| 3252 | + // every site did before the setting existed. | |
| 3253 | + 'canonical_scheme' => Url_Scheme::AUTOMATIC, | |
| 2800 | 3254 | 'robots_txt_content' => '', |
| 3255 | + // Empty map = every AI crawler allowed. Defaults must stay | |
| 3256 | + // permissive so an upgrade never starts blocking a crawler a site | |
| 3257 | + // was happily serving (#657). | |
| 3258 | + 'ai_crawler_rules' => [], | |
| 2801 | 3259 | 'logo_url' => '', |
| 2802 | 3260 | 'favicon_url' => '', |
| 2803 | 3261 | 'apple_touch_icon_url' => '' |
| 2804 | 3262 | ]; |
| @@ -3007,16 +3465,17 @@ | ||
| 3007 | 3465 | // Remove extra whitespace |
| 3008 | 3466 | $title = preg_replace('/\s+/', ' ', $title); |
| 3009 | 3467 | $title = trim($title); |
| 3010 | 3468 | |
| 3011 | - // Ensure title is not too long (60 characters max for SEO) | |
| 3012 | - if (strlen($title) > 60) { | |
| 3013 | - // Try to truncate at word boundary | |
| 3014 | - $title = wp_trim_words($title, 8, '...'); | |
| 3015 | - if (strlen($title) > 60) { | |
| 3016 | - $title = substr($title, 0, 57) . '...'; | |
| 3017 | - } | |
| 3018 | - } | |
| 3469 | + // Ensure title is not too long (60 characters max for SEO). | |
| 3470 | + // All three units here were wrong for non-Latin text: strlen() counts | |
| 3471 | + // BYTES so the gate fired at 20 Thai characters, wp_trim_words() counts | |
| 3472 | + // CHARACTERS on th/ja/zh_* so `8` cut the title to 8 of them, and | |
| 3473 | + // substr() cuts bytes so it split a character mid-sequence (#687). | |
| 3474 | + $title = \ThinkRank\Core\Seo_Text::trim_to_length( | |
| 3475 | + $title, | |
| 3476 | + \ThinkRank\Core\Seo_Text::TITLE_MAX_LENGTH | |
| 3477 | + ); | |
| 3019 | 3478 | |
| 3020 | 3479 | // Ensure title is not empty |
| 3021 | 3480 | if (empty($title)) { |
| 3022 | 3481 | $title = get_bloginfo('name'); |
| @@ -3406,8 +3865,29 @@ | ||
| 3406 | 3865 | * @since 1.0.0 |
| 3407 | 3866 | * @return array Array of sitemap URLs |
| 3408 | 3867 | */ |
| 3409 | 3868 | private function get_sitemap_urls_for_robots(): array { |
| 3869 | + // One wrapper over every return path below, including the #104 extras. | |
| 3870 | + // The Sitemap: line is the only absolute URL of ours in robots.txt and | |
| 3871 | + // the one a crawler follows to find everything else, so it has to carry | |
| 3872 | + // the site's scheme preference (#638). Applied here rather than where | |
| 3873 | + // the body is assembled, because that path also renders a robots.txt a | |
| 3874 | + // site owner typed themselves, and their text is not ours to rewrite. | |
| 3875 | + return array_map( | |
| 3876 | + static function (string $url): string { | |
| 3877 | + return Url_Scheme::apply($url); | |
| 3878 | + }, | |
| 3879 | + $this->collect_sitemap_urls_for_robots() | |
| 3880 | + ); | |
| 3881 | + } | |
| 3882 | + | |
| 3883 | + /** | |
| 3884 | + * The sitemap URLs robots.txt advertises, before the scheme preference. | |
| 3885 | + * | |
| 3886 | + * @since 1.0.0 | |
| 3887 | + * @return array Array of sitemap URLs | |
| 3888 | + */ | |
| 3889 | + private function collect_sitemap_urls_for_robots(): array { | |
| 3410 | 3890 | try { |
| 3411 | 3891 | // Get sitemap settings |
| 3412 | 3892 | $sitemap_generator = new \ThinkRank\SEO\Sitemap_Generator(); |
| 3413 | 3893 | $sitemap_settings = $sitemap_generator->get_settings('site'); |
| @@ -3440,11 +3920,29 @@ | ||
| 3440 | 3920 | } |
| 3441 | 3921 | } |
| 3442 | 3922 | |
| 3443 | 3923 | if ($index_url !== '') { |
| 3444 | - // The index alone — it covers the children and, on a segmented | |
| 3445 | - // install, the local business sitemap too. | |
| 3446 | - return [$index_url]; | |
| 3924 | + // The index covers the children and, on a segmented install, | |
| 3925 | + // the local business sitemap too. | |
| 3926 | + // | |
| 3927 | + // It does not cover a sitemap contributed through | |
| 3928 | + // `thinkrank_additional_sitemaps`: the index is built by this | |
| 3929 | + // plugin's own generator and never lists them. Returning the | |
| 3930 | + // index alone therefore left a contributed sitemap with no | |
| 3931 | + // discovery path at all — absent from robots.txt and absent | |
| 3932 | + // from the index — so Pro's News sitemap was unreachable on any | |
| 3933 | + // install with the index enabled, which is the default (#835). | |
| 3934 | + $contributed = []; | |
| 3935 | + | |
| 3936 | + foreach (\ThinkRank\SEO\Sitemap_Generator::additional_sitemaps() as $path) { | |
| 3937 | + $url = home_url($path); | |
| 3938 | + | |
| 3939 | + if ($url !== $index_url && !in_array($url, $contributed, true)) { | |
| 3940 | + $contributed[] = $url; | |
| 3941 | + } | |
| 3942 | + } | |
| 3943 | + | |
| 3944 | + return array_merge([$index_url], $contributed); | |
| 3447 | 3945 | } |
| 3448 | 3946 | |
| 3449 | 3947 | // Fallback to default if no URLs found |
| 3450 | 3948 | if (empty($sitemap_urls)) { |
| @@ -3456,9 +3954,22 @@ | ||
| 3456 | 3954 | // business sitemap and the sitemaps other plugins register both land |
| 3457 | 3955 | // here for the same reason, so they go through one list (#104). |
| 3458 | 3956 | $extra = []; |
| 3459 | 3957 | |
| 3460 | - if (file_exists(ABSPATH . 'local-sitemap.xml')) { | |
| 3958 | + // Not a file test. Under dynamic delivery the local sitemap is | |
| 3959 | + // served from PHP and no file is ever written, so file_exists() | |
| 3960 | + // silently dropped a sitemap the site really does publish (#752). | |
| 3961 | + // On static sites the file is still what proves it, so both count. | |
| 3962 | + $local_sitemap_published = file_exists(ABSPATH . 'local-sitemap.xml'); | |
| 3963 | + | |
| 3964 | + if (!$local_sitemap_published && class_exists('ThinkRank\\SEO\\Sitemap_Generator')) { | |
| 3965 | + $generator = new \ThinkRank\SEO\Sitemap_Generator(false); | |
| 3966 | + | |
| 3967 | + $local_sitemap_published = 'dynamic' === $generator->resolve_delivery_mode() | |
| 3968 | + && $generator->publishes_local_sitemap(); | |
| 3969 | + } | |
| 3970 | + | |
| 3971 | + if ($local_sitemap_published) { | |
| 3461 | 3972 | $extra[] = '/local-sitemap.xml'; |
| 3462 | 3973 | } |
| 3463 | 3974 | |
| 3464 | 3975 | foreach (\ThinkRank\SEO\Sitemap_Generator::additional_sitemaps() as $path) { |
| @@ -3654,8 +4165,116 @@ | ||
| 3654 | 4165 | return $rules; |
| 3655 | 4166 | } |
| 3656 | 4167 | |
| 3657 | 4168 | /** |
| 4169 | + * Opening fence of the machine-owned AI crawler region. | |
| 4170 | + * | |
| 4171 | + * @since 2.5.0 | |
| 4172 | + * @var string | |
| 4173 | + */ | |
| 4174 | + public const AI_BLOCK_BEGIN = '# BEGIN ThinkRank AI crawlers'; | |
| 4175 | + | |
| 4176 | + /** | |
| 4177 | + * Closing fence of the machine-owned AI crawler region. | |
| 4178 | + * | |
| 4179 | + * @since 2.5.0 | |
| 4180 | + * @var string | |
| 4181 | + */ | |
| 4182 | + public const AI_BLOCK_END = '# END ThinkRank AI crawlers'; | |
| 4183 | + | |
| 4184 | + /** | |
| 4185 | + * Render the fenced AI crawler region for the current settings. | |
| 4186 | + * | |
| 4187 | + * One `User-agent:` / `Disallow: /` record per blocked crawler. Allowed | |
| 4188 | + * crawlers emit nothing at all: `Disallow:` with an empty value is the | |
| 4189 | + * robots.txt way of saying "allow everything", but writing eighteen such | |
| 4190 | + * records to say what silence already says would triple the file and | |
| 4191 | + * invite the reading that an unlisted crawler is therefore refused. | |
| 4192 | + * | |
| 4193 | + * @since 2.5.0 | |
| 4194 | + * | |
| 4195 | + * @param array $settings Site settings. | |
| 4196 | + * @return string Fenced block, newline-terminated, or '' when nothing is blocked. | |
| 4197 | + */ | |
| 4198 | + private function build_ai_crawler_block(array $settings): string { | |
| 4199 | + $blocked = AI_Crawlers::blocked_slugs($settings['ai_crawler_rules'] ?? []); | |
| 4200 | + | |
| 4201 | + if (empty($blocked)) { | |
| 4202 | + return ''; | |
| 4203 | + } | |
| 4204 | + | |
| 4205 | + $agents = AI_Crawlers::all(); | |
| 4206 | + | |
| 4207 | + $lines = [ | |
| 4208 | + self::AI_BLOCK_BEGIN, | |
| 4209 | + '# Managed by ThinkRank — edits between these lines are overwritten.', | |
| 4210 | + ]; | |
| 4211 | + | |
| 4212 | + foreach ($blocked as $slug) { | |
| 4213 | + $lines[] = ''; | |
| 4214 | + $lines[] = 'User-agent: ' . $agents[$slug]['token']; | |
| 4215 | + $lines[] = 'Disallow: /'; | |
| 4216 | + } | |
| 4217 | + | |
| 4218 | + $lines[] = self::AI_BLOCK_END; | |
| 4219 | + | |
| 4220 | + return implode("\n", $lines) . "\n"; | |
| 4221 | + } | |
| 4222 | + | |
| 4223 | + /** | |
| 4224 | + * Remove the fenced AI crawler region from a robots.txt body. | |
| 4225 | + * | |
| 4226 | + * Tolerates a missing closing fence rather than leaving the rest of the | |
| 4227 | + * file swallowed: a truncated write, or someone deleting the END line by | |
| 4228 | + * hand, would otherwise make every subsequent read drop everything below | |
| 4229 | + * the opening fence. | |
| 4230 | + * | |
| 4231 | + * @since 2.5.0 | |
| 4232 | + * | |
| 4233 | + * @param string $body Robots.txt body. | |
| 4234 | + * @return string Body with the region removed. | |
| 4235 | + */ | |
| 4236 | + public function strip_ai_crawler_block(string $body): string { | |
| 4237 | + if (false === strpos($body, self::AI_BLOCK_BEGIN)) { | |
| 4238 | + return $body; | |
| 4239 | + } | |
| 4240 | + | |
| 4241 | + $pattern = '/\R*' . preg_quote(self::AI_BLOCK_BEGIN, '/') | |
| 4242 | + . '.*?(?:' . preg_quote(self::AI_BLOCK_END, '/') . '|\z)\R*/s'; | |
| 4243 | + | |
| 4244 | + return trim((string) preg_replace($pattern, "\n\n", $body, 1)); | |
| 4245 | + } | |
| 4246 | + | |
| 4247 | + /** | |
| 4248 | + * Put the current AI crawler region into a robots.txt body. | |
| 4249 | + * | |
| 4250 | + * Replaces an existing region in place so the block keeps its position in | |
| 4251 | + * a hand-ordered file, and appends when there is none. Everything outside | |
| 4252 | + * the fences is returned untouched — that is the whole point of fencing | |
| 4253 | + * it, since the body is also a free-text field the user edits. | |
| 4254 | + * | |
| 4255 | + * @since 2.5.0 | |
| 4256 | + * | |
| 4257 | + * @param string $body Robots.txt body (fences optional). | |
| 4258 | + * @param array $settings Site settings. | |
| 4259 | + * @return string Body carrying the current region. | |
| 4260 | + */ | |
| 4261 | + private function apply_ai_crawler_block(string $body, array $settings): string { | |
| 4262 | + $stripped = $this->strip_ai_crawler_block($body); | |
| 4263 | + $block = $this->build_ai_crawler_block($settings); | |
| 4264 | + | |
| 4265 | + if ('' === $block) { | |
| 4266 | + return $stripped; | |
| 4267 | + } | |
| 4268 | + | |
| 4269 | + if ('' === trim($stripped)) { | |
| 4270 | + return trim($block); | |
| 4271 | + } | |
| 4272 | + | |
| 4273 | + return rtrim($stripped) . "\n\n" . trim($block); | |
| 4274 | + } | |
| 4275 | + | |
| 4276 | + /** | |
| 3658 | 4277 | * The auto-generated header prepended to the served robots.txt. |
| 3659 | 4278 | * |
| 3660 | 4279 | * Kept separate from the body so it is only ever added at render time with |
| 3661 | 4280 | * a fresh timestamp, never stored or shown in the editable textarea. |
| @@ -3692,11 +4311,16 @@ | ||
| 3692 | 4311 | */ |
| 3693 | 4312 | public function get_served_robots_body(): string { |
| 3694 | 4313 | $settings = $this->get_settings('site'); |
| 3695 | 4314 | |
| 4315 | + // The AI block is stripped from every one of these paths. A physical | |
| 4316 | + // robots.txt we wrote carries it, and the stored override is whatever | |
| 4317 | + // the textarea last held — so without this the block round-trips into | |
| 4318 | + // the editor, gets saved as ordinary body text, and is then appended | |
| 4319 | + // to a second time on the next render. | |
| 3696 | 4320 | $custom = trim((string) ($settings['robots_txt_content'] ?? '')); |
| 3697 | 4321 | if ($custom !== '') { |
| 3698 | - return $this->strip_robots_header($custom); | |
| 4322 | + return $this->strip_ai_crawler_block($this->strip_robots_header($custom)); | |
| 3699 | 4323 | } |
| 3700 | 4324 | |
| 3701 | 4325 | $robots_file = ABSPATH . 'robots.txt'; |
| 3702 | 4326 | if (file_exists($robots_file)) { |
| @@ -3702,13 +4326,13 @@ | ||
| 3702 | 4326 | if (file_exists($robots_file)) { |
| 3703 | 4327 | // phpcs:ignore WordPress.WP.AlternativeFunctions.file_get_contents_file_get_contents, WordPress.PHP.NoSilencedErrors.Discouraged -- an unreadable robots.txt is an expected state answered with an empty string. |
| 3704 | 4328 | $raw = (string) @file_get_contents($robots_file); |
| 3705 | 4329 | if ($raw !== '') { |
| 3706 | - return $this->strip_robots_header($raw); | |
| 4330 | + return $this->strip_ai_crawler_block($this->strip_robots_header($raw)); | |
| 3707 | 4331 | } |
| 3708 | 4332 | } |
| 3709 | 4333 | |
| 3710 | - return trim($this->generate_robots_txt()['content']); | |
| 4334 | + return $this->strip_ai_crawler_block(trim($this->generate_robots_txt()['content'])); | |
| 3711 | 4335 | } |
| 3712 | 4336 | private function get_site_identity_data(array $settings): array { |
| 3713 | 4337 | return [ |
| 3714 | 4338 | 'site_name' => $settings['site_name'] ?? get_bloginfo('name'), |
| @@ -3770,11 +4394,14 @@ | ||
| 3770 | 4394 | $optimization['validation']['valid'] = false; |
| 3771 | 4395 | } |
| 3772 | 4396 | |
| 3773 | 4397 | if (!empty($value) && isset($config['max_length'])) { |
| 3774 | - if (strlen($value) > $config['max_length']) { | |
| 4398 | + // The warning says "characters", so measure and cut in characters: | |
| 4399 | + // strlen()/substr() fired early on non-Latin values and the | |
| 4400 | + // suggested replacement was cut mid-character (#687). | |
| 4401 | + if (mb_strlen($value) > $config['max_length']) { | |
| 3775 | 4402 | $optimization['validation']['warnings'][] = "{$element} exceeds maximum length of {$config['max_length']} characters"; |
| 3776 | - $optimization['optimized_value'] = substr($value, 0, $config['max_length']); | |
| 4403 | + $optimization['optimized_value'] = \ThinkRank\Core\Seo_Text::trim_to_length($value, (int) $config['max_length']); | |
| 3777 | 4404 | } |
| 3778 | 4405 | } |
| 3779 | 4406 | |
| 3780 | 4407 | // SEO-specific optimizations |
| @@ -3813,18 +4440,20 @@ | ||
| 3813 | 4440 | return $optimization; |
| 3814 | 4441 | } |
| 3815 | 4442 | |
| 3816 | 4443 | // Check if it's a local image |
| 3817 | - $attachment_id = attachment_url_to_postid($value); | |
| 4444 | + $attachment_id = Attachment_Lookup::id_from_url($value); | |
| 3818 | 4445 | if ($attachment_id) { |
| 3819 | 4446 | $image_meta = wp_get_attachment_metadata($attachment_id); |
| 3820 | 4447 | |
| 3821 | 4448 | if ($image_meta && isset($image_meta['width'], $image_meta['height'])) { |
| 3822 | - // Check recommended size | |
| 4449 | + // Check recommended size, against the configured file itself | |
| 4450 | + // rather than the upload it may have been generated from. | |
| 3823 | 4451 | if (isset($config['recommended_size'])) { |
| 3824 | 4452 | [$rec_width, $rec_height] = explode('x', $config['recommended_size']); |
| 4453 | + $image_file = Attachment_Lookup::describe($attachment_id, $value); | |
| 3825 | 4454 | |
| 3826 | - if ((int) $image_meta['width'] !== (int) $rec_width || (int) $image_meta['height'] !== (int) $rec_height) { | |
| 4455 | + if ($image_file['width'] !== (int) $rec_width || $image_file['height'] !== (int) $rec_height) { | |
| 3827 | 4456 | $optimization['suggestions'][] = "Consider using {$config['recommended_size']} size for optimal {$element}"; |
| 3828 | 4457 | } |
| 3829 | 4458 | } |
| 3830 | 4459 | |