PluginProbe
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO / 2.14.0
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO v2.14.0
2.14.0 2.13.0 2.12.0 2.11.0 2.10.0 2.9.0 2.8.0 2.7.0 2.6.0 2.5.0 2.4.0 2.3.0 2.2.0 2.1.1 2.1.0 2.0.2 2.0.1 2.0.0 1.32.0 1.31.0 1.30.0 1.29.0 1.28.0 1.27.0 1.26.0 All 55 releases
← All changes | includes/seo/class-site-identity-manager.php +1125 -68 1.32.0 → 2.14.0 View file →
@@ -15,8 +15,13 @@
15 15 declare(strict_types=1);
16 16
17 17 namespace ThinkRank\SEO;
18 18
19 +// Prevent direct access
20 +if (!defined('ABSPATH')) {
21 + exit;
22 +}
23 +
19 24 // Ensure dependencies are loaded
20 25 if (!class_exists('ThinkRank\\SEO\\Abstract_SEO_Manager')) {
21 26 require_once THINKRANK_PLUGIN_DIR . 'includes/seo/class-abstract-seo-manager.php';
22 27 }
@@ -35,8 +40,36 @@
35 40 */
36 41 class Site_Identity_Manager extends Abstract_SEO_Manager {
37 42
38 43 /**
44 + * The per-context title formats as ThinkRank ships them.
45 + *
46 + * These are not in get_default_settings(): the admin screen seeds them on
47 + * first save, so on a real install they are stored values, indistinguishable
48 + * from a template the user typed. The migration needs to tell those two
49 + * apart — it may overwrite a shipped default with an imported template, and
50 + * must never overwrite a choice the user made — so this is the record of
51 + * what "untouched" looks like.
52 + *
53 + * Keep in step with getDefaultSettings() in
54 + * src/admin/components/essential-seo/SiteIdentityTab.js. SiteIdentityTitleFormatDefaultsTest
55 + * fails when the two drift.
56 + *
57 + * @since 2.8.0
58 + * @var array<string, string>
59 + */
60 + public const TITLE_FORMAT_DEFAULTS = [
61 + 'homepage_title' => '%site_title% %sep% %site_description%',
62 + 'post_title' => '%post_title% %sep% %site_title%',
63 + 'page_title' => '%page_title% %sep% %site_title%',
64 + 'category_title' => '%category_title% %sep% %site_title%',
65 + 'tag_title' => '%tag_title% %sep% %site_title%',
66 + 'author_title' => '%author_name% %sep% %site_title%',
67 + 'search_title' => 'Search Results for "%search_term%" %sep% %site_title%',
68 + 'archive_title' => '%archive_title% %sep% %site_title%',
69 + ];
70 +
71 + /**
39 72 * WordPress filesystem instance
40 73 *
41 74 * @since 1.0.0
42 75 * @var \WP_Filesystem_Base|null
@@ -240,13 +273,424 @@
240 273 * Constructor
241 274 *
242 275 * @since 1.0.0
243 276 */
277 + /**
278 + * The square derivatives wp_site_icon() asks for.
279 + *
280 + * Core generates these only through its own Site Icon crop flow, so an
281 + * image chosen as a ThinkRank favicon straight from the media library has
282 + * none of them and every sizes="" declaration is a near miss (#571).
283 + *
284 + * @since 2.3.1
285 + * @var int[]
286 + */
287 + public const ICON_SIZES = [32, 180, 192, 270];
288 +
289 + /**
290 + * Transient holding resolved icon URLs, keyed by configured URL and size.
291 + *
292 + * The site-icon filter runs in wp_head on every FRONT-END request, and
293 + * resolving a URL to its attachment costs an uncached postmeta query. The
294 + * mapping only changes when the icon setting does, so it is cached here and
295 + * dropped on save.
296 + *
297 + * @since 2.3.1
298 + * @var string
299 + */
300 + public const ICON_URL_TRANSIENT = 'thinkrank_site_icon_urls';
301 +
302 + /**
303 + * Marker for the one-time derivative backfill on existing installs.
304 + *
305 + * @since 2.3.1
306 + * @var string
307 + */
308 + public const ICON_BACKFILL_OPTION = 'thinkrank_site_icon_sizes_backfilled';
309 +
310 + /**
311 + * Whether the icon-derivative listener has been registered this request.
312 + *
313 + * Static because `thinkrank_seo_settings_saved` is a global hook — one
314 + * listener serves every instance, and this class is constructed on the
315 + * front end as well as in admin.
316 + *
317 + * @since 2.3.1
318 + * @var bool
319 + */
320 + private static bool $icon_sizes_listener_registered = false;
321 +
322 + /**
323 + * Whether the robots.txt resync listener is registered for this request.
324 + *
325 + * @since 2.14.0
326 + * @var bool
327 + */
328 + private static bool $robots_sync_listener_registered = false;
329 +
330 + /**
331 + * Flag set when a plugin change may have altered the sitemap set.
332 + *
333 + * @since 2.14.0
334 + * @var string
335 + */
336 + public const ROBOTS_RESYNC_OPTION = 'thinkrank_robots_txt_resync_pending';
337 +
244 338 public function __construct() {
245 339 parent::__construct('site_identity');
340 +
341 + if (!self::$icon_sizes_listener_registered) {
342 + self::$icon_sizes_listener_registered = true;
343 + add_action('thinkrank_seo_settings_saved', [$this, 'generate_icon_sizes_on_save'], 10, 2);
344 + // Admin only: resizing is not front-end work, and admin traffic is
345 + // enough to run a one-time backfill promptly.
346 + add_action('admin_init', [self::class, 'maybe_backfill_icon_sizes']);
347 + }
348 +
349 + if (!self::$robots_sync_listener_registered) {
350 + self::$robots_sync_listener_registered = true;
351 +
352 + // A physical robots.txt bypasses PHP entirely, so composing the
353 + // Sitemap block at render time fixes the served output only on
354 + // sites with no file. Activating or deactivating a sitemap
355 + // contributor changes the set, and until #835 nothing rewrote the
356 + // file: the deactivated plugin's sitemap stayed advertised, serving
357 + // HTML to anything that followed it.
358 + add_action('activated_plugin', [self::class, 'flag_robots_txt_resync']);
359 + add_action('deactivated_plugin', [self::class, 'flag_robots_txt_resync']);
360 + add_action('init', [self::class, 'maybe_resync_robots_txt'], 99);
361 + }
246 362 }
247 363
248 364 /**
365 + * Note that the set of sitemap contributors may have changed.
366 + *
367 + * Deliberately unconditional about which plugin: a contributor is anything
368 + * hooking `thinkrank_additional_sitemaps`, which is resolved at runtime and
369 + * cannot be inspected for a plugin that is on its way out.
370 + *
371 + * The rewrite is not done here. `deactivated_plugin` fires inside the
372 + * request that deactivated it, while that plugin's filters are still
373 + * attached, so rendering now still sees the sitemap that is going away —
374 + * measured, not assumed: the first version of this fix wrote the
375 + * deactivated plugin's sitemap straight back into the file. The next
376 + * request has the real plugin set loaded, so the work waits for it.
377 + *
378 + * @since 2.14.0
379 + * @return void
380 + */
381 + public static function flag_robots_txt_resync(): void {
382 + if (!file_exists(ABSPATH . 'robots.txt')) {
383 + return;
384 + }
385 +
386 + update_option(self::ROBOTS_RESYNC_OPTION, 1, false);
387 + }
388 +
389 + /**
390 + * Rewrite the physical robots.txt once, on the request after a change.
391 + *
392 + * @since 2.14.0
393 + * @return void
394 + */
395 + public static function maybe_resync_robots_txt(): void {
396 + if (!get_option(self::ROBOTS_RESYNC_OPTION)) {
397 + return;
398 + }
399 +
400 + // Cleared first, so a render that fatals cannot retry on every request
401 + // for the rest of the site's life.
402 + delete_option(self::ROBOTS_RESYNC_OPTION);
403 +
404 + if (!file_exists(ABSPATH . 'robots.txt')) {
405 + return;
406 + }
407 +
408 + (new self())->sync_robots_txt_file();
409 + }
410 +
411 + /**
412 + * Save settings, then refresh what a new canonical scheme invalidates.
413 + *
414 + * The static sitemap files are written with the scheme in force when they
415 + * were built, and nothing else rebuilds them until a post or term changes.
416 + * So a change of scheme left every `<loc>` on the old one while canonical
417 + * and og:url had already moved (#736). Every writer (the settings route,
418 + * the robots route, the MCP abilities, an import) lands here.
419 + *
420 + * @since 2.7.0
421 + *
422 + * @param string $context_type Context type.
423 + * @param int|null $context_id Context ID.
424 + * @param array $settings Settings to save.
425 + * @return bool
426 + */
427 + public function save_settings(string $context_type, ?int $context_id, array $settings): bool {
428 + if (!self::touches_canonical_scheme($context_type, $context_id, $settings)) {
429 + return parent::save_settings($context_type, $context_id, $settings);
430 + }
431 +
432 + $before = Url_Scheme::preference();
433 + $saved = parent::save_settings($context_type, $context_id, $settings);
434 +
435 + if ($saved) {
436 + $this->on_canonical_scheme_saved($before);
437 + }
438 +
439 + return $saved;
440 + }
441 +
442 + /**
443 + * Whether a save can change the site-wide canonical scheme.
444 + *
445 + * @since 2.7.0
446 + *
447 + * @param string $context_type Context type.
448 + * @param int|null $context_id Context ID.
449 + * @param array $settings Settings being saved.
450 + * @return bool
451 + */
452 + public static function touches_canonical_scheme(string $context_type, ?int $context_id, array $settings): bool {
453 + return 'site' === sanitize_key($context_type)
454 + && empty($context_id)
455 + && array_key_exists('canonical_scheme', $settings);
456 + }
457 +
458 + /**
459 + * Rebuild the static sitemaps when the effective scheme changed.
460 + *
461 + * Compares the effective preference, filter included, so a site whose
462 + * scheme is pinned by `thinkrank_canonical_scheme` does not rebuild on a
463 + * stored value that changes nothing it publishes.
464 + *
465 + * @since 2.7.0
466 + *
467 + * @param string $before Effective scheme before the save.
468 + * @return void
469 + */
470 + protected function on_canonical_scheme_saved(string $before): void {
471 + // The preference is cached for the request; the save just changed it.
472 + Url_Scheme::reset();
473 +
474 + if (Url_Scheme::preference() === $before) {
475 + return;
476 + }
477 +
478 + $this->schedule_sitemap_rebuild();
479 + }
480 +
481 + /**
482 + * Queue a settings-driven sitemap rebuild.
483 + *
484 + * Debounced and run after the response, like any other settings change
485 + * that alters what the sitemap publishes.
486 + *
487 + * @since 2.7.0
488 + * @return void
489 + */
490 + protected function schedule_sitemap_rebuild(): void {
491 + (new Sitemap_Generator(false))->schedule_regeneration();
492 + }
493 +
494 + /**
495 + * Build the icon derivatives for a newly chosen favicon.
496 + *
497 + * Runs on save, which is the only moment the choice changes and the only
498 + * place image work belongs — resolving a size on the front end must stay a
499 + * lookup. Failure is silent by design: a missing derivative degrades to the
500 + * next best file, so a site whose host cannot resize still renders an icon.
501 + *
502 + * @since 2.3.1
503 + *
504 + * @param string $manager_type Settings category that was saved.
505 + * @param array $settings The settings that were written.
506 + * @return void
507 + */
508 + public function generate_icon_sizes_on_save(string $manager_type, array $settings): void {
509 + if ('site_identity' !== $manager_type) {
510 + return;
511 + }
512 +
513 + // The choice, or the derivatives behind it, may have just changed.
514 + delete_transient(self::ICON_URL_TRANSIENT);
515 +
516 + foreach (['favicon_url', 'apple_touch_icon_url'] as $key) {
517 + if (empty($settings[$key]) || !is_string($settings[$key])) {
518 + continue;
519 + }
520 +
521 + $attachment_id = self::icon_attachment_id($settings[$key]);
522 +
523 + if ($attachment_id) {
524 + self::ensure_icon_sizes($attachment_id);
525 + }
526 + }
527 +
528 + // Dropped again after the resizes finish. Resizing is not instant, and a
529 + // front-end request arriving mid-generation would otherwise repopulate
530 + // the transient with the pre-derivative URLs and pin them for the full
531 + // TTL — leaving the sizes= declarations untrue until the next save.
532 + delete_transient(self::ICON_URL_TRANSIENT);
533 + }
534 +
535 + /**
536 + * Build the derivatives for a site that configured its icons before this
537 + * existed.
538 + *
539 + * generate_icon_sizes_on_save() only fires on a settings write, so every
540 + * site with an icon already chosen would keep serving whatever
541 + * wp_get_attachment_image_url() could find — in practice the 150x150
542 + * thumbnail behind a sizes="32x32" declaration — until someone happened to
543 + * re-save Site Identity. That is the bug this is meant to fix, so the
544 + * derivatives are built once on upgrade instead of waiting for a save.
545 + *
546 + * Guarded by its own option rather than the plugin version so it runs once
547 + * and stays cheap: the check is a single autoloaded read on requests after
548 + * the first.
549 + *
550 + * @since 2.3.1
551 + *
552 + * @return void
553 + */
554 + public static function maybe_backfill_icon_sizes(): void {
555 + if (get_option(self::ICON_BACKFILL_OPTION)) {
556 + return;
557 + }
558 +
559 + // Written before the work, not after: a host that cannot resize must
560 + // not retry on every admin request forever.
561 + update_option(self::ICON_BACKFILL_OPTION, time(), true);
562 +
563 + $settings = (new self())->get_settings('site');
564 +
565 + if (!is_array($settings)) {
566 + return;
567 + }
568 +
569 + foreach (['favicon_url', 'apple_touch_icon_url'] as $key) {
570 + if (empty($settings[$key]) || !is_string($settings[$key])) {
571 + continue;
572 + }
573 +
574 + $attachment_id = self::icon_attachment_id($settings[$key]);
575 +
576 + if ($attachment_id) {
577 + self::ensure_icon_sizes($attachment_id);
578 + }
579 + }
580 +
581 + delete_transient(self::ICON_URL_TRANSIENT);
582 + }
583 +
584 + /**
585 + * Attachment ID behind a configured icon URL, or 0 when it is not ours.
586 + *
587 + * attachment_url_to_postid() matches _wp_attached_file, which holds the
588 + * ORIGINAL upload path, so the URL of a generated derivative
589 + * (`logo-512x512.png`) returns 0 — and that is exactly what the media
590 + * picker hands back when the user chooses a size. Attachment_Lookup falls
591 + * back to the original behind it; the fallback started here and moved
592 + * there when every other image lookup turned out to need it (#847).
593 + *
594 + * Shared with SEO_Manager's site-icon filter so both sides of the feature
595 + * agree on which attachment a configured URL means.
596 + *
597 + * @since 2.3.1
598 + *
599 + * @param string $url Configured icon URL.
600 + * @return int Attachment ID, or 0.
601 + */
602 + public static function icon_attachment_id(string $url): int {
603 + return Attachment_Lookup::id_from_url($url);
604 + }
605 +
606 + /**
607 + * Which ICON_SIZES derivatives this attachment still needs.
608 + *
609 + * Split out from the generation so the decision can be asserted on its
610 + * own: whether a size is skipped because it already exists or because it
611 + * would upscale is invisible once both answers are "nothing was built".
612 + *
613 + * A source is measured by its SHORTER edge — a 400x40 banner cannot yield
614 + * a true 192x192 — and anything reporting no dimensions at all (SVGs) is
615 + * left alone.
616 + *
617 + * @since 2.3.1
618 + *
619 + * @param array $meta Attachment metadata.
620 + * @return array<string, array{width: int, height: int, crop: bool}> Sizes to build.
621 + */
622 + public static function missing_icon_sizes(array $meta): array {
623 + $source = min((int) ($meta['width'] ?? 0), (int) ($meta['height'] ?? 0));
624 +
625 + if ($source < 1) {
626 + return [];
627 + }
628 +
629 + $wanted = [];
630 + foreach (self::ICON_SIZES as $size) {
631 + // Never upscale: a stretched source behind an accurate sizes=""
632 + // label is worse than the honest near miss it would replace.
633 + if (isset($meta['sizes']["site_icon-{$size}"]) || $size > $source) {
634 + continue;
635 + }
636 +
637 + $wanted["site_icon-{$size}"] = ['width' => $size, 'height' => $size, 'crop' => true];
638 + }
639 +
640 + return $wanted;
641 + }
642 +
643 + /**
644 + * Generate whatever ICON_SIZES derivatives this attachment is missing.
645 + *
646 + * Only the missing ones, and never one larger than the source: upscaling a
647 + * small favicon would put a blurrier file behind an accurate sizes="" label
648 + * than the honest near-miss it replaced.
649 + *
650 + * @since 2.3.1
651 + *
652 + * @param int $attachment_id Attachment to build derivatives for.
653 + * @return string[] Size names generated, empty when there was nothing to do.
654 + */
655 + public static function ensure_icon_sizes(int $attachment_id): array {
656 + $meta = wp_get_attachment_metadata($attachment_id);
657 +
658 + if (!is_array($meta)) {
659 + return [];
660 + }
661 +
662 + $wanted = self::missing_icon_sizes($meta);
663 +
664 + if (empty($wanted)) {
665 + return [];
666 + }
667 +
668 + $file = get_attached_file($attachment_id);
669 +
670 + if (!$file || !file_exists($file)) {
671 + return [];
672 + }
673 +
674 + $editor = wp_get_image_editor($file);
675 +
676 + if (is_wp_error($editor)) {
677 + return [];
678 + }
679 +
680 + $generated = $editor->multi_resize($wanted);
681 +
682 + if (empty($generated)) {
683 + return [];
684 + }
685 +
686 + $meta['sizes'] = array_merge($meta['sizes'] ?? [], $generated);
687 + wp_update_attachment_metadata($attachment_id, $meta);
688 +
689 + return array_keys($generated);
690 + }
691 +
692 + /**
249 693 * Initialize WordPress filesystem
250 694 *
251 695 * @since 1.0.0
252 696 * @return bool True if filesystem is initialized, false otherwise
@@ -442,8 +886,28 @@
442 886 $body = ($custom !== '' && !$fully_blocked)
443 887 ? $this->strip_robots_header($custom)
444 888 : trim($this->generate_robots_txt()['content']);
445 889
890 + // The per-agent AI directives are machine-owned, so they are composed
891 + // here rather than stored: the textarea holds the user's body, with
892 + // the fenced block stripped out of every read and re-applied on every
893 + // render. A site-wide block already disallows everyone, so adding the
894 + // per-agent group there would be noise restating the same refusal.
895 + // Composed here rather than read from storage, for the same reason as
896 + // the AI block below: the set of sitemaps an install publishes is a
897 + // runtime fact. `robots_txt_content` is a snapshot of it taken at the
898 + // last save, and nothing invalidated that snapshot, so deactivating a
899 + // sitemap provider left its URL advertised and serving HTML (#835).
900 + // Composing it on every render means the advertisement agrees with what
901 + // the install publishes, in both directions, with no cache to expire.
902 + if (!$fully_blocked) {
903 + $body = $this->apply_sitemap_block($body);
904 + }
905 +
906 + if (!$fully_blocked) {
907 + $body = $this->apply_ai_crawler_block($body, $settings);
908 + }
909 +
446 910 if ($body === '') {
447 911 return '';
448 912 }
449 913
@@ -450,8 +914,87 @@
450 914 return $this->robots_txt_header() . $body . "\n";
451 915 }
452 916
453 917 /**
918 + * Replace the generated Sitemap block with the one this install publishes.
919 + *
920 + * @since 2.14.0
921 + * @param string $body Robots.txt body, without the header.
922 + * @return string
923 + */
924 + private function apply_sitemap_block(string $body): string {
925 + $stripped = $this->strip_generated_sitemap_block($body);
926 + $urls = $this->get_sitemap_urls_for_robots();
927 +
928 + if (empty($urls)) {
929 + return $stripped;
930 + }
931 +
932 + $block = '';
933 + foreach ($urls as $url) {
934 + $block .= 'Sitemap: ' . $url . "\n";
935 + }
936 +
937 + if ('' === trim($stripped)) {
938 + return trim($block);
939 + }
940 +
941 + // The grammar build_robots_txt_content() writes: one blank line before
942 + // the block, none inside it. A blank line terminates a record in the
943 + // robots.txt grammar, so a line between every directive is invalid.
944 + return rtrim($stripped) . "\n\n" . trim($block);
945 + }
946 +
947 + /**
948 + * Remove the plugin-written Sitemap block from a stored body.
949 + *
950 + * Only the trailing run of `Sitemap:` lines is removed, which is the exact
951 + * shape `build_robots_txt_content()` writes: a blank line, then nothing but
952 + * `Sitemap:` lines to the end of the body. A `Sitemap:` line anywhere else
953 + * was typed by the site owner and is left exactly where they put it, which
954 + * is why this cannot simply strip every matching line.
955 + *
956 + * @since 2.14.0
957 + * @param string $body Robots.txt body.
958 + * @return string
959 + */
960 + private function strip_generated_sitemap_block(string $body): string {
961 + $lines = preg_split('/\R/', $body);
962 +
963 + if (!is_array($lines)) {
964 + return $body;
965 + }
966 +
967 + $cut = count($lines);
968 +
969 + // Walk back over the trailing block: sitemap lines, and the blank lines
970 + // that separate or pad it. Anything else ends the block.
971 + for ($i = count($lines) - 1; $i >= 0; $i--) {
972 + $line = trim($lines[$i]);
973 +
974 + if ('' === $line) {
975 + $cut = $i;
976 + continue;
977 + }
978 +
979 + if (0 === stripos($line, 'sitemap:')) {
980 + $cut = $i;
981 + continue;
982 + }
983 +
984 + break;
985 + }
986 +
987 + if ($cut >= count($lines)) {
988 + return $body;
989 + }
990 +
991 + // Nothing but sitemap lines in the whole body means there is no owner
992 + // content to keep.
993 + return rtrim(implode("\n", array_slice($lines, 0, $cut)));
994 + }
995 +
996 + /**
454 997 * Resolve the robots.txt actually served to crawlers, with its origin.
455 998 *
456 999 * Lets an API/MCP consumer see the effective output without crawling the
457 1000 * URL. Mirrors serving precedence: a physical robots.txt in the web root is
@@ -507,27 +1050,53 @@
507 1050 * ever sees — the conflict this exists to surface.
508 1051 *
509 1052 * @since 1.31.0
510 1053 *
511 - * @return array{content: string, source: string, is_default: bool, in_sync: bool, url: string}
1054 + * @return array{content: string, source: string, is_default: bool, in_sync: bool, out_of_sync_reason: string, url: string}
512 1055 * The served content and its origin, whether it still reflects the
513 - * saved settings, and the public URL it is served from.
1056 + * body the editor is showing, why it does not when it does not
1057 + * ('file_drift' or 'crawl_blocked'), and the public URL it is
1058 + * served from.
514 1059 */
515 1060 public function get_robots_txt_delivery(): array {
516 1061 $effective = $this->get_effective_robots_txt();
1062 + $settings = $this->get_settings('site');
517 1063
518 1064 // Compare bodies, not raw strings: the auto-generated header carries a
519 1065 // regeneration timestamp that always differs and means nothing here.
520 - $served = $this->strip_robots_header($effective['content']);
521 - $expected = $this->strip_robots_header($this->render_robots_txt());
1066 + // The AI crawler block is composed at render time on both sides, so it
1067 + // is identical by construction and comparing it would only ever report
1068 + // a false drift the admin cannot act on.
1069 + $served = $this->strip_ai_crawler_block($this->strip_robots_header($effective['content']));
522 1070
1071 + // Measure against the body the editor is displaying — get_served_robots_body()
1072 + // — not against render_robots_txt(). Two things made the old comparison
1073 + // report "in sync" while the screen showed rules no crawler receives:
1074 + // a physical file was compared to a freshly rendered body rather than
1075 + // to the stored override the textarea shows, and a site-wide crawl
1076 + // block makes render_robots_txt() return the generated "Disallow: /"
1077 + // on both sides of the comparison, so it always matched.
1078 + $expected = $this->get_served_robots_body();
1079 +
1080 + // Management off: WordPress serves its own default and the editor is not
1081 + // claiming anything is live, so there is nothing to be out of sync with.
1082 + $managed = !empty($settings['robots_txt_enabled']);
1083 + $in_sync = !$managed || $served === $expected;
1084 +
1085 + $reason = '';
1086 + if (!$in_sync) {
1087 + // A crawl block is a deliberate override, not a stale file, and the
1088 + // admin needs to be told which of the two they are looking at.
1089 + $blocked = empty($settings['allow_search_engines'] ?? true) || !get_option('blog_public');
1090 + $reason = $blocked ? 'crawl_blocked' : 'file_drift';
1091 + }
1092 +
523 1093 return [
524 1094 'content' => $effective['content'],
525 1095 'source' => $effective['source'],
526 1096 'is_default' => $effective['is_default'],
527 - // Only a physical file can drift. Every other source is rendered
528 - // from the settings on demand, so it is in sync by construction.
529 - 'in_sync' => $effective['source'] !== 'file' || $served === $expected,
1097 + 'in_sync' => $in_sync,
1098 + 'out_of_sync_reason' => $reason,
530 1099 'url' => home_url('/robots.txt'),
531 1100 ];
532 1101 }
533 1102
@@ -876,17 +1445,19 @@
876 1445 $home_text = $settings['breadcrumb_home_text'] ?? 'Home';
877 1446 if (empty($home_text)) {
878 1447 $optimization['warnings'][] = 'Empty home text reduces accessibility for screen readers';
879 1448 $optimization['score'] -= 15;
880 - } elseif (strlen($home_text) > 20) {
881 - $optimization['suggestions'][] = 'Keep home text concise (current: ' . strlen($home_text) . ' chars)';
1449 + } elseif (mb_strlen($home_text) > 20) {
1450 + // mb_strlen: this number is shown to the user as "chars" (#687).
1451 + $optimization['suggestions'][] = 'Keep home text concise (current: ' . mb_strlen($home_text) . ' chars)';
882 1452 $optimization['score'] -= 5;
883 1453 }
884 1454
885 1455 // Check prefix usage
886 1456 $prefix = $settings['breadcrumb_prefix'] ?? '';
887 - if (!empty($prefix) && strlen($prefix) > 50) {
888 - $optimization['suggestions'][] = 'Breadcrumb prefix is quite long (' . strlen($prefix) . ' chars) - consider shortening';
1457 + if (!empty($prefix) && mb_strlen($prefix) > 50) {
1458 + // mb_strlen: this number is shown to the user as "chars" (#687).
1459 + $optimization['suggestions'][] = 'Breadcrumb prefix is quite long (' . mb_strlen($prefix) . ' chars) - consider shortening';
889 1460 $optimization['score'] -= 5;
890 1461 }
891 1462
892 1463 // Current page display
@@ -1017,13 +1588,16 @@
1017 1588 }
1018 1589
1019 1590 // Additional logo analysis for local images
1020 1591 if (!empty($logo_url) && filter_var($logo_url, FILTER_VALIDATE_URL)) {
1021 - $attachment_id = attachment_url_to_postid($logo_url);
1592 + $attachment_id = Attachment_Lookup::id_from_url($logo_url);
1022 1593 if ($attachment_id) {
1023 1594 $image_meta = wp_get_attachment_metadata($attachment_id);
1024 - $width = isset($image_meta['width']) ? (int) $image_meta['width'] : 0;
1025 - $height = isset($image_meta['height']) ? (int) $image_meta['height'] : 0;
1595 + // The configured file's own size — a logo picked at a generated
1596 + // size is not as large as the upload behind it.
1597 + $logo_file = Attachment_Lookup::describe($attachment_id, $logo_url);
1598 + $width = $logo_file['width'];
1599 + $height = $logo_file['height'];
1026 1600
1027 1601 // SVG logos store 0x0 metadata — no dimension/ratio analysis
1028 1602 // is possible (and dividing by 0 is fatal).
1029 1603 if ($image_meta && $width > 0 && $height > 0) {
@@ -1126,11 +1700,12 @@
1126 1700 $optimization['score'] -= 15;
1127 1701 }
1128 1702 }
1129 1703
1130 - // Business type validation
1131 - if (empty($settings['business_type'])) {
1132 - $optimization['suggestions'][] = 'Select a specific business type for better schema markup';
1704 + // Business type validation (shared rule, one message — #622).
1705 + $business_type = $this->business_type_status($settings);
1706 + if ('suggestion' === $business_type['status']) {
1707 + $optimization['suggestions'][] = $business_type['message'];
1133 1708 $optimization['score'] -= 5;
1134 1709 }
1135 1710
1136 1711 // Email validation
@@ -1660,24 +2235,17 @@
1660 2235 'icon' => '✗'
1661 2236 ];
1662 2237 }
1663 2238
1664 - // Business Type validation
1665 - if (!empty($settings['business_type']) && $settings['business_type'] !== 'LocalBusiness') {
1666 - $field_details[] = [
1667 - 'field' => 'business_type',
1668 - 'label' => 'Business type is selected for proper schema markup.',
1669 - 'status' => 'valid',
1670 - 'icon' => '✓'
1671 - ];
1672 - } else {
1673 - $field_details[] = [
1674 - 'field' => 'business_type',
1675 - 'label' => 'Specific business type selection recommended for better schema markup.',
1676 - 'status' => 'suggestion',
1677 - 'icon' => '⚠'
1678 - ];
1679 - }
2239 + // Business Type validation — see business_type_status() for why there
2240 + // is exactly one rule here now (#622).
2241 + $business_type = $this->business_type_status($settings);
2242 + $field_details[] = [
2243 + 'field' => 'business_type',
2244 + 'label' => $business_type['message'],
2245 + 'status' => $business_type['status'],
2246 + 'icon' => 'valid' === $business_type['status'] ? '✓' : '⚠',
2247 + ];
1680 2248
1681 2249 // Address validation (NAP consistency)
1682 2250 $address_fields = ['business_address', 'business_city', 'business_state', 'business_country'];
1683 2251 $address_complete = true;
@@ -2296,12 +2864,15 @@
2296 2864 'warnings' => [],
2297 2865 'suggestions' => []
2298 2866 ];
2299 2867
2300 - // Validate business name (required for local SEO)
2868 + // Business name is what makes the LocalBusiness schema useful, but it
2869 + // cannot be a blocking error: the toggle is what reveals the business
2870 + // fields, so requiring the name up front makes enabling Local SEO
2871 + // impossible. The frontend already skips the output while the name is
2872 + // empty (see Seo_Manager::output_local_seo_meta_tags()).
2301 2873 if (empty($settings['business_name'])) {
2302 - $validation['errors'][] = 'Business name is required when local SEO is enabled';
2303 - $validation['valid'] = false;
2874 + $validation['warnings'][] = 'Business name is missing - required before local business schema is output';
2304 2875 } elseif (strlen($settings['business_name']) > 100) {
2305 2876 $validation['warnings'][] = 'Business name is very long, consider shortening for better display';
2306 2877 }
2307 2878
@@ -2358,11 +2929,15 @@
2358 2929 } else {
2359 2930 $validation['suggestions'][] = 'Add business hours to improve local search visibility';
2360 2931 }
2361 2932
2362 - // Validate business type
2363 - if (empty($settings['business_type'])) {
2364 - $validation['suggestions'][] = 'Select a specific business type for better schema markup';
2933 + // Business type, through the shared rule (#622). This is the only place
2934 + // it is reported on the generic path: validate_settings() with no tab
2935 + // context attaches basic-info field details, not business-info ones, so
2936 + // without this the setting would go unreported there entirely.
2937 + $business_type = $this->business_type_status($settings);
2938 + if ('suggestion' === $business_type['status']) {
2939 + $validation['suggestions'][] = $business_type['message'];
2365 2940 }
2366 2941
2367 2942 return $validation;
2368 2943 }
@@ -2442,8 +3017,201 @@
2442 3017 return $output;
2443 3018 }
2444 3019
2445 3020 /**
3021 + * Keys the Site Identity screens store beyond the 16 defaults.
3022 + *
3023 + * Title formats, breadcrumb configuration, the hero fields, the business
3024 + * block and the wizard's identity fields are all real settings written by
3025 + * this manager, none of which get_default_settings() names — it seeds only
3026 + * the values a fresh install needs. Gating on defaults alone would stop
3027 + * every one of them saving (#452).
3028 + *
3029 + * @since 2.0.1
3030 + *
3031 + * @return string[]
3032 + */
3033 + /**
3034 + * The stored alternate name(s), shaped for schema output.
3035 + *
3036 + * schema.org and Google both allow `alternateName` to carry one value or
3037 + * several, and the store already round-trips either shape, so this accepts
3038 + * both and normalises: null when there is nothing to publish, a bare string
3039 + * for one name, a list for more. Emitting a one-element array would be
3040 + * valid but noisier than it needs to be.
3041 + *
3042 + * Shared because both WebSite producers need it and must agree — a property
3043 + * added to one and not the other is how #688 happened.
3044 + *
3045 + * @since 2.7.0
3046 + *
3047 + * @param mixed $value Stored alternate_name value.
3048 + * @return string|string[]|null
3049 + */
3050 + public static function alternate_name_for_schema($value) {
3051 + $names = [];
3052 +
3053 + foreach ((array) $value as $name) {
3054 + if (!is_scalar($name)) {
3055 + continue;
3056 + }
3057 +
3058 + $name = trim((string) $name);
3059 +
3060 + if ('' !== $name && !in_array($name, $names, true)) {
3061 + $names[] = $name;
3062 + }
3063 + }
3064 +
3065 + if (empty($names)) {
3066 + return null;
3067 + }
3068 +
3069 + return 1 === count($names) ? $names[0] : $names;
3070 + }
3071 +
3072 + protected function additional_setting_keys(): array {
3073 + return [
3074 + // Title formats, one per context.
3075 + 'homepage_title', 'post_title', 'page_title', 'category_title',
3076 + 'tag_title', 'author_title', 'search_title', 'archive_title',
3077 + // The blog-index homepage's meta description (#897).
3078 + 'homepage_description',
3079 + // Breadcrumbs.
3080 + 'breadcrumb_prefix', 'show_current_page', 'breadcrumb_use_seo_title',
3081 + // Identity, as written by the setup wizard and the importers.
3082 + 'alternate_name', 'identity_type', 'represents',
3083 + 'default_meta_description', 'default_social_image',
3084 + 'social_media_accounts',
3085 + // Schema toggles that live on this screen.
3086 + 'organization_schema', 'knowledge_graph',
3087 + // Robots rules composed by the Robots.txt panel.
3088 + 'custom_robots_rules',
3089 + // Per-agent AI crawler allow/block map (#657).
3090 + 'ai_crawler_rules',
3091 + // Hero section.
3092 + 'hero_title', 'hero_subtitle', 'hero_cta_text', 'hero_cta_url',
3093 + 'hero_background_image',
3094 + // Local SEO / business details.
3095 + 'local_seo_enabled', 'business_type', 'business_name',
3096 + 'business_address', 'business_city', 'business_state',
3097 + 'business_postal_code', 'business_country', 'business_phone',
3098 + 'business_email', 'business_latitude', 'business_longitude',
3099 + 'business_price_range', 'business_hours',
3100 + ];
3101 + }
3102 +
3103 + /**
3104 + * Sanitize settings, normalising the AI crawler rule map.
3105 + *
3106 + * The generic array sanitizer keeps the shape but says nothing about the
3107 + * values: a payload could store `ai_crawler_rules[gptbot] = "maybe"`, or a
3108 + * slug no crawler answers to, and both would round-trip through every
3109 + * later response. Normalising here rather than in the REST handler puts it
3110 + * on the one path every writer shares — the settings route, the robots
3111 + * route and the MCP abilities all land in save_settings() (#657).
3112 + *
3113 + * @since 2.5.0
3114 + *
3115 + * @param array $settings Settings to sanitize.
3116 + * @param string $context_type Context type.
3117 + * @return array Sanitized settings.
3118 + */
3119 + protected function sanitize_settings(array $settings, string $context_type = 'site'): array {
3120 + $sanitized = parent::sanitize_settings($settings, $context_type);
3121 +
3122 + if (array_key_exists('ai_crawler_rules', $sanitized)) {
3123 + $sanitized['ai_crawler_rules'] = AI_Crawlers::normalize_rules($sanitized['ai_crawler_rules']);
3124 + }
3125 +
3126 + // Same reasoning one key up, for the scheme override (#638). Anything
3127 + // that is not one of the three modes means "follow WordPress", and is
3128 + // stored as that rather than kept verbatim — otherwise get-site-identity
3129 + // -settings would report a scheme the site does not actually publish.
3130 + if (array_key_exists('canonical_scheme', $sanitized)) {
3131 + $sanitized['canonical_scheme'] = in_array($sanitized['canonical_scheme'], Url_Scheme::MODES, true)
3132 + ? $sanitized['canonical_scheme']
3133 + : Url_Scheme::AUTOMATIC;
3134 + }
3135 +
3136 + // Same reasoning again for the business type. It goes straight into
3137 + // LocalBusiness schema, so a type that is not in the schema.org
3138 + // vocabulary is invalid structured data — and storing it verbatim would
3139 + // have get-site-identity-settings report a type the site cannot
3140 + // actually publish. An empty value keeps meaning "not set"; anything
3141 + // else unrecognised falls back to the general-purpose root (#623).
3142 + if (array_key_exists('business_type', $sanitized)) {
3143 + $type = (string) $sanitized['business_type'];
3144 +
3145 + if ('' !== $type && !\ThinkRank\Config\Local_Business_Types_Config::is_valid($type)) {
3146 + $type = \ThinkRank\Config\Local_Business_Types_Config::ROOT;
3147 + }
3148 +
3149 + $sanitized['business_type'] = $type;
3150 + }
3151 +
3152 + return $sanitized;
3153 + }
3154 +
3155 + /**
3156 + * schema.org's general-purpose LocalBusiness type.
3157 + *
3158 + * The default, the first option in the control, and a valid answer in its
3159 + * own right — which is the whole point of #622.
3160 + *
3161 + * @since 2.10.0
3162 + * @var string
3163 + */
3164 + private const GENERAL_BUSINESS_TYPE = 'LocalBusiness';
3165 +
3166 + /**
3167 + * The one rule for whether a business type needs the user's attention.
3168 + *
3169 + * There were three, with two wordings and two different conditions. Two
3170 + * fired when the value was empty; the third fired when it WAS
3171 + * `LocalBusiness` — which is the default, the first option in the control
3172 + * and a perfectly valid schema.org type. So the warning appeared out of the
3173 + * box for every site, could not be cleared without choosing a type that
3174 + * might be inaccurate, and on an empty value it appeared three times in two
3175 + * different phrasings, which is why it was reported as showing twice (#622).
3176 + *
3177 + * The rule now: a type is expected, and any type in the vocabulary is a
3178 + * correct answer. Only an unset value is worth prompting about.
3179 + * `LocalBusiness` is the general-purpose answer and is accepted as one —
3180 + * with a note that a more specific type sharpens the schema, phrased as the
3181 + * guidance it is rather than as a fault the user has to clear.
3182 + *
3183 + * @since 2.10.0
3184 + *
3185 + * @param array $settings Site identity settings.
3186 + * @return array{status:string,message:string} `valid` or `suggestion`.
3187 + */
3188 + private function business_type_status(array $settings): array {
3189 + $type = trim((string) ($settings['business_type'] ?? ''));
3190 +
3191 + if ('' === $type) {
3192 + return [
3193 + 'status' => 'suggestion',
3194 + 'message' => __('Select a business type so your local schema describes the right kind of business.', 'thinkrank'),
3195 + ];
3196 + }
3197 +
3198 + // The literal rather than a constant from the expanded type list (#623):
3199 + // that lands on its own branch, and this fix must not wait on it.
3200 + if (self::GENERAL_BUSINESS_TYPE === $type) {
3201 + return [
3202 + 'status' => 'valid',
3203 + 'message' => __('Business type is set to Local Business. A more specific type sharpens your schema, if one fits.', 'thinkrank'),
3204 + ];
3205 + }
3206 +
3207 + return [
3208 + 'status' => 'valid',
3209 + 'message' => __('Business type is selected for proper schema markup.', 'thinkrank'),
3210 + ];
3211 + }
3212 +
3213 + /**
2446 3214 * Get default settings for a context type (implements interface)
2447 3215 *
2448 3216 * @since 1.0.0
2449 3217 *
@@ -2463,9 +3231,32 @@
2463 3231 'breadcrumb_home_text' => 'Home',
2464 3232 'breadcrumb_separator' => '>',
2465 3233 'robots_txt_enabled' => true,
2466 3234 'allow_search_engines' => true,
3235 + // Answer 404 when a content selector in the URL resolved to
3236 + // nothing (#634). On by default, unlike the other new settings
3237 + // here: it changes no URL a visitor or a correct crawler uses, only
3238 + // ones where WordPress resolved nothing and served the blog listing
3239 + // at 200 anyway.
3240 + 'query_protection' => true,
3241 +
3242 + // Feed controls (#635). All three off, so an upgrade changes
3243 + // nothing about what an existing site already sends its
3244 + // subscribers; a brand-new install is seeded with the signature and
3245 + // the noindex on, in Activator::seed_feed_defaults().
3246 + 'feed_excerpt_only' => false,
3247 + 'feed_source_link' => false,
3248 + 'feed_noindex' => false,
3249 +
3250 + // The scheme self-referential URLs go out with (#638). 'automatic'
3251 + // means substitute nothing and follow WordPress, which is what
3252 + // every site did before the setting existed.
3253 + 'canonical_scheme' => Url_Scheme::AUTOMATIC,
2467 3254 'robots_txt_content' => '',
3255 + // Empty map = every AI crawler allowed. Defaults must stay
3256 + // permissive so an upgrade never starts blocking a crawler a site
3257 + // was happily serving (#657).
3258 + 'ai_crawler_rules' => [],
2468 3259 'logo_url' => '',
2469 3260 'favicon_url' => '',
2470 3261 'apple_touch_icon_url' => ''
2471 3262 ];
@@ -2584,10 +3375,13 @@
2584 3375 */
2585 3376 private function prepare_title_placeholders(array $data, string $context, array $settings): array {
2586 3377 $placeholders = [
2587 3378 '%title%' => $data['title'] ?? '',
2588 - '%sitename%' => $settings['site_name'] ?? get_bloginfo('name'),
2589 - '%tagline%' => $settings['tagline'] ?? get_bloginfo('description'),
3379 + // `?:` rather than `??`: these are persisted as '' rather than left
3380 + // unset, and '' is not null, so the null-coalesce never reached the
3381 + // WordPress fallback (#398).
3382 + '%sitename%' => ($settings['site_name'] ?? '') ?: get_bloginfo('name'),
3383 + '%tagline%' => ($settings['tagline'] ?? '') ?: get_bloginfo('description'),
2590 3384 '%separator%' => '', // Will be replaced with actual separator
2591 3385 '%category%' => '',
2592 3386 '%author%' => '',
2593 3387 '%date%' => '',
@@ -2671,16 +3465,17 @@
2671 3465 // Remove extra whitespace
2672 3466 $title = preg_replace('/\s+/', ' ', $title);
2673 3467 $title = trim($title);
2674 3468
2675 - // Ensure title is not too long (60 characters max for SEO)
2676 - if (strlen($title) > 60) {
2677 - // Try to truncate at word boundary
2678 - $title = wp_trim_words($title, 8, '...');
2679 - if (strlen($title) > 60) {
2680 - $title = substr($title, 0, 57) . '...';
2681 - }
2682 - }
3469 + // Ensure title is not too long (60 characters max for SEO).
3470 + // All three units here were wrong for non-Latin text: strlen() counts
3471 + // BYTES so the gate fired at 20 Thai characters, wp_trim_words() counts
3472 + // CHARACTERS on th/ja/zh_* so `8` cut the title to 8 of them, and
3473 + // substr() cuts bytes so it split a character mid-sequence (#687).
3474 + $title = \ThinkRank\Core\Seo_Text::trim_to_length(
3475 + $title,
3476 + \ThinkRank\Core\Seo_Text::TITLE_MAX_LENGTH
3477 + );
2683 3478
2684 3479 // Ensure title is not empty
2685 3480 if (empty($title)) {
2686 3481 $title = get_bloginfo('name');
@@ -3070,8 +3865,29 @@
3070 3865 * @since 1.0.0
3071 3866 * @return array Array of sitemap URLs
3072 3867 */
3073 3868 private function get_sitemap_urls_for_robots(): array {
3869 + // One wrapper over every return path below, including the #104 extras.
3870 + // The Sitemap: line is the only absolute URL of ours in robots.txt and
3871 + // the one a crawler follows to find everything else, so it has to carry
3872 + // the site's scheme preference (#638). Applied here rather than where
3873 + // the body is assembled, because that path also renders a robots.txt a
3874 + // site owner typed themselves, and their text is not ours to rewrite.
3875 + return array_map(
3876 + static function (string $url): string {
3877 + return Url_Scheme::apply($url);
3878 + },
3879 + $this->collect_sitemap_urls_for_robots()
3880 + );
3881 + }
3882 +
3883 + /**
3884 + * The sitemap URLs robots.txt advertises, before the scheme preference.
3885 + *
3886 + * @since 1.0.0
3887 + * @return array Array of sitemap URLs
3888 + */
3889 + private function collect_sitemap_urls_for_robots(): array {
3074 3890 try {
3075 3891 // Get sitemap settings
3076 3892 $sitemap_generator = new \ThinkRank\SEO\Sitemap_Generator();
3077 3893 $sitemap_settings = $sitemap_generator->get_settings('site');
@@ -3083,29 +3899,88 @@
3083 3899
3084 3900 $sitemap_urls = [];
3085 3901 $site_url = home_url();
3086 3902
3087 - // Extract enabled sitemap URLs
3903 + // Extract enabled sitemap URLs. When the index is enabled it is the
3904 + // only entry worth advertising: every child sitemap is already
3905 + // listed inside it, so naming them again in robots.txt is pure
3906 + // redundancy and drifts out of date as soon as a post type is added.
3907 + $index_url = '';
3088 3908 if (!empty($sitemap_settings['sitemap_urls']) && is_array($sitemap_settings['sitemap_urls'])) {
3089 3909 foreach ($sitemap_settings['sitemap_urls'] as $sitemap) {
3090 - if (!empty($sitemap['enabled']) && !empty($sitemap['url'])) {
3091 - $sitemap_urls[] = $site_url . $sitemap['url'];
3910 + if (empty($sitemap['enabled']) || empty($sitemap['url'])) {
3911 + continue;
3092 3912 }
3913 +
3914 + if (($sitemap['type'] ?? '') === 'index') {
3915 + $index_url = $site_url . $sitemap['url'];
3916 + continue;
3917 + }
3918 +
3919 + $sitemap_urls[] = $site_url . $sitemap['url'];
3093 3920 }
3094 3921 }
3095 3922
3923 + if ($index_url !== '') {
3924 + // The index covers the children and, on a segmented install,
3925 + // the local business sitemap too.
3926 + //
3927 + // It does not cover a sitemap contributed through
3928 + // `thinkrank_additional_sitemaps`: the index is built by this
3929 + // plugin's own generator and never lists them. Returning the
3930 + // index alone therefore left a contributed sitemap with no
3931 + // discovery path at all — absent from robots.txt and absent
3932 + // from the index — so Pro's News sitemap was unreachable on any
3933 + // install with the index enabled, which is the default (#835).
3934 + $contributed = [];
3935 +
3936 + foreach (\ThinkRank\SEO\Sitemap_Generator::additional_sitemaps() as $path) {
3937 + $url = home_url($path);
3938 +
3939 + if ($url !== $index_url && !in_array($url, $contributed, true)) {
3940 + $contributed[] = $url;
3941 + }
3942 + }
3943 +
3944 + return array_merge([$index_url], $contributed);
3945 + }
3946 +
3096 3947 // Fallback to default if no URLs found
3097 3948 if (empty($sitemap_urls)) {
3098 3949 $sitemap_urls[] = home_url('/sitemap.xml');
3099 3950 }
3100 3951
3101 - // Advertise the local business sitemap when it exists. In segmented
3102 - // mode it is already listed inside the sitemap index; in single mode
3103 - // there is no index, so robots.txt is its discovery path.
3104 - if (file_exists(ABSPATH . 'local-sitemap.xml')) {
3105 - $local_url = home_url('/local-sitemap.xml');
3106 - if (!in_array($local_url, $sitemap_urls, true)) {
3107 - $sitemap_urls[] = $local_url;
3952 + // No index on this install, so anything not already listed above has
3953 + // no other discovery path — advertise it directly. The local
3954 + // business sitemap and the sitemaps other plugins register both land
3955 + // here for the same reason, so they go through one list (#104).
3956 + $extra = [];
3957 +
3958 + // Not a file test. Under dynamic delivery the local sitemap is
3959 + // served from PHP and no file is ever written, so file_exists()
3960 + // silently dropped a sitemap the site really does publish (#752).
3961 + // On static sites the file is still what proves it, so both count.
3962 + $local_sitemap_published = file_exists(ABSPATH . 'local-sitemap.xml');
3963 +
3964 + if (!$local_sitemap_published && class_exists('ThinkRank\\SEO\\Sitemap_Generator')) {
3965 + $generator = new \ThinkRank\SEO\Sitemap_Generator(false);
3966 +
3967 + $local_sitemap_published = 'dynamic' === $generator->resolve_delivery_mode()
3968 + && $generator->publishes_local_sitemap();
3969 + }
3970 +
3971 + if ($local_sitemap_published) {
3972 + $extra[] = '/local-sitemap.xml';
3973 + }
3974 +
3975 + foreach (\ThinkRank\SEO\Sitemap_Generator::additional_sitemaps() as $path) {
3976 + $extra[] = $path;
3977 + }
3978 +
3979 + foreach ($extra as $path) {
3980 + $url = home_url($path);
3981 + if (!in_array($url, $sitemap_urls, true)) {
3982 + $sitemap_urls[] = $url;
3108 3983 }
3109 3984 }
3110 3985
3111 3986 return $sitemap_urls;
@@ -3195,8 +4070,9 @@
3195 4070 // timestamp on every update.
3196 4071 $content = '';
3197 4072
3198 4073 $current_user_agent = '';
4074 + $sitemap_started = false;
3199 4075
3200 4076 foreach ($rules as $rule) {
3201 4077 $directive = $rule['directive'] ?? '';
3202 4078 $value = $rule['value'] ?? '';
@@ -3217,9 +4093,17 @@
3217 4093 case 'crawl_delay':
3218 4094 $content .= "Crawl-delay: {$value}\n";
3219 4095 break;
3220 4096 case 'sitemap':
3221 - $content .= "\nSitemap: {$value}\n";
4097 + // One blank line separates the Sitemap block from the
4098 + // preceding group, and none appear inside it. A blank line
4099 + // terminates a record in the robots.txt grammar, so putting
4100 + // one between every directive was invalid formatting.
4101 + if (!$sitemap_started) {
4102 + $content .= "\n";
4103 + $sitemap_started = true;
4104 + }
4105 + $content .= "Sitemap: {$value}\n";
3222 4106 break;
3223 4107 }
3224 4108 }
3225 4109
@@ -3226,8 +4110,171 @@
3226 4110 return ltrim($content, "\n");
3227 4111 }
3228 4112
3229 4113 /**
4114 + * Parse a robots.txt body back into the {directive, value} rule shape.
4115 + *
4116 + * generate_robots_txt() returns `rules` alongside `content`, but callers
4117 + * replace `content` with the body actually being served (a stored override
4118 + * or a physical file). The generated rules then described something the
4119 + * response no longer contained. Re-deriving them from the served body keeps
4120 + * the two halves of the payload describing the same document.
4121 + *
4122 + * @since 2.0.1
4123 + *
4124 + * @param string $content Robots.txt body (header optional).
4125 + * @return array<int, array{directive: string, value: string}> Parsed rules.
4126 + */
4127 + public function parse_robots_txt_rules(string $content): array {
4128 + $map = [
4129 + 'user-agent' => 'user_agent',
4130 + 'disallow' => 'disallow',
4131 + 'allow' => 'allow',
4132 + 'crawl-delay' => 'crawl_delay',
4133 + 'sitemap' => 'sitemap',
4134 + ];
4135 +
4136 + $rules = [];
4137 +
4138 + foreach (preg_split('/\r\n|\r|\n/', $this->strip_robots_header($content)) as $line) {
4139 + $line = trim($line);
4140 +
4141 + // Blank lines separate groups and `#` starts a comment; neither is
4142 + // a rule.
4143 + if ($line === '' || str_starts_with($line, '#')) {
4144 + continue;
4145 + }
4146 +
4147 + $parts = explode(':', $line, 2);
4148 + if (count($parts) !== 2) {
4149 + continue;
4150 + }
4151 +
4152 + $field = strtolower(trim($parts[0]));
4153 + if (!isset($map[$field])) {
4154 + continue;
4155 + }
4156 +
4157 + $rules[] = [
4158 + 'directive' => $map[$field],
4159 + // Sitemap values are absolute URLs and contain the `:` the
4160 + // limited explode above deliberately preserved.
4161 + 'value' => trim($parts[1]),
4162 + ];
4163 + }
4164 +
4165 + return $rules;
4166 + }
4167 +
4168 + /**
4169 + * Opening fence of the machine-owned AI crawler region.
4170 + *
4171 + * @since 2.5.0
4172 + * @var string
4173 + */
4174 + public const AI_BLOCK_BEGIN = '# BEGIN ThinkRank AI crawlers';
4175 +
4176 + /**
4177 + * Closing fence of the machine-owned AI crawler region.
4178 + *
4179 + * @since 2.5.0
4180 + * @var string
4181 + */
4182 + public const AI_BLOCK_END = '# END ThinkRank AI crawlers';
4183 +
4184 + /**
4185 + * Render the fenced AI crawler region for the current settings.
4186 + *
4187 + * One `User-agent:` / `Disallow: /` record per blocked crawler. Allowed
4188 + * crawlers emit nothing at all: `Disallow:` with an empty value is the
4189 + * robots.txt way of saying "allow everything", but writing eighteen such
4190 + * records to say what silence already says would triple the file and
4191 + * invite the reading that an unlisted crawler is therefore refused.
4192 + *
4193 + * @since 2.5.0
4194 + *
4195 + * @param array $settings Site settings.
4196 + * @return string Fenced block, newline-terminated, or '' when nothing is blocked.
4197 + */
4198 + private function build_ai_crawler_block(array $settings): string {
4199 + $blocked = AI_Crawlers::blocked_slugs($settings['ai_crawler_rules'] ?? []);
4200 +
4201 + if (empty($blocked)) {
4202 + return '';
4203 + }
4204 +
4205 + $agents = AI_Crawlers::all();
4206 +
4207 + $lines = [
4208 + self::AI_BLOCK_BEGIN,
4209 + '# Managed by ThinkRank — edits between these lines are overwritten.',
4210 + ];
4211 +
4212 + foreach ($blocked as $slug) {
4213 + $lines[] = '';
4214 + $lines[] = 'User-agent: ' . $agents[$slug]['token'];
4215 + $lines[] = 'Disallow: /';
4216 + }
4217 +
4218 + $lines[] = self::AI_BLOCK_END;
4219 +
4220 + return implode("\n", $lines) . "\n";
4221 + }
4222 +
4223 + /**
4224 + * Remove the fenced AI crawler region from a robots.txt body.
4225 + *
4226 + * Tolerates a missing closing fence rather than leaving the rest of the
4227 + * file swallowed: a truncated write, or someone deleting the END line by
4228 + * hand, would otherwise make every subsequent read drop everything below
4229 + * the opening fence.
4230 + *
4231 + * @since 2.5.0
4232 + *
4233 + * @param string $body Robots.txt body.
4234 + * @return string Body with the region removed.
4235 + */
4236 + public function strip_ai_crawler_block(string $body): string {
4237 + if (false === strpos($body, self::AI_BLOCK_BEGIN)) {
4238 + return $body;
4239 + }
4240 +
4241 + $pattern = '/\R*' . preg_quote(self::AI_BLOCK_BEGIN, '/')
4242 + . '.*?(?:' . preg_quote(self::AI_BLOCK_END, '/') . '|\z)\R*/s';
4243 +
4244 + return trim((string) preg_replace($pattern, "\n\n", $body, 1));
4245 + }
4246 +
4247 + /**
4248 + * Put the current AI crawler region into a robots.txt body.
4249 + *
4250 + * Replaces an existing region in place so the block keeps its position in
4251 + * a hand-ordered file, and appends when there is none. Everything outside
4252 + * the fences is returned untouched — that is the whole point of fencing
4253 + * it, since the body is also a free-text field the user edits.
4254 + *
4255 + * @since 2.5.0
4256 + *
4257 + * @param string $body Robots.txt body (fences optional).
4258 + * @param array $settings Site settings.
4259 + * @return string Body carrying the current region.
4260 + */
4261 + private function apply_ai_crawler_block(string $body, array $settings): string {
4262 + $stripped = $this->strip_ai_crawler_block($body);
4263 + $block = $this->build_ai_crawler_block($settings);
4264 +
4265 + if ('' === $block) {
4266 + return $stripped;
4267 + }
4268 +
4269 + if ('' === trim($stripped)) {
4270 + return trim($block);
4271 + }
4272 +
4273 + return rtrim($stripped) . "\n\n" . trim($block);
4274 + }
4275 +
4276 + /**
3230 4277 * The auto-generated header prepended to the served robots.txt.
3231 4278 *
3232 4279 * Kept separate from the body so it is only ever added at render time with
3233 4280 * a fresh timestamp, never stored or shown in the editable textarea.
@@ -3264,11 +4311,16 @@
3264 4311 */
3265 4312 public function get_served_robots_body(): string {
3266 4313 $settings = $this->get_settings('site');
3267 4314
4315 + // The AI block is stripped from every one of these paths. A physical
4316 + // robots.txt we wrote carries it, and the stored override is whatever
4317 + // the textarea last held — so without this the block round-trips into
4318 + // the editor, gets saved as ordinary body text, and is then appended
4319 + // to a second time on the next render.
3268 4320 $custom = trim((string) ($settings['robots_txt_content'] ?? ''));
3269 4321 if ($custom !== '') {
3270 - return $this->strip_robots_header($custom);
4322 + return $this->strip_ai_crawler_block($this->strip_robots_header($custom));
3271 4323 }
3272 4324
3273 4325 $robots_file = ABSPATH . 'robots.txt';
3274 4326 if (file_exists($robots_file)) {
@@ -3274,13 +4326,13 @@
3274 4326 if (file_exists($robots_file)) {
3275 4327 // phpcs:ignore WordPress.WP.AlternativeFunctions.file_get_contents_file_get_contents, WordPress.PHP.NoSilencedErrors.Discouraged -- an unreadable robots.txt is an expected state answered with an empty string.
3276 4328 $raw = (string) @file_get_contents($robots_file);
3277 4329 if ($raw !== '') {
3278 - return $this->strip_robots_header($raw);
4330 + return $this->strip_ai_crawler_block($this->strip_robots_header($raw));
3279 4331 }
3280 4332 }
3281 4333
3282 - return trim($this->generate_robots_txt()['content']);
4334 + return $this->strip_ai_crawler_block(trim($this->generate_robots_txt()['content']));
3283 4335 }
3284 4336 private function get_site_identity_data(array $settings): array {
3285 4337 return [
3286 4338 'site_name' => $settings['site_name'] ?? get_bloginfo('name'),
@@ -3342,11 +4394,14 @@
3342 4394 $optimization['validation']['valid'] = false;
3343 4395 }
3344 4396
3345 4397 if (!empty($value) && isset($config['max_length'])) {
3346 - if (strlen($value) > $config['max_length']) {
4398 + // The warning says "characters", so measure and cut in characters:
4399 + // strlen()/substr() fired early on non-Latin values and the
4400 + // suggested replacement was cut mid-character (#687).
4401 + if (mb_strlen($value) > $config['max_length']) {
3347 4402 $optimization['validation']['warnings'][] = "{$element} exceeds maximum length of {$config['max_length']} characters";
3348 - $optimization['optimized_value'] = substr($value, 0, $config['max_length']);
4403 + $optimization['optimized_value'] = \ThinkRank\Core\Seo_Text::trim_to_length($value, (int) $config['max_length']);
3349 4404 }
3350 4405 }
3351 4406
3352 4407 // SEO-specific optimizations
@@ -3385,18 +4440,20 @@
3385 4440 return $optimization;
3386 4441 }
3387 4442
3388 4443 // Check if it's a local image
3389 - $attachment_id = attachment_url_to_postid($value);
4444 + $attachment_id = Attachment_Lookup::id_from_url($value);
3390 4445 if ($attachment_id) {
3391 4446 $image_meta = wp_get_attachment_metadata($attachment_id);
3392 4447
3393 4448 if ($image_meta && isset($image_meta['width'], $image_meta['height'])) {
3394 - // Check recommended size
4449 + // Check recommended size, against the configured file itself
4450 + // rather than the upload it may have been generated from.
3395 4451 if (isset($config['recommended_size'])) {
3396 4452 [$rec_width, $rec_height] = explode('x', $config['recommended_size']);
4453 + $image_file = Attachment_Lookup::describe($attachment_id, $value);
3397 4454
3398 - if ((int) $image_meta['width'] !== (int) $rec_width || (int) $image_meta['height'] !== (int) $rec_height) {
4455 + if ($image_file['width'] !== (int) $rec_width || $image_file['height'] !== (int) $rec_height) {
3399 4456 $optimization['suggestions'][] = "Consider using {$config['recommended_size']} size for optimal {$element}";
3400 4457 }
3401 4458 }
3402 4459