PluginProbe
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO / 2.14.2
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO v2.14.2
2.14.2 2.14.1 2.14.0 2.13.0 2.12.0 2.11.0 2.10.0 2.9.0 2.8.0 2.7.0 2.6.0 2.5.0 2.4.0 2.3.0 2.2.0 2.1.1 2.1.0 2.0.2 2.0.1 2.0.0 1.32.0 1.31.0 1.30.0 1.29.0 1.28.0 All 57 releases
← All changes | includes/seo/class-site-identity-manager.php +1261 -72 1.29.0 → 2.14.2 View file →
@@ -15,8 +15,13 @@
15 15 declare(strict_types=1);
16 16
17 17 namespace ThinkRank\SEO;
18 18
19 +// Prevent direct access
20 +if (!defined('ABSPATH')) {
21 + exit;
22 +}
23 +
19 24 // Ensure dependencies are loaded
20 25 if (!class_exists('ThinkRank\\SEO\\Abstract_SEO_Manager')) {
21 26 require_once THINKRANK_PLUGIN_DIR . 'includes/seo/class-abstract-seo-manager.php';
22 27 }
@@ -35,8 +40,36 @@
35 40 */
36 41 class Site_Identity_Manager extends Abstract_SEO_Manager {
37 42
38 43 /**
44 + * The per-context title formats as ThinkRank ships them.
45 + *
46 + * These are not in get_default_settings(): the admin screen seeds them on
47 + * first save, so on a real install they are stored values, indistinguishable
48 + * from a template the user typed. The migration needs to tell those two
49 + * apart — it may overwrite a shipped default with an imported template, and
50 + * must never overwrite a choice the user made — so this is the record of
51 + * what "untouched" looks like.
52 + *
53 + * Keep in step with getDefaultSettings() in
54 + * src/admin/components/essential-seo/SiteIdentityTab.js. SiteIdentityTitleFormatDefaultsTest
55 + * fails when the two drift.
56 + *
57 + * @since 2.8.0
58 + * @var array<string, string>
59 + */
60 + public const TITLE_FORMAT_DEFAULTS = [
61 + 'homepage_title' => '%site_title% %sep% %site_description%',
62 + 'post_title' => '%post_title% %sep% %site_title%',
63 + 'page_title' => '%page_title% %sep% %site_title%',
64 + 'category_title' => '%category_title% %sep% %site_title%',
65 + 'tag_title' => '%tag_title% %sep% %site_title%',
66 + 'author_title' => '%author_name% %sep% %site_title%',
67 + 'search_title' => 'Search Results for "%search_term%" %sep% %site_title%',
68 + 'archive_title' => '%archive_title% %sep% %site_title%',
69 + ];
70 +
71 + /**
39 72 * WordPress filesystem instance
40 73 *
41 74 * @since 1.0.0
42 75 * @var \WP_Filesystem_Base|null
@@ -240,13 +273,494 @@
240 273 * Constructor
241 274 *
242 275 * @since 1.0.0
243 276 */
277 + /**
278 + * The square derivatives wp_site_icon() asks for.
279 + *
280 + * Core generates these only through its own Site Icon crop flow, so an
281 + * image chosen as a ThinkRank favicon straight from the media library has
282 + * none of them and every sizes="" declaration is a near miss (#571).
283 + *
284 + * @since 2.3.1
285 + * @var int[]
286 + */
287 + public const ICON_SIZES = [32, 180, 192, 270];
288 +
289 + /**
290 + * Transient holding resolved icon URLs, keyed by configured URL and size.
291 + *
292 + * The site-icon filter runs in wp_head on every FRONT-END request, and
293 + * resolving a URL to its attachment costs an uncached postmeta query. The
294 + * mapping only changes when the icon setting does, so it is cached here and
295 + * dropped on save.
296 + *
297 + * @since 2.3.1
298 + * @var string
299 + */
300 + public const ICON_URL_TRANSIENT = 'thinkrank_site_icon_urls';
301 +
302 + /**
303 + * Marker for the one-time derivative backfill on existing installs.
304 + *
305 + * @since 2.3.1
306 + * @var string
307 + */
308 + public const ICON_BACKFILL_OPTION = 'thinkrank_site_icon_sizes_backfilled';
309 +
310 + /**
311 + * Whether the icon-derivative listener has been registered this request.
312 + *
313 + * Static because `thinkrank_seo_settings_saved` is a global hook — one
314 + * listener serves every instance, and this class is constructed on the
315 + * front end as well as in admin.
316 + *
317 + * @since 2.3.1
318 + * @var bool
319 + */
320 + private static bool $icon_sizes_listener_registered = false;
321 +
322 + /**
323 + * Whether the robots.txt resync listener is registered for this request.
324 + *
325 + * @since 2.14.0
326 + * @var bool
327 + */
328 + private static bool $robots_sync_listener_registered = false;
329 +
330 + /**
331 + * Flag set when a plugin change may have altered the sitemap set.
332 + *
333 + * @since 2.14.0
334 + * @var string
335 + */
336 + public const ROBOTS_RESYNC_OPTION = 'thinkrank_robots_txt_resync_pending';
337 +
338 + /**
339 + * Flag set when a plugin change may have altered the sitemap index.
340 + *
341 + * Separate from ROBOTS_RESYNC_OPTION because the two files exist
342 + * independently: that flag is only set when a physical robots.txt exists,
343 + * and the sitemap index needs rebuilding whether or not it does.
344 + *
345 + * @since 2.15.0
346 + * @var string
347 + */
348 + public const SITEMAP_RESYNC_OPTION = 'thinkrank_sitemap_contributors_changed';
349 +
244 350 public function __construct() {
245 351 parent::__construct('site_identity');
352 +
353 + if (!self::$icon_sizes_listener_registered) {
354 + self::$icon_sizes_listener_registered = true;
355 + add_action('thinkrank_seo_settings_saved', [$this, 'generate_icon_sizes_on_save'], 10, 2);
356 + // Admin only: resizing is not front-end work, and admin traffic is
357 + // enough to run a one-time backfill promptly.
358 + add_action('admin_init', [self::class, 'maybe_backfill_icon_sizes']);
359 + }
360 +
361 + if (!self::$robots_sync_listener_registered) {
362 + self::$robots_sync_listener_registered = true;
363 +
364 + // A physical robots.txt bypasses PHP entirely, so composing the
365 + // Sitemap block at render time fixes the served output only on
366 + // sites with no file. Activating or deactivating a sitemap
367 + // contributor changes the set, and until #835 nothing rewrote the
368 + // file: the deactivated plugin's sitemap stayed advertised, serving
369 + // HTML to anything that followed it.
370 + add_action('activated_plugin', [self::class, 'flag_robots_txt_resync']);
371 + add_action('deactivated_plugin', [self::class, 'flag_robots_txt_resync']);
372 + add_action('init', [self::class, 'maybe_resync_robots_txt'], 99);
373 +
374 + // The sitemap index is a second static file listing the same
375 + // contributors, with its own rebuild path. #835 / #859 resynced
376 + // robots.txt only, so after Pro was deactivated the index kept
377 + // advertising news-sitemap.xml, which then served the home page
378 + // as HTML (#920).
379 + add_action('activated_plugin', [self::class, 'flag_sitemap_resync']);
380 + add_action('deactivated_plugin', [self::class, 'flag_sitemap_resync']);
381 + add_action('init', [self::class, 'maybe_resync_sitemap'], 99);
382 + }
246 383 }
247 384
248 385 /**
386 + * Note that the set of sitemap contributors may have changed.
387 + *
388 + * Deliberately unconditional about which plugin: a contributor is anything
389 + * hooking `thinkrank_additional_sitemaps`, which is resolved at runtime and
390 + * cannot be inspected for a plugin that is on its way out.
391 + *
392 + * The rewrite is not done here. `deactivated_plugin` fires inside the
393 + * request that deactivated it, while that plugin's filters are still
394 + * attached, so rendering now still sees the sitemap that is going away —
395 + * measured, not assumed: the first version of this fix wrote the
396 + * deactivated plugin's sitemap straight back into the file. The next
397 + * request has the real plugin set loaded, so the work waits for it.
398 + *
399 + * @since 2.14.0
400 + * @return void
401 + */
402 + public static function flag_robots_txt_resync(): void {
403 + if (!file_exists(ABSPATH . 'robots.txt')) {
404 + return;
405 + }
406 +
407 + update_option(self::ROBOTS_RESYNC_OPTION, 1, false);
408 + }
409 +
410 + /**
411 + * Rewrite the physical robots.txt once, on the request after a change.
412 + *
413 + * @since 2.14.0
414 + * @return void
415 + */
416 + public static function maybe_resync_robots_txt(): void {
417 + if (!get_option(self::ROBOTS_RESYNC_OPTION)) {
418 + return;
419 + }
420 +
421 + // Cleared first, so a render that fatals cannot retry on every request
422 + // for the rest of the site's life.
423 + delete_option(self::ROBOTS_RESYNC_OPTION);
424 +
425 + if (!file_exists(ABSPATH . 'robots.txt')) {
426 + return;
427 + }
428 +
429 + (new self())->sync_robots_txt_file();
430 + }
431 +
432 + /**
433 + * Note that the set of sitemap contributors may have changed.
434 + *
435 + * Unconditional, unlike flag_robots_txt_resync(): the sitemap files exist
436 + * whether or not robots.txt does. The rebuild waits for the next request
437 + * for the same reason as the robots.txt one, since `deactivated_plugin`
438 + * still runs with the outgoing plugin's `thinkrank_additional_sitemaps`
439 + * callback attached.
440 + *
441 + * @since 2.15.0
442 + * @return void
443 + */
444 + public static function flag_sitemap_resync(): void {
445 + update_option(self::SITEMAP_RESYNC_OPTION, 1, false);
446 + }
447 +
448 + /**
449 + * Queue a sitemap rebuild once, on the request after a contributor change.
450 + *
451 + * Goes through schedule_regeneration(), the debounced and lock-protected
452 + * path a sitemap settings save uses, so a burst of plugin changes (a bulk
453 + * deactivate, say) still produces one rebuild. That path also drops the
454 + * cached dynamic documents, so sites serving the sitemap from PHP drop the
455 + * entry as well.
456 + *
457 + * @since 2.15.0
458 + * @return void
459 + */
460 + public static function maybe_resync_sitemap(): void {
461 + if (!get_option(self::SITEMAP_RESYNC_OPTION)) {
462 + return;
463 + }
464 +
465 + // Cleared first, so a rebuild that fatals cannot be retried on every
466 + // request for the rest of the site's life.
467 + delete_option(self::SITEMAP_RESYNC_OPTION);
468 +
469 + $generator = new Sitemap_Generator(false);
470 + $settings = $generator->get_settings('site');
471 +
472 + // A disabled sitemap has no files to correct. Enabling it later builds
473 + // from the contributors present at that time.
474 + if (empty($settings['enabled'])) {
475 + return;
476 + }
477 +
478 + $generator->schedule_regeneration();
479 + }
480 +
481 + /**
482 + * Save settings, then refresh what a new canonical scheme invalidates.
483 + *
484 + * The static sitemap files are written with the scheme in force when they
485 + * were built, and nothing else rebuilds them until a post or term changes.
486 + * So a change of scheme left every `<loc>` on the old one while canonical
487 + * and og:url had already moved (#736). Every writer (the settings route,
488 + * the robots route, the MCP abilities, an import) lands here.
489 + *
490 + * @since 2.7.0
491 + *
492 + * @param string $context_type Context type.
493 + * @param int|null $context_id Context ID.
494 + * @param array $settings Settings to save.
495 + * @return bool
496 + */
497 + public function save_settings(string $context_type, ?int $context_id, array $settings): bool {
498 + if (!self::touches_canonical_scheme($context_type, $context_id, $settings)) {
499 + return parent::save_settings($context_type, $context_id, $settings);
500 + }
501 +
502 + $before = Url_Scheme::preference();
503 + $saved = parent::save_settings($context_type, $context_id, $settings);
504 +
505 + if ($saved) {
506 + $this->on_canonical_scheme_saved($before);
507 + }
508 +
509 + return $saved;
510 + }
511 +
512 + /**
513 + * Whether a save can change the site-wide canonical scheme.
514 + *
515 + * @since 2.7.0
516 + *
517 + * @param string $context_type Context type.
518 + * @param int|null $context_id Context ID.
519 + * @param array $settings Settings being saved.
520 + * @return bool
521 + */
522 + public static function touches_canonical_scheme(string $context_type, ?int $context_id, array $settings): bool {
523 + return 'site' === sanitize_key($context_type)
524 + && empty($context_id)
525 + && array_key_exists('canonical_scheme', $settings);
526 + }
527 +
528 + /**
529 + * Rebuild the static sitemaps when the effective scheme changed.
530 + *
531 + * Compares the effective preference, filter included, so a site whose
532 + * scheme is pinned by `thinkrank_canonical_scheme` does not rebuild on a
533 + * stored value that changes nothing it publishes.
534 + *
535 + * @since 2.7.0
536 + *
537 + * @param string $before Effective scheme before the save.
538 + * @return void
539 + */
540 + protected function on_canonical_scheme_saved(string $before): void {
541 + // The preference is cached for the request; the save just changed it.
542 + Url_Scheme::reset();
543 +
544 + if (Url_Scheme::preference() === $before) {
545 + return;
546 + }
547 +
548 + $this->schedule_sitemap_rebuild();
549 + }
550 +
551 + /**
552 + * Queue a settings-driven sitemap rebuild.
553 + *
554 + * Debounced and run after the response, like any other settings change
555 + * that alters what the sitemap publishes.
556 + *
557 + * @since 2.7.0
558 + * @return void
559 + */
560 + protected function schedule_sitemap_rebuild(): void {
561 + (new Sitemap_Generator(false))->schedule_regeneration();
562 + }
563 +
564 + /**
565 + * Build the icon derivatives for a newly chosen favicon.
566 + *
567 + * Runs on save, which is the only moment the choice changes and the only
568 + * place image work belongs — resolving a size on the front end must stay a
569 + * lookup. Failure is silent by design: a missing derivative degrades to the
570 + * next best file, so a site whose host cannot resize still renders an icon.
571 + *
572 + * @since 2.3.1
573 + *
574 + * @param string $manager_type Settings category that was saved.
575 + * @param array $settings The settings that were written.
576 + * @return void
577 + */
578 + public function generate_icon_sizes_on_save(string $manager_type, array $settings): void {
579 + if ('site_identity' !== $manager_type) {
580 + return;
581 + }
582 +
583 + // The choice, or the derivatives behind it, may have just changed.
584 + delete_transient(self::ICON_URL_TRANSIENT);
585 +
586 + foreach (['favicon_url', 'apple_touch_icon_url'] as $key) {
587 + if (empty($settings[$key]) || !is_string($settings[$key])) {
588 + continue;
589 + }
590 +
591 + $attachment_id = self::icon_attachment_id($settings[$key]);
592 +
593 + if ($attachment_id) {
594 + self::ensure_icon_sizes($attachment_id);
595 + }
596 + }
597 +
598 + // Dropped again after the resizes finish. Resizing is not instant, and a
599 + // front-end request arriving mid-generation would otherwise repopulate
600 + // the transient with the pre-derivative URLs and pin them for the full
601 + // TTL — leaving the sizes= declarations untrue until the next save.
602 + delete_transient(self::ICON_URL_TRANSIENT);
603 + }
604 +
605 + /**
606 + * Build the derivatives for a site that configured its icons before this
607 + * existed.
608 + *
609 + * generate_icon_sizes_on_save() only fires on a settings write, so every
610 + * site with an icon already chosen would keep serving whatever
611 + * wp_get_attachment_image_url() could find — in practice the 150x150
612 + * thumbnail behind a sizes="32x32" declaration — until someone happened to
613 + * re-save Site Identity. That is the bug this is meant to fix, so the
614 + * derivatives are built once on upgrade instead of waiting for a save.
615 + *
616 + * Guarded by its own option rather than the plugin version so it runs once
617 + * and stays cheap: the check is a single autoloaded read on requests after
618 + * the first.
619 + *
620 + * @since 2.3.1
621 + *
622 + * @return void
623 + */
624 + public static function maybe_backfill_icon_sizes(): void {
625 + if (get_option(self::ICON_BACKFILL_OPTION)) {
626 + return;
627 + }
628 +
629 + // Written before the work, not after: a host that cannot resize must
630 + // not retry on every admin request forever.
631 + update_option(self::ICON_BACKFILL_OPTION, time(), true);
632 +
633 + $settings = (new self())->get_settings('site');
634 +
635 + if (!is_array($settings)) {
636 + return;
637 + }
638 +
639 + foreach (['favicon_url', 'apple_touch_icon_url'] as $key) {
640 + if (empty($settings[$key]) || !is_string($settings[$key])) {
641 + continue;
642 + }
643 +
644 + $attachment_id = self::icon_attachment_id($settings[$key]);
645 +
646 + if ($attachment_id) {
647 + self::ensure_icon_sizes($attachment_id);
648 + }
649 + }
650 +
651 + delete_transient(self::ICON_URL_TRANSIENT);
652 + }
653 +
654 + /**
655 + * Attachment ID behind a configured icon URL, or 0 when it is not ours.
656 + *
657 + * attachment_url_to_postid() matches _wp_attached_file, which holds the
658 + * ORIGINAL upload path, so the URL of a generated derivative
659 + * (`logo-512x512.png`) returns 0 — and that is exactly what the media
660 + * picker hands back when the user chooses a size. Attachment_Lookup falls
661 + * back to the original behind it; the fallback started here and moved
662 + * there when every other image lookup turned out to need it (#847).
663 + *
664 + * Shared with SEO_Manager's site-icon filter so both sides of the feature
665 + * agree on which attachment a configured URL means.
666 + *
667 + * @since 2.3.1
668 + *
669 + * @param string $url Configured icon URL.
670 + * @return int Attachment ID, or 0.
671 + */
672 + public static function icon_attachment_id(string $url): int {
673 + return Attachment_Lookup::id_from_url($url);
674 + }
675 +
676 + /**
677 + * Which ICON_SIZES derivatives this attachment still needs.
678 + *
679 + * Split out from the generation so the decision can be asserted on its
680 + * own: whether a size is skipped because it already exists or because it
681 + * would upscale is invisible once both answers are "nothing was built".
682 + *
683 + * A source is measured by its SHORTER edge — a 400x40 banner cannot yield
684 + * a true 192x192 — and anything reporting no dimensions at all (SVGs) is
685 + * left alone.
686 + *
687 + * @since 2.3.1
688 + *
689 + * @param array $meta Attachment metadata.
690 + * @return array<string, array{width: int, height: int, crop: bool}> Sizes to build.
691 + */
692 + public static function missing_icon_sizes(array $meta): array {
693 + $source = min((int) ($meta['width'] ?? 0), (int) ($meta['height'] ?? 0));
694 +
695 + if ($source < 1) {
696 + return [];
697 + }
698 +
699 + $wanted = [];
700 + foreach (self::ICON_SIZES as $size) {
701 + // Never upscale: a stretched source behind an accurate sizes=""
702 + // label is worse than the honest near miss it would replace.
703 + if (isset($meta['sizes']["site_icon-{$size}"]) || $size > $source) {
704 + continue;
705 + }
706 +
707 + $wanted["site_icon-{$size}"] = ['width' => $size, 'height' => $size, 'crop' => true];
708 + }
709 +
710 + return $wanted;
711 + }
712 +
713 + /**
714 + * Generate whatever ICON_SIZES derivatives this attachment is missing.
715 + *
716 + * Only the missing ones, and never one larger than the source: upscaling a
717 + * small favicon would put a blurrier file behind an accurate sizes="" label
718 + * than the honest near-miss it replaced.
719 + *
720 + * @since 2.3.1
721 + *
722 + * @param int $attachment_id Attachment to build derivatives for.
723 + * @return string[] Size names generated, empty when there was nothing to do.
724 + */
725 + public static function ensure_icon_sizes(int $attachment_id): array {
726 + $meta = wp_get_attachment_metadata($attachment_id);
727 +
728 + if (!is_array($meta)) {
729 + return [];
730 + }
731 +
732 + $wanted = self::missing_icon_sizes($meta);
733 +
734 + if (empty($wanted)) {
735 + return [];
736 + }
737 +
738 + $file = get_attached_file($attachment_id);
739 +
740 + if (!$file || !file_exists($file)) {
741 + return [];
742 + }
743 +
744 + $editor = wp_get_image_editor($file);
745 +
746 + if (is_wp_error($editor)) {
747 + return [];
748 + }
749 +
750 + $generated = $editor->multi_resize($wanted);
751 +
752 + if (empty($generated)) {
753 + return [];
754 + }
755 +
756 + $meta['sizes'] = array_merge($meta['sizes'] ?? [], $generated);
757 + wp_update_attachment_metadata($attachment_id, $meta);
758 +
759 + return array_keys($generated);
760 + }
761 +
762 + /**
249 763 * Initialize WordPress filesystem
250 764 *
251 765 * @since 1.0.0
252 766 * @return bool True if filesystem is initialized, false otherwise
@@ -442,8 +956,28 @@
442 956 $body = ($custom !== '' && !$fully_blocked)
443 957 ? $this->strip_robots_header($custom)
444 958 : trim($this->generate_robots_txt()['content']);
445 959
960 + // The per-agent AI directives are machine-owned, so they are composed
961 + // here rather than stored: the textarea holds the user's body, with
962 + // the fenced block stripped out of every read and re-applied on every
963 + // render. A site-wide block already disallows everyone, so adding the
964 + // per-agent group there would be noise restating the same refusal.
965 + // Composed here rather than read from storage, for the same reason as
966 + // the AI block below: the set of sitemaps an install publishes is a
967 + // runtime fact. `robots_txt_content` is a snapshot of it taken at the
968 + // last save, and nothing invalidated that snapshot, so deactivating a
969 + // sitemap provider left its URL advertised and serving HTML (#835).
970 + // Composing it on every render means the advertisement agrees with what
971 + // the install publishes, in both directions, with no cache to expire.
972 + if (!$fully_blocked) {
973 + $body = $this->apply_sitemap_block($body);
974 + }
975 +
976 + if (!$fully_blocked) {
977 + $body = $this->apply_ai_crawler_block($body, $settings);
978 + }
979 +
446 980 if ($body === '') {
447 981 return '';
448 982 }
449 983
@@ -450,8 +984,87 @@
450 984 return $this->robots_txt_header() . $body . "\n";
451 985 }
452 986
453 987 /**
988 + * Replace the generated Sitemap block with the one this install publishes.
989 + *
990 + * @since 2.14.0
991 + * @param string $body Robots.txt body, without the header.
992 + * @return string
993 + */
994 + private function apply_sitemap_block(string $body): string {
995 + $stripped = $this->strip_generated_sitemap_block($body);
996 + $urls = $this->get_sitemap_urls_for_robots();
997 +
998 + if (empty($urls)) {
999 + return $stripped;
1000 + }
1001 +
1002 + $block = '';
1003 + foreach ($urls as $url) {
1004 + $block .= 'Sitemap: ' . $url . "\n";
1005 + }
1006 +
1007 + if ('' === trim($stripped)) {
1008 + return trim($block);
1009 + }
1010 +
1011 + // The grammar build_robots_txt_content() writes: one blank line before
1012 + // the block, none inside it. A blank line terminates a record in the
1013 + // robots.txt grammar, so a line between every directive is invalid.
1014 + return rtrim($stripped) . "\n\n" . trim($block);
1015 + }
1016 +
1017 + /**
1018 + * Remove the plugin-written Sitemap block from a stored body.
1019 + *
1020 + * Only the trailing run of `Sitemap:` lines is removed, which is the exact
1021 + * shape `build_robots_txt_content()` writes: a blank line, then nothing but
1022 + * `Sitemap:` lines to the end of the body. A `Sitemap:` line anywhere else
1023 + * was typed by the site owner and is left exactly where they put it, which
1024 + * is why this cannot simply strip every matching line.
1025 + *
1026 + * @since 2.14.0
1027 + * @param string $body Robots.txt body.
1028 + * @return string
1029 + */
1030 + private function strip_generated_sitemap_block(string $body): string {
1031 + $lines = preg_split('/\R/', $body);
1032 +
1033 + if (!is_array($lines)) {
1034 + return $body;
1035 + }
1036 +
1037 + $cut = count($lines);
1038 +
1039 + // Walk back over the trailing block: sitemap lines, and the blank lines
1040 + // that separate or pad it. Anything else ends the block.
1041 + for ($i = count($lines) - 1; $i >= 0; $i--) {
1042 + $line = trim($lines[$i]);
1043 +
1044 + if ('' === $line) {
1045 + $cut = $i;
1046 + continue;
1047 + }
1048 +
1049 + if (0 === stripos($line, 'sitemap:')) {
1050 + $cut = $i;
1051 + continue;
1052 + }
1053 +
1054 + break;
1055 + }
1056 +
1057 + if ($cut >= count($lines)) {
1058 + return $body;
1059 + }
1060 +
1061 + // Nothing but sitemap lines in the whole body means there is no owner
1062 + // content to keep.
1063 + return rtrim(implode("\n", array_slice($lines, 0, $cut)));
1064 + }
1065 +
1066 + /**
454 1067 * Resolve the robots.txt actually served to crawlers, with its origin.
455 1068 *
456 1069 * Lets an API/MCP consumer see the effective output without crawling the
457 1070 * URL. Mirrors serving precedence: a physical robots.txt in the web root is
@@ -497,8 +1110,68 @@
497 1110 ];
498 1111 }
499 1112
500 1113 /**
1114 + * Describe how /robots.txt is actually delivered, and whether that still
1115 + * matches the saved settings.
1116 + *
1117 + * The admin screen edits settings, but a physical robots.txt in the web root
1118 + * is served directly by the web server and bypasses the `robots_txt` filter
1119 + * entirely. When those two drift, the editor is showing content no crawler
1120 + * ever sees — the conflict this exists to surface.
1121 + *
1122 + * @since 1.31.0
1123 + *
1124 + * @return array{content: string, source: string, is_default: bool, in_sync: bool, out_of_sync_reason: string, url: string}
1125 + * The served content and its origin, whether it still reflects the
1126 + * body the editor is showing, why it does not when it does not
1127 + * ('file_drift' or 'crawl_blocked'), and the public URL it is
1128 + * served from.
1129 + */
1130 + public function get_robots_txt_delivery(): array {
1131 + $effective = $this->get_effective_robots_txt();
1132 + $settings = $this->get_settings('site');
1133 +
1134 + // Compare bodies, not raw strings: the auto-generated header carries a
1135 + // regeneration timestamp that always differs and means nothing here.
1136 + // The AI crawler block is composed at render time on both sides, so it
1137 + // is identical by construction and comparing it would only ever report
1138 + // a false drift the admin cannot act on.
1139 + $served = $this->strip_ai_crawler_block($this->strip_robots_header($effective['content']));
1140 +
1141 + // Measure against the body the editor is displaying — get_served_robots_body()
1142 + // — not against render_robots_txt(). Two things made the old comparison
1143 + // report "in sync" while the screen showed rules no crawler receives:
1144 + // a physical file was compared to a freshly rendered body rather than
1145 + // to the stored override the textarea shows, and a site-wide crawl
1146 + // block makes render_robots_txt() return the generated "Disallow: /"
1147 + // on both sides of the comparison, so it always matched.
1148 + $expected = $this->get_served_robots_body();
1149 +
1150 + // Management off: WordPress serves its own default and the editor is not
1151 + // claiming anything is live, so there is nothing to be out of sync with.
1152 + $managed = !empty($settings['robots_txt_enabled']);
1153 + $in_sync = !$managed || $served === $expected;
1154 +
1155 + $reason = '';
1156 + if (!$in_sync) {
1157 + // A crawl block is a deliberate override, not a stale file, and the
1158 + // admin needs to be told which of the two they are looking at.
1159 + $blocked = empty($settings['allow_search_engines'] ?? true) || !get_option('blog_public');
1160 + $reason = $blocked ? 'crawl_blocked' : 'file_drift';
1161 + }
1162 +
1163 + return [
1164 + 'content' => $effective['content'],
1165 + 'source' => $effective['source'],
1166 + 'is_default' => $effective['is_default'],
1167 + 'in_sync' => $in_sync,
1168 + 'out_of_sync_reason' => $reason,
1169 + 'url' => home_url('/robots.txt'),
1170 + ];
1171 + }
1172 +
1173 + /**
501 1174 * Keep the physical robots.txt file in step with the saved settings.
502 1175 *
503 1176 * When management is enabled the physical file is the source of truth the
504 1177 * web server serves, so this makes sure it exists and matches the effective
@@ -842,17 +1515,19 @@
842 1515 $home_text = $settings['breadcrumb_home_text'] ?? 'Home';
843 1516 if (empty($home_text)) {
844 1517 $optimization['warnings'][] = 'Empty home text reduces accessibility for screen readers';
845 1518 $optimization['score'] -= 15;
846 - } elseif (strlen($home_text) > 20) {
847 - $optimization['suggestions'][] = 'Keep home text concise (current: ' . strlen($home_text) . ' chars)';
1519 + } elseif (mb_strlen($home_text) > 20) {
1520 + // mb_strlen: this number is shown to the user as "chars" (#687).
1521 + $optimization['suggestions'][] = 'Keep home text concise (current: ' . mb_strlen($home_text) . ' chars)';
848 1522 $optimization['score'] -= 5;
849 1523 }
850 1524
851 1525 // Check prefix usage
852 1526 $prefix = $settings['breadcrumb_prefix'] ?? '';
853 - if (!empty($prefix) && strlen($prefix) > 50) {
854 - $optimization['suggestions'][] = 'Breadcrumb prefix is quite long (' . strlen($prefix) . ' chars) - consider shortening';
1527 + if (!empty($prefix) && mb_strlen($prefix) > 50) {
1528 + // mb_strlen: this number is shown to the user as "chars" (#687).
1529 + $optimization['suggestions'][] = 'Breadcrumb prefix is quite long (' . mb_strlen($prefix) . ' chars) - consider shortening';
855 1530 $optimization['score'] -= 5;
856 1531 }
857 1532
858 1533 // Current page display
@@ -961,9 +1636,9 @@
961 1636 $optimization['suggestions'][] = 'Add a site logo for better branding and professional appearance';
962 1637 $optimization['score'] -= 20;
963 1638 } else {
964 1639 // Validate logo URL and dimensions
965 - if (!filter_var($logo_url, FILTER_VALIDATE_URL)) {
1640 + if (!\ThinkRank\Core\Url_Validator::is_http_url($logo_url)) {
966 1641 $optimization['warnings'][] = 'Logo URL format is invalid';
967 1642 $optimization['score'] -= 15;
968 1643 }
969 1644 }
@@ -982,14 +1657,17 @@
982 1657 $optimization['score'] -= 10;
983 1658 }
984 1659
985 1660 // Additional logo analysis for local images
986 - if (!empty($logo_url) && filter_var($logo_url, FILTER_VALIDATE_URL)) {
987 - $attachment_id = attachment_url_to_postid($logo_url);
1661 + if (!empty($logo_url) && \ThinkRank\Core\Url_Validator::is_http_url($logo_url)) {
1662 + $attachment_id = Attachment_Lookup::id_from_url($logo_url);
988 1663 if ($attachment_id) {
989 1664 $image_meta = wp_get_attachment_metadata($attachment_id);
990 - $width = isset($image_meta['width']) ? (int) $image_meta['width'] : 0;
991 - $height = isset($image_meta['height']) ? (int) $image_meta['height'] : 0;
1665 + // The configured file's own size — a logo picked at a generated
1666 + // size is not as large as the upload behind it.
1667 + $logo_file = Attachment_Lookup::describe($attachment_id, $logo_url);
1668 + $width = $logo_file['width'];
1669 + $height = $logo_file['height'];
992 1670
993 1671 // SVG logos store 0x0 metadata — no dimension/ratio analysis
994 1672 // is possible (and dividing by 0 is fatal).
995 1673 if ($image_meta && $width > 0 && $height > 0) {
@@ -1092,11 +1770,12 @@
1092 1770 $optimization['score'] -= 15;
1093 1771 }
1094 1772 }
1095 1773
1096 - // Business type validation
1097 - if (empty($settings['business_type'])) {
1098 - $optimization['suggestions'][] = 'Select a specific business type for better schema markup';
1774 + // Business type validation (shared rule, one message — #622).
1775 + $business_type = $this->business_type_status($settings);
1776 + if ('suggestion' === $business_type['status']) {
1777 + $optimization['suggestions'][] = $business_type['message'];
1099 1778 $optimization['score'] -= 5;
1100 1779 }
1101 1780
1102 1781 // Email validation
@@ -1466,12 +2145,37 @@
1466 2145 $validation['suggestions'][] = 'Consider making site description longer (120-160 characters)';
1467 2146 }
1468 2147 }
1469 2148
1470 - // Validate logo URL
1471 - if (isset($settings['logo_url']) && !empty($settings['logo_url'])) {
1472 - if (!filter_var($settings['logo_url'], FILTER_VALIDATE_URL)) {
1473 - $validation['errors'][] = 'Logo URL must be a valid URL';
2149 + // Image and link URLs. These are written into src and href
2150 + // attributes (the schema logo, the admin previews, the hero section),
2151 + // so only web URLs are accepted. is_valid() takes any scheme, and
2152 + // "javascript://%0Aalert(1)" passed it and was stored as
2153 + // "javascript://alert(1)" once sanitize_text_field() dropped the %0A.
2154 + // The logo and default social image must be absolute: they are
2155 + // published in schema and Open Graph, which require it. The others
2156 + // may also be a path on this site.
2157 + $url_fields = [
2158 + 'logo_url' => ['Logo URL', false],
2159 + 'default_social_image' => ['Default social image URL', false],
2160 + 'favicon_url' => ['Favicon URL', true],
2161 + 'apple_touch_icon_url' => ['Apple touch icon URL', true],
2162 + 'hero_background_image' => ['Hero background image URL', true],
2163 + 'hero_cta_url' => ['Call-to-action URL', true],
2164 + ];
2165 + foreach ($url_fields as $key => [$label, $allow_path]) {
2166 + if (!isset($settings[$key]) || '' === $settings[$key] || null === $settings[$key]) {
2167 + continue;
2168 + }
2169 +
2170 + $ok = $allow_path
2171 + ? \ThinkRank\Core\Url_Validator::is_http_url_or_path($settings[$key])
2172 + : \ThinkRank\Core\Url_Validator::is_http_url($settings[$key]);
2173 +
2174 + if (!$ok) {
2175 + $validation['errors'][] = $allow_path
2176 + ? sprintf('%s must be an http or https URL, or a path starting with /', $label)
2177 + : sprintf('%s must be an http or https URL', $label);
1474 2178 $validation['valid'] = false;
1475 2179 }
1476 2180 }
1477 2181
@@ -1626,24 +2330,17 @@
1626 2330 'icon' => '✗'
1627 2331 ];
1628 2332 }
1629 2333
1630 - // Business Type validation
1631 - if (!empty($settings['business_type']) && $settings['business_type'] !== 'LocalBusiness') {
1632 - $field_details[] = [
1633 - 'field' => 'business_type',
1634 - 'label' => 'Business type is selected for proper schema markup.',
1635 - 'status' => 'valid',
1636 - 'icon' => '✓'
1637 - ];
1638 - } else {
1639 - $field_details[] = [
1640 - 'field' => 'business_type',
1641 - 'label' => 'Specific business type selection recommended for better schema markup.',
1642 - 'status' => 'suggestion',
1643 - 'icon' => '⚠'
1644 - ];
1645 - }
2334 + // Business Type validation — see business_type_status() for why there
2335 + // is exactly one rule here now (#622).
2336 + $business_type = $this->business_type_status($settings);
2337 + $field_details[] = [
2338 + 'field' => 'business_type',
2339 + 'label' => $business_type['message'],
2340 + 'status' => $business_type['status'],
2341 + 'icon' => 'valid' === $business_type['status'] ? '✓' : '⚠',
2342 + ];
1646 2343
1647 2344 // Address validation (NAP consistency)
1648 2345 $address_fields = ['business_address', 'business_city', 'business_state', 'business_country'];
1649 2346 $address_complete = true;
@@ -1916,9 +2613,9 @@
1916 2613 }
1917 2614
1918 2615 // CTA URL validation
1919 2616 if (!empty($settings['hero_cta_url'])) {
1920 - if (filter_var($settings['hero_cta_url'], FILTER_VALIDATE_URL) || strpos($settings['hero_cta_url'], '/') === 0) {
2617 + if (\ThinkRank\Core\Url_Validator::is_http_url_or_path($settings['hero_cta_url'])) {
1921 2618 $field_details[] = [
1922 2619 'field' => 'hero_cta_url',
1923 2620 'label' => 'Call-to-action URL is properly configured.',
1924 2621 'status' => 'valid',
@@ -1959,9 +2656,9 @@
1959 2656 }
1960 2657
1961 2658 // Site Logo validation (from Site Assets section)
1962 2659 if (!empty($settings['logo_url'])) {
1963 - if (filter_var($settings['logo_url'], FILTER_VALIDATE_URL)) {
2660 + if (\ThinkRank\Core\Url_Validator::is_http_url($settings['logo_url'])) {
1964 2661 $field_details[] = [
1965 2662 'field' => 'logo_url',
1966 2663 'label' => 'Site logo is properly configured.',
1967 2664 'status' => 'valid',
@@ -2262,12 +2959,15 @@
2262 2959 'warnings' => [],
2263 2960 'suggestions' => []
2264 2961 ];
2265 2962
2266 - // Validate business name (required for local SEO)
2963 + // Business name is what makes the LocalBusiness schema useful, but it
2964 + // cannot be a blocking error: the toggle is what reveals the business
2965 + // fields, so requiring the name up front makes enabling Local SEO
2966 + // impossible. The frontend already skips the output while the name is
2967 + // empty (see Seo_Manager::output_local_seo_meta_tags()).
2267 2968 if (empty($settings['business_name'])) {
2268 - $validation['errors'][] = 'Business name is required when local SEO is enabled';
2269 - $validation['valid'] = false;
2969 + $validation['warnings'][] = 'Business name is missing - required before local business schema is output';
2270 2970 } elseif (strlen($settings['business_name']) > 100) {
2271 2971 $validation['warnings'][] = 'Business name is very long, consider shortening for better display';
2272 2972 }
2273 2973
@@ -2324,11 +3024,15 @@
2324 3024 } else {
2325 3025 $validation['suggestions'][] = 'Add business hours to improve local search visibility';
2326 3026 }
2327 3027
2328 - // Validate business type
2329 - if (empty($settings['business_type'])) {
2330 - $validation['suggestions'][] = 'Select a specific business type for better schema markup';
3028 + // Business type, through the shared rule (#622). This is the only place
3029 + // it is reported on the generic path: validate_settings() with no tab
3030 + // context attaches basic-info field details, not business-info ones, so
3031 + // without this the setting would go unreported there entirely.
3032 + $business_type = $this->business_type_status($settings);
3033 + if ('suggestion' === $business_type['status']) {
3034 + $validation['suggestions'][] = $business_type['message'];
2331 3035 }
2332 3036
2333 3037 return $validation;
2334 3038 }
@@ -2408,8 +3112,201 @@
2408 3112 return $output;
2409 3113 }
2410 3114
2411 3115 /**
3116 + * Keys the Site Identity screens store beyond the 16 defaults.
3117 + *
3118 + * Title formats, breadcrumb configuration, the hero fields, the business
3119 + * block and the wizard's identity fields are all real settings written by
3120 + * this manager, none of which get_default_settings() names — it seeds only
3121 + * the values a fresh install needs. Gating on defaults alone would stop
3122 + * every one of them saving (#452).
3123 + *
3124 + * @since 2.0.1
3125 + *
3126 + * @return string[]
3127 + */
3128 + /**
3129 + * The stored alternate name(s), shaped for schema output.
3130 + *
3131 + * schema.org and Google both allow `alternateName` to carry one value or
3132 + * several, and the store already round-trips either shape, so this accepts
3133 + * both and normalises: null when there is nothing to publish, a bare string
3134 + * for one name, a list for more. Emitting a one-element array would be
3135 + * valid but noisier than it needs to be.
3136 + *
3137 + * Shared because both WebSite producers need it and must agree — a property
3138 + * added to one and not the other is how #688 happened.
3139 + *
3140 + * @since 2.7.0
3141 + *
3142 + * @param mixed $value Stored alternate_name value.
3143 + * @return string|string[]|null
3144 + */
3145 + public static function alternate_name_for_schema($value) {
3146 + $names = [];
3147 +
3148 + foreach ((array) $value as $name) {
3149 + if (!is_scalar($name)) {
3150 + continue;
3151 + }
3152 +
3153 + $name = trim((string) $name);
3154 +
3155 + if ('' !== $name && !in_array($name, $names, true)) {
3156 + $names[] = $name;
3157 + }
3158 + }
3159 +
3160 + if (empty($names)) {
3161 + return null;
3162 + }
3163 +
3164 + return 1 === count($names) ? $names[0] : $names;
3165 + }
3166 +
3167 + protected function additional_setting_keys(): array {
3168 + return [
3169 + // Title formats, one per context.
3170 + 'homepage_title', 'post_title', 'page_title', 'category_title',
3171 + 'tag_title', 'author_title', 'search_title', 'archive_title',
3172 + // The blog-index homepage's meta description (#897).
3173 + 'homepage_description',
3174 + // Breadcrumbs.
3175 + 'breadcrumb_prefix', 'show_current_page', 'breadcrumb_use_seo_title',
3176 + // Identity, as written by the setup wizard and the importers.
3177 + 'alternate_name', 'identity_type', 'represents',
3178 + 'default_meta_description', 'default_social_image',
3179 + 'social_media_accounts',
3180 + // Schema toggles that live on this screen.
3181 + 'organization_schema', 'knowledge_graph',
3182 + // Robots rules composed by the Robots.txt panel.
3183 + 'custom_robots_rules',
3184 + // Per-agent AI crawler allow/block map (#657).
3185 + 'ai_crawler_rules',
3186 + // Hero section.
3187 + 'hero_title', 'hero_subtitle', 'hero_cta_text', 'hero_cta_url',
3188 + 'hero_background_image',
3189 + // Local SEO / business details.
3190 + 'local_seo_enabled', 'business_type', 'business_name',
3191 + 'business_address', 'business_city', 'business_state',
3192 + 'business_postal_code', 'business_country', 'business_phone',
3193 + 'business_email', 'business_latitude', 'business_longitude',
3194 + 'business_price_range', 'business_hours',
3195 + ];
3196 + }
3197 +
3198 + /**
3199 + * Sanitize settings, normalising the AI crawler rule map.
3200 + *
3201 + * The generic array sanitizer keeps the shape but says nothing about the
3202 + * values: a payload could store `ai_crawler_rules[gptbot] = "maybe"`, or a
3203 + * slug no crawler answers to, and both would round-trip through every
3204 + * later response. Normalising here rather than in the REST handler puts it
3205 + * on the one path every writer shares — the settings route, the robots
3206 + * route and the MCP abilities all land in save_settings() (#657).
3207 + *
3208 + * @since 2.5.0
3209 + *
3210 + * @param array $settings Settings to sanitize.
3211 + * @param string $context_type Context type.
3212 + * @return array Sanitized settings.
3213 + */
3214 + protected function sanitize_settings(array $settings, string $context_type = 'site'): array {
3215 + $sanitized = parent::sanitize_settings($settings, $context_type);
3216 +
3217 + if (array_key_exists('ai_crawler_rules', $sanitized)) {
3218 + $sanitized['ai_crawler_rules'] = AI_Crawlers::normalize_rules($sanitized['ai_crawler_rules']);
3219 + }
3220 +
3221 + // Same reasoning one key up, for the scheme override (#638). Anything
3222 + // that is not one of the three modes means "follow WordPress", and is
3223 + // stored as that rather than kept verbatim — otherwise get-site-identity
3224 + // -settings would report a scheme the site does not actually publish.
3225 + if (array_key_exists('canonical_scheme', $sanitized)) {
3226 + $sanitized['canonical_scheme'] = in_array($sanitized['canonical_scheme'], Url_Scheme::MODES, true)
3227 + ? $sanitized['canonical_scheme']
3228 + : Url_Scheme::AUTOMATIC;
3229 + }
3230 +
3231 + // Same reasoning again for the business type. It goes straight into
3232 + // LocalBusiness schema, so a type that is not in the schema.org
3233 + // vocabulary is invalid structured data — and storing it verbatim would
3234 + // have get-site-identity-settings report a type the site cannot
3235 + // actually publish. An empty value keeps meaning "not set"; anything
3236 + // else unrecognised falls back to the general-purpose root (#623).
3237 + if (array_key_exists('business_type', $sanitized)) {
3238 + $type = (string) $sanitized['business_type'];
3239 +
3240 + if ('' !== $type && !\ThinkRank\Config\Local_Business_Types_Config::is_valid($type)) {
3241 + $type = \ThinkRank\Config\Local_Business_Types_Config::ROOT;
3242 + }
3243 +
3244 + $sanitized['business_type'] = $type;
3245 + }
3246 +
3247 + return $sanitized;
3248 + }
3249 +
3250 + /**
3251 + * schema.org's general-purpose LocalBusiness type.
3252 + *
3253 + * The default, the first option in the control, and a valid answer in its
3254 + * own right — which is the whole point of #622.
3255 + *
3256 + * @since 2.10.0
3257 + * @var string
3258 + */
3259 + private const GENERAL_BUSINESS_TYPE = 'LocalBusiness';
3260 +
3261 + /**
3262 + * The one rule for whether a business type needs the user's attention.
3263 + *
3264 + * There were three, with two wordings and two different conditions. Two
3265 + * fired when the value was empty; the third fired when it WAS
3266 + * `LocalBusiness` — which is the default, the first option in the control
3267 + * and a perfectly valid schema.org type. So the warning appeared out of the
3268 + * box for every site, could not be cleared without choosing a type that
3269 + * might be inaccurate, and on an empty value it appeared three times in two
3270 + * different phrasings, which is why it was reported as showing twice (#622).
3271 + *
3272 + * The rule now: a type is expected, and any type in the vocabulary is a
3273 + * correct answer. Only an unset value is worth prompting about.
3274 + * `LocalBusiness` is the general-purpose answer and is accepted as one —
3275 + * with a note that a more specific type sharpens the schema, phrased as the
3276 + * guidance it is rather than as a fault the user has to clear.
3277 + *
3278 + * @since 2.10.0
3279 + *
3280 + * @param array $settings Site identity settings.
3281 + * @return array{status:string,message:string} `valid` or `suggestion`.
3282 + */
3283 + private function business_type_status(array $settings): array {
3284 + $type = trim((string) ($settings['business_type'] ?? ''));
3285 +
3286 + if ('' === $type) {
3287 + return [
3288 + 'status' => 'suggestion',
3289 + 'message' => __('Select a business type so your local schema describes the right kind of business.', 'thinkrank'),
3290 + ];
3291 + }
3292 +
3293 + // The literal rather than a constant from the expanded type list (#623):
3294 + // that lands on its own branch, and this fix must not wait on it.
3295 + if (self::GENERAL_BUSINESS_TYPE === $type) {
3296 + return [
3297 + 'status' => 'valid',
3298 + 'message' => __('Business type is set to Local Business. A more specific type sharpens your schema, if one fits.', 'thinkrank'),
3299 + ];
3300 + }
3301 +
3302 + return [
3303 + 'status' => 'valid',
3304 + 'message' => __('Business type is selected for proper schema markup.', 'thinkrank'),
3305 + ];
3306 + }
3307 +
3308 + /**
2412 3309 * Get default settings for a context type (implements interface)
2413 3310 *
2414 3311 * @since 1.0.0
2415 3312 *
@@ -2429,9 +3326,32 @@
2429 3326 'breadcrumb_home_text' => 'Home',
2430 3327 'breadcrumb_separator' => '>',
2431 3328 'robots_txt_enabled' => true,
2432 3329 'allow_search_engines' => true,
3330 + // Answer 404 when a content selector in the URL resolved to
3331 + // nothing (#634). On by default, unlike the other new settings
3332 + // here: it changes no URL a visitor or a correct crawler uses, only
3333 + // ones where WordPress resolved nothing and served the blog listing
3334 + // at 200 anyway.
3335 + 'query_protection' => true,
3336 +
3337 + // Feed controls (#635). All three off, so an upgrade changes
3338 + // nothing about what an existing site already sends its
3339 + // subscribers; a brand-new install is seeded with the signature and
3340 + // the noindex on, in Activator::seed_feed_defaults().
3341 + 'feed_excerpt_only' => false,
3342 + 'feed_source_link' => false,
3343 + 'feed_noindex' => false,
3344 +
3345 + // The scheme self-referential URLs go out with (#638). 'automatic'
3346 + // means substitute nothing and follow WordPress, which is what
3347 + // every site did before the setting existed.
3348 + 'canonical_scheme' => Url_Scheme::AUTOMATIC,
2433 3349 'robots_txt_content' => '',
3350 + // Empty map = every AI crawler allowed. Defaults must stay
3351 + // permissive so an upgrade never starts blocking a crawler a site
3352 + // was happily serving (#657).
3353 + 'ai_crawler_rules' => [],
2434 3354 'logo_url' => '',
2435 3355 'favicon_url' => '',
2436 3356 'apple_touch_icon_url' => ''
2437 3357 ];
@@ -2550,10 +3470,13 @@
2550 3470 */
2551 3471 private function prepare_title_placeholders(array $data, string $context, array $settings): array {
2552 3472 $placeholders = [
2553 3473 '%title%' => $data['title'] ?? '',
2554 - '%sitename%' => $settings['site_name'] ?? get_bloginfo('name'),
2555 - '%tagline%' => $settings['tagline'] ?? get_bloginfo('description'),
3474 + // `?:` rather than `??`: these are persisted as '' rather than left
3475 + // unset, and '' is not null, so the null-coalesce never reached the
3476 + // WordPress fallback (#398).
3477 + '%sitename%' => ($settings['site_name'] ?? '') ?: get_bloginfo('name'),
3478 + '%tagline%' => ($settings['tagline'] ?? '') ?: get_bloginfo('description'),
2556 3479 '%separator%' => '', // Will be replaced with actual separator
2557 3480 '%category%' => '',
2558 3481 '%author%' => '',
2559 3482 '%date%' => '',
@@ -2637,16 +3560,17 @@
2637 3560 // Remove extra whitespace
2638 3561 $title = preg_replace('/\s+/', ' ', $title);
2639 3562 $title = trim($title);
2640 3563
2641 - // Ensure title is not too long (60 characters max for SEO)
2642 - if (strlen($title) > 60) {
2643 - // Try to truncate at word boundary
2644 - $title = wp_trim_words($title, 8, '...');
2645 - if (strlen($title) > 60) {
2646 - $title = substr($title, 0, 57) . '...';
2647 - }
2648 - }
3564 + // Ensure title is not too long (60 characters max for SEO).
3565 + // All three units here were wrong for non-Latin text: strlen() counts
3566 + // BYTES so the gate fired at 20 Thai characters, wp_trim_words() counts
3567 + // CHARACTERS on th/ja/zh_* so `8` cut the title to 8 of them, and
3568 + // substr() cuts bytes so it split a character mid-sequence (#687).
3569 + $title = \ThinkRank\Core\Seo_Text::trim_to_length(
3570 + $title,
3571 + \ThinkRank\Core\Seo_Text::TITLE_MAX_LENGTH
3572 + );
2649 3573
2650 3574 // Ensure title is not empty
2651 3575 if (empty($title)) {
2652 3576 $title = get_bloginfo('name');
@@ -3036,8 +3960,29 @@
3036 3960 * @since 1.0.0
3037 3961 * @return array Array of sitemap URLs
3038 3962 */
3039 3963 private function get_sitemap_urls_for_robots(): array {
3964 + // One wrapper over every return path below, including the #104 extras.
3965 + // The Sitemap: line is the only absolute URL of ours in robots.txt and
3966 + // the one a crawler follows to find everything else, so it has to carry
3967 + // the site's scheme preference (#638). Applied here rather than where
3968 + // the body is assembled, because that path also renders a robots.txt a
3969 + // site owner typed themselves, and their text is not ours to rewrite.
3970 + return array_map(
3971 + static function (string $url): string {
3972 + return Url_Scheme::apply($url);
3973 + },
3974 + $this->collect_sitemap_urls_for_robots()
3975 + );
3976 + }
3977 +
3978 + /**
3979 + * The sitemap URLs robots.txt advertises, before the scheme preference.
3980 + *
3981 + * @since 1.0.0
3982 + * @return array Array of sitemap URLs
3983 + */
3984 + private function collect_sitemap_urls_for_robots(): array {
3040 3985 try {
3041 3986 // Get sitemap settings
3042 3987 $sitemap_generator = new \ThinkRank\SEO\Sitemap_Generator();
3043 3988 $sitemap_settings = $sitemap_generator->get_settings('site');
@@ -3049,29 +3994,88 @@
3049 3994
3050 3995 $sitemap_urls = [];
3051 3996 $site_url = home_url();
3052 3997
3053 - // Extract enabled sitemap URLs
3998 + // Extract enabled sitemap URLs. When the index is enabled it is the
3999 + // only entry worth advertising: every child sitemap is already
4000 + // listed inside it, so naming them again in robots.txt is pure
4001 + // redundancy and drifts out of date as soon as a post type is added.
4002 + $index_url = '';
3054 4003 if (!empty($sitemap_settings['sitemap_urls']) && is_array($sitemap_settings['sitemap_urls'])) {
3055 4004 foreach ($sitemap_settings['sitemap_urls'] as $sitemap) {
3056 - if (!empty($sitemap['enabled']) && !empty($sitemap['url'])) {
3057 - $sitemap_urls[] = $site_url . $sitemap['url'];
4005 + if (empty($sitemap['enabled']) || empty($sitemap['url'])) {
4006 + continue;
3058 4007 }
4008 +
4009 + if (($sitemap['type'] ?? '') === 'index') {
4010 + $index_url = $site_url . $sitemap['url'];
4011 + continue;
4012 + }
4013 +
4014 + $sitemap_urls[] = $site_url . $sitemap['url'];
3059 4015 }
3060 4016 }
3061 4017
4018 + if ($index_url !== '') {
4019 + // The index covers the children and, on a segmented install,
4020 + // the local business sitemap too.
4021 + //
4022 + // It does not cover a sitemap contributed through
4023 + // `thinkrank_additional_sitemaps`: the index is built by this
4024 + // plugin's own generator and never lists them. Returning the
4025 + // index alone therefore left a contributed sitemap with no
4026 + // discovery path at all — absent from robots.txt and absent
4027 + // from the index — so Pro's News sitemap was unreachable on any
4028 + // install with the index enabled, which is the default (#835).
4029 + $contributed = [];
4030 +
4031 + foreach (\ThinkRank\SEO\Sitemap_Generator::additional_sitemaps() as $path) {
4032 + $url = home_url($path);
4033 +
4034 + if ($url !== $index_url && !in_array($url, $contributed, true)) {
4035 + $contributed[] = $url;
4036 + }
4037 + }
4038 +
4039 + return array_merge([$index_url], $contributed);
4040 + }
4041 +
3062 4042 // Fallback to default if no URLs found
3063 4043 if (empty($sitemap_urls)) {
3064 4044 $sitemap_urls[] = home_url('/sitemap.xml');
3065 4045 }
3066 4046
3067 - // Advertise the local business sitemap when it exists. In segmented
3068 - // mode it is already listed inside the sitemap index; in single mode
3069 - // there is no index, so robots.txt is its discovery path.
3070 - if (file_exists(ABSPATH . 'local-sitemap.xml')) {
3071 - $local_url = home_url('/local-sitemap.xml');
3072 - if (!in_array($local_url, $sitemap_urls, true)) {
3073 - $sitemap_urls[] = $local_url;
4047 + // No index on this install, so anything not already listed above has
4048 + // no other discovery path — advertise it directly. The local
4049 + // business sitemap and the sitemaps other plugins register both land
4050 + // here for the same reason, so they go through one list (#104).
4051 + $extra = [];
4052 +
4053 + // Not a file test. Under dynamic delivery the local sitemap is
4054 + // served from PHP and no file is ever written, so file_exists()
4055 + // silently dropped a sitemap the site really does publish (#752).
4056 + // On static sites the file is still what proves it, so both count.
4057 + $local_sitemap_published = file_exists(ABSPATH . 'local-sitemap.xml');
4058 +
4059 + if (!$local_sitemap_published && class_exists('ThinkRank\\SEO\\Sitemap_Generator')) {
4060 + $generator = new \ThinkRank\SEO\Sitemap_Generator(false);
4061 +
4062 + $local_sitemap_published = 'dynamic' === $generator->resolve_delivery_mode()
4063 + && $generator->publishes_local_sitemap();
4064 + }
4065 +
4066 + if ($local_sitemap_published) {
4067 + $extra[] = '/local-sitemap.xml';
4068 + }
4069 +
4070 + foreach (\ThinkRank\SEO\Sitemap_Generator::additional_sitemaps() as $path) {
4071 + $extra[] = $path;
4072 + }
4073 +
4074 + foreach ($extra as $path) {
4075 + $url = home_url($path);
4076 + if (!in_array($url, $sitemap_urls, true)) {
4077 + $sitemap_urls[] = $url;
3074 4078 }
3075 4079 }
3076 4080
3077 4081 return $sitemap_urls;
@@ -3123,9 +4127,12 @@
3123 4127 $validation['warnings'][] = "Path '{$value}' should start with '/'";
3124 4128 }
3125 4129 break;
3126 4130 case 'sitemap':
3127 - if (!filter_var($value, FILTER_VALIDATE_URL)) {
4131 + // Url_Validator, not the raw PHP filter: on a site with an
4132 + // internationalised domain the site's own sitemap URL is
4133 + // non-ASCII and the raw filter refused it.
4134 + if (!\ThinkRank\Core\Url_Validator::is_valid($value)) {
3128 4135 $validation['errors'][] = "Invalid sitemap URL: {$value}";
3129 4136 $validation['valid'] = false;
3130 4137 }
3131 4138 break;
@@ -3161,8 +4168,9 @@
3161 4168 // timestamp on every update.
3162 4169 $content = '';
3163 4170
3164 4171 $current_user_agent = '';
4172 + $sitemap_started = false;
3165 4173
3166 4174 foreach ($rules as $rule) {
3167 4175 $directive = $rule['directive'] ?? '';
3168 4176 $value = $rule['value'] ?? '';
@@ -3183,9 +4191,17 @@
3183 4191 case 'crawl_delay':
3184 4192 $content .= "Crawl-delay: {$value}\n";
3185 4193 break;
3186 4194 case 'sitemap':
3187 - $content .= "\nSitemap: {$value}\n";
4195 + // One blank line separates the Sitemap block from the
4196 + // preceding group, and none appear inside it. A blank line
4197 + // terminates a record in the robots.txt grammar, so putting
4198 + // one between every directive was invalid formatting.
4199 + if (!$sitemap_started) {
4200 + $content .= "\n";
4201 + $sitemap_started = true;
4202 + }
4203 + $content .= "Sitemap: {$value}\n";
3188 4204 break;
3189 4205 }
3190 4206 }
3191 4207
@@ -3192,8 +4208,171 @@
3192 4208 return ltrim($content, "\n");
3193 4209 }
3194 4210
3195 4211 /**
4212 + * Parse a robots.txt body back into the {directive, value} rule shape.
4213 + *
4214 + * generate_robots_txt() returns `rules` alongside `content`, but callers
4215 + * replace `content` with the body actually being served (a stored override
4216 + * or a physical file). The generated rules then described something the
4217 + * response no longer contained. Re-deriving them from the served body keeps
4218 + * the two halves of the payload describing the same document.
4219 + *
4220 + * @since 2.0.1
4221 + *
4222 + * @param string $content Robots.txt body (header optional).
4223 + * @return array<int, array{directive: string, value: string}> Parsed rules.
4224 + */
4225 + public function parse_robots_txt_rules(string $content): array {
4226 + $map = [
4227 + 'user-agent' => 'user_agent',
4228 + 'disallow' => 'disallow',
4229 + 'allow' => 'allow',
4230 + 'crawl-delay' => 'crawl_delay',
4231 + 'sitemap' => 'sitemap',
4232 + ];
4233 +
4234 + $rules = [];
4235 +
4236 + foreach (preg_split('/\r\n|\r|\n/', $this->strip_robots_header($content)) as $line) {
4237 + $line = trim($line);
4238 +
4239 + // Blank lines separate groups and `#` starts a comment; neither is
4240 + // a rule.
4241 + if ($line === '' || str_starts_with($line, '#')) {
4242 + continue;
4243 + }
4244 +
4245 + $parts = explode(':', $line, 2);
4246 + if (count($parts) !== 2) {
4247 + continue;
4248 + }
4249 +
4250 + $field = strtolower(trim($parts[0]));
4251 + if (!isset($map[$field])) {
4252 + continue;
4253 + }
4254 +
4255 + $rules[] = [
4256 + 'directive' => $map[$field],
4257 + // Sitemap values are absolute URLs and contain the `:` the
4258 + // limited explode above deliberately preserved.
4259 + 'value' => trim($parts[1]),
4260 + ];
4261 + }
4262 +
4263 + return $rules;
4264 + }
4265 +
4266 + /**
4267 + * Opening fence of the machine-owned AI crawler region.
4268 + *
4269 + * @since 2.5.0
4270 + * @var string
4271 + */
4272 + public const AI_BLOCK_BEGIN = '# BEGIN ThinkRank AI crawlers';
4273 +
4274 + /**
4275 + * Closing fence of the machine-owned AI crawler region.
4276 + *
4277 + * @since 2.5.0
4278 + * @var string
4279 + */
4280 + public const AI_BLOCK_END = '# END ThinkRank AI crawlers';
4281 +
4282 + /**
4283 + * Render the fenced AI crawler region for the current settings.
4284 + *
4285 + * One `User-agent:` / `Disallow: /` record per blocked crawler. Allowed
4286 + * crawlers emit nothing at all: `Disallow:` with an empty value is the
4287 + * robots.txt way of saying "allow everything", but writing eighteen such
4288 + * records to say what silence already says would triple the file and
4289 + * invite the reading that an unlisted crawler is therefore refused.
4290 + *
4291 + * @since 2.5.0
4292 + *
4293 + * @param array $settings Site settings.
4294 + * @return string Fenced block, newline-terminated, or '' when nothing is blocked.
4295 + */
4296 + private function build_ai_crawler_block(array $settings): string {
4297 + $blocked = AI_Crawlers::blocked_slugs($settings['ai_crawler_rules'] ?? []);
4298 +
4299 + if (empty($blocked)) {
4300 + return '';
4301 + }
4302 +
4303 + $agents = AI_Crawlers::all();
4304 +
4305 + $lines = [
4306 + self::AI_BLOCK_BEGIN,
4307 + '# Managed by ThinkRank — edits between these lines are overwritten.',
4308 + ];
4309 +
4310 + foreach ($blocked as $slug) {
4311 + $lines[] = '';
4312 + $lines[] = 'User-agent: ' . $agents[$slug]['token'];
4313 + $lines[] = 'Disallow: /';
4314 + }
4315 +
4316 + $lines[] = self::AI_BLOCK_END;
4317 +
4318 + return implode("\n", $lines) . "\n";
4319 + }
4320 +
4321 + /**
4322 + * Remove the fenced AI crawler region from a robots.txt body.
4323 + *
4324 + * Tolerates a missing closing fence rather than leaving the rest of the
4325 + * file swallowed: a truncated write, or someone deleting the END line by
4326 + * hand, would otherwise make every subsequent read drop everything below
4327 + * the opening fence.
4328 + *
4329 + * @since 2.5.0
4330 + *
4331 + * @param string $body Robots.txt body.
4332 + * @return string Body with the region removed.
4333 + */
4334 + public function strip_ai_crawler_block(string $body): string {
4335 + if (false === strpos($body, self::AI_BLOCK_BEGIN)) {
4336 + return $body;
4337 + }
4338 +
4339 + $pattern = '/\R*' . preg_quote(self::AI_BLOCK_BEGIN, '/')
4340 + . '.*?(?:' . preg_quote(self::AI_BLOCK_END, '/') . '|\z)\R*/s';
4341 +
4342 + return trim((string) preg_replace($pattern, "\n\n", $body, 1));
4343 + }
4344 +
4345 + /**
4346 + * Put the current AI crawler region into a robots.txt body.
4347 + *
4348 + * Replaces an existing region in place so the block keeps its position in
4349 + * a hand-ordered file, and appends when there is none. Everything outside
4350 + * the fences is returned untouched — that is the whole point of fencing
4351 + * it, since the body is also a free-text field the user edits.
4352 + *
4353 + * @since 2.5.0
4354 + *
4355 + * @param string $body Robots.txt body (fences optional).
4356 + * @param array $settings Site settings.
4357 + * @return string Body carrying the current region.
4358 + */
4359 + private function apply_ai_crawler_block(string $body, array $settings): string {
4360 + $stripped = $this->strip_ai_crawler_block($body);
4361 + $block = $this->build_ai_crawler_block($settings);
4362 +
4363 + if ('' === $block) {
4364 + return $stripped;
4365 + }
4366 +
4367 + if ('' === trim($stripped)) {
4368 + return trim($block);
4369 + }
4370 +
4371 + return rtrim($stripped) . "\n\n" . trim($block);
4372 + }
4373 +
4374 + /**
3196 4375 * The auto-generated header prepended to the served robots.txt.
3197 4376 *
3198 4377 * Kept separate from the body so it is only ever added at render time with
3199 4378 * a fresh timestamp, never stored or shown in the editable textarea.
@@ -3230,11 +4409,16 @@
3230 4409 */
3231 4410 public function get_served_robots_body(): string {
3232 4411 $settings = $this->get_settings('site');
3233 4412
4413 + // The AI block is stripped from every one of these paths. A physical
4414 + // robots.txt we wrote carries it, and the stored override is whatever
4415 + // the textarea last held — so without this the block round-trips into
4416 + // the editor, gets saved as ordinary body text, and is then appended
4417 + // to a second time on the next render.
3234 4418 $custom = trim((string) ($settings['robots_txt_content'] ?? ''));
3235 4419 if ($custom !== '') {
3236 - return $this->strip_robots_header($custom);
4420 + return $this->strip_ai_crawler_block($this->strip_robots_header($custom));
3237 4421 }
3238 4422
3239 4423 $robots_file = ABSPATH . 'robots.txt';
3240 4424 if (file_exists($robots_file)) {
@@ -3240,13 +4424,13 @@
3240 4424 if (file_exists($robots_file)) {
3241 4425 // phpcs:ignore WordPress.WP.AlternativeFunctions.file_get_contents_file_get_contents, WordPress.PHP.NoSilencedErrors.Discouraged -- an unreadable robots.txt is an expected state answered with an empty string.
3242 4426 $raw = (string) @file_get_contents($robots_file);
3243 4427 if ($raw !== '') {
3244 - return $this->strip_robots_header($raw);
4428 + return $this->strip_ai_crawler_block($this->strip_robots_header($raw));
3245 4429 }
3246 4430 }
3247 4431
3248 - return trim($this->generate_robots_txt()['content']);
4432 + return $this->strip_ai_crawler_block(trim($this->generate_robots_txt()['content']));
3249 4433 }
3250 4434 private function get_site_identity_data(array $settings): array {
3251 4435 return [
3252 4436 'site_name' => $settings['site_name'] ?? get_bloginfo('name'),
@@ -3308,11 +4492,14 @@
3308 4492 $optimization['validation']['valid'] = false;
3309 4493 }
3310 4494
3311 4495 if (!empty($value) && isset($config['max_length'])) {
3312 - if (strlen($value) > $config['max_length']) {
4496 + // The warning says "characters", so measure and cut in characters:
4497 + // strlen()/substr() fired early on non-Latin values and the
4498 + // suggested replacement was cut mid-character (#687).
4499 + if (mb_strlen($value) > $config['max_length']) {
3313 4500 $optimization['validation']['warnings'][] = "{$element} exceeds maximum length of {$config['max_length']} characters";
3314 - $optimization['optimized_value'] = substr($value, 0, $config['max_length']);
4501 + $optimization['optimized_value'] = \ThinkRank\Core\Seo_Text::trim_to_length($value, (int) $config['max_length']);
3315 4502 }
3316 4503 }
3317 4504
3318 4505 // SEO-specific optimizations
@@ -3344,25 +4531,27 @@
3344 4531 return $optimization;
3345 4532 }
3346 4533
3347 4534 // Validate URL
3348 - if (!filter_var($value, FILTER_VALIDATE_URL)) {
3349 - $optimization['validation']['errors'][] = "{$element} must be a valid URL";
4535 + if (!\ThinkRank\Core\Url_Validator::is_http_url($value)) {
4536 + $optimization['validation']['errors'][] = "{$element} must be an http or https URL";
3350 4537 $optimization['validation']['valid'] = false;
3351 4538 return $optimization;
3352 4539 }
3353 4540
3354 4541 // Check if it's a local image
3355 - $attachment_id = attachment_url_to_postid($value);
4542 + $attachment_id = Attachment_Lookup::id_from_url($value);
3356 4543 if ($attachment_id) {
3357 4544 $image_meta = wp_get_attachment_metadata($attachment_id);
3358 4545
3359 4546 if ($image_meta && isset($image_meta['width'], $image_meta['height'])) {
3360 - // Check recommended size
4547 + // Check recommended size, against the configured file itself
4548 + // rather than the upload it may have been generated from.
3361 4549 if (isset($config['recommended_size'])) {
3362 4550 [$rec_width, $rec_height] = explode('x', $config['recommended_size']);
4551 + $image_file = Attachment_Lookup::describe($attachment_id, $value);
3363 4552
3364 - if ((int) $image_meta['width'] !== (int) $rec_width || (int) $image_meta['height'] !== (int) $rec_height) {
4553 + if ($image_file['width'] !== (int) $rec_width || $image_file['height'] !== (int) $rec_height) {
3365 4554 $optimization['suggestions'][] = "Consider using {$config['recommended_size']} size for optimal {$element}";
3366 4555 }
3367 4556 }
3368 4557