PluginProbe
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO / 2.14.0
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO v2.14.0
2.14.0 2.13.0 2.12.0 2.11.0 2.10.0 2.9.0 2.8.0 2.7.0 2.6.0 2.5.0 2.4.0 2.3.0 2.2.0 2.1.1 2.1.0 2.0.2 2.0.1 2.0.0 1.32.0 1.31.0 1.30.0 1.29.0 1.28.0 1.27.0 1.26.0 All 55 releases
← All changes | includes/seo/class-site-identity-manager.php +190 -3 2.12.0 → 2.14.0 View file →
@@ -318,8 +318,24 @@
318 318 * @var bool
319 319 */
320 320 private static bool $icon_sizes_listener_registered = false;
321 321
322 + /**
323 + * Whether the robots.txt resync listener is registered for this request.
324 + *
325 + * @since 2.14.0
326 + * @var bool
327 + */
328 + private static bool $robots_sync_listener_registered = false;
329 +
330 + /**
331 + * Flag set when a plugin change may have altered the sitemap set.
332 + *
333 + * @since 2.14.0
334 + * @var string
335 + */
336 + public const ROBOTS_RESYNC_OPTION = 'thinkrank_robots_txt_resync_pending';
337 +
322 338 public function __construct() {
323 339 parent::__construct('site_identity');
324 340
325 341 if (!self::$icon_sizes_listener_registered) {
@@ -328,11 +344,72 @@
328 344 // Admin only: resizing is not front-end work, and admin traffic is
329 345 // enough to run a one-time backfill promptly.
330 346 add_action('admin_init', [self::class, 'maybe_backfill_icon_sizes']);
331 347 }
348 +
349 + if (!self::$robots_sync_listener_registered) {
350 + self::$robots_sync_listener_registered = true;
351 +
352 + // A physical robots.txt bypasses PHP entirely, so composing the
353 + // Sitemap block at render time fixes the served output only on
354 + // sites with no file. Activating or deactivating a sitemap
355 + // contributor changes the set, and until #835 nothing rewrote the
356 + // file: the deactivated plugin's sitemap stayed advertised, serving
357 + // HTML to anything that followed it.
358 + add_action('activated_plugin', [self::class, 'flag_robots_txt_resync']);
359 + add_action('deactivated_plugin', [self::class, 'flag_robots_txt_resync']);
360 + add_action('init', [self::class, 'maybe_resync_robots_txt'], 99);
361 + }
332 362 }
333 363
334 364 /**
365 + * Note that the set of sitemap contributors may have changed.
366 + *
367 + * Deliberately unconditional about which plugin: a contributor is anything
368 + * hooking `thinkrank_additional_sitemaps`, which is resolved at runtime and
369 + * cannot be inspected for a plugin that is on its way out.
370 + *
371 + * The rewrite is not done here. `deactivated_plugin` fires inside the
372 + * request that deactivated it, while that plugin's filters are still
373 + * attached, so rendering now still sees the sitemap that is going away —
374 + * measured, not assumed: the first version of this fix wrote the
375 + * deactivated plugin's sitemap straight back into the file. The next
376 + * request has the real plugin set loaded, so the work waits for it.
377 + *
378 + * @since 2.14.0
379 + * @return void
380 + */
381 + public static function flag_robots_txt_resync(): void {
382 + if (!file_exists(ABSPATH . 'robots.txt')) {
383 + return;
384 + }
385 +
386 + update_option(self::ROBOTS_RESYNC_OPTION, 1, false);
387 + }
388 +
389 + /**
390 + * Rewrite the physical robots.txt once, on the request after a change.
391 + *
392 + * @since 2.14.0
393 + * @return void
394 + */
395 + public static function maybe_resync_robots_txt(): void {
396 + if (!get_option(self::ROBOTS_RESYNC_OPTION)) {
397 + return;
398 + }
399 +
400 + // Cleared first, so a render that fatals cannot retry on every request
401 + // for the rest of the site's life.
402 + delete_option(self::ROBOTS_RESYNC_OPTION);
403 +
404 + if (!file_exists(ABSPATH . 'robots.txt')) {
405 + return;
406 + }
407 +
408 + (new self())->sync_robots_txt_file();
409 + }
410 +
411 + /**
335 412 * Save settings, then refresh what a new canonical scheme invalidates.
336 413 *
337 414 * The static sitemap files are written with the scheme in force when they
338 415 * were built, and nothing else rebuilds them until a post or term changes.
@@ -814,9 +891,20 @@
814 891 // here rather than stored: the textarea holds the user's body, with
815 892 // the fenced block stripped out of every read and re-applied on every
816 893 // render. A site-wide block already disallows everyone, so adding the
817 894 // per-agent group there would be noise restating the same refusal.
895 + // Composed here rather than read from storage, for the same reason as
896 + // the AI block below: the set of sitemaps an install publishes is a
897 + // runtime fact. `robots_txt_content` is a snapshot of it taken at the
898 + // last save, and nothing invalidated that snapshot, so deactivating a
899 + // sitemap provider left its URL advertised and serving HTML (#835).
900 + // Composing it on every render means the advertisement agrees with what
901 + // the install publishes, in both directions, with no cache to expire.
818 902 if (!$fully_blocked) {
903 + $body = $this->apply_sitemap_block($body);
904 + }
905 +
906 + if (!$fully_blocked) {
819 907 $body = $this->apply_ai_crawler_block($body, $settings);
820 908 }
821 909
822 910 if ($body === '') {
@@ -826,8 +914,87 @@
826 914 return $this->robots_txt_header() . $body . "\n";
827 915 }
828 916
829 917 /**
918 + * Replace the generated Sitemap block with the one this install publishes.
919 + *
920 + * @since 2.14.0
921 + * @param string $body Robots.txt body, without the header.
922 + * @return string
923 + */
924 + private function apply_sitemap_block(string $body): string {
925 + $stripped = $this->strip_generated_sitemap_block($body);
926 + $urls = $this->get_sitemap_urls_for_robots();
927 +
928 + if (empty($urls)) {
929 + return $stripped;
930 + }
931 +
932 + $block = '';
933 + foreach ($urls as $url) {
934 + $block .= 'Sitemap: ' . $url . "\n";
935 + }
936 +
937 + if ('' === trim($stripped)) {
938 + return trim($block);
939 + }
940 +
941 + // The grammar build_robots_txt_content() writes: one blank line before
942 + // the block, none inside it. A blank line terminates a record in the
943 + // robots.txt grammar, so a line between every directive is invalid.
944 + return rtrim($stripped) . "\n\n" . trim($block);
945 + }
946 +
947 + /**
948 + * Remove the plugin-written Sitemap block from a stored body.
949 + *
950 + * Only the trailing run of `Sitemap:` lines is removed, which is the exact
951 + * shape `build_robots_txt_content()` writes: a blank line, then nothing but
952 + * `Sitemap:` lines to the end of the body. A `Sitemap:` line anywhere else
953 + * was typed by the site owner and is left exactly where they put it, which
954 + * is why this cannot simply strip every matching line.
955 + *
956 + * @since 2.14.0
957 + * @param string $body Robots.txt body.
958 + * @return string
959 + */
960 + private function strip_generated_sitemap_block(string $body): string {
961 + $lines = preg_split('/\R/', $body);
962 +
963 + if (!is_array($lines)) {
964 + return $body;
965 + }
966 +
967 + $cut = count($lines);
968 +
969 + // Walk back over the trailing block: sitemap lines, and the blank lines
970 + // that separate or pad it. Anything else ends the block.
971 + for ($i = count($lines) - 1; $i >= 0; $i--) {
972 + $line = trim($lines[$i]);
973 +
974 + if ('' === $line) {
975 + $cut = $i;
976 + continue;
977 + }
978 +
979 + if (0 === stripos($line, 'sitemap:')) {
980 + $cut = $i;
981 + continue;
982 + }
983 +
984 + break;
985 + }
986 +
987 + if ($cut >= count($lines)) {
988 + return $body;
989 + }
990 +
991 + // Nothing but sitemap lines in the whole body means there is no owner
992 + // content to keep.
993 + return rtrim(implode("\n", array_slice($lines, 0, $cut)));
994 + }
995 +
996 + /**
830 997 * Resolve the robots.txt actually served to crawlers, with its origin.
831 998 *
832 999 * Lets an API/MCP consumer see the effective output without crawling the
833 1000 * URL. Mirrors serving precedence: a physical robots.txt in the web root is
@@ -2906,8 +3073,10 @@
2906 3073 return [
2907 3074 // Title formats, one per context.
2908 3075 'homepage_title', 'post_title', 'page_title', 'category_title',
2909 3076 'tag_title', 'author_title', 'search_title', 'archive_title',
3077 + // The blog-index homepage's meta description (#897).
3078 + 'homepage_description',
2910 3079 // Breadcrumbs.
2911 3080 'breadcrumb_prefix', 'show_current_page', 'breadcrumb_use_seo_title',
2912 3081 // Identity, as written by the setup wizard and the importers.
2913 3082 'alternate_name', 'identity_type', 'represents',
@@ -3751,11 +3920,29 @@
3751 3920 }
3752 3921 }
3753 3922
3754 3923 if ($index_url !== '') {
3755 - // The index alone — it covers the children and, on a segmented
3756 - // install, the local business sitemap too.
3757 - return [$index_url];
3924 + // The index covers the children and, on a segmented install,
3925 + // the local business sitemap too.
3926 + //
3927 + // It does not cover a sitemap contributed through
3928 + // `thinkrank_additional_sitemaps`: the index is built by this
3929 + // plugin's own generator and never lists them. Returning the
3930 + // index alone therefore left a contributed sitemap with no
3931 + // discovery path at all — absent from robots.txt and absent
3932 + // from the index — so Pro's News sitemap was unreachable on any
3933 + // install with the index enabled, which is the default (#835).
3934 + $contributed = [];
3935 +
3936 + foreach (\ThinkRank\SEO\Sitemap_Generator::additional_sitemaps() as $path) {
3937 + $url = home_url($path);
3938 +
3939 + if ($url !== $index_url && !in_array($url, $contributed, true)) {
3940 + $contributed[] = $url;
3941 + }
3942 + }
3943 +
3944 + return array_merge([$index_url], $contributed);
3758 3945 }
3759 3946
3760 3947 // Fallback to default if no URLs found
3761 3948 if (empty($sitemap_urls)) {