PluginProbe
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO / 2.13.0
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO v2.13.0
2.13.0 2.12.0 2.11.0 2.10.0 2.9.0 2.8.0 2.7.0 2.6.0 2.5.0 2.4.0 2.3.0 2.2.0 2.1.1 2.1.0 2.0.2 2.0.1 2.0.0 1.32.0 1.31.0 1.30.0 1.29.0 1.28.0 1.27.0 1.26.0 1.25.0 All 54 releases
← All changes | includes/seo/class-site-identity-manager.php +507 -63 2.4.0 → 2.13.0 View file →
@@ -40,8 +40,36 @@
40 40 */
41 41 class Site_Identity_Manager extends Abstract_SEO_Manager {
42 42
43 43 /**
44 + * The per-context title formats as ThinkRank ships them.
45 + *
46 + * These are not in get_default_settings(): the admin screen seeds them on
47 + * first save, so on a real install they are stored values, indistinguishable
48 + * from a template the user typed. The migration needs to tell those two
49 + * apart — it may overwrite a shipped default with an imported template, and
50 + * must never overwrite a choice the user made — so this is the record of
51 + * what "untouched" looks like.
52 + *
53 + * Keep in step with getDefaultSettings() in
54 + * src/admin/components/essential-seo/SiteIdentityTab.js. SiteIdentityTitleFormatDefaultsTest
55 + * fails when the two drift.
56 + *
57 + * @since 2.8.0
58 + * @var array<string, string>
59 + */
60 + public const TITLE_FORMAT_DEFAULTS = [
61 + 'homepage_title' => '%site_title% %sep% %site_description%',
62 + 'post_title' => '%post_title% %sep% %site_title%',
63 + 'page_title' => '%page_title% %sep% %site_title%',
64 + 'category_title' => '%category_title% %sep% %site_title%',
65 + 'tag_title' => '%tag_title% %sep% %site_title%',
66 + 'author_title' => '%author_name% %sep% %site_title%',
67 + 'search_title' => 'Search Results for "%search_term%" %sep% %site_title%',
68 + 'archive_title' => '%archive_title% %sep% %site_title%',
69 + ];
70 +
71 + /**
44 72 * WordPress filesystem instance
45 73 *
46 74 * @since 1.0.0
47 75 * @var \WP_Filesystem_Base|null
@@ -303,8 +331,91 @@
303 331 }
304 332 }
305 333
306 334 /**
335 + * Save settings, then refresh what a new canonical scheme invalidates.
336 + *
337 + * The static sitemap files are written with the scheme in force when they
338 + * were built, and nothing else rebuilds them until a post or term changes.
339 + * So a change of scheme left every `<loc>` on the old one while canonical
340 + * and og:url had already moved (#736). Every writer (the settings route,
341 + * the robots route, the MCP abilities, an import) lands here.
342 + *
343 + * @since 2.7.0
344 + *
345 + * @param string $context_type Context type.
346 + * @param int|null $context_id Context ID.
347 + * @param array $settings Settings to save.
348 + * @return bool
349 + */
350 + public function save_settings(string $context_type, ?int $context_id, array $settings): bool {
351 + if (!self::touches_canonical_scheme($context_type, $context_id, $settings)) {
352 + return parent::save_settings($context_type, $context_id, $settings);
353 + }
354 +
355 + $before = Url_Scheme::preference();
356 + $saved = parent::save_settings($context_type, $context_id, $settings);
357 +
358 + if ($saved) {
359 + $this->on_canonical_scheme_saved($before);
360 + }
361 +
362 + return $saved;
363 + }
364 +
365 + /**
366 + * Whether a save can change the site-wide canonical scheme.
367 + *
368 + * @since 2.7.0
369 + *
370 + * @param string $context_type Context type.
371 + * @param int|null $context_id Context ID.
372 + * @param array $settings Settings being saved.
373 + * @return bool
374 + */
375 + public static function touches_canonical_scheme(string $context_type, ?int $context_id, array $settings): bool {
376 + return 'site' === sanitize_key($context_type)
377 + && empty($context_id)
378 + && array_key_exists('canonical_scheme', $settings);
379 + }
380 +
381 + /**
382 + * Rebuild the static sitemaps when the effective scheme changed.
383 + *
384 + * Compares the effective preference, filter included, so a site whose
385 + * scheme is pinned by `thinkrank_canonical_scheme` does not rebuild on a
386 + * stored value that changes nothing it publishes.
387 + *
388 + * @since 2.7.0
389 + *
390 + * @param string $before Effective scheme before the save.
391 + * @return void
392 + */
393 + protected function on_canonical_scheme_saved(string $before): void {
394 + // The preference is cached for the request; the save just changed it.
395 + Url_Scheme::reset();
396 +
397 + if (Url_Scheme::preference() === $before) {
398 + return;
399 + }
400 +
401 + $this->schedule_sitemap_rebuild();
402 + }
403 +
404 + /**
405 + * Queue a settings-driven sitemap rebuild.
406 + *
407 + * Debounced and run after the response, like any other settings change
408 + * that alters what the sitemap publishes.
409 + *
410 + * @since 2.7.0
411 + * @return void
412 + */
413 + protected function schedule_sitemap_rebuild(): void {
414 + (new Sitemap_Generator(false))->schedule_regeneration();
415 + }
416 +
417 + /**
307 418 * Build the icon derivatives for a newly chosen favicon.
308 419 *
309 420 * Runs on save, which is the only moment the choice changes and the only
310 421 * place image work belongs — resolving a size on the front end must stay a
@@ -397,11 +508,12 @@
397 508 * Attachment ID behind a configured icon URL, or 0 when it is not ours.
398 509 *
399 510 * attachment_url_to_postid() matches _wp_attached_file, which holds the
400 511 * ORIGINAL upload path, so the URL of a generated derivative
401 - * (`logo-512.png`) returns 0 — and that is exactly what the media picker
402 - * hands back when the user chooses a size. Strip the dimension suffix and
403 - * try the original once.
512 + * (`logo-512x512.png`) returns 0 — and that is exactly what the media
513 + * picker hands back when the user chooses a size. Attachment_Lookup falls
514 + * back to the original behind it; the fallback started here and moved
515 + * there when every other image lookup turned out to need it (#847).
404 516 *
405 517 * Shared with SEO_Manager's site-icon filter so both sides of the feature
406 518 * agree on which attachment a configured URL means.
407 519 *
@@ -410,21 +522,9 @@
410 522 * @param string $url Configured icon URL.
411 523 * @return int Attachment ID, or 0.
412 524 */
413 525 public static function icon_attachment_id(string $url): int {
414 - $attachment_id = (int) attachment_url_to_postid($url);
415 -
416 - if ($attachment_id) {
417 - return $attachment_id;
418 - }
419 -
420 - $original = preg_replace('/-\d+x\d+(?=\.[a-zA-Z0-9]+$)/', '', $url);
421 -
422 - if (is_string($original) && $original !== $url) {
423 - return (int) attachment_url_to_postid($original);
424 - }
425 -
426 - return 0;
526 + return Attachment_Lookup::id_from_url($url);
427 527 }
428 528
429 529 /**
430 530 * Which ICON_SIZES derivatives this attachment still needs.
@@ -709,8 +809,17 @@
709 809 $body = ($custom !== '' && !$fully_blocked)
710 810 ? $this->strip_robots_header($custom)
711 811 : trim($this->generate_robots_txt()['content']);
712 812
813 + // The per-agent AI directives are machine-owned, so they are composed
814 + // here rather than stored: the textarea holds the user's body, with
815 + // the fenced block stripped out of every read and re-applied on every
816 + // render. A site-wide block already disallows everyone, so adding the
817 + // per-agent group there would be noise restating the same refusal.
818 + if (!$fully_blocked) {
819 + $body = $this->apply_ai_crawler_block($body, $settings);
820 + }
821 +
713 822 if ($body === '') {
714 823 return '';
715 824 }
716 825
@@ -786,9 +895,12 @@
786 895 $settings = $this->get_settings('site');
787 896
788 897 // Compare bodies, not raw strings: the auto-generated header carries a
789 898 // regeneration timestamp that always differs and means nothing here.
790 - $served = $this->strip_robots_header($effective['content']);
899 + // The AI crawler block is composed at render time on both sides, so it
900 + // is identical by construction and comparing it would only ever report
901 + // a false drift the admin cannot act on.
902 + $served = $this->strip_ai_crawler_block($this->strip_robots_header($effective['content']));
791 903
792 904 // Measure against the body the editor is displaying — get_served_robots_body()
793 905 // — not against render_robots_txt(). Two things made the old comparison
794 906 // report "in sync" while the screen showed rules no crawler receives:
@@ -1166,17 +1278,19 @@
1166 1278 $home_text = $settings['breadcrumb_home_text'] ?? 'Home';
1167 1279 if (empty($home_text)) {
1168 1280 $optimization['warnings'][] = 'Empty home text reduces accessibility for screen readers';
1169 1281 $optimization['score'] -= 15;
1170 - } elseif (strlen($home_text) > 20) {
1171 - $optimization['suggestions'][] = 'Keep home text concise (current: ' . strlen($home_text) . ' chars)';
1282 + } elseif (mb_strlen($home_text) > 20) {
1283 + // mb_strlen: this number is shown to the user as "chars" (#687).
1284 + $optimization['suggestions'][] = 'Keep home text concise (current: ' . mb_strlen($home_text) . ' chars)';
1172 1285 $optimization['score'] -= 5;
1173 1286 }
1174 1287
1175 1288 // Check prefix usage
1176 1289 $prefix = $settings['breadcrumb_prefix'] ?? '';
1177 - if (!empty($prefix) && strlen($prefix) > 50) {
1178 - $optimization['suggestions'][] = 'Breadcrumb prefix is quite long (' . strlen($prefix) . ' chars) - consider shortening';
1290 + if (!empty($prefix) && mb_strlen($prefix) > 50) {
1291 + // mb_strlen: this number is shown to the user as "chars" (#687).
1292 + $optimization['suggestions'][] = 'Breadcrumb prefix is quite long (' . mb_strlen($prefix) . ' chars) - consider shortening';
1179 1293 $optimization['score'] -= 5;
1180 1294 }
1181 1295
1182 1296 // Current page display
@@ -1307,13 +1421,16 @@
1307 1421 }
1308 1422
1309 1423 // Additional logo analysis for local images
1310 1424 if (!empty($logo_url) && filter_var($logo_url, FILTER_VALIDATE_URL)) {
1311 - $attachment_id = attachment_url_to_postid($logo_url);
1425 + $attachment_id = Attachment_Lookup::id_from_url($logo_url);
1312 1426 if ($attachment_id) {
1313 1427 $image_meta = wp_get_attachment_metadata($attachment_id);
1314 - $width = isset($image_meta['width']) ? (int) $image_meta['width'] : 0;
1315 - $height = isset($image_meta['height']) ? (int) $image_meta['height'] : 0;
1428 + // The configured file's own size — a logo picked at a generated
1429 + // size is not as large as the upload behind it.
1430 + $logo_file = Attachment_Lookup::describe($attachment_id, $logo_url);
1431 + $width = $logo_file['width'];
1432 + $height = $logo_file['height'];
1316 1433
1317 1434 // SVG logos store 0x0 metadata — no dimension/ratio analysis
1318 1435 // is possible (and dividing by 0 is fatal).
1319 1436 if ($image_meta && $width > 0 && $height > 0) {
@@ -1416,11 +1533,12 @@
1416 1533 $optimization['score'] -= 15;
1417 1534 }
1418 1535 }
1419 1536
1420 - // Business type validation
1421 - if (empty($settings['business_type'])) {
1422 - $optimization['suggestions'][] = 'Select a specific business type for better schema markup';
1537 + // Business type validation (shared rule, one message — #622).
1538 + $business_type = $this->business_type_status($settings);
1539 + if ('suggestion' === $business_type['status']) {
1540 + $optimization['suggestions'][] = $business_type['message'];
1423 1541 $optimization['score'] -= 5;
1424 1542 }
1425 1543
1426 1544 // Email validation
@@ -1950,24 +2068,17 @@
1950 2068 'icon' => '✗'
1951 2069 ];
1952 2070 }
1953 2071
1954 - // Business Type validation
1955 - if (!empty($settings['business_type']) && $settings['business_type'] !== 'LocalBusiness') {
1956 - $field_details[] = [
1957 - 'field' => 'business_type',
1958 - 'label' => 'Business type is selected for proper schema markup.',
1959 - 'status' => 'valid',
1960 - 'icon' => '✓'
1961 - ];
1962 - } else {
1963 - $field_details[] = [
1964 - 'field' => 'business_type',
1965 - 'label' => 'Specific business type selection recommended for better schema markup.',
1966 - 'status' => 'suggestion',
1967 - 'icon' => '⚠'
1968 - ];
1969 - }
2072 + // Business Type validation — see business_type_status() for why there
2073 + // is exactly one rule here now (#622).
2074 + $business_type = $this->business_type_status($settings);
2075 + $field_details[] = [
2076 + 'field' => 'business_type',
2077 + 'label' => $business_type['message'],
2078 + 'status' => $business_type['status'],
2079 + 'icon' => 'valid' === $business_type['status'] ? '✓' : '⚠',
2080 + ];
1970 2081
1971 2082 // Address validation (NAP consistency)
1972 2083 $address_fields = ['business_address', 'business_city', 'business_state', 'business_country'];
1973 2084 $address_complete = true;
@@ -2651,11 +2762,15 @@
2651 2762 } else {
2652 2763 $validation['suggestions'][] = 'Add business hours to improve local search visibility';
2653 2764 }
2654 2765
2655 - // Validate business type
2656 - if (empty($settings['business_type'])) {
2657 - $validation['suggestions'][] = 'Select a specific business type for better schema markup';
2766 + // Business type, through the shared rule (#622). This is the only place
2767 + // it is reported on the generic path: validate_settings() with no tab
2768 + // context attaches basic-info field details, not business-info ones, so
2769 + // without this the setting would go unreported there entirely.
2770 + $business_type = $this->business_type_status($settings);
2771 + if ('suggestion' === $business_type['status']) {
2772 + $validation['suggestions'][] = $business_type['message'];
2658 2773 }
2659 2774
2660 2775 return $validation;
2661 2776 }
@@ -2747,13 +2862,54 @@
2747 2862 * @since 2.0.1
2748 2863 *
2749 2864 * @return string[]
2750 2865 */
2866 + /**
2867 + * The stored alternate name(s), shaped for schema output.
2868 + *
2869 + * schema.org and Google both allow `alternateName` to carry one value or
2870 + * several, and the store already round-trips either shape, so this accepts
2871 + * both and normalises: null when there is nothing to publish, a bare string
2872 + * for one name, a list for more. Emitting a one-element array would be
2873 + * valid but noisier than it needs to be.
2874 + *
2875 + * Shared because both WebSite producers need it and must agree — a property
2876 + * added to one and not the other is how #688 happened.
2877 + *
2878 + * @since 2.7.0
2879 + *
2880 + * @param mixed $value Stored alternate_name value.
2881 + * @return string|string[]|null
2882 + */
2883 + public static function alternate_name_for_schema($value) {
2884 + $names = [];
2885 +
2886 + foreach ((array) $value as $name) {
2887 + if (!is_scalar($name)) {
2888 + continue;
2889 + }
2890 +
2891 + $name = trim((string) $name);
2892 +
2893 + if ('' !== $name && !in_array($name, $names, true)) {
2894 + $names[] = $name;
2895 + }
2896 + }
2897 +
2898 + if (empty($names)) {
2899 + return null;
2900 + }
2901 +
2902 + return 1 === count($names) ? $names[0] : $names;
2903 + }
2904 +
2751 2905 protected function additional_setting_keys(): array {
2752 2906 return [
2753 2907 // Title formats, one per context.
2754 2908 'homepage_title', 'post_title', 'page_title', 'category_title',
2755 2909 'tag_title', 'author_title', 'search_title', 'archive_title',
2910 + // The blog-index homepage's meta description (#897).
2911 + 'homepage_description',
2756 2912 // Breadcrumbs.
2757 2913 'breadcrumb_prefix', 'show_current_page', 'breadcrumb_use_seo_title',
2758 2914 // Identity, as written by the setup wizard and the importers.
2759 2915 'alternate_name', 'identity_type', 'represents',
@@ -2762,8 +2918,10 @@
2762 2918 // Schema toggles that live on this screen.
2763 2919 'organization_schema', 'knowledge_graph',
2764 2920 // Robots rules composed by the Robots.txt panel.
2765 2921 'custom_robots_rules',
2922 + // Per-agent AI crawler allow/block map (#657).
2923 + 'ai_crawler_rules',
2766 2924 // Hero section.
2767 2925 'hero_title', 'hero_subtitle', 'hero_cta_text', 'hero_cta_url',
2768 2926 'hero_background_image',
2769 2927 // Local SEO / business details.
@@ -2775,8 +2933,118 @@
2775 2933 ];
2776 2934 }
2777 2935
2778 2936 /**
2937 + * Sanitize settings, normalising the AI crawler rule map.
2938 + *
2939 + * The generic array sanitizer keeps the shape but says nothing about the
2940 + * values: a payload could store `ai_crawler_rules[gptbot] = "maybe"`, or a
2941 + * slug no crawler answers to, and both would round-trip through every
2942 + * later response. Normalising here rather than in the REST handler puts it
2943 + * on the one path every writer shares — the settings route, the robots
2944 + * route and the MCP abilities all land in save_settings() (#657).
2945 + *
2946 + * @since 2.5.0
2947 + *
2948 + * @param array $settings Settings to sanitize.
2949 + * @param string $context_type Context type.
2950 + * @return array Sanitized settings.
2951 + */
2952 + protected function sanitize_settings(array $settings, string $context_type = 'site'): array {
2953 + $sanitized = parent::sanitize_settings($settings, $context_type);
2954 +
2955 + if (array_key_exists('ai_crawler_rules', $sanitized)) {
2956 + $sanitized['ai_crawler_rules'] = AI_Crawlers::normalize_rules($sanitized['ai_crawler_rules']);
2957 + }
2958 +
2959 + // Same reasoning one key up, for the scheme override (#638). Anything
2960 + // that is not one of the three modes means "follow WordPress", and is
2961 + // stored as that rather than kept verbatim — otherwise get-site-identity
2962 + // -settings would report a scheme the site does not actually publish.
2963 + if (array_key_exists('canonical_scheme', $sanitized)) {
2964 + $sanitized['canonical_scheme'] = in_array($sanitized['canonical_scheme'], Url_Scheme::MODES, true)
2965 + ? $sanitized['canonical_scheme']
2966 + : Url_Scheme::AUTOMATIC;
2967 + }
2968 +
2969 + // Same reasoning again for the business type. It goes straight into
2970 + // LocalBusiness schema, so a type that is not in the schema.org
2971 + // vocabulary is invalid structured data — and storing it verbatim would
2972 + // have get-site-identity-settings report a type the site cannot
2973 + // actually publish. An empty value keeps meaning "not set"; anything
2974 + // else unrecognised falls back to the general-purpose root (#623).
2975 + if (array_key_exists('business_type', $sanitized)) {
2976 + $type = (string) $sanitized['business_type'];
2977 +
2978 + if ('' !== $type && !\ThinkRank\Config\Local_Business_Types_Config::is_valid($type)) {
2979 + $type = \ThinkRank\Config\Local_Business_Types_Config::ROOT;
2980 + }
2981 +
2982 + $sanitized['business_type'] = $type;
2983 + }
2984 +
2985 + return $sanitized;
2986 + }
2987 +
2988 + /**
2989 + * schema.org's general-purpose LocalBusiness type.
2990 + *
2991 + * The default, the first option in the control, and a valid answer in its
2992 + * own right — which is the whole point of #622.
2993 + *
2994 + * @since 2.10.0
2995 + * @var string
2996 + */
2997 + private const GENERAL_BUSINESS_TYPE = 'LocalBusiness';
2998 +
2999 + /**
3000 + * The one rule for whether a business type needs the user's attention.
3001 + *
3002 + * There were three, with two wordings and two different conditions. Two
3003 + * fired when the value was empty; the third fired when it WAS
3004 + * `LocalBusiness` — which is the default, the first option in the control
3005 + * and a perfectly valid schema.org type. So the warning appeared out of the
3006 + * box for every site, could not be cleared without choosing a type that
3007 + * might be inaccurate, and on an empty value it appeared three times in two
3008 + * different phrasings, which is why it was reported as showing twice (#622).
3009 + *
3010 + * The rule now: a type is expected, and any type in the vocabulary is a
3011 + * correct answer. Only an unset value is worth prompting about.
3012 + * `LocalBusiness` is the general-purpose answer and is accepted as one —
3013 + * with a note that a more specific type sharpens the schema, phrased as the
3014 + * guidance it is rather than as a fault the user has to clear.
3015 + *
3016 + * @since 2.10.0
3017 + *
3018 + * @param array $settings Site identity settings.
3019 + * @return array{status:string,message:string} `valid` or `suggestion`.
3020 + */
3021 + private function business_type_status(array $settings): array {
3022 + $type = trim((string) ($settings['business_type'] ?? ''));
3023 +
3024 + if ('' === $type) {
3025 + return [
3026 + 'status' => 'suggestion',
3027 + 'message' => __('Select a business type so your local schema describes the right kind of business.', 'thinkrank'),
3028 + ];
3029 + }
3030 +
3031 + // The literal rather than a constant from the expanded type list (#623):
3032 + // that lands on its own branch, and this fix must not wait on it.
3033 + if (self::GENERAL_BUSINESS_TYPE === $type) {
3034 + return [
3035 + 'status' => 'valid',
3036 + 'message' => __('Business type is set to Local Business. A more specific type sharpens your schema, if one fits.', 'thinkrank'),
3037 + ];
3038 + }
3039 +
3040 + return [
3041 + 'status' => 'valid',
3042 + 'message' => __('Business type is selected for proper schema markup.', 'thinkrank'),
3043 + ];
3044 + }
3045 +
3046 + /**
2779 3047 * Get default settings for a context type (implements interface)
2780 3048 *
2781 3049 * @since 1.0.0
2782 3050 *
@@ -2796,9 +3064,32 @@
2796 3064 'breadcrumb_home_text' => 'Home',
2797 3065 'breadcrumb_separator' => '>',
2798 3066 'robots_txt_enabled' => true,
2799 3067 'allow_search_engines' => true,
3068 + // Answer 404 when a content selector in the URL resolved to
3069 + // nothing (#634). On by default, unlike the other new settings
3070 + // here: it changes no URL a visitor or a correct crawler uses, only
3071 + // ones where WordPress resolved nothing and served the blog listing
3072 + // at 200 anyway.
3073 + 'query_protection' => true,
3074 +
3075 + // Feed controls (#635). All three off, so an upgrade changes
3076 + // nothing about what an existing site already sends its
3077 + // subscribers; a brand-new install is seeded with the signature and
3078 + // the noindex on, in Activator::seed_feed_defaults().
3079 + 'feed_excerpt_only' => false,
3080 + 'feed_source_link' => false,
3081 + 'feed_noindex' => false,
3082 +
3083 + // The scheme self-referential URLs go out with (#638). 'automatic'
3084 + // means substitute nothing and follow WordPress, which is what
3085 + // every site did before the setting existed.
3086 + 'canonical_scheme' => Url_Scheme::AUTOMATIC,
2800 3087 'robots_txt_content' => '',
3088 + // Empty map = every AI crawler allowed. Defaults must stay
3089 + // permissive so an upgrade never starts blocking a crawler a site
3090 + // was happily serving (#657).
3091 + 'ai_crawler_rules' => [],
2801 3092 'logo_url' => '',
2802 3093 'favicon_url' => '',
2803 3094 'apple_touch_icon_url' => ''
2804 3095 ];
@@ -3007,16 +3298,17 @@
3007 3298 // Remove extra whitespace
3008 3299 $title = preg_replace('/\s+/', ' ', $title);
3009 3300 $title = trim($title);
3010 3301
3011 - // Ensure title is not too long (60 characters max for SEO)
3012 - if (strlen($title) > 60) {
3013 - // Try to truncate at word boundary
3014 - $title = wp_trim_words($title, 8, '...');
3015 - if (strlen($title) > 60) {
3016 - $title = substr($title, 0, 57) . '...';
3017 - }
3018 - }
3302 + // Ensure title is not too long (60 characters max for SEO).
3303 + // All three units here were wrong for non-Latin text: strlen() counts
3304 + // BYTES so the gate fired at 20 Thai characters, wp_trim_words() counts
3305 + // CHARACTERS on th/ja/zh_* so `8` cut the title to 8 of them, and
3306 + // substr() cuts bytes so it split a character mid-sequence (#687).
3307 + $title = \ThinkRank\Core\Seo_Text::trim_to_length(
3308 + $title,
3309 + \ThinkRank\Core\Seo_Text::TITLE_MAX_LENGTH
3310 + );
3019 3311
3020 3312 // Ensure title is not empty
3021 3313 if (empty($title)) {
3022 3314 $title = get_bloginfo('name');
@@ -3406,8 +3698,29 @@
3406 3698 * @since 1.0.0
3407 3699 * @return array Array of sitemap URLs
3408 3700 */
3409 3701 private function get_sitemap_urls_for_robots(): array {
3702 + // One wrapper over every return path below, including the #104 extras.
3703 + // The Sitemap: line is the only absolute URL of ours in robots.txt and
3704 + // the one a crawler follows to find everything else, so it has to carry
3705 + // the site's scheme preference (#638). Applied here rather than where
3706 + // the body is assembled, because that path also renders a robots.txt a
3707 + // site owner typed themselves, and their text is not ours to rewrite.
3708 + return array_map(
3709 + static function (string $url): string {
3710 + return Url_Scheme::apply($url);
3711 + },
3712 + $this->collect_sitemap_urls_for_robots()
3713 + );
3714 + }
3715 +
3716 + /**
3717 + * The sitemap URLs robots.txt advertises, before the scheme preference.
3718 + *
3719 + * @since 1.0.0
3720 + * @return array Array of sitemap URLs
3721 + */
3722 + private function collect_sitemap_urls_for_robots(): array {
3410 3723 try {
3411 3724 // Get sitemap settings
3412 3725 $sitemap_generator = new \ThinkRank\SEO\Sitemap_Generator();
3413 3726 $sitemap_settings = $sitemap_generator->get_settings('site');
@@ -3456,9 +3769,22 @@
3456 3769 // business sitemap and the sitemaps other plugins register both land
3457 3770 // here for the same reason, so they go through one list (#104).
3458 3771 $extra = [];
3459 3772
3460 - if (file_exists(ABSPATH . 'local-sitemap.xml')) {
3773 + // Not a file test. Under dynamic delivery the local sitemap is
3774 + // served from PHP and no file is ever written, so file_exists()
3775 + // silently dropped a sitemap the site really does publish (#752).
3776 + // On static sites the file is still what proves it, so both count.
3777 + $local_sitemap_published = file_exists(ABSPATH . 'local-sitemap.xml');
3778 +
3779 + if (!$local_sitemap_published && class_exists('ThinkRank\\SEO\\Sitemap_Generator')) {
3780 + $generator = new \ThinkRank\SEO\Sitemap_Generator(false);
3781 +
3782 + $local_sitemap_published = 'dynamic' === $generator->resolve_delivery_mode()
3783 + && $generator->publishes_local_sitemap();
3784 + }
3785 +
3786 + if ($local_sitemap_published) {
3461 3787 $extra[] = '/local-sitemap.xml';
3462 3788 }
3463 3789
3464 3790 foreach (\ThinkRank\SEO\Sitemap_Generator::additional_sitemaps() as $path) {
@@ -3654,8 +3980,116 @@
3654 3980 return $rules;
3655 3981 }
3656 3982
3657 3983 /**
3984 + * Opening fence of the machine-owned AI crawler region.
3985 + *
3986 + * @since 2.5.0
3987 + * @var string
3988 + */
3989 + public const AI_BLOCK_BEGIN = '# BEGIN ThinkRank AI crawlers';
3990 +
3991 + /**
3992 + * Closing fence of the machine-owned AI crawler region.
3993 + *
3994 + * @since 2.5.0
3995 + * @var string
3996 + */
3997 + public const AI_BLOCK_END = '# END ThinkRank AI crawlers';
3998 +
3999 + /**
4000 + * Render the fenced AI crawler region for the current settings.
4001 + *
4002 + * One `User-agent:` / `Disallow: /` record per blocked crawler. Allowed
4003 + * crawlers emit nothing at all: `Disallow:` with an empty value is the
4004 + * robots.txt way of saying "allow everything", but writing eighteen such
4005 + * records to say what silence already says would triple the file and
4006 + * invite the reading that an unlisted crawler is therefore refused.
4007 + *
4008 + * @since 2.5.0
4009 + *
4010 + * @param array $settings Site settings.
4011 + * @return string Fenced block, newline-terminated, or '' when nothing is blocked.
4012 + */
4013 + private function build_ai_crawler_block(array $settings): string {
4014 + $blocked = AI_Crawlers::blocked_slugs($settings['ai_crawler_rules'] ?? []);
4015 +
4016 + if (empty($blocked)) {
4017 + return '';
4018 + }
4019 +
4020 + $agents = AI_Crawlers::all();
4021 +
4022 + $lines = [
4023 + self::AI_BLOCK_BEGIN,
4024 + '# Managed by ThinkRank — edits between these lines are overwritten.',
4025 + ];
4026 +
4027 + foreach ($blocked as $slug) {
4028 + $lines[] = '';
4029 + $lines[] = 'User-agent: ' . $agents[$slug]['token'];
4030 + $lines[] = 'Disallow: /';
4031 + }
4032 +
4033 + $lines[] = self::AI_BLOCK_END;
4034 +
4035 + return implode("\n", $lines) . "\n";
4036 + }
4037 +
4038 + /**
4039 + * Remove the fenced AI crawler region from a robots.txt body.
4040 + *
4041 + * Tolerates a missing closing fence rather than leaving the rest of the
4042 + * file swallowed: a truncated write, or someone deleting the END line by
4043 + * hand, would otherwise make every subsequent read drop everything below
4044 + * the opening fence.
4045 + *
4046 + * @since 2.5.0
4047 + *
4048 + * @param string $body Robots.txt body.
4049 + * @return string Body with the region removed.
4050 + */
4051 + public function strip_ai_crawler_block(string $body): string {
4052 + if (false === strpos($body, self::AI_BLOCK_BEGIN)) {
4053 + return $body;
4054 + }
4055 +
4056 + $pattern = '/\R*' . preg_quote(self::AI_BLOCK_BEGIN, '/')
4057 + . '.*?(?:' . preg_quote(self::AI_BLOCK_END, '/') . '|\z)\R*/s';
4058 +
4059 + return trim((string) preg_replace($pattern, "\n\n", $body, 1));
4060 + }
4061 +
4062 + /**
4063 + * Put the current AI crawler region into a robots.txt body.
4064 + *
4065 + * Replaces an existing region in place so the block keeps its position in
4066 + * a hand-ordered file, and appends when there is none. Everything outside
4067 + * the fences is returned untouched — that is the whole point of fencing
4068 + * it, since the body is also a free-text field the user edits.
4069 + *
4070 + * @since 2.5.0
4071 + *
4072 + * @param string $body Robots.txt body (fences optional).
4073 + * @param array $settings Site settings.
4074 + * @return string Body carrying the current region.
4075 + */
4076 + private function apply_ai_crawler_block(string $body, array $settings): string {
4077 + $stripped = $this->strip_ai_crawler_block($body);
4078 + $block = $this->build_ai_crawler_block($settings);
4079 +
4080 + if ('' === $block) {
4081 + return $stripped;
4082 + }
4083 +
4084 + if ('' === trim($stripped)) {
4085 + return trim($block);
4086 + }
4087 +
4088 + return rtrim($stripped) . "\n\n" . trim($block);
4089 + }
4090 +
4091 + /**
3658 4092 * The auto-generated header prepended to the served robots.txt.
3659 4093 *
3660 4094 * Kept separate from the body so it is only ever added at render time with
3661 4095 * a fresh timestamp, never stored or shown in the editable textarea.
@@ -3692,11 +4126,16 @@
3692 4126 */
3693 4127 public function get_served_robots_body(): string {
3694 4128 $settings = $this->get_settings('site');
3695 4129
4130 + // The AI block is stripped from every one of these paths. A physical
4131 + // robots.txt we wrote carries it, and the stored override is whatever
4132 + // the textarea last held — so without this the block round-trips into
4133 + // the editor, gets saved as ordinary body text, and is then appended
4134 + // to a second time on the next render.
3696 4135 $custom = trim((string) ($settings['robots_txt_content'] ?? ''));
3697 4136 if ($custom !== '') {
3698 - return $this->strip_robots_header($custom);
4137 + return $this->strip_ai_crawler_block($this->strip_robots_header($custom));
3699 4138 }
3700 4139
3701 4140 $robots_file = ABSPATH . 'robots.txt';
3702 4141 if (file_exists($robots_file)) {
@@ -3702,13 +4141,13 @@
3702 4141 if (file_exists($robots_file)) {
3703 4142 // phpcs:ignore WordPress.WP.AlternativeFunctions.file_get_contents_file_get_contents, WordPress.PHP.NoSilencedErrors.Discouraged -- an unreadable robots.txt is an expected state answered with an empty string.
3704 4143 $raw = (string) @file_get_contents($robots_file);
3705 4144 if ($raw !== '') {
3706 - return $this->strip_robots_header($raw);
4145 + return $this->strip_ai_crawler_block($this->strip_robots_header($raw));
3707 4146 }
3708 4147 }
3709 4148
3710 - return trim($this->generate_robots_txt()['content']);
4149 + return $this->strip_ai_crawler_block(trim($this->generate_robots_txt()['content']));
3711 4150 }
3712 4151 private function get_site_identity_data(array $settings): array {
3713 4152 return [
3714 4153 'site_name' => $settings['site_name'] ?? get_bloginfo('name'),
@@ -3770,11 +4209,14 @@
3770 4209 $optimization['validation']['valid'] = false;
3771 4210 }
3772 4211
3773 4212 if (!empty($value) && isset($config['max_length'])) {
3774 - if (strlen($value) > $config['max_length']) {
4213 + // The warning says "characters", so measure and cut in characters:
4214 + // strlen()/substr() fired early on non-Latin values and the
4215 + // suggested replacement was cut mid-character (#687).
4216 + if (mb_strlen($value) > $config['max_length']) {
3775 4217 $optimization['validation']['warnings'][] = "{$element} exceeds maximum length of {$config['max_length']} characters";
3776 - $optimization['optimized_value'] = substr($value, 0, $config['max_length']);
4218 + $optimization['optimized_value'] = \ThinkRank\Core\Seo_Text::trim_to_length($value, (int) $config['max_length']);
3777 4219 }
3778 4220 }
3779 4221
3780 4222 // SEO-specific optimizations
@@ -3813,18 +4255,20 @@
3813 4255 return $optimization;
3814 4256 }
3815 4257
3816 4258 // Check if it's a local image
3817 - $attachment_id = attachment_url_to_postid($value);
4259 + $attachment_id = Attachment_Lookup::id_from_url($value);
3818 4260 if ($attachment_id) {
3819 4261 $image_meta = wp_get_attachment_metadata($attachment_id);
3820 4262
3821 4263 if ($image_meta && isset($image_meta['width'], $image_meta['height'])) {
3822 - // Check recommended size
4264 + // Check recommended size, against the configured file itself
4265 + // rather than the upload it may have been generated from.
3823 4266 if (isset($config['recommended_size'])) {
3824 4267 [$rec_width, $rec_height] = explode('x', $config['recommended_size']);
4268 + $image_file = Attachment_Lookup::describe($attachment_id, $value);
3825 4269
3826 - if ((int) $image_meta['width'] !== (int) $rec_width || (int) $image_meta['height'] !== (int) $rec_height) {
4270 + if ($image_file['width'] !== (int) $rec_width || $image_file['height'] !== (int) $rec_height) {
3827 4271 $optimization['suggestions'][] = "Consider using {$config['recommended_size']} size for optimal {$element}";
3828 4272 }
3829 4273 }
3830 4274