PluginProbe
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO / 2.10.0
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO v2.10.0
2.11.0 2.10.0 2.9.0 2.8.0 2.7.0 2.6.0 2.5.0 2.4.0 2.3.0 2.2.0 2.1.1 2.1.0 2.0.2 2.0.1 2.0.0 1.32.0 1.31.0 1.30.0 1.29.0 1.28.0 1.27.0 1.26.0 1.25.0 trunk 1.0.0 All 52 releases
← All changes | includes/seo/class-site-identity-manager.php +774 -47 2.0.1 → 2.10.0 View file →
@@ -15,8 +15,13 @@
15 15 declare(strict_types=1);
16 16
17 17 namespace ThinkRank\SEO;
18 18
19 +// Prevent direct access
20 +if (!defined('ABSPATH')) {
21 + exit;
22 +}
23 +
19 24 // Ensure dependencies are loaded
20 25 if (!class_exists('ThinkRank\\SEO\\Abstract_SEO_Manager')) {
21 26 require_once THINKRANK_PLUGIN_DIR . 'includes/seo/class-abstract-seo-manager.php';
22 27 }
@@ -35,8 +40,36 @@
35 40 */
36 41 class Site_Identity_Manager extends Abstract_SEO_Manager {
37 42
38 43 /**
44 + * The per-context title formats as ThinkRank ships them.
45 + *
46 + * These are not in get_default_settings(): the admin screen seeds them on
47 + * first save, so on a real install they are stored values, indistinguishable
48 + * from a template the user typed. The migration needs to tell those two
49 + * apart — it may overwrite a shipped default with an imported template, and
50 + * must never overwrite a choice the user made — so this is the record of
51 + * what "untouched" looks like.
52 + *
53 + * Keep in step with getDefaultSettings() in
54 + * src/admin/components/essential-seo/SiteIdentityTab.js. SiteIdentityTitleFormatDefaultsTest
55 + * fails when the two drift.
56 + *
57 + * @since 2.8.0
58 + * @var array<string, string>
59 + */
60 + public const TITLE_FORMAT_DEFAULTS = [
61 + 'homepage_title' => '%site_title% %sep% %site_description%',
62 + 'post_title' => '%post_title% %sep% %site_title%',
63 + 'page_title' => '%page_title% %sep% %site_title%',
64 + 'category_title' => '%category_title% %sep% %site_title%',
65 + 'tag_title' => '%tag_title% %sep% %site_title%',
66 + 'author_title' => '%author_name% %sep% %site_title%',
67 + 'search_title' => 'Search Results for "%search_term%" %sep% %site_title%',
68 + 'archive_title' => '%archive_title% %sep% %site_title%',
69 + ];
70 +
71 + /**
39 72 * WordPress filesystem instance
40 73 *
41 74 * @since 1.0.0
42 75 * @var \WP_Filesystem_Base|null
@@ -240,13 +273,358 @@
240 273 * Constructor
241 274 *
242 275 * @since 1.0.0
243 276 */
277 + /**
278 + * The square derivatives wp_site_icon() asks for.
279 + *
280 + * Core generates these only through its own Site Icon crop flow, so an
281 + * image chosen as a ThinkRank favicon straight from the media library has
282 + * none of them and every sizes="" declaration is a near miss (#571).
283 + *
284 + * @since 2.3.1
285 + * @var int[]
286 + */
287 + public const ICON_SIZES = [32, 180, 192, 270];
288 +
289 + /**
290 + * Transient holding resolved icon URLs, keyed by configured URL and size.
291 + *
292 + * The site-icon filter runs in wp_head on every FRONT-END request, and
293 + * resolving a URL to its attachment costs an uncached postmeta query. The
294 + * mapping only changes when the icon setting does, so it is cached here and
295 + * dropped on save.
296 + *
297 + * @since 2.3.1
298 + * @var string
299 + */
300 + public const ICON_URL_TRANSIENT = 'thinkrank_site_icon_urls';
301 +
302 + /**
303 + * Marker for the one-time derivative backfill on existing installs.
304 + *
305 + * @since 2.3.1
306 + * @var string
307 + */
308 + public const ICON_BACKFILL_OPTION = 'thinkrank_site_icon_sizes_backfilled';
309 +
310 + /**
311 + * Whether the icon-derivative listener has been registered this request.
312 + *
313 + * Static because `thinkrank_seo_settings_saved` is a global hook — one
314 + * listener serves every instance, and this class is constructed on the
315 + * front end as well as in admin.
316 + *
317 + * @since 2.3.1
318 + * @var bool
319 + */
320 + private static bool $icon_sizes_listener_registered = false;
321 +
244 322 public function __construct() {
245 323 parent::__construct('site_identity');
324 +
325 + if (!self::$icon_sizes_listener_registered) {
326 + self::$icon_sizes_listener_registered = true;
327 + add_action('thinkrank_seo_settings_saved', [$this, 'generate_icon_sizes_on_save'], 10, 2);
328 + // Admin only: resizing is not front-end work, and admin traffic is
329 + // enough to run a one-time backfill promptly.
330 + add_action('admin_init', [self::class, 'maybe_backfill_icon_sizes']);
331 + }
246 332 }
247 333
248 334 /**
335 + * Save settings, then refresh what a new canonical scheme invalidates.
336 + *
337 + * The static sitemap files are written with the scheme in force when they
338 + * were built, and nothing else rebuilds them until a post or term changes.
339 + * So a change of scheme left every `<loc>` on the old one while canonical
340 + * and og:url had already moved (#736). Every writer (the settings route,
341 + * the robots route, the MCP abilities, an import) lands here.
342 + *
343 + * @since 2.7.0
344 + *
345 + * @param string $context_type Context type.
346 + * @param int|null $context_id Context ID.
347 + * @param array $settings Settings to save.
348 + * @return bool
349 + */
350 + public function save_settings(string $context_type, ?int $context_id, array $settings): bool {
351 + if (!self::touches_canonical_scheme($context_type, $context_id, $settings)) {
352 + return parent::save_settings($context_type, $context_id, $settings);
353 + }
354 +
355 + $before = Url_Scheme::preference();
356 + $saved = parent::save_settings($context_type, $context_id, $settings);
357 +
358 + if ($saved) {
359 + $this->on_canonical_scheme_saved($before);
360 + }
361 +
362 + return $saved;
363 + }
364 +
365 + /**
366 + * Whether a save can change the site-wide canonical scheme.
367 + *
368 + * @since 2.7.0
369 + *
370 + * @param string $context_type Context type.
371 + * @param int|null $context_id Context ID.
372 + * @param array $settings Settings being saved.
373 + * @return bool
374 + */
375 + public static function touches_canonical_scheme(string $context_type, ?int $context_id, array $settings): bool {
376 + return 'site' === sanitize_key($context_type)
377 + && empty($context_id)
378 + && array_key_exists('canonical_scheme', $settings);
379 + }
380 +
381 + /**
382 + * Rebuild the static sitemaps when the effective scheme changed.
383 + *
384 + * Compares the effective preference, filter included, so a site whose
385 + * scheme is pinned by `thinkrank_canonical_scheme` does not rebuild on a
386 + * stored value that changes nothing it publishes.
387 + *
388 + * @since 2.7.0
389 + *
390 + * @param string $before Effective scheme before the save.
391 + * @return void
392 + */
393 + protected function on_canonical_scheme_saved(string $before): void {
394 + // The preference is cached for the request; the save just changed it.
395 + Url_Scheme::reset();
396 +
397 + if (Url_Scheme::preference() === $before) {
398 + return;
399 + }
400 +
401 + $this->schedule_sitemap_rebuild();
402 + }
403 +
404 + /**
405 + * Queue a settings-driven sitemap rebuild.
406 + *
407 + * Debounced and run after the response, like any other settings change
408 + * that alters what the sitemap publishes.
409 + *
410 + * @since 2.7.0
411 + * @return void
412 + */
413 + protected function schedule_sitemap_rebuild(): void {
414 + (new Sitemap_Generator(false))->schedule_regeneration();
415 + }
416 +
417 + /**
418 + * Build the icon derivatives for a newly chosen favicon.
419 + *
420 + * Runs on save, which is the only moment the choice changes and the only
421 + * place image work belongs — resolving a size on the front end must stay a
422 + * lookup. Failure is silent by design: a missing derivative degrades to the
423 + * next best file, so a site whose host cannot resize still renders an icon.
424 + *
425 + * @since 2.3.1
426 + *
427 + * @param string $manager_type Settings category that was saved.
428 + * @param array $settings The settings that were written.
429 + * @return void
430 + */
431 + public function generate_icon_sizes_on_save(string $manager_type, array $settings): void {
432 + if ('site_identity' !== $manager_type) {
433 + return;
434 + }
435 +
436 + // The choice, or the derivatives behind it, may have just changed.
437 + delete_transient(self::ICON_URL_TRANSIENT);
438 +
439 + foreach (['favicon_url', 'apple_touch_icon_url'] as $key) {
440 + if (empty($settings[$key]) || !is_string($settings[$key])) {
441 + continue;
442 + }
443 +
444 + $attachment_id = self::icon_attachment_id($settings[$key]);
445 +
446 + if ($attachment_id) {
447 + self::ensure_icon_sizes($attachment_id);
448 + }
449 + }
450 +
451 + // Dropped again after the resizes finish. Resizing is not instant, and a
452 + // front-end request arriving mid-generation would otherwise repopulate
453 + // the transient with the pre-derivative URLs and pin them for the full
454 + // TTL — leaving the sizes= declarations untrue until the next save.
455 + delete_transient(self::ICON_URL_TRANSIENT);
456 + }
457 +
458 + /**
459 + * Build the derivatives for a site that configured its icons before this
460 + * existed.
461 + *
462 + * generate_icon_sizes_on_save() only fires on a settings write, so every
463 + * site with an icon already chosen would keep serving whatever
464 + * wp_get_attachment_image_url() could find — in practice the 150x150
465 + * thumbnail behind a sizes="32x32" declaration — until someone happened to
466 + * re-save Site Identity. That is the bug this is meant to fix, so the
467 + * derivatives are built once on upgrade instead of waiting for a save.
468 + *
469 + * Guarded by its own option rather than the plugin version so it runs once
470 + * and stays cheap: the check is a single autoloaded read on requests after
471 + * the first.
472 + *
473 + * @since 2.3.1
474 + *
475 + * @return void
476 + */
477 + public static function maybe_backfill_icon_sizes(): void {
478 + if (get_option(self::ICON_BACKFILL_OPTION)) {
479 + return;
480 + }
481 +
482 + // Written before the work, not after: a host that cannot resize must
483 + // not retry on every admin request forever.
484 + update_option(self::ICON_BACKFILL_OPTION, time(), true);
485 +
486 + $settings = (new self())->get_settings('site');
487 +
488 + if (!is_array($settings)) {
489 + return;
490 + }
491 +
492 + foreach (['favicon_url', 'apple_touch_icon_url'] as $key) {
493 + if (empty($settings[$key]) || !is_string($settings[$key])) {
494 + continue;
495 + }
496 +
497 + $attachment_id = self::icon_attachment_id($settings[$key]);
498 +
499 + if ($attachment_id) {
500 + self::ensure_icon_sizes($attachment_id);
501 + }
502 + }
503 +
504 + delete_transient(self::ICON_URL_TRANSIENT);
505 + }
506 +
507 + /**
508 + * Attachment ID behind a configured icon URL, or 0 when it is not ours.
509 + *
510 + * attachment_url_to_postid() matches _wp_attached_file, which holds the
511 + * ORIGINAL upload path, so the URL of a generated derivative
512 + * (`logo-512.png`) returns 0 — and that is exactly what the media picker
513 + * hands back when the user chooses a size. Strip the dimension suffix and
514 + * try the original once.
515 + *
516 + * Shared with SEO_Manager's site-icon filter so both sides of the feature
517 + * agree on which attachment a configured URL means.
518 + *
519 + * @since 2.3.1
520 + *
521 + * @param string $url Configured icon URL.
522 + * @return int Attachment ID, or 0.
523 + */
524 + public static function icon_attachment_id(string $url): int {
525 + $attachment_id = (int) attachment_url_to_postid($url);
526 +
527 + if ($attachment_id) {
528 + return $attachment_id;
529 + }
530 +
531 + $original = preg_replace('/-\d+x\d+(?=\.[a-zA-Z0-9]+$)/', '', $url);
532 +
533 + if (is_string($original) && $original !== $url) {
534 + return (int) attachment_url_to_postid($original);
535 + }
536 +
537 + return 0;
538 + }
539 +
540 + /**
541 + * Which ICON_SIZES derivatives this attachment still needs.
542 + *
543 + * Split out from the generation so the decision can be asserted on its
544 + * own: whether a size is skipped because it already exists or because it
545 + * would upscale is invisible once both answers are "nothing was built".
546 + *
547 + * A source is measured by its SHORTER edge — a 400x40 banner cannot yield
548 + * a true 192x192 — and anything reporting no dimensions at all (SVGs) is
549 + * left alone.
550 + *
551 + * @since 2.3.1
552 + *
553 + * @param array $meta Attachment metadata.
554 + * @return array<string, array{width: int, height: int, crop: bool}> Sizes to build.
555 + */
556 + public static function missing_icon_sizes(array $meta): array {
557 + $source = min((int) ($meta['width'] ?? 0), (int) ($meta['height'] ?? 0));
558 +
559 + if ($source < 1) {
560 + return [];
561 + }
562 +
563 + $wanted = [];
564 + foreach (self::ICON_SIZES as $size) {
565 + // Never upscale: a stretched source behind an accurate sizes=""
566 + // label is worse than the honest near miss it would replace.
567 + if (isset($meta['sizes']["site_icon-{$size}"]) || $size > $source) {
568 + continue;
569 + }
570 +
571 + $wanted["site_icon-{$size}"] = ['width' => $size, 'height' => $size, 'crop' => true];
572 + }
573 +
574 + return $wanted;
575 + }
576 +
577 + /**
578 + * Generate whatever ICON_SIZES derivatives this attachment is missing.
579 + *
580 + * Only the missing ones, and never one larger than the source: upscaling a
581 + * small favicon would put a blurrier file behind an accurate sizes="" label
582 + * than the honest near-miss it replaced.
583 + *
584 + * @since 2.3.1
585 + *
586 + * @param int $attachment_id Attachment to build derivatives for.
587 + * @return string[] Size names generated, empty when there was nothing to do.
588 + */
589 + public static function ensure_icon_sizes(int $attachment_id): array {
590 + $meta = wp_get_attachment_metadata($attachment_id);
591 +
592 + if (!is_array($meta)) {
593 + return [];
594 + }
595 +
596 + $wanted = self::missing_icon_sizes($meta);
597 +
598 + if (empty($wanted)) {
599 + return [];
600 + }
601 +
602 + $file = get_attached_file($attachment_id);
603 +
604 + if (!$file || !file_exists($file)) {
605 + return [];
606 + }
607 +
608 + $editor = wp_get_image_editor($file);
609 +
610 + if (is_wp_error($editor)) {
611 + return [];
612 + }
613 +
614 + $generated = $editor->multi_resize($wanted);
615 +
616 + if (empty($generated)) {
617 + return [];
618 + }
619 +
620 + $meta['sizes'] = array_merge($meta['sizes'] ?? [], $generated);
621 + wp_update_attachment_metadata($attachment_id, $meta);
622 +
623 + return array_keys($generated);
624 + }
625 +
626 + /**
249 627 * Initialize WordPress filesystem
250 628 *
251 629 * @since 1.0.0
252 630 * @return bool True if filesystem is initialized, false otherwise
@@ -442,8 +820,17 @@
442 820 $body = ($custom !== '' && !$fully_blocked)
443 821 ? $this->strip_robots_header($custom)
444 822 : trim($this->generate_robots_txt()['content']);
445 823
824 + // The per-agent AI directives are machine-owned, so they are composed
825 + // here rather than stored: the textarea holds the user's body, with
826 + // the fenced block stripped out of every read and re-applied on every
827 + // render. A site-wide block already disallows everyone, so adding the
828 + // per-agent group there would be noise restating the same refusal.
829 + if (!$fully_blocked) {
830 + $body = $this->apply_ai_crawler_block($body, $settings);
831 + }
832 +
446 833 if ($body === '') {
447 834 return '';
448 835 }
449 836
@@ -519,9 +906,12 @@
519 906 $settings = $this->get_settings('site');
520 907
521 908 // Compare bodies, not raw strings: the auto-generated header carries a
522 909 // regeneration timestamp that always differs and means nothing here.
523 - $served = $this->strip_robots_header($effective['content']);
910 + // The AI crawler block is composed at render time on both sides, so it
911 + // is identical by construction and comparing it would only ever report
912 + // a false drift the admin cannot act on.
913 + $served = $this->strip_ai_crawler_block($this->strip_robots_header($effective['content']));
524 914
525 915 // Measure against the body the editor is displaying — get_served_robots_body()
526 916 // — not against render_robots_txt(). Two things made the old comparison
527 917 // report "in sync" while the screen showed rules no crawler receives:
@@ -899,17 +1289,19 @@
899 1289 $home_text = $settings['breadcrumb_home_text'] ?? 'Home';
900 1290 if (empty($home_text)) {
901 1291 $optimization['warnings'][] = 'Empty home text reduces accessibility for screen readers';
902 1292 $optimization['score'] -= 15;
903 - } elseif (strlen($home_text) > 20) {
904 - $optimization['suggestions'][] = 'Keep home text concise (current: ' . strlen($home_text) . ' chars)';
1293 + } elseif (mb_strlen($home_text) > 20) {
1294 + // mb_strlen: this number is shown to the user as "chars" (#687).
1295 + $optimization['suggestions'][] = 'Keep home text concise (current: ' . mb_strlen($home_text) . ' chars)';
905 1296 $optimization['score'] -= 5;
906 1297 }
907 1298
908 1299 // Check prefix usage
909 1300 $prefix = $settings['breadcrumb_prefix'] ?? '';
910 - if (!empty($prefix) && strlen($prefix) > 50) {
911 - $optimization['suggestions'][] = 'Breadcrumb prefix is quite long (' . strlen($prefix) . ' chars) - consider shortening';
1301 + if (!empty($prefix) && mb_strlen($prefix) > 50) {
1302 + // mb_strlen: this number is shown to the user as "chars" (#687).
1303 + $optimization['suggestions'][] = 'Breadcrumb prefix is quite long (' . mb_strlen($prefix) . ' chars) - consider shortening';
912 1304 $optimization['score'] -= 5;
913 1305 }
914 1306
915 1307 // Current page display
@@ -1149,11 +1541,12 @@
1149 1541 $optimization['score'] -= 15;
1150 1542 }
1151 1543 }
1152 1544
1153 - // Business type validation
1154 - if (empty($settings['business_type'])) {
1155 - $optimization['suggestions'][] = 'Select a specific business type for better schema markup';
1545 + // Business type validation (shared rule, one message — #622).
1546 + $business_type = $this->business_type_status($settings);
1547 + if ('suggestion' === $business_type['status']) {
1548 + $optimization['suggestions'][] = $business_type['message'];
1156 1549 $optimization['score'] -= 5;
1157 1550 }
1158 1551
1159 1552 // Email validation
@@ -1683,24 +2076,17 @@
1683 2076 'icon' => '✗'
1684 2077 ];
1685 2078 }
1686 2079
1687 - // Business Type validation
1688 - if (!empty($settings['business_type']) && $settings['business_type'] !== 'LocalBusiness') {
1689 - $field_details[] = [
1690 - 'field' => 'business_type',
1691 - 'label' => 'Business type is selected for proper schema markup.',
1692 - 'status' => 'valid',
1693 - 'icon' => '✓'
1694 - ];
1695 - } else {
1696 - $field_details[] = [
1697 - 'field' => 'business_type',
1698 - 'label' => 'Specific business type selection recommended for better schema markup.',
1699 - 'status' => 'suggestion',
1700 - 'icon' => '⚠'
1701 - ];
1702 - }
2080 + // Business Type validation — see business_type_status() for why there
2081 + // is exactly one rule here now (#622).
2082 + $business_type = $this->business_type_status($settings);
2083 + $field_details[] = [
2084 + 'field' => 'business_type',
2085 + 'label' => $business_type['message'],
2086 + 'status' => $business_type['status'],
2087 + 'icon' => 'valid' === $business_type['status'] ? '✓' : '⚠',
2088 + ];
1703 2089
1704 2090 // Address validation (NAP consistency)
1705 2091 $address_fields = ['business_address', 'business_city', 'business_state', 'business_country'];
1706 2092 $address_complete = true;
@@ -2384,11 +2770,15 @@
2384 2770 } else {
2385 2771 $validation['suggestions'][] = 'Add business hours to improve local search visibility';
2386 2772 }
2387 2773
2388 - // Validate business type
2389 - if (empty($settings['business_type'])) {
2390 - $validation['suggestions'][] = 'Select a specific business type for better schema markup';
2774 + // Business type, through the shared rule (#622). This is the only place
2775 + // it is reported on the generic path: validate_settings() with no tab
2776 + // context attaches basic-info field details, not business-info ones, so
2777 + // without this the setting would go unreported there entirely.
2778 + $business_type = $this->business_type_status($settings);
2779 + if ('suggestion' === $business_type['status']) {
2780 + $validation['suggestions'][] = $business_type['message'];
2391 2781 }
2392 2782
2393 2783 return $validation;
2394 2784 }
@@ -2480,8 +2870,47 @@
2480 2870 * @since 2.0.1
2481 2871 *
2482 2872 * @return string[]
2483 2873 */
2874 + /**
2875 + * The stored alternate name(s), shaped for schema output.
2876 + *
2877 + * schema.org and Google both allow `alternateName` to carry one value or
2878 + * several, and the store already round-trips either shape, so this accepts
2879 + * both and normalises: null when there is nothing to publish, a bare string
2880 + * for one name, a list for more. Emitting a one-element array would be
2881 + * valid but noisier than it needs to be.
2882 + *
2883 + * Shared because both WebSite producers need it and must agree — a property
2884 + * added to one and not the other is how #688 happened.
2885 + *
2886 + * @since 2.7.0
2887 + *
2888 + * @param mixed $value Stored alternate_name value.
2889 + * @return string|string[]|null
2890 + */
2891 + public static function alternate_name_for_schema($value) {
2892 + $names = [];
2893 +
2894 + foreach ((array) $value as $name) {
2895 + if (!is_scalar($name)) {
2896 + continue;
2897 + }
2898 +
2899 + $name = trim((string) $name);
2900 +
2901 + if ('' !== $name && !in_array($name, $names, true)) {
2902 + $names[] = $name;
2903 + }
2904 + }
2905 +
2906 + if (empty($names)) {
2907 + return null;
2908 + }
2909 +
2910 + return 1 === count($names) ? $names[0] : $names;
2911 + }
2912 +
2484 2913 protected function additional_setting_keys(): array {
2485 2914 return [
2486 2915 // Title formats, one per context.
2487 2916 'homepage_title', 'post_title', 'page_title', 'category_title',
@@ -2486,9 +2915,9 @@
2486 2915 // Title formats, one per context.
2487 2916 'homepage_title', 'post_title', 'page_title', 'category_title',
2488 2917 'tag_title', 'author_title', 'search_title', 'archive_title',
2489 2918 // Breadcrumbs.
2490 - 'breadcrumb_prefix', 'show_current_page',
2919 + 'breadcrumb_prefix', 'show_current_page', 'breadcrumb_use_seo_title',
2491 2920 // Identity, as written by the setup wizard and the importers.
2492 2921 'alternate_name', 'identity_type', 'represents',
2493 2922 'default_meta_description', 'default_social_image',
2494 2923 'social_media_accounts',
@@ -2495,8 +2924,10 @@
2495 2924 // Schema toggles that live on this screen.
2496 2925 'organization_schema', 'knowledge_graph',
2497 2926 // Robots rules composed by the Robots.txt panel.
2498 2927 'custom_robots_rules',
2928 + // Per-agent AI crawler allow/block map (#657).
2929 + 'ai_crawler_rules',
2499 2930 // Hero section.
2500 2931 'hero_title', 'hero_subtitle', 'hero_cta_text', 'hero_cta_url',
2501 2932 'hero_background_image',
2502 2933 // Local SEO / business details.
@@ -2508,8 +2939,118 @@
2508 2939 ];
2509 2940 }
2510 2941
2511 2942 /**
2943 + * Sanitize settings, normalising the AI crawler rule map.
2944 + *
2945 + * The generic array sanitizer keeps the shape but says nothing about the
2946 + * values: a payload could store `ai_crawler_rules[gptbot] = "maybe"`, or a
2947 + * slug no crawler answers to, and both would round-trip through every
2948 + * later response. Normalising here rather than in the REST handler puts it
2949 + * on the one path every writer shares — the settings route, the robots
2950 + * route and the MCP abilities all land in save_settings() (#657).
2951 + *
2952 + * @since 2.5.0
2953 + *
2954 + * @param array $settings Settings to sanitize.
2955 + * @param string $context_type Context type.
2956 + * @return array Sanitized settings.
2957 + */
2958 + protected function sanitize_settings(array $settings, string $context_type = 'site'): array {
2959 + $sanitized = parent::sanitize_settings($settings, $context_type);
2960 +
2961 + if (array_key_exists('ai_crawler_rules', $sanitized)) {
2962 + $sanitized['ai_crawler_rules'] = AI_Crawlers::normalize_rules($sanitized['ai_crawler_rules']);
2963 + }
2964 +
2965 + // Same reasoning one key up, for the scheme override (#638). Anything
2966 + // that is not one of the three modes means "follow WordPress", and is
2967 + // stored as that rather than kept verbatim — otherwise get-site-identity
2968 + // -settings would report a scheme the site does not actually publish.
2969 + if (array_key_exists('canonical_scheme', $sanitized)) {
2970 + $sanitized['canonical_scheme'] = in_array($sanitized['canonical_scheme'], Url_Scheme::MODES, true)
2971 + ? $sanitized['canonical_scheme']
2972 + : Url_Scheme::AUTOMATIC;
2973 + }
2974 +
2975 + // Same reasoning again for the business type. It goes straight into
2976 + // LocalBusiness schema, so a type that is not in the schema.org
2977 + // vocabulary is invalid structured data — and storing it verbatim would
2978 + // have get-site-identity-settings report a type the site cannot
2979 + // actually publish. An empty value keeps meaning "not set"; anything
2980 + // else unrecognised falls back to the general-purpose root (#623).
2981 + if (array_key_exists('business_type', $sanitized)) {
2982 + $type = (string) $sanitized['business_type'];
2983 +
2984 + if ('' !== $type && !\ThinkRank\Config\Local_Business_Types_Config::is_valid($type)) {
2985 + $type = \ThinkRank\Config\Local_Business_Types_Config::ROOT;
2986 + }
2987 +
2988 + $sanitized['business_type'] = $type;
2989 + }
2990 +
2991 + return $sanitized;
2992 + }
2993 +
2994 + /**
2995 + * schema.org's general-purpose LocalBusiness type.
2996 + *
2997 + * The default, the first option in the control, and a valid answer in its
2998 + * own right — which is the whole point of #622.
2999 + *
3000 + * @since 2.10.0
3001 + * @var string
3002 + */
3003 + private const GENERAL_BUSINESS_TYPE = 'LocalBusiness';
3004 +
3005 + /**
3006 + * The one rule for whether a business type needs the user's attention.
3007 + *
3008 + * There were three, with two wordings and two different conditions. Two
3009 + * fired when the value was empty; the third fired when it WAS
3010 + * `LocalBusiness` — which is the default, the first option in the control
3011 + * and a perfectly valid schema.org type. So the warning appeared out of the
3012 + * box for every site, could not be cleared without choosing a type that
3013 + * might be inaccurate, and on an empty value it appeared three times in two
3014 + * different phrasings, which is why it was reported as showing twice (#622).
3015 + *
3016 + * The rule now: a type is expected, and any type in the vocabulary is a
3017 + * correct answer. Only an unset value is worth prompting about.
3018 + * `LocalBusiness` is the general-purpose answer and is accepted as one —
3019 + * with a note that a more specific type sharpens the schema, phrased as the
3020 + * guidance it is rather than as a fault the user has to clear.
3021 + *
3022 + * @since 2.10.0
3023 + *
3024 + * @param array $settings Site identity settings.
3025 + * @return array{status:string,message:string} `valid` or `suggestion`.
3026 + */
3027 + private function business_type_status(array $settings): array {
3028 + $type = trim((string) ($settings['business_type'] ?? ''));
3029 +
3030 + if ('' === $type) {
3031 + return [
3032 + 'status' => 'suggestion',
3033 + 'message' => __('Select a business type so your local schema describes the right kind of business.', 'thinkrank'),
3034 + ];
3035 + }
3036 +
3037 + // The literal rather than a constant from the expanded type list (#623):
3038 + // that lands on its own branch, and this fix must not wait on it.
3039 + if (self::GENERAL_BUSINESS_TYPE === $type) {
3040 + return [
3041 + 'status' => 'valid',
3042 + 'message' => __('Business type is set to Local Business. A more specific type sharpens your schema, if one fits.', 'thinkrank'),
3043 + ];
3044 + }
3045 +
3046 + return [
3047 + 'status' => 'valid',
3048 + 'message' => __('Business type is selected for proper schema markup.', 'thinkrank'),
3049 + ];
3050 + }
3051 +
3052 + /**
2512 3053 * Get default settings for a context type (implements interface)
2513 3054 *
2514 3055 * @since 1.0.0
2515 3056 *
@@ -2529,9 +3070,32 @@
2529 3070 'breadcrumb_home_text' => 'Home',
2530 3071 'breadcrumb_separator' => '>',
2531 3072 'robots_txt_enabled' => true,
2532 3073 'allow_search_engines' => true,
3074 + // Answer 404 when a content selector in the URL resolved to
3075 + // nothing (#634). On by default, unlike the other new settings
3076 + // here: it changes no URL a visitor or a correct crawler uses, only
3077 + // ones where WordPress resolved nothing and served the blog listing
3078 + // at 200 anyway.
3079 + 'query_protection' => true,
3080 +
3081 + // Feed controls (#635). All three off, so an upgrade changes
3082 + // nothing about what an existing site already sends its
3083 + // subscribers; a brand-new install is seeded with the signature and
3084 + // the noindex on, in Activator::seed_feed_defaults().
3085 + 'feed_excerpt_only' => false,
3086 + 'feed_source_link' => false,
3087 + 'feed_noindex' => false,
3088 +
3089 + // The scheme self-referential URLs go out with (#638). 'automatic'
3090 + // means substitute nothing and follow WordPress, which is what
3091 + // every site did before the setting existed.
3092 + 'canonical_scheme' => Url_Scheme::AUTOMATIC,
2533 3093 'robots_txt_content' => '',
3094 + // Empty map = every AI crawler allowed. Defaults must stay
3095 + // permissive so an upgrade never starts blocking a crawler a site
3096 + // was happily serving (#657).
3097 + 'ai_crawler_rules' => [],
2534 3098 'logo_url' => '',
2535 3099 'favicon_url' => '',
2536 3100 'apple_touch_icon_url' => ''
2537 3101 ];
@@ -2740,16 +3304,17 @@
2740 3304 // Remove extra whitespace
2741 3305 $title = preg_replace('/\s+/', ' ', $title);
2742 3306 $title = trim($title);
2743 3307
2744 - // Ensure title is not too long (60 characters max for SEO)
2745 - if (strlen($title) > 60) {
2746 - // Try to truncate at word boundary
2747 - $title = wp_trim_words($title, 8, '...');
2748 - if (strlen($title) > 60) {
2749 - $title = substr($title, 0, 57) . '...';
2750 - }
2751 - }
3308 + // Ensure title is not too long (60 characters max for SEO).
3309 + // All three units here were wrong for non-Latin text: strlen() counts
3310 + // BYTES so the gate fired at 20 Thai characters, wp_trim_words() counts
3311 + // CHARACTERS on th/ja/zh_* so `8` cut the title to 8 of them, and
3312 + // substr() cuts bytes so it split a character mid-sequence (#687).
3313 + $title = \ThinkRank\Core\Seo_Text::trim_to_length(
3314 + $title,
3315 + \ThinkRank\Core\Seo_Text::TITLE_MAX_LENGTH
3316 + );
2752 3317
2753 3318 // Ensure title is not empty
2754 3319 if (empty($title)) {
2755 3320 $title = get_bloginfo('name');
@@ -3139,8 +3704,29 @@
3139 3704 * @since 1.0.0
3140 3705 * @return array Array of sitemap URLs
3141 3706 */
3142 3707 private function get_sitemap_urls_for_robots(): array {
3708 + // One wrapper over every return path below, including the #104 extras.
3709 + // The Sitemap: line is the only absolute URL of ours in robots.txt and
3710 + // the one a crawler follows to find everything else, so it has to carry
3711 + // the site's scheme preference (#638). Applied here rather than where
3712 + // the body is assembled, because that path also renders a robots.txt a
3713 + // site owner typed themselves, and their text is not ours to rewrite.
3714 + return array_map(
3715 + static function (string $url): string {
3716 + return Url_Scheme::apply($url);
3717 + },
3718 + $this->collect_sitemap_urls_for_robots()
3719 + );
3720 + }
3721 +
3722 + /**
3723 + * The sitemap URLs robots.txt advertises, before the scheme preference.
3724 + *
3725 + * @since 1.0.0
3726 + * @return array Array of sitemap URLs
3727 + */
3728 + private function collect_sitemap_urls_for_robots(): array {
3143 3729 try {
3144 3730 // Get sitemap settings
3145 3731 $sitemap_generator = new \ThinkRank\SEO\Sitemap_Generator();
3146 3732 $sitemap_settings = $sitemap_generator->get_settings('site');
@@ -3183,14 +3769,39 @@
3183 3769 if (empty($sitemap_urls)) {
3184 3770 $sitemap_urls[] = home_url('/sitemap.xml');
3185 3771 }
3186 3772
3187 - // No index on this install, so the local business sitemap has no
3188 - // other discovery path — advertise it directly.
3189 - if (file_exists(ABSPATH . 'local-sitemap.xml')) {
3190 - $local_url = home_url('/local-sitemap.xml');
3191 - if (!in_array($local_url, $sitemap_urls, true)) {
3192 - $sitemap_urls[] = $local_url;
3773 + // No index on this install, so anything not already listed above has
3774 + // no other discovery path — advertise it directly. The local
3775 + // business sitemap and the sitemaps other plugins register both land
3776 + // here for the same reason, so they go through one list (#104).
3777 + $extra = [];
3778 +
3779 + // Not a file test. Under dynamic delivery the local sitemap is
3780 + // served from PHP and no file is ever written, so file_exists()
3781 + // silently dropped a sitemap the site really does publish (#752).
3782 + // On static sites the file is still what proves it, so both count.
3783 + $local_sitemap_published = file_exists(ABSPATH . 'local-sitemap.xml');
3784 +
3785 + if (!$local_sitemap_published && class_exists('ThinkRank\\SEO\\Sitemap_Generator')) {
3786 + $generator = new \ThinkRank\SEO\Sitemap_Generator(false);
3787 +
3788 + $local_sitemap_published = 'dynamic' === $generator->resolve_delivery_mode()
3789 + && $generator->publishes_local_sitemap();
3790 + }
3791 +
3792 + if ($local_sitemap_published) {
3793 + $extra[] = '/local-sitemap.xml';
3794 + }
3795 +
3796 + foreach (\ThinkRank\SEO\Sitemap_Generator::additional_sitemaps() as $path) {
3797 + $extra[] = $path;
3798 + }
3799 +
3800 + foreach ($extra as $path) {
3801 + $url = home_url($path);
3802 + if (!in_array($url, $sitemap_urls, true)) {
3803 + $sitemap_urls[] = $url;
3193 3804 }
3194 3805 }
3195 3806
3196 3807 return $sitemap_urls;
@@ -3375,8 +3986,116 @@
3375 3986 return $rules;
3376 3987 }
3377 3988
3378 3989 /**
3990 + * Opening fence of the machine-owned AI crawler region.
3991 + *
3992 + * @since 2.5.0
3993 + * @var string
3994 + */
3995 + public const AI_BLOCK_BEGIN = '# BEGIN ThinkRank AI crawlers';
3996 +
3997 + /**
3998 + * Closing fence of the machine-owned AI crawler region.
3999 + *
4000 + * @since 2.5.0
4001 + * @var string
4002 + */
4003 + public const AI_BLOCK_END = '# END ThinkRank AI crawlers';
4004 +
4005 + /**
4006 + * Render the fenced AI crawler region for the current settings.
4007 + *
4008 + * One `User-agent:` / `Disallow: /` record per blocked crawler. Allowed
4009 + * crawlers emit nothing at all: `Disallow:` with an empty value is the
4010 + * robots.txt way of saying "allow everything", but writing eighteen such
4011 + * records to say what silence already says would triple the file and
4012 + * invite the reading that an unlisted crawler is therefore refused.
4013 + *
4014 + * @since 2.5.0
4015 + *
4016 + * @param array $settings Site settings.
4017 + * @return string Fenced block, newline-terminated, or '' when nothing is blocked.
4018 + */
4019 + private function build_ai_crawler_block(array $settings): string {
4020 + $blocked = AI_Crawlers::blocked_slugs($settings['ai_crawler_rules'] ?? []);
4021 +
4022 + if (empty($blocked)) {
4023 + return '';
4024 + }
4025 +
4026 + $agents = AI_Crawlers::all();
4027 +
4028 + $lines = [
4029 + self::AI_BLOCK_BEGIN,
4030 + '# Managed by ThinkRank — edits between these lines are overwritten.',
4031 + ];
4032 +
4033 + foreach ($blocked as $slug) {
4034 + $lines[] = '';
4035 + $lines[] = 'User-agent: ' . $agents[$slug]['token'];
4036 + $lines[] = 'Disallow: /';
4037 + }
4038 +
4039 + $lines[] = self::AI_BLOCK_END;
4040 +
4041 + return implode("\n", $lines) . "\n";
4042 + }
4043 +
4044 + /**
4045 + * Remove the fenced AI crawler region from a robots.txt body.
4046 + *
4047 + * Tolerates a missing closing fence rather than leaving the rest of the
4048 + * file swallowed: a truncated write, or someone deleting the END line by
4049 + * hand, would otherwise make every subsequent read drop everything below
4050 + * the opening fence.
4051 + *
4052 + * @since 2.5.0
4053 + *
4054 + * @param string $body Robots.txt body.
4055 + * @return string Body with the region removed.
4056 + */
4057 + public function strip_ai_crawler_block(string $body): string {
4058 + if (false === strpos($body, self::AI_BLOCK_BEGIN)) {
4059 + return $body;
4060 + }
4061 +
4062 + $pattern = '/\R*' . preg_quote(self::AI_BLOCK_BEGIN, '/')
4063 + . '.*?(?:' . preg_quote(self::AI_BLOCK_END, '/') . '|\z)\R*/s';
4064 +
4065 + return trim((string) preg_replace($pattern, "\n\n", $body, 1));
4066 + }
4067 +
4068 + /**
4069 + * Put the current AI crawler region into a robots.txt body.
4070 + *
4071 + * Replaces an existing region in place so the block keeps its position in
4072 + * a hand-ordered file, and appends when there is none. Everything outside
4073 + * the fences is returned untouched — that is the whole point of fencing
4074 + * it, since the body is also a free-text field the user edits.
4075 + *
4076 + * @since 2.5.0
4077 + *
4078 + * @param string $body Robots.txt body (fences optional).
4079 + * @param array $settings Site settings.
4080 + * @return string Body carrying the current region.
4081 + */
4082 + private function apply_ai_crawler_block(string $body, array $settings): string {
4083 + $stripped = $this->strip_ai_crawler_block($body);
4084 + $block = $this->build_ai_crawler_block($settings);
4085 +
4086 + if ('' === $block) {
4087 + return $stripped;
4088 + }
4089 +
4090 + if ('' === trim($stripped)) {
4091 + return trim($block);
4092 + }
4093 +
4094 + return rtrim($stripped) . "\n\n" . trim($block);
4095 + }
4096 +
4097 + /**
3379 4098 * The auto-generated header prepended to the served robots.txt.
3380 4099 *
3381 4100 * Kept separate from the body so it is only ever added at render time with
3382 4101 * a fresh timestamp, never stored or shown in the editable textarea.
@@ -3413,11 +4132,16 @@
3413 4132 */
3414 4133 public function get_served_robots_body(): string {
3415 4134 $settings = $this->get_settings('site');
3416 4135
4136 + // The AI block is stripped from every one of these paths. A physical
4137 + // robots.txt we wrote carries it, and the stored override is whatever
4138 + // the textarea last held — so without this the block round-trips into
4139 + // the editor, gets saved as ordinary body text, and is then appended
4140 + // to a second time on the next render.
3417 4141 $custom = trim((string) ($settings['robots_txt_content'] ?? ''));
3418 4142 if ($custom !== '') {
3419 - return $this->strip_robots_header($custom);
4143 + return $this->strip_ai_crawler_block($this->strip_robots_header($custom));
3420 4144 }
3421 4145
3422 4146 $robots_file = ABSPATH . 'robots.txt';
3423 4147 if (file_exists($robots_file)) {
@@ -3423,13 +4147,13 @@
3423 4147 if (file_exists($robots_file)) {
3424 4148 // phpcs:ignore WordPress.WP.AlternativeFunctions.file_get_contents_file_get_contents, WordPress.PHP.NoSilencedErrors.Discouraged -- an unreadable robots.txt is an expected state answered with an empty string.
3425 4149 $raw = (string) @file_get_contents($robots_file);
3426 4150 if ($raw !== '') {
3427 - return $this->strip_robots_header($raw);
4151 + return $this->strip_ai_crawler_block($this->strip_robots_header($raw));
3428 4152 }
3429 4153 }
3430 4154
3431 - return trim($this->generate_robots_txt()['content']);
4155 + return $this->strip_ai_crawler_block(trim($this->generate_robots_txt()['content']));
3432 4156 }
3433 4157 private function get_site_identity_data(array $settings): array {
3434 4158 return [
3435 4159 'site_name' => $settings['site_name'] ?? get_bloginfo('name'),
@@ -3491,11 +4215,14 @@
3491 4215 $optimization['validation']['valid'] = false;
3492 4216 }
3493 4217
3494 4218 if (!empty($value) && isset($config['max_length'])) {
3495 - if (strlen($value) > $config['max_length']) {
4219 + // The warning says "characters", so measure and cut in characters:
4220 + // strlen()/substr() fired early on non-Latin values and the
4221 + // suggested replacement was cut mid-character (#687).
4222 + if (mb_strlen($value) > $config['max_length']) {
3496 4223 $optimization['validation']['warnings'][] = "{$element} exceeds maximum length of {$config['max_length']} characters";
3497 - $optimization['optimized_value'] = substr($value, 0, $config['max_length']);
4224 + $optimization['optimized_value'] = \ThinkRank\Core\Seo_Text::trim_to_length($value, (int) $config['max_length']);
3498 4225 }
3499 4226 }
3500 4227
3501 4228 // SEO-specific optimizations