PluginProbe
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO / 2.13.0
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO v2.13.0
2.13.0 2.12.0 2.11.0 2.10.0 2.9.0 2.8.0 2.7.0 2.6.0 2.5.0 2.4.0 2.3.0 2.2.0 2.1.1 2.1.0 2.0.2 2.0.1 2.0.0 1.32.0 1.31.0 1.30.0 1.29.0 1.28.0 1.27.0 1.26.0 1.25.0 All 54 releases
← All changes | includes/seo/class-site-identity-manager.php +970 -61 1.26.0 → 2.13.0 View file →
@@ -15,8 +15,13 @@
15 15 declare(strict_types=1);
16 16
17 17 namespace ThinkRank\SEO;
18 18
19 +// Prevent direct access
20 +if (!defined('ABSPATH')) {
21 + exit;
22 +}
23 +
19 24 // Ensure dependencies are loaded
20 25 if (!class_exists('ThinkRank\\SEO\\Abstract_SEO_Manager')) {
21 26 require_once THINKRANK_PLUGIN_DIR . 'includes/seo/class-abstract-seo-manager.php';
22 27 }
@@ -35,8 +40,36 @@
35 40 */
36 41 class Site_Identity_Manager extends Abstract_SEO_Manager {
37 42
38 43 /**
44 + * The per-context title formats as ThinkRank ships them.
45 + *
46 + * These are not in get_default_settings(): the admin screen seeds them on
47 + * first save, so on a real install they are stored values, indistinguishable
48 + * from a template the user typed. The migration needs to tell those two
49 + * apart — it may overwrite a shipped default with an imported template, and
50 + * must never overwrite a choice the user made — so this is the record of
51 + * what "untouched" looks like.
52 + *
53 + * Keep in step with getDefaultSettings() in
54 + * src/admin/components/essential-seo/SiteIdentityTab.js. SiteIdentityTitleFormatDefaultsTest
55 + * fails when the two drift.
56 + *
57 + * @since 2.8.0
58 + * @var array<string, string>
59 + */
60 + public const TITLE_FORMAT_DEFAULTS = [
61 + 'homepage_title' => '%site_title% %sep% %site_description%',
62 + 'post_title' => '%post_title% %sep% %site_title%',
63 + 'page_title' => '%page_title% %sep% %site_title%',
64 + 'category_title' => '%category_title% %sep% %site_title%',
65 + 'tag_title' => '%tag_title% %sep% %site_title%',
66 + 'author_title' => '%author_name% %sep% %site_title%',
67 + 'search_title' => 'Search Results for "%search_term%" %sep% %site_title%',
68 + 'archive_title' => '%archive_title% %sep% %site_title%',
69 + ];
70 +
71 + /**
39 72 * WordPress filesystem instance
40 73 *
41 74 * @since 1.0.0
42 75 * @var \WP_Filesystem_Base|null
@@ -240,13 +273,347 @@
240 273 * Constructor
241 274 *
242 275 * @since 1.0.0
243 276 */
277 + /**
278 + * The square derivatives wp_site_icon() asks for.
279 + *
280 + * Core generates these only through its own Site Icon crop flow, so an
281 + * image chosen as a ThinkRank favicon straight from the media library has
282 + * none of them and every sizes="" declaration is a near miss (#571).
283 + *
284 + * @since 2.3.1
285 + * @var int[]
286 + */
287 + public const ICON_SIZES = [32, 180, 192, 270];
288 +
289 + /**
290 + * Transient holding resolved icon URLs, keyed by configured URL and size.
291 + *
292 + * The site-icon filter runs in wp_head on every FRONT-END request, and
293 + * resolving a URL to its attachment costs an uncached postmeta query. The
294 + * mapping only changes when the icon setting does, so it is cached here and
295 + * dropped on save.
296 + *
297 + * @since 2.3.1
298 + * @var string
299 + */
300 + public const ICON_URL_TRANSIENT = 'thinkrank_site_icon_urls';
301 +
302 + /**
303 + * Marker for the one-time derivative backfill on existing installs.
304 + *
305 + * @since 2.3.1
306 + * @var string
307 + */
308 + public const ICON_BACKFILL_OPTION = 'thinkrank_site_icon_sizes_backfilled';
309 +
310 + /**
311 + * Whether the icon-derivative listener has been registered this request.
312 + *
313 + * Static because `thinkrank_seo_settings_saved` is a global hook — one
314 + * listener serves every instance, and this class is constructed on the
315 + * front end as well as in admin.
316 + *
317 + * @since 2.3.1
318 + * @var bool
319 + */
320 + private static bool $icon_sizes_listener_registered = false;
321 +
244 322 public function __construct() {
245 323 parent::__construct('site_identity');
324 +
325 + if (!self::$icon_sizes_listener_registered) {
326 + self::$icon_sizes_listener_registered = true;
327 + add_action('thinkrank_seo_settings_saved', [$this, 'generate_icon_sizes_on_save'], 10, 2);
328 + // Admin only: resizing is not front-end work, and admin traffic is
329 + // enough to run a one-time backfill promptly.
330 + add_action('admin_init', [self::class, 'maybe_backfill_icon_sizes']);
331 + }
246 332 }
247 333
248 334 /**
335 + * Save settings, then refresh what a new canonical scheme invalidates.
336 + *
337 + * The static sitemap files are written with the scheme in force when they
338 + * were built, and nothing else rebuilds them until a post or term changes.
339 + * So a change of scheme left every `<loc>` on the old one while canonical
340 + * and og:url had already moved (#736). Every writer (the settings route,
341 + * the robots route, the MCP abilities, an import) lands here.
342 + *
343 + * @since 2.7.0
344 + *
345 + * @param string $context_type Context type.
346 + * @param int|null $context_id Context ID.
347 + * @param array $settings Settings to save.
348 + * @return bool
349 + */
350 + public function save_settings(string $context_type, ?int $context_id, array $settings): bool {
351 + if (!self::touches_canonical_scheme($context_type, $context_id, $settings)) {
352 + return parent::save_settings($context_type, $context_id, $settings);
353 + }
354 +
355 + $before = Url_Scheme::preference();
356 + $saved = parent::save_settings($context_type, $context_id, $settings);
357 +
358 + if ($saved) {
359 + $this->on_canonical_scheme_saved($before);
360 + }
361 +
362 + return $saved;
363 + }
364 +
365 + /**
366 + * Whether a save can change the site-wide canonical scheme.
367 + *
368 + * @since 2.7.0
369 + *
370 + * @param string $context_type Context type.
371 + * @param int|null $context_id Context ID.
372 + * @param array $settings Settings being saved.
373 + * @return bool
374 + */
375 + public static function touches_canonical_scheme(string $context_type, ?int $context_id, array $settings): bool {
376 + return 'site' === sanitize_key($context_type)
377 + && empty($context_id)
378 + && array_key_exists('canonical_scheme', $settings);
379 + }
380 +
381 + /**
382 + * Rebuild the static sitemaps when the effective scheme changed.
383 + *
384 + * Compares the effective preference, filter included, so a site whose
385 + * scheme is pinned by `thinkrank_canonical_scheme` does not rebuild on a
386 + * stored value that changes nothing it publishes.
387 + *
388 + * @since 2.7.0
389 + *
390 + * @param string $before Effective scheme before the save.
391 + * @return void
392 + */
393 + protected function on_canonical_scheme_saved(string $before): void {
394 + // The preference is cached for the request; the save just changed it.
395 + Url_Scheme::reset();
396 +
397 + if (Url_Scheme::preference() === $before) {
398 + return;
399 + }
400 +
401 + $this->schedule_sitemap_rebuild();
402 + }
403 +
404 + /**
405 + * Queue a settings-driven sitemap rebuild.
406 + *
407 + * Debounced and run after the response, like any other settings change
408 + * that alters what the sitemap publishes.
409 + *
410 + * @since 2.7.0
411 + * @return void
412 + */
413 + protected function schedule_sitemap_rebuild(): void {
414 + (new Sitemap_Generator(false))->schedule_regeneration();
415 + }
416 +
417 + /**
418 + * Build the icon derivatives for a newly chosen favicon.
419 + *
420 + * Runs on save, which is the only moment the choice changes and the only
421 + * place image work belongs — resolving a size on the front end must stay a
422 + * lookup. Failure is silent by design: a missing derivative degrades to the
423 + * next best file, so a site whose host cannot resize still renders an icon.
424 + *
425 + * @since 2.3.1
426 + *
427 + * @param string $manager_type Settings category that was saved.
428 + * @param array $settings The settings that were written.
429 + * @return void
430 + */
431 + public function generate_icon_sizes_on_save(string $manager_type, array $settings): void {
432 + if ('site_identity' !== $manager_type) {
433 + return;
434 + }
435 +
436 + // The choice, or the derivatives behind it, may have just changed.
437 + delete_transient(self::ICON_URL_TRANSIENT);
438 +
439 + foreach (['favicon_url', 'apple_touch_icon_url'] as $key) {
440 + if (empty($settings[$key]) || !is_string($settings[$key])) {
441 + continue;
442 + }
443 +
444 + $attachment_id = self::icon_attachment_id($settings[$key]);
445 +
446 + if ($attachment_id) {
447 + self::ensure_icon_sizes($attachment_id);
448 + }
449 + }
450 +
451 + // Dropped again after the resizes finish. Resizing is not instant, and a
452 + // front-end request arriving mid-generation would otherwise repopulate
453 + // the transient with the pre-derivative URLs and pin them for the full
454 + // TTL — leaving the sizes= declarations untrue until the next save.
455 + delete_transient(self::ICON_URL_TRANSIENT);
456 + }
457 +
458 + /**
459 + * Build the derivatives for a site that configured its icons before this
460 + * existed.
461 + *
462 + * generate_icon_sizes_on_save() only fires on a settings write, so every
463 + * site with an icon already chosen would keep serving whatever
464 + * wp_get_attachment_image_url() could find — in practice the 150x150
465 + * thumbnail behind a sizes="32x32" declaration — until someone happened to
466 + * re-save Site Identity. That is the bug this is meant to fix, so the
467 + * derivatives are built once on upgrade instead of waiting for a save.
468 + *
469 + * Guarded by its own option rather than the plugin version so it runs once
470 + * and stays cheap: the check is a single autoloaded read on requests after
471 + * the first.
472 + *
473 + * @since 2.3.1
474 + *
475 + * @return void
476 + */
477 + public static function maybe_backfill_icon_sizes(): void {
478 + if (get_option(self::ICON_BACKFILL_OPTION)) {
479 + return;
480 + }
481 +
482 + // Written before the work, not after: a host that cannot resize must
483 + // not retry on every admin request forever.
484 + update_option(self::ICON_BACKFILL_OPTION, time(), true);
485 +
486 + $settings = (new self())->get_settings('site');
487 +
488 + if (!is_array($settings)) {
489 + return;
490 + }
491 +
492 + foreach (['favicon_url', 'apple_touch_icon_url'] as $key) {
493 + if (empty($settings[$key]) || !is_string($settings[$key])) {
494 + continue;
495 + }
496 +
497 + $attachment_id = self::icon_attachment_id($settings[$key]);
498 +
499 + if ($attachment_id) {
500 + self::ensure_icon_sizes($attachment_id);
501 + }
502 + }
503 +
504 + delete_transient(self::ICON_URL_TRANSIENT);
505 + }
506 +
507 + /**
508 + * Attachment ID behind a configured icon URL, or 0 when it is not ours.
509 + *
510 + * attachment_url_to_postid() matches _wp_attached_file, which holds the
511 + * ORIGINAL upload path, so the URL of a generated derivative
512 + * (`logo-512x512.png`) returns 0 — and that is exactly what the media
513 + * picker hands back when the user chooses a size. Attachment_Lookup falls
514 + * back to the original behind it; the fallback started here and moved
515 + * there when every other image lookup turned out to need it (#847).
516 + *
517 + * Shared with SEO_Manager's site-icon filter so both sides of the feature
518 + * agree on which attachment a configured URL means.
519 + *
520 + * @since 2.3.1
521 + *
522 + * @param string $url Configured icon URL.
523 + * @return int Attachment ID, or 0.
524 + */
525 + public static function icon_attachment_id(string $url): int {
526 + return Attachment_Lookup::id_from_url($url);
527 + }
528 +
529 + /**
530 + * Which ICON_SIZES derivatives this attachment still needs.
531 + *
532 + * Split out from the generation so the decision can be asserted on its
533 + * own: whether a size is skipped because it already exists or because it
534 + * would upscale is invisible once both answers are "nothing was built".
535 + *
536 + * A source is measured by its SHORTER edge — a 400x40 banner cannot yield
537 + * a true 192x192 — and anything reporting no dimensions at all (SVGs) is
538 + * left alone.
539 + *
540 + * @since 2.3.1
541 + *
542 + * @param array $meta Attachment metadata.
543 + * @return array<string, array{width: int, height: int, crop: bool}> Sizes to build.
544 + */
545 + public static function missing_icon_sizes(array $meta): array {
546 + $source = min((int) ($meta['width'] ?? 0), (int) ($meta['height'] ?? 0));
547 +
548 + if ($source < 1) {
549 + return [];
550 + }
551 +
552 + $wanted = [];
553 + foreach (self::ICON_SIZES as $size) {
554 + // Never upscale: a stretched source behind an accurate sizes=""
555 + // label is worse than the honest near miss it would replace.
556 + if (isset($meta['sizes']["site_icon-{$size}"]) || $size > $source) {
557 + continue;
558 + }
559 +
560 + $wanted["site_icon-{$size}"] = ['width' => $size, 'height' => $size, 'crop' => true];
561 + }
562 +
563 + return $wanted;
564 + }
565 +
566 + /**
567 + * Generate whatever ICON_SIZES derivatives this attachment is missing.
568 + *
569 + * Only the missing ones, and never one larger than the source: upscaling a
570 + * small favicon would put a blurrier file behind an accurate sizes="" label
571 + * than the honest near-miss it replaced.
572 + *
573 + * @since 2.3.1
574 + *
575 + * @param int $attachment_id Attachment to build derivatives for.
576 + * @return string[] Size names generated, empty when there was nothing to do.
577 + */
578 + public static function ensure_icon_sizes(int $attachment_id): array {
579 + $meta = wp_get_attachment_metadata($attachment_id);
580 +
581 + if (!is_array($meta)) {
582 + return [];
583 + }
584 +
585 + $wanted = self::missing_icon_sizes($meta);
586 +
587 + if (empty($wanted)) {
588 + return [];
589 + }
590 +
591 + $file = get_attached_file($attachment_id);
592 +
593 + if (!$file || !file_exists($file)) {
594 + return [];
595 + }
596 +
597 + $editor = wp_get_image_editor($file);
598 +
599 + if (is_wp_error($editor)) {
600 + return [];
601 + }
602 +
603 + $generated = $editor->multi_resize($wanted);
604 +
605 + if (empty($generated)) {
606 + return [];
607 + }
608 +
609 + $meta['sizes'] = array_merge($meta['sizes'] ?? [], $generated);
610 + wp_update_attachment_metadata($attachment_id, $meta);
611 +
612 + return array_keys($generated);
613 + }
614 +
615 + /**
249 616 * Initialize WordPress filesystem
250 617 *
251 618 * @since 1.0.0
252 619 * @return bool True if filesystem is initialized, false otherwise
@@ -442,8 +809,17 @@
442 809 $body = ($custom !== '' && !$fully_blocked)
443 810 ? $this->strip_robots_header($custom)
444 811 : trim($this->generate_robots_txt()['content']);
445 812
813 + // The per-agent AI directives are machine-owned, so they are composed
814 + // here rather than stored: the textarea holds the user's body, with
815 + // the fenced block stripped out of every read and re-applied on every
816 + // render. A site-wide block already disallows everyone, so adding the
817 + // per-agent group there would be noise restating the same refusal.
818 + if (!$fully_blocked) {
819 + $body = $this->apply_ai_crawler_block($body, $settings);
820 + }
821 +
446 822 if ($body === '') {
447 823 return '';
448 824 }
449 825
@@ -468,8 +844,9 @@
468 844 $robots_file = ABSPATH . 'robots.txt';
469 845 if (file_exists($robots_file) && is_readable($robots_file)) {
470 846 // phpcs:ignore WordPress.WP.AlternativeFunctions.file_get_contents_file_get_contents -- Reading a public web-root file; WP_Filesystem is not available on front-end requests.
471 847 return [
848 + // phpcs:ignore WordPress.WP.AlternativeFunctions.file_get_contents_file_get_contents -- reads a local file the plugin just located; WP_Filesystem would need credentials on some hosts.
472 849 'content' => (string) file_get_contents($robots_file),
473 850 'is_default' => false,
474 851 'source' => 'file',
475 852 ];
@@ -496,8 +873,68 @@
496 873 ];
497 874 }
498 875
499 876 /**
877 + * Describe how /robots.txt is actually delivered, and whether that still
878 + * matches the saved settings.
879 + *
880 + * The admin screen edits settings, but a physical robots.txt in the web root
881 + * is served directly by the web server and bypasses the `robots_txt` filter
882 + * entirely. When those two drift, the editor is showing content no crawler
883 + * ever sees — the conflict this exists to surface.
884 + *
885 + * @since 1.31.0
886 + *
887 + * @return array{content: string, source: string, is_default: bool, in_sync: bool, out_of_sync_reason: string, url: string}
888 + * The served content and its origin, whether it still reflects the
889 + * body the editor is showing, why it does not when it does not
890 + * ('file_drift' or 'crawl_blocked'), and the public URL it is
891 + * served from.
892 + */
893 + public function get_robots_txt_delivery(): array {
894 + $effective = $this->get_effective_robots_txt();
895 + $settings = $this->get_settings('site');
896 +
897 + // Compare bodies, not raw strings: the auto-generated header carries a
898 + // regeneration timestamp that always differs and means nothing here.
899 + // The AI crawler block is composed at render time on both sides, so it
900 + // is identical by construction and comparing it would only ever report
901 + // a false drift the admin cannot act on.
902 + $served = $this->strip_ai_crawler_block($this->strip_robots_header($effective['content']));
903 +
904 + // Measure against the body the editor is displaying — get_served_robots_body()
905 + // — not against render_robots_txt(). Two things made the old comparison
906 + // report "in sync" while the screen showed rules no crawler receives:
907 + // a physical file was compared to a freshly rendered body rather than
908 + // to the stored override the textarea shows, and a site-wide crawl
909 + // block makes render_robots_txt() return the generated "Disallow: /"
910 + // on both sides of the comparison, so it always matched.
911 + $expected = $this->get_served_robots_body();
912 +
913 + // Management off: WordPress serves its own default and the editor is not
914 + // claiming anything is live, so there is nothing to be out of sync with.
915 + $managed = !empty($settings['robots_txt_enabled']);
916 + $in_sync = !$managed || $served === $expected;
917 +
918 + $reason = '';
919 + if (!$in_sync) {
920 + // A crawl block is a deliberate override, not a stale file, and the
921 + // admin needs to be told which of the two they are looking at.
922 + $blocked = empty($settings['allow_search_engines'] ?? true) || !get_option('blog_public');
923 + $reason = $blocked ? 'crawl_blocked' : 'file_drift';
924 + }
925 +
926 + return [
927 + 'content' => $effective['content'],
928 + 'source' => $effective['source'],
929 + 'is_default' => $effective['is_default'],
930 + 'in_sync' => $in_sync,
931 + 'out_of_sync_reason' => $reason,
932 + 'url' => home_url('/robots.txt'),
933 + ];
934 + }
935 +
936 + /**
500 937 * Keep the physical robots.txt file in step with the saved settings.
501 938 *
502 939 * When management is enabled the physical file is the source of truth the
503 940 * web server serves, so this makes sure it exists and matches the effective
@@ -534,8 +971,9 @@
534 971 // do_robots()/robots_txt filter entirely, so without this those lines
535 972 // are silently dropped. ThinkRank's own filter_robots_txt callback just
536 973 // re-returns this same content (it calls render_robots_txt(), which does
537 974 // not re-apply the filter), so there is no recursion or double-append.
975 + // phpcs:ignore WordPress.NamingConventions.PrefixAllGlobals.NonPrefixedHooknameFound -- WPML/core hook, not ours to name.
538 976 $content = (string) apply_filters('robots_txt', $content, (bool) get_option('blog_public'));
539 977 if ($content === '') {
540 978 return false;
541 979 }
@@ -840,17 +1278,19 @@
840 1278 $home_text = $settings['breadcrumb_home_text'] ?? 'Home';
841 1279 if (empty($home_text)) {
842 1280 $optimization['warnings'][] = 'Empty home text reduces accessibility for screen readers';
843 1281 $optimization['score'] -= 15;
844 - } elseif (strlen($home_text) > 20) {
845 - $optimization['suggestions'][] = 'Keep home text concise (current: ' . strlen($home_text) . ' chars)';
1282 + } elseif (mb_strlen($home_text) > 20) {
1283 + // mb_strlen: this number is shown to the user as "chars" (#687).
1284 + $optimization['suggestions'][] = 'Keep home text concise (current: ' . mb_strlen($home_text) . ' chars)';
846 1285 $optimization['score'] -= 5;
847 1286 }
848 1287
849 1288 // Check prefix usage
850 1289 $prefix = $settings['breadcrumb_prefix'] ?? '';
851 - if (!empty($prefix) && strlen($prefix) > 50) {
852 - $optimization['suggestions'][] = 'Breadcrumb prefix is quite long (' . strlen($prefix) . ' chars) - consider shortening';
1290 + if (!empty($prefix) && mb_strlen($prefix) > 50) {
1291 + // mb_strlen: this number is shown to the user as "chars" (#687).
1292 + $optimization['suggestions'][] = 'Breadcrumb prefix is quite long (' . mb_strlen($prefix) . ' chars) - consider shortening';
853 1293 $optimization['score'] -= 5;
854 1294 }
855 1295
856 1296 // Current page display
@@ -981,13 +1421,16 @@
981 1421 }
982 1422
983 1423 // Additional logo analysis for local images
984 1424 if (!empty($logo_url) && filter_var($logo_url, FILTER_VALIDATE_URL)) {
985 - $attachment_id = attachment_url_to_postid($logo_url);
1425 + $attachment_id = Attachment_Lookup::id_from_url($logo_url);
986 1426 if ($attachment_id) {
987 1427 $image_meta = wp_get_attachment_metadata($attachment_id);
988 - $width = isset($image_meta['width']) ? (int) $image_meta['width'] : 0;
989 - $height = isset($image_meta['height']) ? (int) $image_meta['height'] : 0;
1428 + // The configured file's own size — a logo picked at a generated
1429 + // size is not as large as the upload behind it.
1430 + $logo_file = Attachment_Lookup::describe($attachment_id, $logo_url);
1431 + $width = $logo_file['width'];
1432 + $height = $logo_file['height'];
990 1433
991 1434 // SVG logos store 0x0 metadata — no dimension/ratio analysis
992 1435 // is possible (and dividing by 0 is fatal).
993 1436 if ($image_meta && $width > 0 && $height > 0) {
@@ -1090,11 +1533,12 @@
1090 1533 $optimization['score'] -= 15;
1091 1534 }
1092 1535 }
1093 1536
1094 - // Business type validation
1095 - if (empty($settings['business_type'])) {
1096 - $optimization['suggestions'][] = 'Select a specific business type for better schema markup';
1537 + // Business type validation (shared rule, one message — #622).
1538 + $business_type = $this->business_type_status($settings);
1539 + if ('suggestion' === $business_type['status']) {
1540 + $optimization['suggestions'][] = $business_type['message'];
1097 1541 $optimization['score'] -= 5;
1098 1542 }
1099 1543
1100 1544 // Email validation
@@ -1624,24 +2068,17 @@
1624 2068 'icon' => '✗'
1625 2069 ];
1626 2070 }
1627 2071
1628 - // Business Type validation
1629 - if (!empty($settings['business_type']) && $settings['business_type'] !== 'LocalBusiness') {
1630 - $field_details[] = [
1631 - 'field' => 'business_type',
1632 - 'label' => 'Business type is selected for proper schema markup.',
1633 - 'status' => 'valid',
1634 - 'icon' => '✓'
1635 - ];
1636 - } else {
1637 - $field_details[] = [
1638 - 'field' => 'business_type',
1639 - 'label' => 'Specific business type selection recommended for better schema markup.',
1640 - 'status' => 'suggestion',
1641 - 'icon' => '⚠'
1642 - ];
1643 - }
2072 + // Business Type validation — see business_type_status() for why there
2073 + // is exactly one rule here now (#622).
2074 + $business_type = $this->business_type_status($settings);
2075 + $field_details[] = [
2076 + 'field' => 'business_type',
2077 + 'label' => $business_type['message'],
2078 + 'status' => $business_type['status'],
2079 + 'icon' => 'valid' === $business_type['status'] ? '✓' : '⚠',
2080 + ];
1644 2081
1645 2082 // Address validation (NAP consistency)
1646 2083 $address_fields = ['business_address', 'business_city', 'business_state', 'business_country'];
1647 2084 $address_complete = true;
@@ -2260,12 +2697,15 @@
2260 2697 'warnings' => [],
2261 2698 'suggestions' => []
2262 2699 ];
2263 2700
2264 - // Validate business name (required for local SEO)
2701 + // Business name is what makes the LocalBusiness schema useful, but it
2702 + // cannot be a blocking error: the toggle is what reveals the business
2703 + // fields, so requiring the name up front makes enabling Local SEO
2704 + // impossible. The frontend already skips the output while the name is
2705 + // empty (see Seo_Manager::output_local_seo_meta_tags()).
2265 2706 if (empty($settings['business_name'])) {
2266 - $validation['errors'][] = 'Business name is required when local SEO is enabled';
2267 - $validation['valid'] = false;
2707 + $validation['warnings'][] = 'Business name is missing - required before local business schema is output';
2268 2708 } elseif (strlen($settings['business_name']) > 100) {
2269 2709 $validation['warnings'][] = 'Business name is very long, consider shortening for better display';
2270 2710 }
2271 2711
@@ -2322,11 +2762,15 @@
2322 2762 } else {
2323 2763 $validation['suggestions'][] = 'Add business hours to improve local search visibility';
2324 2764 }
2325 2765
2326 - // Validate business type
2327 - if (empty($settings['business_type'])) {
2328 - $validation['suggestions'][] = 'Select a specific business type for better schema markup';
2766 + // Business type, through the shared rule (#622). This is the only place
2767 + // it is reported on the generic path: validate_settings() with no tab
2768 + // context attaches basic-info field details, not business-info ones, so
2769 + // without this the setting would go unreported there entirely.
2770 + $business_type = $this->business_type_status($settings);
2771 + if ('suggestion' === $business_type['status']) {
2772 + $validation['suggestions'][] = $business_type['message'];
2329 2773 }
2330 2774
2331 2775 return $validation;
2332 2776 }
@@ -2406,8 +2850,201 @@
2406 2850 return $output;
2407 2851 }
2408 2852
2409 2853 /**
2854 + * Keys the Site Identity screens store beyond the 16 defaults.
2855 + *
2856 + * Title formats, breadcrumb configuration, the hero fields, the business
2857 + * block and the wizard's identity fields are all real settings written by
2858 + * this manager, none of which get_default_settings() names — it seeds only
2859 + * the values a fresh install needs. Gating on defaults alone would stop
2860 + * every one of them saving (#452).
2861 + *
2862 + * @since 2.0.1
2863 + *
2864 + * @return string[]
2865 + */
2866 + /**
2867 + * The stored alternate name(s), shaped for schema output.
2868 + *
2869 + * schema.org and Google both allow `alternateName` to carry one value or
2870 + * several, and the store already round-trips either shape, so this accepts
2871 + * both and normalises: null when there is nothing to publish, a bare string
2872 + * for one name, a list for more. Emitting a one-element array would be
2873 + * valid but noisier than it needs to be.
2874 + *
2875 + * Shared because both WebSite producers need it and must agree — a property
2876 + * added to one and not the other is how #688 happened.
2877 + *
2878 + * @since 2.7.0
2879 + *
2880 + * @param mixed $value Stored alternate_name value.
2881 + * @return string|string[]|null
2882 + */
2883 + public static function alternate_name_for_schema($value) {
2884 + $names = [];
2885 +
2886 + foreach ((array) $value as $name) {
2887 + if (!is_scalar($name)) {
2888 + continue;
2889 + }
2890 +
2891 + $name = trim((string) $name);
2892 +
2893 + if ('' !== $name && !in_array($name, $names, true)) {
2894 + $names[] = $name;
2895 + }
2896 + }
2897 +
2898 + if (empty($names)) {
2899 + return null;
2900 + }
2901 +
2902 + return 1 === count($names) ? $names[0] : $names;
2903 + }
2904 +
2905 + protected function additional_setting_keys(): array {
2906 + return [
2907 + // Title formats, one per context.
2908 + 'homepage_title', 'post_title', 'page_title', 'category_title',
2909 + 'tag_title', 'author_title', 'search_title', 'archive_title',
2910 + // The blog-index homepage's meta description (#897).
2911 + 'homepage_description',
2912 + // Breadcrumbs.
2913 + 'breadcrumb_prefix', 'show_current_page', 'breadcrumb_use_seo_title',
2914 + // Identity, as written by the setup wizard and the importers.
2915 + 'alternate_name', 'identity_type', 'represents',
2916 + 'default_meta_description', 'default_social_image',
2917 + 'social_media_accounts',
2918 + // Schema toggles that live on this screen.
2919 + 'organization_schema', 'knowledge_graph',
2920 + // Robots rules composed by the Robots.txt panel.
2921 + 'custom_robots_rules',
2922 + // Per-agent AI crawler allow/block map (#657).
2923 + 'ai_crawler_rules',
2924 + // Hero section.
2925 + 'hero_title', 'hero_subtitle', 'hero_cta_text', 'hero_cta_url',
2926 + 'hero_background_image',
2927 + // Local SEO / business details.
2928 + 'local_seo_enabled', 'business_type', 'business_name',
2929 + 'business_address', 'business_city', 'business_state',
2930 + 'business_postal_code', 'business_country', 'business_phone',
2931 + 'business_email', 'business_latitude', 'business_longitude',
2932 + 'business_price_range', 'business_hours',
2933 + ];
2934 + }
2935 +
2936 + /**
2937 + * Sanitize settings, normalising the AI crawler rule map.
2938 + *
2939 + * The generic array sanitizer keeps the shape but says nothing about the
2940 + * values: a payload could store `ai_crawler_rules[gptbot] = "maybe"`, or a
2941 + * slug no crawler answers to, and both would round-trip through every
2942 + * later response. Normalising here rather than in the REST handler puts it
2943 + * on the one path every writer shares — the settings route, the robots
2944 + * route and the MCP abilities all land in save_settings() (#657).
2945 + *
2946 + * @since 2.5.0
2947 + *
2948 + * @param array $settings Settings to sanitize.
2949 + * @param string $context_type Context type.
2950 + * @return array Sanitized settings.
2951 + */
2952 + protected function sanitize_settings(array $settings, string $context_type = 'site'): array {
2953 + $sanitized = parent::sanitize_settings($settings, $context_type);
2954 +
2955 + if (array_key_exists('ai_crawler_rules', $sanitized)) {
2956 + $sanitized['ai_crawler_rules'] = AI_Crawlers::normalize_rules($sanitized['ai_crawler_rules']);
2957 + }
2958 +
2959 + // Same reasoning one key up, for the scheme override (#638). Anything
2960 + // that is not one of the three modes means "follow WordPress", and is
2961 + // stored as that rather than kept verbatim — otherwise get-site-identity
2962 + // -settings would report a scheme the site does not actually publish.
2963 + if (array_key_exists('canonical_scheme', $sanitized)) {
2964 + $sanitized['canonical_scheme'] = in_array($sanitized['canonical_scheme'], Url_Scheme::MODES, true)
2965 + ? $sanitized['canonical_scheme']
2966 + : Url_Scheme::AUTOMATIC;
2967 + }
2968 +
2969 + // Same reasoning again for the business type. It goes straight into
2970 + // LocalBusiness schema, so a type that is not in the schema.org
2971 + // vocabulary is invalid structured data — and storing it verbatim would
2972 + // have get-site-identity-settings report a type the site cannot
2973 + // actually publish. An empty value keeps meaning "not set"; anything
2974 + // else unrecognised falls back to the general-purpose root (#623).
2975 + if (array_key_exists('business_type', $sanitized)) {
2976 + $type = (string) $sanitized['business_type'];
2977 +
2978 + if ('' !== $type && !\ThinkRank\Config\Local_Business_Types_Config::is_valid($type)) {
2979 + $type = \ThinkRank\Config\Local_Business_Types_Config::ROOT;
2980 + }
2981 +
2982 + $sanitized['business_type'] = $type;
2983 + }
2984 +
2985 + return $sanitized;
2986 + }
2987 +
2988 + /**
2989 + * schema.org's general-purpose LocalBusiness type.
2990 + *
2991 + * The default, the first option in the control, and a valid answer in its
2992 + * own right — which is the whole point of #622.
2993 + *
2994 + * @since 2.10.0
2995 + * @var string
2996 + */
2997 + private const GENERAL_BUSINESS_TYPE = 'LocalBusiness';
2998 +
2999 + /**
3000 + * The one rule for whether a business type needs the user's attention.
3001 + *
3002 + * There were three, with two wordings and two different conditions. Two
3003 + * fired when the value was empty; the third fired when it WAS
3004 + * `LocalBusiness` — which is the default, the first option in the control
3005 + * and a perfectly valid schema.org type. So the warning appeared out of the
3006 + * box for every site, could not be cleared without choosing a type that
3007 + * might be inaccurate, and on an empty value it appeared three times in two
3008 + * different phrasings, which is why it was reported as showing twice (#622).
3009 + *
3010 + * The rule now: a type is expected, and any type in the vocabulary is a
3011 + * correct answer. Only an unset value is worth prompting about.
3012 + * `LocalBusiness` is the general-purpose answer and is accepted as one —
3013 + * with a note that a more specific type sharpens the schema, phrased as the
3014 + * guidance it is rather than as a fault the user has to clear.
3015 + *
3016 + * @since 2.10.0
3017 + *
3018 + * @param array $settings Site identity settings.
3019 + * @return array{status:string,message:string} `valid` or `suggestion`.
3020 + */
3021 + private function business_type_status(array $settings): array {
3022 + $type = trim((string) ($settings['business_type'] ?? ''));
3023 +
3024 + if ('' === $type) {
3025 + return [
3026 + 'status' => 'suggestion',
3027 + 'message' => __('Select a business type so your local schema describes the right kind of business.', 'thinkrank'),
3028 + ];
3029 + }
3030 +
3031 + // The literal rather than a constant from the expanded type list (#623):
3032 + // that lands on its own branch, and this fix must not wait on it.
3033 + if (self::GENERAL_BUSINESS_TYPE === $type) {
3034 + return [
3035 + 'status' => 'valid',
3036 + 'message' => __('Business type is set to Local Business. A more specific type sharpens your schema, if one fits.', 'thinkrank'),
3037 + ];
3038 + }
3039 +
3040 + return [
3041 + 'status' => 'valid',
3042 + 'message' => __('Business type is selected for proper schema markup.', 'thinkrank'),
3043 + ];
3044 + }
3045 +
3046 + /**
2410 3047 * Get default settings for a context type (implements interface)
2411 3048 *
2412 3049 * @since 1.0.0
2413 3050 *
@@ -2427,9 +3064,32 @@
2427 3064 'breadcrumb_home_text' => 'Home',
2428 3065 'breadcrumb_separator' => '>',
2429 3066 'robots_txt_enabled' => true,
2430 3067 'allow_search_engines' => true,
3068 + // Answer 404 when a content selector in the URL resolved to
3069 + // nothing (#634). On by default, unlike the other new settings
3070 + // here: it changes no URL a visitor or a correct crawler uses, only
3071 + // ones where WordPress resolved nothing and served the blog listing
3072 + // at 200 anyway.
3073 + 'query_protection' => true,
3074 +
3075 + // Feed controls (#635). All three off, so an upgrade changes
3076 + // nothing about what an existing site already sends its
3077 + // subscribers; a brand-new install is seeded with the signature and
3078 + // the noindex on, in Activator::seed_feed_defaults().
3079 + 'feed_excerpt_only' => false,
3080 + 'feed_source_link' => false,
3081 + 'feed_noindex' => false,
3082 +
3083 + // The scheme self-referential URLs go out with (#638). 'automatic'
3084 + // means substitute nothing and follow WordPress, which is what
3085 + // every site did before the setting existed.
3086 + 'canonical_scheme' => Url_Scheme::AUTOMATIC,
2431 3087 'robots_txt_content' => '',
3088 + // Empty map = every AI crawler allowed. Defaults must stay
3089 + // permissive so an upgrade never starts blocking a crawler a site
3090 + // was happily serving (#657).
3091 + 'ai_crawler_rules' => [],
2432 3092 'logo_url' => '',
2433 3093 'favicon_url' => '',
2434 3094 'apple_touch_icon_url' => ''
2435 3095 ];
@@ -2548,10 +3208,13 @@
2548 3208 */
2549 3209 private function prepare_title_placeholders(array $data, string $context, array $settings): array {
2550 3210 $placeholders = [
2551 3211 '%title%' => $data['title'] ?? '',
2552 - '%sitename%' => $settings['site_name'] ?? get_bloginfo('name'),
2553 - '%tagline%' => $settings['tagline'] ?? get_bloginfo('description'),
3212 + // `?:` rather than `??`: these are persisted as '' rather than left
3213 + // unset, and '' is not null, so the null-coalesce never reached the
3214 + // WordPress fallback (#398).
3215 + '%sitename%' => ($settings['site_name'] ?? '') ?: get_bloginfo('name'),
3216 + '%tagline%' => ($settings['tagline'] ?? '') ?: get_bloginfo('description'),
2554 3217 '%separator%' => '', // Will be replaced with actual separator
2555 3218 '%category%' => '',
2556 3219 '%author%' => '',
2557 3220 '%date%' => '',
@@ -2635,16 +3298,17 @@
2635 3298 // Remove extra whitespace
2636 3299 $title = preg_replace('/\s+/', ' ', $title);
2637 3300 $title = trim($title);
2638 3301
2639 - // Ensure title is not too long (60 characters max for SEO)
2640 - if (strlen($title) > 60) {
2641 - // Try to truncate at word boundary
2642 - $title = wp_trim_words($title, 8, '...');
2643 - if (strlen($title) > 60) {
2644 - $title = substr($title, 0, 57) . '...';
2645 - }
2646 - }
3302 + // Ensure title is not too long (60 characters max for SEO).
3303 + // All three units here were wrong for non-Latin text: strlen() counts
3304 + // BYTES so the gate fired at 20 Thai characters, wp_trim_words() counts
3305 + // CHARACTERS on th/ja/zh_* so `8` cut the title to 8 of them, and
3306 + // substr() cuts bytes so it split a character mid-sequence (#687).
3307 + $title = \ThinkRank\Core\Seo_Text::trim_to_length(
3308 + $title,
3309 + \ThinkRank\Core\Seo_Text::TITLE_MAX_LENGTH
3310 + );
2647 3311
2648 3312 // Ensure title is not empty
2649 3313 if (empty($title)) {
2650 3314 $title = get_bloginfo('name');
@@ -3034,8 +3698,29 @@
3034 3698 * @since 1.0.0
3035 3699 * @return array Array of sitemap URLs
3036 3700 */
3037 3701 private function get_sitemap_urls_for_robots(): array {
3702 + // One wrapper over every return path below, including the #104 extras.
3703 + // The Sitemap: line is the only absolute URL of ours in robots.txt and
3704 + // the one a crawler follows to find everything else, so it has to carry
3705 + // the site's scheme preference (#638). Applied here rather than where
3706 + // the body is assembled, because that path also renders a robots.txt a
3707 + // site owner typed themselves, and their text is not ours to rewrite.
3708 + return array_map(
3709 + static function (string $url): string {
3710 + return Url_Scheme::apply($url);
3711 + },
3712 + $this->collect_sitemap_urls_for_robots()
3713 + );
3714 + }
3715 +
3716 + /**
3717 + * The sitemap URLs robots.txt advertises, before the scheme preference.
3718 + *
3719 + * @since 1.0.0
3720 + * @return array Array of sitemap URLs
3721 + */
3722 + private function collect_sitemap_urls_for_robots(): array {
3038 3723 try {
3039 3724 // Get sitemap settings
3040 3725 $sitemap_generator = new \ThinkRank\SEO\Sitemap_Generator();
3041 3726 $sitemap_settings = $sitemap_generator->get_settings('site');
@@ -3047,29 +3732,70 @@
3047 3732
3048 3733 $sitemap_urls = [];
3049 3734 $site_url = home_url();
3050 3735
3051 - // Extract enabled sitemap URLs
3736 + // Extract enabled sitemap URLs. When the index is enabled it is the
3737 + // only entry worth advertising: every child sitemap is already
3738 + // listed inside it, so naming them again in robots.txt is pure
3739 + // redundancy and drifts out of date as soon as a post type is added.
3740 + $index_url = '';
3052 3741 if (!empty($sitemap_settings['sitemap_urls']) && is_array($sitemap_settings['sitemap_urls'])) {
3053 3742 foreach ($sitemap_settings['sitemap_urls'] as $sitemap) {
3054 - if (!empty($sitemap['enabled']) && !empty($sitemap['url'])) {
3055 - $sitemap_urls[] = $site_url . $sitemap['url'];
3743 + if (empty($sitemap['enabled']) || empty($sitemap['url'])) {
3744 + continue;
3056 3745 }
3746 +
3747 + if (($sitemap['type'] ?? '') === 'index') {
3748 + $index_url = $site_url . $sitemap['url'];
3749 + continue;
3750 + }
3751 +
3752 + $sitemap_urls[] = $site_url . $sitemap['url'];
3057 3753 }
3058 3754 }
3059 3755
3756 + if ($index_url !== '') {
3757 + // The index alone — it covers the children and, on a segmented
3758 + // install, the local business sitemap too.
3759 + return [$index_url];
3760 + }
3761 +
3060 3762 // Fallback to default if no URLs found
3061 3763 if (empty($sitemap_urls)) {
3062 3764 $sitemap_urls[] = home_url('/sitemap.xml');
3063 3765 }
3064 3766
3065 - // Advertise the local business sitemap when it exists. In segmented
3066 - // mode it is already listed inside the sitemap index; in single mode
3067 - // there is no index, so robots.txt is its discovery path.
3068 - if (file_exists(ABSPATH . 'local-sitemap.xml')) {
3069 - $local_url = home_url('/local-sitemap.xml');
3070 - if (!in_array($local_url, $sitemap_urls, true)) {
3071 - $sitemap_urls[] = $local_url;
3767 + // No index on this install, so anything not already listed above has
3768 + // no other discovery path — advertise it directly. The local
3769 + // business sitemap and the sitemaps other plugins register both land
3770 + // here for the same reason, so they go through one list (#104).
3771 + $extra = [];
3772 +
3773 + // Not a file test. Under dynamic delivery the local sitemap is
3774 + // served from PHP and no file is ever written, so file_exists()
3775 + // silently dropped a sitemap the site really does publish (#752).
3776 + // On static sites the file is still what proves it, so both count.
3777 + $local_sitemap_published = file_exists(ABSPATH . 'local-sitemap.xml');
3778 +
3779 + if (!$local_sitemap_published && class_exists('ThinkRank\\SEO\\Sitemap_Generator')) {
3780 + $generator = new \ThinkRank\SEO\Sitemap_Generator(false);
3781 +
3782 + $local_sitemap_published = 'dynamic' === $generator->resolve_delivery_mode()
3783 + && $generator->publishes_local_sitemap();
3784 + }
3785 +
3786 + if ($local_sitemap_published) {
3787 + $extra[] = '/local-sitemap.xml';
3788 + }
3789 +
3790 + foreach (\ThinkRank\SEO\Sitemap_Generator::additional_sitemaps() as $path) {
3791 + $extra[] = $path;
3792 + }
3793 +
3794 + foreach ($extra as $path) {
3795 + $url = home_url($path);
3796 + if (!in_array($url, $sitemap_urls, true)) {
3797 + $sitemap_urls[] = $url;
3072 3798 }
3073 3799 }
3074 3800
3075 3801 return $sitemap_urls;
@@ -3159,8 +3885,9 @@
3159 3885 // timestamp on every update.
3160 3886 $content = '';
3161 3887
3162 3888 $current_user_agent = '';
3889 + $sitemap_started = false;
3163 3890
3164 3891 foreach ($rules as $rule) {
3165 3892 $directive = $rule['directive'] ?? '';
3166 3893 $value = $rule['value'] ?? '';
@@ -3181,9 +3908,17 @@
3181 3908 case 'crawl_delay':
3182 3909 $content .= "Crawl-delay: {$value}\n";
3183 3910 break;
3184 3911 case 'sitemap':
3185 - $content .= "\nSitemap: {$value}\n";
3912 + // One blank line separates the Sitemap block from the
3913 + // preceding group, and none appear inside it. A blank line
3914 + // terminates a record in the robots.txt grammar, so putting
3915 + // one between every directive was invalid formatting.
3916 + if (!$sitemap_started) {
3917 + $content .= "\n";
3918 + $sitemap_started = true;
3919 + }
3920 + $content .= "Sitemap: {$value}\n";
3186 3921 break;
3187 3922 }
3188 3923 }
3189 3924
@@ -3190,8 +3925,171 @@
3190 3925 return ltrim($content, "\n");
3191 3926 }
3192 3927
3193 3928 /**
3929 + * Parse a robots.txt body back into the {directive, value} rule shape.
3930 + *
3931 + * generate_robots_txt() returns `rules` alongside `content`, but callers
3932 + * replace `content` with the body actually being served (a stored override
3933 + * or a physical file). The generated rules then described something the
3934 + * response no longer contained. Re-deriving them from the served body keeps
3935 + * the two halves of the payload describing the same document.
3936 + *
3937 + * @since 2.0.1
3938 + *
3939 + * @param string $content Robots.txt body (header optional).
3940 + * @return array<int, array{directive: string, value: string}> Parsed rules.
3941 + */
3942 + public function parse_robots_txt_rules(string $content): array {
3943 + $map = [
3944 + 'user-agent' => 'user_agent',
3945 + 'disallow' => 'disallow',
3946 + 'allow' => 'allow',
3947 + 'crawl-delay' => 'crawl_delay',
3948 + 'sitemap' => 'sitemap',
3949 + ];
3950 +
3951 + $rules = [];
3952 +
3953 + foreach (preg_split('/\r\n|\r|\n/', $this->strip_robots_header($content)) as $line) {
3954 + $line = trim($line);
3955 +
3956 + // Blank lines separate groups and `#` starts a comment; neither is
3957 + // a rule.
3958 + if ($line === '' || str_starts_with($line, '#')) {
3959 + continue;
3960 + }
3961 +
3962 + $parts = explode(':', $line, 2);
3963 + if (count($parts) !== 2) {
3964 + continue;
3965 + }
3966 +
3967 + $field = strtolower(trim($parts[0]));
3968 + if (!isset($map[$field])) {
3969 + continue;
3970 + }
3971 +
3972 + $rules[] = [
3973 + 'directive' => $map[$field],
3974 + // Sitemap values are absolute URLs and contain the `:` the
3975 + // limited explode above deliberately preserved.
3976 + 'value' => trim($parts[1]),
3977 + ];
3978 + }
3979 +
3980 + return $rules;
3981 + }
3982 +
3983 + /**
3984 + * Opening fence of the machine-owned AI crawler region.
3985 + *
3986 + * @since 2.5.0
3987 + * @var string
3988 + */
3989 + public const AI_BLOCK_BEGIN = '# BEGIN ThinkRank AI crawlers';
3990 +
3991 + /**
3992 + * Closing fence of the machine-owned AI crawler region.
3993 + *
3994 + * @since 2.5.0
3995 + * @var string
3996 + */
3997 + public const AI_BLOCK_END = '# END ThinkRank AI crawlers';
3998 +
3999 + /**
4000 + * Render the fenced AI crawler region for the current settings.
4001 + *
4002 + * One `User-agent:` / `Disallow: /` record per blocked crawler. Allowed
4003 + * crawlers emit nothing at all: `Disallow:` with an empty value is the
4004 + * robots.txt way of saying "allow everything", but writing eighteen such
4005 + * records to say what silence already says would triple the file and
4006 + * invite the reading that an unlisted crawler is therefore refused.
4007 + *
4008 + * @since 2.5.0
4009 + *
4010 + * @param array $settings Site settings.
4011 + * @return string Fenced block, newline-terminated, or '' when nothing is blocked.
4012 + */
4013 + private function build_ai_crawler_block(array $settings): string {
4014 + $blocked = AI_Crawlers::blocked_slugs($settings['ai_crawler_rules'] ?? []);
4015 +
4016 + if (empty($blocked)) {
4017 + return '';
4018 + }
4019 +
4020 + $agents = AI_Crawlers::all();
4021 +
4022 + $lines = [
4023 + self::AI_BLOCK_BEGIN,
4024 + '# Managed by ThinkRank — edits between these lines are overwritten.',
4025 + ];
4026 +
4027 + foreach ($blocked as $slug) {
4028 + $lines[] = '';
4029 + $lines[] = 'User-agent: ' . $agents[$slug]['token'];
4030 + $lines[] = 'Disallow: /';
4031 + }
4032 +
4033 + $lines[] = self::AI_BLOCK_END;
4034 +
4035 + return implode("\n", $lines) . "\n";
4036 + }
4037 +
4038 + /**
4039 + * Remove the fenced AI crawler region from a robots.txt body.
4040 + *
4041 + * Tolerates a missing closing fence rather than leaving the rest of the
4042 + * file swallowed: a truncated write, or someone deleting the END line by
4043 + * hand, would otherwise make every subsequent read drop everything below
4044 + * the opening fence.
4045 + *
4046 + * @since 2.5.0
4047 + *
4048 + * @param string $body Robots.txt body.
4049 + * @return string Body with the region removed.
4050 + */
4051 + public function strip_ai_crawler_block(string $body): string {
4052 + if (false === strpos($body, self::AI_BLOCK_BEGIN)) {
4053 + return $body;
4054 + }
4055 +
4056 + $pattern = '/\R*' . preg_quote(self::AI_BLOCK_BEGIN, '/')
4057 + . '.*?(?:' . preg_quote(self::AI_BLOCK_END, '/') . '|\z)\R*/s';
4058 +
4059 + return trim((string) preg_replace($pattern, "\n\n", $body, 1));
4060 + }
4061 +
4062 + /**
4063 + * Put the current AI crawler region into a robots.txt body.
4064 + *
4065 + * Replaces an existing region in place so the block keeps its position in
4066 + * a hand-ordered file, and appends when there is none. Everything outside
4067 + * the fences is returned untouched — that is the whole point of fencing
4068 + * it, since the body is also a free-text field the user edits.
4069 + *
4070 + * @since 2.5.0
4071 + *
4072 + * @param string $body Robots.txt body (fences optional).
4073 + * @param array $settings Site settings.
4074 + * @return string Body carrying the current region.
4075 + */
4076 + private function apply_ai_crawler_block(string $body, array $settings): string {
4077 + $stripped = $this->strip_ai_crawler_block($body);
4078 + $block = $this->build_ai_crawler_block($settings);
4079 +
4080 + if ('' === $block) {
4081 + return $stripped;
4082 + }
4083 +
4084 + if ('' === trim($stripped)) {
4085 + return trim($block);
4086 + }
4087 +
4088 + return rtrim($stripped) . "\n\n" . trim($block);
4089 + }
4090 +
4091 + /**
3194 4092 * The auto-generated header prepended to the served robots.txt.
3195 4093 *
3196 4094 * Kept separate from the body so it is only ever added at render time with
3197 4095 * a fresh timestamp, never stored or shown in the editable textarea.
@@ -3228,22 +4126,28 @@
3228 4126 */
3229 4127 public function get_served_robots_body(): string {
3230 4128 $settings = $this->get_settings('site');
3231 4129
4130 + // The AI block is stripped from every one of these paths. A physical
4131 + // robots.txt we wrote carries it, and the stored override is whatever
4132 + // the textarea last held — so without this the block round-trips into
4133 + // the editor, gets saved as ordinary body text, and is then appended
4134 + // to a second time on the next render.
3232 4135 $custom = trim((string) ($settings['robots_txt_content'] ?? ''));
3233 4136 if ($custom !== '') {
3234 - return $this->strip_robots_header($custom);
4137 + return $this->strip_ai_crawler_block($this->strip_robots_header($custom));
3235 4138 }
3236 4139
3237 4140 $robots_file = ABSPATH . 'robots.txt';
3238 4141 if (file_exists($robots_file)) {
4142 + // phpcs:ignore WordPress.WP.AlternativeFunctions.file_get_contents_file_get_contents, WordPress.PHP.NoSilencedErrors.Discouraged -- an unreadable robots.txt is an expected state answered with an empty string.
3239 4143 $raw = (string) @file_get_contents($robots_file);
3240 4144 if ($raw !== '') {
3241 - return $this->strip_robots_header($raw);
4145 + return $this->strip_ai_crawler_block($this->strip_robots_header($raw));
3242 4146 }
3243 4147 }
3244 4148
3245 - return trim($this->generate_robots_txt()['content']);
4149 + return $this->strip_ai_crawler_block(trim($this->generate_robots_txt()['content']));
3246 4150 }
3247 4151 private function get_site_identity_data(array $settings): array {
3248 4152 return [
3249 4153 'site_name' => $settings['site_name'] ?? get_bloginfo('name'),
@@ -3305,11 +4209,14 @@
3305 4209 $optimization['validation']['valid'] = false;
3306 4210 }
3307 4211
3308 4212 if (!empty($value) && isset($config['max_length'])) {
3309 - if (strlen($value) > $config['max_length']) {
4213 + // The warning says "characters", so measure and cut in characters:
4214 + // strlen()/substr() fired early on non-Latin values and the
4215 + // suggested replacement was cut mid-character (#687).
4216 + if (mb_strlen($value) > $config['max_length']) {
3310 4217 $optimization['validation']['warnings'][] = "{$element} exceeds maximum length of {$config['max_length']} characters";
3311 - $optimization['optimized_value'] = substr($value, 0, $config['max_length']);
4218 + $optimization['optimized_value'] = \ThinkRank\Core\Seo_Text::trim_to_length($value, (int) $config['max_length']);
3312 4219 }
3313 4220 }
3314 4221
3315 4222 // SEO-specific optimizations
@@ -3348,18 +4255,20 @@
3348 4255 return $optimization;
3349 4256 }
3350 4257
3351 4258 // Check if it's a local image
3352 - $attachment_id = attachment_url_to_postid($value);
4259 + $attachment_id = Attachment_Lookup::id_from_url($value);
3353 4260 if ($attachment_id) {
3354 4261 $image_meta = wp_get_attachment_metadata($attachment_id);
3355 4262
3356 4263 if ($image_meta && isset($image_meta['width'], $image_meta['height'])) {
3357 - // Check recommended size
4264 + // Check recommended size, against the configured file itself
4265 + // rather than the upload it may have been generated from.
3358 4266 if (isset($config['recommended_size'])) {
3359 4267 [$rec_width, $rec_height] = explode('x', $config['recommended_size']);
4268 + $image_file = Attachment_Lookup::describe($attachment_id, $value);
3360 4269
3361 - if ($image_meta['width'] != $rec_width || $image_meta['height'] != $rec_height) {
4270 + if ($image_file['width'] !== (int) $rec_width || $image_file['height'] !== (int) $rec_height) {
3362 4271 $optimization['suggestions'][] = "Consider using {$config['recommended_size']} size for optimal {$element}";
3363 4272 }
3364 4273 }
3365 4274