PluginProbe
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO / 2.14.2
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO v2.14.2
2.14.3 2.14.2 2.14.1 2.14.0 2.13.0 2.12.0 2.11.0 2.10.0 2.9.0 2.8.0 2.7.0 2.6.0 2.5.0 2.4.0 2.3.0 2.2.0 2.1.1 2.1.0 2.0.2 2.0.1 2.0.0 1.32.0 1.31.0 1.30.0 1.29.0 All 58 releases
← All changes | includes/seo/class-site-identity-manager.php +643 -73 2.5.0 → 2.14.2 View file →
@@ -40,8 +40,36 @@
40 40 */
41 41 class Site_Identity_Manager extends Abstract_SEO_Manager {
42 42
43 43 /**
44 + * The per-context title formats as ThinkRank ships them.
45 + *
46 + * These are not in get_default_settings(): the admin screen seeds them on
47 + * first save, so on a real install they are stored values, indistinguishable
48 + * from a template the user typed. The migration needs to tell those two
49 + * apart — it may overwrite a shipped default with an imported template, and
50 + * must never overwrite a choice the user made — so this is the record of
51 + * what "untouched" looks like.
52 + *
53 + * Keep in step with getDefaultSettings() in
54 + * src/admin/components/essential-seo/SiteIdentityTab.js. SiteIdentityTitleFormatDefaultsTest
55 + * fails when the two drift.
56 + *
57 + * @since 2.8.0
58 + * @var array<string, string>
59 + */
60 + public const TITLE_FORMAT_DEFAULTS = [
61 + 'homepage_title' => '%site_title% %sep% %site_description%',
62 + 'post_title' => '%post_title% %sep% %site_title%',
63 + 'page_title' => '%page_title% %sep% %site_title%',
64 + 'category_title' => '%category_title% %sep% %site_title%',
65 + 'tag_title' => '%tag_title% %sep% %site_title%',
66 + 'author_title' => '%author_name% %sep% %site_title%',
67 + 'search_title' => 'Search Results for "%search_term%" %sep% %site_title%',
68 + 'archive_title' => '%archive_title% %sep% %site_title%',
69 + ];
70 +
71 + /**
44 72 * WordPress filesystem instance
45 73 *
46 74 * @since 1.0.0
47 75 * @var \WP_Filesystem_Base|null
@@ -290,8 +318,36 @@
290 318 * @var bool
291 319 */
292 320 private static bool $icon_sizes_listener_registered = false;
293 321
322 + /**
323 + * Whether the robots.txt resync listener is registered for this request.
324 + *
325 + * @since 2.14.0
326 + * @var bool
327 + */
328 + private static bool $robots_sync_listener_registered = false;
329 +
330 + /**
331 + * Flag set when a plugin change may have altered the sitemap set.
332 + *
333 + * @since 2.14.0
334 + * @var string
335 + */
336 + public const ROBOTS_RESYNC_OPTION = 'thinkrank_robots_txt_resync_pending';
337 +
338 + /**
339 + * Flag set when a plugin change may have altered the sitemap index.
340 + *
341 + * Separate from ROBOTS_RESYNC_OPTION because the two files exist
342 + * independently: that flag is only set when a physical robots.txt exists,
343 + * and the sitemap index needs rebuilding whether or not it does.
344 + *
345 + * @since 2.15.0
346 + * @var string
347 + */
348 + public const SITEMAP_RESYNC_OPTION = 'thinkrank_sitemap_contributors_changed';
349 +
294 350 public function __construct() {
295 351 parent::__construct('site_identity');
296 352
297 353 if (!self::$icon_sizes_listener_registered) {
@@ -300,11 +356,213 @@
300 356 // Admin only: resizing is not front-end work, and admin traffic is
301 357 // enough to run a one-time backfill promptly.
302 358 add_action('admin_init', [self::class, 'maybe_backfill_icon_sizes']);
303 359 }
360 +
361 + if (!self::$robots_sync_listener_registered) {
362 + self::$robots_sync_listener_registered = true;
363 +
364 + // A physical robots.txt bypasses PHP entirely, so composing the
365 + // Sitemap block at render time fixes the served output only on
366 + // sites with no file. Activating or deactivating a sitemap
367 + // contributor changes the set, and until #835 nothing rewrote the
368 + // file: the deactivated plugin's sitemap stayed advertised, serving
369 + // HTML to anything that followed it.
370 + add_action('activated_plugin', [self::class, 'flag_robots_txt_resync']);
371 + add_action('deactivated_plugin', [self::class, 'flag_robots_txt_resync']);
372 + add_action('init', [self::class, 'maybe_resync_robots_txt'], 99);
373 +
374 + // The sitemap index is a second static file listing the same
375 + // contributors, with its own rebuild path. #835 / #859 resynced
376 + // robots.txt only, so after Pro was deactivated the index kept
377 + // advertising news-sitemap.xml, which then served the home page
378 + // as HTML (#920).
379 + add_action('activated_plugin', [self::class, 'flag_sitemap_resync']);
380 + add_action('deactivated_plugin', [self::class, 'flag_sitemap_resync']);
381 + add_action('init', [self::class, 'maybe_resync_sitemap'], 99);
382 + }
304 383 }
305 384
306 385 /**
386 + * Note that the set of sitemap contributors may have changed.
387 + *
388 + * Deliberately unconditional about which plugin: a contributor is anything
389 + * hooking `thinkrank_additional_sitemaps`, which is resolved at runtime and
390 + * cannot be inspected for a plugin that is on its way out.
391 + *
392 + * The rewrite is not done here. `deactivated_plugin` fires inside the
393 + * request that deactivated it, while that plugin's filters are still
394 + * attached, so rendering now still sees the sitemap that is going away —
395 + * measured, not assumed: the first version of this fix wrote the
396 + * deactivated plugin's sitemap straight back into the file. The next
397 + * request has the real plugin set loaded, so the work waits for it.
398 + *
399 + * @since 2.14.0
400 + * @return void
401 + */
402 + public static function flag_robots_txt_resync(): void {
403 + if (!file_exists(ABSPATH . 'robots.txt')) {
404 + return;
405 + }
406 +
407 + update_option(self::ROBOTS_RESYNC_OPTION, 1, false);
408 + }
409 +
410 + /**
411 + * Rewrite the physical robots.txt once, on the request after a change.
412 + *
413 + * @since 2.14.0
414 + * @return void
415 + */
416 + public static function maybe_resync_robots_txt(): void {
417 + if (!get_option(self::ROBOTS_RESYNC_OPTION)) {
418 + return;
419 + }
420 +
421 + // Cleared first, so a render that fatals cannot retry on every request
422 + // for the rest of the site's life.
423 + delete_option(self::ROBOTS_RESYNC_OPTION);
424 +
425 + if (!file_exists(ABSPATH . 'robots.txt')) {
426 + return;
427 + }
428 +
429 + (new self())->sync_robots_txt_file();
430 + }
431 +
432 + /**
433 + * Note that the set of sitemap contributors may have changed.
434 + *
435 + * Unconditional, unlike flag_robots_txt_resync(): the sitemap files exist
436 + * whether or not robots.txt does. The rebuild waits for the next request
437 + * for the same reason as the robots.txt one, since `deactivated_plugin`
438 + * still runs with the outgoing plugin's `thinkrank_additional_sitemaps`
439 + * callback attached.
440 + *
441 + * @since 2.15.0
442 + * @return void
443 + */
444 + public static function flag_sitemap_resync(): void {
445 + update_option(self::SITEMAP_RESYNC_OPTION, 1, false);
446 + }
447 +
448 + /**
449 + * Queue a sitemap rebuild once, on the request after a contributor change.
450 + *
451 + * Goes through schedule_regeneration(), the debounced and lock-protected
452 + * path a sitemap settings save uses, so a burst of plugin changes (a bulk
453 + * deactivate, say) still produces one rebuild. That path also drops the
454 + * cached dynamic documents, so sites serving the sitemap from PHP drop the
455 + * entry as well.
456 + *
457 + * @since 2.15.0
458 + * @return void
459 + */
460 + public static function maybe_resync_sitemap(): void {
461 + if (!get_option(self::SITEMAP_RESYNC_OPTION)) {
462 + return;
463 + }
464 +
465 + // Cleared first, so a rebuild that fatals cannot be retried on every
466 + // request for the rest of the site's life.
467 + delete_option(self::SITEMAP_RESYNC_OPTION);
468 +
469 + $generator = new Sitemap_Generator(false);
470 + $settings = $generator->get_settings('site');
471 +
472 + // A disabled sitemap has no files to correct. Enabling it later builds
473 + // from the contributors present at that time.
474 + if (empty($settings['enabled'])) {
475 + return;
476 + }
477 +
478 + $generator->schedule_regeneration();
479 + }
480 +
481 + /**
482 + * Save settings, then refresh what a new canonical scheme invalidates.
483 + *
484 + * The static sitemap files are written with the scheme in force when they
485 + * were built, and nothing else rebuilds them until a post or term changes.
486 + * So a change of scheme left every `<loc>` on the old one while canonical
487 + * and og:url had already moved (#736). Every writer (the settings route,
488 + * the robots route, the MCP abilities, an import) lands here.
489 + *
490 + * @since 2.7.0
491 + *
492 + * @param string $context_type Context type.
493 + * @param int|null $context_id Context ID.
494 + * @param array $settings Settings to save.
495 + * @return bool
496 + */
497 + public function save_settings(string $context_type, ?int $context_id, array $settings): bool {
498 + if (!self::touches_canonical_scheme($context_type, $context_id, $settings)) {
499 + return parent::save_settings($context_type, $context_id, $settings);
500 + }
501 +
502 + $before = Url_Scheme::preference();
503 + $saved = parent::save_settings($context_type, $context_id, $settings);
504 +
505 + if ($saved) {
506 + $this->on_canonical_scheme_saved($before);
507 + }
508 +
509 + return $saved;
510 + }
511 +
512 + /**
513 + * Whether a save can change the site-wide canonical scheme.
514 + *
515 + * @since 2.7.0
516 + *
517 + * @param string $context_type Context type.
518 + * @param int|null $context_id Context ID.
519 + * @param array $settings Settings being saved.
520 + * @return bool
521 + */
522 + public static function touches_canonical_scheme(string $context_type, ?int $context_id, array $settings): bool {
523 + return 'site' === sanitize_key($context_type)
524 + && empty($context_id)
525 + && array_key_exists('canonical_scheme', $settings);
526 + }
527 +
528 + /**
529 + * Rebuild the static sitemaps when the effective scheme changed.
530 + *
531 + * Compares the effective preference, filter included, so a site whose
532 + * scheme is pinned by `thinkrank_canonical_scheme` does not rebuild on a
533 + * stored value that changes nothing it publishes.
534 + *
535 + * @since 2.7.0
536 + *
537 + * @param string $before Effective scheme before the save.
538 + * @return void
539 + */
540 + protected function on_canonical_scheme_saved(string $before): void {
541 + // The preference is cached for the request; the save just changed it.
542 + Url_Scheme::reset();
543 +
544 + if (Url_Scheme::preference() === $before) {
545 + return;
546 + }
547 +
548 + $this->schedule_sitemap_rebuild();
549 + }
550 +
551 + /**
552 + * Queue a settings-driven sitemap rebuild.
553 + *
554 + * Debounced and run after the response, like any other settings change
555 + * that alters what the sitemap publishes.
556 + *
557 + * @since 2.7.0
558 + * @return void
559 + */
560 + protected function schedule_sitemap_rebuild(): void {
561 + (new Sitemap_Generator(false))->schedule_regeneration();
562 + }
563 +
564 + /**
307 565 * Build the icon derivatives for a newly chosen favicon.
308 566 *
309 567 * Runs on save, which is the only moment the choice changes and the only
310 568 * place image work belongs — resolving a size on the front end must stay a
@@ -397,11 +655,12 @@
397 655 * Attachment ID behind a configured icon URL, or 0 when it is not ours.
398 656 *
399 657 * attachment_url_to_postid() matches _wp_attached_file, which holds the
400 658 * ORIGINAL upload path, so the URL of a generated derivative
401 - * (`logo-512.png`) returns 0 — and that is exactly what the media picker
402 - * hands back when the user chooses a size. Strip the dimension suffix and
403 - * try the original once.
659 + * (`logo-512x512.png`) returns 0 — and that is exactly what the media
660 + * picker hands back when the user chooses a size. Attachment_Lookup falls
661 + * back to the original behind it; the fallback started here and moved
662 + * there when every other image lookup turned out to need it (#847).
404 663 *
405 664 * Shared with SEO_Manager's site-icon filter so both sides of the feature
406 665 * agree on which attachment a configured URL means.
407 666 *
@@ -410,21 +669,9 @@
410 669 * @param string $url Configured icon URL.
411 670 * @return int Attachment ID, or 0.
412 671 */
413 672 public static function icon_attachment_id(string $url): int {
414 - $attachment_id = (int) attachment_url_to_postid($url);
415 -
416 - if ($attachment_id) {
417 - return $attachment_id;
418 - }
419 -
420 - $original = preg_replace('/-\d+x\d+(?=\.[a-zA-Z0-9]+$)/', '', $url);
421 -
422 - if (is_string($original) && $original !== $url) {
423 - return (int) attachment_url_to_postid($original);
424 - }
425 -
426 - return 0;
673 + return Attachment_Lookup::id_from_url($url);
427 674 }
428 675
429 676 /**
430 677 * Which ICON_SIZES derivatives this attachment still needs.
@@ -714,9 +961,20 @@
714 961 // here rather than stored: the textarea holds the user's body, with
715 962 // the fenced block stripped out of every read and re-applied on every
716 963 // render. A site-wide block already disallows everyone, so adding the
717 964 // per-agent group there would be noise restating the same refusal.
965 + // Composed here rather than read from storage, for the same reason as
966 + // the AI block below: the set of sitemaps an install publishes is a
967 + // runtime fact. `robots_txt_content` is a snapshot of it taken at the
968 + // last save, and nothing invalidated that snapshot, so deactivating a
969 + // sitemap provider left its URL advertised and serving HTML (#835).
970 + // Composing it on every render means the advertisement agrees with what
971 + // the install publishes, in both directions, with no cache to expire.
718 972 if (!$fully_blocked) {
973 + $body = $this->apply_sitemap_block($body);
974 + }
975 +
976 + if (!$fully_blocked) {
719 977 $body = $this->apply_ai_crawler_block($body, $settings);
720 978 }
721 979
722 980 if ($body === '') {
@@ -726,8 +984,87 @@
726 984 return $this->robots_txt_header() . $body . "\n";
727 985 }
728 986
729 987 /**
988 + * Replace the generated Sitemap block with the one this install publishes.
989 + *
990 + * @since 2.14.0
991 + * @param string $body Robots.txt body, without the header.
992 + * @return string
993 + */
994 + private function apply_sitemap_block(string $body): string {
995 + $stripped = $this->strip_generated_sitemap_block($body);
996 + $urls = $this->get_sitemap_urls_for_robots();
997 +
998 + if (empty($urls)) {
999 + return $stripped;
1000 + }
1001 +
1002 + $block = '';
1003 + foreach ($urls as $url) {
1004 + $block .= 'Sitemap: ' . $url . "\n";
1005 + }
1006 +
1007 + if ('' === trim($stripped)) {
1008 + return trim($block);
1009 + }
1010 +
1011 + // The grammar build_robots_txt_content() writes: one blank line before
1012 + // the block, none inside it. A blank line terminates a record in the
1013 + // robots.txt grammar, so a line between every directive is invalid.
1014 + return rtrim($stripped) . "\n\n" . trim($block);
1015 + }
1016 +
1017 + /**
1018 + * Remove the plugin-written Sitemap block from a stored body.
1019 + *
1020 + * Only the trailing run of `Sitemap:` lines is removed, which is the exact
1021 + * shape `build_robots_txt_content()` writes: a blank line, then nothing but
1022 + * `Sitemap:` lines to the end of the body. A `Sitemap:` line anywhere else
1023 + * was typed by the site owner and is left exactly where they put it, which
1024 + * is why this cannot simply strip every matching line.
1025 + *
1026 + * @since 2.14.0
1027 + * @param string $body Robots.txt body.
1028 + * @return string
1029 + */
1030 + private function strip_generated_sitemap_block(string $body): string {
1031 + $lines = preg_split('/\R/', $body);
1032 +
1033 + if (!is_array($lines)) {
1034 + return $body;
1035 + }
1036 +
1037 + $cut = count($lines);
1038 +
1039 + // Walk back over the trailing block: sitemap lines, and the blank lines
1040 + // that separate or pad it. Anything else ends the block.
1041 + for ($i = count($lines) - 1; $i >= 0; $i--) {
1042 + $line = trim($lines[$i]);
1043 +
1044 + if ('' === $line) {
1045 + $cut = $i;
1046 + continue;
1047 + }
1048 +
1049 + if (0 === stripos($line, 'sitemap:')) {
1050 + $cut = $i;
1051 + continue;
1052 + }
1053 +
1054 + break;
1055 + }
1056 +
1057 + if ($cut >= count($lines)) {
1058 + return $body;
1059 + }
1060 +
1061 + // Nothing but sitemap lines in the whole body means there is no owner
1062 + // content to keep.
1063 + return rtrim(implode("\n", array_slice($lines, 0, $cut)));
1064 + }
1065 +
1066 + /**
730 1067 * Resolve the robots.txt actually served to crawlers, with its origin.
731 1068 *
732 1069 * Lets an API/MCP consumer see the effective output without crawling the
733 1070 * URL. Mirrors serving precedence: a physical robots.txt in the web root is
@@ -1178,17 +1515,19 @@
1178 1515 $home_text = $settings['breadcrumb_home_text'] ?? 'Home';
1179 1516 if (empty($home_text)) {
1180 1517 $optimization['warnings'][] = 'Empty home text reduces accessibility for screen readers';
1181 1518 $optimization['score'] -= 15;
1182 - } elseif (strlen($home_text) > 20) {
1183 - $optimization['suggestions'][] = 'Keep home text concise (current: ' . strlen($home_text) . ' chars)';
1519 + } elseif (mb_strlen($home_text) > 20) {
1520 + // mb_strlen: this number is shown to the user as "chars" (#687).
1521 + $optimization['suggestions'][] = 'Keep home text concise (current: ' . mb_strlen($home_text) . ' chars)';
1184 1522 $optimization['score'] -= 5;
1185 1523 }
1186 1524
1187 1525 // Check prefix usage
1188 1526 $prefix = $settings['breadcrumb_prefix'] ?? '';
1189 - if (!empty($prefix) && strlen($prefix) > 50) {
1190 - $optimization['suggestions'][] = 'Breadcrumb prefix is quite long (' . strlen($prefix) . ' chars) - consider shortening';
1527 + if (!empty($prefix) && mb_strlen($prefix) > 50) {
1528 + // mb_strlen: this number is shown to the user as "chars" (#687).
1529 + $optimization['suggestions'][] = 'Breadcrumb prefix is quite long (' . mb_strlen($prefix) . ' chars) - consider shortening';
1191 1530 $optimization['score'] -= 5;
1192 1531 }
1193 1532
1194 1533 // Current page display
@@ -1297,9 +1636,9 @@
1297 1636 $optimization['suggestions'][] = 'Add a site logo for better branding and professional appearance';
1298 1637 $optimization['score'] -= 20;
1299 1638 } else {
1300 1639 // Validate logo URL and dimensions
1301 - if (!filter_var($logo_url, FILTER_VALIDATE_URL)) {
1640 + if (!\ThinkRank\Core\Url_Validator::is_http_url($logo_url)) {
1302 1641 $optimization['warnings'][] = 'Logo URL format is invalid';
1303 1642 $optimization['score'] -= 15;
1304 1643 }
1305 1644 }
@@ -1318,14 +1657,17 @@
1318 1657 $optimization['score'] -= 10;
1319 1658 }
1320 1659
1321 1660 // Additional logo analysis for local images
1322 - if (!empty($logo_url) && filter_var($logo_url, FILTER_VALIDATE_URL)) {
1323 - $attachment_id = attachment_url_to_postid($logo_url);
1661 + if (!empty($logo_url) && \ThinkRank\Core\Url_Validator::is_http_url($logo_url)) {
1662 + $attachment_id = Attachment_Lookup::id_from_url($logo_url);
1324 1663 if ($attachment_id) {
1325 1664 $image_meta = wp_get_attachment_metadata($attachment_id);
1326 - $width = isset($image_meta['width']) ? (int) $image_meta['width'] : 0;
1327 - $height = isset($image_meta['height']) ? (int) $image_meta['height'] : 0;
1665 + // The configured file's own size — a logo picked at a generated
1666 + // size is not as large as the upload behind it.
1667 + $logo_file = Attachment_Lookup::describe($attachment_id, $logo_url);
1668 + $width = $logo_file['width'];
1669 + $height = $logo_file['height'];
1328 1670
1329 1671 // SVG logos store 0x0 metadata — no dimension/ratio analysis
1330 1672 // is possible (and dividing by 0 is fatal).
1331 1673 if ($image_meta && $width > 0 && $height > 0) {
@@ -1428,11 +1770,12 @@
1428 1770 $optimization['score'] -= 15;
1429 1771 }
1430 1772 }
1431 1773
1432 - // Business type validation
1433 - if (empty($settings['business_type'])) {
1434 - $optimization['suggestions'][] = 'Select a specific business type for better schema markup';
1774 + // Business type validation (shared rule, one message — #622).
1775 + $business_type = $this->business_type_status($settings);
1776 + if ('suggestion' === $business_type['status']) {
1777 + $optimization['suggestions'][] = $business_type['message'];
1435 1778 $optimization['score'] -= 5;
1436 1779 }
1437 1780
1438 1781 // Email validation
@@ -1802,12 +2145,37 @@
1802 2145 $validation['suggestions'][] = 'Consider making site description longer (120-160 characters)';
1803 2146 }
1804 2147 }
1805 2148
1806 - // Validate logo URL
1807 - if (isset($settings['logo_url']) && !empty($settings['logo_url'])) {
1808 - if (!filter_var($settings['logo_url'], FILTER_VALIDATE_URL)) {
1809 - $validation['errors'][] = 'Logo URL must be a valid URL';
2149 + // Image and link URLs. These are written into src and href
2150 + // attributes (the schema logo, the admin previews, the hero section),
2151 + // so only web URLs are accepted. is_valid() takes any scheme, and
2152 + // "javascript://%0Aalert(1)" passed it and was stored as
2153 + // "javascript://alert(1)" once sanitize_text_field() dropped the %0A.
2154 + // The logo and default social image must be absolute: they are
2155 + // published in schema and Open Graph, which require it. The others
2156 + // may also be a path on this site.
2157 + $url_fields = [
2158 + 'logo_url' => ['Logo URL', false],
2159 + 'default_social_image' => ['Default social image URL', false],
2160 + 'favicon_url' => ['Favicon URL', true],
2161 + 'apple_touch_icon_url' => ['Apple touch icon URL', true],
2162 + 'hero_background_image' => ['Hero background image URL', true],
2163 + 'hero_cta_url' => ['Call-to-action URL', true],
2164 + ];
2165 + foreach ($url_fields as $key => [$label, $allow_path]) {
2166 + if (!isset($settings[$key]) || '' === $settings[$key] || null === $settings[$key]) {
2167 + continue;
2168 + }
2169 +
2170 + $ok = $allow_path
2171 + ? \ThinkRank\Core\Url_Validator::is_http_url_or_path($settings[$key])
2172 + : \ThinkRank\Core\Url_Validator::is_http_url($settings[$key]);
2173 +
2174 + if (!$ok) {
2175 + $validation['errors'][] = $allow_path
2176 + ? sprintf('%s must be an http or https URL, or a path starting with /', $label)
2177 + : sprintf('%s must be an http or https URL', $label);
1810 2178 $validation['valid'] = false;
1811 2179 }
1812 2180 }
1813 2181
@@ -1962,24 +2330,17 @@
1962 2330 'icon' => '✗'
1963 2331 ];
1964 2332 }
1965 2333
1966 - // Business Type validation
1967 - if (!empty($settings['business_type']) && $settings['business_type'] !== 'LocalBusiness') {
1968 - $field_details[] = [
1969 - 'field' => 'business_type',
1970 - 'label' => 'Business type is selected for proper schema markup.',
1971 - 'status' => 'valid',
1972 - 'icon' => '✓'
1973 - ];
1974 - } else {
1975 - $field_details[] = [
1976 - 'field' => 'business_type',
1977 - 'label' => 'Specific business type selection recommended for better schema markup.',
1978 - 'status' => 'suggestion',
1979 - 'icon' => '⚠'
1980 - ];
1981 - }
2334 + // Business Type validation — see business_type_status() for why there
2335 + // is exactly one rule here now (#622).
2336 + $business_type = $this->business_type_status($settings);
2337 + $field_details[] = [
2338 + 'field' => 'business_type',
2339 + 'label' => $business_type['message'],
2340 + 'status' => $business_type['status'],
2341 + 'icon' => 'valid' === $business_type['status'] ? '✓' : '⚠',
2342 + ];
1982 2343
1983 2344 // Address validation (NAP consistency)
1984 2345 $address_fields = ['business_address', 'business_city', 'business_state', 'business_country'];
1985 2346 $address_complete = true;
@@ -2252,9 +2613,9 @@
2252 2613 }
2253 2614
2254 2615 // CTA URL validation
2255 2616 if (!empty($settings['hero_cta_url'])) {
2256 - if (filter_var($settings['hero_cta_url'], FILTER_VALIDATE_URL) || strpos($settings['hero_cta_url'], '/') === 0) {
2617 + if (\ThinkRank\Core\Url_Validator::is_http_url_or_path($settings['hero_cta_url'])) {
2257 2618 $field_details[] = [
2258 2619 'field' => 'hero_cta_url',
2259 2620 'label' => 'Call-to-action URL is properly configured.',
2260 2621 'status' => 'valid',
@@ -2295,9 +2656,9 @@
2295 2656 }
2296 2657
2297 2658 // Site Logo validation (from Site Assets section)
2298 2659 if (!empty($settings['logo_url'])) {
2299 - if (filter_var($settings['logo_url'], FILTER_VALIDATE_URL)) {
2660 + if (\ThinkRank\Core\Url_Validator::is_http_url($settings['logo_url'])) {
2300 2661 $field_details[] = [
2301 2662 'field' => 'logo_url',
2302 2663 'label' => 'Site logo is properly configured.',
2303 2664 'status' => 'valid',
@@ -2663,11 +3024,15 @@
2663 3024 } else {
2664 3025 $validation['suggestions'][] = 'Add business hours to improve local search visibility';
2665 3026 }
2666 3027
2667 - // Validate business type
2668 - if (empty($settings['business_type'])) {
2669 - $validation['suggestions'][] = 'Select a specific business type for better schema markup';
3028 + // Business type, through the shared rule (#622). This is the only place
3029 + // it is reported on the generic path: validate_settings() with no tab
3030 + // context attaches basic-info field details, not business-info ones, so
3031 + // without this the setting would go unreported there entirely.
3032 + $business_type = $this->business_type_status($settings);
3033 + if ('suggestion' === $business_type['status']) {
3034 + $validation['suggestions'][] = $business_type['message'];
2670 3035 }
2671 3036
2672 3037 return $validation;
2673 3038 }
@@ -2759,13 +3124,54 @@
2759 3124 * @since 2.0.1
2760 3125 *
2761 3126 * @return string[]
2762 3127 */
3128 + /**
3129 + * The stored alternate name(s), shaped for schema output.
3130 + *
3131 + * schema.org and Google both allow `alternateName` to carry one value or
3132 + * several, and the store already round-trips either shape, so this accepts
3133 + * both and normalises: null when there is nothing to publish, a bare string
3134 + * for one name, a list for more. Emitting a one-element array would be
3135 + * valid but noisier than it needs to be.
3136 + *
3137 + * Shared because both WebSite producers need it and must agree — a property
3138 + * added to one and not the other is how #688 happened.
3139 + *
3140 + * @since 2.7.0
3141 + *
3142 + * @param mixed $value Stored alternate_name value.
3143 + * @return string|string[]|null
3144 + */
3145 + public static function alternate_name_for_schema($value) {
3146 + $names = [];
3147 +
3148 + foreach ((array) $value as $name) {
3149 + if (!is_scalar($name)) {
3150 + continue;
3151 + }
3152 +
3153 + $name = trim((string) $name);
3154 +
3155 + if ('' !== $name && !in_array($name, $names, true)) {
3156 + $names[] = $name;
3157 + }
3158 + }
3159 +
3160 + if (empty($names)) {
3161 + return null;
3162 + }
3163 +
3164 + return 1 === count($names) ? $names[0] : $names;
3165 + }
3166 +
2763 3167 protected function additional_setting_keys(): array {
2764 3168 return [
2765 3169 // Title formats, one per context.
2766 3170 'homepage_title', 'post_title', 'page_title', 'category_title',
2767 3171 'tag_title', 'author_title', 'search_title', 'archive_title',
3172 + // The blog-index homepage's meta description (#897).
3173 + 'homepage_description',
2768 3174 // Breadcrumbs.
2769 3175 'breadcrumb_prefix', 'show_current_page', 'breadcrumb_use_seo_title',
2770 3176 // Identity, as written by the setup wizard and the importers.
2771 3177 'alternate_name', 'identity_type', 'represents',
@@ -2811,12 +3217,96 @@
2811 3217 if (array_key_exists('ai_crawler_rules', $sanitized)) {
2812 3218 $sanitized['ai_crawler_rules'] = AI_Crawlers::normalize_rules($sanitized['ai_crawler_rules']);
2813 3219 }
2814 3220
3221 + // Same reasoning one key up, for the scheme override (#638). Anything
3222 + // that is not one of the three modes means "follow WordPress", and is
3223 + // stored as that rather than kept verbatim — otherwise get-site-identity
3224 + // -settings would report a scheme the site does not actually publish.
3225 + if (array_key_exists('canonical_scheme', $sanitized)) {
3226 + $sanitized['canonical_scheme'] = in_array($sanitized['canonical_scheme'], Url_Scheme::MODES, true)
3227 + ? $sanitized['canonical_scheme']
3228 + : Url_Scheme::AUTOMATIC;
3229 + }
3230 +
3231 + // Same reasoning again for the business type. It goes straight into
3232 + // LocalBusiness schema, so a type that is not in the schema.org
3233 + // vocabulary is invalid structured data — and storing it verbatim would
3234 + // have get-site-identity-settings report a type the site cannot
3235 + // actually publish. An empty value keeps meaning "not set"; anything
3236 + // else unrecognised falls back to the general-purpose root (#623).
3237 + if (array_key_exists('business_type', $sanitized)) {
3238 + $type = (string) $sanitized['business_type'];
3239 +
3240 + if ('' !== $type && !\ThinkRank\Config\Local_Business_Types_Config::is_valid($type)) {
3241 + $type = \ThinkRank\Config\Local_Business_Types_Config::ROOT;
3242 + }
3243 +
3244 + $sanitized['business_type'] = $type;
3245 + }
3246 +
2815 3247 return $sanitized;
2816 3248 }
2817 3249
2818 3250 /**
3251 + * schema.org's general-purpose LocalBusiness type.
3252 + *
3253 + * The default, the first option in the control, and a valid answer in its
3254 + * own right — which is the whole point of #622.
3255 + *
3256 + * @since 2.10.0
3257 + * @var string
3258 + */
3259 + private const GENERAL_BUSINESS_TYPE = 'LocalBusiness';
3260 +
3261 + /**
3262 + * The one rule for whether a business type needs the user's attention.
3263 + *
3264 + * There were three, with two wordings and two different conditions. Two
3265 + * fired when the value was empty; the third fired when it WAS
3266 + * `LocalBusiness` — which is the default, the first option in the control
3267 + * and a perfectly valid schema.org type. So the warning appeared out of the
3268 + * box for every site, could not be cleared without choosing a type that
3269 + * might be inaccurate, and on an empty value it appeared three times in two
3270 + * different phrasings, which is why it was reported as showing twice (#622).
3271 + *
3272 + * The rule now: a type is expected, and any type in the vocabulary is a
3273 + * correct answer. Only an unset value is worth prompting about.
3274 + * `LocalBusiness` is the general-purpose answer and is accepted as one —
3275 + * with a note that a more specific type sharpens the schema, phrased as the
3276 + * guidance it is rather than as a fault the user has to clear.
3277 + *
3278 + * @since 2.10.0
3279 + *
3280 + * @param array $settings Site identity settings.
3281 + * @return array{status:string,message:string} `valid` or `suggestion`.
3282 + */
3283 + private function business_type_status(array $settings): array {
3284 + $type = trim((string) ($settings['business_type'] ?? ''));
3285 +
3286 + if ('' === $type) {
3287 + return [
3288 + 'status' => 'suggestion',
3289 + 'message' => __('Select a business type so your local schema describes the right kind of business.', 'thinkrank'),
3290 + ];
3291 + }
3292 +
3293 + // The literal rather than a constant from the expanded type list (#623):
3294 + // that lands on its own branch, and this fix must not wait on it.
3295 + if (self::GENERAL_BUSINESS_TYPE === $type) {
3296 + return [
3297 + 'status' => 'valid',
3298 + 'message' => __('Business type is set to Local Business. A more specific type sharpens your schema, if one fits.', 'thinkrank'),
3299 + ];
3300 + }
3301 +
3302 + return [
3303 + 'status' => 'valid',
3304 + 'message' => __('Business type is selected for proper schema markup.', 'thinkrank'),
3305 + ];
3306 + }
3307 +
3308 + /**
2819 3309 * Get default settings for a context type (implements interface)
2820 3310 *
2821 3311 * @since 1.0.0
2822 3312 *
@@ -2836,8 +3326,27 @@
2836 3326 'breadcrumb_home_text' => 'Home',
2837 3327 'breadcrumb_separator' => '>',
2838 3328 'robots_txt_enabled' => true,
2839 3329 'allow_search_engines' => true,
3330 + // Answer 404 when a content selector in the URL resolved to
3331 + // nothing (#634). On by default, unlike the other new settings
3332 + // here: it changes no URL a visitor or a correct crawler uses, only
3333 + // ones where WordPress resolved nothing and served the blog listing
3334 + // at 200 anyway.
3335 + 'query_protection' => true,
3336 +
3337 + // Feed controls (#635). All three off, so an upgrade changes
3338 + // nothing about what an existing site already sends its
3339 + // subscribers; a brand-new install is seeded with the signature and
3340 + // the noindex on, in Activator::seed_feed_defaults().
3341 + 'feed_excerpt_only' => false,
3342 + 'feed_source_link' => false,
3343 + 'feed_noindex' => false,
3344 +
3345 + // The scheme self-referential URLs go out with (#638). 'automatic'
3346 + // means substitute nothing and follow WordPress, which is what
3347 + // every site did before the setting existed.
3348 + 'canonical_scheme' => Url_Scheme::AUTOMATIC,
2840 3349 'robots_txt_content' => '',
2841 3350 // Empty map = every AI crawler allowed. Defaults must stay
2842 3351 // permissive so an upgrade never starts blocking a crawler a site
2843 3352 // was happily serving (#657).
@@ -3051,16 +3560,17 @@
3051 3560 // Remove extra whitespace
3052 3561 $title = preg_replace('/\s+/', ' ', $title);
3053 3562 $title = trim($title);
3054 3563
3055 - // Ensure title is not too long (60 characters max for SEO)
3056 - if (strlen($title) > 60) {
3057 - // Try to truncate at word boundary
3058 - $title = wp_trim_words($title, 8, '...');
3059 - if (strlen($title) > 60) {
3060 - $title = substr($title, 0, 57) . '...';
3061 - }
3062 - }
3564 + // Ensure title is not too long (60 characters max for SEO).
3565 + // All three units here were wrong for non-Latin text: strlen() counts
3566 + // BYTES so the gate fired at 20 Thai characters, wp_trim_words() counts
3567 + // CHARACTERS on th/ja/zh_* so `8` cut the title to 8 of them, and
3568 + // substr() cuts bytes so it split a character mid-sequence (#687).
3569 + $title = \ThinkRank\Core\Seo_Text::trim_to_length(
3570 + $title,
3571 + \ThinkRank\Core\Seo_Text::TITLE_MAX_LENGTH
3572 + );
3063 3573
3064 3574 // Ensure title is not empty
3065 3575 if (empty($title)) {
3066 3576 $title = get_bloginfo('name');
@@ -3450,8 +3960,29 @@
3450 3960 * @since 1.0.0
3451 3961 * @return array Array of sitemap URLs
3452 3962 */
3453 3963 private function get_sitemap_urls_for_robots(): array {
3964 + // One wrapper over every return path below, including the #104 extras.
3965 + // The Sitemap: line is the only absolute URL of ours in robots.txt and
3966 + // the one a crawler follows to find everything else, so it has to carry
3967 + // the site's scheme preference (#638). Applied here rather than where
3968 + // the body is assembled, because that path also renders a robots.txt a
3969 + // site owner typed themselves, and their text is not ours to rewrite.
3970 + return array_map(
3971 + static function (string $url): string {
3972 + return Url_Scheme::apply($url);
3973 + },
3974 + $this->collect_sitemap_urls_for_robots()
3975 + );
3976 + }
3977 +
3978 + /**
3979 + * The sitemap URLs robots.txt advertises, before the scheme preference.
3980 + *
3981 + * @since 1.0.0
3982 + * @return array Array of sitemap URLs
3983 + */
3984 + private function collect_sitemap_urls_for_robots(): array {
3454 3985 try {
3455 3986 // Get sitemap settings
3456 3987 $sitemap_generator = new \ThinkRank\SEO\Sitemap_Generator();
3457 3988 $sitemap_settings = $sitemap_generator->get_settings('site');
@@ -3484,11 +4015,29 @@
3484 4015 }
3485 4016 }
3486 4017
3487 4018 if ($index_url !== '') {
3488 - // The index alone — it covers the children and, on a segmented
3489 - // install, the local business sitemap too.
3490 - return [$index_url];
4019 + // The index covers the children and, on a segmented install,
4020 + // the local business sitemap too.
4021 + //
4022 + // It does not cover a sitemap contributed through
4023 + // `thinkrank_additional_sitemaps`: the index is built by this
4024 + // plugin's own generator and never lists them. Returning the
4025 + // index alone therefore left a contributed sitemap with no
4026 + // discovery path at all — absent from robots.txt and absent
4027 + // from the index — so Pro's News sitemap was unreachable on any
4028 + // install with the index enabled, which is the default (#835).
4029 + $contributed = [];
4030 +
4031 + foreach (\ThinkRank\SEO\Sitemap_Generator::additional_sitemaps() as $path) {
4032 + $url = home_url($path);
4033 +
4034 + if ($url !== $index_url && !in_array($url, $contributed, true)) {
4035 + $contributed[] = $url;
4036 + }
4037 + }
4038 +
4039 + return array_merge([$index_url], $contributed);
3491 4040 }
3492 4041
3493 4042 // Fallback to default if no URLs found
3494 4043 if (empty($sitemap_urls)) {
@@ -3500,9 +4049,22 @@
3500 4049 // business sitemap and the sitemaps other plugins register both land
3501 4050 // here for the same reason, so they go through one list (#104).
3502 4051 $extra = [];
3503 4052
3504 - if (file_exists(ABSPATH . 'local-sitemap.xml')) {
4053 + // Not a file test. Under dynamic delivery the local sitemap is
4054 + // served from PHP and no file is ever written, so file_exists()
4055 + // silently dropped a sitemap the site really does publish (#752).
4056 + // On static sites the file is still what proves it, so both count.
4057 + $local_sitemap_published = file_exists(ABSPATH . 'local-sitemap.xml');
4058 +
4059 + if (!$local_sitemap_published && class_exists('ThinkRank\\SEO\\Sitemap_Generator')) {
4060 + $generator = new \ThinkRank\SEO\Sitemap_Generator(false);
4061 +
4062 + $local_sitemap_published = 'dynamic' === $generator->resolve_delivery_mode()
4063 + && $generator->publishes_local_sitemap();
4064 + }
4065 +
4066 + if ($local_sitemap_published) {
3505 4067 $extra[] = '/local-sitemap.xml';
3506 4068 }
3507 4069
3508 4070 foreach (\ThinkRank\SEO\Sitemap_Generator::additional_sitemaps() as $path) {
@@ -3565,9 +4127,12 @@
3565 4127 $validation['warnings'][] = "Path '{$value}' should start with '/'";
3566 4128 }
3567 4129 break;
3568 4130 case 'sitemap':
3569 - if (!filter_var($value, FILTER_VALIDATE_URL)) {
4131 + // Url_Validator, not the raw PHP filter: on a site with an
4132 + // internationalised domain the site's own sitemap URL is
4133 + // non-ASCII and the raw filter refused it.
4134 + if (!\ThinkRank\Core\Url_Validator::is_valid($value)) {
3570 4135 $validation['errors'][] = "Invalid sitemap URL: {$value}";
3571 4136 $validation['valid'] = false;
3572 4137 }
3573 4138 break;
@@ -3927,11 +4492,14 @@
3927 4492 $optimization['validation']['valid'] = false;
3928 4493 }
3929 4494
3930 4495 if (!empty($value) && isset($config['max_length'])) {
3931 - if (strlen($value) > $config['max_length']) {
4496 + // The warning says "characters", so measure and cut in characters:
4497 + // strlen()/substr() fired early on non-Latin values and the
4498 + // suggested replacement was cut mid-character (#687).
4499 + if (mb_strlen($value) > $config['max_length']) {
3932 4500 $optimization['validation']['warnings'][] = "{$element} exceeds maximum length of {$config['max_length']} characters";
3933 - $optimization['optimized_value'] = substr($value, 0, $config['max_length']);
4501 + $optimization['optimized_value'] = \ThinkRank\Core\Seo_Text::trim_to_length($value, (int) $config['max_length']);
3934 4502 }
3935 4503 }
3936 4504
3937 4505 // SEO-specific optimizations
@@ -3963,25 +4531,27 @@
3963 4531 return $optimization;
3964 4532 }
3965 4533
3966 4534 // Validate URL
3967 - if (!filter_var($value, FILTER_VALIDATE_URL)) {
3968 - $optimization['validation']['errors'][] = "{$element} must be a valid URL";
4535 + if (!\ThinkRank\Core\Url_Validator::is_http_url($value)) {
4536 + $optimization['validation']['errors'][] = "{$element} must be an http or https URL";
3969 4537 $optimization['validation']['valid'] = false;
3970 4538 return $optimization;
3971 4539 }
3972 4540
3973 4541 // Check if it's a local image
3974 - $attachment_id = attachment_url_to_postid($value);
4542 + $attachment_id = Attachment_Lookup::id_from_url($value);
3975 4543 if ($attachment_id) {
3976 4544 $image_meta = wp_get_attachment_metadata($attachment_id);
3977 4545
3978 4546 if ($image_meta && isset($image_meta['width'], $image_meta['height'])) {
3979 - // Check recommended size
4547 + // Check recommended size, against the configured file itself
4548 + // rather than the upload it may have been generated from.
3980 4549 if (isset($config['recommended_size'])) {
3981 4550 [$rec_width, $rec_height] = explode('x', $config['recommended_size']);
4551 + $image_file = Attachment_Lookup::describe($attachment_id, $value);
3982 4552
3983 - if ((int) $image_meta['width'] !== (int) $rec_width || (int) $image_meta['height'] !== (int) $rec_height) {
4553 + if ($image_file['width'] !== (int) $rec_width || $image_file['height'] !== (int) $rec_height) {
3984 4554 $optimization['suggestions'][] = "Consider using {$config['recommended_size']} size for optimal {$element}";
3985 4555 }
3986 4556 }
3987 4557