| @@ -15,8 +15,13 @@ | ||
| 15 | 15 | declare(strict_types=1); |
| 16 | 16 | |
| 17 | 17 | namespace ThinkRank\SEO; |
| 18 | 18 | |
| 19 | +// Prevent direct access | |
| 20 | +if (!defined('ABSPATH')) { | |
| 21 | + exit; | |
| 22 | +} | |
| 23 | + | |
| 19 | 24 | // Ensure dependencies are loaded |
| 20 | 25 | if (!class_exists('ThinkRank\\SEO\\Abstract_SEO_Manager')) { |
| 21 | 26 | require_once THINKRANK_PLUGIN_DIR . 'includes/seo/class-abstract-seo-manager.php'; |
| 22 | 27 | } |
| @@ -35,8 +40,36 @@ | ||
| 35 | 40 | */ |
| 36 | 41 | class Site_Identity_Manager extends Abstract_SEO_Manager { |
| 37 | 42 | |
| 38 | 43 | /** |
| 44 | + * The per-context title formats as ThinkRank ships them. | |
| 45 | + * | |
| 46 | + * These are not in get_default_settings(): the admin screen seeds them on | |
| 47 | + * first save, so on a real install they are stored values, indistinguishable | |
| 48 | + * from a template the user typed. The migration needs to tell those two | |
| 49 | + * apart — it may overwrite a shipped default with an imported template, and | |
| 50 | + * must never overwrite a choice the user made — so this is the record of | |
| 51 | + * what "untouched" looks like. | |
| 52 | + * | |
| 53 | + * Keep in step with getDefaultSettings() in | |
| 54 | + * src/admin/components/essential-seo/SiteIdentityTab.js. SiteIdentityTitleFormatDefaultsTest | |
| 55 | + * fails when the two drift. | |
| 56 | + * | |
| 57 | + * @since 2.8.0 | |
| 58 | + * @var array<string, string> | |
| 59 | + */ | |
| 60 | + public const TITLE_FORMAT_DEFAULTS = [ | |
| 61 | + 'homepage_title' => '%site_title% %sep% %site_description%', | |
| 62 | + 'post_title' => '%post_title% %sep% %site_title%', | |
| 63 | + 'page_title' => '%page_title% %sep% %site_title%', | |
| 64 | + 'category_title' => '%category_title% %sep% %site_title%', | |
| 65 | + 'tag_title' => '%tag_title% %sep% %site_title%', | |
| 66 | + 'author_title' => '%author_name% %sep% %site_title%', | |
| 67 | + 'search_title' => 'Search Results for "%search_term%" %sep% %site_title%', | |
| 68 | + 'archive_title' => '%archive_title% %sep% %site_title%', | |
| 69 | + ]; | |
| 70 | + | |
| 71 | + /** | |
| 39 | 72 | * WordPress filesystem instance |
| 40 | 73 | * |
| 41 | 74 | * @since 1.0.0 |
| 42 | 75 | * @var \WP_Filesystem_Base|null |
| @@ -240,13 +273,494 @@ | ||
| 240 | 273 | * Constructor |
| 241 | 274 | * |
| 242 | 275 | * @since 1.0.0 |
| 243 | 276 | */ |
| 277 | + /** | |
| 278 | + * The square derivatives wp_site_icon() asks for. | |
| 279 | + * | |
| 280 | + * Core generates these only through its own Site Icon crop flow, so an | |
| 281 | + * image chosen as a ThinkRank favicon straight from the media library has | |
| 282 | + * none of them and every sizes="" declaration is a near miss (#571). | |
| 283 | + * | |
| 284 | + * @since 2.3.1 | |
| 285 | + * @var int[] | |
| 286 | + */ | |
| 287 | + public const ICON_SIZES = [32, 180, 192, 270]; | |
| 288 | + | |
| 289 | + /** | |
| 290 | + * Transient holding resolved icon URLs, keyed by configured URL and size. | |
| 291 | + * | |
| 292 | + * The site-icon filter runs in wp_head on every FRONT-END request, and | |
| 293 | + * resolving a URL to its attachment costs an uncached postmeta query. The | |
| 294 | + * mapping only changes when the icon setting does, so it is cached here and | |
| 295 | + * dropped on save. | |
| 296 | + * | |
| 297 | + * @since 2.3.1 | |
| 298 | + * @var string | |
| 299 | + */ | |
| 300 | + public const ICON_URL_TRANSIENT = 'thinkrank_site_icon_urls'; | |
| 301 | + | |
| 302 | + /** | |
| 303 | + * Marker for the one-time derivative backfill on existing installs. | |
| 304 | + * | |
| 305 | + * @since 2.3.1 | |
| 306 | + * @var string | |
| 307 | + */ | |
| 308 | + public const ICON_BACKFILL_OPTION = 'thinkrank_site_icon_sizes_backfilled'; | |
| 309 | + | |
| 310 | + /** | |
| 311 | + * Whether the icon-derivative listener has been registered this request. | |
| 312 | + * | |
| 313 | + * Static because `thinkrank_seo_settings_saved` is a global hook — one | |
| 314 | + * listener serves every instance, and this class is constructed on the | |
| 315 | + * front end as well as in admin. | |
| 316 | + * | |
| 317 | + * @since 2.3.1 | |
| 318 | + * @var bool | |
| 319 | + */ | |
| 320 | + private static bool $icon_sizes_listener_registered = false; | |
| 321 | + | |
| 322 | + /** | |
| 323 | + * Whether the robots.txt resync listener is registered for this request. | |
| 324 | + * | |
| 325 | + * @since 2.14.0 | |
| 326 | + * @var bool | |
| 327 | + */ | |
| 328 | + private static bool $robots_sync_listener_registered = false; | |
| 329 | + | |
| 330 | + /** | |
| 331 | + * Flag set when a plugin change may have altered the sitemap set. | |
| 332 | + * | |
| 333 | + * @since 2.14.0 | |
| 334 | + * @var string | |
| 335 | + */ | |
| 336 | + public const ROBOTS_RESYNC_OPTION = 'thinkrank_robots_txt_resync_pending'; | |
| 337 | + | |
| 338 | + /** | |
| 339 | + * Flag set when a plugin change may have altered the sitemap index. | |
| 340 | + * | |
| 341 | + * Separate from ROBOTS_RESYNC_OPTION because the two files exist | |
| 342 | + * independently: that flag is only set when a physical robots.txt exists, | |
| 343 | + * and the sitemap index needs rebuilding whether or not it does. | |
| 344 | + * | |
| 345 | + * @since 2.15.0 | |
| 346 | + * @var string | |
| 347 | + */ | |
| 348 | + public const SITEMAP_RESYNC_OPTION = 'thinkrank_sitemap_contributors_changed'; | |
| 349 | + | |
| 244 | 350 | public function __construct() { |
| 245 | 351 | parent::__construct('site_identity'); |
| 352 | + | |
| 353 | + if (!self::$icon_sizes_listener_registered) { | |
| 354 | + self::$icon_sizes_listener_registered = true; | |
| 355 | + add_action('thinkrank_seo_settings_saved', [$this, 'generate_icon_sizes_on_save'], 10, 2); | |
| 356 | + // Admin only: resizing is not front-end work, and admin traffic is | |
| 357 | + // enough to run a one-time backfill promptly. | |
| 358 | + add_action('admin_init', [self::class, 'maybe_backfill_icon_sizes']); | |
| 359 | + } | |
| 360 | + | |
| 361 | + if (!self::$robots_sync_listener_registered) { | |
| 362 | + self::$robots_sync_listener_registered = true; | |
| 363 | + | |
| 364 | + // A physical robots.txt bypasses PHP entirely, so composing the | |
| 365 | + // Sitemap block at render time fixes the served output only on | |
| 366 | + // sites with no file. Activating or deactivating a sitemap | |
| 367 | + // contributor changes the set, and until #835 nothing rewrote the | |
| 368 | + // file: the deactivated plugin's sitemap stayed advertised, serving | |
| 369 | + // HTML to anything that followed it. | |
| 370 | + add_action('activated_plugin', [self::class, 'flag_robots_txt_resync']); | |
| 371 | + add_action('deactivated_plugin', [self::class, 'flag_robots_txt_resync']); | |
| 372 | + add_action('init', [self::class, 'maybe_resync_robots_txt'], 99); | |
| 373 | + | |
| 374 | + // The sitemap index is a second static file listing the same | |
| 375 | + // contributors, with its own rebuild path. #835 / #859 resynced | |
| 376 | + // robots.txt only, so after Pro was deactivated the index kept | |
| 377 | + // advertising news-sitemap.xml, which then served the home page | |
| 378 | + // as HTML (#920). | |
| 379 | + add_action('activated_plugin', [self::class, 'flag_sitemap_resync']); | |
| 380 | + add_action('deactivated_plugin', [self::class, 'flag_sitemap_resync']); | |
| 381 | + add_action('init', [self::class, 'maybe_resync_sitemap'], 99); | |
| 382 | + } | |
| 246 | 383 | } |
| 247 | 384 | |
| 248 | 385 | /** |
| 386 | + * Note that the set of sitemap contributors may have changed. | |
| 387 | + * | |
| 388 | + * Deliberately unconditional about which plugin: a contributor is anything | |
| 389 | + * hooking `thinkrank_additional_sitemaps`, which is resolved at runtime and | |
| 390 | + * cannot be inspected for a plugin that is on its way out. | |
| 391 | + * | |
| 392 | + * The rewrite is not done here. `deactivated_plugin` fires inside the | |
| 393 | + * request that deactivated it, while that plugin's filters are still | |
| 394 | + * attached, so rendering now still sees the sitemap that is going away — | |
| 395 | + * measured, not assumed: the first version of this fix wrote the | |
| 396 | + * deactivated plugin's sitemap straight back into the file. The next | |
| 397 | + * request has the real plugin set loaded, so the work waits for it. | |
| 398 | + * | |
| 399 | + * @since 2.14.0 | |
| 400 | + * @return void | |
| 401 | + */ | |
| 402 | + public static function flag_robots_txt_resync(): void { | |
| 403 | + if (!file_exists(ABSPATH . 'robots.txt')) { | |
| 404 | + return; | |
| 405 | + } | |
| 406 | + | |
| 407 | + update_option(self::ROBOTS_RESYNC_OPTION, 1, false); | |
| 408 | + } | |
| 409 | + | |
| 410 | + /** | |
| 411 | + * Rewrite the physical robots.txt once, on the request after a change. | |
| 412 | + * | |
| 413 | + * @since 2.14.0 | |
| 414 | + * @return void | |
| 415 | + */ | |
| 416 | + public static function maybe_resync_robots_txt(): void { | |
| 417 | + if (!get_option(self::ROBOTS_RESYNC_OPTION)) { | |
| 418 | + return; | |
| 419 | + } | |
| 420 | + | |
| 421 | + // Cleared first, so a render that fatals cannot retry on every request | |
| 422 | + // for the rest of the site's life. | |
| 423 | + delete_option(self::ROBOTS_RESYNC_OPTION); | |
| 424 | + | |
| 425 | + if (!file_exists(ABSPATH . 'robots.txt')) { | |
| 426 | + return; | |
| 427 | + } | |
| 428 | + | |
| 429 | + (new self())->sync_robots_txt_file(); | |
| 430 | + } | |
| 431 | + | |
| 432 | + /** | |
| 433 | + * Note that the set of sitemap contributors may have changed. | |
| 434 | + * | |
| 435 | + * Unconditional, unlike flag_robots_txt_resync(): the sitemap files exist | |
| 436 | + * whether or not robots.txt does. The rebuild waits for the next request | |
| 437 | + * for the same reason as the robots.txt one, since `deactivated_plugin` | |
| 438 | + * still runs with the outgoing plugin's `thinkrank_additional_sitemaps` | |
| 439 | + * callback attached. | |
| 440 | + * | |
| 441 | + * @since 2.15.0 | |
| 442 | + * @return void | |
| 443 | + */ | |
| 444 | + public static function flag_sitemap_resync(): void { | |
| 445 | + update_option(self::SITEMAP_RESYNC_OPTION, 1, false); | |
| 446 | + } | |
| 447 | + | |
| 448 | + /** | |
| 449 | + * Queue a sitemap rebuild once, on the request after a contributor change. | |
| 450 | + * | |
| 451 | + * Goes through schedule_regeneration(), the debounced and lock-protected | |
| 452 | + * path a sitemap settings save uses, so a burst of plugin changes (a bulk | |
| 453 | + * deactivate, say) still produces one rebuild. That path also drops the | |
| 454 | + * cached dynamic documents, so sites serving the sitemap from PHP drop the | |
| 455 | + * entry as well. | |
| 456 | + * | |
| 457 | + * @since 2.15.0 | |
| 458 | + * @return void | |
| 459 | + */ | |
| 460 | + public static function maybe_resync_sitemap(): void { | |
| 461 | + if (!get_option(self::SITEMAP_RESYNC_OPTION)) { | |
| 462 | + return; | |
| 463 | + } | |
| 464 | + | |
| 465 | + // Cleared first, so a rebuild that fatals cannot be retried on every | |
| 466 | + // request for the rest of the site's life. | |
| 467 | + delete_option(self::SITEMAP_RESYNC_OPTION); | |
| 468 | + | |
| 469 | + $generator = new Sitemap_Generator(false); | |
| 470 | + $settings = $generator->get_settings('site'); | |
| 471 | + | |
| 472 | + // A disabled sitemap has no files to correct. Enabling it later builds | |
| 473 | + // from the contributors present at that time. | |
| 474 | + if (empty($settings['enabled'])) { | |
| 475 | + return; | |
| 476 | + } | |
| 477 | + | |
| 478 | + $generator->schedule_regeneration(); | |
| 479 | + } | |
| 480 | + | |
| 481 | + /** | |
| 482 | + * Save settings, then refresh what a new canonical scheme invalidates. | |
| 483 | + * | |
| 484 | + * The static sitemap files are written with the scheme in force when they | |
| 485 | + * were built, and nothing else rebuilds them until a post or term changes. | |
| 486 | + * So a change of scheme left every `<loc>` on the old one while canonical | |
| 487 | + * and og:url had already moved (#736). Every writer (the settings route, | |
| 488 | + * the robots route, the MCP abilities, an import) lands here. | |
| 489 | + * | |
| 490 | + * @since 2.7.0 | |
| 491 | + * | |
| 492 | + * @param string $context_type Context type. | |
| 493 | + * @param int|null $context_id Context ID. | |
| 494 | + * @param array $settings Settings to save. | |
| 495 | + * @return bool | |
| 496 | + */ | |
| 497 | + public function save_settings(string $context_type, ?int $context_id, array $settings): bool { | |
| 498 | + if (!self::touches_canonical_scheme($context_type, $context_id, $settings)) { | |
| 499 | + return parent::save_settings($context_type, $context_id, $settings); | |
| 500 | + } | |
| 501 | + | |
| 502 | + $before = Url_Scheme::preference(); | |
| 503 | + $saved = parent::save_settings($context_type, $context_id, $settings); | |
| 504 | + | |
| 505 | + if ($saved) { | |
| 506 | + $this->on_canonical_scheme_saved($before); | |
| 507 | + } | |
| 508 | + | |
| 509 | + return $saved; | |
| 510 | + } | |
| 511 | + | |
| 512 | + /** | |
| 513 | + * Whether a save can change the site-wide canonical scheme. | |
| 514 | + * | |
| 515 | + * @since 2.7.0 | |
| 516 | + * | |
| 517 | + * @param string $context_type Context type. | |
| 518 | + * @param int|null $context_id Context ID. | |
| 519 | + * @param array $settings Settings being saved. | |
| 520 | + * @return bool | |
| 521 | + */ | |
| 522 | + public static function touches_canonical_scheme(string $context_type, ?int $context_id, array $settings): bool { | |
| 523 | + return 'site' === sanitize_key($context_type) | |
| 524 | + && empty($context_id) | |
| 525 | + && array_key_exists('canonical_scheme', $settings); | |
| 526 | + } | |
| 527 | + | |
| 528 | + /** | |
| 529 | + * Rebuild the static sitemaps when the effective scheme changed. | |
| 530 | + * | |
| 531 | + * Compares the effective preference, filter included, so a site whose | |
| 532 | + * scheme is pinned by `thinkrank_canonical_scheme` does not rebuild on a | |
| 533 | + * stored value that changes nothing it publishes. | |
| 534 | + * | |
| 535 | + * @since 2.7.0 | |
| 536 | + * | |
| 537 | + * @param string $before Effective scheme before the save. | |
| 538 | + * @return void | |
| 539 | + */ | |
| 540 | + protected function on_canonical_scheme_saved(string $before): void { | |
| 541 | + // The preference is cached for the request; the save just changed it. | |
| 542 | + Url_Scheme::reset(); | |
| 543 | + | |
| 544 | + if (Url_Scheme::preference() === $before) { | |
| 545 | + return; | |
| 546 | + } | |
| 547 | + | |
| 548 | + $this->schedule_sitemap_rebuild(); | |
| 549 | + } | |
| 550 | + | |
| 551 | + /** | |
| 552 | + * Queue a settings-driven sitemap rebuild. | |
| 553 | + * | |
| 554 | + * Debounced and run after the response, like any other settings change | |
| 555 | + * that alters what the sitemap publishes. | |
| 556 | + * | |
| 557 | + * @since 2.7.0 | |
| 558 | + * @return void | |
| 559 | + */ | |
| 560 | + protected function schedule_sitemap_rebuild(): void { | |
| 561 | + (new Sitemap_Generator(false))->schedule_regeneration(); | |
| 562 | + } | |
| 563 | + | |
| 564 | + /** | |
| 565 | + * Build the icon derivatives for a newly chosen favicon. | |
| 566 | + * | |
| 567 | + * Runs on save, which is the only moment the choice changes and the only | |
| 568 | + * place image work belongs — resolving a size on the front end must stay a | |
| 569 | + * lookup. Failure is silent by design: a missing derivative degrades to the | |
| 570 | + * next best file, so a site whose host cannot resize still renders an icon. | |
| 571 | + * | |
| 572 | + * @since 2.3.1 | |
| 573 | + * | |
| 574 | + * @param string $manager_type Settings category that was saved. | |
| 575 | + * @param array $settings The settings that were written. | |
| 576 | + * @return void | |
| 577 | + */ | |
| 578 | + public function generate_icon_sizes_on_save(string $manager_type, array $settings): void { | |
| 579 | + if ('site_identity' !== $manager_type) { | |
| 580 | + return; | |
| 581 | + } | |
| 582 | + | |
| 583 | + // The choice, or the derivatives behind it, may have just changed. | |
| 584 | + delete_transient(self::ICON_URL_TRANSIENT); | |
| 585 | + | |
| 586 | + foreach (['favicon_url', 'apple_touch_icon_url'] as $key) { | |
| 587 | + if (empty($settings[$key]) || !is_string($settings[$key])) { | |
| 588 | + continue; | |
| 589 | + } | |
| 590 | + | |
| 591 | + $attachment_id = self::icon_attachment_id($settings[$key]); | |
| 592 | + | |
| 593 | + if ($attachment_id) { | |
| 594 | + self::ensure_icon_sizes($attachment_id); | |
| 595 | + } | |
| 596 | + } | |
| 597 | + | |
| 598 | + // Dropped again after the resizes finish. Resizing is not instant, and a | |
| 599 | + // front-end request arriving mid-generation would otherwise repopulate | |
| 600 | + // the transient with the pre-derivative URLs and pin them for the full | |
| 601 | + // TTL — leaving the sizes= declarations untrue until the next save. | |
| 602 | + delete_transient(self::ICON_URL_TRANSIENT); | |
| 603 | + } | |
| 604 | + | |
| 605 | + /** | |
| 606 | + * Build the derivatives for a site that configured its icons before this | |
| 607 | + * existed. | |
| 608 | + * | |
| 609 | + * generate_icon_sizes_on_save() only fires on a settings write, so every | |
| 610 | + * site with an icon already chosen would keep serving whatever | |
| 611 | + * wp_get_attachment_image_url() could find — in practice the 150x150 | |
| 612 | + * thumbnail behind a sizes="32x32" declaration — until someone happened to | |
| 613 | + * re-save Site Identity. That is the bug this is meant to fix, so the | |
| 614 | + * derivatives are built once on upgrade instead of waiting for a save. | |
| 615 | + * | |
| 616 | + * Guarded by its own option rather than the plugin version so it runs once | |
| 617 | + * and stays cheap: the check is a single autoloaded read on requests after | |
| 618 | + * the first. | |
| 619 | + * | |
| 620 | + * @since 2.3.1 | |
| 621 | + * | |
| 622 | + * @return void | |
| 623 | + */ | |
| 624 | + public static function maybe_backfill_icon_sizes(): void { | |
| 625 | + if (get_option(self::ICON_BACKFILL_OPTION)) { | |
| 626 | + return; | |
| 627 | + } | |
| 628 | + | |
| 629 | + // Written before the work, not after: a host that cannot resize must | |
| 630 | + // not retry on every admin request forever. | |
| 631 | + update_option(self::ICON_BACKFILL_OPTION, time(), true); | |
| 632 | + | |
| 633 | + $settings = (new self())->get_settings('site'); | |
| 634 | + | |
| 635 | + if (!is_array($settings)) { | |
| 636 | + return; | |
| 637 | + } | |
| 638 | + | |
| 639 | + foreach (['favicon_url', 'apple_touch_icon_url'] as $key) { | |
| 640 | + if (empty($settings[$key]) || !is_string($settings[$key])) { | |
| 641 | + continue; | |
| 642 | + } | |
| 643 | + | |
| 644 | + $attachment_id = self::icon_attachment_id($settings[$key]); | |
| 645 | + | |
| 646 | + if ($attachment_id) { | |
| 647 | + self::ensure_icon_sizes($attachment_id); | |
| 648 | + } | |
| 649 | + } | |
| 650 | + | |
| 651 | + delete_transient(self::ICON_URL_TRANSIENT); | |
| 652 | + } | |
| 653 | + | |
| 654 | + /** | |
| 655 | + * Attachment ID behind a configured icon URL, or 0 when it is not ours. | |
| 656 | + * | |
| 657 | + * attachment_url_to_postid() matches _wp_attached_file, which holds the | |
| 658 | + * ORIGINAL upload path, so the URL of a generated derivative | |
| 659 | + * (`logo-512x512.png`) returns 0 — and that is exactly what the media | |
| 660 | + * picker hands back when the user chooses a size. Attachment_Lookup falls | |
| 661 | + * back to the original behind it; the fallback started here and moved | |
| 662 | + * there when every other image lookup turned out to need it (#847). | |
| 663 | + * | |
| 664 | + * Shared with SEO_Manager's site-icon filter so both sides of the feature | |
| 665 | + * agree on which attachment a configured URL means. | |
| 666 | + * | |
| 667 | + * @since 2.3.1 | |
| 668 | + * | |
| 669 | + * @param string $url Configured icon URL. | |
| 670 | + * @return int Attachment ID, or 0. | |
| 671 | + */ | |
| 672 | + public static function icon_attachment_id(string $url): int { | |
| 673 | + return Attachment_Lookup::id_from_url($url); | |
| 674 | + } | |
| 675 | + | |
| 676 | + /** | |
| 677 | + * Which ICON_SIZES derivatives this attachment still needs. | |
| 678 | + * | |
| 679 | + * Split out from the generation so the decision can be asserted on its | |
| 680 | + * own: whether a size is skipped because it already exists or because it | |
| 681 | + * would upscale is invisible once both answers are "nothing was built". | |
| 682 | + * | |
| 683 | + * A source is measured by its SHORTER edge — a 400x40 banner cannot yield | |
| 684 | + * a true 192x192 — and anything reporting no dimensions at all (SVGs) is | |
| 685 | + * left alone. | |
| 686 | + * | |
| 687 | + * @since 2.3.1 | |
| 688 | + * | |
| 689 | + * @param array $meta Attachment metadata. | |
| 690 | + * @return array<string, array{width: int, height: int, crop: bool}> Sizes to build. | |
| 691 | + */ | |
| 692 | + public static function missing_icon_sizes(array $meta): array { | |
| 693 | + $source = min((int) ($meta['width'] ?? 0), (int) ($meta['height'] ?? 0)); | |
| 694 | + | |
| 695 | + if ($source < 1) { | |
| 696 | + return []; | |
| 697 | + } | |
| 698 | + | |
| 699 | + $wanted = []; | |
| 700 | + foreach (self::ICON_SIZES as $size) { | |
| 701 | + // Never upscale: a stretched source behind an accurate sizes="" | |
| 702 | + // label is worse than the honest near miss it would replace. | |
| 703 | + if (isset($meta['sizes']["site_icon-{$size}"]) || $size > $source) { | |
| 704 | + continue; | |
| 705 | + } | |
| 706 | + | |
| 707 | + $wanted["site_icon-{$size}"] = ['width' => $size, 'height' => $size, 'crop' => true]; | |
| 708 | + } | |
| 709 | + | |
| 710 | + return $wanted; | |
| 711 | + } | |
| 712 | + | |
| 713 | + /** | |
| 714 | + * Generate whatever ICON_SIZES derivatives this attachment is missing. | |
| 715 | + * | |
| 716 | + * Only the missing ones, and never one larger than the source: upscaling a | |
| 717 | + * small favicon would put a blurrier file behind an accurate sizes="" label | |
| 718 | + * than the honest near-miss it replaced. | |
| 719 | + * | |
| 720 | + * @since 2.3.1 | |
| 721 | + * | |
| 722 | + * @param int $attachment_id Attachment to build derivatives for. | |
| 723 | + * @return string[] Size names generated, empty when there was nothing to do. | |
| 724 | + */ | |
| 725 | + public static function ensure_icon_sizes(int $attachment_id): array { | |
| 726 | + $meta = wp_get_attachment_metadata($attachment_id); | |
| 727 | + | |
| 728 | + if (!is_array($meta)) { | |
| 729 | + return []; | |
| 730 | + } | |
| 731 | + | |
| 732 | + $wanted = self::missing_icon_sizes($meta); | |
| 733 | + | |
| 734 | + if (empty($wanted)) { | |
| 735 | + return []; | |
| 736 | + } | |
| 737 | + | |
| 738 | + $file = get_attached_file($attachment_id); | |
| 739 | + | |
| 740 | + if (!$file || !file_exists($file)) { | |
| 741 | + return []; | |
| 742 | + } | |
| 743 | + | |
| 744 | + $editor = wp_get_image_editor($file); | |
| 745 | + | |
| 746 | + if (is_wp_error($editor)) { | |
| 747 | + return []; | |
| 748 | + } | |
| 749 | + | |
| 750 | + $generated = $editor->multi_resize($wanted); | |
| 751 | + | |
| 752 | + if (empty($generated)) { | |
| 753 | + return []; | |
| 754 | + } | |
| 755 | + | |
| 756 | + $meta['sizes'] = array_merge($meta['sizes'] ?? [], $generated); | |
| 757 | + wp_update_attachment_metadata($attachment_id, $meta); | |
| 758 | + | |
| 759 | + return array_keys($generated); | |
| 760 | + } | |
| 761 | + | |
| 762 | + /** | |
| 249 | 763 | * Initialize WordPress filesystem |
| 250 | 764 | * |
| 251 | 765 | * @since 1.0.0 |
| 252 | 766 | * @return bool True if filesystem is initialized, false otherwise |
| @@ -442,8 +956,28 @@ | ||
| 442 | 956 | $body = ($custom !== '' && !$fully_blocked) |
| 443 | 957 | ? $this->strip_robots_header($custom) |
| 444 | 958 | : trim($this->generate_robots_txt()['content']); |
| 445 | 959 | |
| 960 | + // The per-agent AI directives are machine-owned, so they are composed | |
| 961 | + // here rather than stored: the textarea holds the user's body, with | |
| 962 | + // the fenced block stripped out of every read and re-applied on every | |
| 963 | + // render. A site-wide block already disallows everyone, so adding the | |
| 964 | + // per-agent group there would be noise restating the same refusal. | |
| 965 | + // Composed here rather than read from storage, for the same reason as | |
| 966 | + // the AI block below: the set of sitemaps an install publishes is a | |
| 967 | + // runtime fact. `robots_txt_content` is a snapshot of it taken at the | |
| 968 | + // last save, and nothing invalidated that snapshot, so deactivating a | |
| 969 | + // sitemap provider left its URL advertised and serving HTML (#835). | |
| 970 | + // Composing it on every render means the advertisement agrees with what | |
| 971 | + // the install publishes, in both directions, with no cache to expire. | |
| 972 | + if (!$fully_blocked) { | |
| 973 | + $body = $this->apply_sitemap_block($body); | |
| 974 | + } | |
| 975 | + | |
| 976 | + if (!$fully_blocked) { | |
| 977 | + $body = $this->apply_ai_crawler_block($body, $settings); | |
| 978 | + } | |
| 979 | + | |
| 446 | 980 | if ($body === '') { |
| 447 | 981 | return ''; |
| 448 | 982 | } |
| 449 | 983 | |
| @@ -450,8 +984,87 @@ | ||
| 450 | 984 | return $this->robots_txt_header() . $body . "\n"; |
| 451 | 985 | } |
| 452 | 986 | |
| 453 | 987 | /** |
| 988 | + * Replace the generated Sitemap block with the one this install publishes. | |
| 989 | + * | |
| 990 | + * @since 2.14.0 | |
| 991 | + * @param string $body Robots.txt body, without the header. | |
| 992 | + * @return string | |
| 993 | + */ | |
| 994 | + private function apply_sitemap_block(string $body): string { | |
| 995 | + $stripped = $this->strip_generated_sitemap_block($body); | |
| 996 | + $urls = $this->get_sitemap_urls_for_robots(); | |
| 997 | + | |
| 998 | + if (empty($urls)) { | |
| 999 | + return $stripped; | |
| 1000 | + } | |
| 1001 | + | |
| 1002 | + $block = ''; | |
| 1003 | + foreach ($urls as $url) { | |
| 1004 | + $block .= 'Sitemap: ' . $url . "\n"; | |
| 1005 | + } | |
| 1006 | + | |
| 1007 | + if ('' === trim($stripped)) { | |
| 1008 | + return trim($block); | |
| 1009 | + } | |
| 1010 | + | |
| 1011 | + // The grammar build_robots_txt_content() writes: one blank line before | |
| 1012 | + // the block, none inside it. A blank line terminates a record in the | |
| 1013 | + // robots.txt grammar, so a line between every directive is invalid. | |
| 1014 | + return rtrim($stripped) . "\n\n" . trim($block); | |
| 1015 | + } | |
| 1016 | + | |
| 1017 | + /** | |
| 1018 | + * Remove the plugin-written Sitemap block from a stored body. | |
| 1019 | + * | |
| 1020 | + * Only the trailing run of `Sitemap:` lines is removed, which is the exact | |
| 1021 | + * shape `build_robots_txt_content()` writes: a blank line, then nothing but | |
| 1022 | + * `Sitemap:` lines to the end of the body. A `Sitemap:` line anywhere else | |
| 1023 | + * was typed by the site owner and is left exactly where they put it, which | |
| 1024 | + * is why this cannot simply strip every matching line. | |
| 1025 | + * | |
| 1026 | + * @since 2.14.0 | |
| 1027 | + * @param string $body Robots.txt body. | |
| 1028 | + * @return string | |
| 1029 | + */ | |
| 1030 | + private function strip_generated_sitemap_block(string $body): string { | |
| 1031 | + $lines = preg_split('/\R/', $body); | |
| 1032 | + | |
| 1033 | + if (!is_array($lines)) { | |
| 1034 | + return $body; | |
| 1035 | + } | |
| 1036 | + | |
| 1037 | + $cut = count($lines); | |
| 1038 | + | |
| 1039 | + // Walk back over the trailing block: sitemap lines, and the blank lines | |
| 1040 | + // that separate or pad it. Anything else ends the block. | |
| 1041 | + for ($i = count($lines) - 1; $i >= 0; $i--) { | |
| 1042 | + $line = trim($lines[$i]); | |
| 1043 | + | |
| 1044 | + if ('' === $line) { | |
| 1045 | + $cut = $i; | |
| 1046 | + continue; | |
| 1047 | + } | |
| 1048 | + | |
| 1049 | + if (0 === stripos($line, 'sitemap:')) { | |
| 1050 | + $cut = $i; | |
| 1051 | + continue; | |
| 1052 | + } | |
| 1053 | + | |
| 1054 | + break; | |
| 1055 | + } | |
| 1056 | + | |
| 1057 | + if ($cut >= count($lines)) { | |
| 1058 | + return $body; | |
| 1059 | + } | |
| 1060 | + | |
| 1061 | + // Nothing but sitemap lines in the whole body means there is no owner | |
| 1062 | + // content to keep. | |
| 1063 | + return rtrim(implode("\n", array_slice($lines, 0, $cut))); | |
| 1064 | + } | |
| 1065 | + | |
| 1066 | + /** | |
| 454 | 1067 | * Resolve the robots.txt actually served to crawlers, with its origin. |
| 455 | 1068 | * |
| 456 | 1069 | * Lets an API/MCP consumer see the effective output without crawling the |
| 457 | 1070 | * URL. Mirrors serving precedence: a physical robots.txt in the web root is |
| @@ -497,8 +1110,68 @@ | ||
| 497 | 1110 | ]; |
| 498 | 1111 | } |
| 499 | 1112 | |
| 500 | 1113 | /** |
| 1114 | + * Describe how /robots.txt is actually delivered, and whether that still | |
| 1115 | + * matches the saved settings. | |
| 1116 | + * | |
| 1117 | + * The admin screen edits settings, but a physical robots.txt in the web root | |
| 1118 | + * is served directly by the web server and bypasses the `robots_txt` filter | |
| 1119 | + * entirely. When those two drift, the editor is showing content no crawler | |
| 1120 | + * ever sees — the conflict this exists to surface. | |
| 1121 | + * | |
| 1122 | + * @since 1.31.0 | |
| 1123 | + * | |
| 1124 | + * @return array{content: string, source: string, is_default: bool, in_sync: bool, out_of_sync_reason: string, url: string} | |
| 1125 | + * The served content and its origin, whether it still reflects the | |
| 1126 | + * body the editor is showing, why it does not when it does not | |
| 1127 | + * ('file_drift' or 'crawl_blocked'), and the public URL it is | |
| 1128 | + * served from. | |
| 1129 | + */ | |
| 1130 | + public function get_robots_txt_delivery(): array { | |
| 1131 | + $effective = $this->get_effective_robots_txt(); | |
| 1132 | + $settings = $this->get_settings('site'); | |
| 1133 | + | |
| 1134 | + // Compare bodies, not raw strings: the auto-generated header carries a | |
| 1135 | + // regeneration timestamp that always differs and means nothing here. | |
| 1136 | + // The AI crawler block is composed at render time on both sides, so it | |
| 1137 | + // is identical by construction and comparing it would only ever report | |
| 1138 | + // a false drift the admin cannot act on. | |
| 1139 | + $served = $this->strip_ai_crawler_block($this->strip_robots_header($effective['content'])); | |
| 1140 | + | |
| 1141 | + // Measure against the body the editor is displaying — get_served_robots_body() | |
| 1142 | + // — not against render_robots_txt(). Two things made the old comparison | |
| 1143 | + // report "in sync" while the screen showed rules no crawler receives: | |
| 1144 | + // a physical file was compared to a freshly rendered body rather than | |
| 1145 | + // to the stored override the textarea shows, and a site-wide crawl | |
| 1146 | + // block makes render_robots_txt() return the generated "Disallow: /" | |
| 1147 | + // on both sides of the comparison, so it always matched. | |
| 1148 | + $expected = $this->get_served_robots_body(); | |
| 1149 | + | |
| 1150 | + // Management off: WordPress serves its own default and the editor is not | |
| 1151 | + // claiming anything is live, so there is nothing to be out of sync with. | |
| 1152 | + $managed = !empty($settings['robots_txt_enabled']); | |
| 1153 | + $in_sync = !$managed || $served === $expected; | |
| 1154 | + | |
| 1155 | + $reason = ''; | |
| 1156 | + if (!$in_sync) { | |
| 1157 | + // A crawl block is a deliberate override, not a stale file, and the | |
| 1158 | + // admin needs to be told which of the two they are looking at. | |
| 1159 | + $blocked = empty($settings['allow_search_engines'] ?? true) || !get_option('blog_public'); | |
| 1160 | + $reason = $blocked ? 'crawl_blocked' : 'file_drift'; | |
| 1161 | + } | |
| 1162 | + | |
| 1163 | + return [ | |
| 1164 | + 'content' => $effective['content'], | |
| 1165 | + 'source' => $effective['source'], | |
| 1166 | + 'is_default' => $effective['is_default'], | |
| 1167 | + 'in_sync' => $in_sync, | |
| 1168 | + 'out_of_sync_reason' => $reason, | |
| 1169 | + 'url' => home_url('/robots.txt'), | |
| 1170 | + ]; | |
| 1171 | + } | |
| 1172 | + | |
| 1173 | + /** | |
| 501 | 1174 | * Keep the physical robots.txt file in step with the saved settings. |
| 502 | 1175 | * |
| 503 | 1176 | * When management is enabled the physical file is the source of truth the |
| 504 | 1177 | * web server serves, so this makes sure it exists and matches the effective |
| @@ -842,17 +1515,19 @@ | ||
| 842 | 1515 | $home_text = $settings['breadcrumb_home_text'] ?? 'Home'; |
| 843 | 1516 | if (empty($home_text)) { |
| 844 | 1517 | $optimization['warnings'][] = 'Empty home text reduces accessibility for screen readers'; |
| 845 | 1518 | $optimization['score'] -= 15; |
| 846 | - } elseif (strlen($home_text) > 20) { | |
| 847 | - $optimization['suggestions'][] = 'Keep home text concise (current: ' . strlen($home_text) . ' chars)'; | |
| 1519 | + } elseif (mb_strlen($home_text) > 20) { | |
| 1520 | + // mb_strlen: this number is shown to the user as "chars" (#687). | |
| 1521 | + $optimization['suggestions'][] = 'Keep home text concise (current: ' . mb_strlen($home_text) . ' chars)'; | |
| 848 | 1522 | $optimization['score'] -= 5; |
| 849 | 1523 | } |
| 850 | 1524 | |
| 851 | 1525 | // Check prefix usage |
| 852 | 1526 | $prefix = $settings['breadcrumb_prefix'] ?? ''; |
| 853 | - if (!empty($prefix) && strlen($prefix) > 50) { | |
| 854 | - $optimization['suggestions'][] = 'Breadcrumb prefix is quite long (' . strlen($prefix) . ' chars) - consider shortening'; | |
| 1527 | + if (!empty($prefix) && mb_strlen($prefix) > 50) { | |
| 1528 | + // mb_strlen: this number is shown to the user as "chars" (#687). | |
| 1529 | + $optimization['suggestions'][] = 'Breadcrumb prefix is quite long (' . mb_strlen($prefix) . ' chars) - consider shortening'; | |
| 855 | 1530 | $optimization['score'] -= 5; |
| 856 | 1531 | } |
| 857 | 1532 | |
| 858 | 1533 | // Current page display |
| @@ -961,9 +1636,9 @@ | ||
| 961 | 1636 | $optimization['suggestions'][] = 'Add a site logo for better branding and professional appearance'; |
| 962 | 1637 | $optimization['score'] -= 20; |
| 963 | 1638 | } else { |
| 964 | 1639 | // Validate logo URL and dimensions |
| 965 | - if (!filter_var($logo_url, FILTER_VALIDATE_URL)) { | |
| 1640 | + if (!\ThinkRank\Core\Url_Validator::is_http_url($logo_url)) { | |
| 966 | 1641 | $optimization['warnings'][] = 'Logo URL format is invalid'; |
| 967 | 1642 | $optimization['score'] -= 15; |
| 968 | 1643 | } |
| 969 | 1644 | } |
| @@ -982,14 +1657,17 @@ | ||
| 982 | 1657 | $optimization['score'] -= 10; |
| 983 | 1658 | } |
| 984 | 1659 | |
| 985 | 1660 | // Additional logo analysis for local images |
| 986 | - if (!empty($logo_url) && filter_var($logo_url, FILTER_VALIDATE_URL)) { | |
| 987 | - $attachment_id = attachment_url_to_postid($logo_url); | |
| 1661 | + if (!empty($logo_url) && \ThinkRank\Core\Url_Validator::is_http_url($logo_url)) { | |
| 1662 | + $attachment_id = Attachment_Lookup::id_from_url($logo_url); | |
| 988 | 1663 | if ($attachment_id) { |
| 989 | 1664 | $image_meta = wp_get_attachment_metadata($attachment_id); |
| 990 | - $width = isset($image_meta['width']) ? (int) $image_meta['width'] : 0; | |
| 991 | - $height = isset($image_meta['height']) ? (int) $image_meta['height'] : 0; | |
| 1665 | + // The configured file's own size — a logo picked at a generated | |
| 1666 | + // size is not as large as the upload behind it. | |
| 1667 | + $logo_file = Attachment_Lookup::describe($attachment_id, $logo_url); | |
| 1668 | + $width = $logo_file['width']; | |
| 1669 | + $height = $logo_file['height']; | |
| 992 | 1670 | |
| 993 | 1671 | // SVG logos store 0x0 metadata — no dimension/ratio analysis |
| 994 | 1672 | // is possible (and dividing by 0 is fatal). |
| 995 | 1673 | if ($image_meta && $width > 0 && $height > 0) { |
| @@ -1092,11 +1770,12 @@ | ||
| 1092 | 1770 | $optimization['score'] -= 15; |
| 1093 | 1771 | } |
| 1094 | 1772 | } |
| 1095 | 1773 | |
| 1096 | - // Business type validation | |
| 1097 | - if (empty($settings['business_type'])) { | |
| 1098 | - $optimization['suggestions'][] = 'Select a specific business type for better schema markup'; | |
| 1774 | + // Business type validation (shared rule, one message — #622). | |
| 1775 | + $business_type = $this->business_type_status($settings); | |
| 1776 | + if ('suggestion' === $business_type['status']) { | |
| 1777 | + $optimization['suggestions'][] = $business_type['message']; | |
| 1099 | 1778 | $optimization['score'] -= 5; |
| 1100 | 1779 | } |
| 1101 | 1780 | |
| 1102 | 1781 | // Email validation |
| @@ -1466,12 +2145,37 @@ | ||
| 1466 | 2145 | $validation['suggestions'][] = 'Consider making site description longer (120-160 characters)'; |
| 1467 | 2146 | } |
| 1468 | 2147 | } |
| 1469 | 2148 | |
| 1470 | - // Validate logo URL | |
| 1471 | - if (isset($settings['logo_url']) && !empty($settings['logo_url'])) { | |
| 1472 | - if (!filter_var($settings['logo_url'], FILTER_VALIDATE_URL)) { | |
| 1473 | - $validation['errors'][] = 'Logo URL must be a valid URL'; | |
| 2149 | + // Image and link URLs. These are written into src and href | |
| 2150 | + // attributes (the schema logo, the admin previews, the hero section), | |
| 2151 | + // so only web URLs are accepted. is_valid() takes any scheme, and | |
| 2152 | + // "javascript://%0Aalert(1)" passed it and was stored as | |
| 2153 | + // "javascript://alert(1)" once sanitize_text_field() dropped the %0A. | |
| 2154 | + // The logo and default social image must be absolute: they are | |
| 2155 | + // published in schema and Open Graph, which require it. The others | |
| 2156 | + // may also be a path on this site. | |
| 2157 | + $url_fields = [ | |
| 2158 | + 'logo_url' => ['Logo URL', false], | |
| 2159 | + 'default_social_image' => ['Default social image URL', false], | |
| 2160 | + 'favicon_url' => ['Favicon URL', true], | |
| 2161 | + 'apple_touch_icon_url' => ['Apple touch icon URL', true], | |
| 2162 | + 'hero_background_image' => ['Hero background image URL', true], | |
| 2163 | + 'hero_cta_url' => ['Call-to-action URL', true], | |
| 2164 | + ]; | |
| 2165 | + foreach ($url_fields as $key => [$label, $allow_path]) { | |
| 2166 | + if (!isset($settings[$key]) || '' === $settings[$key] || null === $settings[$key]) { | |
| 2167 | + continue; | |
| 2168 | + } | |
| 2169 | + | |
| 2170 | + $ok = $allow_path | |
| 2171 | + ? \ThinkRank\Core\Url_Validator::is_http_url_or_path($settings[$key]) | |
| 2172 | + : \ThinkRank\Core\Url_Validator::is_http_url($settings[$key]); | |
| 2173 | + | |
| 2174 | + if (!$ok) { | |
| 2175 | + $validation['errors'][] = $allow_path | |
| 2176 | + ? sprintf('%s must be an http or https URL, or a path starting with /', $label) | |
| 2177 | + : sprintf('%s must be an http or https URL', $label); | |
| 1474 | 2178 | $validation['valid'] = false; |
| 1475 | 2179 | } |
| 1476 | 2180 | } |
| 1477 | 2181 | |
| @@ -1626,24 +2330,17 @@ | ||
| 1626 | 2330 | 'icon' => '✗' |
| 1627 | 2331 | ]; |
| 1628 | 2332 | } |
| 1629 | 2333 | |
| 1630 | - // Business Type validation | |
| 1631 | - if (!empty($settings['business_type']) && $settings['business_type'] !== 'LocalBusiness') { | |
| 1632 | - $field_details[] = [ | |
| 1633 | - 'field' => 'business_type', | |
| 1634 | - 'label' => 'Business type is selected for proper schema markup.', | |
| 1635 | - 'status' => 'valid', | |
| 1636 | - 'icon' => '✓' | |
| 1637 | - ]; | |
| 1638 | - } else { | |
| 1639 | - $field_details[] = [ | |
| 1640 | - 'field' => 'business_type', | |
| 1641 | - 'label' => 'Specific business type selection recommended for better schema markup.', | |
| 1642 | - 'status' => 'suggestion', | |
| 1643 | - 'icon' => '⚠' | |
| 1644 | - ]; | |
| 1645 | - } | |
| 2334 | + // Business Type validation — see business_type_status() for why there | |
| 2335 | + // is exactly one rule here now (#622). | |
| 2336 | + $business_type = $this->business_type_status($settings); | |
| 2337 | + $field_details[] = [ | |
| 2338 | + 'field' => 'business_type', | |
| 2339 | + 'label' => $business_type['message'], | |
| 2340 | + 'status' => $business_type['status'], | |
| 2341 | + 'icon' => 'valid' === $business_type['status'] ? '✓' : '⚠', | |
| 2342 | + ]; | |
| 1646 | 2343 | |
| 1647 | 2344 | // Address validation (NAP consistency) |
| 1648 | 2345 | $address_fields = ['business_address', 'business_city', 'business_state', 'business_country']; |
| 1649 | 2346 | $address_complete = true; |
| @@ -1916,9 +2613,9 @@ | ||
| 1916 | 2613 | } |
| 1917 | 2614 | |
| 1918 | 2615 | // CTA URL validation |
| 1919 | 2616 | if (!empty($settings['hero_cta_url'])) { |
| 1920 | - if (filter_var($settings['hero_cta_url'], FILTER_VALIDATE_URL) || strpos($settings['hero_cta_url'], '/') === 0) { | |
| 2617 | + if (\ThinkRank\Core\Url_Validator::is_http_url_or_path($settings['hero_cta_url'])) { | |
| 1921 | 2618 | $field_details[] = [ |
| 1922 | 2619 | 'field' => 'hero_cta_url', |
| 1923 | 2620 | 'label' => 'Call-to-action URL is properly configured.', |
| 1924 | 2621 | 'status' => 'valid', |
| @@ -1959,9 +2656,9 @@ | ||
| 1959 | 2656 | } |
| 1960 | 2657 | |
| 1961 | 2658 | // Site Logo validation (from Site Assets section) |
| 1962 | 2659 | if (!empty($settings['logo_url'])) { |
| 1963 | - if (filter_var($settings['logo_url'], FILTER_VALIDATE_URL)) { | |
| 2660 | + if (\ThinkRank\Core\Url_Validator::is_http_url($settings['logo_url'])) { | |
| 1964 | 2661 | $field_details[] = [ |
| 1965 | 2662 | 'field' => 'logo_url', |
| 1966 | 2663 | 'label' => 'Site logo is properly configured.', |
| 1967 | 2664 | 'status' => 'valid', |
| @@ -2262,12 +2959,15 @@ | ||
| 2262 | 2959 | 'warnings' => [], |
| 2263 | 2960 | 'suggestions' => [] |
| 2264 | 2961 | ]; |
| 2265 | 2962 | |
| 2266 | - // Validate business name (required for local SEO) | |
| 2963 | + // Business name is what makes the LocalBusiness schema useful, but it | |
| 2964 | + // cannot be a blocking error: the toggle is what reveals the business | |
| 2965 | + // fields, so requiring the name up front makes enabling Local SEO | |
| 2966 | + // impossible. The frontend already skips the output while the name is | |
| 2967 | + // empty (see Seo_Manager::output_local_seo_meta_tags()). | |
| 2267 | 2968 | if (empty($settings['business_name'])) { |
| 2268 | - $validation['errors'][] = 'Business name is required when local SEO is enabled'; | |
| 2269 | - $validation['valid'] = false; | |
| 2969 | + $validation['warnings'][] = 'Business name is missing - required before local business schema is output'; | |
| 2270 | 2970 | } elseif (strlen($settings['business_name']) > 100) { |
| 2271 | 2971 | $validation['warnings'][] = 'Business name is very long, consider shortening for better display'; |
| 2272 | 2972 | } |
| 2273 | 2973 | |
| @@ -2324,11 +3024,15 @@ | ||
| 2324 | 3024 | } else { |
| 2325 | 3025 | $validation['suggestions'][] = 'Add business hours to improve local search visibility'; |
| 2326 | 3026 | } |
| 2327 | 3027 | |
| 2328 | - // Validate business type | |
| 2329 | - if (empty($settings['business_type'])) { | |
| 2330 | - $validation['suggestions'][] = 'Select a specific business type for better schema markup'; | |
| 3028 | + // Business type, through the shared rule (#622). This is the only place | |
| 3029 | + // it is reported on the generic path: validate_settings() with no tab | |
| 3030 | + // context attaches basic-info field details, not business-info ones, so | |
| 3031 | + // without this the setting would go unreported there entirely. | |
| 3032 | + $business_type = $this->business_type_status($settings); | |
| 3033 | + if ('suggestion' === $business_type['status']) { | |
| 3034 | + $validation['suggestions'][] = $business_type['message']; | |
| 2331 | 3035 | } |
| 2332 | 3036 | |
| 2333 | 3037 | return $validation; |
| 2334 | 3038 | } |
| @@ -2408,8 +3112,201 @@ | ||
| 2408 | 3112 | return $output; |
| 2409 | 3113 | } |
| 2410 | 3114 | |
| 2411 | 3115 | /** |
| 3116 | + * Keys the Site Identity screens store beyond the 16 defaults. | |
| 3117 | + * | |
| 3118 | + * Title formats, breadcrumb configuration, the hero fields, the business | |
| 3119 | + * block and the wizard's identity fields are all real settings written by | |
| 3120 | + * this manager, none of which get_default_settings() names — it seeds only | |
| 3121 | + * the values a fresh install needs. Gating on defaults alone would stop | |
| 3122 | + * every one of them saving (#452). | |
| 3123 | + * | |
| 3124 | + * @since 2.0.1 | |
| 3125 | + * | |
| 3126 | + * @return string[] | |
| 3127 | + */ | |
| 3128 | + /** | |
| 3129 | + * The stored alternate name(s), shaped for schema output. | |
| 3130 | + * | |
| 3131 | + * schema.org and Google both allow `alternateName` to carry one value or | |
| 3132 | + * several, and the store already round-trips either shape, so this accepts | |
| 3133 | + * both and normalises: null when there is nothing to publish, a bare string | |
| 3134 | + * for one name, a list for more. Emitting a one-element array would be | |
| 3135 | + * valid but noisier than it needs to be. | |
| 3136 | + * | |
| 3137 | + * Shared because both WebSite producers need it and must agree — a property | |
| 3138 | + * added to one and not the other is how #688 happened. | |
| 3139 | + * | |
| 3140 | + * @since 2.7.0 | |
| 3141 | + * | |
| 3142 | + * @param mixed $value Stored alternate_name value. | |
| 3143 | + * @return string|string[]|null | |
| 3144 | + */ | |
| 3145 | + public static function alternate_name_for_schema($value) { | |
| 3146 | + $names = []; | |
| 3147 | + | |
| 3148 | + foreach ((array) $value as $name) { | |
| 3149 | + if (!is_scalar($name)) { | |
| 3150 | + continue; | |
| 3151 | + } | |
| 3152 | + | |
| 3153 | + $name = trim((string) $name); | |
| 3154 | + | |
| 3155 | + if ('' !== $name && !in_array($name, $names, true)) { | |
| 3156 | + $names[] = $name; | |
| 3157 | + } | |
| 3158 | + } | |
| 3159 | + | |
| 3160 | + if (empty($names)) { | |
| 3161 | + return null; | |
| 3162 | + } | |
| 3163 | + | |
| 3164 | + return 1 === count($names) ? $names[0] : $names; | |
| 3165 | + } | |
| 3166 | + | |
| 3167 | + protected function additional_setting_keys(): array { | |
| 3168 | + return [ | |
| 3169 | + // Title formats, one per context. | |
| 3170 | + 'homepage_title', 'post_title', 'page_title', 'category_title', | |
| 3171 | + 'tag_title', 'author_title', 'search_title', 'archive_title', | |
| 3172 | + // The blog-index homepage's meta description (#897). | |
| 3173 | + 'homepage_description', | |
| 3174 | + // Breadcrumbs. | |
| 3175 | + 'breadcrumb_prefix', 'show_current_page', 'breadcrumb_use_seo_title', | |
| 3176 | + // Identity, as written by the setup wizard and the importers. | |
| 3177 | + 'alternate_name', 'identity_type', 'represents', | |
| 3178 | + 'default_meta_description', 'default_social_image', | |
| 3179 | + 'social_media_accounts', | |
| 3180 | + // Schema toggles that live on this screen. | |
| 3181 | + 'organization_schema', 'knowledge_graph', | |
| 3182 | + // Robots rules composed by the Robots.txt panel. | |
| 3183 | + 'custom_robots_rules', | |
| 3184 | + // Per-agent AI crawler allow/block map (#657). | |
| 3185 | + 'ai_crawler_rules', | |
| 3186 | + // Hero section. | |
| 3187 | + 'hero_title', 'hero_subtitle', 'hero_cta_text', 'hero_cta_url', | |
| 3188 | + 'hero_background_image', | |
| 3189 | + // Local SEO / business details. | |
| 3190 | + 'local_seo_enabled', 'business_type', 'business_name', | |
| 3191 | + 'business_address', 'business_city', 'business_state', | |
| 3192 | + 'business_postal_code', 'business_country', 'business_phone', | |
| 3193 | + 'business_email', 'business_latitude', 'business_longitude', | |
| 3194 | + 'business_price_range', 'business_hours', | |
| 3195 | + ]; | |
| 3196 | + } | |
| 3197 | + | |
| 3198 | + /** | |
| 3199 | + * Sanitize settings, normalising the AI crawler rule map. | |
| 3200 | + * | |
| 3201 | + * The generic array sanitizer keeps the shape but says nothing about the | |
| 3202 | + * values: a payload could store `ai_crawler_rules[gptbot] = "maybe"`, or a | |
| 3203 | + * slug no crawler answers to, and both would round-trip through every | |
| 3204 | + * later response. Normalising here rather than in the REST handler puts it | |
| 3205 | + * on the one path every writer shares — the settings route, the robots | |
| 3206 | + * route and the MCP abilities all land in save_settings() (#657). | |
| 3207 | + * | |
| 3208 | + * @since 2.5.0 | |
| 3209 | + * | |
| 3210 | + * @param array $settings Settings to sanitize. | |
| 3211 | + * @param string $context_type Context type. | |
| 3212 | + * @return array Sanitized settings. | |
| 3213 | + */ | |
| 3214 | + protected function sanitize_settings(array $settings, string $context_type = 'site'): array { | |
| 3215 | + $sanitized = parent::sanitize_settings($settings, $context_type); | |
| 3216 | + | |
| 3217 | + if (array_key_exists('ai_crawler_rules', $sanitized)) { | |
| 3218 | + $sanitized['ai_crawler_rules'] = AI_Crawlers::normalize_rules($sanitized['ai_crawler_rules']); | |
| 3219 | + } | |
| 3220 | + | |
| 3221 | + // Same reasoning one key up, for the scheme override (#638). Anything | |
| 3222 | + // that is not one of the three modes means "follow WordPress", and is | |
| 3223 | + // stored as that rather than kept verbatim — otherwise get-site-identity | |
| 3224 | + // -settings would report a scheme the site does not actually publish. | |
| 3225 | + if (array_key_exists('canonical_scheme', $sanitized)) { | |
| 3226 | + $sanitized['canonical_scheme'] = in_array($sanitized['canonical_scheme'], Url_Scheme::MODES, true) | |
| 3227 | + ? $sanitized['canonical_scheme'] | |
| 3228 | + : Url_Scheme::AUTOMATIC; | |
| 3229 | + } | |
| 3230 | + | |
| 3231 | + // Same reasoning again for the business type. It goes straight into | |
| 3232 | + // LocalBusiness schema, so a type that is not in the schema.org | |
| 3233 | + // vocabulary is invalid structured data — and storing it verbatim would | |
| 3234 | + // have get-site-identity-settings report a type the site cannot | |
| 3235 | + // actually publish. An empty value keeps meaning "not set"; anything | |
| 3236 | + // else unrecognised falls back to the general-purpose root (#623). | |
| 3237 | + if (array_key_exists('business_type', $sanitized)) { | |
| 3238 | + $type = (string) $sanitized['business_type']; | |
| 3239 | + | |
| 3240 | + if ('' !== $type && !\ThinkRank\Config\Local_Business_Types_Config::is_valid($type)) { | |
| 3241 | + $type = \ThinkRank\Config\Local_Business_Types_Config::ROOT; | |
| 3242 | + } | |
| 3243 | + | |
| 3244 | + $sanitized['business_type'] = $type; | |
| 3245 | + } | |
| 3246 | + | |
| 3247 | + return $sanitized; | |
| 3248 | + } | |
| 3249 | + | |
| 3250 | + /** | |
| 3251 | + * schema.org's general-purpose LocalBusiness type. | |
| 3252 | + * | |
| 3253 | + * The default, the first option in the control, and a valid answer in its | |
| 3254 | + * own right — which is the whole point of #622. | |
| 3255 | + * | |
| 3256 | + * @since 2.10.0 | |
| 3257 | + * @var string | |
| 3258 | + */ | |
| 3259 | + private const GENERAL_BUSINESS_TYPE = 'LocalBusiness'; | |
| 3260 | + | |
| 3261 | + /** | |
| 3262 | + * The one rule for whether a business type needs the user's attention. | |
| 3263 | + * | |
| 3264 | + * There were three, with two wordings and two different conditions. Two | |
| 3265 | + * fired when the value was empty; the third fired when it WAS | |
| 3266 | + * `LocalBusiness` — which is the default, the first option in the control | |
| 3267 | + * and a perfectly valid schema.org type. So the warning appeared out of the | |
| 3268 | + * box for every site, could not be cleared without choosing a type that | |
| 3269 | + * might be inaccurate, and on an empty value it appeared three times in two | |
| 3270 | + * different phrasings, which is why it was reported as showing twice (#622). | |
| 3271 | + * | |
| 3272 | + * The rule now: a type is expected, and any type in the vocabulary is a | |
| 3273 | + * correct answer. Only an unset value is worth prompting about. | |
| 3274 | + * `LocalBusiness` is the general-purpose answer and is accepted as one — | |
| 3275 | + * with a note that a more specific type sharpens the schema, phrased as the | |
| 3276 | + * guidance it is rather than as a fault the user has to clear. | |
| 3277 | + * | |
| 3278 | + * @since 2.10.0 | |
| 3279 | + * | |
| 3280 | + * @param array $settings Site identity settings. | |
| 3281 | + * @return array{status:string,message:string} `valid` or `suggestion`. | |
| 3282 | + */ | |
| 3283 | + private function business_type_status(array $settings): array { | |
| 3284 | + $type = trim((string) ($settings['business_type'] ?? '')); | |
| 3285 | + | |
| 3286 | + if ('' === $type) { | |
| 3287 | + return [ | |
| 3288 | + 'status' => 'suggestion', | |
| 3289 | + 'message' => __('Select a business type so your local schema describes the right kind of business.', 'thinkrank'), | |
| 3290 | + ]; | |
| 3291 | + } | |
| 3292 | + | |
| 3293 | + // The literal rather than a constant from the expanded type list (#623): | |
| 3294 | + // that lands on its own branch, and this fix must not wait on it. | |
| 3295 | + if (self::GENERAL_BUSINESS_TYPE === $type) { | |
| 3296 | + return [ | |
| 3297 | + 'status' => 'valid', | |
| 3298 | + 'message' => __('Business type is set to Local Business. A more specific type sharpens your schema, if one fits.', 'thinkrank'), | |
| 3299 | + ]; | |
| 3300 | + } | |
| 3301 | + | |
| 3302 | + return [ | |
| 3303 | + 'status' => 'valid', | |
| 3304 | + 'message' => __('Business type is selected for proper schema markup.', 'thinkrank'), | |
| 3305 | + ]; | |
| 3306 | + } | |
| 3307 | + | |
| 3308 | + /** | |
| 2412 | 3309 | * Get default settings for a context type (implements interface) |
| 2413 | 3310 | * |
| 2414 | 3311 | * @since 1.0.0 |
| 2415 | 3312 | * |
| @@ -2429,9 +3326,32 @@ | ||
| 2429 | 3326 | 'breadcrumb_home_text' => 'Home', |
| 2430 | 3327 | 'breadcrumb_separator' => '>', |
| 2431 | 3328 | 'robots_txt_enabled' => true, |
| 2432 | 3329 | 'allow_search_engines' => true, |
| 3330 | + // Answer 404 when a content selector in the URL resolved to | |
| 3331 | + // nothing (#634). On by default, unlike the other new settings | |
| 3332 | + // here: it changes no URL a visitor or a correct crawler uses, only | |
| 3333 | + // ones where WordPress resolved nothing and served the blog listing | |
| 3334 | + // at 200 anyway. | |
| 3335 | + 'query_protection' => true, | |
| 3336 | + | |
| 3337 | + // Feed controls (#635). All three off, so an upgrade changes | |
| 3338 | + // nothing about what an existing site already sends its | |
| 3339 | + // subscribers; a brand-new install is seeded with the signature and | |
| 3340 | + // the noindex on, in Activator::seed_feed_defaults(). | |
| 3341 | + 'feed_excerpt_only' => false, | |
| 3342 | + 'feed_source_link' => false, | |
| 3343 | + 'feed_noindex' => false, | |
| 3344 | + | |
| 3345 | + // The scheme self-referential URLs go out with (#638). 'automatic' | |
| 3346 | + // means substitute nothing and follow WordPress, which is what | |
| 3347 | + // every site did before the setting existed. | |
| 3348 | + 'canonical_scheme' => Url_Scheme::AUTOMATIC, | |
| 2433 | 3349 | 'robots_txt_content' => '', |
| 3350 | + // Empty map = every AI crawler allowed. Defaults must stay | |
| 3351 | + // permissive so an upgrade never starts blocking a crawler a site | |
| 3352 | + // was happily serving (#657). | |
| 3353 | + 'ai_crawler_rules' => [], | |
| 2434 | 3354 | 'logo_url' => '', |
| 2435 | 3355 | 'favicon_url' => '', |
| 2436 | 3356 | 'apple_touch_icon_url' => '' |
| 2437 | 3357 | ]; |
| @@ -2550,10 +3470,13 @@ | ||
| 2550 | 3470 | */ |
| 2551 | 3471 | private function prepare_title_placeholders(array $data, string $context, array $settings): array { |
| 2552 | 3472 | $placeholders = [ |
| 2553 | 3473 | '%title%' => $data['title'] ?? '', |
| 2554 | - '%sitename%' => $settings['site_name'] ?? get_bloginfo('name'), | |
| 2555 | - '%tagline%' => $settings['tagline'] ?? get_bloginfo('description'), | |
| 3474 | + // `?:` rather than `??`: these are persisted as '' rather than left | |
| 3475 | + // unset, and '' is not null, so the null-coalesce never reached the | |
| 3476 | + // WordPress fallback (#398). | |
| 3477 | + '%sitename%' => ($settings['site_name'] ?? '') ?: get_bloginfo('name'), | |
| 3478 | + '%tagline%' => ($settings['tagline'] ?? '') ?: get_bloginfo('description'), | |
| 2556 | 3479 | '%separator%' => '', // Will be replaced with actual separator |
| 2557 | 3480 | '%category%' => '', |
| 2558 | 3481 | '%author%' => '', |
| 2559 | 3482 | '%date%' => '', |
| @@ -2637,16 +3560,17 @@ | ||
| 2637 | 3560 | // Remove extra whitespace |
| 2638 | 3561 | $title = preg_replace('/\s+/', ' ', $title); |
| 2639 | 3562 | $title = trim($title); |
| 2640 | 3563 | |
| 2641 | - // Ensure title is not too long (60 characters max for SEO) | |
| 2642 | - if (strlen($title) > 60) { | |
| 2643 | - // Try to truncate at word boundary | |
| 2644 | - $title = wp_trim_words($title, 8, '...'); | |
| 2645 | - if (strlen($title) > 60) { | |
| 2646 | - $title = substr($title, 0, 57) . '...'; | |
| 2647 | - } | |
| 2648 | - } | |
| 3564 | + // Ensure title is not too long (60 characters max for SEO). | |
| 3565 | + // All three units here were wrong for non-Latin text: strlen() counts | |
| 3566 | + // BYTES so the gate fired at 20 Thai characters, wp_trim_words() counts | |
| 3567 | + // CHARACTERS on th/ja/zh_* so `8` cut the title to 8 of them, and | |
| 3568 | + // substr() cuts bytes so it split a character mid-sequence (#687). | |
| 3569 | + $title = \ThinkRank\Core\Seo_Text::trim_to_length( | |
| 3570 | + $title, | |
| 3571 | + \ThinkRank\Core\Seo_Text::TITLE_MAX_LENGTH | |
| 3572 | + ); | |
| 2649 | 3573 | |
| 2650 | 3574 | // Ensure title is not empty |
| 2651 | 3575 | if (empty($title)) { |
| 2652 | 3576 | $title = get_bloginfo('name'); |
| @@ -3036,8 +3960,29 @@ | ||
| 3036 | 3960 | * @since 1.0.0 |
| 3037 | 3961 | * @return array Array of sitemap URLs |
| 3038 | 3962 | */ |
| 3039 | 3963 | private function get_sitemap_urls_for_robots(): array { |
| 3964 | + // One wrapper over every return path below, including the #104 extras. | |
| 3965 | + // The Sitemap: line is the only absolute URL of ours in robots.txt and | |
| 3966 | + // the one a crawler follows to find everything else, so it has to carry | |
| 3967 | + // the site's scheme preference (#638). Applied here rather than where | |
| 3968 | + // the body is assembled, because that path also renders a robots.txt a | |
| 3969 | + // site owner typed themselves, and their text is not ours to rewrite. | |
| 3970 | + return array_map( | |
| 3971 | + static function (string $url): string { | |
| 3972 | + return Url_Scheme::apply($url); | |
| 3973 | + }, | |
| 3974 | + $this->collect_sitemap_urls_for_robots() | |
| 3975 | + ); | |
| 3976 | + } | |
| 3977 | + | |
| 3978 | + /** | |
| 3979 | + * The sitemap URLs robots.txt advertises, before the scheme preference. | |
| 3980 | + * | |
| 3981 | + * @since 1.0.0 | |
| 3982 | + * @return array Array of sitemap URLs | |
| 3983 | + */ | |
| 3984 | + private function collect_sitemap_urls_for_robots(): array { | |
| 3040 | 3985 | try { |
| 3041 | 3986 | // Get sitemap settings |
| 3042 | 3987 | $sitemap_generator = new \ThinkRank\SEO\Sitemap_Generator(); |
| 3043 | 3988 | $sitemap_settings = $sitemap_generator->get_settings('site'); |
| @@ -3049,29 +3994,88 @@ | ||
| 3049 | 3994 | |
| 3050 | 3995 | $sitemap_urls = []; |
| 3051 | 3996 | $site_url = home_url(); |
| 3052 | 3997 | |
| 3053 | - // Extract enabled sitemap URLs | |
| 3998 | + // Extract enabled sitemap URLs. When the index is enabled it is the | |
| 3999 | + // only entry worth advertising: every child sitemap is already | |
| 4000 | + // listed inside it, so naming them again in robots.txt is pure | |
| 4001 | + // redundancy and drifts out of date as soon as a post type is added. | |
| 4002 | + $index_url = ''; | |
| 3054 | 4003 | if (!empty($sitemap_settings['sitemap_urls']) && is_array($sitemap_settings['sitemap_urls'])) { |
| 3055 | 4004 | foreach ($sitemap_settings['sitemap_urls'] as $sitemap) { |
| 3056 | - if (!empty($sitemap['enabled']) && !empty($sitemap['url'])) { | |
| 3057 | - $sitemap_urls[] = $site_url . $sitemap['url']; | |
| 4005 | + if (empty($sitemap['enabled']) || empty($sitemap['url'])) { | |
| 4006 | + continue; | |
| 3058 | 4007 | } |
| 4008 | + | |
| 4009 | + if (($sitemap['type'] ?? '') === 'index') { | |
| 4010 | + $index_url = $site_url . $sitemap['url']; | |
| 4011 | + continue; | |
| 4012 | + } | |
| 4013 | + | |
| 4014 | + $sitemap_urls[] = $site_url . $sitemap['url']; | |
| 3059 | 4015 | } |
| 3060 | 4016 | } |
| 3061 | 4017 | |
| 4018 | + if ($index_url !== '') { | |
| 4019 | + // The index covers the children and, on a segmented install, | |
| 4020 | + // the local business sitemap too. | |
| 4021 | + // | |
| 4022 | + // It does not cover a sitemap contributed through | |
| 4023 | + // `thinkrank_additional_sitemaps`: the index is built by this | |
| 4024 | + // plugin's own generator and never lists them. Returning the | |
| 4025 | + // index alone therefore left a contributed sitemap with no | |
| 4026 | + // discovery path at all — absent from robots.txt and absent | |
| 4027 | + // from the index — so Pro's News sitemap was unreachable on any | |
| 4028 | + // install with the index enabled, which is the default (#835). | |
| 4029 | + $contributed = []; | |
| 4030 | + | |
| 4031 | + foreach (\ThinkRank\SEO\Sitemap_Generator::additional_sitemaps() as $path) { | |
| 4032 | + $url = home_url($path); | |
| 4033 | + | |
| 4034 | + if ($url !== $index_url && !in_array($url, $contributed, true)) { | |
| 4035 | + $contributed[] = $url; | |
| 4036 | + } | |
| 4037 | + } | |
| 4038 | + | |
| 4039 | + return array_merge([$index_url], $contributed); | |
| 4040 | + } | |
| 4041 | + | |
| 3062 | 4042 | // Fallback to default if no URLs found |
| 3063 | 4043 | if (empty($sitemap_urls)) { |
| 3064 | 4044 | $sitemap_urls[] = home_url('/sitemap.xml'); |
| 3065 | 4045 | } |
| 3066 | 4046 | |
| 3067 | - // Advertise the local business sitemap when it exists. In segmented | |
| 3068 | - // mode it is already listed inside the sitemap index; in single mode | |
| 3069 | - // there is no index, so robots.txt is its discovery path. | |
| 3070 | - if (file_exists(ABSPATH . 'local-sitemap.xml')) { | |
| 3071 | - $local_url = home_url('/local-sitemap.xml'); | |
| 3072 | - if (!in_array($local_url, $sitemap_urls, true)) { | |
| 3073 | - $sitemap_urls[] = $local_url; | |
| 4047 | + // No index on this install, so anything not already listed above has | |
| 4048 | + // no other discovery path — advertise it directly. The local | |
| 4049 | + // business sitemap and the sitemaps other plugins register both land | |
| 4050 | + // here for the same reason, so they go through one list (#104). | |
| 4051 | + $extra = []; | |
| 4052 | + | |
| 4053 | + // Not a file test. Under dynamic delivery the local sitemap is | |
| 4054 | + // served from PHP and no file is ever written, so file_exists() | |
| 4055 | + // silently dropped a sitemap the site really does publish (#752). | |
| 4056 | + // On static sites the file is still what proves it, so both count. | |
| 4057 | + $local_sitemap_published = file_exists(ABSPATH . 'local-sitemap.xml'); | |
| 4058 | + | |
| 4059 | + if (!$local_sitemap_published && class_exists('ThinkRank\\SEO\\Sitemap_Generator')) { | |
| 4060 | + $generator = new \ThinkRank\SEO\Sitemap_Generator(false); | |
| 4061 | + | |
| 4062 | + $local_sitemap_published = 'dynamic' === $generator->resolve_delivery_mode() | |
| 4063 | + && $generator->publishes_local_sitemap(); | |
| 4064 | + } | |
| 4065 | + | |
| 4066 | + if ($local_sitemap_published) { | |
| 4067 | + $extra[] = '/local-sitemap.xml'; | |
| 4068 | + } | |
| 4069 | + | |
| 4070 | + foreach (\ThinkRank\SEO\Sitemap_Generator::additional_sitemaps() as $path) { | |
| 4071 | + $extra[] = $path; | |
| 4072 | + } | |
| 4073 | + | |
| 4074 | + foreach ($extra as $path) { | |
| 4075 | + $url = home_url($path); | |
| 4076 | + if (!in_array($url, $sitemap_urls, true)) { | |
| 4077 | + $sitemap_urls[] = $url; | |
| 3074 | 4078 | } |
| 3075 | 4079 | } |
| 3076 | 4080 | |
| 3077 | 4081 | return $sitemap_urls; |
| @@ -3123,9 +4127,12 @@ | ||
| 3123 | 4127 | $validation['warnings'][] = "Path '{$value}' should start with '/'"; |
| 3124 | 4128 | } |
| 3125 | 4129 | break; |
| 3126 | 4130 | case 'sitemap': |
| 3127 | - if (!filter_var($value, FILTER_VALIDATE_URL)) { | |
| 4131 | + // Url_Validator, not the raw PHP filter: on a site with an | |
| 4132 | + // internationalised domain the site's own sitemap URL is | |
| 4133 | + // non-ASCII and the raw filter refused it. | |
| 4134 | + if (!\ThinkRank\Core\Url_Validator::is_valid($value)) { | |
| 3128 | 4135 | $validation['errors'][] = "Invalid sitemap URL: {$value}"; |
| 3129 | 4136 | $validation['valid'] = false; |
| 3130 | 4137 | } |
| 3131 | 4138 | break; |
| @@ -3161,8 +4168,9 @@ | ||
| 3161 | 4168 | // timestamp on every update. |
| 3162 | 4169 | $content = ''; |
| 3163 | 4170 | |
| 3164 | 4171 | $current_user_agent = ''; |
| 4172 | + $sitemap_started = false; | |
| 3165 | 4173 | |
| 3166 | 4174 | foreach ($rules as $rule) { |
| 3167 | 4175 | $directive = $rule['directive'] ?? ''; |
| 3168 | 4176 | $value = $rule['value'] ?? ''; |
| @@ -3183,9 +4191,17 @@ | ||
| 3183 | 4191 | case 'crawl_delay': |
| 3184 | 4192 | $content .= "Crawl-delay: {$value}\n"; |
| 3185 | 4193 | break; |
| 3186 | 4194 | case 'sitemap': |
| 3187 | - $content .= "\nSitemap: {$value}\n"; | |
| 4195 | + // One blank line separates the Sitemap block from the | |
| 4196 | + // preceding group, and none appear inside it. A blank line | |
| 4197 | + // terminates a record in the robots.txt grammar, so putting | |
| 4198 | + // one between every directive was invalid formatting. | |
| 4199 | + if (!$sitemap_started) { | |
| 4200 | + $content .= "\n"; | |
| 4201 | + $sitemap_started = true; | |
| 4202 | + } | |
| 4203 | + $content .= "Sitemap: {$value}\n"; | |
| 3188 | 4204 | break; |
| 3189 | 4205 | } |
| 3190 | 4206 | } |
| 3191 | 4207 | |
| @@ -3192,8 +4208,171 @@ | ||
| 3192 | 4208 | return ltrim($content, "\n"); |
| 3193 | 4209 | } |
| 3194 | 4210 | |
| 3195 | 4211 | /** |
| 4212 | + * Parse a robots.txt body back into the {directive, value} rule shape. | |
| 4213 | + * | |
| 4214 | + * generate_robots_txt() returns `rules` alongside `content`, but callers | |
| 4215 | + * replace `content` with the body actually being served (a stored override | |
| 4216 | + * or a physical file). The generated rules then described something the | |
| 4217 | + * response no longer contained. Re-deriving them from the served body keeps | |
| 4218 | + * the two halves of the payload describing the same document. | |
| 4219 | + * | |
| 4220 | + * @since 2.0.1 | |
| 4221 | + * | |
| 4222 | + * @param string $content Robots.txt body (header optional). | |
| 4223 | + * @return array<int, array{directive: string, value: string}> Parsed rules. | |
| 4224 | + */ | |
| 4225 | + public function parse_robots_txt_rules(string $content): array { | |
| 4226 | + $map = [ | |
| 4227 | + 'user-agent' => 'user_agent', | |
| 4228 | + 'disallow' => 'disallow', | |
| 4229 | + 'allow' => 'allow', | |
| 4230 | + 'crawl-delay' => 'crawl_delay', | |
| 4231 | + 'sitemap' => 'sitemap', | |
| 4232 | + ]; | |
| 4233 | + | |
| 4234 | + $rules = []; | |
| 4235 | + | |
| 4236 | + foreach (preg_split('/\r\n|\r|\n/', $this->strip_robots_header($content)) as $line) { | |
| 4237 | + $line = trim($line); | |
| 4238 | + | |
| 4239 | + // Blank lines separate groups and `#` starts a comment; neither is | |
| 4240 | + // a rule. | |
| 4241 | + if ($line === '' || str_starts_with($line, '#')) { | |
| 4242 | + continue; | |
| 4243 | + } | |
| 4244 | + | |
| 4245 | + $parts = explode(':', $line, 2); | |
| 4246 | + if (count($parts) !== 2) { | |
| 4247 | + continue; | |
| 4248 | + } | |
| 4249 | + | |
| 4250 | + $field = strtolower(trim($parts[0])); | |
| 4251 | + if (!isset($map[$field])) { | |
| 4252 | + continue; | |
| 4253 | + } | |
| 4254 | + | |
| 4255 | + $rules[] = [ | |
| 4256 | + 'directive' => $map[$field], | |
| 4257 | + // Sitemap values are absolute URLs and contain the `:` the | |
| 4258 | + // limited explode above deliberately preserved. | |
| 4259 | + 'value' => trim($parts[1]), | |
| 4260 | + ]; | |
| 4261 | + } | |
| 4262 | + | |
| 4263 | + return $rules; | |
| 4264 | + } | |
| 4265 | + | |
| 4266 | + /** | |
| 4267 | + * Opening fence of the machine-owned AI crawler region. | |
| 4268 | + * | |
| 4269 | + * @since 2.5.0 | |
| 4270 | + * @var string | |
| 4271 | + */ | |
| 4272 | + public const AI_BLOCK_BEGIN = '# BEGIN ThinkRank AI crawlers'; | |
| 4273 | + | |
| 4274 | + /** | |
| 4275 | + * Closing fence of the machine-owned AI crawler region. | |
| 4276 | + * | |
| 4277 | + * @since 2.5.0 | |
| 4278 | + * @var string | |
| 4279 | + */ | |
| 4280 | + public const AI_BLOCK_END = '# END ThinkRank AI crawlers'; | |
| 4281 | + | |
| 4282 | + /** | |
| 4283 | + * Render the fenced AI crawler region for the current settings. | |
| 4284 | + * | |
| 4285 | + * One `User-agent:` / `Disallow: /` record per blocked crawler. Allowed | |
| 4286 | + * crawlers emit nothing at all: `Disallow:` with an empty value is the | |
| 4287 | + * robots.txt way of saying "allow everything", but writing eighteen such | |
| 4288 | + * records to say what silence already says would triple the file and | |
| 4289 | + * invite the reading that an unlisted crawler is therefore refused. | |
| 4290 | + * | |
| 4291 | + * @since 2.5.0 | |
| 4292 | + * | |
| 4293 | + * @param array $settings Site settings. | |
| 4294 | + * @return string Fenced block, newline-terminated, or '' when nothing is blocked. | |
| 4295 | + */ | |
| 4296 | + private function build_ai_crawler_block(array $settings): string { | |
| 4297 | + $blocked = AI_Crawlers::blocked_slugs($settings['ai_crawler_rules'] ?? []); | |
| 4298 | + | |
| 4299 | + if (empty($blocked)) { | |
| 4300 | + return ''; | |
| 4301 | + } | |
| 4302 | + | |
| 4303 | + $agents = AI_Crawlers::all(); | |
| 4304 | + | |
| 4305 | + $lines = [ | |
| 4306 | + self::AI_BLOCK_BEGIN, | |
| 4307 | + '# Managed by ThinkRank — edits between these lines are overwritten.', | |
| 4308 | + ]; | |
| 4309 | + | |
| 4310 | + foreach ($blocked as $slug) { | |
| 4311 | + $lines[] = ''; | |
| 4312 | + $lines[] = 'User-agent: ' . $agents[$slug]['token']; | |
| 4313 | + $lines[] = 'Disallow: /'; | |
| 4314 | + } | |
| 4315 | + | |
| 4316 | + $lines[] = self::AI_BLOCK_END; | |
| 4317 | + | |
| 4318 | + return implode("\n", $lines) . "\n"; | |
| 4319 | + } | |
| 4320 | + | |
| 4321 | + /** | |
| 4322 | + * Remove the fenced AI crawler region from a robots.txt body. | |
| 4323 | + * | |
| 4324 | + * Tolerates a missing closing fence rather than leaving the rest of the | |
| 4325 | + * file swallowed: a truncated write, or someone deleting the END line by | |
| 4326 | + * hand, would otherwise make every subsequent read drop everything below | |
| 4327 | + * the opening fence. | |
| 4328 | + * | |
| 4329 | + * @since 2.5.0 | |
| 4330 | + * | |
| 4331 | + * @param string $body Robots.txt body. | |
| 4332 | + * @return string Body with the region removed. | |
| 4333 | + */ | |
| 4334 | + public function strip_ai_crawler_block(string $body): string { | |
| 4335 | + if (false === strpos($body, self::AI_BLOCK_BEGIN)) { | |
| 4336 | + return $body; | |
| 4337 | + } | |
| 4338 | + | |
| 4339 | + $pattern = '/\R*' . preg_quote(self::AI_BLOCK_BEGIN, '/') | |
| 4340 | + . '.*?(?:' . preg_quote(self::AI_BLOCK_END, '/') . '|\z)\R*/s'; | |
| 4341 | + | |
| 4342 | + return trim((string) preg_replace($pattern, "\n\n", $body, 1)); | |
| 4343 | + } | |
| 4344 | + | |
| 4345 | + /** | |
| 4346 | + * Put the current AI crawler region into a robots.txt body. | |
| 4347 | + * | |
| 4348 | + * Replaces an existing region in place so the block keeps its position in | |
| 4349 | + * a hand-ordered file, and appends when there is none. Everything outside | |
| 4350 | + * the fences is returned untouched — that is the whole point of fencing | |
| 4351 | + * it, since the body is also a free-text field the user edits. | |
| 4352 | + * | |
| 4353 | + * @since 2.5.0 | |
| 4354 | + * | |
| 4355 | + * @param string $body Robots.txt body (fences optional). | |
| 4356 | + * @param array $settings Site settings. | |
| 4357 | + * @return string Body carrying the current region. | |
| 4358 | + */ | |
| 4359 | + private function apply_ai_crawler_block(string $body, array $settings): string { | |
| 4360 | + $stripped = $this->strip_ai_crawler_block($body); | |
| 4361 | + $block = $this->build_ai_crawler_block($settings); | |
| 4362 | + | |
| 4363 | + if ('' === $block) { | |
| 4364 | + return $stripped; | |
| 4365 | + } | |
| 4366 | + | |
| 4367 | + if ('' === trim($stripped)) { | |
| 4368 | + return trim($block); | |
| 4369 | + } | |
| 4370 | + | |
| 4371 | + return rtrim($stripped) . "\n\n" . trim($block); | |
| 4372 | + } | |
| 4373 | + | |
| 4374 | + /** | |
| 3196 | 4375 | * The auto-generated header prepended to the served robots.txt. |
| 3197 | 4376 | * |
| 3198 | 4377 | * Kept separate from the body so it is only ever added at render time with |
| 3199 | 4378 | * a fresh timestamp, never stored or shown in the editable textarea. |
| @@ -3230,11 +4409,16 @@ | ||
| 3230 | 4409 | */ |
| 3231 | 4410 | public function get_served_robots_body(): string { |
| 3232 | 4411 | $settings = $this->get_settings('site'); |
| 3233 | 4412 | |
| 4413 | + // The AI block is stripped from every one of these paths. A physical | |
| 4414 | + // robots.txt we wrote carries it, and the stored override is whatever | |
| 4415 | + // the textarea last held — so without this the block round-trips into | |
| 4416 | + // the editor, gets saved as ordinary body text, and is then appended | |
| 4417 | + // to a second time on the next render. | |
| 3234 | 4418 | $custom = trim((string) ($settings['robots_txt_content'] ?? '')); |
| 3235 | 4419 | if ($custom !== '') { |
| 3236 | - return $this->strip_robots_header($custom); | |
| 4420 | + return $this->strip_ai_crawler_block($this->strip_robots_header($custom)); | |
| 3237 | 4421 | } |
| 3238 | 4422 | |
| 3239 | 4423 | $robots_file = ABSPATH . 'robots.txt'; |
| 3240 | 4424 | if (file_exists($robots_file)) { |
| @@ -3240,13 +4424,13 @@ | ||
| 3240 | 4424 | if (file_exists($robots_file)) { |
| 3241 | 4425 | // phpcs:ignore WordPress.WP.AlternativeFunctions.file_get_contents_file_get_contents, WordPress.PHP.NoSilencedErrors.Discouraged -- an unreadable robots.txt is an expected state answered with an empty string. |
| 3242 | 4426 | $raw = (string) @file_get_contents($robots_file); |
| 3243 | 4427 | if ($raw !== '') { |
| 3244 | - return $this->strip_robots_header($raw); | |
| 4428 | + return $this->strip_ai_crawler_block($this->strip_robots_header($raw)); | |
| 3245 | 4429 | } |
| 3246 | 4430 | } |
| 3247 | 4431 | |
| 3248 | - return trim($this->generate_robots_txt()['content']); | |
| 4432 | + return $this->strip_ai_crawler_block(trim($this->generate_robots_txt()['content'])); | |
| 3249 | 4433 | } |
| 3250 | 4434 | private function get_site_identity_data(array $settings): array { |
| 3251 | 4435 | return [ |
| 3252 | 4436 | 'site_name' => $settings['site_name'] ?? get_bloginfo('name'), |
| @@ -3308,11 +4492,14 @@ | ||
| 3308 | 4492 | $optimization['validation']['valid'] = false; |
| 3309 | 4493 | } |
| 3310 | 4494 | |
| 3311 | 4495 | if (!empty($value) && isset($config['max_length'])) { |
| 3312 | - if (strlen($value) > $config['max_length']) { | |
| 4496 | + // The warning says "characters", so measure and cut in characters: | |
| 4497 | + // strlen()/substr() fired early on non-Latin values and the | |
| 4498 | + // suggested replacement was cut mid-character (#687). | |
| 4499 | + if (mb_strlen($value) > $config['max_length']) { | |
| 3313 | 4500 | $optimization['validation']['warnings'][] = "{$element} exceeds maximum length of {$config['max_length']} characters"; |
| 3314 | - $optimization['optimized_value'] = substr($value, 0, $config['max_length']); | |
| 4501 | + $optimization['optimized_value'] = \ThinkRank\Core\Seo_Text::trim_to_length($value, (int) $config['max_length']); | |
| 3315 | 4502 | } |
| 3316 | 4503 | } |
| 3317 | 4504 | |
| 3318 | 4505 | // SEO-specific optimizations |
| @@ -3344,25 +4531,27 @@ | ||
| 3344 | 4531 | return $optimization; |
| 3345 | 4532 | } |
| 3346 | 4533 | |
| 3347 | 4534 | // Validate URL |
| 3348 | - if (!filter_var($value, FILTER_VALIDATE_URL)) { | |
| 3349 | - $optimization['validation']['errors'][] = "{$element} must be a valid URL"; | |
| 4535 | + if (!\ThinkRank\Core\Url_Validator::is_http_url($value)) { | |
| 4536 | + $optimization['validation']['errors'][] = "{$element} must be an http or https URL"; | |
| 3350 | 4537 | $optimization['validation']['valid'] = false; |
| 3351 | 4538 | return $optimization; |
| 3352 | 4539 | } |
| 3353 | 4540 | |
| 3354 | 4541 | // Check if it's a local image |
| 3355 | - $attachment_id = attachment_url_to_postid($value); | |
| 4542 | + $attachment_id = Attachment_Lookup::id_from_url($value); | |
| 3356 | 4543 | if ($attachment_id) { |
| 3357 | 4544 | $image_meta = wp_get_attachment_metadata($attachment_id); |
| 3358 | 4545 | |
| 3359 | 4546 | if ($image_meta && isset($image_meta['width'], $image_meta['height'])) { |
| 3360 | - // Check recommended size | |
| 4547 | + // Check recommended size, against the configured file itself | |
| 4548 | + // rather than the upload it may have been generated from. | |
| 3361 | 4549 | if (isset($config['recommended_size'])) { |
| 3362 | 4550 | [$rec_width, $rec_height] = explode('x', $config['recommended_size']); |
| 4551 | + $image_file = Attachment_Lookup::describe($attachment_id, $value); | |
| 3363 | 4552 | |
| 3364 | - if ((int) $image_meta['width'] !== (int) $rec_width || (int) $image_meta['height'] !== (int) $rec_height) { | |
| 4553 | + if ($image_file['width'] !== (int) $rec_width || $image_file['height'] !== (int) $rec_height) { | |
| 3365 | 4554 | $optimization['suggestions'][] = "Consider using {$config['recommended_size']} size for optimal {$element}"; |
| 3366 | 4555 | } |
| 3367 | 4556 | } |
| 3368 | 4557 | |