$headers Extra request headers. */ private static function warm_args( string $device, array $headers = array() ): array { $headers['Sec-CH-UA-Mobile'] = 'mobile' === $device ? '?1' : '?0'; return array( 'timeout' => self::REQUEST_TIMEOUT, 'sslverify' => false, 'user-agent' => 'mobile' === $device ? self::mobile_user_agent() : self::user_agent(), 'headers' => Self_Traffic::headers( $headers ), 'blocking' => true, ); } /** * Warm every copy of $url that devices() names. * * @param array $headers Extra request headers. * @return array{failures: array, body: string} * The failures, and the desktop response body ('' when it failed). */ private static function warm_devices( string $url, array $headers = array() ): array { $failures = array(); $body = ''; foreach ( self::devices() as $device ) { $args = self::warm_args( $device, $headers ); $response = wp_remote_get( $url, $args ); if ( is_wp_error( $response ) ) { $failures[] = array( 'device' => $device, 'error' => $response->get_error_message() ); continue; } $code = (int) wp_remote_retrieve_response_code( $response ); if ( $code >= 400 ) { $failures[] = array( 'device' => $device, 'error' => self::failure_detail( $code, $args['user-agent'] ) ); if ( self::is_firewall_block( $code ) ) { self::remember_firewall_block( $url, $code, $args['user-agent'] ); } continue; } if ( 'desktop' === $device ) { $body = (string) wp_remote_retrieve_body( $response ); } } // The notice goes only once every copy got through. A host that // blocks just the phone user-agent would otherwise have the notice // cleared by each desktop warm right after the phone warm set it. if ( empty( $failures ) ) { self::clear_firewall_block(); } return array( 'failures' => $failures, 'body' => $body ); } /** "phone: " for a failure of the phone copy, so the log says which copy failed. */ private static function device_prefix( string $device ): string { return 'mobile' === $device ? 'phone: ' : ''; } /** * Is this status code the signature of a firewall refusing our warmer? * * 403 and 406 are what bad-bot rules (7G/8G, mod_security, Wordfence) * answer with. We only ever warm our OWN origin, and a page a visitor can * load must be loadable by us too — so these codes mean the request was * judged by its user-agent, not that the page is missing or broken. (#481) */ private static function is_firewall_block( int $code ): bool { return in_array( $code, array( 403, 406 ), true ); } /** * Explain a warm failure in terms the admin can act on. * * A bare "HTTP 403" sent people hunting a broken page; the page is fine, * and the fix is a server rule, so the message has to name the cause and * the exact UA to allow. (#481) */ private static function failure_detail( int $code, string $user_agent = '' ): string { if ( ! self::is_firewall_block( $code ) ) { return sprintf( 'HTTP %d', $code ); } return sprintf( 'HTTP %d — your server\'s firewall is blocking the xSpeed cache warmer by user-agent, so this page was not warmed. Allow the user-agent "%s" (on xCloud this is the 8G firewall\'s bad-bot rule), or change it with the xspeed_preloader_user_agent filter.', $code, '' !== $user_agent ? $user_agent : self::user_agent() ); } /** Option holding the last firewall-shaped warm refusal. */ public const FIREWALL_BLOCK_OPTION = 'xspeed_preloader_firewall_block'; /** * Record that the origin refused a warm by user-agent, for ui_notices(). * * An option rather than a transient: the condition is a server rule that * persists until someone changes it, and a notice that expired on its own * would let a site go back to never warming, silently. Cleared by * clear_firewall_block() on the first warm that succeeds. (#481) */ private static function remember_firewall_block( string $url, int $code, string $user_agent = '' ): void { if ( ! function_exists( 'update_option' ) ) { return; } update_option( self::FIREWALL_BLOCK_OPTION, array( 'url' => $url, 'code' => $code, 'user_agent' => '' !== $user_agent ? $user_agent : self::user_agent(), 'ts' => time(), ), false ); } /** Forget the firewall block once a warm gets through. */ public static function clear_firewall_block(): void { if ( function_exists( 'delete_option' ) && self::firewall_block() ) { delete_option( self::FIREWALL_BLOCK_OPTION ); } } /** The last firewall-shaped refusal, or null when there isn't one. */ public static function firewall_block(): ?array { if ( ! function_exists( 'get_option' ) ) { return null; } $block = get_option( self::FIREWALL_BLOCK_OPTION, null ); return ( is_array( $block ) && ! empty( $block['code'] ) ) ? $block : null; } /** * How many new remote images one warmed page may resolve. */ private static function remote_dimension_limit(): int { /** * Filter the per-page cap on remote dimension lookups. * * @param int $limit Default 20. Values below 1 disable the lookup. */ return (int) apply_filters( 'xspeed_preloader_remote_dimension_limit', self::REMOTE_DIMENSION_LIMIT ); } /** * Why the top-level sitemap fetch failed on this request, or '' when it * succeeded. Set by fetch_sitemap_urls(), read by resolve_queue() — the * reason has to survive the return of an empty array, which is exactly * what it could not do before. Request-scoped; never persisted. (#142) * * @var string */ private static $last_sitemap_error = ''; /** * The sitemap URL the last error refers to. Kept beside the message so * an error entry can carry a real `url` field like every other one, * rather than repeating the URL already inside the message text. */ private static $last_sitemap_url = ''; /** * How the queue for the current crawl was built — 'sitemap', 'fallback' * (enumerated from the database because the sitemap was unreachable), or * 'none'. Surfaced in the state so the panel, REST and CLI can each say * what actually happened instead of reporting a bare zero. (#142) * * @var string */ private static $queue_source = 'none'; /** Largest number of URLs the database fallback will enumerate. */ private const FALLBACK_LIMIT = 500; /** * Kick off a fresh crawl. Returns the initial state. */ public static function start(): array { $opts = Settings_Manager::get( 'preloader' ); $urls = self::resolve_queue( $opts ); // A crawl that queued nothing because the sitemap was unreachable is a // FAILURE, and every layer above needs to be able to say so. It used // to be indistinguishable from success: errors stayed empty, the REST // route returned 200, and the CLI printed a green Success. (#142) $sitemap_error = self::$last_sitemap_error; $errors = array(); if ( '' !== $sitemap_error && empty( $urls ) ) { // Same {url, error, ts} shape every other entry uses. A bare // string here fataled `wp xspeed preloader status`, which // destructures `$e['url']` over the list — and took the MCP // `get_preloader_status` tool down with it, so an agent asking // why the preload failed got "Cannot access offset of type // string on string" instead of the reason this code records. // `url` is the sitemap because that is what failed. (QA F1) $errors[] = array( 'url' => self::$last_sitemap_url, 'error' => $sitemap_error, 'ts' => time(), ); } $state = array( 'running' => ! empty( $urls ), 'started_at' => time(), 'finished_at' => empty( $urls ) ? time() : 0, 'queue' => array_values( $urls ), 'processed' => 0, 'total' => count( $urls ), 'last_url' => '', 'errors' => $errors, // Consumers render on these: the panel needs to distinguish // "not started" from "ran and found nothing", and to tell the // user when the queue came from the fallback rather than the // sitemap they configured. 'source' => self::$queue_source, 'sitemap_error' => $sitemap_error, // Which copies each URL gets, for the panel and the status // command. Each tick refreshes it: turning Separate Mobile // Cache on or off mid-crawl purges the cache, and the rest of // the crawl should fill the copies that now exist. 'devices' => self::devices(), ); set_transient( self::STATE_KEY, $state, self::STATE_TTL ); if ( '' !== $sitemap_error && 'fallback' === self::$queue_source ) { $message = sprintf( /* translators: 1: number of URLs, 2: the sitemap failure detail. */ __( 'Preloader queued %1$d URLs from the site content — %2$s', 'xspeed' ), $state['total'], $sitemap_error ); $severity = Activity_Log::WARN; } elseif ( '' !== $sitemap_error ) { $message = sprintf( /* translators: %s: the sitemap failure detail. */ __( 'Preloader could not start — %s', 'xspeed' ), $sitemap_error ); $severity = Activity_Log::WARN; } else { $message = sprintf( /* translators: 1: number of URLs, 2: plural suffix. */ __( 'Preloader queued %1$d URL%2$s for warming.', 'xspeed' ), $state['total'], 1 === $state['total'] ? '' : 's' ); $severity = $state['total'] > 0 ? Activity_Log::INFO : Activity_Log::WARN; } Activity_Log::record( 'preloader_started', $message, $severity ); // Schedule the first tick ~5 seconds out so the kick-off REST call // returns instantly; wp_schedule_single_event covers the // "process the queue ASAP" path without a heavy synchronous loop. if ( $state['running'] ) { wp_schedule_single_event( time() + 5, self::CRON_HOOK ); } return $state; } /** * Cancel an in-flight crawl. Idempotent. */ public static function stop(): array { $state = self::status(); if ( $state['running'] ) { Activity_Log::record( 'preloader_stopped', sprintf( 'Preloader stopped (%d/%d URLs warmed).', $state['processed'], $state['total'] ), Activity_Log::INFO ); } // Clear scheduled ticks. wp_clear_scheduled_hook( self::CRON_HOOK ); $state['running'] = false; $state['finished_at'] = time(); $state['queue'] = array(); set_transient( self::STATE_KEY, $state, self::STATE_TTL ); return $state; } public static function status(): array { $raw = get_transient( self::STATE_KEY ); if ( ! is_array( $raw ) ) { return self::empty_state(); } return wp_parse_args( $raw, self::empty_state() ); } private static function empty_state(): array { return array( 'running' => false, 'started_at' => 0, 'finished_at' => 0, 'queue' => array(), 'processed' => 0, 'total' => 0, 'last_url' => '', 'errors' => array(), ); } /** * Tick handler — pulls up to batch_size URLs off the queue, warms * each, persists state, and reschedules itself until the queue is * empty. Called via the xspeed_preloader_tick action. */ public static function tick(): void { $state = self::status(); if ( ! $state['running'] || empty( $state['queue'] ) ) { if ( $state['running'] ) { self::mark_complete( $state ); } return; } $opts = Settings_Manager::get( 'preloader' ); $batch = max( 1, min( 50, (int) ( $opts['batch_size'] ?? 5 ) ) ); /** * Filter: xspeed_preload_batch_size * * How much one tick may warm, in the units the batch size setting * counts. The setting is the site owner's * intent; this is for anything that knows a lower ceiling applies — * a CDN with a rate limit in front, say, which will answer a burst * with a challenge and leave the queue looking warmed when it is not. * * Only ever lowers. A filter that raised it would let an add-on * overrule a number the site owner chose, and the reason to reach for * this is always that something cannot take the current rate. * * @param int $batch The batch this tick would otherwise use. */ $ceiling = (int) apply_filters( 'xspeed_preload_batch_size', $batch ); if ( $ceiling > 0 && $ceiling < $batch ) { $batch = $ceiling; } // batch_size caps requests, not URLs, so the load on the origin per // tick is the same with the phone copy on: each URL costs one request // per device. A tick always warms at least one URL. $devices = self::devices(); $state['devices'] = $devices; $per_url = count( $devices ); $requests = 0; while ( ! empty( $state['queue'] ) && ( 0 === $requests || $requests + $per_url <= $batch ) ) { $url = (string) array_shift( $state['queue'] ); self::warm_url( $url, $state ); $state['processed']++; $state['last_url'] = $url; $requests += $per_url; } if ( empty( $state['queue'] ) ) { self::mark_complete( $state ); return; } // More to do — persist + reschedule. Slight delay to avoid // hammering the origin with parallel batches. set_transient( self::STATE_KEY, $state, self::STATE_TTL ); wp_schedule_single_event( time() + 10, self::CRON_HOOK ); } /** * Fire a single warm request for one URL with no queue / no cron * (the "content warmer" path: new post published → warm its URL * immediately). Records an activity event so the user can see in * the Health log that warming happened. * * Best-effort and non-blocking-feeling — uses a short timeout so a * dead origin can't hang the calling request. Returns true if the * fetch completed with a non-error status, false otherwise. */ public static function warm_one( string $url, string $cause = 'manual' ): bool { if ( '' === $url ) { return false; } $result = self::warm_devices( $url ); foreach ( $result['failures'] as $failure ) { Activity_Log::record( 'preloader_warm_failed', sprintf( 'Warm %s failed (%s): %s%s', $cause, $url, self::device_prefix( $failure['device'] ), $failure['error'] ), Activity_Log::WARN ); } if ( ! empty( $result['failures'] ) ) { return false; } Activity_Log::record( 'preloader_warmed_one', sprintf( 'Warmed %s (%s)', $url, $cause ), Activity_Log::INFO ); return true; } private static function warm_url( string $url, array &$state ): void { $result = self::warm_devices( $url, array( 'Accept' => 'text/html,application/xhtml+xml' ) ); foreach ( $result['failures'] as $failure ) { $state['errors'][] = array( 'url' => $url, 'error' => self::device_prefix( $failure['device'] ) . $failure['error'], 'ts' => time(), ); } // Cap retained errors so a broken sitemap doesn't blow the // transient size. $state['errors'] = array_slice( $state['errors'], -20 ); if ( '' !== $result['body'] ) { self::warm_remote_dimensions( $result['body'] ); } } /** * Resolve dimensions for externally hosted images found on a warmed page. * * The crawl already has the HTML in hand, so harvesting image URLs from it * costs nothing extra — and this is the one place where paying for a * remote lookup is free of consequence, because no visitor is waiting. * * An image on another domain has no local file to measure, so the front * end skips it and the page ships without width/height — which is layout * shift, on precisely the sites least able to fix it by hand (a CDN, a * sister site, a shared asset host). Warming here means the NEXT render * finds the dimensions in cache and stamps them, with the visitor paying * nothing. * * Deliberately bounded per page: a crawl should not turn into a scraper * for a page embedding hundreds of third-party images. * * @param string $html The warmed page's HTML. */ private static function warm_remote_dimensions( string $html ): void { if ( '' === $html || ! class_exists( '\XSpeed\Lazy_Loader' ) ) { return; } $opts = Settings_Manager::get( 'lazy' ); if ( empty( $opts['add_missing_dimensions'] ) ) { return; } // Match any , not only one carrying `src`. The URL worth warming // may live in a lazy attribute instead — which is the whole point of // #328 — and resolvable_image_url() below is what knows where to look. if ( ! preg_match_all( '#]*>#i', $html, $m, PREG_SET_ORDER ) ) { return; } $home = wp_parse_url( home_url(), PHP_URL_HOST ); $targets = array(); foreach ( $m as $tag ) { // Only tags MISSING a dimension are worth resolving — one that // already declares both needs nothing. // Same lookbehind as Lazy_Loader::ensure_dimensions(): a bare // `\bwidth=` also matches `data-width=`, so a slider carrying its // own metadata looked already-sized and was skipped from warming. // The two must agree, or the collector skips exactly the tags the // renderer still needs measured. (#333 review round 3, issue 2) if ( preg_match( '#(?= self::remote_dimension_limit() ) { break; } } if ( $targets ) { Lazy_Loader::warm_dimensions( array_keys( $targets ) ); } } private static function mark_complete( array $state ): void { $state['running'] = false; $state['finished_at'] = time(); $state['queue'] = array(); set_transient( self::STATE_KEY, $state, self::STATE_TTL ); Activity_Log::record( 'preloader_completed', sprintf( 'Preloader finished — %d/%d URLs warmed, %d error%s.', $state['processed'], $state['total'], count( $state['errors'] ), 1 === count( $state['errors'] ) ? '' : 's' ), empty( $state['errors'] ) ? Activity_Log::SUCCESS : Activity_Log::WARN ); } /** * Build the URL queue for a fresh crawl: parse the sitemap, follow * nested indexes, drop excluded paths. * * @return string[] */ private static function resolve_queue( array $opts ): array { // Request-scoped statics: reset so a previous crawl in the same // process can't leak its verdict into this one. self::$last_sitemap_error = ''; self::$last_sitemap_url = ''; self::$queue_source = 'none'; $sitemap = trim( (string) ( $opts['sitemap_url'] ?? '' ) ); if ( '' === $sitemap ) { $sitemap = home_url( '/wp-sitemap.xml' ); } $urls = self::fetch_sitemap_urls( $sitemap, 0 ); $from = empty( $urls ) ? 'none' : 'sitemap'; /* * A missing sitemap must not disable the feature. Two very common * setups produce one with no misconfiguration by the user: * `blog_public = 0` (WordPress disables /wp-sitemap.xml outright, * standard on staging and pre-launch sites), and an SEO plugin * filtering `wp_sitemaps_enabled` to false while serving its own * sitemap at a path we were never told about. * * Enumerate warmable URLs straight from the database instead. Only * on a genuine fetch FAILURE — a sitemap that is reachable and * legitimately empty is a real answer, and silently crawling * something else would be worse than doing nothing. (#142) */ if ( empty( $urls ) && '' !== self::$last_sitemap_error ) { $urls = self::fallback_urls(); $from = empty( $urls ) ? 'none' : 'fallback'; } $cache_opts = Settings_Manager::get( 'cache' ); $excluded = is_array( $cache_opts['excluded_urls'] ?? null ) ? $cache_opts['excluded_urls'] : array(); if ( ! empty( $excluded ) ) { $urls = array_filter( $urls, static function ( $u ) use ( $excluded ) { $path = (string) wp_parse_url( $u, PHP_URL_PATH ); foreach ( $excluded as $needle ) { if ( '' !== $needle && false !== strpos( $path, (string) $needle ) ) { return false; } } return true; } ); } // Dedup + cap at 5000 to bound the transient size on huge sites. $urls = array_values( array_unique( $urls ) ); $urls = array_slice( $urls, 0, 5000 ); /* * Commit the verdict only now, AFTER the exclusion filter — the * source describes what we ACTUALLY queued, not what we hoped to. * Setting it earlier let a queue that the exclusions stripped to * nothing still claim `fallback`, so the panel announced "Warmed * from site content" over 0 URLs, and `crawlFailed` (which needs * source !== 'fallback' at total 0) could never become true. * One assignment fixes both. (QA F2/F3 on #155) */ self::$queue_source = empty( $urls ) ? 'none' : $from; return $urls; } /** * Enumerate warmable URLs from the database, for sites whose sitemap * can't be fetched. Deliberately modest in scope: the home page, then * the most recently modified public posts across every public post type. * Newest-first is the right bias — those are the URLs most likely to be * requested and least likely to be warm already. * * Uses WP_Query rather than SQL so post-type registration, status * handling and multisite switching all behave the way the rest of * WordPress does. * * @return string[] */ private static function fallback_urls(): array { $urls = array(); $home = (string) home_url( '/' ); if ( '' !== trim( $home, '/' ) ) { $urls[] = $home; } $types = get_post_types( array( 'public' => true, 'publicly_queryable' => true, ) ); // `page` is public but not publicly_queryable, so the query above // misses it — and pages are exactly what a warm cache wants most. // Only add it when the site has post types at all: an empty list // means there is nothing to enumerate, and constructing a WP_Query // for it would be wasted work. if ( ! empty( $types ) ) { $types['page'] = 'page'; unset( $types['attachment'] ); } if ( empty( $types ) || ! class_exists( '\WP_Query' ) ) { return $urls; } $query = new \WP_Query( array( 'post_type' => array_values( $types ), 'post_status' => 'publish', 'posts_per_page' => self::FALLBACK_LIMIT, 'orderby' => 'modified', 'order' => 'DESC', 'ignore_sticky_posts' => true, 'no_found_rows' => true, 'update_post_meta_cache' => false, 'update_post_term_cache' => false, 'fields' => 'ids', ) ); foreach ( $query->posts as $post_id ) { $permalink = get_permalink( (int) $post_id ); if ( is_string( $permalink ) && '' !== $permalink ) { $urls[] = $permalink; } } return array_values( array_unique( $urls ) ); } /** * Recursive sitemap parser. Depth-limited to 3 so a maliciously * deep index can't stack-overflow. */ private static function fetch_sitemap_urls( string $sitemap_url, int $depth ): array { if ( $depth > 3 ) { return array(); } $res = wp_remote_get( $sitemap_url, array( 'timeout' => self::REQUEST_TIMEOUT, 'sslverify' => false, 'user-agent' => self::user_agent(), 'headers' => Self_Traffic::headers(), ) ); if ( is_wp_error( $res ) ) { // Record WHY, don't just vanish. "Unreachable" and "valid but // empty" both used to collapse into an empty array here, which is // what made a sitemap-less site look like a successful crawl of // zero URLs. Only the top-level fetch is recorded: a nested index // failing is a partial result, not a dead crawl. (#142) if ( 0 === $depth ) { self::$last_sitemap_error = sprintf( /* translators: 1: sitemap URL, 2: error detail. */ __( 'Could not fetch the sitemap at %1$s — %2$s', 'xspeed' ), $sitemap_url, $res->get_error_message() ); self::$last_sitemap_url = (string) $sitemap_url; } return array(); } $code = (int) wp_remote_retrieve_response_code( $res ); if ( $code >= 400 ) { if ( 0 === $depth ) { self::$last_sitemap_error = sprintf( /* translators: 1: sitemap URL, 2: HTTP status code. */ __( 'Could not fetch the sitemap at %1$s — the server returned HTTP %2$d.', 'xspeed' ), $sitemap_url, $code ); self::$last_sitemap_url = (string) $sitemap_url; } return array(); } $body = (string) wp_remote_retrieve_body( $res ); if ( '' === $body ) { return array(); } $urls = array(); // Sitemap index → recurse. if ( false !== strpos( $body, '([^<]+)#i', $body, $matches ) ) { foreach ( $matches[1] as $child ) { $urls = array_merge( $urls, self::fetch_sitemap_urls( trim( $child ), $depth + 1 ) ); } } return $urls; } // URL set → collect. if ( preg_match_all( '#([^<]+)#i', $body, $matches ) ) { foreach ( $matches[1] as $u ) { $u = trim( $u ); if ( '' !== $u && false !== filter_var( $u, FILTER_VALIDATE_URL ) ) { $urls[] = $u; } } } return $urls; } /** * Apply the user's schedule choice. Called on settings change. * Manual = no cron schedule (user must hit "Start now" to crawl). */ public static function apply_schedule( string $schedule ): void { wp_clear_scheduled_hook( 'xspeed_preloader_recurring' ); if ( in_array( $schedule, array( 'hourly', 'daily', 'weekly' ), true ) ) { if ( ! wp_next_scheduled( 'xspeed_preloader_recurring' ) ) { wp_schedule_event( time() + 60, $schedule, 'xspeed_preloader_recurring' ); } } } /** * Recurring schedule hook handler — fires per the user's chosen * cadence and kicks off a fresh crawl unless one is already running. */ public static function recurring_kickoff(): void { $state = self::status(); if ( $state['running'] ) { return; } self::start(); } }