PluginProbe
xSpeed Cache: AI-Powered Performance Hub with MCP, Caching & CDN / 1.4.1
xSpeed Cache: AI-Powered Performance Hub with MCP, Caching & CDN v1.4.1
1.4.1 1.4.0 1.3.7 1.3.6 1.3.5 1.3.4 1.3.3 1.3.2 1.3.1 1.3.0 1.2.4 trunk 1.0.0 1.0.1 1.0.2 1.0.3 1.0.4 1.0.5 1.0.6 1.0.7 1.0.8 1.0.9 1.1.0 1.1.1 1.1.2 All 35 releases
xspeed / includes / class-cache-inventory.php

class-cache-inventory.php in xSpeed Cache: AI-Powered Performance Hub with MCP, Caching & CDN 1.4.1, at includes/class-cache-inventory.php

567 lines 20.1 KB
No matching file
Up and down to move Enter to open Esc to close
Raw Download Zip
1 <?php
2 /**
3 * Cache_Inventory — what is actually in the cache, and what it costs.
4 *
5 * The four stat cards on the dashboard answer "how much"; this answers
6 * "which". A number like "412 cached pages" is only trustworthy if you
7 * can open it and see the 412, and a size of "8.4 MB" is only actionable
8 * if you can see what is taking the space.
9 *
10 * Two storage shapes have to be reconciled here:
11 *
12 * flat XSPEED_CACHE_DIR/<md5>.html written by the PHP path
13 * static XSPEED_CACHE_STATIC_DIR/<host><uri>/index.html
14 * served by the web server
15 *
16 * The static tree encodes the URL in its path, so those entries are exact.
17 * The flat tree is keyed by md5( host . uri . device . … ), which cannot be
18 * reversed — so for those we read the first HEAD_BYTES of the body and pull
19 * the canonical/og:url out of the markup. That resolves the overwhelming
20 * majority of real pages, and an entry we cannot name is reported with a
21 * null URL rather than a guess.
22 *
23 * Cost control: scanning is capped (SCAN_CAP) and the result memoized for
24 * CACHE_TTL, because this runs on an admin click, not on a page render.
25 *
26 * @package XSpeed
27 */
28
29 declare(strict_types=1);
30
31 namespace XSpeed;
32
33 defined( 'ABSPATH' ) || exit;
34
35 final class Cache_Inventory {
36
37 /**
38 * Hard ceiling on entries examined in one pass. A site with 50k cached
39 * pages must not turn one dashboard click into a 50k-file stat storm;
40 * the response says `capped: true` so the UI can say so out loud
41 * instead of quietly showing a partial list as if it were the whole.
42 */
43 public const SCAN_CAP = 750;
44
45 /** Bytes of each flat entry read to look for a canonical URL. */
46 public const HEAD_BYTES = 16384;
47
48 /** Memoization window for a full scan, in seconds. */
49 public const CACHE_TTL = 60;
50
51 public const TRANSIENT_KEY = 'xspeed_cache_inventory';
52
53 /** Activity-log event types that represent a purge. */
54 private const PURGE_TYPES = array( 'cache_purged', 'cache_purge_url', 'cache_purged_url' );
55
56 /**
57 * Resolve a static-tree file back to the URL that produced it.
58 *
59 * `<root>/example.com/blog/post/index.html` → `https://example.com/blog/post/`.
60 * Returns null for anything that isn't shaped like a static entry, so a
61 * stray file in the tree can't become a bogus row.
62 *
63 * Pure — no filesystem access, so the path logic is testable on its own.
64 */
65 public static function url_from_static_path( string $file, string $root, bool $https = true ): ?string {
66 $root = rtrim( str_replace( '\\', '/', $root ), '/' );
67 $file = str_replace( '\\', '/', $file );
68
69 if ( '' === $root || 0 !== strpos( $file, $root . '/' ) ) {
70 return null;
71 }
72 if ( substr( $file, -11 ) !== '/index.html' ) {
73 return null;
74 }
75
76 $rel = substr( $file, strlen( $root ) + 1, -11 );
77 if ( '' === $rel ) {
78 return null;
79 }
80
81 $segments = explode( '/', $rel );
82 $host = array_shift( $segments );
83 // Same allowlist store_static() writes with — anything else is not
84 // ours and must not be presented as a cached page.
85 if ( null === $host || '' === $host || preg_match( '/[^a-zA-Z0-9.\-]/', $host ) ) {
86 return null;
87 }
88
89 $path = '' === implode( '/', $segments ) ? '/' : '/' . implode( '/', $segments ) . '/';
90 // The tree stores decoded names (Cache::static_path()). Encode them
91 // back the way a permalink is spelled, so the row merges with the
92 // flat entry's canonical URL instead of listing the page twice.
93 $path = Cache::normalize_path( $path );
94
95 return ( $https ? 'https://' : 'http://' ) . $host . $path;
96 }
97
98 /**
99 * Pull a page URL out of the top of a cached document.
100 *
101 * Prefers `<link rel="canonical">` — WordPress emits it for singular
102 * views and it is what the site itself considers the page's address.
103 * Falls back to `og:url`. Returns null rather than guessing from, say,
104 * the first anchor in the body.
105 *
106 * Pure.
107 */
108 public static function extract_url_from_html( string $head ): ?string {
109 if ( preg_match( '#<link[^>]+rel=["\']canonical["\'][^>]*>#i', $head, $tag ) ) {
110 if ( preg_match( '#href=["\']([^"\']+)["\']#i', $tag[0], $href ) ) {
111 $url = html_entity_decode( trim( $href[1] ), ENT_QUOTES );
112 if ( 0 === stripos( $url, 'http' ) ) {
113 return $url;
114 }
115 }
116 }
117 if ( preg_match( '#<meta[^>]+property=["\']og:url["\'][^>]*>#i', $head, $tag ) ) {
118 if ( preg_match( '#content=["\']([^"\']+)["\']#i', $tag[0], $content ) ) {
119 $url = html_entity_decode( trim( $content[1] ), ENT_QUOTES );
120 if ( 0 === stripos( $url, 'http' ) ) {
121 return $url;
122 }
123 }
124 }
125 return null;
126 }
127
128 /**
129 * Merge the two storage shapes into one row per page.
130 *
131 * A page served by the web-server rewrite usually exists in BOTH trees.
132 * Listing it twice would make the drill-down disagree with the stat card
133 * it was opened from, so rows are keyed by URL when we know it: the
134 * newest mtime wins for "age", disk bytes add up, and `stored_in` records
135 * which copies exist.
136 *
137 * Entries whose URL is unknown can't be merged with anything — they stay
138 * distinct, keyed by their own path.
139 *
140 * Pure: takes already-collected rows, returns merged rows.
141 *
142 * @param array<int,array<string,mixed>> $rows Raw rows from either tree.
143 * @return array<int,array<string,mixed>> Merged, newest-first.
144 */
145 public static function merge_rows( array $rows ): array {
146 $merged = array();
147
148 foreach ( $rows as $row ) {
149 $url = isset( $row['url'] ) && is_string( $row['url'] ) ? $row['url'] : null;
150 // A flat entry's URL comes from the page's canonical link, which
151 // may carry either escape case; compare in the cache key's one
152 // spelling so a page's flat and static copies stay one row.
153 $key = null !== $url ? 'u:' . Cache::normalize_path( $url ) : 'p:' . (string) ( $row['path'] ?? '' );
154
155 if ( ! isset( $merged[ $key ] ) ) {
156 $merged[ $key ] = $row;
157 continue;
158 }
159
160 $existing = $merged[ $key ];
161 $existing['bytes'] = (int) $existing['bytes'] + (int) $row['bytes'];
162 $existing['mtime'] = max( (int) $existing['mtime'], (int) $row['mtime'] );
163 $existing['stored_in'] = array_values( array_unique( array_merge( (array) $existing['stored_in'], (array) $row['stored_in'] ) ) );
164 $existing['compressed'] += (int) $row['compressed'];
165 sort( $existing['stored_in'] );
166 $merged[ $key ] = $existing;
167 }
168
169 $out = array_values( $merged );
170 usort(
171 $out,
172 static function ( array $a, array $b ): int {
173 return (int) $b['mtime'] <=> (int) $a['mtime'];
174 }
175 );
176
177 return $out;
178 }
179
180 /**
181 * The cached-pages drill-down.
182 *
183 * @param int $limit Rows returned.
184 * @param int $offset Rows skipped.
185 * @param bool $fresh Bypass the memoized scan.
186 * @return array{entries:array<int,array<string,mixed>>,total:int,capped:bool,generated:int}
187 */
188 public static function entries( int $limit = 50, int $offset = 0, bool $fresh = false ): array {
189 $rows = $fresh ? null : get_transient( self::TRANSIENT_KEY );
190 if ( ! is_array( $rows ) || ! isset( $rows['entries'] ) ) {
191 $rows = self::scan();
192 set_transient( self::TRANSIENT_KEY, $rows, self::CACHE_TTL );
193 }
194
195 $all = is_array( $rows['entries'] ) ? $rows['entries'] : array();
196 $limit = max( 1, min( 200, $limit ) );
197 $offset = max( 0, $offset );
198 $now = time();
199
200 $page = array_slice( $all, $offset, $limit );
201 foreach ( $page as $i => $entry ) {
202 $page[ $i ]['age'] = max( 0, $now - (int) $entry['mtime'] );
203 }
204
205 return array(
206 'entries' => array_values( $page ),
207 'total' => count( $all ),
208 'capped' => (bool) ( $rows['capped'] ?? false ),
209 'generated' => (int) ( $rows['generated'] ?? $now ),
210 );
211 }
212
213 /**
214 * Walk both trees. Separated from entries() so the memoization and the
215 * pagination stay readable, and so a test can drive the walk directly.
216 *
217 * @return array{entries:array<int,array<string,mixed>>,capped:bool,generated:int}
218 */
219 public static function scan(): array {
220 $rows = array();
221 $scanned = 0;
222 $capped = false;
223
224 // Flat tree — md5-keyed, URL recovered from the markup.
225 // Flat entries live in per-site buckets since #6
226 // (XSPEED_CACHE_DIR/<host>/<md5>.html); the legacy top-level layout
227 // is still globbed so pre-#6 entries remain visible until they age out.
228 $flat = defined( 'XSPEED_CACHE_DIR' )
229 ? array_merge(
230 (array) glob( XSPEED_CACHE_DIR . '/*.html' ),
231 (array) glob( XSPEED_CACHE_DIR . '/*/*.html' )
232 )
233 : array();
234 $flat = array_values(
235 array_filter(
236 $flat,
237 static function ( $f ) {
238 // min/ and rest/ are separate buckets, not page entries.
239 return ! in_array( basename( dirname( (string) $f ) ), array( 'min', 'rest', 'combined' ), true );
240 }
241 )
242 );
243 foreach ( $flat as $file ) {
244 if ( $scanned >= self::SCAN_CAP ) {
245 $capped = true;
246 break;
247 }
248 ++$scanned;
249
250 $size = (int) @filesize( $file ); // phpcs:ignore WordPress.PHP.NoSilencedErrors.Discouraged -- A file purged between glob() and stat() is expected, not exceptional.
251 $rows[] = array(
252 'url' => self::read_url( $file ),
253 'path' => $file,
254 'key' => basename( $file, '.html' ),
255 'bytes' => $size,
256 'compressed' => self::sibling_size( $file . '.br' ),
257 'mtime' => (int) @filemtime( $file ), // phpcs:ignore WordPress.PHP.NoSilencedErrors.Discouraged -- Same race as filesize() above.
258 'stored_in' => array( 'flat' ),
259 );
260 }
261
262 // Static tree — URL is the path, so no read is needed at all.
263 if ( defined( 'XSPEED_CACHE_STATIC_DIR' ) && is_dir( XSPEED_CACHE_STATIC_DIR ) ) {
264 $https = ! function_exists( 'home_url' ) || 0 === stripos( (string) home_url(), 'https://' );
265 self::walk_static( XSPEED_CACHE_STATIC_DIR, XSPEED_CACHE_STATIC_DIR, $https, $rows, $scanned, $capped );
266 }
267
268 return array(
269 'entries' => self::merge_rows( $rows ),
270 'capped' => $capped,
271 'generated' => time(),
272 );
273 }
274
275 /**
276 * Recursive walk of the static tree, mirroring Cache::rmtree_html()'s
277 * traversal so the two can't disagree about what counts as an entry.
278 *
279 * @param array<int,array<string,mixed>> $rows Collected rows, by reference.
280 * @param int $scanned Files examined so far, by reference.
281 * @param bool $capped Set when SCAN_CAP is hit, by reference.
282 */
283 private static function walk_static( string $dir, string $root, bool $https, array &$rows, int &$scanned, bool &$capped ): void {
284 if ( $capped ) {
285 return;
286 }
287 $entries = @scandir( $dir, SCANDIR_SORT_NONE ); // phpcs:ignore WordPress.PHP.NoSilencedErrors.Discouraged -- An unreadable subdir yields no rows; it must not fatal the drill-down.
288 if ( false === $entries ) {
289 return;
290 }
291
292 foreach ( $entries as $entry ) {
293 if ( '.' === $entry || '..' === $entry ) {
294 continue;
295 }
296 $path = $dir . '/' . $entry;
297
298 if ( is_dir( $path ) ) {
299 self::walk_static( $path, $root, $https, $rows, $scanned, $capped );
300 if ( $capped ) {
301 return;
302 }
303 continue;
304 }
305 if ( 'index.html' !== $entry ) {
306 continue;
307 }
308 if ( $scanned >= self::SCAN_CAP ) {
309 $capped = true;
310 return;
311 }
312 ++$scanned;
313
314 $rows[] = array(
315 'url' => self::url_from_static_path( $path, $root, $https ),
316 'path' => $path,
317 'key' => '',
318 'bytes' => (int) @filesize( $path ), // phpcs:ignore WordPress.PHP.NoSilencedErrors.Discouraged -- Purge race, see scan().
319 'compressed' => self::sibling_size( $path . '.br' ),
320 'mtime' => (int) @filemtime( $path ), // phpcs:ignore WordPress.PHP.NoSilencedErrors.Discouraged -- Purge race, see scan().
321 'stored_in' => array( 'static' ),
322 );
323 }
324 }
325
326 /** Size of a precompressed sibling, or 0 when there isn't one. */
327 private static function sibling_size( string $path ): int {
328 if ( ! is_readable( $path ) ) {
329 return 0;
330 }
331 return (int) @filesize( $path ); // phpcs:ignore WordPress.PHP.NoSilencedErrors.Discouraged -- Purge race, see scan().
332 }
333
334 /**
335 * Read just enough of a cached document to find its canonical URL.
336 *
337 * Deliberately a partial read: WP_Filesystem::get_contents() would pull
338 * whole pages into memory, and at SCAN_CAP entries the difference is
339 * megabytes of pointless I/O for data we discard immediately.
340 */
341 private static function read_url( string $file ): ?string {
342 // phpcs:ignore WordPress.WP.AlternativeFunctions.file_get_contents_file_get_contents, WordPress.PHP.NoSilencedErrors.Discouraged -- Partial read (16KB of head markup); WP_Filesystem has no offset/length API and would read entire pages.
343 $head = @file_get_contents( $file, false, null, 0, self::HEAD_BYTES );
344 if ( ! is_string( $head ) || '' === $head ) {
345 return null;
346 }
347 return self::extract_url_from_html( $head );
348 }
349
350 /**
351 * Where the cache's disk usage actually goes.
352 *
353 * Buckets are what a user can act on — pages, their precompressed
354 * siblings, the REST response cache, minified assets — not what the
355 * code happens to write. `compressed_bytes` is real measured bytes from
356 * `.br` siblings, never an estimate: a made-up compression ratio on a
357 * dashboard is worse than no number.
358 *
359 * @return array<string,mixed>
360 */
361 public static function size_breakdown(): array {
362 $buckets = array(
363 'pages_flat' => array(
364 'label' => __( 'Cached pages (PHP)', 'xspeed' ),
365 'bytes' => 0,
366 'files' => 0,
367 ),
368 'pages_static' => array(
369 'label' => __( 'Cached pages (server-served)', 'xspeed' ),
370 'bytes' => 0,
371 'files' => 0,
372 ),
373 'precompressed' => array(
374 'label' => __( 'Precompressed copies', 'xspeed' ),
375 'bytes' => 0,
376 'files' => 0,
377 ),
378 'metadata' => array(
379 'label' => __( 'Per-entry metadata', 'xspeed' ),
380 'bytes' => 0,
381 'files' => 0,
382 ),
383 'rest' => array(
384 'label' => __( 'REST responses', 'xspeed' ),
385 'bytes' => 0,
386 'files' => 0,
387 ),
388 'assets' => array(
389 'label' => __( 'Minified CSS/JS', 'xspeed' ),
390 'bytes' => 0,
391 'files' => 0,
392 ),
393 );
394
395 if ( defined( 'XSPEED_CACHE_DIR' ) ) {
396 // Both layouts: per-site buckets (#6) and pre-#6 top level.
397 self::add_glob( $buckets['pages_flat'], XSPEED_CACHE_DIR . '/*.html' );
398 self::add_glob( $buckets['metadata'], XSPEED_CACHE_DIR . '/*.meta' );
399 self::add_glob( $buckets['precompressed'], XSPEED_CACHE_DIR . '/*.br' );
400 self::add_glob( $buckets['pages_flat'], XSPEED_CACHE_DIR . '/*/*.html' );
401 self::add_glob( $buckets['metadata'], XSPEED_CACHE_DIR . '/*/*.meta' );
402 self::add_glob( $buckets['precompressed'], XSPEED_CACHE_DIR . '/*/*.br' );
403 self::add_glob( $buckets['rest'], XSPEED_CACHE_DIR . '/rest/*.json' );
404 self::add_glob( $buckets['assets'], XSPEED_CACHE_DIR . '/min/*.css' );
405 self::add_glob( $buckets['assets'], XSPEED_CACHE_DIR . '/min/*.js' );
406 self::add_glob( $buckets['assets'], XSPEED_CACHE_DIR . '/min/combined/*.css' );
407 self::add_glob( $buckets['assets'], XSPEED_CACHE_DIR . '/min/combined/*.js' );
408 // One manifest per minified source, per blog (Asset_Manifest).
409 self::add_glob( $buckets['metadata'], XSPEED_CACHE_DIR . '/min/manifests/*/*.json' );
410 }
411
412 if ( defined( 'XSPEED_CACHE_STATIC_DIR' ) && is_dir( XSPEED_CACHE_STATIC_DIR ) ) {
413 $static = self::measure_static( XSPEED_CACHE_STATIC_DIR );
414 $buckets['pages_static']['bytes'] = $static['html_bytes'];
415 $buckets['pages_static']['files'] = $static['html_files'];
416 $buckets['precompressed']['bytes'] += $static['br_bytes'];
417 $buckets['precompressed']['files'] += $static['br_files'];
418 }
419
420 $total_bytes = 0;
421 $total_files = 0;
422 $out = array();
423 foreach ( $buckets as $key => $bucket ) {
424 $total_bytes += $bucket['bytes'];
425 $total_files += $bucket['files'];
426 $out[] = array(
427 'key' => $key,
428 'label' => $bucket['label'],
429 'bytes' => $bucket['bytes'],
430 'files' => $bucket['files'],
431 );
432 }
433
434 return array(
435 'buckets' => $out,
436 'total_bytes' => $total_bytes,
437 'total_files' => $total_files,
438 // What a visitor actually downloads for the pages that have a
439 // precompressed copy. Pages without one are served compressed by
440 // the web server at request time, which we cannot measure from
441 // here — so this is a floor, and the UI labels it as one.
442 'compressed_bytes' => $buckets['precompressed']['bytes'],
443 'pages' => $buckets['pages_flat']['files'] + $buckets['pages_static']['files'],
444 );
445 }
446
447 /**
448 * Accumulate a glob into a bucket.
449 *
450 * @param array{label:string,bytes:int,files:int} $bucket By reference.
451 */
452 private static function add_glob( array &$bucket, string $pattern ): void {
453 foreach ( (array) glob( $pattern ) as $file ) {
454 if ( ! is_string( $file ) ) {
455 continue;
456 }
457 $bucket['bytes'] += (int) @filesize( $file ); // phpcs:ignore WordPress.PHP.NoSilencedErrors.Discouraged -- Purge race, see scan().
458 ++$bucket['files'];
459 }
460 }
461
462 /**
463 * Total the static tree without collecting per-file rows.
464 *
465 * @return array{html_bytes:int,html_files:int,br_bytes:int,br_files:int}
466 */
467 private static function measure_static( string $dir ): array {
468 $totals = array(
469 'html_bytes' => 0,
470 'html_files' => 0,
471 'br_bytes' => 0,
472 'br_files' => 0,
473 );
474 $entries = @scandir( $dir, SCANDIR_SORT_NONE ); // phpcs:ignore WordPress.PHP.NoSilencedErrors.Discouraged -- Unreadable subdir contributes nothing rather than fataling.
475 if ( false === $entries ) {
476 return $totals;
477 }
478
479 foreach ( $entries as $entry ) {
480 if ( '.' === $entry || '..' === $entry ) {
481 continue;
482 }
483 $path = $dir . '/' . $entry;
484 if ( is_dir( $path ) ) {
485 $sub = self::measure_static( $path );
486 foreach ( $totals as $k => $v ) {
487 $totals[ $k ] = $v + $sub[ $k ];
488 }
489 continue;
490 }
491 if ( 'index.html' === $entry ) {
492 $totals['html_bytes'] += (int) @filesize( $path ); // phpcs:ignore WordPress.PHP.NoSilencedErrors.Discouraged -- Purge race, see scan().
493 ++$totals['html_files'];
494 } elseif ( substr( $entry, -3 ) === '.br' ) {
495 $totals['br_bytes'] += (int) @filesize( $path ); // phpcs:ignore WordPress.PHP.NoSilencedErrors.Discouraged -- Purge race, see scan().
496 ++$totals['br_files'];
497 }
498 }
499
500 return $totals;
501 }
502
503 /**
504 * Filter the activity log down to purge events.
505 *
506 * The log already carries the cause in its message ("post saved",
507 * "settings change", "manual"), which is the whole point of the
508 * drill-down: "Last purge — 4h ago" is trivia until you can see that it
509 * was a post save rather than something clearing the cache every hour.
510 *
511 * Pure with respect to the entries passed in, so the type filter is
512 * testable without a WordPress transient.
513 *
514 * @param array<int,array<string,mixed>> $entries Activity_Log::entries() output.
515 * @return array<int,array<string,mixed>>
516 */
517 public static function filter_purges( array $entries, int $limit = 25 ): array {
518 $out = array();
519 foreach ( $entries as $entry ) {
520 $type = isset( $entry['type'] ) ? (string) $entry['type'] : '';
521 if ( ! in_array( $type, self::PURGE_TYPES, true ) ) {
522 continue;
523 }
524 $out[] = array(
525 'ts' => (int) ( $entry['ts'] ?? 0 ),
526 'type' => $type,
527 'message' => (string) ( $entry['message'] ?? '' ),
528 'severity' => (string) ( $entry['severity'] ?? 'info' ),
529 );
530 if ( count( $out ) >= $limit ) {
531 break;
532 }
533 }
534 return $out;
535 }
536
537 /**
538 * The last-purge drill-down.
539 *
540 * `last_gc` / `gc_removed` / `gc_removed_total` come from the daily
541 * `xspeed_gc` sweep (Cache_GC) so the collector is verifiable from the
542 * dashboard instead of over SSH.
543 *
544 * @return array{events:array<int,array<string,mixed>>,last_purge:int,last_gc:int,gc_removed:int,gc_removed_total:int,gc_next_run:int}
545 */
546 public static function purge_log( int $limit = 25 ): array {
547 $limit = max( 1, min( 50, $limit ) );
548 $events = self::filter_purges( Activity_Log::entries(), $limit );
549 $stats = get_option( 'xspeed_stats', array() );
550 $stats = is_array( $stats ) ? $stats : array();
551
552 return array(
553 'events' => $events,
554 'last_purge' => (int) ( $stats['last_purge'] ?? 0 ),
555 'last_gc' => (int) ( $stats['last_gc'] ?? 0 ),
556 'gc_removed' => (int) ( $stats['gc_removed'] ?? 0 ),
557 'gc_removed_total' => (int) ( $stats['gc_removed_total'] ?? 0 ),
558 'gc_next_run' => (int) wp_next_scheduled( Cache_GC::CRON_HOOK ),
559 );
560 }
561
562 /** Drop the memoized scan — called after a purge so the list can't lie. */
563 public static function invalidate(): void {
564 delete_transient( self::TRANSIENT_KEY );
565 }
566 }
567