PluginProbe
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO / 2.12.0
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO v2.12.0
2.12.0 2.11.0 2.10.0 2.9.0 2.8.0 2.7.0 2.6.0 2.5.0 2.4.0 2.3.0 2.2.0 2.1.1 2.1.0 2.0.2 2.0.1 2.0.0 1.32.0 1.31.0 1.30.0 1.29.0 1.28.0 1.27.0 1.26.0 1.25.0 trunk All 53 releases
thinkrank / includes / seo / class-builder-content.php

class-builder-content.php in ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO 2.12.0, at includes/seo/class-builder-content.php

1,849 lines 72.1 KB
No matching file
Up and down to move Enter to open Esc to close
Raw Download Zip
1 <?php
2 /**
3 * Page-builder content extraction.
4 *
5 * SEO analysis reads `post_content`, which is only the real text on a classic
6 * post. Page builders keep the words somewhere else, and every server-side
7 * scoring path — bulk analysis, the post-list SEO Overview column, the MCP
8 * abilities, cron reports — saw an empty page as a result:
9 *
10 * - Oxygen / Breakdance leave `post_content` completely EMPTY and store the
11 * node tree in postmeta. Nothing to render, nothing to strip: the analyzer
12 * reported "No content" on pages with well over a thousand visible words.
13 * - Elementor does the same via `_elementor_data`.
14 * - Divi 5 and Gutenberg do store block markup in `post_content`, but Divi
15 * keeps module text inside the block's JSON attributes — inside an HTML
16 * comment, which tag stripping removes wholesale.
17 * - Divi 4 and other shortcode builders keep text in shortcode attributes.
18 *
19 * Extraction reads the builder's own stored data rather than invoking its
20 * render engine. Rendering an Oxygen page outside a front-end request is slow,
21 * stateful and can fatal in an admin context, whereas the stored tree is just
22 * JSON — cheap, side-effect free and safe to touch during a bulk run.
23 *
24 * @package ThinkRank\SEO
25 * @since 1.23.0
26 */
27
28 declare(strict_types=1);
29
30 namespace ThinkRank\SEO;
31
32 if (!defined('ABSPATH')) {
33 exit;
34 }
35
36 /**
37 * Resolves the analyzable content of a post, whatever built it.
38 */
39 class Builder_Content {
40
41 /**
42 * Post meta keys that hold builder data, in priority order.
43 *
44 * Several generations of the same builder are listed on purpose: Oxygen 6
45 * is Breakdance under the hood (`_breakdance_data`), while earlier Oxygen
46 * releases used `_oxygen_data` or the shortcode-based
47 * `ct_builder_shortcodes`. A site can only have one of them.
48 *
49 * @var string[]
50 */
51 private const BUILDER_META_KEYS = [
52 '_breakdance_data', // Oxygen 6+ / Breakdance
53 '_oxygen_data', // Oxygen (earlier releases)
54 // Oxygen classic. 4.x writes the tree as JSON to `ct_builder_json`
55 // while still keeping `ct_builder_shortcodes`. A post carrying only
56 // the JSON key used to match no key at all and fall through to an
57 // empty `post_content`, which reads as a one-word page (#776).
58 //
59 // Oxygen 4.8.3 then renamed every `ct_*` post meta key to `_ct_*`
60 // (`oxygen_vsb_update_4_8_3()` runs `oxy_prefix_meta_keys()` on
61 // upgrade, and `oxy_get_post_meta()` only reads the prefixed name
62 // from then on). A current Oxygen classic site therefore has only the
63 // underscored keys, which nothing here listed, so every one of its
64 // pages resolved as empty. The prefixed keys come first because they
65 // are what Oxygen itself reads; the bare ones cover a site that has
66 // not run the migration (or reverted it with `?unprefix_meta`).
67 //
68 // These four are not read in this loop: from_oxygen_classic() pairs
69 // each JSON key with its shortcode sibling so the two forms can be
70 // compared. They are listed here because this list is also what the
71 // word-count index watches and what FAQ detection scans.
72 '_ct_builder_json', // Oxygen classic 4.8.3+ (JSON tree)
73 'ct_builder_json', // Oxygen classic 4.0-4.8.2 (JSON tree)
74 '_ct_builder_shortcodes', // Oxygen classic 4.8.3+ (shortcode tree)
75 'ct_builder_shortcodes', // Oxygen classic < 4.8.3 (shortcode tree)
76 '_elementor_data', // Elementor
77 // Beaver Builder. Published layout first: `_fl_builder_draft` holds
78 // unsaved changes and would score content the visitor cannot see.
79 // Both are arrays of stdClass nodes, which is why the walker below
80 // has to treat objects like arrays (#449).
81 '_fl_builder_data', // Beaver Builder (published)
82 '_fl_builder_draft', // Beaver Builder (unsaved changes)
83 ];
84
85 /**
86 * Bricks' content-area meta key, used when Bricks itself isn't loaded.
87 *
88 * Bricks exposes `BRICKS_DB_PAGE_CONTENT` and renames the underlying key
89 * between generations (it gained the `_2` suffix in 1.7.3), so the
90 * constant is authoritative and this literal is only the fallback for the
91 * contexts where it is undefined — Bricks is a theme, so on an admin or
92 * CLI request against a site that has since switched themes the constant
93 * is simply not there while the post meta still is.
94 *
95 * Bricks stores three areas — header, content and footer. Only the content
96 * area belongs to the post being scored; the header and footer areas live
97 * on Bricks' own template posts and would double-count site chrome into
98 * every page's word count, so they are deliberately not read here.
99 *
100 * @since 2.2.1
101 * @var string
102 */
103 private const BRICKS_CONTENT_META_KEY = '_bricks_page_content_2';
104
105 /**
106 * Bricks' per-post editor-mode meta key, used when Bricks isn't loaded.
107 *
108 * @since 2.2.1
109 * @var string
110 */
111 private const BRICKS_EDITOR_MODE_META_KEY = '_bricks_editor_mode';
112
113 /**
114 * Bricks' components option, used when Bricks itself isn't loaded.
115 *
116 * @since 2.2.1
117 * @var string
118 */
119 private const BRICKS_COMPONENTS_OPTION = 'bricks_components';
120
121 /**
122 * Bricks' element that renders the post's own `post_content`.
123 *
124 * A Bricks page normally discards `post_content` entirely, which is why
125 * anything left there is invisible. Dropping this element onto the canvas
126 * is the one way an author puts it back on the page, so its presence flips
127 * `post_content` from stale leftovers to content the visitor reads.
128 *
129 * @since 2.3.1
130 * @var string
131 */
132 private const BRICKS_POST_CONTENT_ELEMENT = 'post-content';
133
134 /**
135 * Resolved Bricks trees for this request, keyed by post ID.
136 *
137 * Rendering one page asks for the tree about twenty times — every
138 * description, every schema node, the FAQ guard — and resolving it is not
139 * free. `bricks_content_source()` clears `Bricks\Database::$active_templates`
140 * before asking Bricks which content template applies, which defeats
141 * Bricks' own early-return and re-runs its whole template-condition engine;
142 * `expand_bricks_components()` then walks the tree again. Measured on a
143 * Bricks page with no content of its own, that was ten full runs of the
144 * rules engine per request.
145 *
146 * Per-request only, and only ever read back within one page render — a
147 * request that writes Bricks content does not also render it.
148 *
149 * @since 2.3.1
150 * @var array<int,array<int,mixed>>
151 */
152 private static array $bricks_trees = [];
153
154 /**
155 * JSON keys whose values are user-visible text.
156 *
157 * Builder trees mix content with configuration, so a blind string sweep
158 * would count CSS classes and option slugs as words. Matching on the key
159 * keeps the word count honest.
160 *
161 * @var string[]
162 */
163 private const CONTENT_KEYS = [
164 'text', 'title', 'subtitle', 'heading', 'subheading', 'content',
165 'description', 'caption', 'excerpt', 'label', 'value', 'html',
166 'editor', 'quote', 'answer', 'question', 'body', 'button_text',
167 // Oxygen classic keeps an element's copy in `options.ct_content`
168 // (headline, text block, rich text, link and button labels). It is the
169 // field Oxygen's own serializer moves between the tags when it writes
170 // shortcodes (`parse_components_tree()`), and the one Relevanssi and
171 // Oxygen's WPML integration read. Missing from this list, the walker
172 // kept only copy that happened to contain markup: a page of plain
173 // headings and paragraphs lost almost all of its words.
174 'ct_content',
175 // Oxygen's composite elements keep their copy under `options.original`
176 // instead, one key per field. Taken from the list Oxygen itself treats
177 // as text when it serializes (`$options_to_encode`); the numeric price
178 // fields and the progress bar's right-hand percentage are left out.
179 'testimonial_text', 'testimonial_author', 'testimonial_author_info',
180 'icon_box_heading', 'icon_box_text',
181 'pricing_box_package_title', 'pricing_box_package_subtitle', 'pricing_box_content',
182 'progress_bar_left_text',
183 ];
184
185 /**
186 * Oxygen classic's storage generations, as JSON key => shortcode key.
187 *
188 * @since 2.10.0
189 * @var array<string,string>
190 */
191 private const OXYGEN_CLASSIC_KEYS = [
192 '_ct_builder_json' => '_ct_builder_shortcodes',
193 'ct_builder_json' => 'ct_builder_shortcodes',
194 ];
195
196 /**
197 * JSON keys whose values hold a link destination.
198 *
199 * Builders store a link's destination in a structured field separate from
200 * its label, either as a bare URL string or as a `{ url: … }` object.
201 * Neither shape survives a text sweep — the key is not content and a bare
202 * URL contains no `<` — so no `<a>` tag reached the link counters.
203 *
204 * @var string[]
205 */
206 private const URL_KEYS = [
207 'link', 'url', 'href', 'link_url', 'button_link', 'permalink', 'link_to',
208 ];
209
210 /**
211 * JSON keys whose values hold an embedded video's source.
212 *
213 * A builder's video widget keeps its destination in a provider-specific
214 * field — Elementor picks `youtube_url`, `vimeo_url`, `dailymotion_url` or
215 * `hosted_url` according to the chosen source type — none of which is a
216 * link field or a content field, so a video on a builder page reached the
217 * analyzers as nothing at all.
218 *
219 * These are deliberately kept out of URL_KEYS. A video is an embed, not an
220 * outbound link: rendering one as `<a href>` would add a spurious external
221 * link to every page carrying a video and skew the link counts. They are
222 * reconstructed as `<iframe>`/`<video>` instead, which the video detector
223 * recognises and the link and image counters ignore.
224 *
225 * @since 2.3.1
226 * @var string[]
227 */
228 private const VIDEO_KEYS = [
229 'youtube_url', 'vimeo_url', 'dailymotion_url', 'videopress_url',
230 'hosted_url', 'video_url', 'video_src', 'video_link',
231 ];
232
233 /**
234 * File extensions that mean a video source is a file, not a provider page.
235 *
236 * @since 2.3.1
237 * @var string[]
238 */
239 private const VIDEO_FILE_EXTENSIONS = ['mp4', 'webm', 'ogv', 'mov', 'm4v'];
240
241 /**
242 * Source keys to trust for a declared video source type.
243 *
244 * A widget keeps one field per provider and does not clear the others when
245 * the author switches source: an Elementor video moved from YouTube to Self
246 * Hosted still carries the earlier `youtube_url`. Reading whichever key
247 * turns up first then emits the video the author replaced. The widget says
248 * which one it is actually playing, so that is read first and the flat key
249 * sweep is only the fallback for a builder that declares nothing.
250 *
251 * @since 2.3.1
252 * @var array<string,string[]>
253 */
254 private const VIDEO_KEYS_BY_TYPE = [
255 'youtube' => ['youtube_url'],
256 'vimeo' => ['vimeo_url'],
257 'dailymotion' => ['dailymotion_url'],
258 'videopress' => ['videopress_url'],
259 'hosted' => ['hosted_url', 'video_url', 'video_src', 'video_link'],
260 'media' => ['hosted_url', 'video_url', 'video_src', 'video_link'],
261 'file' => ['hosted_url', 'video_url', 'video_src', 'video_link'],
262 'self_hosted' => ['hosted_url', 'video_url', 'video_src', 'video_link'],
263 ];
264
265 /**
266 * Keys a builder uses to name which video source a widget is playing.
267 *
268 * @since 2.3.1
269 * @var string[]
270 */
271 private const VIDEO_TYPE_KEYS = ['video_type', 'videotype', 'video_source', 'source_type'];
272
273 /**
274 * JSON keys whose values hold an image, as a URL string or `{ url, alt }`.
275 *
276 * @var string[]
277 */
278 private const IMAGE_KEYS = [
279 'image', 'src', 'image_url', 'background_image', 'bg_image', 'photo',
280 ];
281
282 /**
283 * JSON keys that carry a heading level for the node's text.
284 *
285 * A builder heading's text is collected (its key is in CONTENT_KEYS) and so
286 * counts toward the word count, but it arrives as bare text with no `<h2>`
287 * wrapper — which is why heading-structure checks saw none.
288 *
289 * @var string[]
290 */
291 private const HEADING_TAG_KEYS = [
292 // Lower-cased on both sides of the comparison, so `headingtag` is the
293 // camelCase `headingTag` Bricks uses throughout its own controls and
294 // which ThinkRank's Bricks elements declare. Without it their section
295 // headings counted as body copy and never reached heading structure.
296 'header_size', 'heading_tag', 'headingtag', 'html_tag', 'title_tag', 'tag', 'level', 'size',
297 ];
298
299 /**
300 * Keys whose value is alternative text for a sibling image.
301 *
302 * @var string[]
303 */
304 private const ALT_KEYS = ['alt', 'alt_text', 'image_alt', 'title'];
305
306 /**
307 * The global post and every global `setup_postdata()` writes.
308 *
309 * Rendering points them at the post being analyzed, then puts each one
310 * back exactly as it was, unset included (#860).
311 *
312 * @since 2.12.0
313 * @var string[]
314 */
315 private const POSTDATA_GLOBALS = [
316 'post', 'id', 'authordata', 'currentday', 'currentmonth',
317 'page', 'pages', 'multipage', 'more', 'numpages',
318 ];
319
320 /**
321 * Resolve the content worth analyzing for a post.
322 *
323 * @param \WP_Post $post Post being analyzed.
324 * @return string HTML/text to analyze.
325 */
326 public static function resolve(\WP_Post $post): string {
327 $raw = (string) $post->post_content;
328
329 // A page built in Gutenberg and then switched to Bricks keeps its old
330 // blocks in `post_content` forever — Bricks never clears them, and
331 // never renders them either. Resolving that first meant the stale draft
332 // beat the tree the visitor actually reads, and it did not stop at the
333 // score: the same string becomes the meta description, og:description,
334 // twitter:description and the schema description. Starting from nothing
335 // sends the resolution straight to Bricks' storage, which is where this
336 // page's words are (#651).
337 //
338 // Only for the stored path. `resolve_markup()` is also called with live
339 // editor content, and the Bricks panel's resolver reads the canvas —
340 // discarding that would replace what the author is typing with the last
341 // save.
342 if (self::bricks_supersedes_post_content((int) $post->ID)) {
343 $raw = '';
344 }
345
346 return self::resolve_markup($raw, $post);
347 }
348
349 /**
350 * Whether Bricks renders this post and throws its `post_content` away.
351 *
352 * True means anything still stored in `post_content` is invisible: it is
353 * not on the page, so it must not be scored, described or published as
354 * structured data. False covers both a post Bricks does not own and a
355 * Bricks page that puts `post_content` back with a Post Content element.
356 *
357 * @since 2.3.1
358 *
359 * @param int $post_id Post being resolved.
360 * @return bool
361 */
362 public static function bricks_supersedes_post_content(int $post_id): bool {
363 $tree = self::bricks_tree($post_id);
364
365 if (empty($tree)) {
366 return false;
367 }
368
369 foreach ($tree as $element) {
370 if (is_array($element)
371 && self::BRICKS_POST_CONTENT_ELEMENT === ($element['name'] ?? null)
372 ) {
373 return false;
374 }
375 }
376
377 return !self::bricks_tree_prints_post_content($tree);
378 }
379
380 /**
381 * Whether a Bricks tree prints the body through a dynamic-data tag.
382 *
383 * The Post Content element is not the only way back onto the page: Bricks'
384 * `{post_content}` tag renders the same thing from inside an ordinary text
385 * element, and a single-post template written that way is a common shape.
386 * Missing it would mean the post's real body is discarded everywhere —
387 * scoring, the meta/og/twitter descriptions, the schema description — for a
388 * page that is displaying it.
389 *
390 * Matched over the encoded tree rather than per setting, because the tag can
391 * sit in any string field of any element and Bricks allows modifiers after
392 * the name (`{post_content:...}`).
393 *
394 * @since 2.3.1
395 *
396 * @param array $tree Bricks element tree.
397 * @return bool
398 */
399 private static function bricks_tree_prints_post_content(array $tree): bool {
400 $encoded = wp_json_encode($tree);
401
402 return is_string($encoded) && false !== stripos($encoded, '{post_content');
403 }
404
405 /**
406 * The post's content as the visitor actually receives it.
407 *
408 * `post_content` for everything except a Bricks page that discards it, and
409 * there the Bricks tree's text. Descriptions are derived from a post's body
410 * in half a dozen places; every one of them wants this rather than the raw
411 * column (#651).
412 *
413 * @since 2.3.1
414 *
415 * @param \WP_Post $post Post being described.
416 * @return string
417 */
418 public static function visible_content(\WP_Post $post): string {
419 $superseding = self::superseding_content($post);
420
421 return '' !== $superseding ? $superseding : (string) $post->post_content;
422 }
423
424 /**
425 * Replacement body text for a post whose `post_content` does not render.
426 *
427 * Empty for every ordinary post, which is what makes this safe to call from
428 * paths that already handle excerpts their own way: they keep that handling
429 * and only a Bricks page is diverted.
430 *
431 * @since 2.3.1
432 *
433 * @param \WP_Post $post Post being described.
434 * @return string Visible body text, or '' when `post_content` is fine.
435 */
436 public static function superseding_content(\WP_Post $post): string {
437 if (!self::bricks_supersedes_post_content((int) $post->ID)) {
438 return '';
439 }
440
441 $bricks = self::from_bricks((int) $post->ID);
442
443 return self::is_blank($bricks) ? '' : $bricks;
444 }
445
446 /**
447 * Body text to derive a description from, when the usual source is wrong.
448 *
449 * A hand-written excerpt is the author's own summary and is correct however
450 * the page is built, so it yields '' here and the caller's normal
451 * `get_the_excerpt()` path keeps it. Only a Bricks page with no excerpt —
452 * where core would derive one from discarded `post_content` — gets diverted.
453 *
454 * @since 2.3.1
455 *
456 * @param \WP_Post $post Post being described.
457 * @return string Text to summarize, or '' to leave the caller's path alone.
458 */
459 public static function superseding_excerpt_source(\WP_Post $post): string {
460 if ('' !== trim((string) $post->post_excerpt)) {
461 return '';
462 }
463
464 return self::superseding_content($post);
465 }
466
467 /**
468 * The Bricks element tree that renders for a post.
469 *
470 * Public because what Bricks puts on the page is not only a scoring
471 * question: the schema graph has to know whether a Bricks element already
472 * publishes the page's FAQ before adding one of its own (#649, #650).
473 *
474 * Flat, in Bricks' own storage shape — `expand_bricks_components()`
475 * appends component definitions to the same list rather than nesting them,
476 * so one `foreach` reaches every element.
477 *
478 * @since 2.3.1
479 *
480 * @param int $post_id Post being resolved.
481 * @return array<int,mixed> Elements, or [] when Bricks renders nothing here.
482 */
483 /**
484 * The builder meta keys, for callers that need to inspect the raw storage
485 * rather than the text extracted from it.
486 *
487 * The SEO Analyzer reads these to answer "is there a ThinkRank FAQ element
488 * on this post?", which is a question about the stored tree, not about the
489 * words in it (#686).
490 *
491 * @since 2.7.0
492 * @return string[]
493 */
494 public static function builder_meta_keys(): array {
495 return self::BUILDER_META_KEYS;
496 }
497
498 public static function bricks_tree(int $post_id): array {
499 if (array_key_exists($post_id, self::$bricks_trees)) {
500 return self::$bricks_trees[$post_id];
501 }
502
503 self::$bricks_trees[$post_id] = self::resolve_bricks_tree($post_id);
504
505 return self::$bricks_trees[$post_id];
506 }
507
508 /**
509 * Discard the resolved-tree memo. Test seam.
510 *
511 * @since 2.3.1
512 * @return void
513 */
514 public static function flush_bricks_cache(): void {
515 self::$bricks_trees = [];
516 }
517
518 /**
519 * Read and resolve a post's Bricks tree, ignoring the memo.
520 *
521 * @since 2.3.1
522 *
523 * @param int $post_id Post being resolved.
524 * @return array<int,mixed>
525 */
526 private static function resolve_bricks_tree(int $post_id): array {
527 if (!self::bricks_owns_post($post_id)) {
528 return [];
529 }
530
531 $source = self::bricks_content_source($post_id);
532 if (!$source) {
533 return [];
534 }
535
536 $stored = get_post_meta($source, self::bricks_meta_key(), true);
537
538 if (is_string($stored)) {
539 $stored = '' === trim($stored) ? null : json_decode($stored, true);
540 }
541
542 if (!is_array($stored) || empty($stored)) {
543 return [];
544 }
545
546 return self::expand_bricks_components($stored);
547 }
548
549 /**
550 * Resolve an arbitrary chunk of editor markup for the given post.
551 *
552 * The editor sends its live content to the scorer so an author sees their
553 * unsaved edits reflected. On a builder page that live string is the raw
554 * builder markup — the block editor hands over Divi's
555 * `<!-- wp:divi/... -->` comments verbatim, because it cannot render
556 * blocks it has no client-side registration for. Analyzed as-is it reads
557 * as zero words, which is how a Divi page could show a correct saved score
558 * while the live Content Analysis panel next to it still said
559 * "No content".
560 *
561 * Running the live string through the same chain as stored content keeps
562 * both paths honest, and falling through to the post's builder storage
563 * covers builders (Oxygen) whose editor content is empty to begin with.
564 *
565 * @since 1.23.0
566 *
567 * @param string $raw Markup to analyze.
568 * @param \WP_Post $post Post the markup belongs to.
569 * @return string Content to analyze.
570 */
571 public static function resolve_markup(string $raw, \WP_Post $post): string {
572 $content = self::render_post_content($raw, $post);
573
574 // Block markup that renders to nothing usually means the builder that
575 // owns those blocks did not register them in this context — Divi 5
576 // loads its module library lazily per-request, so in CLI, REST, admin
577 // and block-editor requests do_blocks() yields an empty string while
578 // the words sit right there in the block attributes. Read them
579 // directly.
580 if (self::is_blank($content)) {
581 $from_blocks = self::from_block_attributes($raw);
582 if (!self::is_blank($from_blocks)) {
583 $content = $from_blocks;
584 }
585 }
586
587 // Only reach for builder storage when the markup yielded nothing — a
588 // classic post must never pay for this.
589 if (self::is_blank($content)) {
590 $builder = self::from_builder_meta((int) $post->ID);
591 if (!self::is_blank($builder)) {
592 $content = $builder;
593 }
594 }
595
596 // A resolution that collapsed to nothing is worse than the raw markup.
597 if (self::is_blank($content) && !self::is_blank($raw)) {
598 $content = $raw;
599 }
600
601 /**
602 * Filter the content ThinkRank analyzes for a post.
603 *
604 * Use this to teach ThinkRank about a builder it does not know, or to
605 * override extraction for one it does.
606 *
607 * @since 1.23.0
608 *
609 * @param string $content Resolved content.
610 * @param \WP_Post $post Post being analyzed.
611 * @param string $raw Markup this resolution started from.
612 */
613 return (string) apply_filters('thinkrank_analyzable_content', $content, $post, $raw);
614 }
615
616 /**
617 * Render blocks and shortcodes found in post_content.
618 *
619 * Best-effort: a third-party block that fatals must not take the whole
620 * score down with it.
621 *
622 * Runs as the post's own context, as it would on the front end. Admin and
623 * REST requests have no current post, so a shortcode reading
624 * `get_the_ID()` got nothing, and one looping a related-posts query left
625 * the global post on the last of them: its `wp_reset_postdata()` goes back
626 * to the main query's post, and there is none. On the Classic Editor this
627 * runs after the form prints its hidden `post_ID` and before the title and
628 * editor, which then showed the related post, and Update saved it over the
629 * original (#860).
630 *
631 * @param string $raw Raw post content.
632 * @param \WP_Post $post Post the content belongs to.
633 * @return string Rendered content.
634 */
635 private static function render_post_content(string $raw, \WP_Post $post): string {
636 if ('' === trim($raw)) {
637 return '';
638 }
639
640 $content = $raw;
641 $previous = self::snapshot_post_globals();
642
643 try {
644 // phpcs:ignore WordPress.WP.GlobalVariablesOverride.Prohibited -- Made current for the render, restored in finally.
645 $GLOBALS['post'] = $post;
646
647 // Fires `the_post`, which these paths never fired before: admin,
648 // REST and cron analysis had no current post at all. That is the
649 // same signal the front-end loop sends and it is what makes
650 // `get_the_ID()` work inside a shortcode, but it is a new call on a
651 // path that runs in bulk — the word-count index resolves every post
652 // it visits — so a theme that counts views on `the_post` will count
653 // them during indexing. Accepted deliberately: without it a
654 // shortcode cannot resolve its own post, which is the bug (#860).
655 if (function_exists('setup_postdata')) {
656 setup_postdata($post);
657 }
658
659 if (function_exists('has_blocks') && function_exists('do_blocks') && has_blocks($raw)) {
660 $content = do_blocks($raw);
661 }
662
663 // Block output can itself contain shortcodes, so this runs either way.
664 if (function_exists('do_shortcode') && strpos($content, '[') !== false) {
665 $content = do_shortcode($content);
666 }
667 } catch (\Throwable $e) {
668 return $raw;
669 } finally {
670 self::restore_post_globals($previous);
671 }
672
673 return self::is_blank($content) ? $raw : $content;
674 }
675
676 /**
677 * The post globals as they are now; a global that is unset has no key.
678 *
679 * @since 2.12.0
680 *
681 * @return array<string,mixed>
682 */
683 private static function snapshot_post_globals(): array {
684 $snapshot = [];
685
686 foreach (self::POSTDATA_GLOBALS as $name) {
687 if (array_key_exists($name, $GLOBALS)) {
688 $snapshot[$name] = $GLOBALS[$name];
689 }
690 }
691
692 return $snapshot;
693 }
694
695 /**
696 * Put the post globals back as snapshot_post_globals() found them.
697 *
698 * Assigned directly rather than through `setup_postdata()`: there may have
699 * been no post to set up, and re-running it would fire `the_post` again.
700 *
701 * @since 2.12.0
702 *
703 * @param array<string,mixed> $snapshot From snapshot_post_globals().
704 * @return void
705 */
706 private static function restore_post_globals(array $snapshot): void {
707 foreach (self::POSTDATA_GLOBALS as $name) {
708 if (array_key_exists($name, $snapshot)) {
709 // phpcs:ignore WordPress.NamingConventions.PrefixAllGlobals.NonPrefixedVariableFound -- Core's own globals, put back as they were.
710 $GLOBALS[$name] = $snapshot[$name];
711 } else {
712 unset($GLOBALS[$name]);
713 }
714 }
715 }
716
717 /**
718 * Extract text from the attributes of parsed blocks.
719 *
720 * @param string $raw Raw post content containing block markup.
721 * @return string Collected text, or '' when nothing was found.
722 */
723 private static function from_block_attributes(string $raw): string {
724 if (!function_exists('parse_blocks') || !function_exists('has_blocks') || !has_blocks($raw)) {
725 return '';
726 }
727
728 try {
729 $blocks = parse_blocks($raw);
730 } catch (\Throwable $e) {
731 return '';
732 }
733
734 $attrs = [];
735 $collect = static function (array $items) use (&$collect, &$attrs): void {
736 foreach ($items as $block) {
737 if (!empty($block['attrs']) && is_array($block['attrs'])) {
738 $attrs[] = $block['attrs'];
739 }
740 if (!empty($block['innerBlocks']) && is_array($block['innerBlocks'])) {
741 $collect($block['innerBlocks']);
742 }
743 }
744 };
745 $collect($blocks);
746
747 return empty($attrs) ? '' : self::text_from_tree($attrs);
748 }
749
750 /**
751 * Everything Bricks contributes to this post's analyzable content.
752 *
753 * Bricks is the only builder here that needs more than a meta key, on
754 * three counts:
755 *
756 * - It leaves its stored tree behind when a post is switched back to the
757 * block editor, so an editor-mode gate has to run first or ThinkRank
758 * scores markup the visitor never sees — the same failure
759 * `_fl_builder_draft` was ordered against in #449.
760 * - A post's content can live on ANOTHER post. Bricks' Templates feature
761 * assigns a content template by condition, and a page using one stores
762 * nothing of its own; reading only the page's meta scores it blank
763 * while the visitor reads a full page.
764 * - Its stored text carries dynamic-data tags and internal element names
765 * that never reach the rendered page.
766 *
767 * @since 2.2.1
768 *
769 * @param int $post_id Post being resolved.
770 * @return string Extracted text, or '' when Bricks has nothing for it.
771 */
772 private static function from_bricks(int $post_id): string {
773 $tree = self::bricks_tree($post_id);
774
775 if (empty($tree)) {
776 return '';
777 }
778
779 return self::strip_bricks_dynamic_tags(
780 self::text_from_tree(self::without_bricks_element_labels($tree))
781 );
782 }
783
784 /**
785 * Whether Bricks — not the block editor — renders this post.
786 *
787 * Bricks writes `bricks` or `wordpress` into its editor-mode meta as the
788 * author toggles between the two, and never clears the content it stored
789 * for the other mode. Only the `wordpress` value is disqualifying: an
790 * absent value is the normal state for a post Bricks built and never
791 * toggled. This follows Bricks' own `Helpers::render_with_bricks()`, which
792 * bails on exactly that one value.
793 *
794 * It deliberately does not match it exactly: the comparison here is
795 * case-insensitive, where Bricks' is strict. Bricks 2.3.12 only ever writes
796 * the value lowercase, so the two agree on everything Bricks itself
797 * stores; they part company only on a value some other integration wrote.
798 * The two shipping today disagree about the casing — SureRank compares
799 * against `'WordPress'`, AIOSEO against `'bricks'` — and of the two ways to
800 * be wrong about `'WordPress'`, blocking costs a score on a page that has
801 * one, while allowing scores stale content the visitor never sees, which is
802 * the failure this gate exists to prevent.
803 *
804 * @since 2.2.1
805 *
806 * @param int $post_id Post being resolved.
807 * @return bool
808 */
809 private static function bricks_owns_post(int $post_id): bool {
810 $mode = get_post_meta($post_id, self::bricks_editor_mode_key(), true);
811
812 // phpcs:ignore WordPress.WP.CapitalPDangit.MisspelledInText -- Bricks' own stored meta value, lower-cased for the comparison.
813 return !(is_string($mode) && 'wordpress' === strtolower(trim($mode)));
814 }
815
816 /**
817 * The post whose Bricks tree actually renders for this post.
818 *
819 * Usually the post itself. When it stores nothing of its own, Bricks falls
820 * back to whichever content template's conditions match, and that template
821 * is a separate post carrying the words the visitor reads.
822 *
823 * Resolution is delegated to Bricks rather than reimplemented: template
824 * conditions are a whole rules engine (post IDs, types, taxonomies,
825 * archives), and a second implementation would drift from it. Bricks
826 * answers through statics, so they are saved and restored around the call —
827 * `set_active_templates()` returns early once populated, and on a
828 * front-end request Bricks has already populated it for the page being
829 * served. Clobbering that would corrupt the render in progress.
830 *
831 * Best-effort by design: any failure returns the post's own data, which is
832 * exactly today's behaviour.
833 *
834 * @since 2.2.1
835 *
836 * @param int $post_id Post being resolved.
837 * @return int Post ID holding the Bricks tree, or 0 when there is none.
838 */
839 private static function bricks_content_source(int $post_id): int {
840 $own = get_post_meta($post_id, self::bricks_meta_key(), true);
841 if ((is_array($own) && !empty($own)) || (is_string($own) && '' !== trim($own))) {
842 return $post_id;
843 }
844
845 if (!class_exists('\\Bricks\\Database')
846 || !method_exists('\\Bricks\\Database', 'set_active_templates')
847 ) {
848 return 0;
849 }
850
851 // `set_active_templates()` writes TWO statics — `$active_templates` and,
852 // when a header template resolves, `$header_position`. Both are saved,
853 // and both are restored in `finally` rather than on the happy path: a
854 // throw part-way through (a third-party hook on
855 // `bricks/database/content_type`, `bricks/builder/data_post_id` or
856 // `bricks/active_templates` is enough) must not leave Bricks' render
857 // state holding this lookup's values. Restoring only after a clean
858 // return is what the `catch` below would otherwise skip.
859 $has_header_position = property_exists('\\Bricks\\Database', 'header_position');
860 $saved_templates = \Bricks\Database::$active_templates;
861 $saved_header_position = $has_header_position ? \Bricks\Database::$header_position : null;
862
863 try {
864 \Bricks\Database::$active_templates = [];
865 \Bricks\Database::set_active_templates($post_id);
866 $template = (int) (\Bricks\Database::$active_templates['content'] ?? 0);
867 } catch (\Throwable $e) {
868 return 0;
869 } finally {
870 \Bricks\Database::$active_templates = $saved_templates;
871 if ($has_header_position) {
872 \Bricks\Database::$header_position = $saved_header_position;
873 }
874 }
875
876 // A template that is the post itself adds nothing over the empty read
877 // above, and would otherwise recurse conceptually.
878 return $template === $post_id ? 0 : $template;
879 }
880
881 /**
882 * Splice component definitions into the tree.
883 *
884 * A Bricks component keeps its markup in the `bricks_components` option,
885 * not on the page. The page stores only an instance: an element carrying
886 * `cid` and, usually, empty `settings`. Walking the page alone therefore
887 * found no words at all, and a page built entirely from components scored
888 * blank — the same failure as a page built from a content template.
889 *
890 * Confirmed on Bricks 2.3.12: `Bricks\Frontend::render_data()` renders the
891 * component's copy from an instance this walker extracted '' from.
892 *
893 * The definition is read straight from the option rather than through
894 * `Bricks\Helpers::get_component_instance()`. That helper resolves an
895 * instance's property overrides, which would be better, but it reads
896 * `Bricks\Database::$global_data['components']` — populated once per
897 * request, and empty in the admin and CLI contexts where bulk scoring
898 * runs. Refreshing it would mean writing to Bricks' live render state, the
899 * same hazard the template resolver is careful to avoid, and gating on it
900 * would make a page score differently in wp-admin than on the front end.
901 * Reading the stored definition is consistent everywhere.
902 *
903 * The trade-off: an instance that overrides a component property is scored
904 * with the component's authored copy rather than the override. That is the
905 * text the component renders by default, and it is much closer than the
906 * nothing this returned before.
907 *
908 * @since 2.2.1
909 *
910 * @param array $tree Bricks content area.
911 * @return array Tree with component elements spliced in after each instance.
912 */
913 private static function expand_bricks_components(array $tree): array {
914 $expanded = [];
915 $open = [];
916
917 $walk = static function (array $elements, int $depth) use (&$walk, &$expanded, &$open): void {
918 foreach ($elements as $element) {
919 $expanded[] = $element;
920
921 if (!is_array($element) || empty($element['cid']) || !is_string($element['cid'])) {
922 continue;
923 }
924
925 $cid = $element['cid'];
926
927 // A component nested inside its own definition would recurse
928 // forever; the depth cap covers deep but legitimate nesting.
929 if (isset($open[$cid]) || $depth > 4) {
930 continue;
931 }
932
933 $children = self::bricks_component_elements($cid);
934 if (empty($children)) {
935 continue;
936 }
937
938 // Re-entrant per branch, not per page: the guard is released
939 // after the walk so a second instance further along the page
940 // still expands, rather than being mistaken for recursion.
941 //
942 // That does NOT double the word count — `text_from_tree()`
943 // ends in `array_unique()`, which collapses a repeated
944 // component's copy the same way it collapses a value repeated
945 // across responsive breakpoints. Expanding both instances is
946 // about not silently dropping the second one's structure.
947 $open[$cid] = true;
948 $walk($children, $depth + 1);
949 unset($open[$cid]);
950 }
951 };
952
953 $walk($tree, 0);
954
955 return $expanded;
956 }
957
958 /**
959 * The stored elements of one Bricks component.
960 *
961 * @since 2.2.1
962 *
963 * @param string $cid Component id held by an instance element.
964 * @return array Component elements, or [] when it cannot be resolved.
965 */
966 private static function bricks_component_elements(string $cid): array {
967 $components = get_option(self::bricks_constant('BRICKS_DB_COMPONENTS', self::BRICKS_COMPONENTS_OPTION), []);
968
969 if (!is_array($components)) {
970 return [];
971 }
972
973 foreach ($components as $component) {
974 $component = self::as_children($component);
975 if (null === $component) {
976 continue;
977 }
978
979 if (isset($component['id']) && $component['id'] === $cid && !empty($component['elements'])) {
980 return is_array($component['elements']) ? $component['elements'] : [];
981 }
982 }
983
984 return [];
985 }
986
987 /**
988 * Drop each Bricks element's internal name before the tree is walked.
989 *
990 * A Bricks element carries an optional top-level `label` — the nickname an
991 * author types in the Structure panel to find it again ("Hero headline",
992 * "CTA row"). It is builder chrome and is never rendered, but `label` is in
993 * CONTENT_KEYS because it is real content for other builders' form fields,
994 * so it was being counted as page copy.
995 *
996 * Only the element's own `label` is removed. A `label` inside `settings`
997 * is a rendered field label and stays.
998 *
999 * @since 2.2.1
1000 *
1001 * @param array $tree Bricks content area.
1002 * @return array Tree with element nicknames removed.
1003 */
1004 private static function without_bricks_element_labels(array $tree): array {
1005 foreach ($tree as $index => $element) {
1006 if (is_array($element) && isset($element['id'], $element['label'])) {
1007 unset($tree[$index]['label']);
1008 }
1009 }
1010
1011 return $tree;
1012 }
1013
1014 /**
1015 * Remove Bricks dynamic-data tags from extracted text.
1016 *
1017 * Bricks stores `{post_title}`, `{post_meta:price}`, `{echo:my_fn}` and the
1018 * like verbatim and resolves them when it renders. Extraction reads the
1019 * stored tree, so without this the placeholders were counted as words, and
1020 * a heading whose text is `{post_title}` reported the literal token as its
1021 * heading text.
1022 *
1023 * The pattern is deliberately narrower than Bricks' own
1024 * (`/{([\wÀ-ÖØ-öø-ÿ\-\s\.\/:\(\)...]+)}/u`), which also matches braces
1025 * containing spaces. Bricks only substitutes tags that resolve to a
1026 * registered provider and leaves anything else on the page as literal text,
1027 * so the broad pattern would delete prose the visitor can actually read.
1028 * Matching only tag-shaped tokens keeps every real sentence and still
1029 * removes every placeholder — the same trade-off SureRank makes.
1030 *
1031 * @since 2.2.1
1032 *
1033 * @param string $text Extracted text.
1034 * @return string Text with placeholders removed.
1035 */
1036 private static function strip_bricks_dynamic_tags(string $text): string {
1037 $stripped = preg_replace('/\{[a-z0-9_][a-z0-9_:\-\.]*\}/i', '', $text);
1038
1039 if (null === $stripped) {
1040 return $text;
1041 }
1042
1043 // Collapse the runs of spaces a removed tag leaves mid-sentence,
1044 // without touching the newlines that separate collected nodes.
1045 $tidied = preg_replace('/[ \t]{2,}/', ' ', $stripped);
1046
1047 return null === $tidied ? $stripped : $tidied;
1048 }
1049
1050 /**
1051 * Bricks' content-area meta key, preferring Bricks' own constant.
1052 *
1053 * @since 2.2.1
1054 *
1055 * @return string
1056 */
1057 private static function bricks_meta_key(): string {
1058 return self::bricks_constant('BRICKS_DB_PAGE_CONTENT', self::BRICKS_CONTENT_META_KEY);
1059 }
1060
1061 /**
1062 * Bricks' editor-mode meta key, preferring Bricks' own constant.
1063 *
1064 * @since 2.2.1
1065 *
1066 * @return string
1067 */
1068 private static function bricks_editor_mode_key(): string {
1069 return self::bricks_constant('BRICKS_DB_EDITOR_MODE', self::BRICKS_EDITOR_MODE_META_KEY);
1070 }
1071
1072 /**
1073 * Read one of Bricks' key-name constants, falling back to the literal.
1074 *
1075 * @since 2.2.1
1076 *
1077 * @param string $name Constant name.
1078 * @param string $fallback Key to use when the constant is unavailable.
1079 * @return string
1080 */
1081 private static function bricks_constant(string $name, string $fallback): string {
1082 if (defined($name)) {
1083 $value = constant($name);
1084 if (is_string($value) && '' !== trim($value)) {
1085 return $value;
1086 }
1087 }
1088
1089 return $fallback;
1090 }
1091
1092 /**
1093 * Pull text out of whichever builder stored this post.
1094 *
1095 * @param int $post_id Post ID.
1096 * @return string Extracted text, or '' when no builder data was found.
1097 */
1098 private static function from_builder_meta(int $post_id): string {
1099 // Bricks first: it is the only builder whose content can live on
1100 // another post, and the only one gated on an editor mode.
1101 $bricks = self::from_bricks($post_id);
1102 if (!self::is_blank($bricks)) {
1103 return $bricks;
1104 }
1105
1106 foreach (self::BUILDER_META_KEYS as $key) {
1107 if (self::is_oxygen_classic_key($key)) {
1108 // Resolved as a pair, once, at the first of its keys.
1109 if ('_ct_builder_json' !== $key) {
1110 continue;
1111 }
1112
1113 $oxygen = self::from_oxygen_classic($post_id);
1114 if (!self::is_blank($oxygen)) {
1115 return $oxygen;
1116 }
1117 continue;
1118 }
1119
1120 $stored = get_post_meta($post_id, $key, true);
1121
1122 if (is_string($stored) && '' !== trim($stored)) {
1123 $decoded = json_decode($stored, true);
1124
1125 // JSON node tree (Breakdance/Oxygen 6, Elementor).
1126 if (is_array($decoded)) {
1127 $text = self::text_from_tree($decoded);
1128 if (!self::is_blank($text)) {
1129 return $text;
1130 }
1131 continue;
1132 }
1133
1134 continue;
1135 }
1136
1137 // Some builders store an already-decoded tree — an array for most,
1138 // an array of objects for Beaver Builder (#449).
1139 $tree = self::as_children($stored);
1140 if (null !== $tree) {
1141 $text = self::text_from_tree($tree);
1142 if (!self::is_blank($text)) {
1143 return $text;
1144 }
1145 }
1146 }
1147
1148 return '';
1149 }
1150
1151 /**
1152 * Whether a meta key is one of Oxygen classic's storage keys.
1153 *
1154 * @since 2.10.0
1155 *
1156 * @param string $key Meta key.
1157 * @return bool
1158 */
1159 private static function is_oxygen_classic_key(string $key): bool {
1160 return isset(self::OXYGEN_CLASSIC_KEYS[$key]) || in_array($key, self::OXYGEN_CLASSIC_KEYS, true);
1161 }
1162
1163 /**
1164 * Text of an Oxygen classic page, from whichever stored form holds more.
1165 *
1166 * Oxygen 4.x keeps the same tree twice: as JSON, and as the shortcodes it
1167 * used before 4.0. The JSON is preferred because it carries copy the
1168 * shortcode form hides (a composite element's text is base64-encoded
1169 * inside `ct_options`, which is configuration and stripped). It is not
1170 * trusted blindly, though. Reading `ct_builder_json` first once meant a
1171 * key missing from CONTENT_KEYS silently threw the page away while the
1172 * shortcode copy sat unread next to it, because a non-empty JSON result
1173 * stopped the search. Comparing the two means the next such gap costs
1174 * nothing: the richer form wins.
1175 *
1176 * A generation is only read as a pair. The prefixed keys are what Oxygen
1177 * 4.8.3+ reads, so an unprefixed leftover next to them is stale.
1178 *
1179 * @since 2.10.0
1180 *
1181 * @param int $post_id Post ID.
1182 * @return string Extracted text, or '' when Oxygen classic stored nothing.
1183 */
1184 private static function from_oxygen_classic(int $post_id): string {
1185 foreach (self::OXYGEN_CLASSIC_KEYS as $json_key => $shortcode_key) {
1186 $json = get_post_meta($post_id, $json_key, true);
1187 $shortcodes = get_post_meta($post_id, $shortcode_key, true);
1188
1189 $from_json = '';
1190 if (is_string($json) && '' !== trim($json)) {
1191 $decoded = json_decode($json, true);
1192 if (is_array($decoded)) {
1193 // `[oxygen data="..."]` is a dynamic-data placeholder
1194 // Oxygen fills at render time. The shortcode path drops it
1195 // with every other tag, so it goes here too or the two
1196 // forms would disagree on the same page.
1197 $from_json = (string) preg_replace(
1198 '/\[oxygen\b[^\]]*\]/i',
1199 ' ',
1200 self::text_from_tree($decoded)
1201 );
1202 }
1203 }
1204
1205 $from_shortcodes = '';
1206 if (is_string($shortcodes) && strpos($shortcodes, '[') !== false) {
1207 $from_shortcodes = self::text_from_shortcodes($shortcodes);
1208 }
1209
1210 if (self::is_blank($from_json) && self::is_blank($from_shortcodes)) {
1211 continue;
1212 }
1213
1214 return self::visible_word_count($from_json) >= self::visible_word_count($from_shortcodes)
1215 ? $from_json
1216 : $from_shortcodes;
1217 }
1218
1219 return '';
1220 }
1221
1222 /**
1223 * Rough count of the words a visitor would read in extracted text.
1224 *
1225 * Only used to compare two extractions of the same page, so it needs to
1226 * be consistent rather than locale-exact.
1227 *
1228 * @since 2.10.0
1229 *
1230 * @param string $text Extracted text or markup.
1231 * @return int
1232 */
1233 private static function visible_word_count(string $text): int {
1234 $plain = trim((string) preg_replace('/\s+/u', ' ', wp_strip_all_tags($text)));
1235
1236 return '' === $plain ? 0 : count(explode(' ', $plain));
1237 }
1238
1239 /**
1240 * Shortcode attributes that carry copy a visitor reads.
1241 *
1242 * An allow-list, not a deny-list. Oxygen Classic tags carry far more
1243 * attributes than they do copy — `id`, `class`, `selector`, `url`,
1244 * `ct_options` and friends — and a deny-list silently admits every
1245 * attribute a future builder release invents, which is how markup ends up
1246 * being counted as prose.
1247 *
1248 * @var string[]
1249 */
1250 private const SHORTCODE_TEXT_ATTRIBUTES = [
1251 'text',
1252 'content',
1253 'heading',
1254 'title',
1255 'subtitle',
1256 'label',
1257 'caption',
1258 'description',
1259 'alt',
1260 'button_text',
1261 'link_text',
1262 ];
1263
1264 /**
1265 * Extract readable text from a shortcode tree, without rendering it.
1266 *
1267 * Oxygen Classic is the only builder whose storage is shortcodes rather
1268 * than JSON, and the previous implementation handed the string to
1269 * `do_shortcode()`. That silently depends on Oxygen having registered its
1270 * `ct_*` handlers in the current request — which it has on a front-end
1271 * view, and has not during bulk analysis, the post-list column, cron or
1272 * REST/MCP. With no handlers registered `do_shortcode()` returns its input
1273 * unchanged, so the raw shortcode source was scored as if it were the
1274 * page's prose: `[ct_section`, `id="section-1"` and the rest counted toward
1275 * the word count, while the actual copy sitting in `text="..."` attributes
1276 * was never counted at all (#776).
1277 *
1278 * `strip_shortcodes()` is no help either — it also only knows registered
1279 * shortcodes, so it leaves the same text untouched.
1280 *
1281 * Reading the stored tree directly is what every other builder here already
1282 * does, and it matches the class's stated design: no render engine, no
1283 * dependency on load order, safe during a bulk run.
1284 *
1285 * Parsing unconditionally, rather than rendering when Oxygen happens to be
1286 * loaded and parsing otherwise, is deliberate. It makes the extracted text
1287 * the same in every context, so the score in the editor matches the score
1288 * from a bulk run or from MCP. The old code produced whichever of the two
1289 * the request happened to allow, which is why the same post could report
1290 * two different word counts depending on how it was asked.
1291 *
1292 * The trade-off is that rendered output (resolved images, links, anything
1293 * Oxygen pulls in from a reusable part) is no longer reflected here. For
1294 * what this text feeds — word count, content scoring, meta-description
1295 * fallbacks and schema text — that markup was never the point, and counting
1296 * it only when the builder happened to be booted was the bug.
1297 *
1298 * @since 2.10.0
1299 *
1300 * @param string $stored Raw shortcode source.
1301 * @return string Extracted text.
1302 */
1303 private static function text_from_shortcodes(string $stored): string {
1304 // Oxygen stores each element's settings as a JSON blob in `ct_options`.
1305 // It is configuration, never copy, and it contains braces and brackets
1306 // that would otherwise confuse the tag scan below, so it goes first.
1307 //
1308 // The blob is matched as a balanced JSON object, not as "up to the
1309 // next quote". Oxygen wraps it in single quotes but does not escape
1310 // an apostrophe inside it (`"nicename":"Bob's Plumbing"`), so the
1311 // quote-to-quote match stopped mid-value and the rest of the blob,
1312 // `s Plumbing"}'` and all, was left in the tag and leaked into the
1313 // text. Strings inside the object are skipped whole, so neither a quote
1314 // nor a brace inside a value can end the match early.
1315 $source = (string) preg_replace(
1316 '/\sct_options\s*=\s*\'(?<obj>\{(?:[^{}"]++|"(?:[^"\\\\]|\\\\.)*+"|(?&obj))*+\})\'/s',
1317 '',
1318 $stored
1319 );
1320
1321 // Anything not shaped like Oxygen's JSON blob keeps the old,
1322 // quote-delimited strip.
1323 $source = (string) preg_replace(
1324 '/\sct_options\s*=\s*(["\']).*?\1/s',
1325 '',
1326 $source
1327 );
1328
1329 $attributes = implode('|', array_map(
1330 static fn(string $name): string => preg_quote($name, '/'),
1331 self::SHORTCODE_TEXT_ATTRIBUTES
1332 ));
1333
1334 // Replace each shortcode tag with whatever readable copy its attributes
1335 // carry. Text between tags is left exactly where it is, so the result
1336 // keeps the page's reading order rather than hoisting all the headings
1337 // to the front.
1338 // The attribute blob is matched quote-aware rather than as "anything up
1339 // to the first `]`". Oxygen copy contains brackets often enough to
1340 // matter — "Best tools [2026]", "[Updated] our policy" — and a naive
1341 // scan ends the tag inside the `text` attribute, dropping the copy
1342 // before the bracket and leaking the stray `"]` after it into the
1343 // prose. Which is this bug's own failure mode: the wrong text scored.
1344 //
1345 // A tag name must start with a letter or underscore. `[2026]` is not a
1346 // shortcode anyone can register, and scanning it as one dropped the
1347 // year out of "Best tools [2026]".
1348 $text = (string) preg_replace_callback(
1349 '/\[\/?[a-zA-Z_][a-zA-Z0-9_-]*((?:[^\]"\']|"[^"]*"|\'[^\']*\')*)\]/',
1350 static function (array $matches) use ($attributes): string {
1351 if ('' === trim($matches[1])) {
1352 return ' ';
1353 }
1354
1355 if (!preg_match_all(
1356 '/\b(' . $attributes . ')\s*=\s*(["\'])(.*?)\2/s',
1357 $matches[1],
1358 $found,
1359 PREG_SET_ORDER
1360 )) {
1361 return ' ';
1362 }
1363
1364 $parts = [];
1365 foreach ($found as $attribute) {
1366 $value = trim($attribute[3]);
1367
1368 // An attribute holding markup or a JSON fragment is
1369 // configuration that happens to share a name with a copy
1370 // field, not something a visitor reads.
1371 if ('' === $value || preg_match('/^[\[{<]/', $value)) {
1372 continue;
1373 }
1374
1375 $parts[] = $value;
1376 }
1377
1378 return empty($parts) ? ' ' : ' ' . implode(' ', $parts) . ' ';
1379 },
1380 $source
1381 );
1382
1383 // Oxygen escapes square brackets in an element's copy before writing
1384 // it between the tags, so that "Best tools [2026]" cannot be mistaken
1385 // for a shortcode (`oxygen_vsb_filter_shortcode_content_encode()`).
1386 // Decoded only now, after the tag scan, for the same reason; left
1387 // encoded, the placeholders were scored as words of their own.
1388 $text = str_replace(
1389 ['_OXY_OPENING_BRACKET_', '_OXY_CLOSING_BRACKET_'],
1390 ['[', ']'],
1391 $text
1392 );
1393
1394 // Entities are stored encoded in attributes (&amp;, &#8217;), and would
1395 // otherwise be counted as words.
1396 $text = html_entity_decode($text, ENT_QUOTES | ENT_HTML5, 'UTF-8');
1397
1398 return trim((string) preg_replace('/\s+/u', ' ', $text));
1399 }
1400
1401 /**
1402 * A node's children, whether it stores them as an array or an object.
1403 *
1404 * The walker used to return immediately on `!is_array($node)`, so an
1405 * object node was dropped along with its entire subtree — silently, as
1406 * `''`, which the caller reads as "this builder stored nothing" rather
1407 * than "this walker cannot read this shape".
1408 *
1409 * Beaver Builder stores `_fl_builder_data` as an array of stdClass nodes,
1410 * each with a stdClass `settings` object, so every node would have been
1411 * dropped and adding its meta key alone would have looked like it worked
1412 * and changed nothing. Not BB-specific: any builder storing objects hits
1413 * this, and that shape will come up again (#449).
1414 *
1415 * @since 2.1.0
1416 *
1417 * @param mixed $node Candidate node.
1418 * @return array<string|int,mixed>|null Traversable children, or null.
1419 */
1420 private static function as_children($node): ?array {
1421 if (is_array($node)) {
1422 return $node;
1423 }
1424
1425 // Deliberately not is_object(): a builder can store a value object
1426 // (DateTime, a WP_Post) whose properties are not content, and
1427 // get_object_vars() on those yields noise. stdClass is what the
1428 // JSON/serialize round-trip produces, which is the shape we want.
1429 if ($node instanceof \stdClass) {
1430 return get_object_vars($node);
1431 }
1432
1433 return null;
1434 }
1435
1436 /**
1437 * Walk a builder node tree and collect the user-visible text.
1438 *
1439 * Values are joined with block-level markup so downstream heading, link and
1440 * image detection keeps working on the result.
1441 *
1442 * @param array $tree Decoded builder tree.
1443 * @return string Collected HTML.
1444 */
1445 private static function text_from_tree(array $tree): string {
1446 $collected = [];
1447
1448 // Strings already represented inside reconstructed markup, so the plain
1449 // sweep below doesn't emit a link label or heading a second time and
1450 // double it in the word count.
1451 $consumed = [];
1452
1453 // Pass 1 — rebuild <a>, <img> and <hN> from node *shape*. This has to
1454 // happen per node rather than per leaf: a link's label and its
1455 // destination are separate sibling fields, so once the tree is
1456 // flattened to leaves the pairing is gone.
1457 $reconstruct = static function ($node) use (&$reconstruct, &$collected, &$consumed): void {
1458 $node = self::as_children($node);
1459 if (null === $node) {
1460 return;
1461 }
1462
1463 $markup = self::markup_for_node($node, $consumed);
1464 if ('' !== $markup) {
1465 $collected[] = $markup;
1466 }
1467
1468 foreach ($node as $child_key => $child) {
1469 // A `link` / `image` sub-object is a destination descriptor the
1470 // parent has already folded into its markup. Descending into it
1471 // would emit the same URL a second time as a bare link, and
1472 // would turn an image's own `url` field into a spurious <a>.
1473 if (is_string($child_key)
1474 && (in_array(strtolower($child_key), self::URL_KEYS, true)
1475 || in_array(strtolower($child_key), self::IMAGE_KEYS, true)
1476 || in_array(strtolower($child_key), self::VIDEO_KEYS, true))
1477 ) {
1478 continue;
1479 }
1480
1481 $reconstruct($child);
1482 }
1483 };
1484 $reconstruct($tree);
1485
1486 // Pass 2 — remaining visible text.
1487 $walk = static function ($node, $key = null) use (&$walk, &$collected, &$consumed): void {
1488 $children = self::as_children($node);
1489 if (null !== $children) {
1490 foreach ($children as $child_key => $child) {
1491 $walk($child, is_string($child_key) ? $child_key : $key);
1492 }
1493 return;
1494 }
1495
1496 if (!is_string($node) || '' === trim($node)) {
1497 return;
1498 }
1499
1500 // Already inside a reconstructed tag.
1501 if (in_array($node, $consumed, true)) {
1502 return;
1503 }
1504
1505 $is_content_key = is_string($key)
1506 && in_array(strtolower($key), self::CONTENT_KEYS, true);
1507
1508 // Markup is content wherever it appears; bare strings only count
1509 // when their key says they are content, so slugs and class names
1510 // stay out of the word count.
1511 if ($is_content_key || strpos($node, '<') !== false) {
1512 $collected[] = $node;
1513 }
1514 };
1515
1516 $walk($tree);
1517
1518 if (empty($collected)) {
1519 return '';
1520 }
1521
1522 // De-duplicate: builder trees often repeat a value across responsive
1523 // breakpoints, which would otherwise multiply the word count.
1524 $collected = array_unique($collected);
1525
1526 return implode("\n", $collected);
1527 }
1528
1529 /**
1530 * The video source a node is actually playing, if any.
1531 *
1532 * @since 2.3.1
1533 *
1534 * @param array $node Builder node.
1535 * @return string Video source, or '' when the node carries none.
1536 */
1537 private static function video_from(array $node): string {
1538 foreach ($node as $key => $value) {
1539 if (!is_string($key) || !is_string($value)) {
1540 continue;
1541 }
1542
1543 if (!in_array(strtolower($key), self::VIDEO_TYPE_KEYS, true)) {
1544 continue;
1545 }
1546
1547 $keys = self::VIDEO_KEYS_BY_TYPE[strtolower(trim($value))] ?? null;
1548 if (null === $keys) {
1549 continue;
1550 }
1551
1552 // A recognised video_type settles it, including when that
1553 // provider's own field is empty. Falling through to the flat sweep
1554 // there handed back whichever sibling key happened to come first in
1555 // node order — the stale youtube_url left behind after switching
1556 // the widget to a hosted file, which is exactly what keying on the
1557 // declared type is meant to prevent.
1558 $declared = self::url_from($node, $keys);
1559
1560 return self::is_video_source($declared) ? $declared : '';
1561 }
1562
1563 $url = self::url_from($node, self::VIDEO_KEYS);
1564
1565 return self::is_video_source($url) ? $url : '';
1566 }
1567
1568 /**
1569 * Whether a value can be a video source.
1570 *
1571 * `looks_like_url()` also accepts `#anchor`, `mailto:` and `tel:`, which a
1572 * link node may legitimately hold but a video cannot: `<iframe src="#top">`
1573 * is not a video and would reach a video sitemap as one.
1574 *
1575 * @since 2.3.1
1576 *
1577 * @param string $url Candidate source.
1578 * @return bool
1579 */
1580 private static function is_video_source(string $url): bool {
1581 return '' !== $url
1582 && (1 === preg_match('#^(https?:)?//#i', $url) || str_starts_with($url, '/'));
1583 }
1584
1585 /**
1586 * Whether a video source points at a file rather than a provider page.
1587 *
1588 * @since 2.3.1
1589 *
1590 * @param string $url Video source.
1591 * @return bool
1592 */
1593 private static function is_video_file(string $url): bool {
1594 $path = (string) wp_parse_url($url, PHP_URL_PATH);
1595 $ext = strtolower((string) pathinfo($path, PATHINFO_EXTENSION));
1596
1597 return in_array($ext, self::VIDEO_FILE_EXTENSIONS, true);
1598 }
1599
1600 /**
1601 * Rebuild the HTML a single builder node represents, if any.
1602 *
1603 * Looks only at the node's own fields (plus one level of nesting, because
1604 * builders commonly wrap a destination as `{ url: … }`). Returns an empty
1605 * string for the vast majority of nodes, which are layout or configuration.
1606 *
1607 * Any leaf string folded into the returned markup is appended to $consumed
1608 * so the plain-text sweep doesn't count it twice.
1609 *
1610 * @param array $node Builder node.
1611 * @param array $consumed Collects strings represented in the returned markup.
1612 * @return string Reconstructed HTML, or '' when the node carries none.
1613 */
1614 private static function markup_for_node(array $node, array &$consumed): string {
1615 $text = self::first_value($node, self::CONTENT_KEYS);
1616 $url = self::url_from($node, self::URL_KEYS);
1617 $image = self::image_from($node);
1618 $video = self::video_from($node);
1619 $tag = self::heading_tag_from($node);
1620
1621 $parts = [];
1622
1623 // Video: an embed shape rather than a link, so the video detector can
1624 // see it while the link counters do not mistake it for an outbound
1625 // link. A file source becomes <video src>, anything else an <iframe>,
1626 // matching how the builder itself renders the two cases.
1627 if ('' !== $video) {
1628 $parts[] = self::is_video_file($video)
1629 ? sprintf('<video src="%s"></video>', esc_url_raw($video))
1630 : sprintf('<iframe src="%s"></iframe>', esc_url_raw($video));
1631 }
1632
1633 // Image: alt text matters as much as the tag, since alt checks run over
1634 // whatever this returns.
1635 if ('' !== $image['url']) {
1636 $alt = '' !== $image['alt'] ? $image['alt'] : (string) self::first_value($node, self::ALT_KEYS);
1637 if ('' !== $alt) {
1638 $consumed[] = $alt;
1639 }
1640 $parts[] = sprintf(
1641 '<img src="%s" alt="%s" />',
1642 esc_url_raw($image['url']),
1643 htmlspecialchars($alt, ENT_QUOTES)
1644 );
1645 }
1646
1647 if ('' !== $text) {
1648 $inner = $text;
1649
1650 if ('' !== $url) {
1651 $consumed[] = $text;
1652 $inner = sprintf('<a href="%s">%s</a>', esc_url_raw($url), $text);
1653 }
1654
1655 if ('' !== $tag) {
1656 $consumed[] = $text;
1657 $parts[] = sprintf('<%1$s>%2$s</%1$s>', $tag, $inner);
1658 } elseif ('' !== $url) {
1659 $parts[] = $inner;
1660 }
1661 } elseif ('' !== $url) {
1662 // A destination with no label still counts as a link for link
1663 // checks; the URL doubles as its anchor text.
1664 $parts[] = sprintf('<a href="%1$s">%1$s</a>', esc_url_raw($url));
1665 }
1666
1667 return implode("\n", $parts);
1668 }
1669
1670 /**
1671 * First non-empty scalar value under any of the given keys.
1672 *
1673 * @param array $node Builder node.
1674 * @param string[] $keys Candidate keys.
1675 * @return string Trimmed value, or '' when none match.
1676 */
1677 private static function first_value(array $node, array $keys): string {
1678 foreach ($node as $key => $value) {
1679 if (!is_string($key) || !is_string($value)) {
1680 continue;
1681 }
1682 if (in_array(strtolower($key), $keys, true) && '' !== trim($value)) {
1683 return trim($value);
1684 }
1685 }
1686
1687 return '';
1688 }
1689
1690 /**
1691 * Link destination held by a node, as a bare string or a `{ url: … }` object.
1692 *
1693 * @param array $node Builder node.
1694 * @param string[] $keys Candidate keys.
1695 * @return string URL, or '' when the node holds none.
1696 */
1697 private static function url_from(array $node, array $keys): string {
1698 foreach ($node as $key => $value) {
1699 if (!is_string($key) || !in_array(strtolower($key), $keys, true)) {
1700 continue;
1701 }
1702
1703 if (is_string($value) && self::looks_like_url($value)) {
1704 return trim($value);
1705 }
1706
1707 // Elementor and Breakdance both nest the destination one level down.
1708 $nested_values = self::as_children($value);
1709 if (null !== $nested_values) {
1710 foreach ($nested_values as $nested_key => $nested) {
1711 if (is_string($nested_key)
1712 && in_array(strtolower($nested_key), ['url', 'href', 'permalink'], true)
1713 && is_string($nested)
1714 && self::looks_like_url($nested)
1715 ) {
1716 return trim($nested);
1717 }
1718 }
1719 }
1720 }
1721
1722 return '';
1723 }
1724
1725 /**
1726 * Image URL and alt text held by a node.
1727 *
1728 * @param array $node Builder node.
1729 * @return array{url:string,alt:string}
1730 */
1731 private static function image_from(array $node): array {
1732 foreach ($node as $key => $value) {
1733 if (!is_string($key) || !in_array(strtolower($key), self::IMAGE_KEYS, true)) {
1734 continue;
1735 }
1736
1737 if (is_string($value) && self::looks_like_url($value)) {
1738 return ['url' => trim($value), 'alt' => ''];
1739 }
1740
1741 $nested_values = self::as_children($value);
1742 if (null !== $nested_values) {
1743 $url = '';
1744 $alt = '';
1745 foreach ($nested_values as $nested_key => $nested) {
1746 if (!is_string($nested_key) || !is_string($nested)) {
1747 continue;
1748 }
1749 $nested_key = strtolower($nested_key);
1750 if ('' === $url && in_array($nested_key, ['url', 'src'], true) && self::looks_like_url($nested)) {
1751 $url = trim($nested);
1752 }
1753 if ('' === $alt && in_array($nested_key, self::ALT_KEYS, true)) {
1754 $alt = trim($nested);
1755 }
1756 }
1757 if ('' !== $url) {
1758 return ['url' => $url, 'alt' => $alt];
1759 }
1760 }
1761 }
1762
1763 return ['url' => '', 'alt' => ''];
1764 }
1765
1766 /**
1767 * Heading tag a node asks for, normalised to h1–h6.
1768 *
1769 * Accepts both the `h2` form and a bare level like `2`.
1770 *
1771 * @param array $node Builder node.
1772 * @return string Tag name, or '' when the node is not a heading.
1773 */
1774 private static function heading_tag_from(array $node): string {
1775 foreach ($node as $key => $value) {
1776 if (!is_string($key) || !in_array(strtolower($key), self::HEADING_TAG_KEYS, true)) {
1777 continue;
1778 }
1779
1780 if (is_string($value) && preg_match('/^h([1-6])$/i', trim($value), $m)) {
1781 return 'h' . $m[1];
1782 }
1783
1784 // A bare level only counts under a key that unambiguously means one;
1785 // `size` and `tag` carry values like "large" or "div" far more often.
1786 if (is_numeric($value)
1787 && in_array(strtolower($key), ['level'], true)
1788 && (int) $value >= 1 && (int) $value <= 6
1789 ) {
1790 return 'h' . (int) $value;
1791 }
1792 }
1793
1794 return '';
1795 }
1796
1797 /**
1798 * Whether a string is plausibly a link or asset destination.
1799 *
1800 * Deliberately permissive about relative paths — builders store internal
1801 * links that way — but rejects the option slugs and CSS values that make up
1802 * most of a builder tree.
1803 *
1804 * @param string $value Candidate.
1805 * @return bool
1806 */
1807 private static function looks_like_url(string $value): bool {
1808 $value = trim($value);
1809
1810 if ('' === $value || strlen($value) > 2048) {
1811 return false;
1812 }
1813
1814 if (preg_match('#^(https?:)?//#i', $value) || str_starts_with($value, '/')) {
1815 return true;
1816 }
1817
1818 // Protocol-ish destinations a link node can legitimately hold.
1819 return (bool) preg_match('#^(mailto:|tel:|\#)#i', $value);
1820 }
1821
1822 /**
1823 * Whether a value carries nothing worth analyzing.
1824 *
1825 * Readable text is the usual signal, but not the only one: a page can be
1826 * made entirely of media. A builder section holding just a gallery
1827 * reconstructs to `<img>` tags and one holding just a video widget to a
1828 * single `<iframe>` — both strip to an empty string, so a text-only test
1829 * discarded them here and the page fell through to the next builder key,
1830 * then to the raw markup, and finally reported as having no content at all.
1831 *
1832 * Comments are dropped before the tag test: the raw markup this class falls
1833 * back to on a builder page is unrendered block comments, which must stay
1834 * blank rather than be mistaken for reconstructed media.
1835 *
1836 * @param string $value Candidate content.
1837 * @return bool
1838 */
1839 private static function is_blank(string $value): bool {
1840 if ('' !== trim(wp_strip_all_tags($value))) {
1841 return false;
1842 }
1843
1844 $without_comments = (string) preg_replace('~<!--.*?-->~s', '', $value);
1845
1846 return 1 !== preg_match('~<(?:a|img|iframe|video|source)\b~i', $without_comments);
1847 }
1848 }
1849