PluginProbe
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO / 2.10.0
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO v2.10.0
2.10.0 2.9.0 2.8.0 2.7.0 2.6.0 2.5.0 2.4.0 2.3.0 2.2.0 2.1.1 2.1.0 2.0.2 2.0.1 2.0.0 1.32.0 1.31.0 1.30.0 1.29.0 1.28.0 1.27.0 1.26.0 1.25.0 trunk 1.0.0 1.0.1 All 51 releases
← All changes | includes/seo/class-builder-content.php +1468 -24 1.28.0 → 2.10.0 View file →
@@ -50,13 +50,109 @@
50 50 */
51 51 private const BUILDER_META_KEYS = [
52 52 '_breakdance_data', // Oxygen 6+ / Breakdance
53 53 '_oxygen_data', // Oxygen (earlier releases)
54 - 'ct_builder_shortcodes', // Oxygen classic
54 + // Oxygen classic. 4.x writes the tree as JSON to `ct_builder_json`
55 + // while still keeping `ct_builder_shortcodes`. A post carrying only
56 + // the JSON key used to match no key at all and fall through to an
57 + // empty `post_content`, which reads as a one-word page (#776).
58 + //
59 + // Oxygen 4.8.3 then renamed every `ct_*` post meta key to `_ct_*`
60 + // (`oxygen_vsb_update_4_8_3()` runs `oxy_prefix_meta_keys()` on
61 + // upgrade, and `oxy_get_post_meta()` only reads the prefixed name
62 + // from then on). A current Oxygen classic site therefore has only the
63 + // underscored keys, which nothing here listed, so every one of its
64 + // pages resolved as empty. The prefixed keys come first because they
65 + // are what Oxygen itself reads; the bare ones cover a site that has
66 + // not run the migration (or reverted it with `?unprefix_meta`).
67 + //
68 + // These four are not read in this loop: from_oxygen_classic() pairs
69 + // each JSON key with its shortcode sibling so the two forms can be
70 + // compared. They are listed here because this list is also what the
71 + // word-count index watches and what FAQ detection scans.
72 + '_ct_builder_json', // Oxygen classic 4.8.3+ (JSON tree)
73 + 'ct_builder_json', // Oxygen classic 4.0-4.8.2 (JSON tree)
74 + '_ct_builder_shortcodes', // Oxygen classic 4.8.3+ (shortcode tree)
75 + 'ct_builder_shortcodes', // Oxygen classic < 4.8.3 (shortcode tree)
55 76 '_elementor_data', // Elementor
77 + // Beaver Builder. Published layout first: `_fl_builder_draft` holds
78 + // unsaved changes and would score content the visitor cannot see.
79 + // Both are arrays of stdClass nodes, which is why the walker below
80 + // has to treat objects like arrays (#449).
81 + '_fl_builder_data', // Beaver Builder (published)
82 + '_fl_builder_draft', // Beaver Builder (unsaved changes)
56 83 ];
57 84
58 85 /**
86 + * Bricks' content-area meta key, used when Bricks itself isn't loaded.
87 + *
88 + * Bricks exposes `BRICKS_DB_PAGE_CONTENT` and renames the underlying key
89 + * between generations (it gained the `_2` suffix in 1.7.3), so the
90 + * constant is authoritative and this literal is only the fallback for the
91 + * contexts where it is undefined — Bricks is a theme, so on an admin or
92 + * CLI request against a site that has since switched themes the constant
93 + * is simply not there while the post meta still is.
94 + *
95 + * Bricks stores three areas — header, content and footer. Only the content
96 + * area belongs to the post being scored; the header and footer areas live
97 + * on Bricks' own template posts and would double-count site chrome into
98 + * every page's word count, so they are deliberately not read here.
99 + *
100 + * @since 2.2.1
101 + * @var string
102 + */
103 + private const BRICKS_CONTENT_META_KEY = '_bricks_page_content_2';
104 +
105 + /**
106 + * Bricks' per-post editor-mode meta key, used when Bricks isn't loaded.
107 + *
108 + * @since 2.2.1
109 + * @var string
110 + */
111 + private const BRICKS_EDITOR_MODE_META_KEY = '_bricks_editor_mode';
112 +
113 + /**
114 + * Bricks' components option, used when Bricks itself isn't loaded.
115 + *
116 + * @since 2.2.1
117 + * @var string
118 + */
119 + private const BRICKS_COMPONENTS_OPTION = 'bricks_components';
120 +
121 + /**
122 + * Bricks' element that renders the post's own `post_content`.
123 + *
124 + * A Bricks page normally discards `post_content` entirely, which is why
125 + * anything left there is invisible. Dropping this element onto the canvas
126 + * is the one way an author puts it back on the page, so its presence flips
127 + * `post_content` from stale leftovers to content the visitor reads.
128 + *
129 + * @since 2.3.1
130 + * @var string
131 + */
132 + private const BRICKS_POST_CONTENT_ELEMENT = 'post-content';
133 +
134 + /**
135 + * Resolved Bricks trees for this request, keyed by post ID.
136 + *
137 + * Rendering one page asks for the tree about twenty times — every
138 + * description, every schema node, the FAQ guard — and resolving it is not
139 + * free. `bricks_content_source()` clears `Bricks\Database::$active_templates`
140 + * before asking Bricks which content template applies, which defeats
141 + * Bricks' own early-return and re-runs its whole template-condition engine;
142 + * `expand_bricks_components()` then walks the tree again. Measured on a
143 + * Bricks page with no content of its own, that was ten full runs of the
144 + * rules engine per request.
145 + *
146 + * Per-request only, and only ever read back within one page render — a
147 + * request that writes Bricks content does not also render it.
148 + *
149 + * @since 2.3.1
150 + * @var array<int,array<int,mixed>>
151 + */
152 + private static array $bricks_trees = [];
153 +
154 + /**
59 155 * JSON keys whose values are user-visible text.
60 156 *
61 157 * Builder trees mix content with configuration, so a blind string sweep
62 158 * would count CSS classes and option slugs as words. Matching on the key
@@ -67,11 +163,148 @@
67 163 private const CONTENT_KEYS = [
68 164 'text', 'title', 'subtitle', 'heading', 'subheading', 'content',
69 165 'description', 'caption', 'excerpt', 'label', 'value', 'html',
70 166 'editor', 'quote', 'answer', 'question', 'body', 'button_text',
167 + // Oxygen classic keeps an element's copy in `options.ct_content`
168 + // (headline, text block, rich text, link and button labels). It is the
169 + // field Oxygen's own serializer moves between the tags when it writes
170 + // shortcodes (`parse_components_tree()`), and the one Relevanssi and
171 + // Oxygen's WPML integration read. Missing from this list, the walker
172 + // kept only copy that happened to contain markup: a page of plain
173 + // headings and paragraphs lost almost all of its words.
174 + 'ct_content',
175 + // Oxygen's composite elements keep their copy under `options.original`
176 + // instead, one key per field. Taken from the list Oxygen itself treats
177 + // as text when it serializes (`$options_to_encode`); the numeric price
178 + // fields and the progress bar's right-hand percentage are left out.
179 + 'testimonial_text', 'testimonial_author', 'testimonial_author_info',
180 + 'icon_box_heading', 'icon_box_text',
181 + 'pricing_box_package_title', 'pricing_box_package_subtitle', 'pricing_box_content',
182 + 'progress_bar_left_text',
71 183 ];
72 184
73 185 /**
186 + * Oxygen classic's storage generations, as JSON key => shortcode key.
187 + *
188 + * @since 2.10.0
189 + * @var array<string,string>
190 + */
191 + private const OXYGEN_CLASSIC_KEYS = [
192 + '_ct_builder_json' => '_ct_builder_shortcodes',
193 + 'ct_builder_json' => 'ct_builder_shortcodes',
194 + ];
195 +
196 + /**
197 + * JSON keys whose values hold a link destination.
198 + *
199 + * Builders store a link's destination in a structured field separate from
200 + * its label, either as a bare URL string or as a `{ url: … }` object.
201 + * Neither shape survives a text sweep — the key is not content and a bare
202 + * URL contains no `<` — so no `<a>` tag reached the link counters.
203 + *
204 + * @var string[]
205 + */
206 + private const URL_KEYS = [
207 + 'link', 'url', 'href', 'link_url', 'button_link', 'permalink', 'link_to',
208 + ];
209 +
210 + /**
211 + * JSON keys whose values hold an embedded video's source.
212 + *
213 + * A builder's video widget keeps its destination in a provider-specific
214 + * field — Elementor picks `youtube_url`, `vimeo_url`, `dailymotion_url` or
215 + * `hosted_url` according to the chosen source type — none of which is a
216 + * link field or a content field, so a video on a builder page reached the
217 + * analyzers as nothing at all.
218 + *
219 + * These are deliberately kept out of URL_KEYS. A video is an embed, not an
220 + * outbound link: rendering one as `<a href>` would add a spurious external
221 + * link to every page carrying a video and skew the link counts. They are
222 + * reconstructed as `<iframe>`/`<video>` instead, which the video detector
223 + * recognises and the link and image counters ignore.
224 + *
225 + * @since 2.3.1
226 + * @var string[]
227 + */
228 + private const VIDEO_KEYS = [
229 + 'youtube_url', 'vimeo_url', 'dailymotion_url', 'videopress_url',
230 + 'hosted_url', 'video_url', 'video_src', 'video_link',
231 + ];
232 +
233 + /**
234 + * File extensions that mean a video source is a file, not a provider page.
235 + *
236 + * @since 2.3.1
237 + * @var string[]
238 + */
239 + private const VIDEO_FILE_EXTENSIONS = ['mp4', 'webm', 'ogv', 'mov', 'm4v'];
240 +
241 + /**
242 + * Source keys to trust for a declared video source type.
243 + *
244 + * A widget keeps one field per provider and does not clear the others when
245 + * the author switches source: an Elementor video moved from YouTube to Self
246 + * Hosted still carries the earlier `youtube_url`. Reading whichever key
247 + * turns up first then emits the video the author replaced. The widget says
248 + * which one it is actually playing, so that is read first and the flat key
249 + * sweep is only the fallback for a builder that declares nothing.
250 + *
251 + * @since 2.3.1
252 + * @var array<string,string[]>
253 + */
254 + private const VIDEO_KEYS_BY_TYPE = [
255 + 'youtube' => ['youtube_url'],
256 + 'vimeo' => ['vimeo_url'],
257 + 'dailymotion' => ['dailymotion_url'],
258 + 'videopress' => ['videopress_url'],
259 + 'hosted' => ['hosted_url', 'video_url', 'video_src', 'video_link'],
260 + 'media' => ['hosted_url', 'video_url', 'video_src', 'video_link'],
261 + 'file' => ['hosted_url', 'video_url', 'video_src', 'video_link'],
262 + 'self_hosted' => ['hosted_url', 'video_url', 'video_src', 'video_link'],
263 + ];
264 +
265 + /**
266 + * Keys a builder uses to name which video source a widget is playing.
267 + *
268 + * @since 2.3.1
269 + * @var string[]
270 + */
271 + private const VIDEO_TYPE_KEYS = ['video_type', 'videotype', 'video_source', 'source_type'];
272 +
273 + /**
274 + * JSON keys whose values hold an image, as a URL string or `{ url, alt }`.
275 + *
276 + * @var string[]
277 + */
278 + private const IMAGE_KEYS = [
279 + 'image', 'src', 'image_url', 'background_image', 'bg_image', 'photo',
280 + ];
281 +
282 + /**
283 + * JSON keys that carry a heading level for the node's text.
284 + *
285 + * A builder heading's text is collected (its key is in CONTENT_KEYS) and so
286 + * counts toward the word count, but it arrives as bare text with no `<h2>`
287 + * wrapper — which is why heading-structure checks saw none.
288 + *
289 + * @var string[]
290 + */
291 + private const HEADING_TAG_KEYS = [
292 + // Lower-cased on both sides of the comparison, so `headingtag` is the
293 + // camelCase `headingTag` Bricks uses throughout its own controls and
294 + // which ThinkRank's Bricks elements declare. Without it their section
295 + // headings counted as body copy and never reached heading structure.
296 + 'header_size', 'heading_tag', 'headingtag', 'html_tag', 'title_tag', 'tag', 'level', 'size',
297 + ];
298 +
299 + /**
300 + * Keys whose value is alternative text for a sibling image.
301 + *
302 + * @var string[]
303 + */
304 + private const ALT_KEYS = ['alt', 'alt_text', 'image_alt', 'title'];
305 +
306 + /**
74 307 * Resolve the content worth analyzing for a post.
75 308 *
76 309 * @param \WP_Post $post Post being analyzed.
77 310 * @return string HTML/text to analyze.
@@ -76,12 +309,231 @@
76 309 * @param \WP_Post $post Post being analyzed.
77 310 * @return string HTML/text to analyze.
78 311 */
79 312 public static function resolve(\WP_Post $post): string {
80 - return self::resolve_markup((string) $post->post_content, $post);
313 + $raw = (string) $post->post_content;
314 +
315 + // A page built in Gutenberg and then switched to Bricks keeps its old
316 + // blocks in `post_content` forever — Bricks never clears them, and
317 + // never renders them either. Resolving that first meant the stale draft
318 + // beat the tree the visitor actually reads, and it did not stop at the
319 + // score: the same string becomes the meta description, og:description,
320 + // twitter:description and the schema description. Starting from nothing
321 + // sends the resolution straight to Bricks' storage, which is where this
322 + // page's words are (#651).
323 + //
324 + // Only for the stored path. `resolve_markup()` is also called with live
325 + // editor content, and the Bricks panel's resolver reads the canvas —
326 + // discarding that would replace what the author is typing with the last
327 + // save.
328 + if (self::bricks_supersedes_post_content((int) $post->ID)) {
329 + $raw = '';
330 + }
331 +
332 + return self::resolve_markup($raw, $post);
81 333 }
82 334
83 335 /**
336 + * Whether Bricks renders this post and throws its `post_content` away.
337 + *
338 + * True means anything still stored in `post_content` is invisible: it is
339 + * not on the page, so it must not be scored, described or published as
340 + * structured data. False covers both a post Bricks does not own and a
341 + * Bricks page that puts `post_content` back with a Post Content element.
342 + *
343 + * @since 2.3.1
344 + *
345 + * @param int $post_id Post being resolved.
346 + * @return bool
347 + */
348 + public static function bricks_supersedes_post_content(int $post_id): bool {
349 + $tree = self::bricks_tree($post_id);
350 +
351 + if (empty($tree)) {
352 + return false;
353 + }
354 +
355 + foreach ($tree as $element) {
356 + if (is_array($element)
357 + && self::BRICKS_POST_CONTENT_ELEMENT === ($element['name'] ?? null)
358 + ) {
359 + return false;
360 + }
361 + }
362 +
363 + return !self::bricks_tree_prints_post_content($tree);
364 + }
365 +
366 + /**
367 + * Whether a Bricks tree prints the body through a dynamic-data tag.
368 + *
369 + * The Post Content element is not the only way back onto the page: Bricks'
370 + * `{post_content}` tag renders the same thing from inside an ordinary text
371 + * element, and a single-post template written that way is a common shape.
372 + * Missing it would mean the post's real body is discarded everywhere —
373 + * scoring, the meta/og/twitter descriptions, the schema description — for a
374 + * page that is displaying it.
375 + *
376 + * Matched over the encoded tree rather than per setting, because the tag can
377 + * sit in any string field of any element and Bricks allows modifiers after
378 + * the name (`{post_content:...}`).
379 + *
380 + * @since 2.3.1
381 + *
382 + * @param array $tree Bricks element tree.
383 + * @return bool
384 + */
385 + private static function bricks_tree_prints_post_content(array $tree): bool {
386 + $encoded = wp_json_encode($tree);
387 +
388 + return is_string($encoded) && false !== stripos($encoded, '{post_content');
389 + }
390 +
391 + /**
392 + * The post's content as the visitor actually receives it.
393 + *
394 + * `post_content` for everything except a Bricks page that discards it, and
395 + * there the Bricks tree's text. Descriptions are derived from a post's body
396 + * in half a dozen places; every one of them wants this rather than the raw
397 + * column (#651).
398 + *
399 + * @since 2.3.1
400 + *
401 + * @param \WP_Post $post Post being described.
402 + * @return string
403 + */
404 + public static function visible_content(\WP_Post $post): string {
405 + $superseding = self::superseding_content($post);
406 +
407 + return '' !== $superseding ? $superseding : (string) $post->post_content;
408 + }
409 +
410 + /**
411 + * Replacement body text for a post whose `post_content` does not render.
412 + *
413 + * Empty for every ordinary post, which is what makes this safe to call from
414 + * paths that already handle excerpts their own way: they keep that handling
415 + * and only a Bricks page is diverted.
416 + *
417 + * @since 2.3.1
418 + *
419 + * @param \WP_Post $post Post being described.
420 + * @return string Visible body text, or '' when `post_content` is fine.
421 + */
422 + public static function superseding_content(\WP_Post $post): string {
423 + if (!self::bricks_supersedes_post_content((int) $post->ID)) {
424 + return '';
425 + }
426 +
427 + $bricks = self::from_bricks((int) $post->ID);
428 +
429 + return self::is_blank($bricks) ? '' : $bricks;
430 + }
431 +
432 + /**
433 + * Body text to derive a description from, when the usual source is wrong.
434 + *
435 + * A hand-written excerpt is the author's own summary and is correct however
436 + * the page is built, so it yields '' here and the caller's normal
437 + * `get_the_excerpt()` path keeps it. Only a Bricks page with no excerpt —
438 + * where core would derive one from discarded `post_content` — gets diverted.
439 + *
440 + * @since 2.3.1
441 + *
442 + * @param \WP_Post $post Post being described.
443 + * @return string Text to summarize, or '' to leave the caller's path alone.
444 + */
445 + public static function superseding_excerpt_source(\WP_Post $post): string {
446 + if ('' !== trim((string) $post->post_excerpt)) {
447 + return '';
448 + }
449 +
450 + return self::superseding_content($post);
451 + }
452 +
453 + /**
454 + * The Bricks element tree that renders for a post.
455 + *
456 + * Public because what Bricks puts on the page is not only a scoring
457 + * question: the schema graph has to know whether a Bricks element already
458 + * publishes the page's FAQ before adding one of its own (#649, #650).
459 + *
460 + * Flat, in Bricks' own storage shape — `expand_bricks_components()`
461 + * appends component definitions to the same list rather than nesting them,
462 + * so one `foreach` reaches every element.
463 + *
464 + * @since 2.3.1
465 + *
466 + * @param int $post_id Post being resolved.
467 + * @return array<int,mixed> Elements, or [] when Bricks renders nothing here.
468 + */
469 + /**
470 + * The builder meta keys, for callers that need to inspect the raw storage
471 + * rather than the text extracted from it.
472 + *
473 + * The SEO Analyzer reads these to answer "is there a ThinkRank FAQ element
474 + * on this post?", which is a question about the stored tree, not about the
475 + * words in it (#686).
476 + *
477 + * @since 2.7.0
478 + * @return string[]
479 + */
480 + public static function builder_meta_keys(): array {
481 + return self::BUILDER_META_KEYS;
482 + }
483 +
484 + public static function bricks_tree(int $post_id): array {
485 + if (array_key_exists($post_id, self::$bricks_trees)) {
486 + return self::$bricks_trees[$post_id];
487 + }
488 +
489 + self::$bricks_trees[$post_id] = self::resolve_bricks_tree($post_id);
490 +
491 + return self::$bricks_trees[$post_id];
492 + }
493 +
494 + /**
495 + * Discard the resolved-tree memo. Test seam.
496 + *
497 + * @since 2.3.1
498 + * @return void
499 + */
500 + public static function flush_bricks_cache(): void {
501 + self::$bricks_trees = [];
502 + }
503 +
504 + /**
505 + * Read and resolve a post's Bricks tree, ignoring the memo.
506 + *
507 + * @since 2.3.1
508 + *
509 + * @param int $post_id Post being resolved.
510 + * @return array<int,mixed>
511 + */
512 + private static function resolve_bricks_tree(int $post_id): array {
513 + if (!self::bricks_owns_post($post_id)) {
514 + return [];
515 + }
516 +
517 + $source = self::bricks_content_source($post_id);
518 + if (!$source) {
519 + return [];
520 + }
521 +
522 + $stored = get_post_meta($source, self::bricks_meta_key(), true);
523 +
524 + if (is_string($stored)) {
525 + $stored = '' === trim($stored) ? null : json_decode($stored, true);
526 + }
527 +
528 + if (!is_array($stored) || empty($stored)) {
529 + return [];
530 + }
531 +
532 + return self::expand_bricks_components($stored);
533 + }
534 +
535 + /**
84 536 * Resolve an arbitrary chunk of editor markup for the given post.
85 537 *
86 538 * The editor sends its live content to the scorer so an author sees their
87 539 * unsaved edits reflected. On a builder page that live string is the raw
@@ -196,10 +648,10 @@
196 648 return '';
197 649 }
198 650
199 651 $attrs = [];
200 - $collect = static function (array $list) use (&$collect, &$attrs): void {
201 - foreach ($list as $block) {
652 + $collect = static function (array $items) use (&$collect, &$attrs): void {
653 + foreach ($items as $block) {
202 654 if (!empty($block['attrs']) && is_array($block['attrs'])) {
203 655 $attrs[] = $block['attrs'];
204 656 }
205 657 if (!empty($block['innerBlocks']) && is_array($block['innerBlocks'])) {
@@ -212,8 +664,350 @@
212 664 return empty($attrs) ? '' : self::text_from_tree($attrs);
213 665 }
214 666
215 667 /**
668 + * Everything Bricks contributes to this post's analyzable content.
669 + *
670 + * Bricks is the only builder here that needs more than a meta key, on
671 + * three counts:
672 + *
673 + * - It leaves its stored tree behind when a post is switched back to the
674 + * block editor, so an editor-mode gate has to run first or ThinkRank
675 + * scores markup the visitor never sees — the same failure
676 + * `_fl_builder_draft` was ordered against in #449.
677 + * - A post's content can live on ANOTHER post. Bricks' Templates feature
678 + * assigns a content template by condition, and a page using one stores
679 + * nothing of its own; reading only the page's meta scores it blank
680 + * while the visitor reads a full page.
681 + * - Its stored text carries dynamic-data tags and internal element names
682 + * that never reach the rendered page.
683 + *
684 + * @since 2.2.1
685 + *
686 + * @param int $post_id Post being resolved.
687 + * @return string Extracted text, or '' when Bricks has nothing for it.
688 + */
689 + private static function from_bricks(int $post_id): string {
690 + $tree = self::bricks_tree($post_id);
691 +
692 + if (empty($tree)) {
693 + return '';
694 + }
695 +
696 + return self::strip_bricks_dynamic_tags(
697 + self::text_from_tree(self::without_bricks_element_labels($tree))
698 + );
699 + }
700 +
701 + /**
702 + * Whether Bricks — not the block editor — renders this post.
703 + *
704 + * Bricks writes `bricks` or `wordpress` into its editor-mode meta as the
705 + * author toggles between the two, and never clears the content it stored
706 + * for the other mode. Only the `wordpress` value is disqualifying: an
707 + * absent value is the normal state for a post Bricks built and never
708 + * toggled. This follows Bricks' own `Helpers::render_with_bricks()`, which
709 + * bails on exactly that one value.
710 + *
711 + * It deliberately does not match it exactly: the comparison here is
712 + * case-insensitive, where Bricks' is strict. Bricks 2.3.12 only ever writes
713 + * the value lowercase, so the two agree on everything Bricks itself
714 + * stores; they part company only on a value some other integration wrote.
715 + * The two shipping today disagree about the casing — SureRank compares
716 + * against `'WordPress'`, AIOSEO against `'bricks'` — and of the two ways to
717 + * be wrong about `'WordPress'`, blocking costs a score on a page that has
718 + * one, while allowing scores stale content the visitor never sees, which is
719 + * the failure this gate exists to prevent.
720 + *
721 + * @since 2.2.1
722 + *
723 + * @param int $post_id Post being resolved.
724 + * @return bool
725 + */
726 + private static function bricks_owns_post(int $post_id): bool {
727 + $mode = get_post_meta($post_id, self::bricks_editor_mode_key(), true);
728 +
729 + // phpcs:ignore WordPress.WP.CapitalPDangit.MisspelledInText -- Bricks' own stored meta value, lower-cased for the comparison.
730 + return !(is_string($mode) && 'wordpress' === strtolower(trim($mode)));
731 + }
732 +
733 + /**
734 + * The post whose Bricks tree actually renders for this post.
735 + *
736 + * Usually the post itself. When it stores nothing of its own, Bricks falls
737 + * back to whichever content template's conditions match, and that template
738 + * is a separate post carrying the words the visitor reads.
739 + *
740 + * Resolution is delegated to Bricks rather than reimplemented: template
741 + * conditions are a whole rules engine (post IDs, types, taxonomies,
742 + * archives), and a second implementation would drift from it. Bricks
743 + * answers through statics, so they are saved and restored around the call —
744 + * `set_active_templates()` returns early once populated, and on a
745 + * front-end request Bricks has already populated it for the page being
746 + * served. Clobbering that would corrupt the render in progress.
747 + *
748 + * Best-effort by design: any failure returns the post's own data, which is
749 + * exactly today's behaviour.
750 + *
751 + * @since 2.2.1
752 + *
753 + * @param int $post_id Post being resolved.
754 + * @return int Post ID holding the Bricks tree, or 0 when there is none.
755 + */
756 + private static function bricks_content_source(int $post_id): int {
757 + $own = get_post_meta($post_id, self::bricks_meta_key(), true);
758 + if ((is_array($own) && !empty($own)) || (is_string($own) && '' !== trim($own))) {
759 + return $post_id;
760 + }
761 +
762 + if (!class_exists('\\Bricks\\Database')
763 + || !method_exists('\\Bricks\\Database', 'set_active_templates')
764 + ) {
765 + return 0;
766 + }
767 +
768 + // `set_active_templates()` writes TWO statics — `$active_templates` and,
769 + // when a header template resolves, `$header_position`. Both are saved,
770 + // and both are restored in `finally` rather than on the happy path: a
771 + // throw part-way through (a third-party hook on
772 + // `bricks/database/content_type`, `bricks/builder/data_post_id` or
773 + // `bricks/active_templates` is enough) must not leave Bricks' render
774 + // state holding this lookup's values. Restoring only after a clean
775 + // return is what the `catch` below would otherwise skip.
776 + $has_header_position = property_exists('\\Bricks\\Database', 'header_position');
777 + $saved_templates = \Bricks\Database::$active_templates;
778 + $saved_header_position = $has_header_position ? \Bricks\Database::$header_position : null;
779 +
780 + try {
781 + \Bricks\Database::$active_templates = [];
782 + \Bricks\Database::set_active_templates($post_id);
783 + $template = (int) (\Bricks\Database::$active_templates['content'] ?? 0);
784 + } catch (\Throwable $e) {
785 + return 0;
786 + } finally {
787 + \Bricks\Database::$active_templates = $saved_templates;
788 + if ($has_header_position) {
789 + \Bricks\Database::$header_position = $saved_header_position;
790 + }
791 + }
792 +
793 + // A template that is the post itself adds nothing over the empty read
794 + // above, and would otherwise recurse conceptually.
795 + return $template === $post_id ? 0 : $template;
796 + }
797 +
798 + /**
799 + * Splice component definitions into the tree.
800 + *
801 + * A Bricks component keeps its markup in the `bricks_components` option,
802 + * not on the page. The page stores only an instance: an element carrying
803 + * `cid` and, usually, empty `settings`. Walking the page alone therefore
804 + * found no words at all, and a page built entirely from components scored
805 + * blank — the same failure as a page built from a content template.
806 + *
807 + * Confirmed on Bricks 2.3.12: `Bricks\Frontend::render_data()` renders the
808 + * component's copy from an instance this walker extracted '' from.
809 + *
810 + * The definition is read straight from the option rather than through
811 + * `Bricks\Helpers::get_component_instance()`. That helper resolves an
812 + * instance's property overrides, which would be better, but it reads
813 + * `Bricks\Database::$global_data['components']` — populated once per
814 + * request, and empty in the admin and CLI contexts where bulk scoring
815 + * runs. Refreshing it would mean writing to Bricks' live render state, the
816 + * same hazard the template resolver is careful to avoid, and gating on it
817 + * would make a page score differently in wp-admin than on the front end.
818 + * Reading the stored definition is consistent everywhere.
819 + *
820 + * The trade-off: an instance that overrides a component property is scored
821 + * with the component's authored copy rather than the override. That is the
822 + * text the component renders by default, and it is much closer than the
823 + * nothing this returned before.
824 + *
825 + * @since 2.2.1
826 + *
827 + * @param array $tree Bricks content area.
828 + * @return array Tree with component elements spliced in after each instance.
829 + */
830 + private static function expand_bricks_components(array $tree): array {
831 + $expanded = [];
832 + $open = [];
833 +
834 + $walk = static function (array $elements, int $depth) use (&$walk, &$expanded, &$open): void {
835 + foreach ($elements as $element) {
836 + $expanded[] = $element;
837 +
838 + if (!is_array($element) || empty($element['cid']) || !is_string($element['cid'])) {
839 + continue;
840 + }
841 +
842 + $cid = $element['cid'];
843 +
844 + // A component nested inside its own definition would recurse
845 + // forever; the depth cap covers deep but legitimate nesting.
846 + if (isset($open[$cid]) || $depth > 4) {
847 + continue;
848 + }
849 +
850 + $children = self::bricks_component_elements($cid);
851 + if (empty($children)) {
852 + continue;
853 + }
854 +
855 + // Re-entrant per branch, not per page: the guard is released
856 + // after the walk so a second instance further along the page
857 + // still expands, rather than being mistaken for recursion.
858 + //
859 + // That does NOT double the word count — `text_from_tree()`
860 + // ends in `array_unique()`, which collapses a repeated
861 + // component's copy the same way it collapses a value repeated
862 + // across responsive breakpoints. Expanding both instances is
863 + // about not silently dropping the second one's structure.
864 + $open[$cid] = true;
865 + $walk($children, $depth + 1);
866 + unset($open[$cid]);
867 + }
868 + };
869 +
870 + $walk($tree, 0);
871 +
872 + return $expanded;
873 + }
874 +
875 + /**
876 + * The stored elements of one Bricks component.
877 + *
878 + * @since 2.2.1
879 + *
880 + * @param string $cid Component id held by an instance element.
881 + * @return array Component elements, or [] when it cannot be resolved.
882 + */
883 + private static function bricks_component_elements(string $cid): array {
884 + $components = get_option(self::bricks_constant('BRICKS_DB_COMPONENTS', self::BRICKS_COMPONENTS_OPTION), []);
885 +
886 + if (!is_array($components)) {
887 + return [];
888 + }
889 +
890 + foreach ($components as $component) {
891 + $component = self::as_children($component);
892 + if (null === $component) {
893 + continue;
894 + }
895 +
896 + if (isset($component['id']) && $component['id'] === $cid && !empty($component['elements'])) {
897 + return is_array($component['elements']) ? $component['elements'] : [];
898 + }
899 + }
900 +
901 + return [];
902 + }
903 +
904 + /**
905 + * Drop each Bricks element's internal name before the tree is walked.
906 + *
907 + * A Bricks element carries an optional top-level `label` — the nickname an
908 + * author types in the Structure panel to find it again ("Hero headline",
909 + * "CTA row"). It is builder chrome and is never rendered, but `label` is in
910 + * CONTENT_KEYS because it is real content for other builders' form fields,
911 + * so it was being counted as page copy.
912 + *
913 + * Only the element's own `label` is removed. A `label` inside `settings`
914 + * is a rendered field label and stays.
915 + *
916 + * @since 2.2.1
917 + *
918 + * @param array $tree Bricks content area.
919 + * @return array Tree with element nicknames removed.
920 + */
921 + private static function without_bricks_element_labels(array $tree): array {
922 + foreach ($tree as $index => $element) {
923 + if (is_array($element) && isset($element['id'], $element['label'])) {
924 + unset($tree[$index]['label']);
925 + }
926 + }
927 +
928 + return $tree;
929 + }
930 +
931 + /**
932 + * Remove Bricks dynamic-data tags from extracted text.
933 + *
934 + * Bricks stores `{post_title}`, `{post_meta:price}`, `{echo:my_fn}` and the
935 + * like verbatim and resolves them when it renders. Extraction reads the
936 + * stored tree, so without this the placeholders were counted as words, and
937 + * a heading whose text is `{post_title}` reported the literal token as its
938 + * heading text.
939 + *
940 + * The pattern is deliberately narrower than Bricks' own
941 + * (`/{([\wÀ-ÖØ-öø-ÿ\-\s\.\/:\(\)...]+)}/u`), which also matches braces
942 + * containing spaces. Bricks only substitutes tags that resolve to a
943 + * registered provider and leaves anything else on the page as literal text,
944 + * so the broad pattern would delete prose the visitor can actually read.
945 + * Matching only tag-shaped tokens keeps every real sentence and still
946 + * removes every placeholder — the same trade-off SureRank makes.
947 + *
948 + * @since 2.2.1
949 + *
950 + * @param string $text Extracted text.
951 + * @return string Text with placeholders removed.
952 + */
953 + private static function strip_bricks_dynamic_tags(string $text): string {
954 + $stripped = preg_replace('/\{[a-z0-9_][a-z0-9_:\-\.]*\}/i', '', $text);
955 +
956 + if (null === $stripped) {
957 + return $text;
958 + }
959 +
960 + // Collapse the runs of spaces a removed tag leaves mid-sentence,
961 + // without touching the newlines that separate collected nodes.
962 + $tidied = preg_replace('/[ \t]{2,}/', ' ', $stripped);
963 +
964 + return null === $tidied ? $stripped : $tidied;
965 + }
966 +
967 + /**
968 + * Bricks' content-area meta key, preferring Bricks' own constant.
969 + *
970 + * @since 2.2.1
971 + *
972 + * @return string
973 + */
974 + private static function bricks_meta_key(): string {
975 + return self::bricks_constant('BRICKS_DB_PAGE_CONTENT', self::BRICKS_CONTENT_META_KEY);
976 + }
977 +
978 + /**
979 + * Bricks' editor-mode meta key, preferring Bricks' own constant.
980 + *
981 + * @since 2.2.1
982 + *
983 + * @return string
984 + */
985 + private static function bricks_editor_mode_key(): string {
986 + return self::bricks_constant('BRICKS_DB_EDITOR_MODE', self::BRICKS_EDITOR_MODE_META_KEY);
987 + }
988 +
989 + /**
990 + * Read one of Bricks' key-name constants, falling back to the literal.
991 + *
992 + * @since 2.2.1
993 + *
994 + * @param string $name Constant name.
995 + * @param string $fallback Key to use when the constant is unavailable.
996 + * @return string
997 + */
998 + private static function bricks_constant(string $name, string $fallback): string {
999 + if (defined($name)) {
1000 + $value = constant($name);
1001 + if (is_string($value) && '' !== trim($value)) {
1002 + return $value;
1003 + }
1004 + }
1005 +
1006 + return $fallback;
1007 + }
1008 +
1009 + /**
216 1010 * Pull text out of whichever builder stored this post.
217 1011 *
218 1012 * @param int $post_id Post ID.
219 1013 * @return string Extracted text, or '' when no builder data was found.
@@ -218,9 +1012,29 @@
218 1012 * @param int $post_id Post ID.
219 1013 * @return string Extracted text, or '' when no builder data was found.
220 1014 */
221 1015 private static function from_builder_meta(int $post_id): string {
1016 + // Bricks first: it is the only builder whose content can live on
1017 + // another post, and the only one gated on an editor mode.
1018 + $bricks = self::from_bricks($post_id);
1019 + if (!self::is_blank($bricks)) {
1020 + return $bricks;
1021 + }
1022 +
222 1023 foreach (self::BUILDER_META_KEYS as $key) {
1024 + if (self::is_oxygen_classic_key($key)) {
1025 + // Resolved as a pair, once, at the first of its keys.
1026 + if ('_ct_builder_json' !== $key) {
1027 + continue;
1028 + }
1029 +
1030 + $oxygen = self::from_oxygen_classic($post_id);
1031 + if (!self::is_blank($oxygen)) {
1032 + return $oxygen;
1033 + }
1034 + continue;
1035 + }
1036 +
223 1037 $stored = get_post_meta($post_id, $key, true);
224 1038
225 1039 if (is_string($stored) && '' !== trim($stored)) {
226 1040 $decoded = json_decode($stored, true);
@@ -233,26 +1047,16 @@
233 1047 }
234 1048 continue;
235 1049 }
236 1050
237 - // Shortcode tree (Oxygen classic).
238 - if (strpos($stored, '[') !== false && function_exists('do_shortcode')) {
239 - try {
240 - $rendered = do_shortcode($stored);
241 - } catch (\Throwable $e) {
242 - $rendered = $stored;
243 - }
244 - if (!self::is_blank($rendered)) {
245 - return $rendered;
246 - }
247 - }
248 -
249 1051 continue;
250 1052 }
251 1053
252 - // Some builders store an already-decoded array.
253 - if (is_array($stored)) {
254 - $text = self::text_from_tree($stored);
1054 + // Some builders store an already-decoded tree — an array for most,
1055 + // an array of objects for Beaver Builder (#449).
1056 + $tree = self::as_children($stored);
1057 + if (null !== $tree) {
1058 + $text = self::text_from_tree($tree);
255 1059 if (!self::is_blank($text)) {
256 1060 return $text;
257 1061 }
258 1062 }
@@ -261,8 +1065,293 @@
261 1065 return '';
262 1066 }
263 1067
264 1068 /**
1069 + * Whether a meta key is one of Oxygen classic's storage keys.
1070 + *
1071 + * @since 2.10.0
1072 + *
1073 + * @param string $key Meta key.
1074 + * @return bool
1075 + */
1076 + private static function is_oxygen_classic_key(string $key): bool {
1077 + return isset(self::OXYGEN_CLASSIC_KEYS[$key]) || in_array($key, self::OXYGEN_CLASSIC_KEYS, true);
1078 + }
1079 +
1080 + /**
1081 + * Text of an Oxygen classic page, from whichever stored form holds more.
1082 + *
1083 + * Oxygen 4.x keeps the same tree twice: as JSON, and as the shortcodes it
1084 + * used before 4.0. The JSON is preferred because it carries copy the
1085 + * shortcode form hides (a composite element's text is base64-encoded
1086 + * inside `ct_options`, which is configuration and stripped). It is not
1087 + * trusted blindly, though. Reading `ct_builder_json` first once meant a
1088 + * key missing from CONTENT_KEYS silently threw the page away while the
1089 + * shortcode copy sat unread next to it, because a non-empty JSON result
1090 + * stopped the search. Comparing the two means the next such gap costs
1091 + * nothing: the richer form wins.
1092 + *
1093 + * A generation is only read as a pair. The prefixed keys are what Oxygen
1094 + * 4.8.3+ reads, so an unprefixed leftover next to them is stale.
1095 + *
1096 + * @since 2.10.0
1097 + *
1098 + * @param int $post_id Post ID.
1099 + * @return string Extracted text, or '' when Oxygen classic stored nothing.
1100 + */
1101 + private static function from_oxygen_classic(int $post_id): string {
1102 + foreach (self::OXYGEN_CLASSIC_KEYS as $json_key => $shortcode_key) {
1103 + $json = get_post_meta($post_id, $json_key, true);
1104 + $shortcodes = get_post_meta($post_id, $shortcode_key, true);
1105 +
1106 + $from_json = '';
1107 + if (is_string($json) && '' !== trim($json)) {
1108 + $decoded = json_decode($json, true);
1109 + if (is_array($decoded)) {
1110 + // `[oxygen data="..."]` is a dynamic-data placeholder
1111 + // Oxygen fills at render time. The shortcode path drops it
1112 + // with every other tag, so it goes here too or the two
1113 + // forms would disagree on the same page.
1114 + $from_json = (string) preg_replace(
1115 + '/\[oxygen\b[^\]]*\]/i',
1116 + ' ',
1117 + self::text_from_tree($decoded)
1118 + );
1119 + }
1120 + }
1121 +
1122 + $from_shortcodes = '';
1123 + if (is_string($shortcodes) && strpos($shortcodes, '[') !== false) {
1124 + $from_shortcodes = self::text_from_shortcodes($shortcodes);
1125 + }
1126 +
1127 + if (self::is_blank($from_json) && self::is_blank($from_shortcodes)) {
1128 + continue;
1129 + }
1130 +
1131 + return self::visible_word_count($from_json) >= self::visible_word_count($from_shortcodes)
1132 + ? $from_json
1133 + : $from_shortcodes;
1134 + }
1135 +
1136 + return '';
1137 + }
1138 +
1139 + /**
1140 + * Rough count of the words a visitor would read in extracted text.
1141 + *
1142 + * Only used to compare two extractions of the same page, so it needs to
1143 + * be consistent rather than locale-exact.
1144 + *
1145 + * @since 2.10.0
1146 + *
1147 + * @param string $text Extracted text or markup.
1148 + * @return int
1149 + */
1150 + private static function visible_word_count(string $text): int {
1151 + $plain = trim((string) preg_replace('/\s+/u', ' ', wp_strip_all_tags($text)));
1152 +
1153 + return '' === $plain ? 0 : count(explode(' ', $plain));
1154 + }
1155 +
1156 + /**
1157 + * Shortcode attributes that carry copy a visitor reads.
1158 + *
1159 + * An allow-list, not a deny-list. Oxygen Classic tags carry far more
1160 + * attributes than they do copy — `id`, `class`, `selector`, `url`,
1161 + * `ct_options` and friends — and a deny-list silently admits every
1162 + * attribute a future builder release invents, which is how markup ends up
1163 + * being counted as prose.
1164 + *
1165 + * @var string[]
1166 + */
1167 + private const SHORTCODE_TEXT_ATTRIBUTES = [
1168 + 'text',
1169 + 'content',
1170 + 'heading',
1171 + 'title',
1172 + 'subtitle',
1173 + 'label',
1174 + 'caption',
1175 + 'description',
1176 + 'alt',
1177 + 'button_text',
1178 + 'link_text',
1179 + ];
1180 +
1181 + /**
1182 + * Extract readable text from a shortcode tree, without rendering it.
1183 + *
1184 + * Oxygen Classic is the only builder whose storage is shortcodes rather
1185 + * than JSON, and the previous implementation handed the string to
1186 + * `do_shortcode()`. That silently depends on Oxygen having registered its
1187 + * `ct_*` handlers in the current request — which it has on a front-end
1188 + * view, and has not during bulk analysis, the post-list column, cron or
1189 + * REST/MCP. With no handlers registered `do_shortcode()` returns its input
1190 + * unchanged, so the raw shortcode source was scored as if it were the
1191 + * page's prose: `[ct_section`, `id="section-1"` and the rest counted toward
1192 + * the word count, while the actual copy sitting in `text="..."` attributes
1193 + * was never counted at all (#776).
1194 + *
1195 + * `strip_shortcodes()` is no help either — it also only knows registered
1196 + * shortcodes, so it leaves the same text untouched.
1197 + *
1198 + * Reading the stored tree directly is what every other builder here already
1199 + * does, and it matches the class's stated design: no render engine, no
1200 + * dependency on load order, safe during a bulk run.
1201 + *
1202 + * Parsing unconditionally, rather than rendering when Oxygen happens to be
1203 + * loaded and parsing otherwise, is deliberate. It makes the extracted text
1204 + * the same in every context, so the score in the editor matches the score
1205 + * from a bulk run or from MCP. The old code produced whichever of the two
1206 + * the request happened to allow, which is why the same post could report
1207 + * two different word counts depending on how it was asked.
1208 + *
1209 + * The trade-off is that rendered output (resolved images, links, anything
1210 + * Oxygen pulls in from a reusable part) is no longer reflected here. For
1211 + * what this text feeds — word count, content scoring, meta-description
1212 + * fallbacks and schema text — that markup was never the point, and counting
1213 + * it only when the builder happened to be booted was the bug.
1214 + *
1215 + * @since 2.10.0
1216 + *
1217 + * @param string $stored Raw shortcode source.
1218 + * @return string Extracted text.
1219 + */
1220 + private static function text_from_shortcodes(string $stored): string {
1221 + // Oxygen stores each element's settings as a JSON blob in `ct_options`.
1222 + // It is configuration, never copy, and it contains braces and brackets
1223 + // that would otherwise confuse the tag scan below, so it goes first.
1224 + //
1225 + // The blob is matched as a balanced JSON object, not as "up to the
1226 + // next quote". Oxygen wraps it in single quotes but does not escape
1227 + // an apostrophe inside it (`"nicename":"Bob's Plumbing"`), so the
1228 + // quote-to-quote match stopped mid-value and the rest of the blob,
1229 + // `s Plumbing"}'` and all, was left in the tag and leaked into the
1230 + // text. Strings inside the object are skipped whole, so neither a quote
1231 + // nor a brace inside a value can end the match early.
1232 + $source = (string) preg_replace(
1233 + '/\sct_options\s*=\s*\'(?<obj>\{(?:[^{}"]++|"(?:[^"\\\\]|\\\\.)*+"|(?&obj))*+\})\'/s',
1234 + '',
1235 + $stored
1236 + );
1237 +
1238 + // Anything not shaped like Oxygen's JSON blob keeps the old,
1239 + // quote-delimited strip.
1240 + $source = (string) preg_replace(
1241 + '/\sct_options\s*=\s*(["\']).*?\1/s',
1242 + '',
1243 + $source
1244 + );
1245 +
1246 + $attributes = implode('|', array_map(
1247 + static fn(string $name): string => preg_quote($name, '/'),
1248 + self::SHORTCODE_TEXT_ATTRIBUTES
1249 + ));
1250 +
1251 + // Replace each shortcode tag with whatever readable copy its attributes
1252 + // carry. Text between tags is left exactly where it is, so the result
1253 + // keeps the page's reading order rather than hoisting all the headings
1254 + // to the front.
1255 + // The attribute blob is matched quote-aware rather than as "anything up
1256 + // to the first `]`". Oxygen copy contains brackets often enough to
1257 + // matter — "Best tools [2026]", "[Updated] our policy" — and a naive
1258 + // scan ends the tag inside the `text` attribute, dropping the copy
1259 + // before the bracket and leaking the stray `"]` after it into the
1260 + // prose. Which is this bug's own failure mode: the wrong text scored.
1261 + //
1262 + // A tag name must start with a letter or underscore. `[2026]` is not a
1263 + // shortcode anyone can register, and scanning it as one dropped the
1264 + // year out of "Best tools [2026]".
1265 + $text = (string) preg_replace_callback(
1266 + '/\[\/?[a-zA-Z_][a-zA-Z0-9_-]*((?:[^\]"\']|"[^"]*"|\'[^\']*\')*)\]/',
1267 + static function (array $matches) use ($attributes): string {
1268 + if ('' === trim($matches[1])) {
1269 + return ' ';
1270 + }
1271 +
1272 + if (!preg_match_all(
1273 + '/\b(' . $attributes . ')\s*=\s*(["\'])(.*?)\2/s',
1274 + $matches[1],
1275 + $found,
1276 + PREG_SET_ORDER
1277 + )) {
1278 + return ' ';
1279 + }
1280 +
1281 + $parts = [];
1282 + foreach ($found as $attribute) {
1283 + $value = trim($attribute[3]);
1284 +
1285 + // An attribute holding markup or a JSON fragment is
1286 + // configuration that happens to share a name with a copy
1287 + // field, not something a visitor reads.
1288 + if ('' === $value || preg_match('/^[\[{<]/', $value)) {
1289 + continue;
1290 + }
1291 +
1292 + $parts[] = $value;
1293 + }
1294 +
1295 + return empty($parts) ? ' ' : ' ' . implode(' ', $parts) . ' ';
1296 + },
1297 + $source
1298 + );
1299 +
1300 + // Oxygen escapes square brackets in an element's copy before writing
1301 + // it between the tags, so that "Best tools [2026]" cannot be mistaken
1302 + // for a shortcode (`oxygen_vsb_filter_shortcode_content_encode()`).
1303 + // Decoded only now, after the tag scan, for the same reason; left
1304 + // encoded, the placeholders were scored as words of their own.
1305 + $text = str_replace(
1306 + ['_OXY_OPENING_BRACKET_', '_OXY_CLOSING_BRACKET_'],
1307 + ['[', ']'],
1308 + $text
1309 + );
1310 +
1311 + // Entities are stored encoded in attributes (&amp;, &#8217;), and would
1312 + // otherwise be counted as words.
1313 + $text = html_entity_decode($text, ENT_QUOTES | ENT_HTML5, 'UTF-8');
1314 +
1315 + return trim((string) preg_replace('/\s+/u', ' ', $text));
1316 + }
1317 +
1318 + /**
1319 + * A node's children, whether it stores them as an array or an object.
1320 + *
1321 + * The walker used to return immediately on `!is_array($node)`, so an
1322 + * object node was dropped along with its entire subtree — silently, as
1323 + * `''`, which the caller reads as "this builder stored nothing" rather
1324 + * than "this walker cannot read this shape".
1325 + *
1326 + * Beaver Builder stores `_fl_builder_data` as an array of stdClass nodes,
1327 + * each with a stdClass `settings` object, so every node would have been
1328 + * dropped and adding its meta key alone would have looked like it worked
1329 + * and changed nothing. Not BB-specific: any builder storing objects hits
1330 + * this, and that shape will come up again (#449).
1331 + *
1332 + * @since 2.1.0
1333 + *
1334 + * @param mixed $node Candidate node.
1335 + * @return array<string|int,mixed>|null Traversable children, or null.
1336 + */
1337 + private static function as_children($node): ?array {
1338 + if (is_array($node)) {
1339 + return $node;
1340 + }
1341 +
1342 + // Deliberately not is_object(): a builder can store a value object
1343 + // (DateTime, a WP_Post) whose properties are not content, and
1344 + // get_object_vars() on those yields noise. stdClass is what the
1345 + // JSON/serialize round-trip produces, which is the shape we want.
1346 + if ($node instanceof \stdClass) {
1347 + return get_object_vars($node);
1348 + }
1349 +
1350 + return null;
1351 + }
1352 +
1353 + /**
265 1354 * Walk a builder node tree and collect the user-visible text.
266 1355 *
267 1356 * Values are joined with block-level markup so downstream heading, link and
268 1357 * image detection keeps working on the result.
@@ -272,11 +1361,51 @@
272 1361 */
273 1362 private static function text_from_tree(array $tree): string {
274 1363 $collected = [];
275 1364
276 - $walk = static function ($node, $key = null) use (&$walk, &$collected): void {
277 - if (is_array($node)) {
278 - foreach ($node as $child_key => $child) {
1365 + // Strings already represented inside reconstructed markup, so the plain
1366 + // sweep below doesn't emit a link label or heading a second time and
1367 + // double it in the word count.
1368 + $consumed = [];
1369 +
1370 + // Pass 1 — rebuild <a>, <img> and <hN> from node *shape*. This has to
1371 + // happen per node rather than per leaf: a link's label and its
1372 + // destination are separate sibling fields, so once the tree is
1373 + // flattened to leaves the pairing is gone.
1374 + $reconstruct = static function ($node) use (&$reconstruct, &$collected, &$consumed): void {
1375 + $node = self::as_children($node);
1376 + if (null === $node) {
1377 + return;
1378 + }
1379 +
1380 + $markup = self::markup_for_node($node, $consumed);
1381 + if ('' !== $markup) {
1382 + $collected[] = $markup;
1383 + }
1384 +
1385 + foreach ($node as $child_key => $child) {
1386 + // A `link` / `image` sub-object is a destination descriptor the
1387 + // parent has already folded into its markup. Descending into it
1388 + // would emit the same URL a second time as a bare link, and
1389 + // would turn an image's own `url` field into a spurious <a>.
1390 + if (is_string($child_key)
1391 + && (in_array(strtolower($child_key), self::URL_KEYS, true)
1392 + || in_array(strtolower($child_key), self::IMAGE_KEYS, true)
1393 + || in_array(strtolower($child_key), self::VIDEO_KEYS, true))
1394 + ) {
1395 + continue;
1396 + }
1397 +
1398 + $reconstruct($child);
1399 + }
1400 + };
1401 + $reconstruct($tree);
1402 +
1403 + // Pass 2 — remaining visible text.
1404 + $walk = static function ($node, $key = null) use (&$walk, &$collected, &$consumed): void {
1405 + $children = self::as_children($node);
1406 + if (null !== $children) {
1407 + foreach ($children as $child_key => $child) {
279 1408 $walk($child, is_string($child_key) ? $child_key : $key);
280 1409 }
281 1410 return;
282 1411 }
@@ -284,8 +1413,13 @@
284 1413 if (!is_string($node) || '' === trim($node)) {
285 1414 return;
286 1415 }
287 1416
1417 + // Already inside a reconstructed tag.
1418 + if (in_array($node, $consumed, true)) {
1419 + return;
1420 + }
1421 +
288 1422 $is_content_key = is_string($key)
289 1423 && in_array(strtolower($key), self::CONTENT_KEYS, true);
290 1424
291 1425 // Markup is content wherever it appears; bare strings only count
@@ -309,13 +1443,323 @@
309 1443 return implode("\n", $collected);
310 1444 }
311 1445
312 1446 /**
313 - * Whether a value carries no readable text.
1447 + * The video source a node is actually playing, if any.
314 1448 *
1449 + * @since 2.3.1
1450 + *
1451 + * @param array $node Builder node.
1452 + * @return string Video source, or '' when the node carries none.
1453 + */
1454 + private static function video_from(array $node): string {
1455 + foreach ($node as $key => $value) {
1456 + if (!is_string($key) || !is_string($value)) {
1457 + continue;
1458 + }
1459 +
1460 + if (!in_array(strtolower($key), self::VIDEO_TYPE_KEYS, true)) {
1461 + continue;
1462 + }
1463 +
1464 + $keys = self::VIDEO_KEYS_BY_TYPE[strtolower(trim($value))] ?? null;
1465 + if (null === $keys) {
1466 + continue;
1467 + }
1468 +
1469 + // A recognised video_type settles it, including when that
1470 + // provider's own field is empty. Falling through to the flat sweep
1471 + // there handed back whichever sibling key happened to come first in
1472 + // node order — the stale youtube_url left behind after switching
1473 + // the widget to a hosted file, which is exactly what keying on the
1474 + // declared type is meant to prevent.
1475 + $declared = self::url_from($node, $keys);
1476 +
1477 + return self::is_video_source($declared) ? $declared : '';
1478 + }
1479 +
1480 + $url = self::url_from($node, self::VIDEO_KEYS);
1481 +
1482 + return self::is_video_source($url) ? $url : '';
1483 + }
1484 +
1485 + /**
1486 + * Whether a value can be a video source.
1487 + *
1488 + * `looks_like_url()` also accepts `#anchor`, `mailto:` and `tel:`, which a
1489 + * link node may legitimately hold but a video cannot: `<iframe src="#top">`
1490 + * is not a video and would reach a video sitemap as one.
1491 + *
1492 + * @since 2.3.1
1493 + *
1494 + * @param string $url Candidate source.
1495 + * @return bool
1496 + */
1497 + private static function is_video_source(string $url): bool {
1498 + return '' !== $url
1499 + && (1 === preg_match('#^(https?:)?//#i', $url) || str_starts_with($url, '/'));
1500 + }
1501 +
1502 + /**
1503 + * Whether a video source points at a file rather than a provider page.
1504 + *
1505 + * @since 2.3.1
1506 + *
1507 + * @param string $url Video source.
1508 + * @return bool
1509 + */
1510 + private static function is_video_file(string $url): bool {
1511 + $path = (string) wp_parse_url($url, PHP_URL_PATH);
1512 + $ext = strtolower((string) pathinfo($path, PATHINFO_EXTENSION));
1513 +
1514 + return in_array($ext, self::VIDEO_FILE_EXTENSIONS, true);
1515 + }
1516 +
1517 + /**
1518 + * Rebuild the HTML a single builder node represents, if any.
1519 + *
1520 + * Looks only at the node's own fields (plus one level of nesting, because
1521 + * builders commonly wrap a destination as `{ url: … }`). Returns an empty
1522 + * string for the vast majority of nodes, which are layout or configuration.
1523 + *
1524 + * Any leaf string folded into the returned markup is appended to $consumed
1525 + * so the plain-text sweep doesn't count it twice.
1526 + *
1527 + * @param array $node Builder node.
1528 + * @param array $consumed Collects strings represented in the returned markup.
1529 + * @return string Reconstructed HTML, or '' when the node carries none.
1530 + */
1531 + private static function markup_for_node(array $node, array &$consumed): string {
1532 + $text = self::first_value($node, self::CONTENT_KEYS);
1533 + $url = self::url_from($node, self::URL_KEYS);
1534 + $image = self::image_from($node);
1535 + $video = self::video_from($node);
1536 + $tag = self::heading_tag_from($node);
1537 +
1538 + $parts = [];
1539 +
1540 + // Video: an embed shape rather than a link, so the video detector can
1541 + // see it while the link counters do not mistake it for an outbound
1542 + // link. A file source becomes <video src>, anything else an <iframe>,
1543 + // matching how the builder itself renders the two cases.
1544 + if ('' !== $video) {
1545 + $parts[] = self::is_video_file($video)
1546 + ? sprintf('<video src="%s"></video>', esc_url_raw($video))
1547 + : sprintf('<iframe src="%s"></iframe>', esc_url_raw($video));
1548 + }
1549 +
1550 + // Image: alt text matters as much as the tag, since alt checks run over
1551 + // whatever this returns.
1552 + if ('' !== $image['url']) {
1553 + $alt = '' !== $image['alt'] ? $image['alt'] : (string) self::first_value($node, self::ALT_KEYS);
1554 + if ('' !== $alt) {
1555 + $consumed[] = $alt;
1556 + }
1557 + $parts[] = sprintf(
1558 + '<img src="%s" alt="%s" />',
1559 + esc_url_raw($image['url']),
1560 + htmlspecialchars($alt, ENT_QUOTES)
1561 + );
1562 + }
1563 +
1564 + if ('' !== $text) {
1565 + $inner = $text;
1566 +
1567 + if ('' !== $url) {
1568 + $consumed[] = $text;
1569 + $inner = sprintf('<a href="%s">%s</a>', esc_url_raw($url), $text);
1570 + }
1571 +
1572 + if ('' !== $tag) {
1573 + $consumed[] = $text;
1574 + $parts[] = sprintf('<%1$s>%2$s</%1$s>', $tag, $inner);
1575 + } elseif ('' !== $url) {
1576 + $parts[] = $inner;
1577 + }
1578 + } elseif ('' !== $url) {
1579 + // A destination with no label still counts as a link for link
1580 + // checks; the URL doubles as its anchor text.
1581 + $parts[] = sprintf('<a href="%1$s">%1$s</a>', esc_url_raw($url));
1582 + }
1583 +
1584 + return implode("\n", $parts);
1585 + }
1586 +
1587 + /**
1588 + * First non-empty scalar value under any of the given keys.
1589 + *
1590 + * @param array $node Builder node.
1591 + * @param string[] $keys Candidate keys.
1592 + * @return string Trimmed value, or '' when none match.
1593 + */
1594 + private static function first_value(array $node, array $keys): string {
1595 + foreach ($node as $key => $value) {
1596 + if (!is_string($key) || !is_string($value)) {
1597 + continue;
1598 + }
1599 + if (in_array(strtolower($key), $keys, true) && '' !== trim($value)) {
1600 + return trim($value);
1601 + }
1602 + }
1603 +
1604 + return '';
1605 + }
1606 +
1607 + /**
1608 + * Link destination held by a node, as a bare string or a `{ url: … }` object.
1609 + *
1610 + * @param array $node Builder node.
1611 + * @param string[] $keys Candidate keys.
1612 + * @return string URL, or '' when the node holds none.
1613 + */
1614 + private static function url_from(array $node, array $keys): string {
1615 + foreach ($node as $key => $value) {
1616 + if (!is_string($key) || !in_array(strtolower($key), $keys, true)) {
1617 + continue;
1618 + }
1619 +
1620 + if (is_string($value) && self::looks_like_url($value)) {
1621 + return trim($value);
1622 + }
1623 +
1624 + // Elementor and Breakdance both nest the destination one level down.
1625 + $nested_values = self::as_children($value);
1626 + if (null !== $nested_values) {
1627 + foreach ($nested_values as $nested_key => $nested) {
1628 + if (is_string($nested_key)
1629 + && in_array(strtolower($nested_key), ['url', 'href', 'permalink'], true)
1630 + && is_string($nested)
1631 + && self::looks_like_url($nested)
1632 + ) {
1633 + return trim($nested);
1634 + }
1635 + }
1636 + }
1637 + }
1638 +
1639 + return '';
1640 + }
1641 +
1642 + /**
1643 + * Image URL and alt text held by a node.
1644 + *
1645 + * @param array $node Builder node.
1646 + * @return array{url:string,alt:string}
1647 + */
1648 + private static function image_from(array $node): array {
1649 + foreach ($node as $key => $value) {
1650 + if (!is_string($key) || !in_array(strtolower($key), self::IMAGE_KEYS, true)) {
1651 + continue;
1652 + }
1653 +
1654 + if (is_string($value) && self::looks_like_url($value)) {
1655 + return ['url' => trim($value), 'alt' => ''];
1656 + }
1657 +
1658 + $nested_values = self::as_children($value);
1659 + if (null !== $nested_values) {
1660 + $url = '';
1661 + $alt = '';
1662 + foreach ($nested_values as $nested_key => $nested) {
1663 + if (!is_string($nested_key) || !is_string($nested)) {
1664 + continue;
1665 + }
1666 + $nested_key = strtolower($nested_key);
1667 + if ('' === $url && in_array($nested_key, ['url', 'src'], true) && self::looks_like_url($nested)) {
1668 + $url = trim($nested);
1669 + }
1670 + if ('' === $alt && in_array($nested_key, self::ALT_KEYS, true)) {
1671 + $alt = trim($nested);
1672 + }
1673 + }
1674 + if ('' !== $url) {
1675 + return ['url' => $url, 'alt' => $alt];
1676 + }
1677 + }
1678 + }
1679 +
1680 + return ['url' => '', 'alt' => ''];
1681 + }
1682 +
1683 + /**
1684 + * Heading tag a node asks for, normalised to h1–h6.
1685 + *
1686 + * Accepts both the `h2` form and a bare level like `2`.
1687 + *
1688 + * @param array $node Builder node.
1689 + * @return string Tag name, or '' when the node is not a heading.
1690 + */
1691 + private static function heading_tag_from(array $node): string {
1692 + foreach ($node as $key => $value) {
1693 + if (!is_string($key) || !in_array(strtolower($key), self::HEADING_TAG_KEYS, true)) {
1694 + continue;
1695 + }
1696 +
1697 + if (is_string($value) && preg_match('/^h([1-6])$/i', trim($value), $m)) {
1698 + return 'h' . $m[1];
1699 + }
1700 +
1701 + // A bare level only counts under a key that unambiguously means one;
1702 + // `size` and `tag` carry values like "large" or "div" far more often.
1703 + if (is_numeric($value)
1704 + && in_array(strtolower($key), ['level'], true)
1705 + && (int) $value >= 1 && (int) $value <= 6
1706 + ) {
1707 + return 'h' . (int) $value;
1708 + }
1709 + }
1710 +
1711 + return '';
1712 + }
1713 +
1714 + /**
1715 + * Whether a string is plausibly a link or asset destination.
1716 + *
1717 + * Deliberately permissive about relative paths — builders store internal
1718 + * links that way — but rejects the option slugs and CSS values that make up
1719 + * most of a builder tree.
1720 + *
1721 + * @param string $value Candidate.
1722 + * @return bool
1723 + */
1724 + private static function looks_like_url(string $value): bool {
1725 + $value = trim($value);
1726 +
1727 + if ('' === $value || strlen($value) > 2048) {
1728 + return false;
1729 + }
1730 +
1731 + if (preg_match('#^(https?:)?//#i', $value) || str_starts_with($value, '/')) {
1732 + return true;
1733 + }
1734 +
1735 + // Protocol-ish destinations a link node can legitimately hold.
1736 + return (bool) preg_match('#^(mailto:|tel:|\#)#i', $value);
1737 + }
1738 +
1739 + /**
1740 + * Whether a value carries nothing worth analyzing.
1741 + *
1742 + * Readable text is the usual signal, but not the only one: a page can be
1743 + * made entirely of media. A builder section holding just a gallery
1744 + * reconstructs to `<img>` tags and one holding just a video widget to a
1745 + * single `<iframe>` — both strip to an empty string, so a text-only test
1746 + * discarded them here and the page fell through to the next builder key,
1747 + * then to the raw markup, and finally reported as having no content at all.
1748 + *
1749 + * Comments are dropped before the tag test: the raw markup this class falls
1750 + * back to on a builder page is unrendered block comments, which must stay
1751 + * blank rather than be mistaken for reconstructed media.
1752 + *
315 1753 * @param string $value Candidate content.
316 1754 * @return bool
317 1755 */
318 1756 private static function is_blank(string $value): bool {
319 - return '' === trim(wp_strip_all_tags($value));
1757 + if ('' !== trim(wp_strip_all_tags($value))) {
1758 + return false;
1759 + }
1760 +
1761 + $without_comments = (string) preg_replace('~<!--.*?-->~s', '', $value);
1762 +
1763 + return 1 !== preg_match('~<(?:a|img|iframe|video|source)\b~i', $without_comments);
320 1764 }
321 1765 }