PluginProbe
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO / 2.14.2
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO v2.14.2
2.14.2 2.14.1 2.14.0 2.13.0 2.12.0 2.11.0 2.10.0 2.9.0 2.8.0 2.7.0 2.6.0 2.5.0 2.4.0 2.3.0 2.2.0 2.1.1 2.1.0 2.0.2 2.0.1 2.0.0 1.32.0 1.31.0 1.30.0 1.29.0 1.28.0 All 57 releases
← All changes | includes/seo/class-builder-content.php +1633 -80 2.0.0 → 2.14.2 View file →
@@ -40,23 +40,157 @@
40 40
41 41 /**
42 42 * Post meta keys that hold builder data, in priority order.
43 43 *
44 - * Several generations of the same builder are listed on purpose: Oxygen 6
45 - * is Breakdance under the hood (`_breakdance_data`), while earlier Oxygen
46 - * releases used `_oxygen_data` or the shortcode-based
47 - * `ct_builder_shortcodes`. A site can only have one of them.
44 + * Several generations of the same builder are listed on purpose. Oxygen 6
45 + * is Breakdance under the hood and writes the same tree, under its own
46 + * prefix: Breakdance keeps it in `_breakdance_data`, Oxygen 6 in
47 + * `_oxygen_data` (the key is `__bdox('_meta_prefix') . 'data'`, and the
48 + * prefix is `_oxygen_` under Oxygen). Both store it inside a
49 + * `tree_json_string` envelope, see unwrap_tree_envelope(). Earlier Oxygen
50 + * releases used the shortcode-based `ct_builder_shortcodes` and its JSON
51 + * sibling. A site can only have one of them.
48 52 *
49 53 * @var string[]
50 54 */
51 55 private const BUILDER_META_KEYS = [
52 - '_breakdance_data', // Oxygen 6+ / Breakdance
53 - '_oxygen_data', // Oxygen (earlier releases)
54 - 'ct_builder_shortcodes', // Oxygen classic
56 + '_breakdance_data', // Breakdance
57 + '_oxygen_data', // Oxygen 6+ (Breakdance engine, Oxygen prefix)
58 + // Oxygen classic. 4.x writes the tree as JSON to `ct_builder_json`
59 + // while still keeping `ct_builder_shortcodes`. A post carrying only
60 + // the JSON key used to match no key at all and fall through to an
61 + // empty `post_content`, which reads as a one-word page (#776).
62 + //
63 + // Oxygen 4.8.3 then renamed every `ct_*` post meta key to `_ct_*`
64 + // (`oxygen_vsb_update_4_8_3()` runs `oxy_prefix_meta_keys()` on
65 + // upgrade, and `oxy_get_post_meta()` only reads the prefixed name
66 + // from then on). A current Oxygen classic site therefore has only the
67 + // underscored keys, which nothing here listed, so every one of its
68 + // pages resolved as empty. The prefixed keys come first because they
69 + // are what Oxygen itself reads; the bare ones cover a site that has
70 + // not run the migration (or reverted it with `?unprefix_meta`).
71 + //
72 + // These four are not read in this loop: from_oxygen_classic() pairs
73 + // each JSON key with its shortcode sibling so the two forms can be
74 + // compared. They are listed here because this list is also what the
75 + // word-count index watches and what FAQ detection scans.
76 + '_ct_builder_json', // Oxygen classic 4.8.3+ (JSON tree)
77 + 'ct_builder_json', // Oxygen classic 4.0-4.8.2 (JSON tree)
78 + '_ct_builder_shortcodes', // Oxygen classic 4.8.3+ (shortcode tree)
79 + 'ct_builder_shortcodes', // Oxygen classic < 4.8.3 (shortcode tree)
55 80 '_elementor_data', // Elementor
81 + // Beaver Builder. Published layout first: `_fl_builder_draft` holds
82 + // unsaved changes and would score content the visitor cannot see.
83 + // Both are arrays of stdClass nodes, which is why the walker below
84 + // has to treat objects like arrays (#449).
85 + '_fl_builder_data', // Beaver Builder (published)
86 + '_fl_builder_draft', // Beaver Builder (unsaved changes)
56 87 ];
57 88
58 89 /**
90 + * Bricks' content-area meta key, used when Bricks itself isn't loaded.
91 + *
92 + * Bricks exposes `BRICKS_DB_PAGE_CONTENT` and renames the underlying key
93 + * between generations (it gained the `_2` suffix in 1.7.3), so the
94 + * constant is authoritative and this literal is only the fallback for the
95 + * contexts where it is undefined — Bricks is a theme, so on an admin or
96 + * CLI request against a site that has since switched themes the constant
97 + * is simply not there while the post meta still is.
98 + *
99 + * Bricks stores three areas — header, content and footer. Only the content
100 + * area belongs to the post being scored; the header and footer areas live
101 + * on Bricks' own template posts and would double-count site chrome into
102 + * every page's word count, so they are deliberately not read here.
103 + *
104 + * @since 2.2.1
105 + * @var string
106 + */
107 + private const BRICKS_CONTENT_META_KEY = '_bricks_page_content_2';
108 +
109 + /**
110 + * Bricks' per-post editor-mode meta key, used when Bricks isn't loaded.
111 + *
112 + * @since 2.2.1
113 + * @var string
114 + */
115 + private const BRICKS_EDITOR_MODE_META_KEY = '_bricks_editor_mode';
116 +
117 + /**
118 + * Bricks' components option, used when Bricks itself isn't loaded.
119 + *
120 + * @since 2.2.1
121 + * @var string
122 + */
123 + private const BRICKS_COMPONENTS_OPTION = 'bricks_components';
124 +
125 + /**
126 + * Bricks' element that renders the post's own `post_content`.
127 + *
128 + * A Bricks page normally discards `post_content` entirely, which is why
129 + * anything left there is invisible. Dropping this element onto the canvas
130 + * is the one way an author puts it back on the page, so its presence flips
131 + * `post_content` from stale leftovers to content the visitor reads.
132 + *
133 + * @since 2.3.1
134 + * @var string
135 + */
136 + private const BRICKS_POST_CONTENT_ELEMENT = 'post-content';
137 +
138 + /**
139 + * Bricks' Heading element, and the tag it renders when none is stored.
140 + *
141 + * Bricks leaves a setting out of storage while it equals its default, so a
142 + * Heading left on its default tag is stored with no `tag` at all. Bricks
143 + * 2.4.1 renders it as `h3` (`Element_Heading::$tag`, overridable by the
144 + * active theme style's `tag`), and the walker, which only wraps text whose
145 + * node names a tag, read it as body copy (#908).
146 + *
147 + * @since 2.15.0
148 + * @var string
149 + */
150 + private const BRICKS_HEADING_ELEMENT = 'heading';
151 +
152 + /**
153 + * Tag a Bricks Heading renders when neither it nor a theme style sets one.
154 + *
155 + * @since 2.15.0
156 + * @var string
157 + */
158 + private const BRICKS_HEADING_DEFAULT_TAG = 'h3';
159 +
160 + /**
161 + * Tag an Elementor Heading widget renders when `header_size` is not stored.
162 + *
163 + * Elementor saves `settings.toJSON({ remove: ['default'] })`, so a heading
164 + * left on its default size has no `header_size` in `_elementor_data`, and
165 + * that default is `h2` (#908).
166 + *
167 + * @since 2.15.0
168 + * @var string
169 + */
170 + private const ELEMENTOR_HEADING_DEFAULT_TAG = 'h2';
171 +
172 + /**
173 + * Resolved Bricks trees for this request, keyed by post ID.
174 + *
175 + * Rendering one page asks for the tree about twenty times — every
176 + * description, every schema node, the FAQ guard — and resolving it is not
177 + * free. `bricks_content_source()` clears `Bricks\Database::$active_templates`
178 + * before asking Bricks which content template applies, which defeats
179 + * Bricks' own early-return and re-runs its whole template-condition engine;
180 + * `expand_bricks_components()` then walks the tree again. Measured on a
181 + * Bricks page with no content of its own, that was ten full runs of the
182 + * rules engine per request.
183 + *
184 + * Per-request only, and only ever read back within one page render — a
185 + * request that writes Bricks content does not also render it.
186 + *
187 + * @since 2.3.1
188 + * @var array<int,array<int,mixed>>
189 + */
190 + private static array $bricks_trees = [];
191 +
192 + /**
59 193 * JSON keys whose values are user-visible text.
60 194 *
61 195 * Builder trees mix content with configuration, so a blind string sweep
62 196 * would count CSS classes and option slugs as words. Matching on the key
@@ -67,11 +201,38 @@
67 201 private const CONTENT_KEYS = [
68 202 'text', 'title', 'subtitle', 'heading', 'subheading', 'content',
69 203 'description', 'caption', 'excerpt', 'label', 'value', 'html',
70 204 'editor', 'quote', 'answer', 'question', 'body', 'button_text',
205 + // Oxygen classic keeps an element's copy in `options.ct_content`
206 + // (headline, text block, rich text, link and button labels). It is the
207 + // field Oxygen's own serializer moves between the tags when it writes
208 + // shortcodes (`parse_components_tree()`), and the one Relevanssi and
209 + // Oxygen's WPML integration read. Missing from this list, the walker
210 + // kept only copy that happened to contain markup: a page of plain
211 + // headings and paragraphs lost almost all of its words.
212 + 'ct_content',
213 + // Oxygen's composite elements keep their copy under `options.original`
214 + // instead, one key per field. Taken from the list Oxygen itself treats
215 + // as text when it serializes (`$options_to_encode`); the numeric price
216 + // fields and the progress bar's right-hand percentage are left out.
217 + 'testimonial_text', 'testimonial_author', 'testimonial_author_info',
218 + 'icon_box_heading', 'icon_box_text',
219 + 'pricing_box_package_title', 'pricing_box_package_subtitle', 'pricing_box_content',
220 + 'progress_bar_left_text',
71 221 ];
72 222
73 223 /**
224 + * Oxygen classic's storage generations, as JSON key => shortcode key.
225 + *
226 + * @since 2.10.0
227 + * @var array<string,string>
228 + */
229 + private const OXYGEN_CLASSIC_KEYS = [
230 + '_ct_builder_json' => '_ct_builder_shortcodes',
231 + 'ct_builder_json' => 'ct_builder_shortcodes',
232 + ];
233 +
234 + /**
74 235 * JSON keys whose values hold a link destination.
75 236 *
76 237 * Builders store a link's destination in a structured field separate from
77 238 * its label, either as a bare URL string or as a `{ url: … }` object.
@@ -84,8 +245,71 @@
84 245 'link', 'url', 'href', 'link_url', 'button_link', 'permalink', 'link_to',
85 246 ];
86 247
87 248 /**
249 + * JSON keys whose values hold an embedded video's source.
250 + *
251 + * A builder's video widget keeps its destination in a provider-specific
252 + * field — Elementor picks `youtube_url`, `vimeo_url`, `dailymotion_url` or
253 + * `hosted_url` according to the chosen source type — none of which is a
254 + * link field or a content field, so a video on a builder page reached the
255 + * analyzers as nothing at all.
256 + *
257 + * These are deliberately kept out of URL_KEYS. A video is an embed, not an
258 + * outbound link: rendering one as `<a href>` would add a spurious external
259 + * link to every page carrying a video and skew the link counts. They are
260 + * reconstructed as `<iframe>`/`<video>` instead, which the video detector
261 + * recognises and the link and image counters ignore.
262 + *
263 + * @since 2.3.1
264 + * @var string[]
265 + */
266 + private const VIDEO_KEYS = [
267 + 'youtube_url', 'vimeo_url', 'dailymotion_url', 'videopress_url',
268 + 'hosted_url', 'video_url', 'video_src', 'video_link',
269 + ];
270 +
271 + /**
272 + * File extensions that mean a video source is a file, not a provider page.
273 + *
274 + * @since 2.3.1
275 + * @var string[]
276 + */
277 + private const VIDEO_FILE_EXTENSIONS = ['mp4', 'webm', 'ogv', 'mov', 'm4v'];
278 +
279 + /**
280 + * Source keys to trust for a declared video source type.
281 + *
282 + * A widget keeps one field per provider and does not clear the others when
283 + * the author switches source: an Elementor video moved from YouTube to Self
284 + * Hosted still carries the earlier `youtube_url`. Reading whichever key
285 + * turns up first then emits the video the author replaced. The widget says
286 + * which one it is actually playing, so that is read first and the flat key
287 + * sweep is only the fallback for a builder that declares nothing.
288 + *
289 + * @since 2.3.1
290 + * @var array<string,string[]>
291 + */
292 + private const VIDEO_KEYS_BY_TYPE = [
293 + 'youtube' => ['youtube_url'],
294 + 'vimeo' => ['vimeo_url'],
295 + 'dailymotion' => ['dailymotion_url'],
296 + 'videopress' => ['videopress_url'],
297 + 'hosted' => ['hosted_url', 'video_url', 'video_src', 'video_link'],
298 + 'media' => ['hosted_url', 'video_url', 'video_src', 'video_link'],
299 + 'file' => ['hosted_url', 'video_url', 'video_src', 'video_link'],
300 + 'self_hosted' => ['hosted_url', 'video_url', 'video_src', 'video_link'],
301 + ];
302 +
303 + /**
304 + * Keys a builder uses to name which video source a widget is playing.
305 + *
306 + * @since 2.3.1
307 + * @var string[]
308 + */
309 + private const VIDEO_TYPE_KEYS = ['video_type', 'videotype', 'video_source', 'source_type'];
310 +
311 + /**
88 312 * JSON keys whose values hold an image, as a URL string or `{ url, alt }`.
89 313 *
90 314 * @var string[]
91 315 */
@@ -102,9 +326,13 @@
102 326 *
103 327 * @var string[]
104 328 */
105 329 private const HEADING_TAG_KEYS = [
106 - 'header_size', 'heading_tag', 'html_tag', 'title_tag', 'tag', 'level', 'size',
330 + // Lower-cased on both sides of the comparison, so `headingtag` is the
331 + // camelCase `headingTag` Bricks uses throughout its own controls and
332 + // which ThinkRank's Bricks elements declare. Without it their section
333 + // headings counted as body copy and never reached heading structure.
334 + 'header_size', 'heading_tag', 'headingtag', 'html_tag', 'title_tag', 'tag', 'level', 'size',
107 335 ];
108 336
109 337 /**
110 338 * Keys whose value is alternative text for a sibling image.
@@ -113,8 +341,22 @@
113 341 */
114 342 private const ALT_KEYS = ['alt', 'alt_text', 'image_alt', 'title'];
115 343
116 344 /**
345 + * The global post and every global `setup_postdata()` writes.
346 + *
347 + * Rendering points them at the post being analyzed, then puts each one
348 + * back exactly as it was, unset included (#860).
349 + *
350 + * @since 2.12.0
351 + * @var string[]
352 + */
353 + private const POSTDATA_GLOBALS = [
354 + 'post', 'id', 'authordata', 'currentday', 'currentmonth',
355 + 'page', 'pages', 'multipage', 'more', 'numpages',
356 + ];
357 +
358 + /**
117 359 * Resolve the content worth analyzing for a post.
118 360 *
119 361 * @param \WP_Post $post Post being analyzed.
120 362 * @return string HTML/text to analyze.
@@ -119,12 +361,301 @@
119 361 * @param \WP_Post $post Post being analyzed.
120 362 * @return string HTML/text to analyze.
121 363 */
122 364 public static function resolve(\WP_Post $post): string {
123 - return self::resolve_markup((string) $post->post_content, $post);
365 + $raw = (string) $post->post_content;
366 +
367 + // A page built in Gutenberg and then switched to Bricks keeps its old
368 + // blocks in `post_content` forever — Bricks never clears them, and
369 + // never renders them either. Resolving that first meant the stale draft
370 + // beat the tree the visitor actually reads, and it did not stop at the
371 + // score: the same string becomes the meta description, og:description,
372 + // twitter:description and the schema description. Starting from nothing
373 + // sends the resolution straight to Bricks' storage, which is where this
374 + // page's words are (#651).
375 + //
376 + // Only for the stored path. `resolve_markup()` is also called with live
377 + // editor content, and the Bricks panel's resolver reads the canvas —
378 + // discarding that would replace what the author is typing with the last
379 + // save.
380 + if (self::bricks_supersedes_post_content((int) $post->ID)) {
381 + $raw = '';
382 + }
383 +
384 + return self::resolve_markup($raw, $post);
124 385 }
125 386
126 387 /**
388 + * Whether Bricks renders this post and throws its `post_content` away.
389 + *
390 + * True means anything still stored in `post_content` is invisible: it is
391 + * not on the page, so it must not be scored, described or published as
392 + * structured data. False covers both a post Bricks does not own and a
393 + * Bricks page that puts `post_content` back with a Post Content element.
394 + *
395 + * @since 2.3.1
396 + *
397 + * @param int $post_id Post being resolved.
398 + * @return bool
399 + */
400 + public static function bricks_supersedes_post_content(int $post_id): bool {
401 + $tree = self::bricks_tree($post_id);
402 +
403 + if (empty($tree)) {
404 + return false;
405 + }
406 +
407 + foreach ($tree as $element) {
408 + if (is_array($element)
409 + && self::BRICKS_POST_CONTENT_ELEMENT === ($element['name'] ?? null)
410 + ) {
411 + return false;
412 + }
413 + }
414 +
415 + return !self::bricks_tree_prints_post_content($tree);
416 + }
417 +
418 + /**
419 + * Whether a Bricks tree prints the body through a dynamic-data tag.
420 + *
421 + * The Post Content element is not the only way back onto the page: Bricks'
422 + * `{post_content}` tag renders the same thing from inside an ordinary text
423 + * element, and a single-post template written that way is a common shape.
424 + * Missing it would mean the post's real body is discarded everywhere —
425 + * scoring, the meta/og/twitter descriptions, the schema description — for a
426 + * page that is displaying it.
427 + *
428 + * Matched over the encoded tree rather than per setting, because the tag can
429 + * sit in any string field of any element and Bricks allows modifiers after
430 + * the name (`{post_content:...}`).
431 + *
432 + * @since 2.3.1
433 + *
434 + * @param array $tree Bricks element tree.
435 + * @return bool
436 + */
437 + private static function bricks_tree_prints_post_content(array $tree): bool {
438 + $encoded = wp_json_encode($tree);
439 +
440 + return is_string($encoded) && false !== stripos($encoded, '{post_content');
441 + }
442 +
443 + /**
444 + * The post's content as the visitor actually receives it.
445 + *
446 + * `post_content` for everything except a Bricks page that discards it, and
447 + * there the Bricks tree's text. Descriptions are derived from a post's body
448 + * in half a dozen places; every one of them wants this rather than the raw
449 + * column (#651).
450 + *
451 + * @since 2.3.1
452 + *
453 + * @param \WP_Post $post Post being described.
454 + * @return string
455 + */
456 + public static function visible_content(\WP_Post $post): string {
457 + $superseding = self::superseding_content($post);
458 +
459 + return '' !== $superseding ? $superseding : (string) $post->post_content;
460 + }
461 +
462 + /**
463 + * Replacement body text for a post whose `post_content` does not render.
464 + *
465 + * Empty for every ordinary post, which is what makes this safe to call from
466 + * paths that already handle excerpts their own way: they keep that handling
467 + * and only a Bricks page is diverted.
468 + *
469 + * @since 2.3.1
470 + *
471 + * @param \WP_Post $post Post being described.
472 + * @return string Visible body text, or '' when `post_content` is fine.
473 + */
474 + public static function superseding_content(\WP_Post $post): string {
475 + if (!self::bricks_supersedes_post_content((int) $post->ID)) {
476 + return '';
477 + }
478 +
479 + $bricks = self::from_bricks((int) $post->ID);
480 +
481 + return self::is_blank($bricks) ? '' : $bricks;
482 + }
483 +
484 + /**
485 + * Body text to derive a description from, when the usual source is wrong.
486 + *
487 + * A hand-written excerpt is the author's own summary and is correct however
488 + * the page is built, so it yields '' here and the caller's normal
489 + * `get_the_excerpt()` path keeps it. Only a Bricks page with no excerpt —
490 + * where core would derive one from discarded `post_content` — gets diverted.
491 + *
492 + * @since 2.3.1
493 + *
494 + * @param \WP_Post $post Post being described.
495 + * @return string Text to summarize, or '' to leave the caller's path alone.
496 + */
497 + public static function superseding_excerpt_source(\WP_Post $post): string {
498 + if ('' !== trim((string) $post->post_excerpt)) {
499 + return '';
500 + }
501 +
502 + return self::superseding_content($post);
503 + }
504 +
505 + /**
506 + * The Bricks element tree that renders for a post.
507 + *
508 + * Public because what Bricks puts on the page is not only a scoring
509 + * question: the schema graph has to know whether a Bricks element already
510 + * publishes the page's FAQ before adding one of its own (#649, #650).
511 + *
512 + * Flat, in Bricks' own storage shape — `expand_bricks_components()`
513 + * appends component definitions to the same list rather than nesting them,
514 + * so one `foreach` reaches every element.
515 + *
516 + * @since 2.3.1
517 + *
518 + * @param int $post_id Post being resolved.
519 + * @return array<int,mixed> Elements, or [] when Bricks renders nothing here.
520 + */
521 + /**
522 + * The builder meta keys, for callers that need to inspect the raw storage
523 + * rather than the text extracted from it.
524 + *
525 + * The SEO Analyzer reads these to answer "is there a ThinkRank FAQ element
526 + * on this post?", which is a question about the stored tree, not about the
527 + * words in it (#686).
528 + *
529 + * @since 2.7.0
530 + * @return string[]
531 + */
532 + public static function builder_meta_keys(): array {
533 + return self::BUILDER_META_KEYS;
534 + }
535 +
536 + public static function bricks_tree(int $post_id): array {
537 + if (array_key_exists($post_id, self::$bricks_trees)) {
538 + return self::$bricks_trees[$post_id];
539 + }
540 +
541 + self::$bricks_trees[$post_id] = self::resolve_bricks_tree($post_id);
542 +
543 + return self::$bricks_trees[$post_id];
544 + }
545 +
546 + /**
547 + * Discard the resolved-tree memo. Test seam.
548 + *
549 + * @since 2.3.1
550 + * @return void
551 + */
552 + public static function flush_bricks_cache(): void {
553 + self::$bricks_trees = [];
554 + }
555 +
556 + /**
557 + * Read and resolve a post's Bricks tree, ignoring the memo.
558 + *
559 + * @since 2.3.1
560 + *
561 + * @param int $post_id Post being resolved.
562 + * @return array<int,mixed>
563 + */
564 + private static function resolve_bricks_tree(int $post_id): array {
565 + if (!self::bricks_owns_post($post_id)) {
566 + return [];
567 + }
568 +
569 + $source = self::bricks_content_source($post_id);
570 + if (!$source) {
571 + return [];
572 + }
573 +
574 + $stored = get_post_meta($source, self::bricks_meta_key(), true);
575 +
576 + if (is_string($stored)) {
577 + $stored = '' === trim($stored) ? null : json_decode($stored, true);
578 + }
579 +
580 + if (!is_array($stored) || empty($stored)) {
581 + return [];
582 + }
583 +
584 + return self::expand_bricks_components(self::bricks_render_order($stored));
585 + }
586 +
587 + /**
588 + * A flat Bricks element list, in the order Bricks renders it.
589 + *
590 + * Bricks stores one flat list and links it with `parent` and `children`
591 + * ids. `Frontend::render_data()` renders the root elements in list order
592 + * and each element's children in the order of its `children` array, so a
593 + * child's position in the list says nothing about where it appears on the
594 + * page. Walking the list as stored put a section's contents wherever they
595 + * happened to be saved (#907).
596 + *
597 + * Anything the walk does not reach (an orphan, a cycle) keeps its stored
598 + * position after the rest, so no copy is dropped.
599 + *
600 + * @since 2.15.0
601 + *
602 + * @param array $elements Flat Bricks element list.
603 + * @return array The same elements, in render order.
604 + */
605 + private static function bricks_render_order(array $elements): array {
606 + $by_id = [];
607 + foreach ($elements as $index => $element) {
608 + $id = is_array($element) ? ($element['id'] ?? null) : null;
609 + if (is_scalar($id) && '' !== (string) $id && !isset($by_id[(string) $id])) {
610 + $by_id[(string) $id] = $index;
611 + }
612 + }
613 +
614 + if (empty($by_id)) {
615 + return $elements;
616 + }
617 +
618 + $ordered = [];
619 + $placed = [];
620 +
621 + $place = static function ($index) use (&$place, &$ordered, &$placed, $elements, $by_id): void {
622 + if (isset($placed[$index])) {
623 + return;
624 + }
625 +
626 + $placed[$index] = true;
627 + $ordered[] = $elements[$index];
628 +
629 + $children = is_array($elements[$index]) ? ($elements[$index]['children'] ?? []) : [];
630 + if (!is_array($children)) {
631 + return;
632 + }
633 +
634 + foreach ($children as $child_id) {
635 + if (is_scalar($child_id) && isset($by_id[(string) $child_id])) {
636 + $place($by_id[(string) $child_id]);
637 + }
638 + }
639 + };
640 +
641 + foreach ($elements as $index => $element) {
642 + $parent = is_array($element) ? ($element['parent'] ?? null) : null;
643 + if (empty($parent) || !is_scalar($parent) || !isset($by_id[(string) $parent])) {
644 + $place($index);
645 + }
646 + }
647 +
648 + foreach ($elements as $index => $element) {
649 + if (!isset($placed[$index])) {
650 + $ordered[] = $element;
651 + }
652 + }
653 +
654 + return $ordered;
655 + }
656 +
657 + /**
127 658 * Resolve an arbitrary chunk of editor markup for the given post.
128 659 *
129 660 * The editor sends its live content to the scorer so an author sees their
130 661 * unsaved edits reflected. On a builder page that live string is the raw
@@ -145,9 +676,9 @@
145 676 * @param \WP_Post $post Post the markup belongs to.
146 677 * @return string Content to analyze.
147 678 */
148 679 public static function resolve_markup(string $raw, \WP_Post $post): string {
149 - $content = self::render_post_content($raw);
680 + $content = self::render_post_content($raw, $post);
150 681
151 682 // Block markup that renders to nothing usually means the builder that
152 683 // owns those blocks did not register them in this context — Divi 5
153 684 // loads its module library lazily per-request, so in CLI, REST, admin
@@ -195,19 +726,74 @@
195 726 *
196 727 * Best-effort: a third-party block that fatals must not take the whole
197 728 * score down with it.
198 729 *
199 - * @param string $raw Raw post content.
730 + * Runs as the post's own context, as it would on the front end. Admin and
731 + * REST requests have no current post, so a shortcode reading
732 + * `get_the_ID()` got nothing, and one looping a related-posts query left
733 + * the global post on the last of them: its `wp_reset_postdata()` goes back
734 + * to the main query's post, and there is none. On the Classic Editor this
735 + * runs after the form prints its hidden `post_ID` and before the title and
736 + * editor, which then showed the related post, and Update saved it over the
737 + * original (#860).
738 + *
739 + * Secondary queries also run with front-end statuses. In wp-admin core
740 + * marks every `WP_Query` as an admin query and, when no `post_status` is
741 + * set, adds the statuses the admin post list shows, draft among them, so
742 + * a related-posts shortcode listed drafts the front end never shows and
743 + * the editor-load analysis disagreed with REST and the page (#902).
744 + *
745 + * The main query points at the post too. Restoring the globals afterwards
746 + * (#860) did not reach between shortcodes: a related-posts loop's own
747 + * `wp_reset_postdata()` still found no post on the main query, so every
748 + * later shortcode in the same render saw the last looped post (#903).
749 + *
750 + * @param string $raw Raw post content.
751 + * @param \WP_Post $post Post the content belongs to.
200 752 * @return string Rendered content.
201 753 */
202 - private static function render_post_content(string $raw): string {
754 + private static function render_post_content(string $raw, \WP_Post $post): string {
203 755 if ('' === trim($raw)) {
204 756 return '';
205 757 }
206 758
207 - $content = $raw;
759 + $content = $raw;
760 + $previous = self::snapshot_post_globals();
208 761
762 + // After pre_get_posts core reads `is_admin` only to add the admin
763 + // list's statuses when none were asked for, so queries that set
764 + // `post_status`, and the main query, are untouched.
765 + $front_end_statuses = static function ($query): void {
766 + if ($query instanceof \WP_Query && !$query->is_main_query()) {
767 + $query->is_admin = false;
768 + }
769 + };
770 +
771 + // `wp_reset_postdata()` returns to the main query's post, which admin
772 + // and REST requests do not have. Restored in finally, null included.
773 + $main_query = (isset($GLOBALS['wp_query']) && $GLOBALS['wp_query'] instanceof \WP_Query) ? $GLOBALS['wp_query'] : null;
774 + $main_query_post = $main_query ? $main_query->post : null;
775 +
209 776 try {
777 + // phpcs:ignore WordPress.WP.GlobalVariablesOverride.Prohibited -- Made current for the render, restored in finally.
778 + $GLOBALS['post'] = $post;
779 + add_action('pre_get_posts', $front_end_statuses, PHP_INT_MIN);
780 + if ($main_query) {
781 + $main_query->post = $post;
782 + }
783 +
784 + // Fires `the_post`, which these paths never fired before: admin,
785 + // REST and cron analysis had no current post at all. That is the
786 + // same signal the front-end loop sends and it is what makes
787 + // `get_the_ID()` work inside a shortcode, but it is a new call on a
788 + // path that runs in bulk — the word-count index resolves every post
789 + // it visits — so a theme that counts views on `the_post` will count
790 + // them during indexing. Accepted deliberately: without it a
791 + // shortcode cannot resolve its own post, which is the bug (#860).
792 + if (function_exists('setup_postdata')) {
793 + setup_postdata($post);
794 + }
795 +
210 796 if (function_exists('has_blocks') && function_exists('do_blocks') && has_blocks($raw)) {
211 797 $content = do_blocks($raw);
212 798 }
213 799
@@ -216,8 +802,14 @@
216 802 $content = do_shortcode($content);
217 803 }
218 804 } catch (\Throwable $e) {
219 805 return $raw;
806 + } finally {
807 + if ($main_query) {
808 + $main_query->post = $main_query_post;
809 + }
810 + remove_action('pre_get_posts', $front_end_statuses, PHP_INT_MIN);
811 + self::restore_post_globals($previous);
220 812 }
221 813
222 814 return self::is_blank($content) ? $raw : $content;
223 815 }
@@ -222,8 +814,49 @@
222 814 return self::is_blank($content) ? $raw : $content;
223 815 }
224 816
225 817 /**
818 + * The post globals as they are now; a global that is unset has no key.
819 + *
820 + * @since 2.12.0
821 + *
822 + * @return array<string,mixed>
823 + */
824 + private static function snapshot_post_globals(): array {
825 + $snapshot = [];
826 +
827 + foreach (self::POSTDATA_GLOBALS as $name) {
828 + if (array_key_exists($name, $GLOBALS)) {
829 + $snapshot[$name] = $GLOBALS[$name];
830 + }
831 + }
832 +
833 + return $snapshot;
834 + }
835 +
836 + /**
837 + * Put the post globals back as snapshot_post_globals() found them.
838 + *
839 + * Assigned directly rather than through `setup_postdata()`: there may have
840 + * been no post to set up, and re-running it would fire `the_post` again.
841 + *
842 + * @since 2.12.0
843 + *
844 + * @param array<string,mixed> $snapshot From snapshot_post_globals().
845 + * @return void
846 + */
847 + private static function restore_post_globals(array $snapshot): void {
848 + foreach (self::POSTDATA_GLOBALS as $name) {
849 + if (array_key_exists($name, $snapshot)) {
850 + // phpcs:ignore WordPress.NamingConventions.PrefixAllGlobals.NonPrefixedVariableFound -- Core's own globals, put back as they were.
851 + $GLOBALS[$name] = $snapshot[$name];
852 + } else {
853 + unset($GLOBALS[$name]);
854 + }
855 + }
856 + }
857 +
858 + /**
226 859 * Extract text from the attributes of parsed blocks.
227 860 *
228 861 * @param string $raw Raw post content containing block markup.
229 862 * @return string Collected text, or '' when nothing was found.
@@ -255,8 +888,458 @@
255 888 return empty($attrs) ? '' : self::text_from_tree($attrs);
256 889 }
257 890
258 891 /**
892 + * Everything Bricks contributes to this post's analyzable content.
893 + *
894 + * Bricks is the only builder here that needs more than a meta key, on
895 + * three counts:
896 + *
897 + * - It leaves its stored tree behind when a post is switched back to the
898 + * block editor, so an editor-mode gate has to run first or ThinkRank
899 + * scores markup the visitor never sees — the same failure
900 + * `_fl_builder_draft` was ordered against in #449.
901 + * - A post's content can live on ANOTHER post. Bricks' Templates feature
902 + * assigns a content template by condition, and a page using one stores
903 + * nothing of its own; reading only the page's meta scores it blank
904 + * while the visitor reads a full page.
905 + * - Its stored text carries dynamic-data tags and internal element names
906 + * that never reach the rendered page.
907 + *
908 + * @since 2.2.1
909 + *
910 + * @param int $post_id Post being resolved.
911 + * @return string Extracted text, or '' when Bricks has nothing for it.
912 + */
913 + private static function from_bricks(int $post_id): string {
914 + $tree = self::bricks_tree($post_id);
915 +
916 + if (empty($tree)) {
917 + return '';
918 + }
919 +
920 + return self::strip_bricks_dynamic_tags(
921 + self::text_from_tree(self::with_bricks_heading_tags(self::without_bricks_element_labels($tree)))
922 + );
923 + }
924 +
925 + /**
926 + * Whether Bricks — not the block editor — renders this post.
927 + *
928 + * Bricks writes `bricks` or `wordpress` into its editor-mode meta as the
929 + * author toggles between the two, and never clears the content it stored
930 + * for the other mode. Only the `wordpress` value is disqualifying: an
931 + * absent value is the normal state for a post Bricks built and never
932 + * toggled. This follows Bricks' own `Helpers::render_with_bricks()`, which
933 + * bails on exactly that one value.
934 + *
935 + * It deliberately does not match it exactly: the comparison here is
936 + * case-insensitive, where Bricks' is strict. Bricks 2.3.12 only ever writes
937 + * the value lowercase, so the two agree on everything Bricks itself
938 + * stores; they part company only on a value some other integration wrote.
939 + * The two shipping today disagree about the casing — SureRank compares
940 + * against `'WordPress'`, AIOSEO against `'bricks'` — and of the two ways to
941 + * be wrong about `'WordPress'`, blocking costs a score on a page that has
942 + * one, while allowing scores stale content the visitor never sees, which is
943 + * the failure this gate exists to prevent.
944 + *
945 + * @since 2.2.1
946 + *
947 + * @param int $post_id Post being resolved.
948 + * @return bool
949 + */
950 + private static function bricks_owns_post(int $post_id): bool {
951 + $mode = get_post_meta($post_id, self::bricks_editor_mode_key(), true);
952 +
953 + // phpcs:ignore WordPress.WP.CapitalPDangit.MisspelledInText -- Bricks' own stored meta value, lower-cased for the comparison.
954 + return !(is_string($mode) && 'wordpress' === strtolower(trim($mode)));
955 + }
956 +
957 + /**
958 + * The post whose Bricks tree actually renders for this post.
959 + *
960 + * Usually the post itself. When it stores nothing of its own, Bricks falls
961 + * back to whichever content template's conditions match, and that template
962 + * is a separate post carrying the words the visitor reads.
963 + *
964 + * Resolution is delegated to Bricks rather than reimplemented: template
965 + * conditions are a whole rules engine (post IDs, types, taxonomies,
966 + * archives), and a second implementation would drift from it. Bricks
967 + * answers through statics, so they are saved and restored around the call —
968 + * `set_active_templates()` returns early once populated, and on a
969 + * front-end request Bricks has already populated it for the page being
970 + * served. Clobbering that would corrupt the render in progress.
971 + *
972 + * Best-effort by design: any failure returns the post's own data, which is
973 + * exactly today's behaviour.
974 + *
975 + * @since 2.2.1
976 + *
977 + * @param int $post_id Post being resolved.
978 + * @return int Post ID holding the Bricks tree, or 0 when there is none.
979 + */
980 + private static function bricks_content_source(int $post_id): int {
981 + $own = get_post_meta($post_id, self::bricks_meta_key(), true);
982 + if ((is_array($own) && !empty($own)) || (is_string($own) && '' !== trim($own))) {
983 + return $post_id;
984 + }
985 +
986 + if (!class_exists('\\Bricks\\Database')
987 + || !method_exists('\\Bricks\\Database', 'set_active_templates')
988 + ) {
989 + return 0;
990 + }
991 +
992 + // `set_active_templates()` writes TWO statics — `$active_templates` and,
993 + // when a header template resolves, `$header_position`. Both are saved,
994 + // and both are restored in `finally` rather than on the happy path: a
995 + // throw part-way through (a third-party hook on
996 + // `bricks/database/content_type`, `bricks/builder/data_post_id` or
997 + // `bricks/active_templates` is enough) must not leave Bricks' render
998 + // state holding this lookup's values. Restoring only after a clean
999 + // return is what the `catch` below would otherwise skip.
1000 + $has_header_position = property_exists('\\Bricks\\Database', 'header_position');
1001 + $saved_templates = \Bricks\Database::$active_templates;
1002 + $saved_header_position = $has_header_position ? \Bricks\Database::$header_position : null;
1003 +
1004 + try {
1005 + \Bricks\Database::$active_templates = [];
1006 + \Bricks\Database::set_active_templates($post_id);
1007 + $template = (int) (\Bricks\Database::$active_templates['content'] ?? 0);
1008 + } catch (\Throwable $e) {
1009 + return 0;
1010 + } finally {
1011 + \Bricks\Database::$active_templates = $saved_templates;
1012 + if ($has_header_position) {
1013 + \Bricks\Database::$header_position = $saved_header_position;
1014 + }
1015 + }
1016 +
1017 + // A template that is the post itself adds nothing over the empty read
1018 + // above, and would otherwise recurse conceptually.
1019 + return $template === $post_id ? 0 : $template;
1020 + }
1021 +
1022 + /**
1023 + * Splice component definitions into the tree.
1024 + *
1025 + * A Bricks component keeps its markup in the `bricks_components` option,
1026 + * not on the page. The page stores only an instance: an element carrying
1027 + * `cid` and, usually, empty `settings`. Walking the page alone therefore
1028 + * found no words at all, and a page built entirely from components scored
1029 + * blank — the same failure as a page built from a content template.
1030 + *
1031 + * Confirmed on Bricks 2.3.12: `Bricks\Frontend::render_data()` renders the
1032 + * component's copy from an instance this walker extracted '' from.
1033 + *
1034 + * The definition is read straight from the option rather than through
1035 + * `Bricks\Helpers::get_component_instance()`. That helper resolves an
1036 + * instance's property overrides, which would be better, but it reads
1037 + * `Bricks\Database::$global_data['components']` — populated once per
1038 + * request, and empty in the admin and CLI contexts where bulk scoring
1039 + * runs. Refreshing it would mean writing to Bricks' live render state, the
1040 + * same hazard the template resolver is careful to avoid, and gating on it
1041 + * would make a page score differently in wp-admin than on the front end.
1042 + * Reading the stored definition is consistent everywhere.
1043 + *
1044 + * The trade-off: an instance that overrides a component property is scored
1045 + * with the component's authored copy rather than the override. That is the
1046 + * text the component renders by default, and it is much closer than the
1047 + * nothing this returned before.
1048 + *
1049 + * @since 2.2.1
1050 + *
1051 + * @param array $tree Bricks content area.
1052 + * @return array Tree with component elements spliced in after each instance.
1053 + */
1054 + private static function expand_bricks_components(array $tree): array {
1055 + $expanded = [];
1056 + $open = [];
1057 +
1058 + $walk = static function (array $elements, int $depth) use (&$walk, &$expanded, &$open): void {
1059 + foreach ($elements as $element) {
1060 + $expanded[] = $element;
1061 +
1062 + if (!is_array($element) || empty($element['cid']) || !is_string($element['cid'])) {
1063 + continue;
1064 + }
1065 +
1066 + $cid = $element['cid'];
1067 +
1068 + // A component nested inside its own definition would recurse
1069 + // forever; the depth cap covers deep but legitimate nesting.
1070 + if (isset($open[$cid]) || $depth > 4) {
1071 + continue;
1072 + }
1073 +
1074 + $children = self::bricks_component_elements($cid);
1075 + if (empty($children)) {
1076 + continue;
1077 + }
1078 +
1079 + // Re-entrant per branch, not per page: the guard is released
1080 + // after the walk so a second instance further along the page
1081 + // still expands, rather than being mistaken for recursion.
1082 + //
1083 + // That does NOT double the word count — `text_from_tree()`
1084 + // ends in `array_unique()`, which collapses a repeated
1085 + // component's copy the same way it collapses a value repeated
1086 + // across responsive breakpoints. Expanding both instances is
1087 + // about not silently dropping the second one's structure.
1088 + $open[$cid] = true;
1089 + $walk($children, $depth + 1);
1090 + unset($open[$cid]);
1091 + }
1092 + };
1093 +
1094 + $walk($tree, 0);
1095 +
1096 + return $expanded;
1097 + }
1098 +
1099 + /**
1100 + * The stored elements of one Bricks component.
1101 + *
1102 + * @since 2.2.1
1103 + *
1104 + * @param string $cid Component id held by an instance element.
1105 + * @return array Component elements, or [] when it cannot be resolved.
1106 + */
1107 + private static function bricks_component_elements(string $cid): array {
1108 + $components = get_option(self::bricks_constant('BRICKS_DB_COMPONENTS', self::BRICKS_COMPONENTS_OPTION), []);
1109 +
1110 + if (!is_array($components)) {
1111 + return [];
1112 + }
1113 +
1114 + foreach ($components as $component) {
1115 + $component = self::as_children($component);
1116 + if (null === $component) {
1117 + continue;
1118 + }
1119 +
1120 + if (isset($component['id']) && $component['id'] === $cid && !empty($component['elements'])) {
1121 + return is_array($component['elements']) ? self::bricks_render_order($component['elements']) : [];
1122 + }
1123 + }
1124 +
1125 + return [];
1126 + }
1127 +
1128 + /**
1129 + * Drop each Bricks element's internal name before the tree is walked.
1130 + *
1131 + * A Bricks element carries an optional top-level `label` — the nickname an
1132 + * author types in the Structure panel to find it again ("Hero headline",
1133 + * "CTA row"). It is builder chrome and is never rendered, but `label` is in
1134 + * CONTENT_KEYS because it is real content for other builders' form fields,
1135 + * so it was being counted as page copy.
1136 + *
1137 + * Only the element's own `label` is removed. A `label` inside `settings`
1138 + * is a rendered field label and stays.
1139 + *
1140 + * @since 2.2.1
1141 + *
1142 + * @param array $tree Bricks content area.
1143 + * @return array Tree with element nicknames removed.
1144 + */
1145 + private static function without_bricks_element_labels(array $tree): array {
1146 + foreach ($tree as $index => $element) {
1147 + if (is_array($element) && isset($element['id'], $element['label'])) {
1148 + unset($tree[$index]['label']);
1149 + }
1150 + }
1151 +
1152 + return $tree;
1153 + }
1154 +
1155 + /**
1156 + * Give each Bricks Heading the tag it renders with when none is stored.
1157 + *
1158 + * Applied on the Bricks path only. Most other builders' text nodes carry no
1159 + * tag because they are not headings, so a generic "text without a tag is a
1160 + * heading" rule in heading_tag_from() would turn every paragraph into one.
1161 + *
1162 + * A `tag` of `custom` is left alone: the element then renders its
1163 + * `customTag`, which is not necessarily a heading.
1164 + *
1165 + * @since 2.15.0
1166 + *
1167 + * @param array $tree Bricks content area.
1168 + * @return array Tree with each untagged Heading's default tag filled in.
1169 + */
1170 + private static function with_bricks_heading_tags(array $tree): array {
1171 + $default = null;
1172 +
1173 + foreach ($tree as $index => $element) {
1174 + if (!is_array($element) || self::BRICKS_HEADING_ELEMENT !== ($element['name'] ?? null)) {
1175 + continue;
1176 + }
1177 +
1178 + $settings = $element['settings'] ?? [];
1179 + if (!is_array($settings)) {
1180 + continue;
1181 + }
1182 +
1183 + $tag = $settings['tag'] ?? '';
1184 + if (is_string($tag) && '' !== trim($tag)) {
1185 + continue;
1186 + }
1187 +
1188 + if (null === $default) {
1189 + $default = self::bricks_default_heading_tag();
1190 + }
1191 +
1192 + $settings['tag'] = $default;
1193 + $tree[$index]['settings'] = $settings;
1194 + }
1195 +
1196 + return $tree;
1197 + }
1198 +
1199 + /**
1200 + * The tag Bricks gives a Heading that does not set one.
1201 + *
1202 + * The active theme style can change it. Bricks only loads theme styles for
1203 + * a front-end render, so in admin, REST and CLI requests this is the
1204 + * element's own default.
1205 + *
1206 + * @since 2.15.0
1207 + *
1208 + * @return string Heading tag, h1 to h6.
1209 + */
1210 + private static function bricks_default_heading_tag(): string {
1211 + if (class_exists('\\Bricks\\Theme_Styles')
1212 + && method_exists('\\Bricks\\Theme_Styles', 'get_setting_by_key')
1213 + ) {
1214 + try {
1215 + $styled = \Bricks\Theme_Styles::get_setting_by_key(self::BRICKS_HEADING_ELEMENT, 'tag');
1216 + } catch (\Throwable $e) {
1217 + $styled = null;
1218 + }
1219 +
1220 + if (is_string($styled) && preg_match('/^h[1-6]$/i', trim($styled))) {
1221 + return strtolower(trim($styled));
1222 + }
1223 + }
1224 +
1225 + return self::BRICKS_HEADING_DEFAULT_TAG;
1226 + }
1227 +
1228 + /**
1229 + * Give each Elementor Heading widget its default `header_size` if unstored.
1230 + *
1231 + * @since 2.15.0
1232 + *
1233 + * @param array $elements Decoded `_elementor_data`.
1234 + * @return array The same tree, with untagged Heading widgets tagged.
1235 + */
1236 + private static function with_elementor_heading_tags(array $elements): array {
1237 + foreach ($elements as $index => $element) {
1238 + if (!is_array($element)) {
1239 + continue;
1240 + }
1241 +
1242 + if ('heading' === ($element['widgetType'] ?? null)) {
1243 + $settings = $element['settings'] ?? [];
1244 + if (is_array($settings)) {
1245 + $size = $settings['header_size'] ?? '';
1246 + if (!is_string($size) || '' === trim($size)) {
1247 + $settings['header_size'] = self::ELEMENTOR_HEADING_DEFAULT_TAG;
1248 + $element['settings'] = $settings;
1249 + }
1250 + }
1251 + }
1252 +
1253 + if (!empty($element['elements']) && is_array($element['elements'])) {
1254 + $element['elements'] = self::with_elementor_heading_tags($element['elements']);
1255 + }
1256 +
1257 + $elements[$index] = $element;
1258 + }
1259 +
1260 + return $elements;
1261 + }
1262 +
1263 + /**
1264 + * Remove Bricks dynamic-data tags from extracted text.
1265 + *
1266 + * Bricks stores `{post_title}`, `{post_meta:price}`, `{echo:my_fn}` and the
1267 + * like verbatim and resolves them when it renders. Extraction reads the
1268 + * stored tree, so without this the placeholders were counted as words, and
1269 + * a heading whose text is `{post_title}` reported the literal token as its
1270 + * heading text.
1271 + *
1272 + * The pattern is deliberately narrower than Bricks' own
1273 + * (`/{([\wÀ-ÖØ-öø-ÿ\-\s\.\/:\(\)...]+)}/u`), which also matches braces
1274 + * containing spaces. Bricks only substitutes tags that resolve to a
1275 + * registered provider and leaves anything else on the page as literal text,
1276 + * so the broad pattern would delete prose the visitor can actually read.
1277 + * Matching only tag-shaped tokens keeps every real sentence and still
1278 + * removes every placeholder — the same trade-off SureRank makes.
1279 + *
1280 + * @since 2.2.1
1281 + *
1282 + * @param string $text Extracted text.
1283 + * @return string Text with placeholders removed.
1284 + */
1285 + private static function strip_bricks_dynamic_tags(string $text): string {
1286 + $stripped = preg_replace('/\{[a-z0-9_][a-z0-9_:\-\.]*\}/i', '', $text);
1287 +
1288 + if (null === $stripped) {
1289 + return $text;
1290 + }
1291 +
1292 + // Collapse the runs of spaces a removed tag leaves mid-sentence,
1293 + // without touching the newlines that separate collected nodes.
1294 + $tidied = preg_replace('/[ \t]{2,}/', ' ', $stripped);
1295 +
1296 + return null === $tidied ? $stripped : $tidied;
1297 + }
1298 +
1299 + /**
1300 + * Bricks' content-area meta key, preferring Bricks' own constant.
1301 + *
1302 + * @since 2.2.1
1303 + *
1304 + * @return string
1305 + */
1306 + private static function bricks_meta_key(): string {
1307 + return self::bricks_constant('BRICKS_DB_PAGE_CONTENT', self::BRICKS_CONTENT_META_KEY);
1308 + }
1309 +
1310 + /**
1311 + * Bricks' editor-mode meta key, preferring Bricks' own constant.
1312 + *
1313 + * @since 2.2.1
1314 + *
1315 + * @return string
1316 + */
1317 + private static function bricks_editor_mode_key(): string {
1318 + return self::bricks_constant('BRICKS_DB_EDITOR_MODE', self::BRICKS_EDITOR_MODE_META_KEY);
1319 + }
1320 +
1321 + /**
1322 + * Read one of Bricks' key-name constants, falling back to the literal.
1323 + *
1324 + * @since 2.2.1
1325 + *
1326 + * @param string $name Constant name.
1327 + * @param string $fallback Key to use when the constant is unavailable.
1328 + * @return string
1329 + */
1330 + private static function bricks_constant(string $name, string $fallback): string {
1331 + if (defined($name)) {
1332 + $value = constant($name);
1333 + if (is_string($value) && '' !== trim($value)) {
1334 + return $value;
1335 + }
1336 + }
1337 +
1338 + return $fallback;
1339 + }
1340 +
1341 + /**
259 1342 * Pull text out of whichever builder stored this post.
260 1343 *
261 1344 * @param int $post_id Post ID.
262 1345 * @return string Extracted text, or '' when no builder data was found.
@@ -261,17 +1344,40 @@
261 1344 * @param int $post_id Post ID.
262 1345 * @return string Extracted text, or '' when no builder data was found.
263 1346 */
264 1347 private static function from_builder_meta(int $post_id): string {
1348 + // Bricks first: it is the only builder whose content can live on
1349 + // another post, and the only one gated on an editor mode.
1350 + $bricks = self::from_bricks($post_id);
1351 + if (!self::is_blank($bricks)) {
1352 + return $bricks;
1353 + }
1354 +
265 1355 foreach (self::BUILDER_META_KEYS as $key) {
1356 + if (self::is_oxygen_classic_key($key)) {
1357 + // Resolved as a pair, once, at the first of its keys.
1358 + if ('_ct_builder_json' !== $key) {
1359 + continue;
1360 + }
1361 +
1362 + $oxygen = self::from_oxygen_classic($post_id);
1363 + if (!self::is_blank($oxygen)) {
1364 + return $oxygen;
1365 + }
1366 + continue;
1367 + }
1368 +
266 1369 $stored = get_post_meta($post_id, $key, true);
267 1370
268 1371 if (is_string($stored) && '' !== trim($stored)) {
269 1372 $decoded = json_decode($stored, true);
1373 + if ('_elementor_data' === $key && is_array($decoded)) {
1374 + $decoded = self::with_elementor_heading_tags($decoded);
1375 + }
270 1376
271 1377 // JSON node tree (Breakdance/Oxygen 6, Elementor).
272 1378 if (is_array($decoded)) {
273 - $text = self::text_from_tree($decoded);
1379 + $text = self::text_from_tree(self::unwrap_tree_envelope($decoded));
274 1380 if (!self::is_blank($text)) {
275 1381 return $text;
276 1382 }
277 1383 continue;
@@ -276,26 +1382,17 @@
276 1382 }
277 1383 continue;
278 1384 }
279 1385
280 - // Shortcode tree (Oxygen classic).
281 - if (strpos($stored, '[') !== false && function_exists('do_shortcode')) {
282 - try {
283 - $rendered = do_shortcode($stored);
284 - } catch (\Throwable $e) {
285 - $rendered = $stored;
286 - }
287 - if (!self::is_blank($rendered)) {
288 - return $rendered;
289 - }
290 - }
291 -
292 1386 continue;
293 1387 }
294 1388
295 - // Some builders store an already-decoded array.
296 - if (is_array($stored)) {
297 - $text = self::text_from_tree($stored);
1389 + // Some builders store an already-decoded tree — an array for most,
1390 + // an array of objects for Beaver Builder (#449).
1391 + $tree = self::as_children($stored);
1392 + if (null !== $tree) {
1393 + $tree = self::unwrap_tree_envelope($tree);
1394 + $text = self::text_from_tree($tree);
298 1395 if (!self::is_blank($text)) {
299 1396 return $text;
300 1397 }
301 1398 }
@@ -304,92 +1401,447 @@
304 1401 return '';
305 1402 }
306 1403
307 1404 /**
1405 + * The node tree inside a Breakdance / Oxygen 6 storage envelope.
1406 + *
1407 + * Neither builder stores its tree directly. The meta value is
1408 + * `{"tree_json_string": "<the tree, JSON-encoded again>"}`, so one
1409 + * json_decode() yields the envelope, not the tree. Walked as a tree, the
1410 + * envelope is a single string leaf: kept whole as "content" when any
1411 + * element held rich text (the encoded JSON then reached scoring, the
1412 + * get-post-content ability and Markdown for AI), dropped when none did,
1413 + * leaving the page empty (#905).
1414 + *
1415 + * An envelope whose inner string does not decode returns an empty tree,
1416 + * never the string: handing the raw JSON back to the walker would bring
1417 + * the JSON-as-content failure back on corrupt data. Anything that is not
1418 + * an envelope is returned unchanged, so a bare tree still resolves.
1419 + *
1420 + * @since 2.15.0
1421 + *
1422 + * @param array $decoded Decoded meta value.
1423 + * @return array The node tree.
1424 + */
1425 + private static function unwrap_tree_envelope(array $decoded): array {
1426 + if (!array_key_exists('tree_json_string', $decoded)) {
1427 + return $decoded;
1428 + }
1429 +
1430 + $inner = is_string($decoded['tree_json_string'])
1431 + ? json_decode($decoded['tree_json_string'], true)
1432 + : $decoded['tree_json_string'];
1433 +
1434 + if (is_array($inner)) {
1435 + return $inner;
1436 + }
1437 +
1438 + // Re-serialised envelopes can carry the tree as an object.
1439 + $inner = self::as_children($inner);
1440 +
1441 + return null !== $inner ? $inner : [];
1442 + }
1443 +
1444 + /**
1445 + * Whether a meta key is one of Oxygen classic's storage keys.
1446 + *
1447 + * @since 2.10.0
1448 + *
1449 + * @param string $key Meta key.
1450 + * @return bool
1451 + */
1452 + private static function is_oxygen_classic_key(string $key): bool {
1453 + return isset(self::OXYGEN_CLASSIC_KEYS[$key]) || in_array($key, self::OXYGEN_CLASSIC_KEYS, true);
1454 + }
1455 +
1456 + /**
1457 + * Text of an Oxygen classic page, from whichever stored form holds more.
1458 + *
1459 + * Oxygen 4.x keeps the same tree twice: as JSON, and as the shortcodes it
1460 + * used before 4.0. The JSON is preferred because it carries copy the
1461 + * shortcode form hides (a composite element's text is base64-encoded
1462 + * inside `ct_options`, which is configuration and stripped). It is not
1463 + * trusted blindly, though. Reading `ct_builder_json` first once meant a
1464 + * key missing from CONTENT_KEYS silently threw the page away while the
1465 + * shortcode copy sat unread next to it, because a non-empty JSON result
1466 + * stopped the search. Comparing the two means the next such gap costs
1467 + * nothing: the richer form wins.
1468 + *
1469 + * A generation is only read as a pair. The prefixed keys are what Oxygen
1470 + * 4.8.3+ reads, so an unprefixed leftover next to them is stale.
1471 + *
1472 + * @since 2.10.0
1473 + *
1474 + * @param int $post_id Post ID.
1475 + * @return string Extracted text, or '' when Oxygen classic stored nothing.
1476 + */
1477 + private static function from_oxygen_classic(int $post_id): string {
1478 + foreach (self::OXYGEN_CLASSIC_KEYS as $json_key => $shortcode_key) {
1479 + $json = get_post_meta($post_id, $json_key, true);
1480 + $shortcodes = get_post_meta($post_id, $shortcode_key, true);
1481 +
1482 + $from_json = '';
1483 + if (is_string($json) && '' !== trim($json)) {
1484 + $decoded = json_decode($json, true);
1485 + if (is_array($decoded)) {
1486 + // `[oxygen data="..."]` is a dynamic-data placeholder
1487 + // Oxygen fills at render time. The shortcode path drops it
1488 + // with every other tag, so it goes here too or the two
1489 + // forms would disagree on the same page.
1490 + $from_json = (string) preg_replace(
1491 + '/\[oxygen\b[^\]]*\]/i',
1492 + ' ',
1493 + self::text_from_tree($decoded)
1494 + );
1495 + }
1496 + }
1497 +
1498 + $from_shortcodes = '';
1499 + if (is_string($shortcodes) && strpos($shortcodes, '[') !== false) {
1500 + $from_shortcodes = self::text_from_shortcodes($shortcodes);
1501 + }
1502 +
1503 + if (self::is_blank($from_json) && self::is_blank($from_shortcodes)) {
1504 + continue;
1505 + }
1506 +
1507 + return self::visible_word_count($from_json) >= self::visible_word_count($from_shortcodes)
1508 + ? $from_json
1509 + : $from_shortcodes;
1510 + }
1511 +
1512 + return '';
1513 + }
1514 +
1515 + /**
1516 + * Rough count of the words a visitor would read in extracted text.
1517 + *
1518 + * Only used to compare two extractions of the same page, so it needs to
1519 + * be consistent rather than locale-exact.
1520 + *
1521 + * @since 2.10.0
1522 + *
1523 + * @param string $text Extracted text or markup.
1524 + * @return int
1525 + */
1526 + private static function visible_word_count(string $text): int {
1527 + $plain = trim((string) preg_replace('/\s+/u', ' ', wp_strip_all_tags($text)));
1528 +
1529 + return '' === $plain ? 0 : count(explode(' ', $plain));
1530 + }
1531 +
1532 + /**
1533 + * Shortcode attributes that carry copy a visitor reads.
1534 + *
1535 + * An allow-list, not a deny-list. Oxygen Classic tags carry far more
1536 + * attributes than they do copy — `id`, `class`, `selector`, `url`,
1537 + * `ct_options` and friends — and a deny-list silently admits every
1538 + * attribute a future builder release invents, which is how markup ends up
1539 + * being counted as prose.
1540 + *
1541 + * @var string[]
1542 + */
1543 + private const SHORTCODE_TEXT_ATTRIBUTES = [
1544 + 'text',
1545 + 'content',
1546 + 'heading',
1547 + 'title',
1548 + 'subtitle',
1549 + 'label',
1550 + 'caption',
1551 + 'description',
1552 + 'alt',
1553 + 'button_text',
1554 + 'link_text',
1555 + ];
1556 +
1557 + /**
1558 + * Extract readable text from a shortcode tree, without rendering it.
1559 + *
1560 + * Oxygen Classic is the only builder whose storage is shortcodes rather
1561 + * than JSON, and the previous implementation handed the string to
1562 + * `do_shortcode()`. That silently depends on Oxygen having registered its
1563 + * `ct_*` handlers in the current request — which it has on a front-end
1564 + * view, and has not during bulk analysis, the post-list column, cron or
1565 + * REST/MCP. With no handlers registered `do_shortcode()` returns its input
1566 + * unchanged, so the raw shortcode source was scored as if it were the
1567 + * page's prose: `[ct_section`, `id="section-1"` and the rest counted toward
1568 + * the word count, while the actual copy sitting in `text="..."` attributes
1569 + * was never counted at all (#776).
1570 + *
1571 + * `strip_shortcodes()` is no help either — it also only knows registered
1572 + * shortcodes, so it leaves the same text untouched.
1573 + *
1574 + * Reading the stored tree directly is what every other builder here already
1575 + * does, and it matches the class's stated design: no render engine, no
1576 + * dependency on load order, safe during a bulk run.
1577 + *
1578 + * Parsing unconditionally, rather than rendering when Oxygen happens to be
1579 + * loaded and parsing otherwise, is deliberate. It makes the extracted text
1580 + * the same in every context, so the score in the editor matches the score
1581 + * from a bulk run or from MCP. The old code produced whichever of the two
1582 + * the request happened to allow, which is why the same post could report
1583 + * two different word counts depending on how it was asked.
1584 + *
1585 + * The trade-off is that rendered output (resolved images, links, anything
1586 + * Oxygen pulls in from a reusable part) is no longer reflected here. For
1587 + * what this text feeds — word count, content scoring, meta-description
1588 + * fallbacks and schema text — that markup was never the point, and counting
1589 + * it only when the builder happened to be booted was the bug.
1590 + *
1591 + * @since 2.10.0
1592 + *
1593 + * @param string $stored Raw shortcode source.
1594 + * @return string Extracted text.
1595 + */
1596 + private static function text_from_shortcodes(string $stored): string {
1597 + // Oxygen stores each element's settings as a JSON blob in `ct_options`.
1598 + // It is configuration, never copy, and it contains braces and brackets
1599 + // that would otherwise confuse the tag scan below, so it goes first.
1600 + //
1601 + // The blob is matched as a balanced JSON object, not as "up to the
1602 + // next quote". Oxygen wraps it in single quotes but does not escape
1603 + // an apostrophe inside it (`"nicename":"Bob's Plumbing"`), so the
1604 + // quote-to-quote match stopped mid-value and the rest of the blob,
1605 + // `s Plumbing"}'` and all, was left in the tag and leaked into the
1606 + // text. Strings inside the object are skipped whole, so neither a quote
1607 + // nor a brace inside a value can end the match early.
1608 + $source = (string) preg_replace(
1609 + '/\sct_options\s*=\s*\'(?<obj>\{(?:[^{}"]++|"(?:[^"\\\\]|\\\\.)*+"|(?&obj))*+\})\'/s',
1610 + '',
1611 + $stored
1612 + );
1613 +
1614 + // Anything not shaped like Oxygen's JSON blob keeps the old,
1615 + // quote-delimited strip.
1616 + $source = (string) preg_replace(
1617 + '/\sct_options\s*=\s*(["\']).*?\1/s',
1618 + '',
1619 + $source
1620 + );
1621 +
1622 + $attributes = implode('|', array_map(
1623 + static fn(string $name): string => preg_quote($name, '/'),
1624 + self::SHORTCODE_TEXT_ATTRIBUTES
1625 + ));
1626 +
1627 + // Replace each shortcode tag with whatever readable copy its attributes
1628 + // carry. Text between tags is left exactly where it is, so the result
1629 + // keeps the page's reading order rather than hoisting all the headings
1630 + // to the front.
1631 + // The attribute blob is matched quote-aware rather than as "anything up
1632 + // to the first `]`". Oxygen copy contains brackets often enough to
1633 + // matter — "Best tools [2026]", "[Updated] our policy" — and a naive
1634 + // scan ends the tag inside the `text` attribute, dropping the copy
1635 + // before the bracket and leaking the stray `"]` after it into the
1636 + // prose. Which is this bug's own failure mode: the wrong text scored.
1637 + //
1638 + // A tag name must start with a letter or underscore. `[2026]` is not a
1639 + // shortcode anyone can register, and scanning it as one dropped the
1640 + // year out of "Best tools [2026]".
1641 + $text = (string) preg_replace_callback(
1642 + '/\[\/?[a-zA-Z_][a-zA-Z0-9_-]*((?:[^\]"\']|"[^"]*"|\'[^\']*\')*)\]/',
1643 + static function (array $matches) use ($attributes): string {
1644 + if ('' === trim($matches[1])) {
1645 + return ' ';
1646 + }
1647 +
1648 + if (!preg_match_all(
1649 + '/\b(' . $attributes . ')\s*=\s*(["\'])(.*?)\2/s',
1650 + $matches[1],
1651 + $found,
1652 + PREG_SET_ORDER
1653 + )) {
1654 + return ' ';
1655 + }
1656 +
1657 + $parts = [];
1658 + foreach ($found as $attribute) {
1659 + $value = trim($attribute[3]);
1660 +
1661 + // An attribute holding markup or a JSON fragment is
1662 + // configuration that happens to share a name with a copy
1663 + // field, not something a visitor reads.
1664 + if ('' === $value || preg_match('/^[\[{<]/', $value)) {
1665 + continue;
1666 + }
1667 +
1668 + $parts[] = $value;
1669 + }
1670 +
1671 + return empty($parts) ? ' ' : ' ' . implode(' ', $parts) . ' ';
1672 + },
1673 + $source
1674 + );
1675 +
1676 + // Oxygen escapes square brackets in an element's copy before writing
1677 + // it between the tags, so that "Best tools [2026]" cannot be mistaken
1678 + // for a shortcode (`oxygen_vsb_filter_shortcode_content_encode()`).
1679 + // Decoded only now, after the tag scan, for the same reason; left
1680 + // encoded, the placeholders were scored as words of their own.
1681 + $text = str_replace(
1682 + ['_OXY_OPENING_BRACKET_', '_OXY_CLOSING_BRACKET_'],
1683 + ['[', ']'],
1684 + $text
1685 + );
1686 +
1687 + // Entities are stored encoded in attributes (&amp;, &#8217;), and would
1688 + // otherwise be counted as words.
1689 + $text = html_entity_decode($text, ENT_QUOTES | ENT_HTML5, 'UTF-8');
1690 +
1691 + return trim((string) preg_replace('/\s+/u', ' ', $text));
1692 + }
1693 +
1694 + /**
1695 + * A node's children, whether it stores them as an array or an object.
1696 + *
1697 + * The walker used to return immediately on `!is_array($node)`, so an
1698 + * object node was dropped along with its entire subtree — silently, as
1699 + * `''`, which the caller reads as "this builder stored nothing" rather
1700 + * than "this walker cannot read this shape".
1701 + *
1702 + * Beaver Builder stores `_fl_builder_data` as an array of stdClass nodes,
1703 + * each with a stdClass `settings` object, so every node would have been
1704 + * dropped and adding its meta key alone would have looked like it worked
1705 + * and changed nothing. Not BB-specific: any builder storing objects hits
1706 + * this, and that shape will come up again (#449).
1707 + *
1708 + * @since 2.1.0
1709 + *
1710 + * @param mixed $node Candidate node.
1711 + * @return array<string|int,mixed>|null Traversable children, or null.
1712 + */
1713 + private static function as_children($node): ?array {
1714 + if (is_array($node)) {
1715 + return $node;
1716 + }
1717 +
1718 + // Deliberately not is_object(): a builder can store a value object
1719 + // (DateTime, a WP_Post) whose properties are not content, and
1720 + // get_object_vars() on those yields noise. stdClass is what the
1721 + // JSON/serialize round-trip produces, which is the shape we want.
1722 + if ($node instanceof \stdClass) {
1723 + return get_object_vars($node);
1724 + }
1725 +
1726 + return null;
1727 + }
1728 +
1729 + /**
308 1730 * Walk a builder node tree and collect the user-visible text.
309 1731 *
310 1732 * Values are joined with block-level markup so downstream heading, link and
311 1733 * image detection keeps working on the result.
312 1734 *
1735 + * One depth-first walk, so the output follows the tree's own order, which
1736 + * is the order the builders read here render in. This used to be two
1737 + * passes over the whole tree, one for the reconstructed headings, links and
1738 + * images and one for the remaining text, and the output followed pass
1739 + * order: every heading and button on the page first, every paragraph after
1740 + * them. That order became the meta description, og:description, the schema
1741 + * description and Pro's Markdown for AI document (#907).
1742 + *
313 1743 * @param array $tree Decoded builder tree.
314 1744 * @return string Collected HTML.
315 1745 */
316 1746 private static function text_from_tree(array $tree): string {
317 - $collected = [];
1747 + // Each entry is [value, is_markup], in tree order.
1748 + $entries = [];
318 1749
319 1750 // Strings already represented inside reconstructed markup, so the plain
320 - // sweep below doesn't emit a link label or heading a second time and
321 - // double it in the word count.
1751 + // text doesn't emit a link label or heading a second time and double it
1752 + // in the word count. Applied after the walk, against the whole tree:
1753 + // a string folded into markup anywhere is dropped everywhere, exactly
1754 + // as it was when the markup pass ran over the whole tree first. Checking
1755 + // it during the walk instead would let a bare copy that appears before
1756 + // its heading through.
322 1757 $consumed = [];
323 1758
324 - // Pass 1 — rebuild <a>, <img> and <hN> from node *shape*. This has to
325 - // happen per node rather than per leaf: a link's label and its
326 - // destination are separate sibling fields, so once the tree is
327 - // flattened to leaves the pairing is gone.
328 - $reconstruct = static function ($node) use (&$reconstruct, &$collected, &$consumed): void {
329 - if (!is_array($node)) {
1759 + // Markup is content wherever it appears; bare strings only count when
1760 + // their key says they are content, so slugs and class names stay out of
1761 + // the word count.
1762 + $leaf = static function ($value, $key) use (&$entries): void {
1763 + if (!is_string($value) || '' === trim($value)) {
330 1764 return;
331 1765 }
332 1766
333 - $markup = self::markup_for_node($node, $consumed);
334 - if ('' !== $markup) {
335 - $collected[] = $markup;
1767 + $is_content_key = is_string($key)
1768 + && in_array(strtolower($key), self::CONTENT_KEYS, true);
1769 +
1770 + if ($is_content_key || strpos($value, '<') !== false) {
1771 + $entries[] = [$value, false];
336 1772 }
1773 + };
337 1774
338 - foreach ($node as $child_key => $child) {
339 - // A `link` / `image` sub-object is a destination descriptor the
340 - // parent has already folded into its markup. Descending into it
341 - // would emit the same URL a second time as a bare link, and
342 - // would turn an image's own `url` field into a spurious <a>.
343 - if (is_string($child_key)
344 - && (in_array(strtolower($child_key), self::URL_KEYS, true)
345 - || in_array(strtolower($child_key), self::IMAGE_KEYS, true))
346 - ) {
347 - continue;
348 - }
1775 + // Text only, no reconstruction. Used for a `link` / `image` / video
1776 + // sub-object: a destination descriptor the parent has already folded
1777 + // into its markup. Rebuilding inside it would emit the same URL a second
1778 + // time as a bare link and turn an image's own `url` field into a
1779 + // spurious <a>, but any copy it carries still counts.
1780 + $sweep = static function ($node, $key) use (&$sweep, $leaf): void {
1781 + $children = self::as_children($node);
1782 + if (null === $children) {
1783 + $leaf($node, $key);
1784 + return;
1785 + }
349 1786
350 - $reconstruct($child);
1787 + foreach ($children as $child_key => $child) {
1788 + $sweep($child, is_string($child_key) ? $child_key : $key);
351 1789 }
352 1790 };
353 - $reconstruct($tree);
354 1791
355 - // Pass 2 — remaining visible text.
356 - $walk = static function ($node, $key = null) use (&$walk, &$collected, &$consumed): void {
357 - if (is_array($node)) {
358 - foreach ($node as $child_key => $child) {
359 - $walk($child, is_string($child_key) ? $child_key : $key);
360 - }
1792 + // Rebuild <a>, <img> and <hN> from node *shape*, then carry on through
1793 + // the node's own fields in order. This has to happen per node rather
1794 + // than per leaf: a link's label and its destination are separate
1795 + // sibling fields, so once the tree is flattened to leaves the pairing
1796 + // is gone.
1797 + $walk = static function ($node, $key = null) use (&$walk, $sweep, $leaf, &$entries, &$consumed): void {
1798 + $children = self::as_children($node);
1799 + if (null === $children) {
1800 + $leaf($node, $key);
361 1801 return;
362 1802 }
363 1803
364 - if (!is_string($node) || '' === trim($node)) {
365 - return;
1804 + $markup = self::markup_for_node($children, $consumed);
1805 + if ('' !== $markup) {
1806 + $entries[] = [$markup, true];
366 1807 }
367 1808
368 - // Already inside a reconstructed tag.
369 - if (in_array($node, $consumed, true)) {
370 - return;
371 - }
1809 + foreach ($children as $child_key => $child) {
1810 + $next_key = is_string($child_key) ? $child_key : $key;
372 1811
373 - $is_content_key = is_string($key)
374 - && in_array(strtolower($key), self::CONTENT_KEYS, true);
1812 + if (is_string($child_key)
1813 + && (in_array(strtolower($child_key), self::URL_KEYS, true)
1814 + || in_array(strtolower($child_key), self::IMAGE_KEYS, true)
1815 + || in_array(strtolower($child_key), self::VIDEO_KEYS, true))
1816 + ) {
1817 + $sweep($child, $next_key);
1818 + continue;
1819 + }
375 1820
376 - // Markup is content wherever it appears; bare strings only count
377 - // when their key says they are content, so slugs and class names
378 - // stay out of the word count.
379 - if ($is_content_key || strpos($node, '<') !== false) {
380 - $collected[] = $node;
1821 + $walk($child, $next_key);
381 1822 }
382 1823 };
383 1824
384 1825 $walk($tree);
385 1826
1827 + $collected = [];
1828 + foreach ($entries as [$value, $is_markup]) {
1829 + // Already inside a reconstructed tag.
1830 + if (!$is_markup && in_array($value, $consumed, true)) {
1831 + continue;
1832 + }
1833 +
1834 + $collected[] = $value;
1835 + }
1836 +
386 1837 if (empty($collected)) {
387 1838 return '';
388 1839 }
389 1840
390 1841 // De-duplicate: builder trees often repeat a value across responsive
391 - // breakpoints, which would otherwise multiply the word count.
1842 + // breakpoints, which would otherwise multiply the word count. Keeps the
1843 + // first occurrence, so a repeat never moves a value later in the page.
392 1844 $collected = array_unique($collected);
393 1845
394 1846 return implode("\n", $collected);
395 1847 }
@@ -394,8 +1846,79 @@
394 1846 return implode("\n", $collected);
395 1847 }
396 1848
397 1849 /**
1850 + * The video source a node is actually playing, if any.
1851 + *
1852 + * @since 2.3.1
1853 + *
1854 + * @param array $node Builder node.
1855 + * @return string Video source, or '' when the node carries none.
1856 + */
1857 + private static function video_from(array $node): string {
1858 + foreach ($node as $key => $value) {
1859 + if (!is_string($key) || !is_string($value)) {
1860 + continue;
1861 + }
1862 +
1863 + if (!in_array(strtolower($key), self::VIDEO_TYPE_KEYS, true)) {
1864 + continue;
1865 + }
1866 +
1867 + $keys = self::VIDEO_KEYS_BY_TYPE[strtolower(trim($value))] ?? null;
1868 + if (null === $keys) {
1869 + continue;
1870 + }
1871 +
1872 + // A recognised video_type settles it, including when that
1873 + // provider's own field is empty. Falling through to the flat sweep
1874 + // there handed back whichever sibling key happened to come first in
1875 + // node order — the stale youtube_url left behind after switching
1876 + // the widget to a hosted file, which is exactly what keying on the
1877 + // declared type is meant to prevent.
1878 + $declared = self::url_from($node, $keys);
1879 +
1880 + return self::is_video_source($declared) ? $declared : '';
1881 + }
1882 +
1883 + $url = self::url_from($node, self::VIDEO_KEYS);
1884 +
1885 + return self::is_video_source($url) ? $url : '';
1886 + }
1887 +
1888 + /**
1889 + * Whether a value can be a video source.
1890 + *
1891 + * `looks_like_url()` also accepts `#anchor`, `mailto:` and `tel:`, which a
1892 + * link node may legitimately hold but a video cannot: `<iframe src="#top">`
1893 + * is not a video and would reach a video sitemap as one.
1894 + *
1895 + * @since 2.3.1
1896 + *
1897 + * @param string $url Candidate source.
1898 + * @return bool
1899 + */
1900 + private static function is_video_source(string $url): bool {
1901 + return '' !== $url
1902 + && (1 === preg_match('#^(https?:)?//#i', $url) || str_starts_with($url, '/'));
1903 + }
1904 +
1905 + /**
1906 + * Whether a video source points at a file rather than a provider page.
1907 + *
1908 + * @since 2.3.1
1909 + *
1910 + * @param string $url Video source.
1911 + * @return bool
1912 + */
1913 + private static function is_video_file(string $url): bool {
1914 + $path = (string) wp_parse_url($url, PHP_URL_PATH);
1915 + $ext = strtolower((string) pathinfo($path, PATHINFO_EXTENSION));
1916 +
1917 + return in_array($ext, self::VIDEO_FILE_EXTENSIONS, true);
1918 + }
1919 +
1920 + /**
398 1921 * Rebuild the HTML a single builder node represents, if any.
399 1922 *
400 1923 * Looks only at the node's own fields (plus one level of nesting, because
401 1924 * builders commonly wrap a destination as `{ url: … }`). Returns an empty
@@ -411,12 +1934,23 @@
411 1934 private static function markup_for_node(array $node, array &$consumed): string {
412 1935 $text = self::first_value($node, self::CONTENT_KEYS);
413 1936 $url = self::url_from($node, self::URL_KEYS);
414 1937 $image = self::image_from($node);
1938 + $video = self::video_from($node);
415 1939 $tag = self::heading_tag_from($node);
416 1940
417 1941 $parts = [];
418 1942
1943 + // Video: an embed shape rather than a link, so the video detector can
1944 + // see it while the link counters do not mistake it for an outbound
1945 + // link. A file source becomes <video src>, anything else an <iframe>,
1946 + // matching how the builder itself renders the two cases.
1947 + if ('' !== $video) {
1948 + $parts[] = self::is_video_file($video)
1949 + ? sprintf('<video src="%s"></video>', esc_url_raw($video))
1950 + : sprintf('<iframe src="%s"></iframe>', esc_url_raw($video));
1951 + }
1952 +
419 1953 // Image: alt text matters as much as the tag, since alt checks run over
420 1954 // whatever this returns.
421 1955 if ('' !== $image['url']) {
422 1956 $alt = '' !== $image['alt'] ? $image['alt'] : (string) self::first_value($node, self::ALT_KEYS);
@@ -490,10 +2024,11 @@
490 2024 return trim($value);
491 2025 }
492 2026
493 2027 // Elementor and Breakdance both nest the destination one level down.
494 - if (is_array($value)) {
495 - foreach ($value as $nested_key => $nested) {
2028 + $nested_values = self::as_children($value);
2029 + if (null !== $nested_values) {
2030 + foreach ($nested_values as $nested_key => $nested) {
496 2031 if (is_string($nested_key)
497 2032 && in_array(strtolower($nested_key), ['url', 'href', 'permalink'], true)
498 2033 && is_string($nested)
499 2034 && self::looks_like_url($nested)
@@ -522,12 +2057,13 @@
522 2057 if (is_string($value) && self::looks_like_url($value)) {
523 2058 return ['url' => trim($value), 'alt' => ''];
524 2059 }
525 2060
526 - if (is_array($value)) {
2061 + $nested_values = self::as_children($value);
2062 + if (null !== $nested_values) {
527 2063 $url = '';
528 2064 $alt = '';
529 - foreach ($value as $nested_key => $nested) {
2065 + foreach ($nested_values as $nested_key => $nested) {
530 2066 if (!is_string($nested_key) || !is_string($nested)) {
531 2067 continue;
532 2068 }
533 2069 $nested_key = strtolower($nested_key);
@@ -603,13 +2139,30 @@
603 2139 return (bool) preg_match('#^(mailto:|tel:|\#)#i', $value);
604 2140 }
605 2141
606 2142 /**
607 - * Whether a value carries no readable text.
2143 + * Whether a value carries nothing worth analyzing.
608 2144 *
2145 + * Readable text is the usual signal, but not the only one: a page can be
2146 + * made entirely of media. A builder section holding just a gallery
2147 + * reconstructs to `<img>` tags and one holding just a video widget to a
2148 + * single `<iframe>` — both strip to an empty string, so a text-only test
2149 + * discarded them here and the page fell through to the next builder key,
2150 + * then to the raw markup, and finally reported as having no content at all.
2151 + *
2152 + * Comments are dropped before the tag test: the raw markup this class falls
2153 + * back to on a builder page is unrendered block comments, which must stay
2154 + * blank rather than be mistaken for reconstructed media.
2155 + *
609 2156 * @param string $value Candidate content.
610 2157 * @return bool
611 2158 */
612 2159 private static function is_blank(string $value): bool {
613 - return '' === trim(wp_strip_all_tags($value));
2160 + if ('' !== trim(wp_strip_all_tags($value))) {
2161 + return false;
2162 + }
2163 +
2164 + $without_comments = (string) preg_replace('~<!--.*?-->~s', '', $value);
2165 +
2166 + return 1 !== preg_match('~<(?:a|img|iframe|video|source)\b~i', $without_comments);
614 2167 }
615 2168 }