PluginProbe
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO / 2.14.2
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO v2.14.2
2.14.2 2.14.1 2.14.0 2.13.0 2.12.0 2.11.0 2.10.0 2.9.0 2.8.0 2.7.0 2.6.0 2.5.0 2.4.0 2.3.0 2.2.0 2.1.1 2.1.0 2.0.2 2.0.1 2.0.0 1.32.0 1.31.0 1.30.0 1.29.0 1.28.0 All 57 releases
← All changes | includes/seo/class-builder-content.php +791 -74 2.4.0 → 2.14.2 View file →
@@ -40,19 +40,44 @@
40 40
41 41 /**
42 42 * Post meta keys that hold builder data, in priority order.
43 43 *
44 - * Several generations of the same builder are listed on purpose: Oxygen 6
45 - * is Breakdance under the hood (`_breakdance_data`), while earlier Oxygen
46 - * releases used `_oxygen_data` or the shortcode-based
47 - * `ct_builder_shortcodes`. A site can only have one of them.
44 + * Several generations of the same builder are listed on purpose. Oxygen 6
45 + * is Breakdance under the hood and writes the same tree, under its own
46 + * prefix: Breakdance keeps it in `_breakdance_data`, Oxygen 6 in
47 + * `_oxygen_data` (the key is `__bdox('_meta_prefix') . 'data'`, and the
48 + * prefix is `_oxygen_` under Oxygen). Both store it inside a
49 + * `tree_json_string` envelope, see unwrap_tree_envelope(). Earlier Oxygen
50 + * releases used the shortcode-based `ct_builder_shortcodes` and its JSON
51 + * sibling. A site can only have one of them.
48 52 *
49 53 * @var string[]
50 54 */
51 55 private const BUILDER_META_KEYS = [
52 - '_breakdance_data', // Oxygen 6+ / Breakdance
53 - '_oxygen_data', // Oxygen (earlier releases)
54 - 'ct_builder_shortcodes', // Oxygen classic
56 + '_breakdance_data', // Breakdance
57 + '_oxygen_data', // Oxygen 6+ (Breakdance engine, Oxygen prefix)
58 + // Oxygen classic. 4.x writes the tree as JSON to `ct_builder_json`
59 + // while still keeping `ct_builder_shortcodes`. A post carrying only
60 + // the JSON key used to match no key at all and fall through to an
61 + // empty `post_content`, which reads as a one-word page (#776).
62 + //
63 + // Oxygen 4.8.3 then renamed every `ct_*` post meta key to `_ct_*`
64 + // (`oxygen_vsb_update_4_8_3()` runs `oxy_prefix_meta_keys()` on
65 + // upgrade, and `oxy_get_post_meta()` only reads the prefixed name
66 + // from then on). A current Oxygen classic site therefore has only the
67 + // underscored keys, which nothing here listed, so every one of its
68 + // pages resolved as empty. The prefixed keys come first because they
69 + // are what Oxygen itself reads; the bare ones cover a site that has
70 + // not run the migration (or reverted it with `?unprefix_meta`).
71 + //
72 + // These four are not read in this loop: from_oxygen_classic() pairs
73 + // each JSON key with its shortcode sibling so the two forms can be
74 + // compared. They are listed here because this list is also what the
75 + // word-count index watches and what FAQ detection scans.
76 + '_ct_builder_json', // Oxygen classic 4.8.3+ (JSON tree)
77 + 'ct_builder_json', // Oxygen classic 4.0-4.8.2 (JSON tree)
78 + '_ct_builder_shortcodes', // Oxygen classic 4.8.3+ (shortcode tree)
79 + 'ct_builder_shortcodes', // Oxygen classic < 4.8.3 (shortcode tree)
55 80 '_elementor_data', // Elementor
56 81 // Beaver Builder. Published layout first: `_fl_builder_draft` holds
57 82 // unsaved changes and would score content the visitor cannot see.
58 83 // Both are arrays of stdClass nodes, which is why the walker below
@@ -110,8 +135,42 @@
110 135 */
111 136 private const BRICKS_POST_CONTENT_ELEMENT = 'post-content';
112 137
113 138 /**
139 + * Bricks' Heading element, and the tag it renders when none is stored.
140 + *
141 + * Bricks leaves a setting out of storage while it equals its default, so a
142 + * Heading left on its default tag is stored with no `tag` at all. Bricks
143 + * 2.4.1 renders it as `h3` (`Element_Heading::$tag`, overridable by the
144 + * active theme style's `tag`), and the walker, which only wraps text whose
145 + * node names a tag, read it as body copy (#908).
146 + *
147 + * @since 2.15.0
148 + * @var string
149 + */
150 + private const BRICKS_HEADING_ELEMENT = 'heading';
151 +
152 + /**
153 + * Tag a Bricks Heading renders when neither it nor a theme style sets one.
154 + *
155 + * @since 2.15.0
156 + * @var string
157 + */
158 + private const BRICKS_HEADING_DEFAULT_TAG = 'h3';
159 +
160 + /**
161 + * Tag an Elementor Heading widget renders when `header_size` is not stored.
162 + *
163 + * Elementor saves `settings.toJSON({ remove: ['default'] })`, so a heading
164 + * left on its default size has no `header_size` in `_elementor_data`, and
165 + * that default is `h2` (#908).
166 + *
167 + * @since 2.15.0
168 + * @var string
169 + */
170 + private const ELEMENTOR_HEADING_DEFAULT_TAG = 'h2';
171 +
172 + /**
114 173 * Resolved Bricks trees for this request, keyed by post ID.
115 174 *
116 175 * Rendering one page asks for the tree about twenty times — every
117 176 * description, every schema node, the FAQ guard — and resolving it is not
@@ -142,11 +201,38 @@
142 201 private const CONTENT_KEYS = [
143 202 'text', 'title', 'subtitle', 'heading', 'subheading', 'content',
144 203 'description', 'caption', 'excerpt', 'label', 'value', 'html',
145 204 'editor', 'quote', 'answer', 'question', 'body', 'button_text',
205 + // Oxygen classic keeps an element's copy in `options.ct_content`
206 + // (headline, text block, rich text, link and button labels). It is the
207 + // field Oxygen's own serializer moves between the tags when it writes
208 + // shortcodes (`parse_components_tree()`), and the one Relevanssi and
209 + // Oxygen's WPML integration read. Missing from this list, the walker
210 + // kept only copy that happened to contain markup: a page of plain
211 + // headings and paragraphs lost almost all of its words.
212 + 'ct_content',
213 + // Oxygen's composite elements keep their copy under `options.original`
214 + // instead, one key per field. Taken from the list Oxygen itself treats
215 + // as text when it serializes (`$options_to_encode`); the numeric price
216 + // fields and the progress bar's right-hand percentage are left out.
217 + 'testimonial_text', 'testimonial_author', 'testimonial_author_info',
218 + 'icon_box_heading', 'icon_box_text',
219 + 'pricing_box_package_title', 'pricing_box_package_subtitle', 'pricing_box_content',
220 + 'progress_bar_left_text',
146 221 ];
147 222
148 223 /**
224 + * Oxygen classic's storage generations, as JSON key => shortcode key.
225 + *
226 + * @since 2.10.0
227 + * @var array<string,string>
228 + */
229 + private const OXYGEN_CLASSIC_KEYS = [
230 + '_ct_builder_json' => '_ct_builder_shortcodes',
231 + 'ct_builder_json' => 'ct_builder_shortcodes',
232 + ];
233 +
234 + /**
149 235 * JSON keys whose values hold a link destination.
150 236 *
151 237 * Builders store a link's destination in a structured field separate from
152 238 * its label, either as a bare URL string or as a `{ url: … }` object.
@@ -255,8 +341,22 @@
255 341 */
256 342 private const ALT_KEYS = ['alt', 'alt_text', 'image_alt', 'title'];
257 343
258 344 /**
345 + * The global post and every global `setup_postdata()` writes.
346 + *
347 + * Rendering points them at the post being analyzed, then puts each one
348 + * back exactly as it was, unset included (#860).
349 + *
350 + * @since 2.12.0
351 + * @var string[]
352 + */
353 + private const POSTDATA_GLOBALS = [
354 + 'post', 'id', 'authordata', 'currentday', 'currentmonth',
355 + 'page', 'pages', 'multipage', 'more', 'numpages',
356 + ];
357 +
358 + /**
259 359 * Resolve the content worth analyzing for a post.
260 360 *
261 361 * @param \WP_Post $post Post being analyzed.
262 362 * @return string HTML/text to analyze.
@@ -417,8 +517,23 @@
417 517 *
418 518 * @param int $post_id Post being resolved.
419 519 * @return array<int,mixed> Elements, or [] when Bricks renders nothing here.
420 520 */
521 + /**
522 + * The builder meta keys, for callers that need to inspect the raw storage
523 + * rather than the text extracted from it.
524 + *
525 + * The SEO Analyzer reads these to answer "is there a ThinkRank FAQ element
526 + * on this post?", which is a question about the stored tree, not about the
527 + * words in it (#686).
528 + *
529 + * @since 2.7.0
530 + * @return string[]
531 + */
532 + public static function builder_meta_keys(): array {
533 + return self::BUILDER_META_KEYS;
534 + }
535 +
421 536 public static function bricks_tree(int $post_id): array {
422 537 if (array_key_exists($post_id, self::$bricks_trees)) {
423 538 return self::$bricks_trees[$post_id];
424 539 }
@@ -465,12 +580,82 @@
465 580 if (!is_array($stored) || empty($stored)) {
466 581 return [];
467 582 }
468 583
469 - return self::expand_bricks_components($stored);
584 + return self::expand_bricks_components(self::bricks_render_order($stored));
470 585 }
471 586
472 587 /**
588 + * A flat Bricks element list, in the order Bricks renders it.
589 + *
590 + * Bricks stores one flat list and links it with `parent` and `children`
591 + * ids. `Frontend::render_data()` renders the root elements in list order
592 + * and each element's children in the order of its `children` array, so a
593 + * child's position in the list says nothing about where it appears on the
594 + * page. Walking the list as stored put a section's contents wherever they
595 + * happened to be saved (#907).
596 + *
597 + * Anything the walk does not reach (an orphan, a cycle) keeps its stored
598 + * position after the rest, so no copy is dropped.
599 + *
600 + * @since 2.15.0
601 + *
602 + * @param array $elements Flat Bricks element list.
603 + * @return array The same elements, in render order.
604 + */
605 + private static function bricks_render_order(array $elements): array {
606 + $by_id = [];
607 + foreach ($elements as $index => $element) {
608 + $id = is_array($element) ? ($element['id'] ?? null) : null;
609 + if (is_scalar($id) && '' !== (string) $id && !isset($by_id[(string) $id])) {
610 + $by_id[(string) $id] = $index;
611 + }
612 + }
613 +
614 + if (empty($by_id)) {
615 + return $elements;
616 + }
617 +
618 + $ordered = [];
619 + $placed = [];
620 +
621 + $place = static function ($index) use (&$place, &$ordered, &$placed, $elements, $by_id): void {
622 + if (isset($placed[$index])) {
623 + return;
624 + }
625 +
626 + $placed[$index] = true;
627 + $ordered[] = $elements[$index];
628 +
629 + $children = is_array($elements[$index]) ? ($elements[$index]['children'] ?? []) : [];
630 + if (!is_array($children)) {
631 + return;
632 + }
633 +
634 + foreach ($children as $child_id) {
635 + if (is_scalar($child_id) && isset($by_id[(string) $child_id])) {
636 + $place($by_id[(string) $child_id]);
637 + }
638 + }
639 + };
640 +
641 + foreach ($elements as $index => $element) {
642 + $parent = is_array($element) ? ($element['parent'] ?? null) : null;
643 + if (empty($parent) || !is_scalar($parent) || !isset($by_id[(string) $parent])) {
644 + $place($index);
645 + }
646 + }
647 +
648 + foreach ($elements as $index => $element) {
649 + if (!isset($placed[$index])) {
650 + $ordered[] = $element;
651 + }
652 + }
653 +
654 + return $ordered;
655 + }
656 +
657 + /**
473 658 * Resolve an arbitrary chunk of editor markup for the given post.
474 659 *
475 660 * The editor sends its live content to the scorer so an author sees their
476 661 * unsaved edits reflected. On a builder page that live string is the raw
@@ -491,9 +676,9 @@
491 676 * @param \WP_Post $post Post the markup belongs to.
492 677 * @return string Content to analyze.
493 678 */
494 679 public static function resolve_markup(string $raw, \WP_Post $post): string {
495 - $content = self::render_post_content($raw);
680 + $content = self::render_post_content($raw, $post);
496 681
497 682 // Block markup that renders to nothing usually means the builder that
498 683 // owns those blocks did not register them in this context — Divi 5
499 684 // loads its module library lazily per-request, so in CLI, REST, admin
@@ -541,19 +726,74 @@
541 726 *
542 727 * Best-effort: a third-party block that fatals must not take the whole
543 728 * score down with it.
544 729 *
545 - * @param string $raw Raw post content.
730 + * Runs as the post's own context, as it would on the front end. Admin and
731 + * REST requests have no current post, so a shortcode reading
732 + * `get_the_ID()` got nothing, and one looping a related-posts query left
733 + * the global post on the last of them: its `wp_reset_postdata()` goes back
734 + * to the main query's post, and there is none. On the Classic Editor this
735 + * runs after the form prints its hidden `post_ID` and before the title and
736 + * editor, which then showed the related post, and Update saved it over the
737 + * original (#860).
738 + *
739 + * Secondary queries also run with front-end statuses. In wp-admin core
740 + * marks every `WP_Query` as an admin query and, when no `post_status` is
741 + * set, adds the statuses the admin post list shows, draft among them, so
742 + * a related-posts shortcode listed drafts the front end never shows and
743 + * the editor-load analysis disagreed with REST and the page (#902).
744 + *
745 + * The main query points at the post too. Restoring the globals afterwards
746 + * (#860) did not reach between shortcodes: a related-posts loop's own
747 + * `wp_reset_postdata()` still found no post on the main query, so every
748 + * later shortcode in the same render saw the last looped post (#903).
749 + *
750 + * @param string $raw Raw post content.
751 + * @param \WP_Post $post Post the content belongs to.
546 752 * @return string Rendered content.
547 753 */
548 - private static function render_post_content(string $raw): string {
754 + private static function render_post_content(string $raw, \WP_Post $post): string {
549 755 if ('' === trim($raw)) {
550 756 return '';
551 757 }
552 758
553 - $content = $raw;
759 + $content = $raw;
760 + $previous = self::snapshot_post_globals();
554 761
762 + // After pre_get_posts core reads `is_admin` only to add the admin
763 + // list's statuses when none were asked for, so queries that set
764 + // `post_status`, and the main query, are untouched.
765 + $front_end_statuses = static function ($query): void {
766 + if ($query instanceof \WP_Query && !$query->is_main_query()) {
767 + $query->is_admin = false;
768 + }
769 + };
770 +
771 + // `wp_reset_postdata()` returns to the main query's post, which admin
772 + // and REST requests do not have. Restored in finally, null included.
773 + $main_query = (isset($GLOBALS['wp_query']) && $GLOBALS['wp_query'] instanceof \WP_Query) ? $GLOBALS['wp_query'] : null;
774 + $main_query_post = $main_query ? $main_query->post : null;
775 +
555 776 try {
777 + // phpcs:ignore WordPress.WP.GlobalVariablesOverride.Prohibited -- Made current for the render, restored in finally.
778 + $GLOBALS['post'] = $post;
779 + add_action('pre_get_posts', $front_end_statuses, PHP_INT_MIN);
780 + if ($main_query) {
781 + $main_query->post = $post;
782 + }
783 +
784 + // Fires `the_post`, which these paths never fired before: admin,
785 + // REST and cron analysis had no current post at all. That is the
786 + // same signal the front-end loop sends and it is what makes
787 + // `get_the_ID()` work inside a shortcode, but it is a new call on a
788 + // path that runs in bulk — the word-count index resolves every post
789 + // it visits — so a theme that counts views on `the_post` will count
790 + // them during indexing. Accepted deliberately: without it a
791 + // shortcode cannot resolve its own post, which is the bug (#860).
792 + if (function_exists('setup_postdata')) {
793 + setup_postdata($post);
794 + }
795 +
556 796 if (function_exists('has_blocks') && function_exists('do_blocks') && has_blocks($raw)) {
557 797 $content = do_blocks($raw);
558 798 }
559 799
@@ -562,8 +802,14 @@
562 802 $content = do_shortcode($content);
563 803 }
564 804 } catch (\Throwable $e) {
565 805 return $raw;
806 + } finally {
807 + if ($main_query) {
808 + $main_query->post = $main_query_post;
809 + }
810 + remove_action('pre_get_posts', $front_end_statuses, PHP_INT_MIN);
811 + self::restore_post_globals($previous);
566 812 }
567 813
568 814 return self::is_blank($content) ? $raw : $content;
569 815 }
@@ -568,8 +814,49 @@
568 814 return self::is_blank($content) ? $raw : $content;
569 815 }
570 816
571 817 /**
818 + * The post globals as they are now; a global that is unset has no key.
819 + *
820 + * @since 2.12.0
821 + *
822 + * @return array<string,mixed>
823 + */
824 + private static function snapshot_post_globals(): array {
825 + $snapshot = [];
826 +
827 + foreach (self::POSTDATA_GLOBALS as $name) {
828 + if (array_key_exists($name, $GLOBALS)) {
829 + $snapshot[$name] = $GLOBALS[$name];
830 + }
831 + }
832 +
833 + return $snapshot;
834 + }
835 +
836 + /**
837 + * Put the post globals back as snapshot_post_globals() found them.
838 + *
839 + * Assigned directly rather than through `setup_postdata()`: there may have
840 + * been no post to set up, and re-running it would fire `the_post` again.
841 + *
842 + * @since 2.12.0
843 + *
844 + * @param array<string,mixed> $snapshot From snapshot_post_globals().
845 + * @return void
846 + */
847 + private static function restore_post_globals(array $snapshot): void {
848 + foreach (self::POSTDATA_GLOBALS as $name) {
849 + if (array_key_exists($name, $snapshot)) {
850 + // phpcs:ignore WordPress.NamingConventions.PrefixAllGlobals.NonPrefixedVariableFound -- Core's own globals, put back as they were.
851 + $GLOBALS[$name] = $snapshot[$name];
852 + } else {
853 + unset($GLOBALS[$name]);
854 + }
855 + }
856 + }
857 +
858 + /**
572 859 * Extract text from the attributes of parsed blocks.
573 860 *
574 861 * @param string $raw Raw post content containing block markup.
575 862 * @return string Collected text, or '' when nothing was found.
@@ -630,9 +917,9 @@
630 917 return '';
631 918 }
632 919
633 920 return self::strip_bricks_dynamic_tags(
634 - self::text_from_tree(self::without_bricks_element_labels($tree))
921 + self::text_from_tree(self::with_bricks_heading_tags(self::without_bricks_element_labels($tree)))
635 922 );
636 923 }
637 924
638 925 /**
@@ -830,9 +1117,9 @@
830 1117 continue;
831 1118 }
832 1119
833 1120 if (isset($component['id']) && $component['id'] === $cid && !empty($component['elements'])) {
834 - return is_array($component['elements']) ? $component['elements'] : [];
1121 + return is_array($component['elements']) ? self::bricks_render_order($component['elements']) : [];
835 1122 }
836 1123 }
837 1124
838 1125 return [];
@@ -865,8 +1152,116 @@
865 1152 return $tree;
866 1153 }
867 1154
868 1155 /**
1156 + * Give each Bricks Heading the tag it renders with when none is stored.
1157 + *
1158 + * Applied on the Bricks path only. Most other builders' text nodes carry no
1159 + * tag because they are not headings, so a generic "text without a tag is a
1160 + * heading" rule in heading_tag_from() would turn every paragraph into one.
1161 + *
1162 + * A `tag` of `custom` is left alone: the element then renders its
1163 + * `customTag`, which is not necessarily a heading.
1164 + *
1165 + * @since 2.15.0
1166 + *
1167 + * @param array $tree Bricks content area.
1168 + * @return array Tree with each untagged Heading's default tag filled in.
1169 + */
1170 + private static function with_bricks_heading_tags(array $tree): array {
1171 + $default = null;
1172 +
1173 + foreach ($tree as $index => $element) {
1174 + if (!is_array($element) || self::BRICKS_HEADING_ELEMENT !== ($element['name'] ?? null)) {
1175 + continue;
1176 + }
1177 +
1178 + $settings = $element['settings'] ?? [];
1179 + if (!is_array($settings)) {
1180 + continue;
1181 + }
1182 +
1183 + $tag = $settings['tag'] ?? '';
1184 + if (is_string($tag) && '' !== trim($tag)) {
1185 + continue;
1186 + }
1187 +
1188 + if (null === $default) {
1189 + $default = self::bricks_default_heading_tag();
1190 + }
1191 +
1192 + $settings['tag'] = $default;
1193 + $tree[$index]['settings'] = $settings;
1194 + }
1195 +
1196 + return $tree;
1197 + }
1198 +
1199 + /**
1200 + * The tag Bricks gives a Heading that does not set one.
1201 + *
1202 + * The active theme style can change it. Bricks only loads theme styles for
1203 + * a front-end render, so in admin, REST and CLI requests this is the
1204 + * element's own default.
1205 + *
1206 + * @since 2.15.0
1207 + *
1208 + * @return string Heading tag, h1 to h6.
1209 + */
1210 + private static function bricks_default_heading_tag(): string {
1211 + if (class_exists('\\Bricks\\Theme_Styles')
1212 + && method_exists('\\Bricks\\Theme_Styles', 'get_setting_by_key')
1213 + ) {
1214 + try {
1215 + $styled = \Bricks\Theme_Styles::get_setting_by_key(self::BRICKS_HEADING_ELEMENT, 'tag');
1216 + } catch (\Throwable $e) {
1217 + $styled = null;
1218 + }
1219 +
1220 + if (is_string($styled) && preg_match('/^h[1-6]$/i', trim($styled))) {
1221 + return strtolower(trim($styled));
1222 + }
1223 + }
1224 +
1225 + return self::BRICKS_HEADING_DEFAULT_TAG;
1226 + }
1227 +
1228 + /**
1229 + * Give each Elementor Heading widget its default `header_size` if unstored.
1230 + *
1231 + * @since 2.15.0
1232 + *
1233 + * @param array $elements Decoded `_elementor_data`.
1234 + * @return array The same tree, with untagged Heading widgets tagged.
1235 + */
1236 + private static function with_elementor_heading_tags(array $elements): array {
1237 + foreach ($elements as $index => $element) {
1238 + if (!is_array($element)) {
1239 + continue;
1240 + }
1241 +
1242 + if ('heading' === ($element['widgetType'] ?? null)) {
1243 + $settings = $element['settings'] ?? [];
1244 + if (is_array($settings)) {
1245 + $size = $settings['header_size'] ?? '';
1246 + if (!is_string($size) || '' === trim($size)) {
1247 + $settings['header_size'] = self::ELEMENTOR_HEADING_DEFAULT_TAG;
1248 + $element['settings'] = $settings;
1249 + }
1250 + }
1251 + }
1252 +
1253 + if (!empty($element['elements']) && is_array($element['elements'])) {
1254 + $element['elements'] = self::with_elementor_heading_tags($element['elements']);
1255 + }
1256 +
1257 + $elements[$index] = $element;
1258 + }
1259 +
1260 + return $elements;
1261 + }
1262 +
1263 + /**
869 1264 * Remove Bricks dynamic-data tags from extracted text.
870 1265 *
871 1266 * Bricks stores `{post_title}`, `{post_meta:price}`, `{echo:my_fn}` and the
872 1267 * like verbatim and resolves them when it renders. Extraction reads the
@@ -957,16 +1352,32 @@
957 1352 return $bricks;
958 1353 }
959 1354
960 1355 foreach (self::BUILDER_META_KEYS as $key) {
1356 + if (self::is_oxygen_classic_key($key)) {
1357 + // Resolved as a pair, once, at the first of its keys.
1358 + if ('_ct_builder_json' !== $key) {
1359 + continue;
1360 + }
1361 +
1362 + $oxygen = self::from_oxygen_classic($post_id);
1363 + if (!self::is_blank($oxygen)) {
1364 + return $oxygen;
1365 + }
1366 + continue;
1367 + }
1368 +
961 1369 $stored = get_post_meta($post_id, $key, true);
962 1370
963 1371 if (is_string($stored) && '' !== trim($stored)) {
964 1372 $decoded = json_decode($stored, true);
1373 + if ('_elementor_data' === $key && is_array($decoded)) {
1374 + $decoded = self::with_elementor_heading_tags($decoded);
1375 + }
965 1376
966 1377 // JSON node tree (Breakdance/Oxygen 6, Elementor).
967 1378 if (is_array($decoded)) {
968 - $text = self::text_from_tree($decoded);
1379 + $text = self::text_from_tree(self::unwrap_tree_envelope($decoded));
969 1380 if (!self::is_blank($text)) {
970 1381 return $text;
971 1382 }
972 1383 continue;
@@ -971,20 +1382,8 @@
971 1382 }
972 1383 continue;
973 1384 }
974 1385
975 - // Shortcode tree (Oxygen classic).
976 - if (strpos($stored, '[') !== false && function_exists('do_shortcode')) {
977 - try {
978 - $rendered = do_shortcode($stored);
979 - } catch (\Throwable $e) {
980 - $rendered = $stored;
981 - }
982 - if (!self::is_blank($rendered)) {
983 - return $rendered;
984 - }
985 - }
986 -
987 1386 continue;
988 1387 }
989 1388
990 1389 // Some builders store an already-decoded tree — an array for most,
@@ -990,8 +1389,9 @@
990 1389 // Some builders store an already-decoded tree — an array for most,
991 1390 // an array of objects for Beaver Builder (#449).
992 1391 $tree = self::as_children($stored);
993 1392 if (null !== $tree) {
1393 + $tree = self::unwrap_tree_envelope($tree);
994 1394 $text = self::text_from_tree($tree);
995 1395 if (!self::is_blank($text)) {
996 1396 return $text;
997 1397 }
@@ -1001,8 +1401,298 @@
1001 1401 return '';
1002 1402 }
1003 1403
1004 1404 /**
1405 + * The node tree inside a Breakdance / Oxygen 6 storage envelope.
1406 + *
1407 + * Neither builder stores its tree directly. The meta value is
1408 + * `{"tree_json_string": "<the tree, JSON-encoded again>"}`, so one
1409 + * json_decode() yields the envelope, not the tree. Walked as a tree, the
1410 + * envelope is a single string leaf: kept whole as "content" when any
1411 + * element held rich text (the encoded JSON then reached scoring, the
1412 + * get-post-content ability and Markdown for AI), dropped when none did,
1413 + * leaving the page empty (#905).
1414 + *
1415 + * An envelope whose inner string does not decode returns an empty tree,
1416 + * never the string: handing the raw JSON back to the walker would bring
1417 + * the JSON-as-content failure back on corrupt data. Anything that is not
1418 + * an envelope is returned unchanged, so a bare tree still resolves.
1419 + *
1420 + * @since 2.15.0
1421 + *
1422 + * @param array $decoded Decoded meta value.
1423 + * @return array The node tree.
1424 + */
1425 + private static function unwrap_tree_envelope(array $decoded): array {
1426 + if (!array_key_exists('tree_json_string', $decoded)) {
1427 + return $decoded;
1428 + }
1429 +
1430 + $inner = is_string($decoded['tree_json_string'])
1431 + ? json_decode($decoded['tree_json_string'], true)
1432 + : $decoded['tree_json_string'];
1433 +
1434 + if (is_array($inner)) {
1435 + return $inner;
1436 + }
1437 +
1438 + // Re-serialised envelopes can carry the tree as an object.
1439 + $inner = self::as_children($inner);
1440 +
1441 + return null !== $inner ? $inner : [];
1442 + }
1443 +
1444 + /**
1445 + * Whether a meta key is one of Oxygen classic's storage keys.
1446 + *
1447 + * @since 2.10.0
1448 + *
1449 + * @param string $key Meta key.
1450 + * @return bool
1451 + */
1452 + private static function is_oxygen_classic_key(string $key): bool {
1453 + return isset(self::OXYGEN_CLASSIC_KEYS[$key]) || in_array($key, self::OXYGEN_CLASSIC_KEYS, true);
1454 + }
1455 +
1456 + /**
1457 + * Text of an Oxygen classic page, from whichever stored form holds more.
1458 + *
1459 + * Oxygen 4.x keeps the same tree twice: as JSON, and as the shortcodes it
1460 + * used before 4.0. The JSON is preferred because it carries copy the
1461 + * shortcode form hides (a composite element's text is base64-encoded
1462 + * inside `ct_options`, which is configuration and stripped). It is not
1463 + * trusted blindly, though. Reading `ct_builder_json` first once meant a
1464 + * key missing from CONTENT_KEYS silently threw the page away while the
1465 + * shortcode copy sat unread next to it, because a non-empty JSON result
1466 + * stopped the search. Comparing the two means the next such gap costs
1467 + * nothing: the richer form wins.
1468 + *
1469 + * A generation is only read as a pair. The prefixed keys are what Oxygen
1470 + * 4.8.3+ reads, so an unprefixed leftover next to them is stale.
1471 + *
1472 + * @since 2.10.0
1473 + *
1474 + * @param int $post_id Post ID.
1475 + * @return string Extracted text, or '' when Oxygen classic stored nothing.
1476 + */
1477 + private static function from_oxygen_classic(int $post_id): string {
1478 + foreach (self::OXYGEN_CLASSIC_KEYS as $json_key => $shortcode_key) {
1479 + $json = get_post_meta($post_id, $json_key, true);
1480 + $shortcodes = get_post_meta($post_id, $shortcode_key, true);
1481 +
1482 + $from_json = '';
1483 + if (is_string($json) && '' !== trim($json)) {
1484 + $decoded = json_decode($json, true);
1485 + if (is_array($decoded)) {
1486 + // `[oxygen data="..."]` is a dynamic-data placeholder
1487 + // Oxygen fills at render time. The shortcode path drops it
1488 + // with every other tag, so it goes here too or the two
1489 + // forms would disagree on the same page.
1490 + $from_json = (string) preg_replace(
1491 + '/\[oxygen\b[^\]]*\]/i',
1492 + ' ',
1493 + self::text_from_tree($decoded)
1494 + );
1495 + }
1496 + }
1497 +
1498 + $from_shortcodes = '';
1499 + if (is_string($shortcodes) && strpos($shortcodes, '[') !== false) {
1500 + $from_shortcodes = self::text_from_shortcodes($shortcodes);
1501 + }
1502 +
1503 + if (self::is_blank($from_json) && self::is_blank($from_shortcodes)) {
1504 + continue;
1505 + }
1506 +
1507 + return self::visible_word_count($from_json) >= self::visible_word_count($from_shortcodes)
1508 + ? $from_json
1509 + : $from_shortcodes;
1510 + }
1511 +
1512 + return '';
1513 + }
1514 +
1515 + /**
1516 + * Rough count of the words a visitor would read in extracted text.
1517 + *
1518 + * Only used to compare two extractions of the same page, so it needs to
1519 + * be consistent rather than locale-exact.
1520 + *
1521 + * @since 2.10.0
1522 + *
1523 + * @param string $text Extracted text or markup.
1524 + * @return int
1525 + */
1526 + private static function visible_word_count(string $text): int {
1527 + $plain = trim((string) preg_replace('/\s+/u', ' ', wp_strip_all_tags($text)));
1528 +
1529 + return '' === $plain ? 0 : count(explode(' ', $plain));
1530 + }
1531 +
1532 + /**
1533 + * Shortcode attributes that carry copy a visitor reads.
1534 + *
1535 + * An allow-list, not a deny-list. Oxygen Classic tags carry far more
1536 + * attributes than they do copy — `id`, `class`, `selector`, `url`,
1537 + * `ct_options` and friends — and a deny-list silently admits every
1538 + * attribute a future builder release invents, which is how markup ends up
1539 + * being counted as prose.
1540 + *
1541 + * @var string[]
1542 + */
1543 + private const SHORTCODE_TEXT_ATTRIBUTES = [
1544 + 'text',
1545 + 'content',
1546 + 'heading',
1547 + 'title',
1548 + 'subtitle',
1549 + 'label',
1550 + 'caption',
1551 + 'description',
1552 + 'alt',
1553 + 'button_text',
1554 + 'link_text',
1555 + ];
1556 +
1557 + /**
1558 + * Extract readable text from a shortcode tree, without rendering it.
1559 + *
1560 + * Oxygen Classic is the only builder whose storage is shortcodes rather
1561 + * than JSON, and the previous implementation handed the string to
1562 + * `do_shortcode()`. That silently depends on Oxygen having registered its
1563 + * `ct_*` handlers in the current request — which it has on a front-end
1564 + * view, and has not during bulk analysis, the post-list column, cron or
1565 + * REST/MCP. With no handlers registered `do_shortcode()` returns its input
1566 + * unchanged, so the raw shortcode source was scored as if it were the
1567 + * page's prose: `[ct_section`, `id="section-1"` and the rest counted toward
1568 + * the word count, while the actual copy sitting in `text="..."` attributes
1569 + * was never counted at all (#776).
1570 + *
1571 + * `strip_shortcodes()` is no help either — it also only knows registered
1572 + * shortcodes, so it leaves the same text untouched.
1573 + *
1574 + * Reading the stored tree directly is what every other builder here already
1575 + * does, and it matches the class's stated design: no render engine, no
1576 + * dependency on load order, safe during a bulk run.
1577 + *
1578 + * Parsing unconditionally, rather than rendering when Oxygen happens to be
1579 + * loaded and parsing otherwise, is deliberate. It makes the extracted text
1580 + * the same in every context, so the score in the editor matches the score
1581 + * from a bulk run or from MCP. The old code produced whichever of the two
1582 + * the request happened to allow, which is why the same post could report
1583 + * two different word counts depending on how it was asked.
1584 + *
1585 + * The trade-off is that rendered output (resolved images, links, anything
1586 + * Oxygen pulls in from a reusable part) is no longer reflected here. For
1587 + * what this text feeds — word count, content scoring, meta-description
1588 + * fallbacks and schema text — that markup was never the point, and counting
1589 + * it only when the builder happened to be booted was the bug.
1590 + *
1591 + * @since 2.10.0
1592 + *
1593 + * @param string $stored Raw shortcode source.
1594 + * @return string Extracted text.
1595 + */
1596 + private static function text_from_shortcodes(string $stored): string {
1597 + // Oxygen stores each element's settings as a JSON blob in `ct_options`.
1598 + // It is configuration, never copy, and it contains braces and brackets
1599 + // that would otherwise confuse the tag scan below, so it goes first.
1600 + //
1601 + // The blob is matched as a balanced JSON object, not as "up to the
1602 + // next quote". Oxygen wraps it in single quotes but does not escape
1603 + // an apostrophe inside it (`"nicename":"Bob's Plumbing"`), so the
1604 + // quote-to-quote match stopped mid-value and the rest of the blob,
1605 + // `s Plumbing"}'` and all, was left in the tag and leaked into the
1606 + // text. Strings inside the object are skipped whole, so neither a quote
1607 + // nor a brace inside a value can end the match early.
1608 + $source = (string) preg_replace(
1609 + '/\sct_options\s*=\s*\'(?<obj>\{(?:[^{}"]++|"(?:[^"\\\\]|\\\\.)*+"|(?&obj))*+\})\'/s',
1610 + '',
1611 + $stored
1612 + );
1613 +
1614 + // Anything not shaped like Oxygen's JSON blob keeps the old,
1615 + // quote-delimited strip.
1616 + $source = (string) preg_replace(
1617 + '/\sct_options\s*=\s*(["\']).*?\1/s',
1618 + '',
1619 + $source
1620 + );
1621 +
1622 + $attributes = implode('|', array_map(
1623 + static fn(string $name): string => preg_quote($name, '/'),
1624 + self::SHORTCODE_TEXT_ATTRIBUTES
1625 + ));
1626 +
1627 + // Replace each shortcode tag with whatever readable copy its attributes
1628 + // carry. Text between tags is left exactly where it is, so the result
1629 + // keeps the page's reading order rather than hoisting all the headings
1630 + // to the front.
1631 + // The attribute blob is matched quote-aware rather than as "anything up
1632 + // to the first `]`". Oxygen copy contains brackets often enough to
1633 + // matter — "Best tools [2026]", "[Updated] our policy" — and a naive
1634 + // scan ends the tag inside the `text` attribute, dropping the copy
1635 + // before the bracket and leaking the stray `"]` after it into the
1636 + // prose. Which is this bug's own failure mode: the wrong text scored.
1637 + //
1638 + // A tag name must start with a letter or underscore. `[2026]` is not a
1639 + // shortcode anyone can register, and scanning it as one dropped the
1640 + // year out of "Best tools [2026]".
1641 + $text = (string) preg_replace_callback(
1642 + '/\[\/?[a-zA-Z_][a-zA-Z0-9_-]*((?:[^\]"\']|"[^"]*"|\'[^\']*\')*)\]/',
1643 + static function (array $matches) use ($attributes): string {
1644 + if ('' === trim($matches[1])) {
1645 + return ' ';
1646 + }
1647 +
1648 + if (!preg_match_all(
1649 + '/\b(' . $attributes . ')\s*=\s*(["\'])(.*?)\2/s',
1650 + $matches[1],
1651 + $found,
1652 + PREG_SET_ORDER
1653 + )) {
1654 + return ' ';
1655 + }
1656 +
1657 + $parts = [];
1658 + foreach ($found as $attribute) {
1659 + $value = trim($attribute[3]);
1660 +
1661 + // An attribute holding markup or a JSON fragment is
1662 + // configuration that happens to share a name with a copy
1663 + // field, not something a visitor reads.
1664 + if ('' === $value || preg_match('/^[\[{<]/', $value)) {
1665 + continue;
1666 + }
1667 +
1668 + $parts[] = $value;
1669 + }
1670 +
1671 + return empty($parts) ? ' ' : ' ' . implode(' ', $parts) . ' ';
1672 + },
1673 + $source
1674 + );
1675 +
1676 + // Oxygen escapes square brackets in an element's copy before writing
1677 + // it between the tags, so that "Best tools [2026]" cannot be mistaken
1678 + // for a shortcode (`oxygen_vsb_filter_shortcode_content_encode()`).
1679 + // Decoded only now, after the tag scan, for the same reason; left
1680 + // encoded, the placeholders were scored as words of their own.
1681 + $text = str_replace(
1682 + ['_OXY_OPENING_BRACKET_', '_OXY_CLOSING_BRACKET_'],
1683 + ['[', ']'],
1684 + $text
1685 + );
1686 +
1687 + // Entities are stored encoded in attributes (&amp;, &#8217;), and would
1688 + // otherwise be counted as words.
1689 + $text = html_entity_decode($text, ENT_QUOTES | ENT_HTML5, 'UTF-8');
1690 +
1691 + return trim((string) preg_replace('/\s+/u', ' ', $text));
1692 + }
1693 +
1694 + /**
1005 1695 * A node's children, whether it stores them as an array or an object.
1006 1696 *
1007 1697 * The walker used to return immediately on `!is_array($node)`, so an
1008 1698 * object node was dropped along with its entire subtree — silently, as
@@ -1041,90 +1731,117 @@
1041 1731 *
1042 1732 * Values are joined with block-level markup so downstream heading, link and
1043 1733 * image detection keeps working on the result.
1044 1734 *
1735 + * One depth-first walk, so the output follows the tree's own order, which
1736 + * is the order the builders read here render in. This used to be two
1737 + * passes over the whole tree, one for the reconstructed headings, links and
1738 + * images and one for the remaining text, and the output followed pass
1739 + * order: every heading and button on the page first, every paragraph after
1740 + * them. That order became the meta description, og:description, the schema
1741 + * description and Pro's Markdown for AI document (#907).
1742 + *
1045 1743 * @param array $tree Decoded builder tree.
1046 1744 * @return string Collected HTML.
1047 1745 */
1048 1746 private static function text_from_tree(array $tree): string {
1049 - $collected = [];
1747 + // Each entry is [value, is_markup], in tree order.
1748 + $entries = [];
1050 1749
1051 1750 // Strings already represented inside reconstructed markup, so the plain
1052 - // sweep below doesn't emit a link label or heading a second time and
1053 - // double it in the word count.
1751 + // text doesn't emit a link label or heading a second time and double it
1752 + // in the word count. Applied after the walk, against the whole tree:
1753 + // a string folded into markup anywhere is dropped everywhere, exactly
1754 + // as it was when the markup pass ran over the whole tree first. Checking
1755 + // it during the walk instead would let a bare copy that appears before
1756 + // its heading through.
1054 1757 $consumed = [];
1055 1758
1056 - // Pass 1 — rebuild <a>, <img> and <hN> from node *shape*. This has to
1057 - // happen per node rather than per leaf: a link's label and its
1058 - // destination are separate sibling fields, so once the tree is
1059 - // flattened to leaves the pairing is gone.
1060 - $reconstruct = static function ($node) use (&$reconstruct, &$collected, &$consumed): void {
1061 - $node = self::as_children($node);
1062 - if (null === $node) {
1759 + // Markup is content wherever it appears; bare strings only count when
1760 + // their key says they are content, so slugs and class names stay out of
1761 + // the word count.
1762 + $leaf = static function ($value, $key) use (&$entries): void {
1763 + if (!is_string($value) || '' === trim($value)) {
1063 1764 return;
1064 1765 }
1065 1766
1066 - $markup = self::markup_for_node($node, $consumed);
1767 + $is_content_key = is_string($key)
1768 + && in_array(strtolower($key), self::CONTENT_KEYS, true);
1769 +
1770 + if ($is_content_key || strpos($value, '<') !== false) {
1771 + $entries[] = [$value, false];
1772 + }
1773 + };
1774 +
1775 + // Text only, no reconstruction. Used for a `link` / `image` / video
1776 + // sub-object: a destination descriptor the parent has already folded
1777 + // into its markup. Rebuilding inside it would emit the same URL a second
1778 + // time as a bare link and turn an image's own `url` field into a
1779 + // spurious <a>, but any copy it carries still counts.
1780 + $sweep = static function ($node, $key) use (&$sweep, $leaf): void {
1781 + $children = self::as_children($node);
1782 + if (null === $children) {
1783 + $leaf($node, $key);
1784 + return;
1785 + }
1786 +
1787 + foreach ($children as $child_key => $child) {
1788 + $sweep($child, is_string($child_key) ? $child_key : $key);
1789 + }
1790 + };
1791 +
1792 + // Rebuild <a>, <img> and <hN> from node *shape*, then carry on through
1793 + // the node's own fields in order. This has to happen per node rather
1794 + // than per leaf: a link's label and its destination are separate
1795 + // sibling fields, so once the tree is flattened to leaves the pairing
1796 + // is gone.
1797 + $walk = static function ($node, $key = null) use (&$walk, $sweep, $leaf, &$entries, &$consumed): void {
1798 + $children = self::as_children($node);
1799 + if (null === $children) {
1800 + $leaf($node, $key);
1801 + return;
1802 + }
1803 +
1804 + $markup = self::markup_for_node($children, $consumed);
1067 1805 if ('' !== $markup) {
1068 - $collected[] = $markup;
1806 + $entries[] = [$markup, true];
1069 1807 }
1070 1808
1071 - foreach ($node as $child_key => $child) {
1072 - // A `link` / `image` sub-object is a destination descriptor the
1073 - // parent has already folded into its markup. Descending into it
1074 - // would emit the same URL a second time as a bare link, and
1075 - // would turn an image's own `url` field into a spurious <a>.
1809 + foreach ($children as $child_key => $child) {
1810 + $next_key = is_string($child_key) ? $child_key : $key;
1811 +
1076 1812 if (is_string($child_key)
1077 1813 && (in_array(strtolower($child_key), self::URL_KEYS, true)
1078 1814 || in_array(strtolower($child_key), self::IMAGE_KEYS, true)
1079 1815 || in_array(strtolower($child_key), self::VIDEO_KEYS, true))
1080 1816 ) {
1817 + $sweep($child, $next_key);
1081 1818 continue;
1082 1819 }
1083 1820
1084 - $reconstruct($child);
1821 + $walk($child, $next_key);
1085 1822 }
1086 1823 };
1087 - $reconstruct($tree);
1088 1824
1089 - // Pass 2 — remaining visible text.
1090 - $walk = static function ($node, $key = null) use (&$walk, &$collected, &$consumed): void {
1091 - $children = self::as_children($node);
1092 - if (null !== $children) {
1093 - foreach ($children as $child_key => $child) {
1094 - $walk($child, is_string($child_key) ? $child_key : $key);
1095 - }
1096 - return;
1097 - }
1825 + $walk($tree);
1098 1826
1099 - if (!is_string($node) || '' === trim($node)) {
1100 - return;
1101 - }
1102 -
1827 + $collected = [];
1828 + foreach ($entries as [$value, $is_markup]) {
1103 1829 // Already inside a reconstructed tag.
1104 - if (in_array($node, $consumed, true)) {
1105 - return;
1830 + if (!$is_markup && in_array($value, $consumed, true)) {
1831 + continue;
1106 1832 }
1107 1833
1108 - $is_content_key = is_string($key)
1109 - && in_array(strtolower($key), self::CONTENT_KEYS, true);
1834 + $collected[] = $value;
1835 + }
1110 1836
1111 - // Markup is content wherever it appears; bare strings only count
1112 - // when their key says they are content, so slugs and class names
1113 - // stay out of the word count.
1114 - if ($is_content_key || strpos($node, '<') !== false) {
1115 - $collected[] = $node;
1116 - }
1117 - };
1118 -
1119 - $walk($tree);
1120 -
1121 1837 if (empty($collected)) {
1122 1838 return '';
1123 1839 }
1124 1840
1125 1841 // De-duplicate: builder trees often repeat a value across responsive
1126 - // breakpoints, which would otherwise multiply the word count.
1842 + // breakpoints, which would otherwise multiply the word count. Keeps the
1843 + // first occurrence, so a repeat never moves a value later in the page.
1127 1844 $collected = array_unique($collected);
1128 1845
1129 1846 return implode("\n", $collected);
1130 1847 }