PluginProbe
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO / 2.10.0
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO v2.10.0
2.10.0 2.9.0 2.8.0 2.7.0 2.6.0 2.5.0 2.4.0 2.3.0 2.2.0 2.1.1 2.1.0 2.0.2 2.0.1 2.0.0 1.32.0 1.31.0 1.30.0 1.29.0 1.28.0 1.27.0 1.26.0 1.25.0 trunk 1.0.0 1.0.1 All 51 releases
← All changes | includes/seo/class-builder-content.php +327 -13 2.5.0 → 2.10.0 View file →
@@ -50,9 +50,30 @@
50 50 */
51 51 private const BUILDER_META_KEYS = [
52 52 '_breakdance_data', // Oxygen 6+ / Breakdance
53 53 '_oxygen_data', // Oxygen (earlier releases)
54 - 'ct_builder_shortcodes', // Oxygen classic
54 + // Oxygen classic. 4.x writes the tree as JSON to `ct_builder_json`
55 + // while still keeping `ct_builder_shortcodes`. A post carrying only
56 + // the JSON key used to match no key at all and fall through to an
57 + // empty `post_content`, which reads as a one-word page (#776).
58 + //
59 + // Oxygen 4.8.3 then renamed every `ct_*` post meta key to `_ct_*`
60 + // (`oxygen_vsb_update_4_8_3()` runs `oxy_prefix_meta_keys()` on
61 + // upgrade, and `oxy_get_post_meta()` only reads the prefixed name
62 + // from then on). A current Oxygen classic site therefore has only the
63 + // underscored keys, which nothing here listed, so every one of its
64 + // pages resolved as empty. The prefixed keys come first because they
65 + // are what Oxygen itself reads; the bare ones cover a site that has
66 + // not run the migration (or reverted it with `?unprefix_meta`).
67 + //
68 + // These four are not read in this loop: from_oxygen_classic() pairs
69 + // each JSON key with its shortcode sibling so the two forms can be
70 + // compared. They are listed here because this list is also what the
71 + // word-count index watches and what FAQ detection scans.
72 + '_ct_builder_json', // Oxygen classic 4.8.3+ (JSON tree)
73 + 'ct_builder_json', // Oxygen classic 4.0-4.8.2 (JSON tree)
74 + '_ct_builder_shortcodes', // Oxygen classic 4.8.3+ (shortcode tree)
75 + 'ct_builder_shortcodes', // Oxygen classic < 4.8.3 (shortcode tree)
55 76 '_elementor_data', // Elementor
56 77 // Beaver Builder. Published layout first: `_fl_builder_draft` holds
57 78 // unsaved changes and would score content the visitor cannot see.
58 79 // Both are arrays of stdClass nodes, which is why the walker below
@@ -142,11 +163,38 @@
142 163 private const CONTENT_KEYS = [
143 164 'text', 'title', 'subtitle', 'heading', 'subheading', 'content',
144 165 'description', 'caption', 'excerpt', 'label', 'value', 'html',
145 166 'editor', 'quote', 'answer', 'question', 'body', 'button_text',
167 + // Oxygen classic keeps an element's copy in `options.ct_content`
168 + // (headline, text block, rich text, link and button labels). It is the
169 + // field Oxygen's own serializer moves between the tags when it writes
170 + // shortcodes (`parse_components_tree()`), and the one Relevanssi and
171 + // Oxygen's WPML integration read. Missing from this list, the walker
172 + // kept only copy that happened to contain markup: a page of plain
173 + // headings and paragraphs lost almost all of its words.
174 + 'ct_content',
175 + // Oxygen's composite elements keep their copy under `options.original`
176 + // instead, one key per field. Taken from the list Oxygen itself treats
177 + // as text when it serializes (`$options_to_encode`); the numeric price
178 + // fields and the progress bar's right-hand percentage are left out.
179 + 'testimonial_text', 'testimonial_author', 'testimonial_author_info',
180 + 'icon_box_heading', 'icon_box_text',
181 + 'pricing_box_package_title', 'pricing_box_package_subtitle', 'pricing_box_content',
182 + 'progress_bar_left_text',
146 183 ];
147 184
148 185 /**
186 + * Oxygen classic's storage generations, as JSON key => shortcode key.
187 + *
188 + * @since 2.10.0
189 + * @var array<string,string>
190 + */
191 + private const OXYGEN_CLASSIC_KEYS = [
192 + '_ct_builder_json' => '_ct_builder_shortcodes',
193 + 'ct_builder_json' => 'ct_builder_shortcodes',
194 + ];
195 +
196 + /**
149 197 * JSON keys whose values hold a link destination.
150 198 *
151 199 * Builders store a link's destination in a structured field separate from
152 200 * its label, either as a bare URL string or as a `{ url: … }` object.
@@ -417,8 +465,23 @@
417 465 *
418 466 * @param int $post_id Post being resolved.
419 467 * @return array<int,mixed> Elements, or [] when Bricks renders nothing here.
420 468 */
469 + /**
470 + * The builder meta keys, for callers that need to inspect the raw storage
471 + * rather than the text extracted from it.
472 + *
473 + * The SEO Analyzer reads these to answer "is there a ThinkRank FAQ element
474 + * on this post?", which is a question about the stored tree, not about the
475 + * words in it (#686).
476 + *
477 + * @since 2.7.0
478 + * @return string[]
479 + */
480 + public static function builder_meta_keys(): array {
481 + return self::BUILDER_META_KEYS;
482 + }
483 +
421 484 public static function bricks_tree(int $post_id): array {
422 485 if (array_key_exists($post_id, self::$bricks_trees)) {
423 486 return self::$bricks_trees[$post_id];
424 487 }
@@ -957,8 +1020,21 @@
957 1020 return $bricks;
958 1021 }
959 1022
960 1023 foreach (self::BUILDER_META_KEYS as $key) {
1024 + if (self::is_oxygen_classic_key($key)) {
1025 + // Resolved as a pair, once, at the first of its keys.
1026 + if ('_ct_builder_json' !== $key) {
1027 + continue;
1028 + }
1029 +
1030 + $oxygen = self::from_oxygen_classic($post_id);
1031 + if (!self::is_blank($oxygen)) {
1032 + return $oxygen;
1033 + }
1034 + continue;
1035 + }
1036 +
961 1037 $stored = get_post_meta($post_id, $key, true);
962 1038
963 1039 if (is_string($stored) && '' !== trim($stored)) {
964 1040 $decoded = json_decode($stored, true);
@@ -971,20 +1047,8 @@
971 1047 }
972 1048 continue;
973 1049 }
974 1050
975 - // Shortcode tree (Oxygen classic).
976 - if (strpos($stored, '[') !== false && function_exists('do_shortcode')) {
977 - try {
978 - $rendered = do_shortcode($stored);
979 - } catch (\Throwable $e) {
980 - $rendered = $stored;
981 - }
982 - if (!self::is_blank($rendered)) {
983 - return $rendered;
984 - }
985 - }
986 -
987 1051 continue;
988 1052 }
989 1053
990 1054 // Some builders store an already-decoded tree — an array for most,
@@ -998,8 +1062,258 @@
998 1062 }
999 1063 }
1000 1064
1001 1065 return '';
1066 + }
1067 +
1068 + /**
1069 + * Whether a meta key is one of Oxygen classic's storage keys.
1070 + *
1071 + * @since 2.10.0
1072 + *
1073 + * @param string $key Meta key.
1074 + * @return bool
1075 + */
1076 + private static function is_oxygen_classic_key(string $key): bool {
1077 + return isset(self::OXYGEN_CLASSIC_KEYS[$key]) || in_array($key, self::OXYGEN_CLASSIC_KEYS, true);
1078 + }
1079 +
1080 + /**
1081 + * Text of an Oxygen classic page, from whichever stored form holds more.
1082 + *
1083 + * Oxygen 4.x keeps the same tree twice: as JSON, and as the shortcodes it
1084 + * used before 4.0. The JSON is preferred because it carries copy the
1085 + * shortcode form hides (a composite element's text is base64-encoded
1086 + * inside `ct_options`, which is configuration and stripped). It is not
1087 + * trusted blindly, though. Reading `ct_builder_json` first once meant a
1088 + * key missing from CONTENT_KEYS silently threw the page away while the
1089 + * shortcode copy sat unread next to it, because a non-empty JSON result
1090 + * stopped the search. Comparing the two means the next such gap costs
1091 + * nothing: the richer form wins.
1092 + *
1093 + * A generation is only read as a pair. The prefixed keys are what Oxygen
1094 + * 4.8.3+ reads, so an unprefixed leftover next to them is stale.
1095 + *
1096 + * @since 2.10.0
1097 + *
1098 + * @param int $post_id Post ID.
1099 + * @return string Extracted text, or '' when Oxygen classic stored nothing.
1100 + */
1101 + private static function from_oxygen_classic(int $post_id): string {
1102 + foreach (self::OXYGEN_CLASSIC_KEYS as $json_key => $shortcode_key) {
1103 + $json = get_post_meta($post_id, $json_key, true);
1104 + $shortcodes = get_post_meta($post_id, $shortcode_key, true);
1105 +
1106 + $from_json = '';
1107 + if (is_string($json) && '' !== trim($json)) {
1108 + $decoded = json_decode($json, true);
1109 + if (is_array($decoded)) {
1110 + // `[oxygen data="..."]` is a dynamic-data placeholder
1111 + // Oxygen fills at render time. The shortcode path drops it
1112 + // with every other tag, so it goes here too or the two
1113 + // forms would disagree on the same page.
1114 + $from_json = (string) preg_replace(
1115 + '/\[oxygen\b[^\]]*\]/i',
1116 + ' ',
1117 + self::text_from_tree($decoded)
1118 + );
1119 + }
1120 + }
1121 +
1122 + $from_shortcodes = '';
1123 + if (is_string($shortcodes) && strpos($shortcodes, '[') !== false) {
1124 + $from_shortcodes = self::text_from_shortcodes($shortcodes);
1125 + }
1126 +
1127 + if (self::is_blank($from_json) && self::is_blank($from_shortcodes)) {
1128 + continue;
1129 + }
1130 +
1131 + return self::visible_word_count($from_json) >= self::visible_word_count($from_shortcodes)
1132 + ? $from_json
1133 + : $from_shortcodes;
1134 + }
1135 +
1136 + return '';
1137 + }
1138 +
1139 + /**
1140 + * Rough count of the words a visitor would read in extracted text.
1141 + *
1142 + * Only used to compare two extractions of the same page, so it needs to
1143 + * be consistent rather than locale-exact.
1144 + *
1145 + * @since 2.10.0
1146 + *
1147 + * @param string $text Extracted text or markup.
1148 + * @return int
1149 + */
1150 + private static function visible_word_count(string $text): int {
1151 + $plain = trim((string) preg_replace('/\s+/u', ' ', wp_strip_all_tags($text)));
1152 +
1153 + return '' === $plain ? 0 : count(explode(' ', $plain));
1154 + }
1155 +
1156 + /**
1157 + * Shortcode attributes that carry copy a visitor reads.
1158 + *
1159 + * An allow-list, not a deny-list. Oxygen Classic tags carry far more
1160 + * attributes than they do copy — `id`, `class`, `selector`, `url`,
1161 + * `ct_options` and friends — and a deny-list silently admits every
1162 + * attribute a future builder release invents, which is how markup ends up
1163 + * being counted as prose.
1164 + *
1165 + * @var string[]
1166 + */
1167 + private const SHORTCODE_TEXT_ATTRIBUTES = [
1168 + 'text',
1169 + 'content',
1170 + 'heading',
1171 + 'title',
1172 + 'subtitle',
1173 + 'label',
1174 + 'caption',
1175 + 'description',
1176 + 'alt',
1177 + 'button_text',
1178 + 'link_text',
1179 + ];
1180 +
1181 + /**
1182 + * Extract readable text from a shortcode tree, without rendering it.
1183 + *
1184 + * Oxygen Classic is the only builder whose storage is shortcodes rather
1185 + * than JSON, and the previous implementation handed the string to
1186 + * `do_shortcode()`. That silently depends on Oxygen having registered its
1187 + * `ct_*` handlers in the current request — which it has on a front-end
1188 + * view, and has not during bulk analysis, the post-list column, cron or
1189 + * REST/MCP. With no handlers registered `do_shortcode()` returns its input
1190 + * unchanged, so the raw shortcode source was scored as if it were the
1191 + * page's prose: `[ct_section`, `id="section-1"` and the rest counted toward
1192 + * the word count, while the actual copy sitting in `text="..."` attributes
1193 + * was never counted at all (#776).
1194 + *
1195 + * `strip_shortcodes()` is no help either — it also only knows registered
1196 + * shortcodes, so it leaves the same text untouched.
1197 + *
1198 + * Reading the stored tree directly is what every other builder here already
1199 + * does, and it matches the class's stated design: no render engine, no
1200 + * dependency on load order, safe during a bulk run.
1201 + *
1202 + * Parsing unconditionally, rather than rendering when Oxygen happens to be
1203 + * loaded and parsing otherwise, is deliberate. It makes the extracted text
1204 + * the same in every context, so the score in the editor matches the score
1205 + * from a bulk run or from MCP. The old code produced whichever of the two
1206 + * the request happened to allow, which is why the same post could report
1207 + * two different word counts depending on how it was asked.
1208 + *
1209 + * The trade-off is that rendered output (resolved images, links, anything
1210 + * Oxygen pulls in from a reusable part) is no longer reflected here. For
1211 + * what this text feeds — word count, content scoring, meta-description
1212 + * fallbacks and schema text — that markup was never the point, and counting
1213 + * it only when the builder happened to be booted was the bug.
1214 + *
1215 + * @since 2.10.0
1216 + *
1217 + * @param string $stored Raw shortcode source.
1218 + * @return string Extracted text.
1219 + */
1220 + private static function text_from_shortcodes(string $stored): string {
1221 + // Oxygen stores each element's settings as a JSON blob in `ct_options`.
1222 + // It is configuration, never copy, and it contains braces and brackets
1223 + // that would otherwise confuse the tag scan below, so it goes first.
1224 + //
1225 + // The blob is matched as a balanced JSON object, not as "up to the
1226 + // next quote". Oxygen wraps it in single quotes but does not escape
1227 + // an apostrophe inside it (`"nicename":"Bob's Plumbing"`), so the
1228 + // quote-to-quote match stopped mid-value and the rest of the blob,
1229 + // `s Plumbing"}'` and all, was left in the tag and leaked into the
1230 + // text. Strings inside the object are skipped whole, so neither a quote
1231 + // nor a brace inside a value can end the match early.
1232 + $source = (string) preg_replace(
1233 + '/\sct_options\s*=\s*\'(?<obj>\{(?:[^{}"]++|"(?:[^"\\\\]|\\\\.)*+"|(?&obj))*+\})\'/s',
1234 + '',
1235 + $stored
1236 + );
1237 +
1238 + // Anything not shaped like Oxygen's JSON blob keeps the old,
1239 + // quote-delimited strip.
1240 + $source = (string) preg_replace(
1241 + '/\sct_options\s*=\s*(["\']).*?\1/s',
1242 + '',
1243 + $source
1244 + );
1245 +
1246 + $attributes = implode('|', array_map(
1247 + static fn(string $name): string => preg_quote($name, '/'),
1248 + self::SHORTCODE_TEXT_ATTRIBUTES
1249 + ));
1250 +
1251 + // Replace each shortcode tag with whatever readable copy its attributes
1252 + // carry. Text between tags is left exactly where it is, so the result
1253 + // keeps the page's reading order rather than hoisting all the headings
1254 + // to the front.
1255 + // The attribute blob is matched quote-aware rather than as "anything up
1256 + // to the first `]`". Oxygen copy contains brackets often enough to
1257 + // matter — "Best tools [2026]", "[Updated] our policy" — and a naive
1258 + // scan ends the tag inside the `text` attribute, dropping the copy
1259 + // before the bracket and leaking the stray `"]` after it into the
1260 + // prose. Which is this bug's own failure mode: the wrong text scored.
1261 + //
1262 + // A tag name must start with a letter or underscore. `[2026]` is not a
1263 + // shortcode anyone can register, and scanning it as one dropped the
1264 + // year out of "Best tools [2026]".
1265 + $text = (string) preg_replace_callback(
1266 + '/\[\/?[a-zA-Z_][a-zA-Z0-9_-]*((?:[^\]"\']|"[^"]*"|\'[^\']*\')*)\]/',
1267 + static function (array $matches) use ($attributes): string {
1268 + if ('' === trim($matches[1])) {
1269 + return ' ';
1270 + }
1271 +
1272 + if (!preg_match_all(
1273 + '/\b(' . $attributes . ')\s*=\s*(["\'])(.*?)\2/s',
1274 + $matches[1],
1275 + $found,
1276 + PREG_SET_ORDER
1277 + )) {
1278 + return ' ';
1279 + }
1280 +
1281 + $parts = [];
1282 + foreach ($found as $attribute) {
1283 + $value = trim($attribute[3]);
1284 +
1285 + // An attribute holding markup or a JSON fragment is
1286 + // configuration that happens to share a name with a copy
1287 + // field, not something a visitor reads.
1288 + if ('' === $value || preg_match('/^[\[{<]/', $value)) {
1289 + continue;
1290 + }
1291 +
1292 + $parts[] = $value;
1293 + }
1294 +
1295 + return empty($parts) ? ' ' : ' ' . implode(' ', $parts) . ' ';
1296 + },
1297 + $source
1298 + );
1299 +
1300 + // Oxygen escapes square brackets in an element's copy before writing
1301 + // it between the tags, so that "Best tools [2026]" cannot be mistaken
1302 + // for a shortcode (`oxygen_vsb_filter_shortcode_content_encode()`).
1303 + // Decoded only now, after the tag scan, for the same reason; left
1304 + // encoded, the placeholders were scored as words of their own.
1305 + $text = str_replace(
1306 + ['_OXY_OPENING_BRACKET_', '_OXY_CLOSING_BRACKET_'],
1307 + ['[', ']'],
1308 + $text
1309 + );
1310 +
1311 + // Entities are stored encoded in attributes (&amp;, &#8217;), and would
1312 + // otherwise be counted as words.
1313 + $text = html_entity_decode($text, ENT_QUOTES | ENT_HTML5, 'UTF-8');
1314 +
1315 + return trim((string) preg_replace('/\s+/u', ' ', $text));
1002 1316 }
1003 1317
1004 1318 /**
1005 1319 * A node's children, whether it stores them as an array or an object.