PluginProbe
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO / 2.10.0
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO v2.10.0
2.10.0 2.9.0 2.8.0 2.7.0 2.6.0 2.5.0 2.4.0 2.3.0 2.2.0 2.1.1 2.1.0 2.0.2 2.0.1 2.0.0 1.32.0 1.31.0 1.30.0 1.29.0 1.28.0 1.27.0 1.26.0 1.25.0 trunk 1.0.0 1.0.1 All 51 releases
← All changes | includes/seo/class-builder-content.php +739 -36 2.3.0 → 2.10.0 View file →
@@ -50,9 +50,30 @@
50 50 */
51 51 private const BUILDER_META_KEYS = [
52 52 '_breakdance_data', // Oxygen 6+ / Breakdance
53 53 '_oxygen_data', // Oxygen (earlier releases)
54 - 'ct_builder_shortcodes', // Oxygen classic
54 + // Oxygen classic. 4.x writes the tree as JSON to `ct_builder_json`
55 + // while still keeping `ct_builder_shortcodes`. A post carrying only
56 + // the JSON key used to match no key at all and fall through to an
57 + // empty `post_content`, which reads as a one-word page (#776).
58 + //
59 + // Oxygen 4.8.3 then renamed every `ct_*` post meta key to `_ct_*`
60 + // (`oxygen_vsb_update_4_8_3()` runs `oxy_prefix_meta_keys()` on
61 + // upgrade, and `oxy_get_post_meta()` only reads the prefixed name
62 + // from then on). A current Oxygen classic site therefore has only the
63 + // underscored keys, which nothing here listed, so every one of its
64 + // pages resolved as empty. The prefixed keys come first because they
65 + // are what Oxygen itself reads; the bare ones cover a site that has
66 + // not run the migration (or reverted it with `?unprefix_meta`).
67 + //
68 + // These four are not read in this loop: from_oxygen_classic() pairs
69 + // each JSON key with its shortcode sibling so the two forms can be
70 + // compared. They are listed here because this list is also what the
71 + // word-count index watches and what FAQ detection scans.
72 + '_ct_builder_json', // Oxygen classic 4.8.3+ (JSON tree)
73 + 'ct_builder_json', // Oxygen classic 4.0-4.8.2 (JSON tree)
74 + '_ct_builder_shortcodes', // Oxygen classic 4.8.3+ (shortcode tree)
75 + 'ct_builder_shortcodes', // Oxygen classic < 4.8.3 (shortcode tree)
55 76 '_elementor_data', // Elementor
56 77 // Beaver Builder. Published layout first: `_fl_builder_draft` holds
57 78 // unsaved changes and would score content the visitor cannot see.
58 79 // Both are arrays of stdClass nodes, which is why the walker below
@@ -97,8 +118,41 @@
97 118 */
98 119 private const BRICKS_COMPONENTS_OPTION = 'bricks_components';
99 120
100 121 /**
122 + * Bricks' element that renders the post's own `post_content`.
123 + *
124 + * A Bricks page normally discards `post_content` entirely, which is why
125 + * anything left there is invisible. Dropping this element onto the canvas
126 + * is the one way an author puts it back on the page, so its presence flips
127 + * `post_content` from stale leftovers to content the visitor reads.
128 + *
129 + * @since 2.3.1
130 + * @var string
131 + */
132 + private const BRICKS_POST_CONTENT_ELEMENT = 'post-content';
133 +
134 + /**
135 + * Resolved Bricks trees for this request, keyed by post ID.
136 + *
137 + * Rendering one page asks for the tree about twenty times — every
138 + * description, every schema node, the FAQ guard — and resolving it is not
139 + * free. `bricks_content_source()` clears `Bricks\Database::$active_templates`
140 + * before asking Bricks which content template applies, which defeats
141 + * Bricks' own early-return and re-runs its whole template-condition engine;
142 + * `expand_bricks_components()` then walks the tree again. Measured on a
143 + * Bricks page with no content of its own, that was ten full runs of the
144 + * rules engine per request.
145 + *
146 + * Per-request only, and only ever read back within one page render — a
147 + * request that writes Bricks content does not also render it.
148 + *
149 + * @since 2.3.1
150 + * @var array<int,array<int,mixed>>
151 + */
152 + private static array $bricks_trees = [];
153 +
154 + /**
101 155 * JSON keys whose values are user-visible text.
102 156 *
103 157 * Builder trees mix content with configuration, so a blind string sweep
104 158 * would count CSS classes and option slugs as words. Matching on the key
@@ -109,11 +163,38 @@
109 163 private const CONTENT_KEYS = [
110 164 'text', 'title', 'subtitle', 'heading', 'subheading', 'content',
111 165 'description', 'caption', 'excerpt', 'label', 'value', 'html',
112 166 'editor', 'quote', 'answer', 'question', 'body', 'button_text',
167 + // Oxygen classic keeps an element's copy in `options.ct_content`
168 + // (headline, text block, rich text, link and button labels). It is the
169 + // field Oxygen's own serializer moves between the tags when it writes
170 + // shortcodes (`parse_components_tree()`), and the one Relevanssi and
171 + // Oxygen's WPML integration read. Missing from this list, the walker
172 + // kept only copy that happened to contain markup: a page of plain
173 + // headings and paragraphs lost almost all of its words.
174 + 'ct_content',
175 + // Oxygen's composite elements keep their copy under `options.original`
176 + // instead, one key per field. Taken from the list Oxygen itself treats
177 + // as text when it serializes (`$options_to_encode`); the numeric price
178 + // fields and the progress bar's right-hand percentage are left out.
179 + 'testimonial_text', 'testimonial_author', 'testimonial_author_info',
180 + 'icon_box_heading', 'icon_box_text',
181 + 'pricing_box_package_title', 'pricing_box_package_subtitle', 'pricing_box_content',
182 + 'progress_bar_left_text',
113 183 ];
114 184
115 185 /**
186 + * Oxygen classic's storage generations, as JSON key => shortcode key.
187 + *
188 + * @since 2.10.0
189 + * @var array<string,string>
190 + */
191 + private const OXYGEN_CLASSIC_KEYS = [
192 + '_ct_builder_json' => '_ct_builder_shortcodes',
193 + 'ct_builder_json' => 'ct_builder_shortcodes',
194 + ];
195 +
196 + /**
116 197 * JSON keys whose values hold a link destination.
117 198 *
118 199 * Builders store a link's destination in a structured field separate from
119 200 * its label, either as a bare URL string or as a `{ url: … }` object.
@@ -126,8 +207,71 @@
126 207 'link', 'url', 'href', 'link_url', 'button_link', 'permalink', 'link_to',
127 208 ];
128 209
129 210 /**
211 + * JSON keys whose values hold an embedded video's source.
212 + *
213 + * A builder's video widget keeps its destination in a provider-specific
214 + * field — Elementor picks `youtube_url`, `vimeo_url`, `dailymotion_url` or
215 + * `hosted_url` according to the chosen source type — none of which is a
216 + * link field or a content field, so a video on a builder page reached the
217 + * analyzers as nothing at all.
218 + *
219 + * These are deliberately kept out of URL_KEYS. A video is an embed, not an
220 + * outbound link: rendering one as `<a href>` would add a spurious external
221 + * link to every page carrying a video and skew the link counts. They are
222 + * reconstructed as `<iframe>`/`<video>` instead, which the video detector
223 + * recognises and the link and image counters ignore.
224 + *
225 + * @since 2.3.1
226 + * @var string[]
227 + */
228 + private const VIDEO_KEYS = [
229 + 'youtube_url', 'vimeo_url', 'dailymotion_url', 'videopress_url',
230 + 'hosted_url', 'video_url', 'video_src', 'video_link',
231 + ];
232 +
233 + /**
234 + * File extensions that mean a video source is a file, not a provider page.
235 + *
236 + * @since 2.3.1
237 + * @var string[]
238 + */
239 + private const VIDEO_FILE_EXTENSIONS = ['mp4', 'webm', 'ogv', 'mov', 'm4v'];
240 +
241 + /**
242 + * Source keys to trust for a declared video source type.
243 + *
244 + * A widget keeps one field per provider and does not clear the others when
245 + * the author switches source: an Elementor video moved from YouTube to Self
246 + * Hosted still carries the earlier `youtube_url`. Reading whichever key
247 + * turns up first then emits the video the author replaced. The widget says
248 + * which one it is actually playing, so that is read first and the flat key
249 + * sweep is only the fallback for a builder that declares nothing.
250 + *
251 + * @since 2.3.1
252 + * @var array<string,string[]>
253 + */
254 + private const VIDEO_KEYS_BY_TYPE = [
255 + 'youtube' => ['youtube_url'],
256 + 'vimeo' => ['vimeo_url'],
257 + 'dailymotion' => ['dailymotion_url'],
258 + 'videopress' => ['videopress_url'],
259 + 'hosted' => ['hosted_url', 'video_url', 'video_src', 'video_link'],
260 + 'media' => ['hosted_url', 'video_url', 'video_src', 'video_link'],
261 + 'file' => ['hosted_url', 'video_url', 'video_src', 'video_link'],
262 + 'self_hosted' => ['hosted_url', 'video_url', 'video_src', 'video_link'],
263 + ];
264 +
265 + /**
266 + * Keys a builder uses to name which video source a widget is playing.
267 + *
268 + * @since 2.3.1
269 + * @var string[]
270 + */
271 + private const VIDEO_TYPE_KEYS = ['video_type', 'videotype', 'video_source', 'source_type'];
272 +
273 + /**
130 274 * JSON keys whose values hold an image, as a URL string or `{ url, alt }`.
131 275 *
132 276 * @var string[]
133 277 */
@@ -144,9 +288,13 @@
144 288 *
145 289 * @var string[]
146 290 */
147 291 private const HEADING_TAG_KEYS = [
148 - 'header_size', 'heading_tag', 'html_tag', 'title_tag', 'tag', 'level', 'size',
292 + // Lower-cased on both sides of the comparison, so `headingtag` is the
293 + // camelCase `headingTag` Bricks uses throughout its own controls and
294 + // which ThinkRank's Bricks elements declare. Without it their section
295 + // headings counted as body copy and never reached heading structure.
296 + 'header_size', 'heading_tag', 'headingtag', 'html_tag', 'title_tag', 'tag', 'level', 'size',
149 297 ];
150 298
151 299 /**
152 300 * Keys whose value is alternative text for a sibling image.
@@ -161,12 +309,231 @@
161 309 * @param \WP_Post $post Post being analyzed.
162 310 * @return string HTML/text to analyze.
163 311 */
164 312 public static function resolve(\WP_Post $post): string {
165 - return self::resolve_markup((string) $post->post_content, $post);
313 + $raw = (string) $post->post_content;
314 +
315 + // A page built in Gutenberg and then switched to Bricks keeps its old
316 + // blocks in `post_content` forever — Bricks never clears them, and
317 + // never renders them either. Resolving that first meant the stale draft
318 + // beat the tree the visitor actually reads, and it did not stop at the
319 + // score: the same string becomes the meta description, og:description,
320 + // twitter:description and the schema description. Starting from nothing
321 + // sends the resolution straight to Bricks' storage, which is where this
322 + // page's words are (#651).
323 + //
324 + // Only for the stored path. `resolve_markup()` is also called with live
325 + // editor content, and the Bricks panel's resolver reads the canvas —
326 + // discarding that would replace what the author is typing with the last
327 + // save.
328 + if (self::bricks_supersedes_post_content((int) $post->ID)) {
329 + $raw = '';
330 + }
331 +
332 + return self::resolve_markup($raw, $post);
166 333 }
167 334
168 335 /**
336 + * Whether Bricks renders this post and throws its `post_content` away.
337 + *
338 + * True means anything still stored in `post_content` is invisible: it is
339 + * not on the page, so it must not be scored, described or published as
340 + * structured data. False covers both a post Bricks does not own and a
341 + * Bricks page that puts `post_content` back with a Post Content element.
342 + *
343 + * @since 2.3.1
344 + *
345 + * @param int $post_id Post being resolved.
346 + * @return bool
347 + */
348 + public static function bricks_supersedes_post_content(int $post_id): bool {
349 + $tree = self::bricks_tree($post_id);
350 +
351 + if (empty($tree)) {
352 + return false;
353 + }
354 +
355 + foreach ($tree as $element) {
356 + if (is_array($element)
357 + && self::BRICKS_POST_CONTENT_ELEMENT === ($element['name'] ?? null)
358 + ) {
359 + return false;
360 + }
361 + }
362 +
363 + return !self::bricks_tree_prints_post_content($tree);
364 + }
365 +
366 + /**
367 + * Whether a Bricks tree prints the body through a dynamic-data tag.
368 + *
369 + * The Post Content element is not the only way back onto the page: Bricks'
370 + * `{post_content}` tag renders the same thing from inside an ordinary text
371 + * element, and a single-post template written that way is a common shape.
372 + * Missing it would mean the post's real body is discarded everywhere —
373 + * scoring, the meta/og/twitter descriptions, the schema description — for a
374 + * page that is displaying it.
375 + *
376 + * Matched over the encoded tree rather than per setting, because the tag can
377 + * sit in any string field of any element and Bricks allows modifiers after
378 + * the name (`{post_content:...}`).
379 + *
380 + * @since 2.3.1
381 + *
382 + * @param array $tree Bricks element tree.
383 + * @return bool
384 + */
385 + private static function bricks_tree_prints_post_content(array $tree): bool {
386 + $encoded = wp_json_encode($tree);
387 +
388 + return is_string($encoded) && false !== stripos($encoded, '{post_content');
389 + }
390 +
391 + /**
392 + * The post's content as the visitor actually receives it.
393 + *
394 + * `post_content` for everything except a Bricks page that discards it, and
395 + * there the Bricks tree's text. Descriptions are derived from a post's body
396 + * in half a dozen places; every one of them wants this rather than the raw
397 + * column (#651).
398 + *
399 + * @since 2.3.1
400 + *
401 + * @param \WP_Post $post Post being described.
402 + * @return string
403 + */
404 + public static function visible_content(\WP_Post $post): string {
405 + $superseding = self::superseding_content($post);
406 +
407 + return '' !== $superseding ? $superseding : (string) $post->post_content;
408 + }
409 +
410 + /**
411 + * Replacement body text for a post whose `post_content` does not render.
412 + *
413 + * Empty for every ordinary post, which is what makes this safe to call from
414 + * paths that already handle excerpts their own way: they keep that handling
415 + * and only a Bricks page is diverted.
416 + *
417 + * @since 2.3.1
418 + *
419 + * @param \WP_Post $post Post being described.
420 + * @return string Visible body text, or '' when `post_content` is fine.
421 + */
422 + public static function superseding_content(\WP_Post $post): string {
423 + if (!self::bricks_supersedes_post_content((int) $post->ID)) {
424 + return '';
425 + }
426 +
427 + $bricks = self::from_bricks((int) $post->ID);
428 +
429 + return self::is_blank($bricks) ? '' : $bricks;
430 + }
431 +
432 + /**
433 + * Body text to derive a description from, when the usual source is wrong.
434 + *
435 + * A hand-written excerpt is the author's own summary and is correct however
436 + * the page is built, so it yields '' here and the caller's normal
437 + * `get_the_excerpt()` path keeps it. Only a Bricks page with no excerpt —
438 + * where core would derive one from discarded `post_content` — gets diverted.
439 + *
440 + * @since 2.3.1
441 + *
442 + * @param \WP_Post $post Post being described.
443 + * @return string Text to summarize, or '' to leave the caller's path alone.
444 + */
445 + public static function superseding_excerpt_source(\WP_Post $post): string {
446 + if ('' !== trim((string) $post->post_excerpt)) {
447 + return '';
448 + }
449 +
450 + return self::superseding_content($post);
451 + }
452 +
453 + /**
454 + * The Bricks element tree that renders for a post.
455 + *
456 + * Public because what Bricks puts on the page is not only a scoring
457 + * question: the schema graph has to know whether a Bricks element already
458 + * publishes the page's FAQ before adding one of its own (#649, #650).
459 + *
460 + * Flat, in Bricks' own storage shape — `expand_bricks_components()`
461 + * appends component definitions to the same list rather than nesting them,
462 + * so one `foreach` reaches every element.
463 + *
464 + * @since 2.3.1
465 + *
466 + * @param int $post_id Post being resolved.
467 + * @return array<int,mixed> Elements, or [] when Bricks renders nothing here.
468 + */
469 + /**
470 + * The builder meta keys, for callers that need to inspect the raw storage
471 + * rather than the text extracted from it.
472 + *
473 + * The SEO Analyzer reads these to answer "is there a ThinkRank FAQ element
474 + * on this post?", which is a question about the stored tree, not about the
475 + * words in it (#686).
476 + *
477 + * @since 2.7.0
478 + * @return string[]
479 + */
480 + public static function builder_meta_keys(): array {
481 + return self::BUILDER_META_KEYS;
482 + }
483 +
484 + public static function bricks_tree(int $post_id): array {
485 + if (array_key_exists($post_id, self::$bricks_trees)) {
486 + return self::$bricks_trees[$post_id];
487 + }
488 +
489 + self::$bricks_trees[$post_id] = self::resolve_bricks_tree($post_id);
490 +
491 + return self::$bricks_trees[$post_id];
492 + }
493 +
494 + /**
495 + * Discard the resolved-tree memo. Test seam.
496 + *
497 + * @since 2.3.1
498 + * @return void
499 + */
500 + public static function flush_bricks_cache(): void {
501 + self::$bricks_trees = [];
502 + }
503 +
504 + /**
505 + * Read and resolve a post's Bricks tree, ignoring the memo.
506 + *
507 + * @since 2.3.1
508 + *
509 + * @param int $post_id Post being resolved.
510 + * @return array<int,mixed>
511 + */
512 + private static function resolve_bricks_tree(int $post_id): array {
513 + if (!self::bricks_owns_post($post_id)) {
514 + return [];
515 + }
516 +
517 + $source = self::bricks_content_source($post_id);
518 + if (!$source) {
519 + return [];
520 + }
521 +
522 + $stored = get_post_meta($source, self::bricks_meta_key(), true);
523 +
524 + if (is_string($stored)) {
525 + $stored = '' === trim($stored) ? null : json_decode($stored, true);
526 + }
527 +
528 + if (!is_array($stored) || empty($stored)) {
529 + return [];
530 + }
531 +
532 + return self::expand_bricks_components($stored);
533 + }
534 +
535 + /**
169 536 * Resolve an arbitrary chunk of editor markup for the given post.
170 537 *
171 538 * The editor sends its live content to the scorer so an author sees their
172 539 * unsaved edits reflected. On a builder page that live string is the raw
@@ -319,31 +686,16 @@
319 686 * @param int $post_id Post being resolved.
320 687 * @return string Extracted text, or '' when Bricks has nothing for it.
321 688 */
322 689 private static function from_bricks(int $post_id): string {
323 - if (!self::bricks_owns_post($post_id)) {
324 - return '';
325 - }
690 + $tree = self::bricks_tree($post_id);
326 691
327 - $source = self::bricks_content_source($post_id);
328 - if (!$source) {
692 + if (empty($tree)) {
329 693 return '';
330 694 }
331 695
332 - $stored = get_post_meta($source, self::bricks_meta_key(), true);
333 -
334 - if (is_string($stored)) {
335 - $stored = '' === trim($stored) ? null : json_decode($stored, true);
336 - }
337 -
338 - if (!is_array($stored) || empty($stored)) {
339 - return '';
340 - }
341 -
342 696 return self::strip_bricks_dynamic_tags(
343 - self::text_from_tree(
344 - self::without_bricks_element_labels(self::expand_bricks_components($stored))
345 - )
697 + self::text_from_tree(self::without_bricks_element_labels($tree))
346 698 );
347 699 }
348 700
349 701 /**
@@ -668,8 +1020,21 @@
668 1020 return $bricks;
669 1021 }
670 1022
671 1023 foreach (self::BUILDER_META_KEYS as $key) {
1024 + if (self::is_oxygen_classic_key($key)) {
1025 + // Resolved as a pair, once, at the first of its keys.
1026 + if ('_ct_builder_json' !== $key) {
1027 + continue;
1028 + }
1029 +
1030 + $oxygen = self::from_oxygen_classic($post_id);
1031 + if (!self::is_blank($oxygen)) {
1032 + return $oxygen;
1033 + }
1034 + continue;
1035 + }
1036 +
672 1037 $stored = get_post_meta($post_id, $key, true);
673 1038
674 1039 if (is_string($stored) && '' !== trim($stored)) {
675 1040 $decoded = json_decode($stored, true);
@@ -682,20 +1047,8 @@
682 1047 }
683 1048 continue;
684 1049 }
685 1050
686 - // Shortcode tree (Oxygen classic).
687 - if (strpos($stored, '[') !== false && function_exists('do_shortcode')) {
688 - try {
689 - $rendered = do_shortcode($stored);
690 - } catch (\Throwable $e) {
691 - $rendered = $stored;
692 - }
693 - if (!self::is_blank($rendered)) {
694 - return $rendered;
695 - }
696 - }
697 -
698 1051 continue;
699 1052 }
700 1053
701 1054 // Some builders store an already-decoded tree — an array for most,
@@ -712,8 +1065,258 @@
712 1065 return '';
713 1066 }
714 1067
715 1068 /**
1069 + * Whether a meta key is one of Oxygen classic's storage keys.
1070 + *
1071 + * @since 2.10.0
1072 + *
1073 + * @param string $key Meta key.
1074 + * @return bool
1075 + */
1076 + private static function is_oxygen_classic_key(string $key): bool {
1077 + return isset(self::OXYGEN_CLASSIC_KEYS[$key]) || in_array($key, self::OXYGEN_CLASSIC_KEYS, true);
1078 + }
1079 +
1080 + /**
1081 + * Text of an Oxygen classic page, from whichever stored form holds more.
1082 + *
1083 + * Oxygen 4.x keeps the same tree twice: as JSON, and as the shortcodes it
1084 + * used before 4.0. The JSON is preferred because it carries copy the
1085 + * shortcode form hides (a composite element's text is base64-encoded
1086 + * inside `ct_options`, which is configuration and stripped). It is not
1087 + * trusted blindly, though. Reading `ct_builder_json` first once meant a
1088 + * key missing from CONTENT_KEYS silently threw the page away while the
1089 + * shortcode copy sat unread next to it, because a non-empty JSON result
1090 + * stopped the search. Comparing the two means the next such gap costs
1091 + * nothing: the richer form wins.
1092 + *
1093 + * A generation is only read as a pair. The prefixed keys are what Oxygen
1094 + * 4.8.3+ reads, so an unprefixed leftover next to them is stale.
1095 + *
1096 + * @since 2.10.0
1097 + *
1098 + * @param int $post_id Post ID.
1099 + * @return string Extracted text, or '' when Oxygen classic stored nothing.
1100 + */
1101 + private static function from_oxygen_classic(int $post_id): string {
1102 + foreach (self::OXYGEN_CLASSIC_KEYS as $json_key => $shortcode_key) {
1103 + $json = get_post_meta($post_id, $json_key, true);
1104 + $shortcodes = get_post_meta($post_id, $shortcode_key, true);
1105 +
1106 + $from_json = '';
1107 + if (is_string($json) && '' !== trim($json)) {
1108 + $decoded = json_decode($json, true);
1109 + if (is_array($decoded)) {
1110 + // `[oxygen data="..."]` is a dynamic-data placeholder
1111 + // Oxygen fills at render time. The shortcode path drops it
1112 + // with every other tag, so it goes here too or the two
1113 + // forms would disagree on the same page.
1114 + $from_json = (string) preg_replace(
1115 + '/\[oxygen\b[^\]]*\]/i',
1116 + ' ',
1117 + self::text_from_tree($decoded)
1118 + );
1119 + }
1120 + }
1121 +
1122 + $from_shortcodes = '';
1123 + if (is_string($shortcodes) && strpos($shortcodes, '[') !== false) {
1124 + $from_shortcodes = self::text_from_shortcodes($shortcodes);
1125 + }
1126 +
1127 + if (self::is_blank($from_json) && self::is_blank($from_shortcodes)) {
1128 + continue;
1129 + }
1130 +
1131 + return self::visible_word_count($from_json) >= self::visible_word_count($from_shortcodes)
1132 + ? $from_json
1133 + : $from_shortcodes;
1134 + }
1135 +
1136 + return '';
1137 + }
1138 +
1139 + /**
1140 + * Rough count of the words a visitor would read in extracted text.
1141 + *
1142 + * Only used to compare two extractions of the same page, so it needs to
1143 + * be consistent rather than locale-exact.
1144 + *
1145 + * @since 2.10.0
1146 + *
1147 + * @param string $text Extracted text or markup.
1148 + * @return int
1149 + */
1150 + private static function visible_word_count(string $text): int {
1151 + $plain = trim((string) preg_replace('/\s+/u', ' ', wp_strip_all_tags($text)));
1152 +
1153 + return '' === $plain ? 0 : count(explode(' ', $plain));
1154 + }
1155 +
1156 + /**
1157 + * Shortcode attributes that carry copy a visitor reads.
1158 + *
1159 + * An allow-list, not a deny-list. Oxygen Classic tags carry far more
1160 + * attributes than they do copy — `id`, `class`, `selector`, `url`,
1161 + * `ct_options` and friends — and a deny-list silently admits every
1162 + * attribute a future builder release invents, which is how markup ends up
1163 + * being counted as prose.
1164 + *
1165 + * @var string[]
1166 + */
1167 + private const SHORTCODE_TEXT_ATTRIBUTES = [
1168 + 'text',
1169 + 'content',
1170 + 'heading',
1171 + 'title',
1172 + 'subtitle',
1173 + 'label',
1174 + 'caption',
1175 + 'description',
1176 + 'alt',
1177 + 'button_text',
1178 + 'link_text',
1179 + ];
1180 +
1181 + /**
1182 + * Extract readable text from a shortcode tree, without rendering it.
1183 + *
1184 + * Oxygen Classic is the only builder whose storage is shortcodes rather
1185 + * than JSON, and the previous implementation handed the string to
1186 + * `do_shortcode()`. That silently depends on Oxygen having registered its
1187 + * `ct_*` handlers in the current request — which it has on a front-end
1188 + * view, and has not during bulk analysis, the post-list column, cron or
1189 + * REST/MCP. With no handlers registered `do_shortcode()` returns its input
1190 + * unchanged, so the raw shortcode source was scored as if it were the
1191 + * page's prose: `[ct_section`, `id="section-1"` and the rest counted toward
1192 + * the word count, while the actual copy sitting in `text="..."` attributes
1193 + * was never counted at all (#776).
1194 + *
1195 + * `strip_shortcodes()` is no help either — it also only knows registered
1196 + * shortcodes, so it leaves the same text untouched.
1197 + *
1198 + * Reading the stored tree directly is what every other builder here already
1199 + * does, and it matches the class's stated design: no render engine, no
1200 + * dependency on load order, safe during a bulk run.
1201 + *
1202 + * Parsing unconditionally, rather than rendering when Oxygen happens to be
1203 + * loaded and parsing otherwise, is deliberate. It makes the extracted text
1204 + * the same in every context, so the score in the editor matches the score
1205 + * from a bulk run or from MCP. The old code produced whichever of the two
1206 + * the request happened to allow, which is why the same post could report
1207 + * two different word counts depending on how it was asked.
1208 + *
1209 + * The trade-off is that rendered output (resolved images, links, anything
1210 + * Oxygen pulls in from a reusable part) is no longer reflected here. For
1211 + * what this text feeds — word count, content scoring, meta-description
1212 + * fallbacks and schema text — that markup was never the point, and counting
1213 + * it only when the builder happened to be booted was the bug.
1214 + *
1215 + * @since 2.10.0
1216 + *
1217 + * @param string $stored Raw shortcode source.
1218 + * @return string Extracted text.
1219 + */
1220 + private static function text_from_shortcodes(string $stored): string {
1221 + // Oxygen stores each element's settings as a JSON blob in `ct_options`.
1222 + // It is configuration, never copy, and it contains braces and brackets
1223 + // that would otherwise confuse the tag scan below, so it goes first.
1224 + //
1225 + // The blob is matched as a balanced JSON object, not as "up to the
1226 + // next quote". Oxygen wraps it in single quotes but does not escape
1227 + // an apostrophe inside it (`"nicename":"Bob's Plumbing"`), so the
1228 + // quote-to-quote match stopped mid-value and the rest of the blob,
1229 + // `s Plumbing"}'` and all, was left in the tag and leaked into the
1230 + // text. Strings inside the object are skipped whole, so neither a quote
1231 + // nor a brace inside a value can end the match early.
1232 + $source = (string) preg_replace(
1233 + '/\sct_options\s*=\s*\'(?<obj>\{(?:[^{}"]++|"(?:[^"\\\\]|\\\\.)*+"|(?&obj))*+\})\'/s',
1234 + '',
1235 + $stored
1236 + );
1237 +
1238 + // Anything not shaped like Oxygen's JSON blob keeps the old,
1239 + // quote-delimited strip.
1240 + $source = (string) preg_replace(
1241 + '/\sct_options\s*=\s*(["\']).*?\1/s',
1242 + '',
1243 + $source
1244 + );
1245 +
1246 + $attributes = implode('|', array_map(
1247 + static fn(string $name): string => preg_quote($name, '/'),
1248 + self::SHORTCODE_TEXT_ATTRIBUTES
1249 + ));
1250 +
1251 + // Replace each shortcode tag with whatever readable copy its attributes
1252 + // carry. Text between tags is left exactly where it is, so the result
1253 + // keeps the page's reading order rather than hoisting all the headings
1254 + // to the front.
1255 + // The attribute blob is matched quote-aware rather than as "anything up
1256 + // to the first `]`". Oxygen copy contains brackets often enough to
1257 + // matter — "Best tools [2026]", "[Updated] our policy" — and a naive
1258 + // scan ends the tag inside the `text` attribute, dropping the copy
1259 + // before the bracket and leaking the stray `"]` after it into the
1260 + // prose. Which is this bug's own failure mode: the wrong text scored.
1261 + //
1262 + // A tag name must start with a letter or underscore. `[2026]` is not a
1263 + // shortcode anyone can register, and scanning it as one dropped the
1264 + // year out of "Best tools [2026]".
1265 + $text = (string) preg_replace_callback(
1266 + '/\[\/?[a-zA-Z_][a-zA-Z0-9_-]*((?:[^\]"\']|"[^"]*"|\'[^\']*\')*)\]/',
1267 + static function (array $matches) use ($attributes): string {
1268 + if ('' === trim($matches[1])) {
1269 + return ' ';
1270 + }
1271 +
1272 + if (!preg_match_all(
1273 + '/\b(' . $attributes . ')\s*=\s*(["\'])(.*?)\2/s',
1274 + $matches[1],
1275 + $found,
1276 + PREG_SET_ORDER
1277 + )) {
1278 + return ' ';
1279 + }
1280 +
1281 + $parts = [];
1282 + foreach ($found as $attribute) {
1283 + $value = trim($attribute[3]);
1284 +
1285 + // An attribute holding markup or a JSON fragment is
1286 + // configuration that happens to share a name with a copy
1287 + // field, not something a visitor reads.
1288 + if ('' === $value || preg_match('/^[\[{<]/', $value)) {
1289 + continue;
1290 + }
1291 +
1292 + $parts[] = $value;
1293 + }
1294 +
1295 + return empty($parts) ? ' ' : ' ' . implode(' ', $parts) . ' ';
1296 + },
1297 + $source
1298 + );
1299 +
1300 + // Oxygen escapes square brackets in an element's copy before writing
1301 + // it between the tags, so that "Best tools [2026]" cannot be mistaken
1302 + // for a shortcode (`oxygen_vsb_filter_shortcode_content_encode()`).
1303 + // Decoded only now, after the tag scan, for the same reason; left
1304 + // encoded, the placeholders were scored as words of their own.
1305 + $text = str_replace(
1306 + ['_OXY_OPENING_BRACKET_', '_OXY_CLOSING_BRACKET_'],
1307 + ['[', ']'],
1308 + $text
1309 + );
1310 +
1311 + // Entities are stored encoded in attributes (&amp;, &#8217;), and would
1312 + // otherwise be counted as words.
1313 + $text = html_entity_decode($text, ENT_QUOTES | ENT_HTML5, 'UTF-8');
1314 +
1315 + return trim((string) preg_replace('/\s+/u', ' ', $text));
1316 + }
1317 +
1318 + /**
716 1319 * A node's children, whether it stores them as an array or an object.
717 1320 *
718 1321 * The walker used to return immediately on `!is_array($node)`, so an
719 1322 * object node was dropped along with its entire subtree — silently, as
@@ -785,9 +1388,10 @@
785 1388 // would emit the same URL a second time as a bare link, and
786 1389 // would turn an image's own `url` field into a spurious <a>.
787 1390 if (is_string($child_key)
788 1391 && (in_array(strtolower($child_key), self::URL_KEYS, true)
789 - || in_array(strtolower($child_key), self::IMAGE_KEYS, true))
1392 + || in_array(strtolower($child_key), self::IMAGE_KEYS, true)
1393 + || in_array(strtolower($child_key), self::VIDEO_KEYS, true))
790 1394 ) {
791 1395 continue;
792 1396 }
793 1397
@@ -839,8 +1443,79 @@
839 1443 return implode("\n", $collected);
840 1444 }
841 1445
842 1446 /**
1447 + * The video source a node is actually playing, if any.
1448 + *
1449 + * @since 2.3.1
1450 + *
1451 + * @param array $node Builder node.
1452 + * @return string Video source, or '' when the node carries none.
1453 + */
1454 + private static function video_from(array $node): string {
1455 + foreach ($node as $key => $value) {
1456 + if (!is_string($key) || !is_string($value)) {
1457 + continue;
1458 + }
1459 +
1460 + if (!in_array(strtolower($key), self::VIDEO_TYPE_KEYS, true)) {
1461 + continue;
1462 + }
1463 +
1464 + $keys = self::VIDEO_KEYS_BY_TYPE[strtolower(trim($value))] ?? null;
1465 + if (null === $keys) {
1466 + continue;
1467 + }
1468 +
1469 + // A recognised video_type settles it, including when that
1470 + // provider's own field is empty. Falling through to the flat sweep
1471 + // there handed back whichever sibling key happened to come first in
1472 + // node order — the stale youtube_url left behind after switching
1473 + // the widget to a hosted file, which is exactly what keying on the
1474 + // declared type is meant to prevent.
1475 + $declared = self::url_from($node, $keys);
1476 +
1477 + return self::is_video_source($declared) ? $declared : '';
1478 + }
1479 +
1480 + $url = self::url_from($node, self::VIDEO_KEYS);
1481 +
1482 + return self::is_video_source($url) ? $url : '';
1483 + }
1484 +
1485 + /**
1486 + * Whether a value can be a video source.
1487 + *
1488 + * `looks_like_url()` also accepts `#anchor`, `mailto:` and `tel:`, which a
1489 + * link node may legitimately hold but a video cannot: `<iframe src="#top">`
1490 + * is not a video and would reach a video sitemap as one.
1491 + *
1492 + * @since 2.3.1
1493 + *
1494 + * @param string $url Candidate source.
1495 + * @return bool
1496 + */
1497 + private static function is_video_source(string $url): bool {
1498 + return '' !== $url
1499 + && (1 === preg_match('#^(https?:)?//#i', $url) || str_starts_with($url, '/'));
1500 + }
1501 +
1502 + /**
1503 + * Whether a video source points at a file rather than a provider page.
1504 + *
1505 + * @since 2.3.1
1506 + *
1507 + * @param string $url Video source.
1508 + * @return bool
1509 + */
1510 + private static function is_video_file(string $url): bool {
1511 + $path = (string) wp_parse_url($url, PHP_URL_PATH);
1512 + $ext = strtolower((string) pathinfo($path, PATHINFO_EXTENSION));
1513 +
1514 + return in_array($ext, self::VIDEO_FILE_EXTENSIONS, true);
1515 + }
1516 +
1517 + /**
843 1518 * Rebuild the HTML a single builder node represents, if any.
844 1519 *
845 1520 * Looks only at the node's own fields (plus one level of nesting, because
846 1521 * builders commonly wrap a destination as `{ url: … }`). Returns an empty
@@ -856,12 +1531,23 @@
856 1531 private static function markup_for_node(array $node, array &$consumed): string {
857 1532 $text = self::first_value($node, self::CONTENT_KEYS);
858 1533 $url = self::url_from($node, self::URL_KEYS);
859 1534 $image = self::image_from($node);
1535 + $video = self::video_from($node);
860 1536 $tag = self::heading_tag_from($node);
861 1537
862 1538 $parts = [];
863 1539
1540 + // Video: an embed shape rather than a link, so the video detector can
1541 + // see it while the link counters do not mistake it for an outbound
1542 + // link. A file source becomes <video src>, anything else an <iframe>,
1543 + // matching how the builder itself renders the two cases.
1544 + if ('' !== $video) {
1545 + $parts[] = self::is_video_file($video)
1546 + ? sprintf('<video src="%s"></video>', esc_url_raw($video))
1547 + : sprintf('<iframe src="%s"></iframe>', esc_url_raw($video));
1548 + }
1549 +
864 1550 // Image: alt text matters as much as the tag, since alt checks run over
865 1551 // whatever this returns.
866 1552 if ('' !== $image['url']) {
867 1553 $alt = '' !== $image['alt'] ? $image['alt'] : (string) self::first_value($node, self::ALT_KEYS);
@@ -1050,13 +1736,30 @@
1050 1736 return (bool) preg_match('#^(mailto:|tel:|\#)#i', $value);
1051 1737 }
1052 1738
1053 1739 /**
1054 - * Whether a value carries no readable text.
1740 + * Whether a value carries nothing worth analyzing.
1055 1741 *
1742 + * Readable text is the usual signal, but not the only one: a page can be
1743 + * made entirely of media. A builder section holding just a gallery
1744 + * reconstructs to `<img>` tags and one holding just a video widget to a
1745 + * single `<iframe>` — both strip to an empty string, so a text-only test
1746 + * discarded them here and the page fell through to the next builder key,
1747 + * then to the raw markup, and finally reported as having no content at all.
1748 + *
1749 + * Comments are dropped before the tag test: the raw markup this class falls
1750 + * back to on a builder page is unrendered block comments, which must stay
1751 + * blank rather than be mistaken for reconstructed media.
1752 + *
1056 1753 * @param string $value Candidate content.
1057 1754 * @return bool
1058 1755 */
1059 1756 private static function is_blank(string $value): bool {
1060 - return '' === trim(wp_strip_all_tags($value));
1757 + if ('' !== trim(wp_strip_all_tags($value))) {
1758 + return false;
1759 + }
1760 +
1761 + $without_comments = (string) preg_replace('~<!--.*?-->~s', '', $value);
1762 +
1763 + return 1 !== preg_match('~<(?:a|img|iframe|video|source)\b~i', $without_comments);
1061 1764 }
1062 1765 }