PluginProbe
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO / 2.14.2
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO v2.14.2
2.14.2 2.14.1 2.14.0 2.13.0 2.12.0 2.11.0 2.10.0 2.9.0 2.8.0 2.7.0 2.6.0 2.5.0 2.4.0 2.3.0 2.2.0 2.1.1 2.1.0 2.0.2 2.0.1 2.0.0 1.32.0 1.31.0 1.30.0 1.29.0 1.28.0 All 57 releases
← All changes | includes/ai/class-seo-score-calculator.php +728 -68 1.30.0 → 2.14.2 View file →
@@ -41,8 +41,41 @@
41 41 */
42 42 private Database $database;
43 43
44 44 /**
45 + * Memoised collected performance measurement, and whether it was resolved.
46 + *
47 + * Two factors read it and both may be asked for on every post in a list, so
48 + * the lookup happens once per calculator. `null` is a real answer here — the
49 + * separate flag keeps "not looked up yet" distinct from "nothing measured".
50 + *
51 + * @since 2.3.1
52 + * @var array|null
53 + */
54 + private ?array $measured_performance = null;
55 +
56 + /**
57 + * @since 2.3.1
58 + * @var bool
59 + */
60 + private bool $measured_performance_resolved = false;
61 +
62 + /**
63 + * Length bands the editor scores against, in characters.
64 + *
65 + * Public so every surface that judges a title or description — the editor
66 + * score and the Bulk Snippets problem filter — reads one set of numbers.
67 + * Before these existed the bands were literals inside the scoring methods,
68 + * and a second screen would have had to copy them and drift (#727).
69 + *
70 + * @since 2.8.0
71 + */
72 + public const TITLE_OPTIMAL_MIN = 35;
73 + public const TITLE_OPTIMAL_MAX = 60;
74 + public const DESCRIPTION_OPTIMAL_MIN = 120;
75 + public const DESCRIPTION_OPTIMAL_MAX = 160;
76 +
77 + /**
45 78 * 2025 SEO scoring factors (Q1 2025 Google Algorithm)
46 79 * Based on First Page Sage research and Google's latest updates
47 80 *
48 81 * @var array
@@ -107,8 +140,9 @@
107 140 'grade' => $result['grade'],
108 141 ]];
109 142 $result['keyword_checks'] = $this->analyze_keyword_checks($content_data, $metadata, $keywords);
110 143 }
144 + $result['keywords'] = $this->keyword_placements($content_data, $metadata, $keywords);
111 145
112 146 return $result;
113 147 }
114 148
@@ -218,8 +252,9 @@
218 252 $best['target_keyword'] = $best_keyword;
219 253 $best['target_keywords'] = $keywords;
220 254 $best['keyword_results'] = $per_keyword;
221 255 $best['keyword_checks'] = $this->analyze_keyword_checks($content_data, $metadata, $keywords);
256 + $best['keywords'] = $this->keyword_placements($content_data, $metadata, $keywords);
222 257
223 258 return $best;
224 259 }
225 260
@@ -234,21 +269,26 @@
234 269 * @param string[] $keywords Target keywords.
235 270 * @return array<string,array{passed:bool,matched_keywords:string[]}>
236 271 */
237 272 private function analyze_keyword_checks(array $content_data, array $metadata, array $keywords): array {
238 - $title = strtolower((string) ($metadata['title'] ?? $content_data['title'] ?? ''));
239 - $description = strtolower((string) ($metadata['description'] ?? ''));
240 - $content = strtolower(wp_strip_all_tags((string) ($content_data['content'] ?? '')));
273 + $title = self::lower((string) ($metadata['title'] ?? $content_data['title'] ?? ''));
274 + $description = self::lower((string) ($metadata['description'] ?? ''));
275 + $content = self::lower(self::plain_text((string) ($content_data['content'] ?? '')));
241 276
242 277 $alts = '';
243 278 foreach ((array) ($content_data['images'] ?? []) as $image) {
244 - $alts .= ' ' . strtolower((string) ($image['alt'] ?? ''));
279 + $alts .= ' ' . self::lower((string) ($image['alt'] ?? ''));
245 280 }
246 281
247 - // Build a searchable slug haystack from the URL path (hyphens/underscores
248 - // become spaces so multi-word keywords can match).
249 - $path = (string) (wp_parse_url((string) ($content_data['url'] ?? ''), PHP_URL_PATH) ?? '');
250 - $slug = strtolower(str_replace(['-', '_', '/'], ' ', trim($path, '/')));
282 + // Build a searchable slug haystack from the post's OWN slug — never the
283 + // full URL path. The path carries ancestors, category bases and date
284 + // segments, so a child of /clinical-trials/ reported "keyword in slug"
285 + // for a page actually slugged `contact-us`. It also breaks the other
286 + // way: an unpublished post has no pretty permalink (get_permalink()
287 + // returns ?p=123), so the path held no slug at all and every draft
288 + // scored "no match" until it was published. Hyphens/underscores become
289 + // spaces so multi-word keywords can match.
290 + $slug = self::lower(self::slug_haystack($content_data));
251 291
252 292 $haystacks = [
253 293 'title' => trim($title),
254 294 'meta_description' => trim($description),
@@ -260,10 +300,9 @@
260 300 $checks = [];
261 301 foreach ($haystacks as $location => $haystack) {
262 302 $matched = [];
263 303 foreach ($keywords as $keyword) {
264 - $needle = strtolower(trim($keyword));
265 - if ($needle !== '' && $haystack !== '' && strpos($haystack, $needle) !== false) {
304 + if ($this->keyword_matches($haystack, self::lower(trim($keyword)))) {
266 305 $matched[] = $keyword;
267 306 }
268 307 }
269 308 $checks[$location] = [
@@ -275,8 +314,288 @@
275 314 return $checks;
276 315 }
277 316
278 317 /**
318 + * The keyword placements the editor draws one gauge segment for, in the
319 + * order a reader meets them (#729).
320 + *
321 + * @since 2.11.0
322 + * @var string[]
323 + */
324 + public const PLACEMENTS = ['title', 'meta_description', 'slug', 'first_paragraph', 'subheading', 'content', 'image_alt'];
325 +
326 + /**
327 + * Characters of plain text read as the opening when the content has no
328 + * paragraph tag.
329 + *
330 + * @since 2.11.0
331 + */
332 + private const OPENING_CHARS = 300;
333 +
334 + /**
335 + * Where each focus keyword is placed, keyword by keyword (#729).
336 + *
337 + * analyze_keyword_checks() answers "does ANY keyword appear here" for
338 + * five places; this answers "where does THIS keyword appear" for seven,
339 + * so the editor can show each keyword's own gauge. Same matcher, so a
340 + * keyword counts as a word (not a fragment) and a keyword in a script
341 + * written without spaces (Thai, Chinese, Japanese) still matches.
342 + *
343 + * `where` names the heading or alt text that matched, so the editor can
344 + * say which one.
345 + *
346 + * @since 2.11.0
347 + *
348 + * @param array $content_data Content analysis data.
349 + * @param array $metadata Post metadata (title, description).
350 + * @param string[] $keywords Focus keywords.
351 + * @return array<int, array{keyword: string, passed: int, total: int, placements: array<string, array{passed: bool, where: string}>}>
352 + */
353 + public function keyword_placements(array $content_data, array $metadata, array $keywords): array {
354 + $html = (string) ($content_data['content'] ?? '');
355 + $plain = self::lower(self::plain_text($html));
356 +
357 + $headings = [];
358 + $source = isset($content_data['headings']) && is_array($content_data['headings']) ? $content_data['headings'] : $this->extract_headings($html);
359 + foreach ($source as $heading) {
360 + $text = trim((string) ($heading['text'] ?? ''));
361 + if ((int) ($heading['level'] ?? 0) >= 2 && '' !== $text) {
362 + $headings[] = $text;
363 + }
364 + }
365 +
366 + $alts = [];
367 + foreach ((array) ($content_data['images'] ?? []) as $image) {
368 + $alt = trim((string) ($image['alt'] ?? ''));
369 + if ('' !== $alt) {
370 + $alts[] = $alt;
371 + }
372 + }
373 +
374 + $single = [
375 + 'title' => self::lower((string) ($metadata['title'] ?? $content_data['title'] ?? '')),
376 + 'meta_description' => self::lower((string) ($metadata['description'] ?? '')),
377 + 'slug' => self::lower(self::slug_haystack($content_data)),
378 + 'first_paragraph' => self::lower(self::opening($html)),
379 + 'content' => $plain,
380 + ];
381 +
382 + $out = [];
383 + foreach ($keywords as $keyword) {
384 + $needle = self::lower(trim((string) $keyword));
385 + $placements = [];
386 +
387 + foreach (self::PLACEMENTS as $placement) {
388 + if ('subheading' === $placement || 'image_alt' === $placement) {
389 + $where = '';
390 + foreach ('subheading' === $placement ? $headings : $alts as $text) {
391 + if ($this->keyword_matches(self::lower($text), $needle)) {
392 + $where = $text;
393 + break;
394 + }
395 + }
396 + $placements[$placement] = ['passed' => '' !== $where, 'where' => $where];
397 + continue;
398 + }
399 +
400 + $placements[$placement] = ['passed' => $this->keyword_matches($single[$placement], $needle), 'where' => ''];
401 + }
402 +
403 + $out[] = [
404 + 'keyword' => (string) $keyword,
405 + 'passed' => count(array_filter(array_column($placements, 'passed'))),
406 + 'total' => count(self::PLACEMENTS),
407 + 'placements' => $placements,
408 + ];
409 + }
410 +
411 + return $out;
412 + }
413 +
414 + /**
415 + * The post's own slug as searchable text: hyphens and underscores become
416 + * spaces so a multi-word keyword can match, and a slug WordPress
417 + * percent-encoded (Thai, Cyrillic, Chinese) is decoded, or it could never
418 + * match a keyword typed in that script.
419 + *
420 + * Never the full URL path: the path carries ancestors, category bases and
421 + * date segments, so a child of /clinical-trials/ reported "keyword in slug"
422 + * for a page actually slugged `contact-us`. An unpublished post has no
423 + * pretty permalink either, so the path held no slug at all.
424 + *
425 + * @param array $content_data Content analysis data.
426 + * @return string
427 + */
428 + private static function slug_haystack(array $content_data): string {
429 + $slug = (string) ($content_data['slug'] ?? '');
430 + if ('' === $slug) {
431 + // Draft with no slug assigned yet: score what WordPress would
432 + // generate from the title, which is what the editor shows as the
433 + // proposed URL — so the check reads the same before and after
434 + // publishing instead of flipping.
435 + $slug = sanitize_title((string) ($content_data['title'] ?? ''));
436 + }
437 +
438 + return trim(str_replace(['-', '_'], ' ', rawurldecode($slug)));
439 + }
440 +
441 + /**
442 + * The opening of the content: its first paragraph with text, or the first
443 + * few hundred characters when it has none. Counted in characters, not
444 + * words, so a language written without spaces is not read as one word.
445 + *
446 + * @param string $html Content HTML.
447 + * @return string Plain text.
448 + */
449 + private static function opening(string $html): string {
450 + if (preg_match_all('/<p\b[^>]*>(.*?)<\/p>/isu', $html, $matches)) {
451 + foreach ($matches[1] as $paragraph) {
452 + $text = self::collapse_whitespace(wp_strip_all_tags($paragraph));
453 + if ('' !== $text) {
454 + return $text;
455 + }
456 + }
457 + }
458 +
459 + $text = self::plain_text($html);
460 +
461 + return function_exists('mb_substr') ? mb_substr($text, 0, self::OPENING_CHARS) : substr($text, 0, self::OPENING_CHARS);
462 + }
463 +
464 + /**
465 + * Content as plain text, with a space where each tag was. wp_strip_all_tags()
466 + * alone joins neighbouring blocks — "…coffee grinder</h3><p>A good…" became
467 + * "coffee grinderA good" — so a keyword at the end of a heading or a
468 + * paragraph was no longer a word and did not match.
469 + *
470 + * @param string $html Content HTML.
471 + * @return string
472 + */
473 + private static function plain_text(string $html): string {
474 + $spaced = preg_replace('/<[^>]+>/', ' $0 ', $html);
475 +
476 + return self::collapse_whitespace(wp_strip_all_tags(null === $spaced ? $html : (string) $spaced));
477 + }
478 +
479 + /**
480 + * Runs of whitespace down to one space.
481 + *
482 + * The `/u` pass is the one that understands a multibyte space, but
483 + * preg_replace() answers null on bytes that are not valid UTF-8 rather than
484 + * throwing — and casting that null to a string blanked the haystack, so a
485 + * post carrying one mojibake byte (a Latin-1 paste, an old import) reported
486 + * every keyword as missing from its body, its opening, and every
487 + * subheading. The gauge said 0/7 and told the author to add a keyword that
488 + * was already there.
489 + *
490 + * Falls back to the byte-wise collapse, which is what this did before the
491 + * multibyte work added the modifier. Same reasoning keyword_matches()
492 + * already records for its own PCRE failure: a pattern PCRE refuses must not
493 + * be reported as a confident "no match".
494 + *
495 + * @param string $text Text to collapse.
496 + * @return string
497 + */
498 + private static function collapse_whitespace(string $text): string {
499 + $collapsed = preg_replace('/\s+/u', ' ', $text);
500 +
501 + if (null === $collapsed) {
502 + $collapsed = preg_replace('/\s+/', ' ', $text);
503 + }
504 +
505 + return trim(null === $collapsed ? $text : (string) $collapsed);
506 + }
507 +
508 + /**
509 + * Lowercase in any script. strtolower() only folds ASCII, so "Кофе" never
510 + * matched "кофе".
511 + *
512 + * @param string $text Text.
513 + * @return string
514 + */
515 + private static function lower(string $text): string {
516 + return function_exists('mb_strtolower') ? mb_strtolower($text, 'UTF-8') : strtolower($text);
517 + }
518 +
519 + /**
520 + * Scripts written without spaces between words.
521 + *
522 + * @since 2.1.0
523 + * @var string
524 + */
525 + private const SCRIPTIO_CONTINUA = '/[\p{Han}\p{Hiragana}\p{Katakana}\p{Thai}\p{Lao}\p{Khmer}\p{Myanmar}]/u';
526 +
527 + /**
528 + * Whether a keyword appears in a haystack as a word rather than as a
529 + * fragment of a longer one.
530 + *
531 + * The five keyword checks used a plain strpos(), so any substring hit
532 + * counted: "test coronavirus" matched "la|test coronavirus|news", "art"
533 + * matched "start", "cat" matched "category". The panel then confidently
534 + * reported a keyword placement that does not exist (#416). Same class of
535 + * problem #71 fixed in the Image SEO rewriter, and the same remedy.
536 + *
537 + * Both arguments are expected lowercased already.
538 + *
539 + * @since 2.1.0
540 + *
541 + * @param string $haystack Text to search.
542 + * @param string $needle Keyword, lowercased and trimmed.
543 + * @return bool
544 + */
545 + private function keyword_matches(string $haystack, string $needle): bool {
546 + if ($needle === '' || $haystack === '') {
547 + return false;
548 + }
549 +
550 + if (!$this->supports_word_boundaries($needle)) {
551 + return strpos($haystack, $needle) !== false;
552 + }
553 +
554 + $matched = preg_match('/\b' . preg_quote($needle, '/') . '\b/u', $haystack);
555 +
556 + // PCRE refusing the pattern — invalid UTF-8 in the keyword, a
557 + // backtrack limit — must not be reported as a confident "no match".
558 + // Fall back to the behaviour this replaced rather than invent a
559 + // negative the user cannot explain.
560 + if ($matched === false) {
561 + return strpos($haystack, $needle) !== false;
562 + }
563 +
564 + return $matched === 1;
565 + }
566 +
567 + /**
568 + * Whether \b can express "this keyword, as a word" for this keyword.
569 + *
570 + * It asserts a transition between a word and a non-word character, which
571 + * only means something where words are separated. Two cases where it is
572 + * not, both verified against PCRE rather than assumed:
573 + *
574 + * - the keyword's own edges are not word characters ("c++", "#seo"), so
575 + * no boundary can assert there and a real match is lost;
576 + * - scripts written without spaces, where the neighbouring characters
577 + * are word characters too — "冠状病毒" inside "最新冠状病毒新闻" is a
578 + * legitimate match that \b never sees.
579 + *
580 + * Accented Latin and Cyrillic need no special handling: PHP's /u modifier
581 + * turns on Unicode character properties, so "café" correctly does not
582 + * match "cafés" and "коронавирус" does not match "коронавирусный".
583 + *
584 + * @since 2.1.0
585 + *
586 + * @param string $needle Keyword, lowercased and trimmed.
587 + * @return bool
588 + */
589 + private function supports_word_boundaries(string $needle): bool {
590 + if (preg_match(self::SCRIPTIO_CONTINUA, $needle)) {
591 + return false;
592 + }
593 +
594 + return preg_match('/^\w/u', $needle) === 1 && preg_match('/\w$/u', $needle) === 1;
595 + }
596 +
597 + /**
279 598 * Compute the SEO score for a single target keyword.
280 599 *
281 600 * @param array $content_data Content analysis data
282 601 * @param array $metadata Post metadata
@@ -388,9 +707,9 @@
388 707 $total_score += $technical_result['score'];
389 708 $suggestions = array_merge($suggestions, $technical_result['suggestions']);
390 709
391 710 try {
392 - $prioritized_suggestions = $this->prioritize_suggestions($suggestions);
711 + $prioritized_suggestions = $this->prioritize_suggestions($suggestions, $scores);
393 712 $grade = $this->get_grade_from_score($total_score);
394 713
395 714 return [
396 715 'overall_score' => min(100, $total_score),
@@ -539,9 +858,9 @@
539 858 $title_length = mb_strlen($title);
540 859
541 860 // 2025 length optimization (6 points). 60 characters is the recommended
542 861 // maximum for best SERP visibility before Google truncates the title.
543 - if ($title_length >= 35 && $title_length <= 60) {
862 + if ($title_length >= self::TITLE_OPTIMAL_MIN && $title_length <= self::TITLE_OPTIMAL_MAX) {
544 863 $score += 6;
545 864 } elseif ($title_length >= 25 && $title_length <= 75) {
546 865 $score += 4;
547 866 $suggestions[] = 'Optimize title length to 35-60 characters for better SERP visibility';
@@ -815,27 +1134,69 @@
815 1134
816 1135 // Consider it a semantic match if 70% of keyword parts are present
817 1136 return ($matches / count($keyword_parts)) >= 0.7;
818 1137 }
819 - private function prioritize_suggestions(array $suggestions): array {
1138 + private function prioritize_suggestions(array $suggestions, array $scores = []): array {
1139 + // Map each suggestion back to the factor that emitted it, so priority
1140 + // can rank by the points the factor actually lost instead of keyword-
1141 + // matching the advice text — which sorted a 2-point title tweak above
1142 + // a 6-point thin-content loss and contradicted the row's own impact
1143 + // tag (#408).
1144 + $by_text = [];
1145 + foreach ($scores as $factor => $result) {
1146 + if (!is_array($result) || empty($result['suggestions']) || !is_array($result['suggestions'])) {
1147 + continue;
1148 + }
1149 + $lost = max(0, (float) ($result['max_score'] ?? 0) - (float) ($result['score'] ?? 0));
1150 + foreach ($result['suggestions'] as $text) {
1151 + if (is_string($text) && !isset($by_text[$text])) {
1152 + $by_text[$text] = ['factor' => (string) $factor, 'lost' => $lost];
1153 + }
1154 + }
1155 + }
1156 +
820 1157 $prioritized = [];
821 -
1158 +
822 1159 foreach ($suggestions as $suggestion) {
823 - $priority = $this->determine_suggestion_priority($suggestion);
1160 + $origin = $by_text[$suggestion] ?? null;
1161 +
1162 + // A factor already at full marks loses nothing to this advice —
1163 + // it was occupying list positions (sometimes at "High") while
1164 + // recovering zero points. Dropped rather than sorted last.
1165 + if (null !== $origin && $origin['lost'] <= 0) {
1166 + continue;
1167 + }
1168 +
1169 + if (null !== $origin) {
1170 + $priority = $origin['lost'] >= 4 ? 'High' : ($origin['lost'] >= 2 ? 'Medium' : 'Low');
1171 + } else {
1172 + // No factor attached (defensive: a filter-added or legacy
1173 + // suggestion) — the old keyword map is the fallback.
1174 + $priority = $this->determine_suggestion_priority($suggestion);
1175 + }
1176 +
824 1177 $prioritized[] = [
825 1178 'text' => $suggestion,
826 1179 'priority' => $priority,
827 1180 'impact' => $this->estimate_impact($suggestion),
828 1181 'effort' => $this->estimate_effort($suggestion),
1182 + 'factor' => $origin['factor'] ?? null,
1183 + 'points_recoverable' => $origin['lost'] ?? null,
829 1184 ];
830 1185 }
831 -
832 - // Sort by priority (High > Medium > Low)
1186 +
1187 + // Biggest recoverable loss first; keyword-mapped stragglers (no
1188 + // factor) sort within their priority band after the measured rows.
833 1189 usort($prioritized, function($a, $b) {
1190 + $al = $a['points_recoverable'] ?? -1;
1191 + $bl = $b['points_recoverable'] ?? -1;
1192 + if ($al !== $bl) {
1193 + return $bl <=> $al;
1194 + }
834 1195 $priority_order = ['High' => 3, 'Medium' => 2, 'Low' => 1];
835 1196 return $priority_order[$b['priority']] - $priority_order[$a['priority']];
836 1197 });
837 -
1198 +
838 1199 return $prioritized;
839 1200 }
840 1201
841 1202 /**
@@ -902,14 +1263,16 @@
902 1263 if (empty($content) || empty($target_keyword)) {
903 1264 return 0.0;
904 1265 }
905 1266
906 - $content_lower = strtolower(wp_strip_all_tags($content));
1267 + // Occurrences and the word count both come from the reading text, so
1268 + // shortcode syntax is in neither.
1269 + $content_lower = strtolower(self::reading_text_of($content));
907 1270 $keyword_lower = strtolower($target_keyword);
908 1271
909 1272 // Calculate keyword and semantic term frequency
910 1273 $keyword_count = substr_count($content_lower, $keyword_lower);
911 - $word_count = $this->calculate_word_count_js_style($content_lower);
1274 + $word_count = \ThinkRank\SEO\Word_Count_Index::count_words($content_lower);
912 1275
913 1276 if ($word_count === 0) {
914 1277 return 0.0;
915 1278 }
@@ -988,10 +1351,13 @@
988 1351 private function keyword_density(string $content, string $target_keyword): float {
989 1352 if (empty($content) || empty($target_keyword)) {
990 1353 return 0.0;
991 1354 }
992 - $plain = strtolower(wp_strip_all_tags($content));
993 - $word_count = $this->calculate_word_count_js_style($plain);
1355 + // Numerator and denominator from the same reading text. Counting the
1356 + // keyword in wp_strip_all_tags() output found it inside shortcode
1357 + // attributes the word count no longer includes.
1358 + $plain = strtolower(self::reading_text_of($content));
1359 + $word_count = \ThinkRank\SEO\Word_Count_Index::count_words($plain);
994 1360 if ($word_count === 0) {
995 1361 return 0.0;
996 1362 }
997 1363 $occurrences = substr_count($plain, strtolower($target_keyword));
@@ -1026,10 +1392,10 @@
1026 1392 private function keyword_in_first_paragraph(string $content, string $target_keyword): bool {
1027 1393 if (empty($content) || empty($target_keyword)) {
1028 1394 return false;
1029 1395 }
1030 - $plain = strtolower(wp_strip_all_tags($content));
1031 - $words = preg_split('/\s+/', trim($plain), -1, PREG_SPLIT_NO_EMPTY) ?: [];
1396 + $plain = strtolower(self::reading_text_of($content));
1397 + $words = preg_split('/\s+/', $plain, -1, PREG_SPLIT_NO_EMPTY) ?: [];
1032 1398 $window = array_slice($words, 0, max(50, (int) ceil(count($words) * 0.1)));
1033 1399 return strpos(implode(' ', $window), strtolower($target_keyword)) !== false;
1034 1400 }
1035 1401
@@ -1259,9 +1625,24 @@
1259 1625 if (!class_exists('\ThinkRank\SEO\Builder_Content')) {
1260 1626 require_once THINKRANK_PLUGIN_DIR . 'includes/seo/class-builder-content.php';
1261 1627 }
1262 1628
1263 - return \ThinkRank\SEO\Builder_Content::resolve_markup($live_content, $post);
1629 + // Bind the live markup to the post the render makes current. Since #862
1630 + // the resolver runs setup_postdata() on this post, so a shortcode that
1631 + // builds its output from the current post's content — get_the_content(),
1632 + // get_post()->post_content, as a table of contents or a reading-time
1633 + // shortcode does — otherwise read the last saved body while the unsaved
1634 + // markup rendered around it, and lagged a save behind (#864).
1635 + //
1636 + // A clone, not the caller's object: the post is current only for the
1637 + // duration of the render and the caller's $post must come back
1638 + // unchanged. `thinkrank_analyzable_content` receives the clone too,
1639 + // which is what makes $post->post_content there agree with the markup
1640 + // being analyzed on the live path.
1641 + $bound = clone $post;
1642 + $bound->post_content = $live_content;
1643 +
1644 + return \ThinkRank\SEO\Builder_Content::resolve_markup($live_content, $bound);
1264 1645 }
1265 1646
1266 1647 public function analyze_post_content(int $post_id): array {
1267 1648 $post = get_post($post_id);
@@ -1274,11 +1655,12 @@
1274 1655
1275 1656 // Extract headings from content
1276 1657 $headings = $this->extract_headings($content);
1277 1658
1278 - // Count words using JavaScript-compatible method
1279 - $plain_text = wp_strip_all_tags($content);
1280 - $word_count = $this->calculate_word_count_js_style($plain_text);
1659 + // Count the markup, not wp_strip_all_tags() output: stripping deletes
1660 + // tags without a space, so "five</p><p>six" would already be one word
1661 + // before the counter saw it.
1662 + $word_count = $this->calculate_word_count_js_style($content);
1281 1663
1282 1664 // Calculate readability
1283 1665 $readability_score = $this->calculate_readability_score($content);
1284 1666
@@ -1305,9 +1687,11 @@
1305 1687 'images' => $images,
1306 1688 'url' => $url,
1307 1689 'slug' => $post->post_name,
1308 1690 'post_modified' => $post->post_modified,
1309 - 'schema_present' => $this->detect_schema_present($content) || $this->thinkrank_global_schema_active($post->post_type),
1691 + 'schema_present' => $this->detect_schema_present($content)
1692 + || $this->thinkrank_global_schema_active($post->post_type)
1693 + || $this->thinkrank_deployed_schema_active($post),
1310 1694 ];
1311 1695 }
1312 1696
1313 1697 /**
@@ -1396,9 +1780,9 @@
1396 1780 // 2. Long paragraphs (flag paragraphs over 150 words).
1397 1781 $long_paragraphs = 0;
1398 1782 if (preg_match_all('/<p[^>]*>(.*?)<\/p>/is', $content, $matches)) {
1399 1783 foreach ($matches[1] as $paragraph) {
1400 - if ($this->calculate_word_count_js_style(wp_strip_all_tags($paragraph)) > 150) {
1784 + if ($this->calculate_word_count_js_style($paragraph) > 150) {
1401 1785 $long_paragraphs++;
1402 1786 }
1403 1787 }
1404 1788 }
@@ -1453,22 +1837,22 @@
1453 1837 * @param string $content Content text
1454 1838 * @return float Readability score
1455 1839 */
1456 1840 private function calculate_readability_score(string $content): float {
1457 - $text = wp_strip_all_tags($content);
1841 + // Sentences, words and syllables all come from one reading text. The
1842 + // word count drops shortcode syntax and punctuation-only tokens; when
1843 + // sentences and syllables were still read off wp_strip_all_tags()
1844 + // output, "[vc_column width="1/2"]" added syllables (and, with a "." in
1845 + // an attribute, sentences) to a word count that did not include it,
1846 + // and a WPBakery page's Flesch fell from 65 to 46.
1847 + $text = self::reading_text_of($content);
1458 1848
1459 - if (empty($text)) {
1849 + if ('' === $text) {
1460 1850 return 0;
1461 1851 }
1462 1852
1463 - // Count sentences (approximate)
1464 - $sentences = preg_split('/[.!?]+/', $text, -1, PREG_SPLIT_NO_EMPTY);
1465 - $sentence_count = count($sentences);
1466 -
1467 - // Count words
1468 - $word_count = $this->calculate_word_count_js_style(wp_strip_all_tags($text));
1469 -
1470 - // Count syllables (approximate)
1853 + $sentence_count = self::count_sentences($text);
1854 + $word_count = \ThinkRank\SEO\Word_Count_Index::count_words($text);
1471 1855 $syllable_count = $this->count_syllables($text);
1472 1856
1473 1857 if ($sentence_count === 0 || $word_count === 0) {
1474 1858 return 0;
@@ -1480,19 +1864,59 @@
1480 1864 return max(0, min(100, $score));
1481 1865 }
1482 1866
1483 1867 /**
1868 + * Count the sentences in reading text.
1869 + *
1870 + * A sentence is a run of text between terminal punctuation that holds at
1871 + * least one letter or digit. A fragment with no word in it (the space
1872 + * after the final full stop, a stray "!" between two "?") is not a
1873 + * sentence. contentAnalysis.js countSentences() applies the same rule.
1874 + *
1875 + * @since 2.14.2
1876 + *
1877 + * @param string $text Text as returned by {@see self::reading_text_of()}.
1878 + * @return int
1879 + */
1880 + private static function count_sentences(string $text): int {
1881 + $fragments = preg_split('/[.!?]+/', $text, -1, PREG_SPLIT_NO_EMPTY) ?: [];
1882 +
1883 + return count(preg_grep('/[\p{L}\p{N}]/u', $fragments) ?: []);
1884 + }
1885 +
1886 + /**
1484 1887 * Count syllables in text (approximate)
1485 1888 *
1889 + * Expects reading text ({@see self::reading_text_of()}); tags are not
1890 + * stripped here, so a decoded "<" in prose is not read as a tag.
1891 + *
1486 1892 * @param string $text Text to analyze
1487 1893 * @return int Syllable count
1488 1894 */
1489 1895 private function count_syllables(string $text): int {
1490 - $words = preg_split('/\s+/', trim(strtolower(wp_strip_all_tags($text))), -1, PREG_SPLIT_NO_EMPTY);
1896 + $words = preg_split('/\s+/', trim(strtolower($text)), -1, PREG_SPLIT_NO_EMPTY) ?: [];
1491 1897 $syllables = 0;
1492 1898
1493 1899 foreach ($words as $word) {
1494 - $syllables += max(1, preg_match_all('/[aeiouy]+/', $word));
1900 + $word = preg_replace('/[^a-z]/', '', $word);
1901 + if ($word === '') {
1902 + continue;
1903 + }
1904 +
1905 + $groups = preg_match_all('/[aeiouy]+/', $word);
1906 +
1907 + // Standard Flesch heuristic: a trailing silent e does not form a
1908 + // syllable ("make", "time", "these") — but only when a consonant
1909 + // precedes it (a vowel+e ending like "movie" already shares its
1910 + // group) and never for consonant-le ("table"), which does count.
1911 + // Without this the counter inflated syllables/word by ~0.2-0.3 on
1912 + // ordinary prose, driving raw Flesch negative and the UI to a
1913 + // clamped "Very Difficult (0)" (#407).
1914 + if ($groups > 1 && preg_match('/[^aeiouy]e$/', $word) && !str_ends_with($word, 'le')) {
1915 + $groups--;
1916 + }
1917 +
1918 + $syllables += max(1, $groups);
1495 1919 }
1496 1920
1497 1921 return $syllables;
1498 1922 }
@@ -1611,8 +2035,45 @@
1611 2035 $all_settings = get_option('thinkrank_global_seo_settings', []);
1612 2036 return !empty($all_settings[$post_type]['schema_type']);
1613 2037 }
1614 2038
2039 + /**
2040 + * Whether the Schema Manager has an active deployed schema for this post.
2041 + *
2042 + * Per-post schema deployed from the editor's Schema tab is stored in the
2043 + * Schema Manager's own table and emitted at wp_head by
2044 + * Frontend\SEO_Manager::output_site_schema_markup(). Neither
2045 + * detect_schema_present() (body scan) nor thinkrank_global_schema_active()
2046 + * (post-type option) sees it, so without this the score reported "no
2047 + * structured data" for posts that do emit it.
2048 + *
2049 + * Mirrors the context_type whitelist the emitter and the metabox both use, so
2050 + * the lookup targets the same row the front end reads.
2051 + *
2052 + * @param \WP_Post $post Post being scored.
2053 + * @return bool
2054 + */
2055 + private function thinkrank_deployed_schema_active(\WP_Post $post): bool {
2056 + if (!class_exists('ThinkRank\\SEO\\Schema_Management_System')) {
2057 + $manager_file = THINKRANK_PLUGIN_DIR . 'includes/seo/class-schema-management-system.php';
2058 + if (!file_exists($manager_file)) {
2059 + return false;
2060 + }
2061 + require_once $manager_file;
2062 + }
2063 +
2064 + $context_type = in_array($post->post_type, ['site', 'post', 'page', 'product'], true)
2065 + ? $post->post_type
2066 + : 'post';
2067 +
2068 + try {
2069 + $manager = new \ThinkRank\SEO\Schema_Management_System();
2070 + return !empty($manager->get_deployed_schemas($context_type, (int) $post->ID));
2071 + } catch (\Throwable $e) {
2072 + return false;
2073 + }
2074 + }
2075 +
1615 2076 private function analyze_images(string $content): array {
1616 2077 $images = [];
1617 2078
1618 2079 if (preg_match_all('/<img[^>]+>/i', $content, $matches)) {
@@ -1739,20 +2200,101 @@
1739 2200 * @param array $content_data Content analysis data
1740 2201 * @return array Scoring result
1741 2202 */
1742 2203 private function score_mobile_experience(array $content_data): array {
1743 - // Mobile experience is theme/site-level, not controlled by post content.
1744 - // Award full credit (benefit of the doubt) instead of a fixed partial
1745 - // that caps every post's ceiling.
2204 + $max = $this->scoring_factors['mobile_experience'];
2205 +
2206 + // Mobile experience is theme/site-level, not controlled by post content
2207 + // — but the plugin already measures it. When a mobile Lighthouse score
2208 + // has been collected, score against it; the "benefit of the doubt" below
2209 + // is for sites nobody has measured, not for sites measured as slow.
2210 + $performance_score = $this->measured_performance_score();
2211 +
2212 + if ($performance_score === null) {
2213 + return [
2214 + 'score' => $max,
2215 + 'max_score' => $max,
2216 + 'suggestions' => ['Ensure mobile-first design and fast loading on mobile devices'],
2217 + 'details' => ['mobile_score' => 'Assumed adequate', 'measured' => false],
2218 + ];
2219 + }
2220 +
2221 + $score = (int) round($max * $performance_score / 100);
2222 +
1746 2223 return [
1747 - 'score' => $this->scoring_factors['mobile_experience'],
1748 - 'max_score' => $this->scoring_factors['mobile_experience'],
1749 - 'suggestions' => ['Ensure mobile-first design and fast loading on mobile devices'],
1750 - 'details' => ['mobile_score' => 'Assumed adequate'],
2224 + 'score' => $score,
2225 + 'max_score' => $max,
2226 + 'suggestions' => $score < $max
2227 + ? ['Improve mobile page speed: the last PageSpeed run scored ' . $performance_score . '/100 on mobile']
2228 + : [],
2229 + 'details' => [
2230 + 'mobile_score' => $performance_score,
2231 + 'measured' => true,
2232 + 'source' => 'pagespeed_mobile',
2233 + ],
1751 2234 ];
1752 2235 }
1753 2236
1754 2237 /**
2238 + * The last collected mobile Lighthouse score, or null when unmeasured.
2239 + *
2240 + * Memoised per instance: compute_score() asks twice, and a post-list screen
2241 + * scores a page of posts at a time.
2242 + *
2243 + * Every failure — no performance module, no collected row, an unreadable
2244 + * table — resolves to null, which the callers read as "not measured" and
2245 + * answer with the full-credit fallback. A site is never penalised for
2246 + * ThinkRank being unable to look.
2247 + *
2248 + * @since 2.3.1
2249 + * @return int|null Score 0-100, or null when nothing has been collected.
2250 + */
2251 + private function measured_performance_score(): ?int {
2252 + $measurement = $this->measured_performance();
2253 +
2254 + if ($measurement === null || !isset($measurement['performance_score'])) {
2255 + return null;
2256 + }
2257 +
2258 + $score = $measurement['performance_score'];
2259 +
2260 + if (!is_numeric($score)) {
2261 + return null;
2262 + }
2263 +
2264 + return (int) round(max(0, min(100, (float) $score)));
2265 + }
2266 +
2267 + /**
2268 + * The last collected mobile measurement, or null when there is none.
2269 + *
2270 + * @since 2.3.1
2271 + * @return array|null { core_web_vitals: array, performance_score: float|null }
2272 + */
2273 + private function measured_performance(): ?array {
2274 + if ($this->measured_performance_resolved) {
2275 + return $this->measured_performance;
2276 + }
2277 +
2278 + $this->measured_performance_resolved = true;
2279 +
2280 + if (!class_exists('ThinkRank\\SEO\\Performance_Monitoring_Manager')) {
2281 + return null;
2282 + }
2283 +
2284 + try {
2285 + $manager = new \ThinkRank\SEO\Performance_Monitoring_Manager();
2286 + // Mobile deliberately: Google indexes mobile-first, and it is the
2287 + // device the mobile_experience factor is named after.
2288 + $this->measured_performance = $manager->get_stored_performance_measurement('mobile');
2289 + } catch (\Throwable $e) {
2290 + $this->measured_performance = null;
2291 + }
2292 +
2293 + return $this->measured_performance;
2294 + }
2295 +
2296 + /**
1755 2297 * Score core web vitals - 2025 version (3 points)
1756 2298 *
1757 2299 * @param array $content_data Content analysis data
1758 2300 * @return array Scoring result
@@ -1757,20 +2299,98 @@
1757 2299 * @param array $content_data Content analysis data
1758 2300 * @return array Scoring result
1759 2301 */
1760 2302 private function score_core_web_vitals(array $content_data): array {
1761 - // Core Web Vitals are a runtime/performance signal, not derivable from
1762 - // post content. Award full credit (benefit of the doubt) rather than a
1763 - // fixed partial that caps every post's ceiling.
2303 + $max = $this->scoring_factors['core_web_vitals'];
2304 +
2305 + // Not derivable from post content — but it is measured, and the audit
2306 + // stores LCP, INP and CLS with a rating each. Score against those when
2307 + // they exist; fall back to the benefit of the doubt when they do not.
2308 + $rated = $this->measured_vitals_score();
2309 +
2310 + if ($rated === null) {
2311 + return [
2312 + 'score' => $max,
2313 + 'max_score' => $max,
2314 + 'suggestions' => ['Optimize Core Web Vitals: LCP, INP, and CLS for better user experience'],
2315 + 'details' => ['vitals_status' => 'Assumed adequate', 'measured' => false],
2316 + ];
2317 + }
2318 +
2319 + $score = (int) round($max * $rated['average'] / 100);
2320 +
1764 2321 return [
1765 - 'score' => $this->scoring_factors['core_web_vitals'],
1766 - 'max_score' => $this->scoring_factors['core_web_vitals'],
1767 - 'suggestions' => ['Optimize Core Web Vitals: LCP, INP, and CLS for better user experience'],
1768 - 'details' => ['vitals_status' => 'Assumed adequate'],
2322 + 'score' => $score,
2323 + 'max_score' => $max,
2324 + // Gated on the measurement, not the rounded score: two good metrics
2325 + // and one needing improvement averages 88.33, which rounds to the
2326 + // full 3 of 3 and used to swallow the suggestion naming the metric
2327 + // that is actually failing.
2328 + 'suggestions' => !empty($rated['failing'])
2329 + ? ['Optimize Core Web Vitals: ' . implode(', ', $rated['failing']) . ' below target on mobile']
2330 + : [],
2331 + 'details' => [
2332 + 'vitals_status' => $rated['statuses'],
2333 + 'measured' => true,
2334 + 'source' => 'pagespeed_mobile',
2335 + ],
1769 2336 ];
1770 2337 }
1771 2338
1772 2339 /**
2340 + * Rate the collected Core Web Vitals, or null when none were measured.
2341 + *
2342 + * Reuses the per-metric score the performance module already assigns
2343 + * (good 100, needs improvement 65, poor 30) rather than inventing a second
2344 + * scale, so the SEO score and the performance card cannot disagree about
2345 + * whether a metric is healthy.
2346 + *
2347 + * Metrics with no stored value — fcp is not always collected — are skipped
2348 + * rather than counted as failures.
2349 + *
2350 + * @since 2.3.1
2351 + * @return array|null { average: float, statuses: array, failing: string[] }
2352 + */
2353 + private function measured_vitals_score(): ?array {
2354 + $measurement = $this->measured_performance();
2355 + $vitals = $measurement['core_web_vitals'] ?? null;
2356 +
2357 + if (!is_array($vitals)) {
2358 + return null;
2359 + }
2360 +
2361 + $scores = [];
2362 + $statuses = [];
2363 + $failing = [];
2364 +
2365 + // The three Google ranks on. fcp is diagnostic and not a Core Web Vital.
2366 + foreach (['lcp', 'inp', 'cls'] as $metric) {
2367 + $data = $vitals[$metric] ?? null;
2368 +
2369 + if (!is_array($data) || !isset($data['value'], $data['score']) || $data['value'] === null) {
2370 + continue;
2371 + }
2372 +
2373 + $scores[] = (float) $data['score'];
2374 + $statuses[$metric] = $data['status'] ?? 'unknown';
2375 +
2376 + if (($data['status'] ?? '') !== 'good') {
2377 + $failing[] = strtoupper($metric);
2378 + }
2379 + }
2380 +
2381 + if (empty($scores)) {
2382 + return null;
2383 + }
2384 +
2385 + return [
2386 + 'average' => array_sum($scores) / count($scores),
2387 + 'statuses' => $statuses,
2388 + 'failing' => $failing,
2389 + ];
2390 + }
2391 +
2392 + /**
1773 2393 * Score internal linking - declining importance (1 point)
1774 2394 *
1775 2395 * @param array $content_data Content analysis data
1776 2396 * @return array Scoring result
@@ -1800,9 +2420,9 @@
1800 2420 $suggestions = [];
1801 2421
1802 2422 // Meta description check
1803 2423 $meta_desc = $metadata['description'] ?? '';
1804 - if (!empty($meta_desc) && mb_strlen($meta_desc) >= 120 && mb_strlen($meta_desc) <= 160) {
2424 + if (!empty($meta_desc) && mb_strlen($meta_desc) >= self::DESCRIPTION_OPTIMAL_MIN && mb_strlen($meta_desc) <= self::DESCRIPTION_OPTIMAL_MAX) {
1805 2425 $score += 0.5;
1806 2426 } else {
1807 2427 $suggestions[] = 'Add a compelling meta description (120-160 characters)';
1808 2428 }
@@ -1884,11 +2504,12 @@
1884 2504
1885 2505 // Extract headings from content
1886 2506 $headings = $this->extract_headings($content);
1887 2507
1888 - // Count words using JavaScript-compatible method
1889 - $plain_text = wp_strip_all_tags($content);
1890 - $word_count = $this->calculate_word_count_js_style($plain_text);
2508 + // Count the markup, not wp_strip_all_tags() output: stripping deletes
2509 + // tags without a space, so "five</p><p>six" would already be one word
2510 + // before the counter saw it.
2511 + $word_count = $this->calculate_word_count_js_style($content);
1891 2512
1892 2513 // Calculate readability
1893 2514 $readability_score = $this->calculate_readability_score($content);
1894 2515
@@ -1915,27 +2536,66 @@
1915 2536 'images' => $images,
1916 2537 'url' => $url,
1917 2538 'slug' => $post->post_name,
1918 2539 'post_modified' => $post->post_modified,
1919 - 'schema_present' => $this->detect_schema_present($content) || $this->thinkrank_global_schema_active($post->post_type),
2540 + 'schema_present' => $this->detect_schema_present($content)
2541 + || $this->thinkrank_global_schema_active($post->post_type)
2542 + || $this->thinkrank_deployed_schema_active($post),
1920 2543 ];
1921 2544 }
1922 2545
1923 2546 /**
1924 - * Calculate word count using JavaScript-compatible method
1925 - * Matches the logic in contentAnalysis.js for consistency
2547 + * Count the words a reader reads in HTML or text.
1926 2548 *
1927 - * @param string $text Text to count words in
2549 + * The same extraction and counting as the thin content report
2550 + * ({@see \ThinkRank\SEO\Word_Count_Index::reading_text()} and
2551 + * {@see \ThinkRank\SEO\Word_Count_Index::count_words()}), so the editor
2552 + * score and the report give one number for one page. Splitting on
2553 + * whitespace counted shortcode syntax as words: a WPBakery page with 216
2554 + * words of prose scored a word count of 325 while the report said 216
2555 + * (#893 fixed the report only). It also counted tokens of punctuation
2556 + * alone, such as a full stop after a link.
2557 + *
2558 + * Always words, whatever the locale's unit, because every threshold that
2559 + * reads this value (content length, long paragraphs, headings per 300
2560 + * words, readability, keyword density) is in words.
2561 + *
2562 + * contentAnalysis.js calculateWordCount() applies the same two rules in
2563 + * the editor.
2564 + *
2565 + * @param string $text HTML or text to count words in.
1928 2566 * @return int Word count
1929 2567 */
1930 2568 private function calculate_word_count_js_style(string $text): int {
1931 - if (empty($text)) {
2569 + if ('' === $text) {
1932 2570 return 0;
1933 2571 }
1934 2572
1935 - // Match JavaScript: trim, split by whitespace, filter empty
1936 - $words = preg_split('/\s+/', trim($text), -1, PREG_SPLIT_NO_EMPTY);
1937 - return count($words);
2573 + return \ThinkRank\SEO\Word_Count_Index::count_words(self::reading_text_of($text));
2574 + }
2575 +
2576 + /**
2577 + * The text a reader reads in HTML or text, as the word count sees it.
2578 + *
2579 + * Every check that divides by the word count (readability, keyword
2580 + * density, topic relevance) reads its numerator from this same text, so
2581 + * shortcode syntax is never on one side of a ratio and not the other.
2582 + *
2583 + * @since 2.14.2
2584 + *
2585 + * @param string $content HTML or text.
2586 + * @return string Plain text, whitespace collapsed.
2587 + */
2588 + private static function reading_text_of(string $content): string {
2589 + if ('' === $content) {
2590 + return '';
2591 + }
2592 +
2593 + if (!class_exists('\ThinkRank\SEO\Word_Count_Index')) {
2594 + require_once THINKRANK_PLUGIN_DIR . 'includes/seo/class-word-count-index.php';
2595 + }
2596 +
2597 + return \ThinkRank\SEO\Word_Count_Index::reading_text($content);
1938 2598 }
1939 2599
1940 2600 /**
1941 2601 * Get existing score data for a post from database