PluginProbe
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO / 2.14.2
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO v2.14.2
2.14.2 2.14.1 2.14.0 2.13.0 2.12.0 2.11.0 2.10.0 2.9.0 2.8.0 2.7.0 2.6.0 2.5.0 2.4.0 2.3.0 2.2.0 2.1.1 2.1.0 2.0.2 2.0.1 2.0.0 1.32.0 1.31.0 1.30.0 1.29.0 1.28.0 All 57 releases
← All changes | includes/ai/class-seo-score-calculator.php +677 -71 2.0.0 → 2.14.2 View file →
@@ -41,8 +41,41 @@
41 41 */
42 42 private Database $database;
43 43
44 44 /**
45 + * Memoised collected performance measurement, and whether it was resolved.
46 + *
47 + * Two factors read it and both may be asked for on every post in a list, so
48 + * the lookup happens once per calculator. `null` is a real answer here — the
49 + * separate flag keeps "not looked up yet" distinct from "nothing measured".
50 + *
51 + * @since 2.3.1
52 + * @var array|null
53 + */
54 + private ?array $measured_performance = null;
55 +
56 + /**
57 + * @since 2.3.1
58 + * @var bool
59 + */
60 + private bool $measured_performance_resolved = false;
61 +
62 + /**
63 + * Length bands the editor scores against, in characters.
64 + *
65 + * Public so every surface that judges a title or description — the editor
66 + * score and the Bulk Snippets problem filter — reads one set of numbers.
67 + * Before these existed the bands were literals inside the scoring methods,
68 + * and a second screen would have had to copy them and drift (#727).
69 + *
70 + * @since 2.8.0
71 + */
72 + public const TITLE_OPTIMAL_MIN = 35;
73 + public const TITLE_OPTIMAL_MAX = 60;
74 + public const DESCRIPTION_OPTIMAL_MIN = 120;
75 + public const DESCRIPTION_OPTIMAL_MAX = 160;
76 +
77 + /**
45 78 * 2025 SEO scoring factors (Q1 2025 Google Algorithm)
46 79 * Based on First Page Sage research and Google's latest updates
47 80 *
48 81 * @var array
@@ -107,8 +140,9 @@
107 140 'grade' => $result['grade'],
108 141 ]];
109 142 $result['keyword_checks'] = $this->analyze_keyword_checks($content_data, $metadata, $keywords);
110 143 }
144 + $result['keywords'] = $this->keyword_placements($content_data, $metadata, $keywords);
111 145
112 146 return $result;
113 147 }
114 148
@@ -218,8 +252,9 @@
218 252 $best['target_keyword'] = $best_keyword;
219 253 $best['target_keywords'] = $keywords;
220 254 $best['keyword_results'] = $per_keyword;
221 255 $best['keyword_checks'] = $this->analyze_keyword_checks($content_data, $metadata, $keywords);
256 + $best['keywords'] = $this->keyword_placements($content_data, $metadata, $keywords);
222 257
223 258 return $best;
224 259 }
225 260
@@ -234,15 +269,15 @@
234 269 * @param string[] $keywords Target keywords.
235 270 * @return array<string,array{passed:bool,matched_keywords:string[]}>
236 271 */
237 272 private function analyze_keyword_checks(array $content_data, array $metadata, array $keywords): array {
238 - $title = strtolower((string) ($metadata['title'] ?? $content_data['title'] ?? ''));
239 - $description = strtolower((string) ($metadata['description'] ?? ''));
240 - $content = strtolower(wp_strip_all_tags((string) ($content_data['content'] ?? '')));
273 + $title = self::lower((string) ($metadata['title'] ?? $content_data['title'] ?? ''));
274 + $description = self::lower((string) ($metadata['description'] ?? ''));
275 + $content = self::lower(self::plain_text((string) ($content_data['content'] ?? '')));
241 276
242 277 $alts = '';
243 278 foreach ((array) ($content_data['images'] ?? []) as $image) {
244 - $alts .= ' ' . strtolower((string) ($image['alt'] ?? ''));
279 + $alts .= ' ' . self::lower((string) ($image['alt'] ?? ''));
245 280 }
246 281
247 282 // Build a searchable slug haystack from the post's OWN slug — never the
248 283 // full URL path. The path carries ancestors, category bases and date
@@ -251,17 +286,9 @@
251 286 // way: an unpublished post has no pretty permalink (get_permalink()
252 287 // returns ?p=123), so the path held no slug at all and every draft
253 288 // scored "no match" until it was published. Hyphens/underscores become
254 289 // spaces so multi-word keywords can match.
255 - $slug_source = (string) ($content_data['slug'] ?? '');
256 - if ($slug_source === '') {
257 - // Draft with no slug assigned yet: score what WordPress would
258 - // generate from the title, which is what the editor shows as the
259 - // proposed URL — so the check reads the same before and after
260 - // publishing instead of flipping.
261 - $slug_source = sanitize_title((string) ($content_data['title'] ?? ''));
262 - }
263 - $slug = strtolower(str_replace(['-', '_'], ' ', $slug_source));
290 + $slug = self::lower(self::slug_haystack($content_data));
264 291
265 292 $haystacks = [
266 293 'title' => trim($title),
267 294 'meta_description' => trim($description),
@@ -273,10 +300,9 @@
273 300 $checks = [];
274 301 foreach ($haystacks as $location => $haystack) {
275 302 $matched = [];
276 303 foreach ($keywords as $keyword) {
277 - $needle = strtolower(trim($keyword));
278 - if ($needle !== '' && $haystack !== '' && strpos($haystack, $needle) !== false) {
304 + if ($this->keyword_matches($haystack, self::lower(trim($keyword)))) {
279 305 $matched[] = $keyword;
280 306 }
281 307 }
282 308 $checks[$location] = [
@@ -288,8 +314,288 @@
288 314 return $checks;
289 315 }
290 316
291 317 /**
318 + * The keyword placements the editor draws one gauge segment for, in the
319 + * order a reader meets them (#729).
320 + *
321 + * @since 2.11.0
322 + * @var string[]
323 + */
324 + public const PLACEMENTS = ['title', 'meta_description', 'slug', 'first_paragraph', 'subheading', 'content', 'image_alt'];
325 +
326 + /**
327 + * Characters of plain text read as the opening when the content has no
328 + * paragraph tag.
329 + *
330 + * @since 2.11.0
331 + */
332 + private const OPENING_CHARS = 300;
333 +
334 + /**
335 + * Where each focus keyword is placed, keyword by keyword (#729).
336 + *
337 + * analyze_keyword_checks() answers "does ANY keyword appear here" for
338 + * five places; this answers "where does THIS keyword appear" for seven,
339 + * so the editor can show each keyword's own gauge. Same matcher, so a
340 + * keyword counts as a word (not a fragment) and a keyword in a script
341 + * written without spaces (Thai, Chinese, Japanese) still matches.
342 + *
343 + * `where` names the heading or alt text that matched, so the editor can
344 + * say which one.
345 + *
346 + * @since 2.11.0
347 + *
348 + * @param array $content_data Content analysis data.
349 + * @param array $metadata Post metadata (title, description).
350 + * @param string[] $keywords Focus keywords.
351 + * @return array<int, array{keyword: string, passed: int, total: int, placements: array<string, array{passed: bool, where: string}>}>
352 + */
353 + public function keyword_placements(array $content_data, array $metadata, array $keywords): array {
354 + $html = (string) ($content_data['content'] ?? '');
355 + $plain = self::lower(self::plain_text($html));
356 +
357 + $headings = [];
358 + $source = isset($content_data['headings']) && is_array($content_data['headings']) ? $content_data['headings'] : $this->extract_headings($html);
359 + foreach ($source as $heading) {
360 + $text = trim((string) ($heading['text'] ?? ''));
361 + if ((int) ($heading['level'] ?? 0) >= 2 && '' !== $text) {
362 + $headings[] = $text;
363 + }
364 + }
365 +
366 + $alts = [];
367 + foreach ((array) ($content_data['images'] ?? []) as $image) {
368 + $alt = trim((string) ($image['alt'] ?? ''));
369 + if ('' !== $alt) {
370 + $alts[] = $alt;
371 + }
372 + }
373 +
374 + $single = [
375 + 'title' => self::lower((string) ($metadata['title'] ?? $content_data['title'] ?? '')),
376 + 'meta_description' => self::lower((string) ($metadata['description'] ?? '')),
377 + 'slug' => self::lower(self::slug_haystack($content_data)),
378 + 'first_paragraph' => self::lower(self::opening($html)),
379 + 'content' => $plain,
380 + ];
381 +
382 + $out = [];
383 + foreach ($keywords as $keyword) {
384 + $needle = self::lower(trim((string) $keyword));
385 + $placements = [];
386 +
387 + foreach (self::PLACEMENTS as $placement) {
388 + if ('subheading' === $placement || 'image_alt' === $placement) {
389 + $where = '';
390 + foreach ('subheading' === $placement ? $headings : $alts as $text) {
391 + if ($this->keyword_matches(self::lower($text), $needle)) {
392 + $where = $text;
393 + break;
394 + }
395 + }
396 + $placements[$placement] = ['passed' => '' !== $where, 'where' => $where];
397 + continue;
398 + }
399 +
400 + $placements[$placement] = ['passed' => $this->keyword_matches($single[$placement], $needle), 'where' => ''];
401 + }
402 +
403 + $out[] = [
404 + 'keyword' => (string) $keyword,
405 + 'passed' => count(array_filter(array_column($placements, 'passed'))),
406 + 'total' => count(self::PLACEMENTS),
407 + 'placements' => $placements,
408 + ];
409 + }
410 +
411 + return $out;
412 + }
413 +
414 + /**
415 + * The post's own slug as searchable text: hyphens and underscores become
416 + * spaces so a multi-word keyword can match, and a slug WordPress
417 + * percent-encoded (Thai, Cyrillic, Chinese) is decoded, or it could never
418 + * match a keyword typed in that script.
419 + *
420 + * Never the full URL path: the path carries ancestors, category bases and
421 + * date segments, so a child of /clinical-trials/ reported "keyword in slug"
422 + * for a page actually slugged `contact-us`. An unpublished post has no
423 + * pretty permalink either, so the path held no slug at all.
424 + *
425 + * @param array $content_data Content analysis data.
426 + * @return string
427 + */
428 + private static function slug_haystack(array $content_data): string {
429 + $slug = (string) ($content_data['slug'] ?? '');
430 + if ('' === $slug) {
431 + // Draft with no slug assigned yet: score what WordPress would
432 + // generate from the title, which is what the editor shows as the
433 + // proposed URL — so the check reads the same before and after
434 + // publishing instead of flipping.
435 + $slug = sanitize_title((string) ($content_data['title'] ?? ''));
436 + }
437 +
438 + return trim(str_replace(['-', '_'], ' ', rawurldecode($slug)));
439 + }
440 +
441 + /**
442 + * The opening of the content: its first paragraph with text, or the first
443 + * few hundred characters when it has none. Counted in characters, not
444 + * words, so a language written without spaces is not read as one word.
445 + *
446 + * @param string $html Content HTML.
447 + * @return string Plain text.
448 + */
449 + private static function opening(string $html): string {
450 + if (preg_match_all('/<p\b[^>]*>(.*?)<\/p>/isu', $html, $matches)) {
451 + foreach ($matches[1] as $paragraph) {
452 + $text = self::collapse_whitespace(wp_strip_all_tags($paragraph));
453 + if ('' !== $text) {
454 + return $text;
455 + }
456 + }
457 + }
458 +
459 + $text = self::plain_text($html);
460 +
461 + return function_exists('mb_substr') ? mb_substr($text, 0, self::OPENING_CHARS) : substr($text, 0, self::OPENING_CHARS);
462 + }
463 +
464 + /**
465 + * Content as plain text, with a space where each tag was. wp_strip_all_tags()
466 + * alone joins neighbouring blocks — "…coffee grinder</h3><p>A good…" became
467 + * "coffee grinderA good" — so a keyword at the end of a heading or a
468 + * paragraph was no longer a word and did not match.
469 + *
470 + * @param string $html Content HTML.
471 + * @return string
472 + */
473 + private static function plain_text(string $html): string {
474 + $spaced = preg_replace('/<[^>]+>/', ' $0 ', $html);
475 +
476 + return self::collapse_whitespace(wp_strip_all_tags(null === $spaced ? $html : (string) $spaced));
477 + }
478 +
479 + /**
480 + * Runs of whitespace down to one space.
481 + *
482 + * The `/u` pass is the one that understands a multibyte space, but
483 + * preg_replace() answers null on bytes that are not valid UTF-8 rather than
484 + * throwing — and casting that null to a string blanked the haystack, so a
485 + * post carrying one mojibake byte (a Latin-1 paste, an old import) reported
486 + * every keyword as missing from its body, its opening, and every
487 + * subheading. The gauge said 0/7 and told the author to add a keyword that
488 + * was already there.
489 + *
490 + * Falls back to the byte-wise collapse, which is what this did before the
491 + * multibyte work added the modifier. Same reasoning keyword_matches()
492 + * already records for its own PCRE failure: a pattern PCRE refuses must not
493 + * be reported as a confident "no match".
494 + *
495 + * @param string $text Text to collapse.
496 + * @return string
497 + */
498 + private static function collapse_whitespace(string $text): string {
499 + $collapsed = preg_replace('/\s+/u', ' ', $text);
500 +
501 + if (null === $collapsed) {
502 + $collapsed = preg_replace('/\s+/', ' ', $text);
503 + }
504 +
505 + return trim(null === $collapsed ? $text : (string) $collapsed);
506 + }
507 +
508 + /**
509 + * Lowercase in any script. strtolower() only folds ASCII, so "Кофе" never
510 + * matched "кофе".
511 + *
512 + * @param string $text Text.
513 + * @return string
514 + */
515 + private static function lower(string $text): string {
516 + return function_exists('mb_strtolower') ? mb_strtolower($text, 'UTF-8') : strtolower($text);
517 + }
518 +
519 + /**
520 + * Scripts written without spaces between words.
521 + *
522 + * @since 2.1.0
523 + * @var string
524 + */
525 + private const SCRIPTIO_CONTINUA = '/[\p{Han}\p{Hiragana}\p{Katakana}\p{Thai}\p{Lao}\p{Khmer}\p{Myanmar}]/u';
526 +
527 + /**
528 + * Whether a keyword appears in a haystack as a word rather than as a
529 + * fragment of a longer one.
530 + *
531 + * The five keyword checks used a plain strpos(), so any substring hit
532 + * counted: "test coronavirus" matched "la|test coronavirus|news", "art"
533 + * matched "start", "cat" matched "category". The panel then confidently
534 + * reported a keyword placement that does not exist (#416). Same class of
535 + * problem #71 fixed in the Image SEO rewriter, and the same remedy.
536 + *
537 + * Both arguments are expected lowercased already.
538 + *
539 + * @since 2.1.0
540 + *
541 + * @param string $haystack Text to search.
542 + * @param string $needle Keyword, lowercased and trimmed.
543 + * @return bool
544 + */
545 + private function keyword_matches(string $haystack, string $needle): bool {
546 + if ($needle === '' || $haystack === '') {
547 + return false;
548 + }
549 +
550 + if (!$this->supports_word_boundaries($needle)) {
551 + return strpos($haystack, $needle) !== false;
552 + }
553 +
554 + $matched = preg_match('/\b' . preg_quote($needle, '/') . '\b/u', $haystack);
555 +
556 + // PCRE refusing the pattern — invalid UTF-8 in the keyword, a
557 + // backtrack limit — must not be reported as a confident "no match".
558 + // Fall back to the behaviour this replaced rather than invent a
559 + // negative the user cannot explain.
560 + if ($matched === false) {
561 + return strpos($haystack, $needle) !== false;
562 + }
563 +
564 + return $matched === 1;
565 + }
566 +
567 + /**
568 + * Whether \b can express "this keyword, as a word" for this keyword.
569 + *
570 + * It asserts a transition between a word and a non-word character, which
571 + * only means something where words are separated. Two cases where it is
572 + * not, both verified against PCRE rather than assumed:
573 + *
574 + * - the keyword's own edges are not word characters ("c++", "#seo"), so
575 + * no boundary can assert there and a real match is lost;
576 + * - scripts written without spaces, where the neighbouring characters
577 + * are word characters too — "冠状病毒" inside "最新冠状病毒新闻" is a
578 + * legitimate match that \b never sees.
579 + *
580 + * Accented Latin and Cyrillic need no special handling: PHP's /u modifier
581 + * turns on Unicode character properties, so "café" correctly does not
582 + * match "cafés" and "коронавирус" does not match "коронавирусный".
583 + *
584 + * @since 2.1.0
585 + *
586 + * @param string $needle Keyword, lowercased and trimmed.
587 + * @return bool
588 + */
589 + private function supports_word_boundaries(string $needle): bool {
590 + if (preg_match(self::SCRIPTIO_CONTINUA, $needle)) {
591 + return false;
592 + }
593 +
594 + return preg_match('/^\w/u', $needle) === 1 && preg_match('/\w$/u', $needle) === 1;
595 + }
596 +
597 + /**
292 598 * Compute the SEO score for a single target keyword.
293 599 *
294 600 * @param array $content_data Content analysis data
295 601 * @param array $metadata Post metadata
@@ -401,9 +707,9 @@
401 707 $total_score += $technical_result['score'];
402 708 $suggestions = array_merge($suggestions, $technical_result['suggestions']);
403 709
404 710 try {
405 - $prioritized_suggestions = $this->prioritize_suggestions($suggestions);
711 + $prioritized_suggestions = $this->prioritize_suggestions($suggestions, $scores);
406 712 $grade = $this->get_grade_from_score($total_score);
407 713
408 714 return [
409 715 'overall_score' => min(100, $total_score),
@@ -552,9 +858,9 @@
552 858 $title_length = mb_strlen($title);
553 859
554 860 // 2025 length optimization (6 points). 60 characters is the recommended
555 861 // maximum for best SERP visibility before Google truncates the title.
556 - if ($title_length >= 35 && $title_length <= 60) {
862 + if ($title_length >= self::TITLE_OPTIMAL_MIN && $title_length <= self::TITLE_OPTIMAL_MAX) {
557 863 $score += 6;
558 864 } elseif ($title_length >= 25 && $title_length <= 75) {
559 865 $score += 4;
560 866 $suggestions[] = 'Optimize title length to 35-60 characters for better SERP visibility';
@@ -828,27 +1134,69 @@
828 1134
829 1135 // Consider it a semantic match if 70% of keyword parts are present
830 1136 return ($matches / count($keyword_parts)) >= 0.7;
831 1137 }
832 - private function prioritize_suggestions(array $suggestions): array {
1138 + private function prioritize_suggestions(array $suggestions, array $scores = []): array {
1139 + // Map each suggestion back to the factor that emitted it, so priority
1140 + // can rank by the points the factor actually lost instead of keyword-
1141 + // matching the advice text — which sorted a 2-point title tweak above
1142 + // a 6-point thin-content loss and contradicted the row's own impact
1143 + // tag (#408).
1144 + $by_text = [];
1145 + foreach ($scores as $factor => $result) {
1146 + if (!is_array($result) || empty($result['suggestions']) || !is_array($result['suggestions'])) {
1147 + continue;
1148 + }
1149 + $lost = max(0, (float) ($result['max_score'] ?? 0) - (float) ($result['score'] ?? 0));
1150 + foreach ($result['suggestions'] as $text) {
1151 + if (is_string($text) && !isset($by_text[$text])) {
1152 + $by_text[$text] = ['factor' => (string) $factor, 'lost' => $lost];
1153 + }
1154 + }
1155 + }
1156 +
833 1157 $prioritized = [];
834 -
1158 +
835 1159 foreach ($suggestions as $suggestion) {
836 - $priority = $this->determine_suggestion_priority($suggestion);
1160 + $origin = $by_text[$suggestion] ?? null;
1161 +
1162 + // A factor already at full marks loses nothing to this advice —
1163 + // it was occupying list positions (sometimes at "High") while
1164 + // recovering zero points. Dropped rather than sorted last.
1165 + if (null !== $origin && $origin['lost'] <= 0) {
1166 + continue;
1167 + }
1168 +
1169 + if (null !== $origin) {
1170 + $priority = $origin['lost'] >= 4 ? 'High' : ($origin['lost'] >= 2 ? 'Medium' : 'Low');
1171 + } else {
1172 + // No factor attached (defensive: a filter-added or legacy
1173 + // suggestion) — the old keyword map is the fallback.
1174 + $priority = $this->determine_suggestion_priority($suggestion);
1175 + }
1176 +
837 1177 $prioritized[] = [
838 1178 'text' => $suggestion,
839 1179 'priority' => $priority,
840 1180 'impact' => $this->estimate_impact($suggestion),
841 1181 'effort' => $this->estimate_effort($suggestion),
1182 + 'factor' => $origin['factor'] ?? null,
1183 + 'points_recoverable' => $origin['lost'] ?? null,
842 1184 ];
843 1185 }
844 -
845 - // Sort by priority (High > Medium > Low)
1186 +
1187 + // Biggest recoverable loss first; keyword-mapped stragglers (no
1188 + // factor) sort within their priority band after the measured rows.
846 1189 usort($prioritized, function($a, $b) {
1190 + $al = $a['points_recoverable'] ?? -1;
1191 + $bl = $b['points_recoverable'] ?? -1;
1192 + if ($al !== $bl) {
1193 + return $bl <=> $al;
1194 + }
847 1195 $priority_order = ['High' => 3, 'Medium' => 2, 'Low' => 1];
848 1196 return $priority_order[$b['priority']] - $priority_order[$a['priority']];
849 1197 });
850 -
1198 +
851 1199 return $prioritized;
852 1200 }
853 1201
854 1202 /**
@@ -915,14 +1263,16 @@
915 1263 if (empty($content) || empty($target_keyword)) {
916 1264 return 0.0;
917 1265 }
918 1266
919 - $content_lower = strtolower(wp_strip_all_tags($content));
1267 + // Occurrences and the word count both come from the reading text, so
1268 + // shortcode syntax is in neither.
1269 + $content_lower = strtolower(self::reading_text_of($content));
920 1270 $keyword_lower = strtolower($target_keyword);
921 1271
922 1272 // Calculate keyword and semantic term frequency
923 1273 $keyword_count = substr_count($content_lower, $keyword_lower);
924 - $word_count = $this->calculate_word_count_js_style($content_lower);
1274 + $word_count = \ThinkRank\SEO\Word_Count_Index::count_words($content_lower);
925 1275
926 1276 if ($word_count === 0) {
927 1277 return 0.0;
928 1278 }
@@ -1001,10 +1351,13 @@
1001 1351 private function keyword_density(string $content, string $target_keyword): float {
1002 1352 if (empty($content) || empty($target_keyword)) {
1003 1353 return 0.0;
1004 1354 }
1005 - $plain = strtolower(wp_strip_all_tags($content));
1006 - $word_count = $this->calculate_word_count_js_style($plain);
1355 + // Numerator and denominator from the same reading text. Counting the
1356 + // keyword in wp_strip_all_tags() output found it inside shortcode
1357 + // attributes the word count no longer includes.
1358 + $plain = strtolower(self::reading_text_of($content));
1359 + $word_count = \ThinkRank\SEO\Word_Count_Index::count_words($plain);
1007 1360 if ($word_count === 0) {
1008 1361 return 0.0;
1009 1362 }
1010 1363 $occurrences = substr_count($plain, strtolower($target_keyword));
@@ -1039,10 +1392,10 @@
1039 1392 private function keyword_in_first_paragraph(string $content, string $target_keyword): bool {
1040 1393 if (empty($content) || empty($target_keyword)) {
1041 1394 return false;
1042 1395 }
1043 - $plain = strtolower(wp_strip_all_tags($content));
1044 - $words = preg_split('/\s+/', trim($plain), -1, PREG_SPLIT_NO_EMPTY) ?: [];
1396 + $plain = strtolower(self::reading_text_of($content));
1397 + $words = preg_split('/\s+/', $plain, -1, PREG_SPLIT_NO_EMPTY) ?: [];
1045 1398 $window = array_slice($words, 0, max(50, (int) ceil(count($words) * 0.1)));
1046 1399 return strpos(implode(' ', $window), strtolower($target_keyword)) !== false;
1047 1400 }
1048 1401
@@ -1272,9 +1625,24 @@
1272 1625 if (!class_exists('\ThinkRank\SEO\Builder_Content')) {
1273 1626 require_once THINKRANK_PLUGIN_DIR . 'includes/seo/class-builder-content.php';
1274 1627 }
1275 1628
1276 - return \ThinkRank\SEO\Builder_Content::resolve_markup($live_content, $post);
1629 + // Bind the live markup to the post the render makes current. Since #862
1630 + // the resolver runs setup_postdata() on this post, so a shortcode that
1631 + // builds its output from the current post's content — get_the_content(),
1632 + // get_post()->post_content, as a table of contents or a reading-time
1633 + // shortcode does — otherwise read the last saved body while the unsaved
1634 + // markup rendered around it, and lagged a save behind (#864).
1635 + //
1636 + // A clone, not the caller's object: the post is current only for the
1637 + // duration of the render and the caller's $post must come back
1638 + // unchanged. `thinkrank_analyzable_content` receives the clone too,
1639 + // which is what makes $post->post_content there agree with the markup
1640 + // being analyzed on the live path.
1641 + $bound = clone $post;
1642 + $bound->post_content = $live_content;
1643 +
1644 + return \ThinkRank\SEO\Builder_Content::resolve_markup($live_content, $bound);
1277 1645 }
1278 1646
1279 1647 public function analyze_post_content(int $post_id): array {
1280 1648 $post = get_post($post_id);
@@ -1287,11 +1655,12 @@
1287 1655
1288 1656 // Extract headings from content
1289 1657 $headings = $this->extract_headings($content);
1290 1658
1291 - // Count words using JavaScript-compatible method
1292 - $plain_text = wp_strip_all_tags($content);
1293 - $word_count = $this->calculate_word_count_js_style($plain_text);
1659 + // Count the markup, not wp_strip_all_tags() output: stripping deletes
1660 + // tags without a space, so "five</p><p>six" would already be one word
1661 + // before the counter saw it.
1662 + $word_count = $this->calculate_word_count_js_style($content);
1294 1663
1295 1664 // Calculate readability
1296 1665 $readability_score = $this->calculate_readability_score($content);
1297 1666
@@ -1411,9 +1780,9 @@
1411 1780 // 2. Long paragraphs (flag paragraphs over 150 words).
1412 1781 $long_paragraphs = 0;
1413 1782 if (preg_match_all('/<p[^>]*>(.*?)<\/p>/is', $content, $matches)) {
1414 1783 foreach ($matches[1] as $paragraph) {
1415 - if ($this->calculate_word_count_js_style(wp_strip_all_tags($paragraph)) > 150) {
1784 + if ($this->calculate_word_count_js_style($paragraph) > 150) {
1416 1785 $long_paragraphs++;
1417 1786 }
1418 1787 }
1419 1788 }
@@ -1468,22 +1837,22 @@
1468 1837 * @param string $content Content text
1469 1838 * @return float Readability score
1470 1839 */
1471 1840 private function calculate_readability_score(string $content): float {
1472 - $text = wp_strip_all_tags($content);
1841 + // Sentences, words and syllables all come from one reading text. The
1842 + // word count drops shortcode syntax and punctuation-only tokens; when
1843 + // sentences and syllables were still read off wp_strip_all_tags()
1844 + // output, "[vc_column width="1/2"]" added syllables (and, with a "." in
1845 + // an attribute, sentences) to a word count that did not include it,
1846 + // and a WPBakery page's Flesch fell from 65 to 46.
1847 + $text = self::reading_text_of($content);
1473 1848
1474 - if (empty($text)) {
1849 + if ('' === $text) {
1475 1850 return 0;
1476 1851 }
1477 1852
1478 - // Count sentences (approximate)
1479 - $sentences = preg_split('/[.!?]+/', $text, -1, PREG_SPLIT_NO_EMPTY);
1480 - $sentence_count = count($sentences);
1481 -
1482 - // Count words
1483 - $word_count = $this->calculate_word_count_js_style(wp_strip_all_tags($text));
1484 -
1485 - // Count syllables (approximate)
1853 + $sentence_count = self::count_sentences($text);
1854 + $word_count = \ThinkRank\SEO\Word_Count_Index::count_words($text);
1486 1855 $syllable_count = $this->count_syllables($text);
1487 1856
1488 1857 if ($sentence_count === 0 || $word_count === 0) {
1489 1858 return 0;
@@ -1495,19 +1864,59 @@
1495 1864 return max(0, min(100, $score));
1496 1865 }
1497 1866
1498 1867 /**
1868 + * Count the sentences in reading text.
1869 + *
1870 + * A sentence is a run of text between terminal punctuation that holds at
1871 + * least one letter or digit. A fragment with no word in it (the space
1872 + * after the final full stop, a stray "!" between two "?") is not a
1873 + * sentence. contentAnalysis.js countSentences() applies the same rule.
1874 + *
1875 + * @since 2.14.2
1876 + *
1877 + * @param string $text Text as returned by {@see self::reading_text_of()}.
1878 + * @return int
1879 + */
1880 + private static function count_sentences(string $text): int {
1881 + $fragments = preg_split('/[.!?]+/', $text, -1, PREG_SPLIT_NO_EMPTY) ?: [];
1882 +
1883 + return count(preg_grep('/[\p{L}\p{N}]/u', $fragments) ?: []);
1884 + }
1885 +
1886 + /**
1499 1887 * Count syllables in text (approximate)
1500 1888 *
1889 + * Expects reading text ({@see self::reading_text_of()}); tags are not
1890 + * stripped here, so a decoded "<" in prose is not read as a tag.
1891 + *
1501 1892 * @param string $text Text to analyze
1502 1893 * @return int Syllable count
1503 1894 */
1504 1895 private function count_syllables(string $text): int {
1505 - $words = preg_split('/\s+/', trim(strtolower(wp_strip_all_tags($text))), -1, PREG_SPLIT_NO_EMPTY);
1896 + $words = preg_split('/\s+/', trim(strtolower($text)), -1, PREG_SPLIT_NO_EMPTY) ?: [];
1506 1897 $syllables = 0;
1507 1898
1508 1899 foreach ($words as $word) {
1509 - $syllables += max(1, preg_match_all('/[aeiouy]+/', $word));
1900 + $word = preg_replace('/[^a-z]/', '', $word);
1901 + if ($word === '') {
1902 + continue;
1903 + }
1904 +
1905 + $groups = preg_match_all('/[aeiouy]+/', $word);
1906 +
1907 + // Standard Flesch heuristic: a trailing silent e does not form a
1908 + // syllable ("make", "time", "these") — but only when a consonant
1909 + // precedes it (a vowel+e ending like "movie" already shares its
1910 + // group) and never for consonant-le ("table"), which does count.
1911 + // Without this the counter inflated syllables/word by ~0.2-0.3 on
1912 + // ordinary prose, driving raw Flesch negative and the UI to a
1913 + // clamped "Very Difficult (0)" (#407).
1914 + if ($groups > 1 && preg_match('/[^aeiouy]e$/', $word) && !str_ends_with($word, 'le')) {
1915 + $groups--;
1916 + }
1917 +
1918 + $syllables += max(1, $groups);
1510 1919 }
1511 1920
1512 1921 return $syllables;
1513 1922 }
@@ -1791,20 +2200,101 @@
1791 2200 * @param array $content_data Content analysis data
1792 2201 * @return array Scoring result
1793 2202 */
1794 2203 private function score_mobile_experience(array $content_data): array {
1795 - // Mobile experience is theme/site-level, not controlled by post content.
1796 - // Award full credit (benefit of the doubt) instead of a fixed partial
1797 - // that caps every post's ceiling.
2204 + $max = $this->scoring_factors['mobile_experience'];
2205 +
2206 + // Mobile experience is theme/site-level, not controlled by post content
2207 + // — but the plugin already measures it. When a mobile Lighthouse score
2208 + // has been collected, score against it; the "benefit of the doubt" below
2209 + // is for sites nobody has measured, not for sites measured as slow.
2210 + $performance_score = $this->measured_performance_score();
2211 +
2212 + if ($performance_score === null) {
2213 + return [
2214 + 'score' => $max,
2215 + 'max_score' => $max,
2216 + 'suggestions' => ['Ensure mobile-first design and fast loading on mobile devices'],
2217 + 'details' => ['mobile_score' => 'Assumed adequate', 'measured' => false],
2218 + ];
2219 + }
2220 +
2221 + $score = (int) round($max * $performance_score / 100);
2222 +
1798 2223 return [
1799 - 'score' => $this->scoring_factors['mobile_experience'],
1800 - 'max_score' => $this->scoring_factors['mobile_experience'],
1801 - 'suggestions' => ['Ensure mobile-first design and fast loading on mobile devices'],
1802 - 'details' => ['mobile_score' => 'Assumed adequate'],
2224 + 'score' => $score,
2225 + 'max_score' => $max,
2226 + 'suggestions' => $score < $max
2227 + ? ['Improve mobile page speed: the last PageSpeed run scored ' . $performance_score . '/100 on mobile']
2228 + : [],
2229 + 'details' => [
2230 + 'mobile_score' => $performance_score,
2231 + 'measured' => true,
2232 + 'source' => 'pagespeed_mobile',
2233 + ],
1803 2234 ];
1804 2235 }
1805 2236
1806 2237 /**
2238 + * The last collected mobile Lighthouse score, or null when unmeasured.
2239 + *
2240 + * Memoised per instance: compute_score() asks twice, and a post-list screen
2241 + * scores a page of posts at a time.
2242 + *
2243 + * Every failure — no performance module, no collected row, an unreadable
2244 + * table — resolves to null, which the callers read as "not measured" and
2245 + * answer with the full-credit fallback. A site is never penalised for
2246 + * ThinkRank being unable to look.
2247 + *
2248 + * @since 2.3.1
2249 + * @return int|null Score 0-100, or null when nothing has been collected.
2250 + */
2251 + private function measured_performance_score(): ?int {
2252 + $measurement = $this->measured_performance();
2253 +
2254 + if ($measurement === null || !isset($measurement['performance_score'])) {
2255 + return null;
2256 + }
2257 +
2258 + $score = $measurement['performance_score'];
2259 +
2260 + if (!is_numeric($score)) {
2261 + return null;
2262 + }
2263 +
2264 + return (int) round(max(0, min(100, (float) $score)));
2265 + }
2266 +
2267 + /**
2268 + * The last collected mobile measurement, or null when there is none.
2269 + *
2270 + * @since 2.3.1
2271 + * @return array|null { core_web_vitals: array, performance_score: float|null }
2272 + */
2273 + private function measured_performance(): ?array {
2274 + if ($this->measured_performance_resolved) {
2275 + return $this->measured_performance;
2276 + }
2277 +
2278 + $this->measured_performance_resolved = true;
2279 +
2280 + if (!class_exists('ThinkRank\\SEO\\Performance_Monitoring_Manager')) {
2281 + return null;
2282 + }
2283 +
2284 + try {
2285 + $manager = new \ThinkRank\SEO\Performance_Monitoring_Manager();
2286 + // Mobile deliberately: Google indexes mobile-first, and it is the
2287 + // device the mobile_experience factor is named after.
2288 + $this->measured_performance = $manager->get_stored_performance_measurement('mobile');
2289 + } catch (\Throwable $e) {
2290 + $this->measured_performance = null;
2291 + }
2292 +
2293 + return $this->measured_performance;
2294 + }
2295 +
2296 + /**
1807 2297 * Score core web vitals - 2025 version (3 points)
1808 2298 *
1809 2299 * @param array $content_data Content analysis data
1810 2300 * @return array Scoring result
@@ -1809,20 +2299,98 @@
1809 2299 * @param array $content_data Content analysis data
1810 2300 * @return array Scoring result
1811 2301 */
1812 2302 private function score_core_web_vitals(array $content_data): array {
1813 - // Core Web Vitals are a runtime/performance signal, not derivable from
1814 - // post content. Award full credit (benefit of the doubt) rather than a
1815 - // fixed partial that caps every post's ceiling.
2303 + $max = $this->scoring_factors['core_web_vitals'];
2304 +
2305 + // Not derivable from post content — but it is measured, and the audit
2306 + // stores LCP, INP and CLS with a rating each. Score against those when
2307 + // they exist; fall back to the benefit of the doubt when they do not.
2308 + $rated = $this->measured_vitals_score();
2309 +
2310 + if ($rated === null) {
2311 + return [
2312 + 'score' => $max,
2313 + 'max_score' => $max,
2314 + 'suggestions' => ['Optimize Core Web Vitals: LCP, INP, and CLS for better user experience'],
2315 + 'details' => ['vitals_status' => 'Assumed adequate', 'measured' => false],
2316 + ];
2317 + }
2318 +
2319 + $score = (int) round($max * $rated['average'] / 100);
2320 +
1816 2321 return [
1817 - 'score' => $this->scoring_factors['core_web_vitals'],
1818 - 'max_score' => $this->scoring_factors['core_web_vitals'],
1819 - 'suggestions' => ['Optimize Core Web Vitals: LCP, INP, and CLS for better user experience'],
1820 - 'details' => ['vitals_status' => 'Assumed adequate'],
2322 + 'score' => $score,
2323 + 'max_score' => $max,
2324 + // Gated on the measurement, not the rounded score: two good metrics
2325 + // and one needing improvement averages 88.33, which rounds to the
2326 + // full 3 of 3 and used to swallow the suggestion naming the metric
2327 + // that is actually failing.
2328 + 'suggestions' => !empty($rated['failing'])
2329 + ? ['Optimize Core Web Vitals: ' . implode(', ', $rated['failing']) . ' below target on mobile']
2330 + : [],
2331 + 'details' => [
2332 + 'vitals_status' => $rated['statuses'],
2333 + 'measured' => true,
2334 + 'source' => 'pagespeed_mobile',
2335 + ],
1821 2336 ];
1822 2337 }
1823 2338
1824 2339 /**
2340 + * Rate the collected Core Web Vitals, or null when none were measured.
2341 + *
2342 + * Reuses the per-metric score the performance module already assigns
2343 + * (good 100, needs improvement 65, poor 30) rather than inventing a second
2344 + * scale, so the SEO score and the performance card cannot disagree about
2345 + * whether a metric is healthy.
2346 + *
2347 + * Metrics with no stored value — fcp is not always collected — are skipped
2348 + * rather than counted as failures.
2349 + *
2350 + * @since 2.3.1
2351 + * @return array|null { average: float, statuses: array, failing: string[] }
2352 + */
2353 + private function measured_vitals_score(): ?array {
2354 + $measurement = $this->measured_performance();
2355 + $vitals = $measurement['core_web_vitals'] ?? null;
2356 +
2357 + if (!is_array($vitals)) {
2358 + return null;
2359 + }
2360 +
2361 + $scores = [];
2362 + $statuses = [];
2363 + $failing = [];
2364 +
2365 + // The three Google ranks on. fcp is diagnostic and not a Core Web Vital.
2366 + foreach (['lcp', 'inp', 'cls'] as $metric) {
2367 + $data = $vitals[$metric] ?? null;
2368 +
2369 + if (!is_array($data) || !isset($data['value'], $data['score']) || $data['value'] === null) {
2370 + continue;
2371 + }
2372 +
2373 + $scores[] = (float) $data['score'];
2374 + $statuses[$metric] = $data['status'] ?? 'unknown';
2375 +
2376 + if (($data['status'] ?? '') !== 'good') {
2377 + $failing[] = strtoupper($metric);
2378 + }
2379 + }
2380 +
2381 + if (empty($scores)) {
2382 + return null;
2383 + }
2384 +
2385 + return [
2386 + 'average' => array_sum($scores) / count($scores),
2387 + 'statuses' => $statuses,
2388 + 'failing' => $failing,
2389 + ];
2390 + }
2391 +
2392 + /**
1825 2393 * Score internal linking - declining importance (1 point)
1826 2394 *
1827 2395 * @param array $content_data Content analysis data
1828 2396 * @return array Scoring result
@@ -1852,9 +2420,9 @@
1852 2420 $suggestions = [];
1853 2421
1854 2422 // Meta description check
1855 2423 $meta_desc = $metadata['description'] ?? '';
1856 - if (!empty($meta_desc) && mb_strlen($meta_desc) >= 120 && mb_strlen($meta_desc) <= 160) {
2424 + if (!empty($meta_desc) && mb_strlen($meta_desc) >= self::DESCRIPTION_OPTIMAL_MIN && mb_strlen($meta_desc) <= self::DESCRIPTION_OPTIMAL_MAX) {
1857 2425 $score += 0.5;
1858 2426 } else {
1859 2427 $suggestions[] = 'Add a compelling meta description (120-160 characters)';
1860 2428 }
@@ -1936,11 +2504,12 @@
1936 2504
1937 2505 // Extract headings from content
1938 2506 $headings = $this->extract_headings($content);
1939 2507
1940 - // Count words using JavaScript-compatible method
1941 - $plain_text = wp_strip_all_tags($content);
1942 - $word_count = $this->calculate_word_count_js_style($plain_text);
2508 + // Count the markup, not wp_strip_all_tags() output: stripping deletes
2509 + // tags without a space, so "five</p><p>six" would already be one word
2510 + // before the counter saw it.
2511 + $word_count = $this->calculate_word_count_js_style($content);
1943 2512
1944 2513 // Calculate readability
1945 2514 $readability_score = $this->calculate_readability_score($content);
1946 2515
@@ -1974,22 +2543,59 @@
1974 2543 ];
1975 2544 }
1976 2545
1977 2546 /**
1978 - * Calculate word count using JavaScript-compatible method
1979 - * Matches the logic in contentAnalysis.js for consistency
2547 + * Count the words a reader reads in HTML or text.
1980 2548 *
1981 - * @param string $text Text to count words in
2549 + * The same extraction and counting as the thin content report
2550 + * ({@see \ThinkRank\SEO\Word_Count_Index::reading_text()} and
2551 + * {@see \ThinkRank\SEO\Word_Count_Index::count_words()}), so the editor
2552 + * score and the report give one number for one page. Splitting on
2553 + * whitespace counted shortcode syntax as words: a WPBakery page with 216
2554 + * words of prose scored a word count of 325 while the report said 216
2555 + * (#893 fixed the report only). It also counted tokens of punctuation
2556 + * alone, such as a full stop after a link.
2557 + *
2558 + * Always words, whatever the locale's unit, because every threshold that
2559 + * reads this value (content length, long paragraphs, headings per 300
2560 + * words, readability, keyword density) is in words.
2561 + *
2562 + * contentAnalysis.js calculateWordCount() applies the same two rules in
2563 + * the editor.
2564 + *
2565 + * @param string $text HTML or text to count words in.
1982 2566 * @return int Word count
1983 2567 */
1984 2568 private function calculate_word_count_js_style(string $text): int {
1985 - if (empty($text)) {
2569 + if ('' === $text) {
1986 2570 return 0;
1987 2571 }
1988 2572
1989 - // Match JavaScript: trim, split by whitespace, filter empty
1990 - $words = preg_split('/\s+/', trim($text), -1, PREG_SPLIT_NO_EMPTY);
1991 - return count($words);
2573 + return \ThinkRank\SEO\Word_Count_Index::count_words(self::reading_text_of($text));
2574 + }
2575 +
2576 + /**
2577 + * The text a reader reads in HTML or text, as the word count sees it.
2578 + *
2579 + * Every check that divides by the word count (readability, keyword
2580 + * density, topic relevance) reads its numerator from this same text, so
2581 + * shortcode syntax is never on one side of a ratio and not the other.
2582 + *
2583 + * @since 2.14.2
2584 + *
2585 + * @param string $content HTML or text.
2586 + * @return string Plain text, whitespace collapsed.
2587 + */
2588 + private static function reading_text_of(string $content): string {
2589 + if ('' === $content) {
2590 + return '';
2591 + }
2592 +
2593 + if (!class_exists('\ThinkRank\SEO\Word_Count_Index')) {
2594 + require_once THINKRANK_PLUGIN_DIR . 'includes/seo/class-word-count-index.php';
2595 + }
2596 +
2597 + return \ThinkRank\SEO\Word_Count_Index::reading_text($content);
1992 2598 }
1993 2599
1994 2600 /**
1995 2601 * Get existing score data for a post from database