PluginProbe
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO / 2.14.2
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO v2.14.2
2.14.2 2.14.1 2.14.0 2.13.0 2.12.0 2.11.0 2.10.0 2.9.0 2.8.0 2.7.0 2.6.0 2.5.0 2.4.0 2.3.0 2.2.0 2.1.1 2.1.0 2.0.2 2.0.1 2.0.0 1.32.0 1.31.0 1.30.0 1.29.0 1.28.0 All 57 releases
← All changes | includes/ai/class-seo-score-calculator.php +609 -63 2.0.1 → 2.14.2 View file →
@@ -41,8 +41,41 @@
41 41 */
42 42 private Database $database;
43 43
44 44 /**
45 + * Memoised collected performance measurement, and whether it was resolved.
46 + *
47 + * Two factors read it and both may be asked for on every post in a list, so
48 + * the lookup happens once per calculator. `null` is a real answer here — the
49 + * separate flag keeps "not looked up yet" distinct from "nothing measured".
50 + *
51 + * @since 2.3.1
52 + * @var array|null
53 + */
54 + private ?array $measured_performance = null;
55 +
56 + /**
57 + * @since 2.3.1
58 + * @var bool
59 + */
60 + private bool $measured_performance_resolved = false;
61 +
62 + /**
63 + * Length bands the editor scores against, in characters.
64 + *
65 + * Public so every surface that judges a title or description — the editor
66 + * score and the Bulk Snippets problem filter — reads one set of numbers.
67 + * Before these existed the bands were literals inside the scoring methods,
68 + * and a second screen would have had to copy them and drift (#727).
69 + *
70 + * @since 2.8.0
71 + */
72 + public const TITLE_OPTIMAL_MIN = 35;
73 + public const TITLE_OPTIMAL_MAX = 60;
74 + public const DESCRIPTION_OPTIMAL_MIN = 120;
75 + public const DESCRIPTION_OPTIMAL_MAX = 160;
76 +
77 + /**
45 78 * 2025 SEO scoring factors (Q1 2025 Google Algorithm)
46 79 * Based on First Page Sage research and Google's latest updates
47 80 *
48 81 * @var array
@@ -107,8 +140,9 @@
107 140 'grade' => $result['grade'],
108 141 ]];
109 142 $result['keyword_checks'] = $this->analyze_keyword_checks($content_data, $metadata, $keywords);
110 143 }
144 + $result['keywords'] = $this->keyword_placements($content_data, $metadata, $keywords);
111 145
112 146 return $result;
113 147 }
114 148
@@ -218,8 +252,9 @@
218 252 $best['target_keyword'] = $best_keyword;
219 253 $best['target_keywords'] = $keywords;
220 254 $best['keyword_results'] = $per_keyword;
221 255 $best['keyword_checks'] = $this->analyze_keyword_checks($content_data, $metadata, $keywords);
256 + $best['keywords'] = $this->keyword_placements($content_data, $metadata, $keywords);
222 257
223 258 return $best;
224 259 }
225 260
@@ -234,15 +269,15 @@
234 269 * @param string[] $keywords Target keywords.
235 270 * @return array<string,array{passed:bool,matched_keywords:string[]}>
236 271 */
237 272 private function analyze_keyword_checks(array $content_data, array $metadata, array $keywords): array {
238 - $title = strtolower((string) ($metadata['title'] ?? $content_data['title'] ?? ''));
239 - $description = strtolower((string) ($metadata['description'] ?? ''));
240 - $content = strtolower(wp_strip_all_tags((string) ($content_data['content'] ?? '')));
273 + $title = self::lower((string) ($metadata['title'] ?? $content_data['title'] ?? ''));
274 + $description = self::lower((string) ($metadata['description'] ?? ''));
275 + $content = self::lower(self::plain_text((string) ($content_data['content'] ?? '')));
241 276
242 277 $alts = '';
243 278 foreach ((array) ($content_data['images'] ?? []) as $image) {
244 - $alts .= ' ' . strtolower((string) ($image['alt'] ?? ''));
279 + $alts .= ' ' . self::lower((string) ($image['alt'] ?? ''));
245 280 }
246 281
247 282 // Build a searchable slug haystack from the post's OWN slug — never the
248 283 // full URL path. The path carries ancestors, category bases and date
@@ -251,17 +286,9 @@
251 286 // way: an unpublished post has no pretty permalink (get_permalink()
252 287 // returns ?p=123), so the path held no slug at all and every draft
253 288 // scored "no match" until it was published. Hyphens/underscores become
254 289 // spaces so multi-word keywords can match.
255 - $slug_source = (string) ($content_data['slug'] ?? '');
256 - if ($slug_source === '') {
257 - // Draft with no slug assigned yet: score what WordPress would
258 - // generate from the title, which is what the editor shows as the
259 - // proposed URL — so the check reads the same before and after
260 - // publishing instead of flipping.
261 - $slug_source = sanitize_title((string) ($content_data['title'] ?? ''));
262 - }
263 - $slug = strtolower(str_replace(['-', '_'], ' ', $slug_source));
290 + $slug = self::lower(self::slug_haystack($content_data));
264 291
265 292 $haystacks = [
266 293 'title' => trim($title),
267 294 'meta_description' => trim($description),
@@ -273,10 +300,9 @@
273 300 $checks = [];
274 301 foreach ($haystacks as $location => $haystack) {
275 302 $matched = [];
276 303 foreach ($keywords as $keyword) {
277 - $needle = strtolower(trim($keyword));
278 - if ($needle !== '' && $haystack !== '' && strpos($haystack, $needle) !== false) {
304 + if ($this->keyword_matches($haystack, self::lower(trim($keyword)))) {
279 305 $matched[] = $keyword;
280 306 }
281 307 }
282 308 $checks[$location] = [
@@ -288,8 +314,288 @@
288 314 return $checks;
289 315 }
290 316
291 317 /**
318 + * The keyword placements the editor draws one gauge segment for, in the
319 + * order a reader meets them (#729).
320 + *
321 + * @since 2.11.0
322 + * @var string[]
323 + */
324 + public const PLACEMENTS = ['title', 'meta_description', 'slug', 'first_paragraph', 'subheading', 'content', 'image_alt'];
325 +
326 + /**
327 + * Characters of plain text read as the opening when the content has no
328 + * paragraph tag.
329 + *
330 + * @since 2.11.0
331 + */
332 + private const OPENING_CHARS = 300;
333 +
334 + /**
335 + * Where each focus keyword is placed, keyword by keyword (#729).
336 + *
337 + * analyze_keyword_checks() answers "does ANY keyword appear here" for
338 + * five places; this answers "where does THIS keyword appear" for seven,
339 + * so the editor can show each keyword's own gauge. Same matcher, so a
340 + * keyword counts as a word (not a fragment) and a keyword in a script
341 + * written without spaces (Thai, Chinese, Japanese) still matches.
342 + *
343 + * `where` names the heading or alt text that matched, so the editor can
344 + * say which one.
345 + *
346 + * @since 2.11.0
347 + *
348 + * @param array $content_data Content analysis data.
349 + * @param array $metadata Post metadata (title, description).
350 + * @param string[] $keywords Focus keywords.
351 + * @return array<int, array{keyword: string, passed: int, total: int, placements: array<string, array{passed: bool, where: string}>}>
352 + */
353 + public function keyword_placements(array $content_data, array $metadata, array $keywords): array {
354 + $html = (string) ($content_data['content'] ?? '');
355 + $plain = self::lower(self::plain_text($html));
356 +
357 + $headings = [];
358 + $source = isset($content_data['headings']) && is_array($content_data['headings']) ? $content_data['headings'] : $this->extract_headings($html);
359 + foreach ($source as $heading) {
360 + $text = trim((string) ($heading['text'] ?? ''));
361 + if ((int) ($heading['level'] ?? 0) >= 2 && '' !== $text) {
362 + $headings[] = $text;
363 + }
364 + }
365 +
366 + $alts = [];
367 + foreach ((array) ($content_data['images'] ?? []) as $image) {
368 + $alt = trim((string) ($image['alt'] ?? ''));
369 + if ('' !== $alt) {
370 + $alts[] = $alt;
371 + }
372 + }
373 +
374 + $single = [
375 + 'title' => self::lower((string) ($metadata['title'] ?? $content_data['title'] ?? '')),
376 + 'meta_description' => self::lower((string) ($metadata['description'] ?? '')),
377 + 'slug' => self::lower(self::slug_haystack($content_data)),
378 + 'first_paragraph' => self::lower(self::opening($html)),
379 + 'content' => $plain,
380 + ];
381 +
382 + $out = [];
383 + foreach ($keywords as $keyword) {
384 + $needle = self::lower(trim((string) $keyword));
385 + $placements = [];
386 +
387 + foreach (self::PLACEMENTS as $placement) {
388 + if ('subheading' === $placement || 'image_alt' === $placement) {
389 + $where = '';
390 + foreach ('subheading' === $placement ? $headings : $alts as $text) {
391 + if ($this->keyword_matches(self::lower($text), $needle)) {
392 + $where = $text;
393 + break;
394 + }
395 + }
396 + $placements[$placement] = ['passed' => '' !== $where, 'where' => $where];
397 + continue;
398 + }
399 +
400 + $placements[$placement] = ['passed' => $this->keyword_matches($single[$placement], $needle), 'where' => ''];
401 + }
402 +
403 + $out[] = [
404 + 'keyword' => (string) $keyword,
405 + 'passed' => count(array_filter(array_column($placements, 'passed'))),
406 + 'total' => count(self::PLACEMENTS),
407 + 'placements' => $placements,
408 + ];
409 + }
410 +
411 + return $out;
412 + }
413 +
414 + /**
415 + * The post's own slug as searchable text: hyphens and underscores become
416 + * spaces so a multi-word keyword can match, and a slug WordPress
417 + * percent-encoded (Thai, Cyrillic, Chinese) is decoded, or it could never
418 + * match a keyword typed in that script.
419 + *
420 + * Never the full URL path: the path carries ancestors, category bases and
421 + * date segments, so a child of /clinical-trials/ reported "keyword in slug"
422 + * for a page actually slugged `contact-us`. An unpublished post has no
423 + * pretty permalink either, so the path held no slug at all.
424 + *
425 + * @param array $content_data Content analysis data.
426 + * @return string
427 + */
428 + private static function slug_haystack(array $content_data): string {
429 + $slug = (string) ($content_data['slug'] ?? '');
430 + if ('' === $slug) {
431 + // Draft with no slug assigned yet: score what WordPress would
432 + // generate from the title, which is what the editor shows as the
433 + // proposed URL — so the check reads the same before and after
434 + // publishing instead of flipping.
435 + $slug = sanitize_title((string) ($content_data['title'] ?? ''));
436 + }
437 +
438 + return trim(str_replace(['-', '_'], ' ', rawurldecode($slug)));
439 + }
440 +
441 + /**
442 + * The opening of the content: its first paragraph with text, or the first
443 + * few hundred characters when it has none. Counted in characters, not
444 + * words, so a language written without spaces is not read as one word.
445 + *
446 + * @param string $html Content HTML.
447 + * @return string Plain text.
448 + */
449 + private static function opening(string $html): string {
450 + if (preg_match_all('/<p\b[^>]*>(.*?)<\/p>/isu', $html, $matches)) {
451 + foreach ($matches[1] as $paragraph) {
452 + $text = self::collapse_whitespace(wp_strip_all_tags($paragraph));
453 + if ('' !== $text) {
454 + return $text;
455 + }
456 + }
457 + }
458 +
459 + $text = self::plain_text($html);
460 +
461 + return function_exists('mb_substr') ? mb_substr($text, 0, self::OPENING_CHARS) : substr($text, 0, self::OPENING_CHARS);
462 + }
463 +
464 + /**
465 + * Content as plain text, with a space where each tag was. wp_strip_all_tags()
466 + * alone joins neighbouring blocks — "…coffee grinder</h3><p>A good…" became
467 + * "coffee grinderA good" — so a keyword at the end of a heading or a
468 + * paragraph was no longer a word and did not match.
469 + *
470 + * @param string $html Content HTML.
471 + * @return string
472 + */
473 + private static function plain_text(string $html): string {
474 + $spaced = preg_replace('/<[^>]+>/', ' $0 ', $html);
475 +
476 + return self::collapse_whitespace(wp_strip_all_tags(null === $spaced ? $html : (string) $spaced));
477 + }
478 +
479 + /**
480 + * Runs of whitespace down to one space.
481 + *
482 + * The `/u` pass is the one that understands a multibyte space, but
483 + * preg_replace() answers null on bytes that are not valid UTF-8 rather than
484 + * throwing — and casting that null to a string blanked the haystack, so a
485 + * post carrying one mojibake byte (a Latin-1 paste, an old import) reported
486 + * every keyword as missing from its body, its opening, and every
487 + * subheading. The gauge said 0/7 and told the author to add a keyword that
488 + * was already there.
489 + *
490 + * Falls back to the byte-wise collapse, which is what this did before the
491 + * multibyte work added the modifier. Same reasoning keyword_matches()
492 + * already records for its own PCRE failure: a pattern PCRE refuses must not
493 + * be reported as a confident "no match".
494 + *
495 + * @param string $text Text to collapse.
496 + * @return string
497 + */
498 + private static function collapse_whitespace(string $text): string {
499 + $collapsed = preg_replace('/\s+/u', ' ', $text);
500 +
501 + if (null === $collapsed) {
502 + $collapsed = preg_replace('/\s+/', ' ', $text);
503 + }
504 +
505 + return trim(null === $collapsed ? $text : (string) $collapsed);
506 + }
507 +
508 + /**
509 + * Lowercase in any script. strtolower() only folds ASCII, so "Кофе" never
510 + * matched "кофе".
511 + *
512 + * @param string $text Text.
513 + * @return string
514 + */
515 + private static function lower(string $text): string {
516 + return function_exists('mb_strtolower') ? mb_strtolower($text, 'UTF-8') : strtolower($text);
517 + }
518 +
519 + /**
520 + * Scripts written without spaces between words.
521 + *
522 + * @since 2.1.0
523 + * @var string
524 + */
525 + private const SCRIPTIO_CONTINUA = '/[\p{Han}\p{Hiragana}\p{Katakana}\p{Thai}\p{Lao}\p{Khmer}\p{Myanmar}]/u';
526 +
527 + /**
528 + * Whether a keyword appears in a haystack as a word rather than as a
529 + * fragment of a longer one.
530 + *
531 + * The five keyword checks used a plain strpos(), so any substring hit
532 + * counted: "test coronavirus" matched "la|test coronavirus|news", "art"
533 + * matched "start", "cat" matched "category". The panel then confidently
534 + * reported a keyword placement that does not exist (#416). Same class of
535 + * problem #71 fixed in the Image SEO rewriter, and the same remedy.
536 + *
537 + * Both arguments are expected lowercased already.
538 + *
539 + * @since 2.1.0
540 + *
541 + * @param string $haystack Text to search.
542 + * @param string $needle Keyword, lowercased and trimmed.
543 + * @return bool
544 + */
545 + private function keyword_matches(string $haystack, string $needle): bool {
546 + if ($needle === '' || $haystack === '') {
547 + return false;
548 + }
549 +
550 + if (!$this->supports_word_boundaries($needle)) {
551 + return strpos($haystack, $needle) !== false;
552 + }
553 +
554 + $matched = preg_match('/\b' . preg_quote($needle, '/') . '\b/u', $haystack);
555 +
556 + // PCRE refusing the pattern — invalid UTF-8 in the keyword, a
557 + // backtrack limit — must not be reported as a confident "no match".
558 + // Fall back to the behaviour this replaced rather than invent a
559 + // negative the user cannot explain.
560 + if ($matched === false) {
561 + return strpos($haystack, $needle) !== false;
562 + }
563 +
564 + return $matched === 1;
565 + }
566 +
567 + /**
568 + * Whether \b can express "this keyword, as a word" for this keyword.
569 + *
570 + * It asserts a transition between a word and a non-word character, which
571 + * only means something where words are separated. Two cases where it is
572 + * not, both verified against PCRE rather than assumed:
573 + *
574 + * - the keyword's own edges are not word characters ("c++", "#seo"), so
575 + * no boundary can assert there and a real match is lost;
576 + * - scripts written without spaces, where the neighbouring characters
577 + * are word characters too — "冠状病毒" inside "最新冠状病毒新闻" is a
578 + * legitimate match that \b never sees.
579 + *
580 + * Accented Latin and Cyrillic need no special handling: PHP's /u modifier
581 + * turns on Unicode character properties, so "café" correctly does not
582 + * match "cafés" and "коронавирус" does not match "коронавирусный".
583 + *
584 + * @since 2.1.0
585 + *
586 + * @param string $needle Keyword, lowercased and trimmed.
587 + * @return bool
588 + */
589 + private function supports_word_boundaries(string $needle): bool {
590 + if (preg_match(self::SCRIPTIO_CONTINUA, $needle)) {
591 + return false;
592 + }
593 +
594 + return preg_match('/^\w/u', $needle) === 1 && preg_match('/\w$/u', $needle) === 1;
595 + }
596 +
597 + /**
292 598 * Compute the SEO score for a single target keyword.
293 599 *
294 600 * @param array $content_data Content analysis data
295 601 * @param array $metadata Post metadata
@@ -552,9 +858,9 @@
552 858 $title_length = mb_strlen($title);
553 859
554 860 // 2025 length optimization (6 points). 60 characters is the recommended
555 861 // maximum for best SERP visibility before Google truncates the title.
556 - if ($title_length >= 35 && $title_length <= 60) {
862 + if ($title_length >= self::TITLE_OPTIMAL_MIN && $title_length <= self::TITLE_OPTIMAL_MAX) {
557 863 $score += 6;
558 864 } elseif ($title_length >= 25 && $title_length <= 75) {
559 865 $score += 4;
560 866 $suggestions[] = 'Optimize title length to 35-60 characters for better SERP visibility';
@@ -957,14 +1263,16 @@
957 1263 if (empty($content) || empty($target_keyword)) {
958 1264 return 0.0;
959 1265 }
960 1266
961 - $content_lower = strtolower(wp_strip_all_tags($content));
1267 + // Occurrences and the word count both come from the reading text, so
1268 + // shortcode syntax is in neither.
1269 + $content_lower = strtolower(self::reading_text_of($content));
962 1270 $keyword_lower = strtolower($target_keyword);
963 1271
964 1272 // Calculate keyword and semantic term frequency
965 1273 $keyword_count = substr_count($content_lower, $keyword_lower);
966 - $word_count = $this->calculate_word_count_js_style($content_lower);
1274 + $word_count = \ThinkRank\SEO\Word_Count_Index::count_words($content_lower);
967 1275
968 1276 if ($word_count === 0) {
969 1277 return 0.0;
970 1278 }
@@ -1043,10 +1351,13 @@
1043 1351 private function keyword_density(string $content, string $target_keyword): float {
1044 1352 if (empty($content) || empty($target_keyword)) {
1045 1353 return 0.0;
1046 1354 }
1047 - $plain = strtolower(wp_strip_all_tags($content));
1048 - $word_count = $this->calculate_word_count_js_style($plain);
1355 + // Numerator and denominator from the same reading text. Counting the
1356 + // keyword in wp_strip_all_tags() output found it inside shortcode
1357 + // attributes the word count no longer includes.
1358 + $plain = strtolower(self::reading_text_of($content));
1359 + $word_count = \ThinkRank\SEO\Word_Count_Index::count_words($plain);
1049 1360 if ($word_count === 0) {
1050 1361 return 0.0;
1051 1362 }
1052 1363 $occurrences = substr_count($plain, strtolower($target_keyword));
@@ -1081,10 +1392,10 @@
1081 1392 private function keyword_in_first_paragraph(string $content, string $target_keyword): bool {
1082 1393 if (empty($content) || empty($target_keyword)) {
1083 1394 return false;
1084 1395 }
1085 - $plain = strtolower(wp_strip_all_tags($content));
1086 - $words = preg_split('/\s+/', trim($plain), -1, PREG_SPLIT_NO_EMPTY) ?: [];
1396 + $plain = strtolower(self::reading_text_of($content));
1397 + $words = preg_split('/\s+/', $plain, -1, PREG_SPLIT_NO_EMPTY) ?: [];
1087 1398 $window = array_slice($words, 0, max(50, (int) ceil(count($words) * 0.1)));
1088 1399 return strpos(implode(' ', $window), strtolower($target_keyword)) !== false;
1089 1400 }
1090 1401
@@ -1314,9 +1625,24 @@
1314 1625 if (!class_exists('\ThinkRank\SEO\Builder_Content')) {
1315 1626 require_once THINKRANK_PLUGIN_DIR . 'includes/seo/class-builder-content.php';
1316 1627 }
1317 1628
1318 - return \ThinkRank\SEO\Builder_Content::resolve_markup($live_content, $post);
1629 + // Bind the live markup to the post the render makes current. Since #862
1630 + // the resolver runs setup_postdata() on this post, so a shortcode that
1631 + // builds its output from the current post's content — get_the_content(),
1632 + // get_post()->post_content, as a table of contents or a reading-time
1633 + // shortcode does — otherwise read the last saved body while the unsaved
1634 + // markup rendered around it, and lagged a save behind (#864).
1635 + //
1636 + // A clone, not the caller's object: the post is current only for the
1637 + // duration of the render and the caller's $post must come back
1638 + // unchanged. `thinkrank_analyzable_content` receives the clone too,
1639 + // which is what makes $post->post_content there agree with the markup
1640 + // being analyzed on the live path.
1641 + $bound = clone $post;
1642 + $bound->post_content = $live_content;
1643 +
1644 + return \ThinkRank\SEO\Builder_Content::resolve_markup($live_content, $bound);
1319 1645 }
1320 1646
1321 1647 public function analyze_post_content(int $post_id): array {
1322 1648 $post = get_post($post_id);
@@ -1329,11 +1655,12 @@
1329 1655
1330 1656 // Extract headings from content
1331 1657 $headings = $this->extract_headings($content);
1332 1658
1333 - // Count words using JavaScript-compatible method
1334 - $plain_text = wp_strip_all_tags($content);
1335 - $word_count = $this->calculate_word_count_js_style($plain_text);
1659 + // Count the markup, not wp_strip_all_tags() output: stripping deletes
1660 + // tags without a space, so "five</p><p>six" would already be one word
1661 + // before the counter saw it.
1662 + $word_count = $this->calculate_word_count_js_style($content);
1336 1663
1337 1664 // Calculate readability
1338 1665 $readability_score = $this->calculate_readability_score($content);
1339 1666
@@ -1453,9 +1780,9 @@
1453 1780 // 2. Long paragraphs (flag paragraphs over 150 words).
1454 1781 $long_paragraphs = 0;
1455 1782 if (preg_match_all('/<p[^>]*>(.*?)<\/p>/is', $content, $matches)) {
1456 1783 foreach ($matches[1] as $paragraph) {
1457 - if ($this->calculate_word_count_js_style(wp_strip_all_tags($paragraph)) > 150) {
1784 + if ($this->calculate_word_count_js_style($paragraph) > 150) {
1458 1785 $long_paragraphs++;
1459 1786 }
1460 1787 }
1461 1788 }
@@ -1510,22 +1837,22 @@
1510 1837 * @param string $content Content text
1511 1838 * @return float Readability score
1512 1839 */
1513 1840 private function calculate_readability_score(string $content): float {
1514 - $text = wp_strip_all_tags($content);
1841 + // Sentences, words and syllables all come from one reading text. The
1842 + // word count drops shortcode syntax and punctuation-only tokens; when
1843 + // sentences and syllables were still read off wp_strip_all_tags()
1844 + // output, "[vc_column width="1/2"]" added syllables (and, with a "." in
1845 + // an attribute, sentences) to a word count that did not include it,
1846 + // and a WPBakery page's Flesch fell from 65 to 46.
1847 + $text = self::reading_text_of($content);
1515 1848
1516 - if (empty($text)) {
1849 + if ('' === $text) {
1517 1850 return 0;
1518 1851 }
1519 1852
1520 - // Count sentences (approximate)
1521 - $sentences = preg_split('/[.!?]+/', $text, -1, PREG_SPLIT_NO_EMPTY);
1522 - $sentence_count = count($sentences);
1523 -
1524 - // Count words
1525 - $word_count = $this->calculate_word_count_js_style(wp_strip_all_tags($text));
1526 -
1527 - // Count syllables (approximate)
1853 + $sentence_count = self::count_sentences($text);
1854 + $word_count = \ThinkRank\SEO\Word_Count_Index::count_words($text);
1528 1855 $syllable_count = $this->count_syllables($text);
1529 1856
1530 1857 if ($sentence_count === 0 || $word_count === 0) {
1531 1858 return 0;
@@ -1537,15 +1864,37 @@
1537 1864 return max(0, min(100, $score));
1538 1865 }
1539 1866
1540 1867 /**
1868 + * Count the sentences in reading text.
1869 + *
1870 + * A sentence is a run of text between terminal punctuation that holds at
1871 + * least one letter or digit. A fragment with no word in it (the space
1872 + * after the final full stop, a stray "!" between two "?") is not a
1873 + * sentence. contentAnalysis.js countSentences() applies the same rule.
1874 + *
1875 + * @since 2.14.2
1876 + *
1877 + * @param string $text Text as returned by {@see self::reading_text_of()}.
1878 + * @return int
1879 + */
1880 + private static function count_sentences(string $text): int {
1881 + $fragments = preg_split('/[.!?]+/', $text, -1, PREG_SPLIT_NO_EMPTY) ?: [];
1882 +
1883 + return count(preg_grep('/[\p{L}\p{N}]/u', $fragments) ?: []);
1884 + }
1885 +
1886 + /**
1541 1887 * Count syllables in text (approximate)
1542 1888 *
1889 + * Expects reading text ({@see self::reading_text_of()}); tags are not
1890 + * stripped here, so a decoded "<" in prose is not read as a tag.
1891 + *
1543 1892 * @param string $text Text to analyze
1544 1893 * @return int Syllable count
1545 1894 */
1546 1895 private function count_syllables(string $text): int {
1547 - $words = preg_split('/\s+/', trim(strtolower(wp_strip_all_tags($text))), -1, PREG_SPLIT_NO_EMPTY);
1896 + $words = preg_split('/\s+/', trim(strtolower($text)), -1, PREG_SPLIT_NO_EMPTY) ?: [];
1548 1897 $syllables = 0;
1549 1898
1550 1899 foreach ($words as $word) {
1551 1900 $word = preg_replace('/[^a-z]/', '', $word);
@@ -1851,20 +2200,101 @@
1851 2200 * @param array $content_data Content analysis data
1852 2201 * @return array Scoring result
1853 2202 */
1854 2203 private function score_mobile_experience(array $content_data): array {
1855 - // Mobile experience is theme/site-level, not controlled by post content.
1856 - // Award full credit (benefit of the doubt) instead of a fixed partial
1857 - // that caps every post's ceiling.
2204 + $max = $this->scoring_factors['mobile_experience'];
2205 +
2206 + // Mobile experience is theme/site-level, not controlled by post content
2207 + // — but the plugin already measures it. When a mobile Lighthouse score
2208 + // has been collected, score against it; the "benefit of the doubt" below
2209 + // is for sites nobody has measured, not for sites measured as slow.
2210 + $performance_score = $this->measured_performance_score();
2211 +
2212 + if ($performance_score === null) {
2213 + return [
2214 + 'score' => $max,
2215 + 'max_score' => $max,
2216 + 'suggestions' => ['Ensure mobile-first design and fast loading on mobile devices'],
2217 + 'details' => ['mobile_score' => 'Assumed adequate', 'measured' => false],
2218 + ];
2219 + }
2220 +
2221 + $score = (int) round($max * $performance_score / 100);
2222 +
1858 2223 return [
1859 - 'score' => $this->scoring_factors['mobile_experience'],
1860 - 'max_score' => $this->scoring_factors['mobile_experience'],
1861 - 'suggestions' => ['Ensure mobile-first design and fast loading on mobile devices'],
1862 - 'details' => ['mobile_score' => 'Assumed adequate'],
2224 + 'score' => $score,
2225 + 'max_score' => $max,
2226 + 'suggestions' => $score < $max
2227 + ? ['Improve mobile page speed: the last PageSpeed run scored ' . $performance_score . '/100 on mobile']
2228 + : [],
2229 + 'details' => [
2230 + 'mobile_score' => $performance_score,
2231 + 'measured' => true,
2232 + 'source' => 'pagespeed_mobile',
2233 + ],
1863 2234 ];
1864 2235 }
1865 2236
1866 2237 /**
2238 + * The last collected mobile Lighthouse score, or null when unmeasured.
2239 + *
2240 + * Memoised per instance: compute_score() asks twice, and a post-list screen
2241 + * scores a page of posts at a time.
2242 + *
2243 + * Every failure — no performance module, no collected row, an unreadable
2244 + * table — resolves to null, which the callers read as "not measured" and
2245 + * answer with the full-credit fallback. A site is never penalised for
2246 + * ThinkRank being unable to look.
2247 + *
2248 + * @since 2.3.1
2249 + * @return int|null Score 0-100, or null when nothing has been collected.
2250 + */
2251 + private function measured_performance_score(): ?int {
2252 + $measurement = $this->measured_performance();
2253 +
2254 + if ($measurement === null || !isset($measurement['performance_score'])) {
2255 + return null;
2256 + }
2257 +
2258 + $score = $measurement['performance_score'];
2259 +
2260 + if (!is_numeric($score)) {
2261 + return null;
2262 + }
2263 +
2264 + return (int) round(max(0, min(100, (float) $score)));
2265 + }
2266 +
2267 + /**
2268 + * The last collected mobile measurement, or null when there is none.
2269 + *
2270 + * @since 2.3.1
2271 + * @return array|null { core_web_vitals: array, performance_score: float|null }
2272 + */
2273 + private function measured_performance(): ?array {
2274 + if ($this->measured_performance_resolved) {
2275 + return $this->measured_performance;
2276 + }
2277 +
2278 + $this->measured_performance_resolved = true;
2279 +
2280 + if (!class_exists('ThinkRank\\SEO\\Performance_Monitoring_Manager')) {
2281 + return null;
2282 + }
2283 +
2284 + try {
2285 + $manager = new \ThinkRank\SEO\Performance_Monitoring_Manager();
2286 + // Mobile deliberately: Google indexes mobile-first, and it is the
2287 + // device the mobile_experience factor is named after.
2288 + $this->measured_performance = $manager->get_stored_performance_measurement('mobile');
2289 + } catch (\Throwable $e) {
2290 + $this->measured_performance = null;
2291 + }
2292 +
2293 + return $this->measured_performance;
2294 + }
2295 +
2296 + /**
1867 2297 * Score core web vitals - 2025 version (3 points)
1868 2298 *
1869 2299 * @param array $content_data Content analysis data
1870 2300 * @return array Scoring result
@@ -1869,20 +2299,98 @@
1869 2299 * @param array $content_data Content analysis data
1870 2300 * @return array Scoring result
1871 2301 */
1872 2302 private function score_core_web_vitals(array $content_data): array {
1873 - // Core Web Vitals are a runtime/performance signal, not derivable from
1874 - // post content. Award full credit (benefit of the doubt) rather than a
1875 - // fixed partial that caps every post's ceiling.
2303 + $max = $this->scoring_factors['core_web_vitals'];
2304 +
2305 + // Not derivable from post content — but it is measured, and the audit
2306 + // stores LCP, INP and CLS with a rating each. Score against those when
2307 + // they exist; fall back to the benefit of the doubt when they do not.
2308 + $rated = $this->measured_vitals_score();
2309 +
2310 + if ($rated === null) {
2311 + return [
2312 + 'score' => $max,
2313 + 'max_score' => $max,
2314 + 'suggestions' => ['Optimize Core Web Vitals: LCP, INP, and CLS for better user experience'],
2315 + 'details' => ['vitals_status' => 'Assumed adequate', 'measured' => false],
2316 + ];
2317 + }
2318 +
2319 + $score = (int) round($max * $rated['average'] / 100);
2320 +
1876 2321 return [
1877 - 'score' => $this->scoring_factors['core_web_vitals'],
1878 - 'max_score' => $this->scoring_factors['core_web_vitals'],
1879 - 'suggestions' => ['Optimize Core Web Vitals: LCP, INP, and CLS for better user experience'],
1880 - 'details' => ['vitals_status' => 'Assumed adequate'],
2322 + 'score' => $score,
2323 + 'max_score' => $max,
2324 + // Gated on the measurement, not the rounded score: two good metrics
2325 + // and one needing improvement averages 88.33, which rounds to the
2326 + // full 3 of 3 and used to swallow the suggestion naming the metric
2327 + // that is actually failing.
2328 + 'suggestions' => !empty($rated['failing'])
2329 + ? ['Optimize Core Web Vitals: ' . implode(', ', $rated['failing']) . ' below target on mobile']
2330 + : [],
2331 + 'details' => [
2332 + 'vitals_status' => $rated['statuses'],
2333 + 'measured' => true,
2334 + 'source' => 'pagespeed_mobile',
2335 + ],
1881 2336 ];
1882 2337 }
1883 2338
1884 2339 /**
2340 + * Rate the collected Core Web Vitals, or null when none were measured.
2341 + *
2342 + * Reuses the per-metric score the performance module already assigns
2343 + * (good 100, needs improvement 65, poor 30) rather than inventing a second
2344 + * scale, so the SEO score and the performance card cannot disagree about
2345 + * whether a metric is healthy.
2346 + *
2347 + * Metrics with no stored value — fcp is not always collected — are skipped
2348 + * rather than counted as failures.
2349 + *
2350 + * @since 2.3.1
2351 + * @return array|null { average: float, statuses: array, failing: string[] }
2352 + */
2353 + private function measured_vitals_score(): ?array {
2354 + $measurement = $this->measured_performance();
2355 + $vitals = $measurement['core_web_vitals'] ?? null;
2356 +
2357 + if (!is_array($vitals)) {
2358 + return null;
2359 + }
2360 +
2361 + $scores = [];
2362 + $statuses = [];
2363 + $failing = [];
2364 +
2365 + // The three Google ranks on. fcp is diagnostic and not a Core Web Vital.
2366 + foreach (['lcp', 'inp', 'cls'] as $metric) {
2367 + $data = $vitals[$metric] ?? null;
2368 +
2369 + if (!is_array($data) || !isset($data['value'], $data['score']) || $data['value'] === null) {
2370 + continue;
2371 + }
2372 +
2373 + $scores[] = (float) $data['score'];
2374 + $statuses[$metric] = $data['status'] ?? 'unknown';
2375 +
2376 + if (($data['status'] ?? '') !== 'good') {
2377 + $failing[] = strtoupper($metric);
2378 + }
2379 + }
2380 +
2381 + if (empty($scores)) {
2382 + return null;
2383 + }
2384 +
2385 + return [
2386 + 'average' => array_sum($scores) / count($scores),
2387 + 'statuses' => $statuses,
2388 + 'failing' => $failing,
2389 + ];
2390 + }
2391 +
2392 + /**
1885 2393 * Score internal linking - declining importance (1 point)
1886 2394 *
1887 2395 * @param array $content_data Content analysis data
1888 2396 * @return array Scoring result
@@ -1912,9 +2420,9 @@
1912 2420 $suggestions = [];
1913 2421
1914 2422 // Meta description check
1915 2423 $meta_desc = $metadata['description'] ?? '';
1916 - if (!empty($meta_desc) && mb_strlen($meta_desc) >= 120 && mb_strlen($meta_desc) <= 160) {
2424 + if (!empty($meta_desc) && mb_strlen($meta_desc) >= self::DESCRIPTION_OPTIMAL_MIN && mb_strlen($meta_desc) <= self::DESCRIPTION_OPTIMAL_MAX) {
1917 2425 $score += 0.5;
1918 2426 } else {
1919 2427 $suggestions[] = 'Add a compelling meta description (120-160 characters)';
1920 2428 }
@@ -1996,11 +2504,12 @@
1996 2504
1997 2505 // Extract headings from content
1998 2506 $headings = $this->extract_headings($content);
1999 2507
2000 - // Count words using JavaScript-compatible method
2001 - $plain_text = wp_strip_all_tags($content);
2002 - $word_count = $this->calculate_word_count_js_style($plain_text);
2508 + // Count the markup, not wp_strip_all_tags() output: stripping deletes
2509 + // tags without a space, so "five</p><p>six" would already be one word
2510 + // before the counter saw it.
2511 + $word_count = $this->calculate_word_count_js_style($content);
2003 2512
2004 2513 // Calculate readability
2005 2514 $readability_score = $this->calculate_readability_score($content);
2006 2515
@@ -2034,22 +2543,59 @@
2034 2543 ];
2035 2544 }
2036 2545
2037 2546 /**
2038 - * Calculate word count using JavaScript-compatible method
2039 - * Matches the logic in contentAnalysis.js for consistency
2547 + * Count the words a reader reads in HTML or text.
2040 2548 *
2041 - * @param string $text Text to count words in
2549 + * The same extraction and counting as the thin content report
2550 + * ({@see \ThinkRank\SEO\Word_Count_Index::reading_text()} and
2551 + * {@see \ThinkRank\SEO\Word_Count_Index::count_words()}), so the editor
2552 + * score and the report give one number for one page. Splitting on
2553 + * whitespace counted shortcode syntax as words: a WPBakery page with 216
2554 + * words of prose scored a word count of 325 while the report said 216
2555 + * (#893 fixed the report only). It also counted tokens of punctuation
2556 + * alone, such as a full stop after a link.
2557 + *
2558 + * Always words, whatever the locale's unit, because every threshold that
2559 + * reads this value (content length, long paragraphs, headings per 300
2560 + * words, readability, keyword density) is in words.
2561 + *
2562 + * contentAnalysis.js calculateWordCount() applies the same two rules in
2563 + * the editor.
2564 + *
2565 + * @param string $text HTML or text to count words in.
2042 2566 * @return int Word count
2043 2567 */
2044 2568 private function calculate_word_count_js_style(string $text): int {
2045 - if (empty($text)) {
2569 + if ('' === $text) {
2046 2570 return 0;
2047 2571 }
2048 2572
2049 - // Match JavaScript: trim, split by whitespace, filter empty
2050 - $words = preg_split('/\s+/', trim($text), -1, PREG_SPLIT_NO_EMPTY);
2051 - return count($words);
2573 + return \ThinkRank\SEO\Word_Count_Index::count_words(self::reading_text_of($text));
2574 + }
2575 +
2576 + /**
2577 + * The text a reader reads in HTML or text, as the word count sees it.
2578 + *
2579 + * Every check that divides by the word count (readability, keyword
2580 + * density, topic relevance) reads its numerator from this same text, so
2581 + * shortcode syntax is never on one side of a ratio and not the other.
2582 + *
2583 + * @since 2.14.2
2584 + *
2585 + * @param string $content HTML or text.
2586 + * @return string Plain text, whitespace collapsed.
2587 + */
2588 + private static function reading_text_of(string $content): string {
2589 + if ('' === $content) {
2590 + return '';
2591 + }
2592 +
2593 + if (!class_exists('\ThinkRank\SEO\Word_Count_Index')) {
2594 + require_once THINKRANK_PLUGIN_DIR . 'includes/seo/class-word-count-index.php';
2595 + }
2596 +
2597 + return \ThinkRank\SEO\Word_Count_Index::reading_text($content);
2052 2598 }
2053 2599
2054 2600 /**
2055 2601 * Get existing score data for a post from database