PluginProbe
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO / 2.14.2
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO v2.14.2
2.14.2 2.14.1 2.14.0 2.13.0 2.12.0 2.11.0 2.10.0 2.9.0 2.8.0 2.7.0 2.6.0 2.5.0 2.4.0 2.3.0 2.2.0 2.1.1 2.1.0 2.0.2 2.0.1 2.0.0 1.32.0 1.31.0 1.30.0 1.29.0 1.28.0 All 57 releases
← All changes | includes/ai/class-seo-score-calculator.php +792 -102 1.28.0 → 2.14.2 View file →
@@ -41,8 +41,41 @@
41 41 */
42 42 private Database $database;
43 43
44 44 /**
45 + * Memoised collected performance measurement, and whether it was resolved.
46 + *
47 + * Two factors read it and both may be asked for on every post in a list, so
48 + * the lookup happens once per calculator. `null` is a real answer here — the
49 + * separate flag keeps "not looked up yet" distinct from "nothing measured".
50 + *
51 + * @since 2.3.1
52 + * @var array|null
53 + */
54 + private ?array $measured_performance = null;
55 +
56 + /**
57 + * @since 2.3.1
58 + * @var bool
59 + */
60 + private bool $measured_performance_resolved = false;
61 +
62 + /**
63 + * Length bands the editor scores against, in characters.
64 + *
65 + * Public so every surface that judges a title or description — the editor
66 + * score and the Bulk Snippets problem filter — reads one set of numbers.
67 + * Before these existed the bands were literals inside the scoring methods,
68 + * and a second screen would have had to copy them and drift (#727).
69 + *
70 + * @since 2.8.0
71 + */
72 + public const TITLE_OPTIMAL_MIN = 35;
73 + public const TITLE_OPTIMAL_MAX = 60;
74 + public const DESCRIPTION_OPTIMAL_MIN = 120;
75 + public const DESCRIPTION_OPTIMAL_MAX = 160;
76 +
77 + /**
45 78 * 2025 SEO scoring factors (Q1 2025 Google Algorithm)
46 79 * Based on First Page Sage research and Google's latest updates
47 80 *
48 81 * @var array
@@ -107,8 +140,9 @@
107 140 'grade' => $result['grade'],
108 141 ]];
109 142 $result['keyword_checks'] = $this->analyze_keyword_checks($content_data, $metadata, $keywords);
110 143 }
144 + $result['keywords'] = $this->keyword_placements($content_data, $metadata, $keywords);
111 145
112 146 return $result;
113 147 }
114 148
@@ -218,8 +252,9 @@
218 252 $best['target_keyword'] = $best_keyword;
219 253 $best['target_keywords'] = $keywords;
220 254 $best['keyword_results'] = $per_keyword;
221 255 $best['keyword_checks'] = $this->analyze_keyword_checks($content_data, $metadata, $keywords);
256 + $best['keywords'] = $this->keyword_placements($content_data, $metadata, $keywords);
222 257
223 258 return $best;
224 259 }
225 260
@@ -234,21 +269,26 @@
234 269 * @param string[] $keywords Target keywords.
235 270 * @return array<string,array{passed:bool,matched_keywords:string[]}>
236 271 */
237 272 private function analyze_keyword_checks(array $content_data, array $metadata, array $keywords): array {
238 - $title = strtolower((string) ($metadata['title'] ?? $content_data['title'] ?? ''));
239 - $description = strtolower((string) ($metadata['description'] ?? ''));
240 - $content = strtolower(wp_strip_all_tags((string) ($content_data['content'] ?? '')));
273 + $title = self::lower((string) ($metadata['title'] ?? $content_data['title'] ?? ''));
274 + $description = self::lower((string) ($metadata['description'] ?? ''));
275 + $content = self::lower(self::plain_text((string) ($content_data['content'] ?? '')));
241 276
242 277 $alts = '';
243 278 foreach ((array) ($content_data['images'] ?? []) as $image) {
244 - $alts .= ' ' . strtolower((string) ($image['alt'] ?? ''));
279 + $alts .= ' ' . self::lower((string) ($image['alt'] ?? ''));
245 280 }
246 281
247 - // Build a searchable slug haystack from the URL path (hyphens/underscores
248 - // become spaces so multi-word keywords can match).
249 - $path = (string) (wp_parse_url((string) ($content_data['url'] ?? ''), PHP_URL_PATH) ?? '');
250 - $slug = strtolower(str_replace(['-', '_', '/'], ' ', trim($path, '/')));
282 + // Build a searchable slug haystack from the post's OWN slug — never the
283 + // full URL path. The path carries ancestors, category bases and date
284 + // segments, so a child of /clinical-trials/ reported "keyword in slug"
285 + // for a page actually slugged `contact-us`. It also breaks the other
286 + // way: an unpublished post has no pretty permalink (get_permalink()
287 + // returns ?p=123), so the path held no slug at all and every draft
288 + // scored "no match" until it was published. Hyphens/underscores become
289 + // spaces so multi-word keywords can match.
290 + $slug = self::lower(self::slug_haystack($content_data));
251 291
252 292 $haystacks = [
253 293 'title' => trim($title),
254 294 'meta_description' => trim($description),
@@ -260,10 +300,9 @@
260 300 $checks = [];
261 301 foreach ($haystacks as $location => $haystack) {
262 302 $matched = [];
263 303 foreach ($keywords as $keyword) {
264 - $needle = strtolower(trim($keyword));
265 - if ($needle !== '' && $haystack !== '' && strpos($haystack, $needle) !== false) {
304 + if ($this->keyword_matches($haystack, self::lower(trim($keyword)))) {
266 305 $matched[] = $keyword;
267 306 }
268 307 }
269 308 $checks[$location] = [
@@ -275,8 +314,288 @@
275 314 return $checks;
276 315 }
277 316
278 317 /**
318 + * The keyword placements the editor draws one gauge segment for, in the
319 + * order a reader meets them (#729).
320 + *
321 + * @since 2.11.0
322 + * @var string[]
323 + */
324 + public const PLACEMENTS = ['title', 'meta_description', 'slug', 'first_paragraph', 'subheading', 'content', 'image_alt'];
325 +
326 + /**
327 + * Characters of plain text read as the opening when the content has no
328 + * paragraph tag.
329 + *
330 + * @since 2.11.0
331 + */
332 + private const OPENING_CHARS = 300;
333 +
334 + /**
335 + * Where each focus keyword is placed, keyword by keyword (#729).
336 + *
337 + * analyze_keyword_checks() answers "does ANY keyword appear here" for
338 + * five places; this answers "where does THIS keyword appear" for seven,
339 + * so the editor can show each keyword's own gauge. Same matcher, so a
340 + * keyword counts as a word (not a fragment) and a keyword in a script
341 + * written without spaces (Thai, Chinese, Japanese) still matches.
342 + *
343 + * `where` names the heading or alt text that matched, so the editor can
344 + * say which one.
345 + *
346 + * @since 2.11.0
347 + *
348 + * @param array $content_data Content analysis data.
349 + * @param array $metadata Post metadata (title, description).
350 + * @param string[] $keywords Focus keywords.
351 + * @return array<int, array{keyword: string, passed: int, total: int, placements: array<string, array{passed: bool, where: string}>}>
352 + */
353 + public function keyword_placements(array $content_data, array $metadata, array $keywords): array {
354 + $html = (string) ($content_data['content'] ?? '');
355 + $plain = self::lower(self::plain_text($html));
356 +
357 + $headings = [];
358 + $source = isset($content_data['headings']) && is_array($content_data['headings']) ? $content_data['headings'] : $this->extract_headings($html);
359 + foreach ($source as $heading) {
360 + $text = trim((string) ($heading['text'] ?? ''));
361 + if ((int) ($heading['level'] ?? 0) >= 2 && '' !== $text) {
362 + $headings[] = $text;
363 + }
364 + }
365 +
366 + $alts = [];
367 + foreach ((array) ($content_data['images'] ?? []) as $image) {
368 + $alt = trim((string) ($image['alt'] ?? ''));
369 + if ('' !== $alt) {
370 + $alts[] = $alt;
371 + }
372 + }
373 +
374 + $single = [
375 + 'title' => self::lower((string) ($metadata['title'] ?? $content_data['title'] ?? '')),
376 + 'meta_description' => self::lower((string) ($metadata['description'] ?? '')),
377 + 'slug' => self::lower(self::slug_haystack($content_data)),
378 + 'first_paragraph' => self::lower(self::opening($html)),
379 + 'content' => $plain,
380 + ];
381 +
382 + $out = [];
383 + foreach ($keywords as $keyword) {
384 + $needle = self::lower(trim((string) $keyword));
385 + $placements = [];
386 +
387 + foreach (self::PLACEMENTS as $placement) {
388 + if ('subheading' === $placement || 'image_alt' === $placement) {
389 + $where = '';
390 + foreach ('subheading' === $placement ? $headings : $alts as $text) {
391 + if ($this->keyword_matches(self::lower($text), $needle)) {
392 + $where = $text;
393 + break;
394 + }
395 + }
396 + $placements[$placement] = ['passed' => '' !== $where, 'where' => $where];
397 + continue;
398 + }
399 +
400 + $placements[$placement] = ['passed' => $this->keyword_matches($single[$placement], $needle), 'where' => ''];
401 + }
402 +
403 + $out[] = [
404 + 'keyword' => (string) $keyword,
405 + 'passed' => count(array_filter(array_column($placements, 'passed'))),
406 + 'total' => count(self::PLACEMENTS),
407 + 'placements' => $placements,
408 + ];
409 + }
410 +
411 + return $out;
412 + }
413 +
414 + /**
415 + * The post's own slug as searchable text: hyphens and underscores become
416 + * spaces so a multi-word keyword can match, and a slug WordPress
417 + * percent-encoded (Thai, Cyrillic, Chinese) is decoded, or it could never
418 + * match a keyword typed in that script.
419 + *
420 + * Never the full URL path: the path carries ancestors, category bases and
421 + * date segments, so a child of /clinical-trials/ reported "keyword in slug"
422 + * for a page actually slugged `contact-us`. An unpublished post has no
423 + * pretty permalink either, so the path held no slug at all.
424 + *
425 + * @param array $content_data Content analysis data.
426 + * @return string
427 + */
428 + private static function slug_haystack(array $content_data): string {
429 + $slug = (string) ($content_data['slug'] ?? '');
430 + if ('' === $slug) {
431 + // Draft with no slug assigned yet: score what WordPress would
432 + // generate from the title, which is what the editor shows as the
433 + // proposed URL — so the check reads the same before and after
434 + // publishing instead of flipping.
435 + $slug = sanitize_title((string) ($content_data['title'] ?? ''));
436 + }
437 +
438 + return trim(str_replace(['-', '_'], ' ', rawurldecode($slug)));
439 + }
440 +
441 + /**
442 + * The opening of the content: its first paragraph with text, or the first
443 + * few hundred characters when it has none. Counted in characters, not
444 + * words, so a language written without spaces is not read as one word.
445 + *
446 + * @param string $html Content HTML.
447 + * @return string Plain text.
448 + */
449 + private static function opening(string $html): string {
450 + if (preg_match_all('/<p\b[^>]*>(.*?)<\/p>/isu', $html, $matches)) {
451 + foreach ($matches[1] as $paragraph) {
452 + $text = self::collapse_whitespace(wp_strip_all_tags($paragraph));
453 + if ('' !== $text) {
454 + return $text;
455 + }
456 + }
457 + }
458 +
459 + $text = self::plain_text($html);
460 +
461 + return function_exists('mb_substr') ? mb_substr($text, 0, self::OPENING_CHARS) : substr($text, 0, self::OPENING_CHARS);
462 + }
463 +
464 + /**
465 + * Content as plain text, with a space where each tag was. wp_strip_all_tags()
466 + * alone joins neighbouring blocks — "…coffee grinder</h3><p>A good…" became
467 + * "coffee grinderA good" — so a keyword at the end of a heading or a
468 + * paragraph was no longer a word and did not match.
469 + *
470 + * @param string $html Content HTML.
471 + * @return string
472 + */
473 + private static function plain_text(string $html): string {
474 + $spaced = preg_replace('/<[^>]+>/', ' $0 ', $html);
475 +
476 + return self::collapse_whitespace(wp_strip_all_tags(null === $spaced ? $html : (string) $spaced));
477 + }
478 +
479 + /**
480 + * Runs of whitespace down to one space.
481 + *
482 + * The `/u` pass is the one that understands a multibyte space, but
483 + * preg_replace() answers null on bytes that are not valid UTF-8 rather than
484 + * throwing — and casting that null to a string blanked the haystack, so a
485 + * post carrying one mojibake byte (a Latin-1 paste, an old import) reported
486 + * every keyword as missing from its body, its opening, and every
487 + * subheading. The gauge said 0/7 and told the author to add a keyword that
488 + * was already there.
489 + *
490 + * Falls back to the byte-wise collapse, which is what this did before the
491 + * multibyte work added the modifier. Same reasoning keyword_matches()
492 + * already records for its own PCRE failure: a pattern PCRE refuses must not
493 + * be reported as a confident "no match".
494 + *
495 + * @param string $text Text to collapse.
496 + * @return string
497 + */
498 + private static function collapse_whitespace(string $text): string {
499 + $collapsed = preg_replace('/\s+/u', ' ', $text);
500 +
501 + if (null === $collapsed) {
502 + $collapsed = preg_replace('/\s+/', ' ', $text);
503 + }
504 +
505 + return trim(null === $collapsed ? $text : (string) $collapsed);
506 + }
507 +
508 + /**
509 + * Lowercase in any script. strtolower() only folds ASCII, so "Кофе" never
510 + * matched "кофе".
511 + *
512 + * @param string $text Text.
513 + * @return string
514 + */
515 + private static function lower(string $text): string {
516 + return function_exists('mb_strtolower') ? mb_strtolower($text, 'UTF-8') : strtolower($text);
517 + }
518 +
519 + /**
520 + * Scripts written without spaces between words.
521 + *
522 + * @since 2.1.0
523 + * @var string
524 + */
525 + private const SCRIPTIO_CONTINUA = '/[\p{Han}\p{Hiragana}\p{Katakana}\p{Thai}\p{Lao}\p{Khmer}\p{Myanmar}]/u';
526 +
527 + /**
528 + * Whether a keyword appears in a haystack as a word rather than as a
529 + * fragment of a longer one.
530 + *
531 + * The five keyword checks used a plain strpos(), so any substring hit
532 + * counted: "test coronavirus" matched "la|test coronavirus|news", "art"
533 + * matched "start", "cat" matched "category". The panel then confidently
534 + * reported a keyword placement that does not exist (#416). Same class of
535 + * problem #71 fixed in the Image SEO rewriter, and the same remedy.
536 + *
537 + * Both arguments are expected lowercased already.
538 + *
539 + * @since 2.1.0
540 + *
541 + * @param string $haystack Text to search.
542 + * @param string $needle Keyword, lowercased and trimmed.
543 + * @return bool
544 + */
545 + private function keyword_matches(string $haystack, string $needle): bool {
546 + if ($needle === '' || $haystack === '') {
547 + return false;
548 + }
549 +
550 + if (!$this->supports_word_boundaries($needle)) {
551 + return strpos($haystack, $needle) !== false;
552 + }
553 +
554 + $matched = preg_match('/\b' . preg_quote($needle, '/') . '\b/u', $haystack);
555 +
556 + // PCRE refusing the pattern — invalid UTF-8 in the keyword, a
557 + // backtrack limit — must not be reported as a confident "no match".
558 + // Fall back to the behaviour this replaced rather than invent a
559 + // negative the user cannot explain.
560 + if ($matched === false) {
561 + return strpos($haystack, $needle) !== false;
562 + }
563 +
564 + return $matched === 1;
565 + }
566 +
567 + /**
568 + * Whether \b can express "this keyword, as a word" for this keyword.
569 + *
570 + * It asserts a transition between a word and a non-word character, which
571 + * only means something where words are separated. Two cases where it is
572 + * not, both verified against PCRE rather than assumed:
573 + *
574 + * - the keyword's own edges are not word characters ("c++", "#seo"), so
575 + * no boundary can assert there and a real match is lost;
576 + * - scripts written without spaces, where the neighbouring characters
577 + * are word characters too — "冠状病毒" inside "最新冠状病毒新闻" is a
578 + * legitimate match that \b never sees.
579 + *
580 + * Accented Latin and Cyrillic need no special handling: PHP's /u modifier
581 + * turns on Unicode character properties, so "café" correctly does not
582 + * match "cafés" and "коронавирус" does not match "коронавирусный".
583 + *
584 + * @since 2.1.0
585 + *
586 + * @param string $needle Keyword, lowercased and trimmed.
587 + * @return bool
588 + */
589 + private function supports_word_boundaries(string $needle): bool {
590 + if (preg_match(self::SCRIPTIO_CONTINUA, $needle)) {
591 + return false;
592 + }
593 +
594 + return preg_match('/^\w/u', $needle) === 1 && preg_match('/\w$/u', $needle) === 1;
595 + }
596 +
597 + /**
279 598 * Compute the SEO score for a single target keyword.
280 599 *
281 600 * @param array $content_data Content analysis data
282 601 * @param array $metadata Post metadata
@@ -281,8 +600,10 @@
281 600 * @param array $content_data Content analysis data
282 601 * @param array $metadata Post metadata
283 602 * @param array $options Additional options (expects scalar target_keyword)
284 603 * @return array Complete scoring result
604 + *
605 + * @throws \Exception On failure.
285 606 */
286 607 private function compute_score(array $content_data, array $metadata, array $options = []): array {
287 608 $scores = [];
288 609 $suggestions = [];
@@ -386,9 +707,9 @@
386 707 $total_score += $technical_result['score'];
387 708 $suggestions = array_merge($suggestions, $technical_result['suggestions']);
388 709
389 710 try {
390 - $prioritized_suggestions = $this->prioritize_suggestions($suggestions);
711 + $prioritized_suggestions = $this->prioritize_suggestions($suggestions, $scores);
391 712 $grade = $this->get_grade_from_score($total_score);
392 713
393 714 return [
394 715 'overall_score' => min(100, $total_score),
@@ -443,9 +764,9 @@
443 764 } elseif ($word_count >= 300) {
444 765 $score += 3;
445 766 $suggestions[] = 'Content is thin - aim for 600+ words minimum';
446 767 } else {
447 - $score += 1;
768 + $score++;
448 769 $suggestions[] = 'Content too shallow - Google prioritizes comprehensive, satisfying content';
449 770 }
450 771
451 772 // Keyword presence & placement (8 points) - deterministic, replaces the
@@ -505,12 +826,16 @@
505 826 * @param int $word_count Word count
506 827 * @return string Depth assessment
507 828 */
508 829 private function assess_content_depth_2025(int $word_count): string {
509 - if ($word_count >= 3000) return 'Comprehensive';
510 - if ($word_count >= 2000) return 'Detailed';
511 - if ($word_count >= 1200) return 'Adequate';
512 - if ($word_count >= 800) return 'Basic';
830 + if ($word_count >= 3000) { return 'Comprehensive';
831 + }
832 + if ($word_count >= 2000) { return 'Detailed';
833 + }
834 + if ($word_count >= 1200) { return 'Adequate';
835 + }
836 + if ($word_count >= 800) { return 'Basic';
837 + }
513 838 return 'Insufficient';
514 839 }
515 840
516 841 /**
@@ -533,15 +858,15 @@
533 858 $title_length = mb_strlen($title);
534 859
535 860 // 2025 length optimization (6 points). 60 characters is the recommended
536 861 // maximum for best SERP visibility before Google truncates the title.
537 - if ($title_length >= 35 && $title_length <= 60) {
862 + if ($title_length >= self::TITLE_OPTIMAL_MIN && $title_length <= self::TITLE_OPTIMAL_MAX) {
538 863 $score += 6;
539 864 } elseif ($title_length >= 25 && $title_length <= 75) {
540 865 $score += 4;
541 866 $suggestions[] = 'Optimize title length to 35-60 characters for better SERP visibility';
542 867 } else {
543 - $score += 1;
868 + $score++;
544 869 $suggestions[] = $title_length < 25 ?
545 870 'Title too short - aim for 35-60 characters' :
546 871 'Title too long - risk truncation in search results';
547 872 }
@@ -589,14 +914,14 @@
589 914 $has_power_word = $this->title_has_power_word($title);
590 915 $has_sentiment = $this->title_has_sentiment_word($title);
591 916
592 917 if ($has_number || $has_power_word) {
593 - $score += 1;
918 + $score++;
594 919 } else {
595 920 $suggestions[] = 'Add a number or a power word to the title to boost click-through rate';
596 921 }
597 922 if ($has_sentiment) {
598 - $score += 1;
923 + $score++;
599 924 } else {
600 925 $suggestions[] = 'Use an emotional/sentiment word in the title to make it more compelling';
601 926 }
602 927
@@ -671,9 +996,9 @@
671 996
672 997 // A slug under ~75 chars keeps the URL clean and fully visible in SERPs.
673 998 $slug_length = strlen($slug);
674 999 if ($slug === '' || $slug_length <= 75) {
675 - $score += 1;
1000 + $score++;
676 1001 } else {
677 1002 $suggestions[] = 'Shorten the URL slug - long URLs are harder to read and share';
678 1003 }
679 1004
@@ -809,27 +1134,69 @@
809 1134
810 1135 // Consider it a semantic match if 70% of keyword parts are present
811 1136 return ($matches / count($keyword_parts)) >= 0.7;
812 1137 }
813 - private function prioritize_suggestions(array $suggestions): array {
1138 + private function prioritize_suggestions(array $suggestions, array $scores = []): array {
1139 + // Map each suggestion back to the factor that emitted it, so priority
1140 + // can rank by the points the factor actually lost instead of keyword-
1141 + // matching the advice text — which sorted a 2-point title tweak above
1142 + // a 6-point thin-content loss and contradicted the row's own impact
1143 + // tag (#408).
1144 + $by_text = [];
1145 + foreach ($scores as $factor => $result) {
1146 + if (!is_array($result) || empty($result['suggestions']) || !is_array($result['suggestions'])) {
1147 + continue;
1148 + }
1149 + $lost = max(0, (float) ($result['max_score'] ?? 0) - (float) ($result['score'] ?? 0));
1150 + foreach ($result['suggestions'] as $text) {
1151 + if (is_string($text) && !isset($by_text[$text])) {
1152 + $by_text[$text] = ['factor' => (string) $factor, 'lost' => $lost];
1153 + }
1154 + }
1155 + }
1156 +
814 1157 $prioritized = [];
815 -
1158 +
816 1159 foreach ($suggestions as $suggestion) {
817 - $priority = $this->determine_suggestion_priority($suggestion);
1160 + $origin = $by_text[$suggestion] ?? null;
1161 +
1162 + // A factor already at full marks loses nothing to this advice —
1163 + // it was occupying list positions (sometimes at "High") while
1164 + // recovering zero points. Dropped rather than sorted last.
1165 + if (null !== $origin && $origin['lost'] <= 0) {
1166 + continue;
1167 + }
1168 +
1169 + if (null !== $origin) {
1170 + $priority = $origin['lost'] >= 4 ? 'High' : ($origin['lost'] >= 2 ? 'Medium' : 'Low');
1171 + } else {
1172 + // No factor attached (defensive: a filter-added or legacy
1173 + // suggestion) — the old keyword map is the fallback.
1174 + $priority = $this->determine_suggestion_priority($suggestion);
1175 + }
1176 +
818 1177 $prioritized[] = [
819 1178 'text' => $suggestion,
820 1179 'priority' => $priority,
821 1180 'impact' => $this->estimate_impact($suggestion),
822 1181 'effort' => $this->estimate_effort($suggestion),
1182 + 'factor' => $origin['factor'] ?? null,
1183 + 'points_recoverable' => $origin['lost'] ?? null,
823 1184 ];
824 1185 }
825 -
826 - // Sort by priority (High > Medium > Low)
1186 +
1187 + // Biggest recoverable loss first; keyword-mapped stragglers (no
1188 + // factor) sort within their priority band after the measured rows.
827 1189 usort($prioritized, function($a, $b) {
1190 + $al = $a['points_recoverable'] ?? -1;
1191 + $bl = $b['points_recoverable'] ?? -1;
1192 + if ($al !== $bl) {
1193 + return $bl <=> $al;
1194 + }
828 1195 $priority_order = ['High' => 3, 'Medium' => 2, 'Low' => 1];
829 1196 return $priority_order[$b['priority']] - $priority_order[$a['priority']];
830 1197 });
831 -
1198 +
832 1199 return $prioritized;
833 1200 }
834 1201
835 1202 /**
@@ -866,11 +1233,14 @@
866 1233 * @return string Impact level
867 1234 */
868 1235 private function estimate_impact(string $suggestion): string {
869 1236 // Simple heuristic - can be enhanced with ML
870 - if (strpos(strtolower($suggestion), 'title') !== false) return 'High';
871 - if (strpos(strtolower($suggestion), 'content') !== false) return 'High';
872 - if (strpos(strtolower($suggestion), 'keyword') !== false) return 'Medium';
1237 + if (strpos(strtolower($suggestion), 'title') !== false) { return 'High';
1238 + }
1239 + if (strpos(strtolower($suggestion), 'content') !== false) { return 'High';
1240 + }
1241 + if (strpos(strtolower($suggestion), 'keyword') !== false) { return 'Medium';
1242 + }
873 1243 return 'Low';
874 1244 }
875 1245
876 1246 /**
@@ -880,11 +1250,14 @@
880 1250 * @return string Effort level
881 1251 */
882 1252 private function estimate_effort(string $suggestion): string {
883 1253 // Simple heuristic - can be enhanced with ML
884 - if (strpos(strtolower($suggestion), 'rewrite') !== false) return 'High';
885 - if (strpos(strtolower($suggestion), 'add') !== false) return 'Medium';
886 - if (strpos(strtolower($suggestion), 'optimize') !== false) return 'Medium';
1254 + if (strpos(strtolower($suggestion), 'rewrite') !== false) { return 'High';
1255 + }
1256 + if (strpos(strtolower($suggestion), 'add') !== false) { return 'Medium';
1257 + }
1258 + if (strpos(strtolower($suggestion), 'optimize') !== false) { return 'Medium';
1259 + }
887 1260 return 'Low';
888 1261 }
889 1262 private function calculate_topic_relevance(string $content, string $target_keyword): float {
890 1263 if (empty($content) || empty($target_keyword)) {
@@ -890,14 +1263,16 @@
890 1263 if (empty($content) || empty($target_keyword)) {
891 1264 return 0.0;
892 1265 }
893 1266
894 - $content_lower = strtolower(wp_strip_all_tags($content));
1267 + // Occurrences and the word count both come from the reading text, so
1268 + // shortcode syntax is in neither.
1269 + $content_lower = strtolower(self::reading_text_of($content));
895 1270 $keyword_lower = strtolower($target_keyword);
896 1271
897 1272 // Calculate keyword and semantic term frequency
898 1273 $keyword_count = substr_count($content_lower, $keyword_lower);
899 - $word_count = $this->calculate_word_count_js_style($content_lower);
1274 + $word_count = \ThinkRank\SEO\Word_Count_Index::count_words($content_lower);
900 1275
901 1276 if ($word_count === 0) {
902 1277 return 0.0;
903 1278 }
@@ -958,9 +1333,9 @@
958 1333 $terms = array_merge($terms, ['optimization', 'search engine', 'ranking', 'visibility']);
959 1334 }
960 1335
961 1336 // WordPress-related terms
962 - if (strpos($keyword_lower, 'wordpress') !== false) {
1337 + if (strpos($keyword_lower, 'wordpress') !== false) { // phpcs:ignore WordPress.WP.CapitalPDangit.MisspelledInText -- lowercase on purpose: the haystack is strtolower()ed.
963 1338 $terms = array_merge($terms, ['wp', 'plugin', 'theme', 'cms']);
964 1339 }
965 1340
966 1341 return $terms;
@@ -976,10 +1351,13 @@
976 1351 private function keyword_density(string $content, string $target_keyword): float {
977 1352 if (empty($content) || empty($target_keyword)) {
978 1353 return 0.0;
979 1354 }
980 - $plain = strtolower(wp_strip_all_tags($content));
981 - $word_count = $this->calculate_word_count_js_style($plain);
1355 + // Numerator and denominator from the same reading text. Counting the
1356 + // keyword in wp_strip_all_tags() output found it inside shortcode
1357 + // attributes the word count no longer includes.
1358 + $plain = strtolower(self::reading_text_of($content));
1359 + $word_count = \ThinkRank\SEO\Word_Count_Index::count_words($plain);
982 1360 if ($word_count === 0) {
983 1361 return 0.0;
984 1362 }
985 1363 $occurrences = substr_count($plain, strtolower($target_keyword));
@@ -1014,10 +1392,10 @@
1014 1392 private function keyword_in_first_paragraph(string $content, string $target_keyword): bool {
1015 1393 if (empty($content) || empty($target_keyword)) {
1016 1394 return false;
1017 1395 }
1018 - $plain = strtolower(wp_strip_all_tags($content));
1019 - $words = preg_split('/\s+/', trim($plain), -1, PREG_SPLIT_NO_EMPTY) ?: [];
1396 + $plain = strtolower(self::reading_text_of($content));
1397 + $words = preg_split('/\s+/', $plain, -1, PREG_SPLIT_NO_EMPTY) ?: [];
1020 1398 $window = array_slice($words, 0, max(50, (int) ceil(count($words) * 0.1)));
1021 1399 return strpos(implode(' ', $window), strtolower($target_keyword)) !== false;
1022 1400 }
1023 1401
@@ -1059,9 +1437,12 @@
1059 1437 $ts = strtotime($datetime);
1060 1438 if ($ts === false) {
1061 1439 return null;
1062 1440 }
1063 - $now = function_exists('current_time') ? (int) current_time('timestamp') : time();
1441 + // strtotime() returns a real unix timestamp, so this must compare against
1442 + // one: current_time('timestamp') is offset by the site timezone and made
1443 + // every "days ago" figure wrong by that offset.
1444 + $now = time();
1064 1445 return (int) floor(($now - $ts) / 86400);
1065 1446 }
1066 1447
1067 1448 /**
@@ -1170,12 +1551,16 @@
1170 1551 * @param int $word_count Word count
1171 1552 * @return string Depth assessment
1172 1553 */
1173 1554 private function assess_content_depth(int $word_count): string {
1174 - if ($word_count >= 2000) return 'Comprehensive';
1175 - if ($word_count >= 1000) return 'Detailed';
1176 - if ($word_count >= 500) return 'Moderate';
1177 - if ($word_count >= 300) return 'Basic';
1555 + if ($word_count >= 2000) { return 'Comprehensive';
1556 + }
1557 + if ($word_count >= 1000) { return 'Detailed';
1558 + }
1559 + if ($word_count >= 500) { return 'Moderate';
1560 + }
1561 + if ($word_count >= 300) { return 'Basic';
1562 + }
1178 1563 return 'Insufficient';
1179 1564 }
1180 1565 private function score_content_freshness(array $content_data): array {
1181 1566 $max_score = $this->scoring_factors['content_freshness'];
@@ -1240,9 +1625,24 @@
1240 1625 if (!class_exists('\ThinkRank\SEO\Builder_Content')) {
1241 1626 require_once THINKRANK_PLUGIN_DIR . 'includes/seo/class-builder-content.php';
1242 1627 }
1243 1628
1244 - return \ThinkRank\SEO\Builder_Content::resolve_markup($live_content, $post);
1629 + // Bind the live markup to the post the render makes current. Since #862
1630 + // the resolver runs setup_postdata() on this post, so a shortcode that
1631 + // builds its output from the current post's content — get_the_content(),
1632 + // get_post()->post_content, as a table of contents or a reading-time
1633 + // shortcode does — otherwise read the last saved body while the unsaved
1634 + // markup rendered around it, and lagged a save behind (#864).
1635 + //
1636 + // A clone, not the caller's object: the post is current only for the
1637 + // duration of the render and the caller's $post must come back
1638 + // unchanged. `thinkrank_analyzable_content` receives the clone too,
1639 + // which is what makes $post->post_content there agree with the markup
1640 + // being analyzed on the live path.
1641 + $bound = clone $post;
1642 + $bound->post_content = $live_content;
1643 +
1644 + return \ThinkRank\SEO\Builder_Content::resolve_markup($live_content, $bound);
1245 1645 }
1246 1646
1247 1647 public function analyze_post_content(int $post_id): array {
1248 1648 $post = get_post($post_id);
@@ -1255,11 +1655,12 @@
1255 1655
1256 1656 // Extract headings from content
1257 1657 $headings = $this->extract_headings($content);
1258 1658
1259 - // Count words using JavaScript-compatible method
1260 - $plain_text = wp_strip_all_tags($content);
1261 - $word_count = $this->calculate_word_count_js_style($plain_text);
1659 + // Count the markup, not wp_strip_all_tags() output: stripping deletes
1660 + // tags without a space, so "five</p><p>six" would already be one word
1661 + // before the counter saw it.
1662 + $word_count = $this->calculate_word_count_js_style($content);
1262 1663
1263 1664 // Calculate readability
1264 1665 $readability_score = $this->calculate_readability_score($content);
1265 1666
@@ -1286,9 +1687,11 @@
1286 1687 'images' => $images,
1287 1688 'url' => $url,
1288 1689 'slug' => $post->post_name,
1289 1690 'post_modified' => $post->post_modified,
1290 - 'schema_present' => $this->detect_schema_present($content) || $this->thinkrank_global_schema_active($post->post_type),
1691 + 'schema_present' => $this->detect_schema_present($content)
1692 + || $this->thinkrank_global_schema_active($post->post_type)
1693 + || $this->thinkrank_deployed_schema_active($post),
1291 1694 ];
1292 1695 }
1293 1696
1294 1697 /**
@@ -1377,9 +1780,9 @@
1377 1780 // 2. Long paragraphs (flag paragraphs over 150 words).
1378 1781 $long_paragraphs = 0;
1379 1782 if (preg_match_all('/<p[^>]*>(.*?)<\/p>/is', $content, $matches)) {
1380 1783 foreach ($matches[1] as $paragraph) {
1381 - if ($this->calculate_word_count_js_style(wp_strip_all_tags($paragraph)) > 150) {
1784 + if ($this->calculate_word_count_js_style($paragraph) > 150) {
1382 1785 $long_paragraphs++;
1383 1786 }
1384 1787 }
1385 1788 }
@@ -1434,22 +1837,22 @@
1434 1837 * @param string $content Content text
1435 1838 * @return float Readability score
1436 1839 */
1437 1840 private function calculate_readability_score(string $content): float {
1438 - $text = wp_strip_all_tags($content);
1841 + // Sentences, words and syllables all come from one reading text. The
1842 + // word count drops shortcode syntax and punctuation-only tokens; when
1843 + // sentences and syllables were still read off wp_strip_all_tags()
1844 + // output, "[vc_column width="1/2"]" added syllables (and, with a "." in
1845 + // an attribute, sentences) to a word count that did not include it,
1846 + // and a WPBakery page's Flesch fell from 65 to 46.
1847 + $text = self::reading_text_of($content);
1439 1848
1440 - if (empty($text)) {
1849 + if ('' === $text) {
1441 1850 return 0;
1442 1851 }
1443 1852
1444 - // Count sentences (approximate)
1445 - $sentences = preg_split('/[.!?]+/', $text, -1, PREG_SPLIT_NO_EMPTY);
1446 - $sentence_count = count($sentences);
1447 -
1448 - // Count words
1449 - $word_count = $this->calculate_word_count_js_style(wp_strip_all_tags($text));
1450 -
1451 - // Count syllables (approximate)
1853 + $sentence_count = self::count_sentences($text);
1854 + $word_count = \ThinkRank\SEO\Word_Count_Index::count_words($text);
1452 1855 $syllable_count = $this->count_syllables($text);
1453 1856
1454 1857 if ($sentence_count === 0 || $word_count === 0) {
1455 1858 return 0;
@@ -1461,19 +1864,59 @@
1461 1864 return max(0, min(100, $score));
1462 1865 }
1463 1866
1464 1867 /**
1868 + * Count the sentences in reading text.
1869 + *
1870 + * A sentence is a run of text between terminal punctuation that holds at
1871 + * least one letter or digit. A fragment with no word in it (the space
1872 + * after the final full stop, a stray "!" between two "?") is not a
1873 + * sentence. contentAnalysis.js countSentences() applies the same rule.
1874 + *
1875 + * @since 2.14.2
1876 + *
1877 + * @param string $text Text as returned by {@see self::reading_text_of()}.
1878 + * @return int
1879 + */
1880 + private static function count_sentences(string $text): int {
1881 + $fragments = preg_split('/[.!?]+/', $text, -1, PREG_SPLIT_NO_EMPTY) ?: [];
1882 +
1883 + return count(preg_grep('/[\p{L}\p{N}]/u', $fragments) ?: []);
1884 + }
1885 +
1886 + /**
1465 1887 * Count syllables in text (approximate)
1466 1888 *
1889 + * Expects reading text ({@see self::reading_text_of()}); tags are not
1890 + * stripped here, so a decoded "<" in prose is not read as a tag.
1891 + *
1467 1892 * @param string $text Text to analyze
1468 1893 * @return int Syllable count
1469 1894 */
1470 1895 private function count_syllables(string $text): int {
1471 - $words = preg_split('/\s+/', trim(strtolower(wp_strip_all_tags($text))), -1, PREG_SPLIT_NO_EMPTY);
1896 + $words = preg_split('/\s+/', trim(strtolower($text)), -1, PREG_SPLIT_NO_EMPTY) ?: [];
1472 1897 $syllables = 0;
1473 1898
1474 1899 foreach ($words as $word) {
1475 - $syllables += max(1, preg_match_all('/[aeiouy]+/', $word));
1900 + $word = preg_replace('/[^a-z]/', '', $word);
1901 + if ($word === '') {
1902 + continue;
1903 + }
1904 +
1905 + $groups = preg_match_all('/[aeiouy]+/', $word);
1906 +
1907 + // Standard Flesch heuristic: a trailing silent e does not form a
1908 + // syllable ("make", "time", "these") — but only when a consonant
1909 + // precedes it (a vowel+e ending like "movie" already shares its
1910 + // group) and never for consonant-le ("table"), which does count.
1911 + // Without this the counter inflated syllables/word by ~0.2-0.3 on
1912 + // ordinary prose, driving raw Flesch negative and the UI to a
1913 + // clamped "Very Difficult (0)" (#407).
1914 + if ($groups > 1 && preg_match('/[^aeiouy]e$/', $word) && !str_ends_with($word, 'le')) {
1915 + $groups--;
1916 + }
1917 +
1918 + $syllables += max(1, $groups);
1476 1919 }
1477 1920
1478 1921 return $syllables;
1479 1922 }
@@ -1592,8 +2035,45 @@
1592 2035 $all_settings = get_option('thinkrank_global_seo_settings', []);
1593 2036 return !empty($all_settings[$post_type]['schema_type']);
1594 2037 }
1595 2038
2039 + /**
2040 + * Whether the Schema Manager has an active deployed schema for this post.
2041 + *
2042 + * Per-post schema deployed from the editor's Schema tab is stored in the
2043 + * Schema Manager's own table and emitted at wp_head by
2044 + * Frontend\SEO_Manager::output_site_schema_markup(). Neither
2045 + * detect_schema_present() (body scan) nor thinkrank_global_schema_active()
2046 + * (post-type option) sees it, so without this the score reported "no
2047 + * structured data" for posts that do emit it.
2048 + *
2049 + * Mirrors the context_type whitelist the emitter and the metabox both use, so
2050 + * the lookup targets the same row the front end reads.
2051 + *
2052 + * @param \WP_Post $post Post being scored.
2053 + * @return bool
2054 + */
2055 + private function thinkrank_deployed_schema_active(\WP_Post $post): bool {
2056 + if (!class_exists('ThinkRank\\SEO\\Schema_Management_System')) {
2057 + $manager_file = THINKRANK_PLUGIN_DIR . 'includes/seo/class-schema-management-system.php';
2058 + if (!file_exists($manager_file)) {
2059 + return false;
2060 + }
2061 + require_once $manager_file;
2062 + }
2063 +
2064 + $context_type = in_array($post->post_type, ['site', 'post', 'page', 'product'], true)
2065 + ? $post->post_type
2066 + : 'post';
2067 +
2068 + try {
2069 + $manager = new \ThinkRank\SEO\Schema_Management_System();
2070 + return !empty($manager->get_deployed_schemas($context_type, (int) $post->ID));
2071 + } catch (\Throwable $e) {
2072 + return false;
2073 + }
2074 + }
2075 +
1596 2076 private function analyze_images(string $content): array {
1597 2077 $images = [];
1598 2078
1599 2079 if (preg_match_all('/<img[^>]+>/i', $content, $matches)) {
@@ -1635,10 +2115,10 @@
1635 2115 $insert_data = [
1636 2116 'post_id' => $post_id,
1637 2117 'user_id' => $user_id,
1638 2118 'overall_score' => $score_data['overall_score'],
1639 - 'score_breakdown' => json_encode($score_data['score_breakdown']),
1640 - 'suggestions' => json_encode($score_data['suggestions']),
2119 + 'score_breakdown' => wp_json_encode($score_data['score_breakdown']),
2120 + 'suggestions' => wp_json_encode($score_data['suggestions']),
1641 2121 'grade' => $score_data['grade'],
1642 2122 'algorithm_version' => $score_data['algorithm_version'] ?? '2024.1',
1643 2123 'calculated_at' => $score_data['calculated_at'],
1644 2124 'created_at' => current_time('mysql'),
@@ -1720,20 +2200,101 @@
1720 2200 * @param array $content_data Content analysis data
1721 2201 * @return array Scoring result
1722 2202 */
1723 2203 private function score_mobile_experience(array $content_data): array {
1724 - // Mobile experience is theme/site-level, not controlled by post content.
1725 - // Award full credit (benefit of the doubt) instead of a fixed partial
1726 - // that caps every post's ceiling.
2204 + $max = $this->scoring_factors['mobile_experience'];
2205 +
2206 + // Mobile experience is theme/site-level, not controlled by post content
2207 + // — but the plugin already measures it. When a mobile Lighthouse score
2208 + // has been collected, score against it; the "benefit of the doubt" below
2209 + // is for sites nobody has measured, not for sites measured as slow.
2210 + $performance_score = $this->measured_performance_score();
2211 +
2212 + if ($performance_score === null) {
2213 + return [
2214 + 'score' => $max,
2215 + 'max_score' => $max,
2216 + 'suggestions' => ['Ensure mobile-first design and fast loading on mobile devices'],
2217 + 'details' => ['mobile_score' => 'Assumed adequate', 'measured' => false],
2218 + ];
2219 + }
2220 +
2221 + $score = (int) round($max * $performance_score / 100);
2222 +
1727 2223 return [
1728 - 'score' => $this->scoring_factors['mobile_experience'],
1729 - 'max_score' => $this->scoring_factors['mobile_experience'],
1730 - 'suggestions' => ['Ensure mobile-first design and fast loading on mobile devices'],
1731 - 'details' => ['mobile_score' => 'Assumed adequate'],
2224 + 'score' => $score,
2225 + 'max_score' => $max,
2226 + 'suggestions' => $score < $max
2227 + ? ['Improve mobile page speed: the last PageSpeed run scored ' . $performance_score . '/100 on mobile']
2228 + : [],
2229 + 'details' => [
2230 + 'mobile_score' => $performance_score,
2231 + 'measured' => true,
2232 + 'source' => 'pagespeed_mobile',
2233 + ],
1732 2234 ];
1733 2235 }
1734 2236
1735 2237 /**
2238 + * The last collected mobile Lighthouse score, or null when unmeasured.
2239 + *
2240 + * Memoised per instance: compute_score() asks twice, and a post-list screen
2241 + * scores a page of posts at a time.
2242 + *
2243 + * Every failure — no performance module, no collected row, an unreadable
2244 + * table — resolves to null, which the callers read as "not measured" and
2245 + * answer with the full-credit fallback. A site is never penalised for
2246 + * ThinkRank being unable to look.
2247 + *
2248 + * @since 2.3.1
2249 + * @return int|null Score 0-100, or null when nothing has been collected.
2250 + */
2251 + private function measured_performance_score(): ?int {
2252 + $measurement = $this->measured_performance();
2253 +
2254 + if ($measurement === null || !isset($measurement['performance_score'])) {
2255 + return null;
2256 + }
2257 +
2258 + $score = $measurement['performance_score'];
2259 +
2260 + if (!is_numeric($score)) {
2261 + return null;
2262 + }
2263 +
2264 + return (int) round(max(0, min(100, (float) $score)));
2265 + }
2266 +
2267 + /**
2268 + * The last collected mobile measurement, or null when there is none.
2269 + *
2270 + * @since 2.3.1
2271 + * @return array|null { core_web_vitals: array, performance_score: float|null }
2272 + */
2273 + private function measured_performance(): ?array {
2274 + if ($this->measured_performance_resolved) {
2275 + return $this->measured_performance;
2276 + }
2277 +
2278 + $this->measured_performance_resolved = true;
2279 +
2280 + if (!class_exists('ThinkRank\\SEO\\Performance_Monitoring_Manager')) {
2281 + return null;
2282 + }
2283 +
2284 + try {
2285 + $manager = new \ThinkRank\SEO\Performance_Monitoring_Manager();
2286 + // Mobile deliberately: Google indexes mobile-first, and it is the
2287 + // device the mobile_experience factor is named after.
2288 + $this->measured_performance = $manager->get_stored_performance_measurement('mobile');
2289 + } catch (\Throwable $e) {
2290 + $this->measured_performance = null;
2291 + }
2292 +
2293 + return $this->measured_performance;
2294 + }
2295 +
2296 + /**
1736 2297 * Score core web vitals - 2025 version (3 points)
1737 2298 *
1738 2299 * @param array $content_data Content analysis data
1739 2300 * @return array Scoring result
@@ -1738,20 +2299,98 @@
1738 2299 * @param array $content_data Content analysis data
1739 2300 * @return array Scoring result
1740 2301 */
1741 2302 private function score_core_web_vitals(array $content_data): array {
1742 - // Core Web Vitals are a runtime/performance signal, not derivable from
1743 - // post content. Award full credit (benefit of the doubt) rather than a
1744 - // fixed partial that caps every post's ceiling.
2303 + $max = $this->scoring_factors['core_web_vitals'];
2304 +
2305 + // Not derivable from post content — but it is measured, and the audit
2306 + // stores LCP, INP and CLS with a rating each. Score against those when
2307 + // they exist; fall back to the benefit of the doubt when they do not.
2308 + $rated = $this->measured_vitals_score();
2309 +
2310 + if ($rated === null) {
2311 + return [
2312 + 'score' => $max,
2313 + 'max_score' => $max,
2314 + 'suggestions' => ['Optimize Core Web Vitals: LCP, INP, and CLS for better user experience'],
2315 + 'details' => ['vitals_status' => 'Assumed adequate', 'measured' => false],
2316 + ];
2317 + }
2318 +
2319 + $score = (int) round($max * $rated['average'] / 100);
2320 +
1745 2321 return [
1746 - 'score' => $this->scoring_factors['core_web_vitals'],
1747 - 'max_score' => $this->scoring_factors['core_web_vitals'],
1748 - 'suggestions' => ['Optimize Core Web Vitals: LCP, INP, and CLS for better user experience'],
1749 - 'details' => ['vitals_status' => 'Assumed adequate'],
2322 + 'score' => $score,
2323 + 'max_score' => $max,
2324 + // Gated on the measurement, not the rounded score: two good metrics
2325 + // and one needing improvement averages 88.33, which rounds to the
2326 + // full 3 of 3 and used to swallow the suggestion naming the metric
2327 + // that is actually failing.
2328 + 'suggestions' => !empty($rated['failing'])
2329 + ? ['Optimize Core Web Vitals: ' . implode(', ', $rated['failing']) . ' below target on mobile']
2330 + : [],
2331 + 'details' => [
2332 + 'vitals_status' => $rated['statuses'],
2333 + 'measured' => true,
2334 + 'source' => 'pagespeed_mobile',
2335 + ],
1750 2336 ];
1751 2337 }
1752 2338
1753 2339 /**
2340 + * Rate the collected Core Web Vitals, or null when none were measured.
2341 + *
2342 + * Reuses the per-metric score the performance module already assigns
2343 + * (good 100, needs improvement 65, poor 30) rather than inventing a second
2344 + * scale, so the SEO score and the performance card cannot disagree about
2345 + * whether a metric is healthy.
2346 + *
2347 + * Metrics with no stored value — fcp is not always collected — are skipped
2348 + * rather than counted as failures.
2349 + *
2350 + * @since 2.3.1
2351 + * @return array|null { average: float, statuses: array, failing: string[] }
2352 + */
2353 + private function measured_vitals_score(): ?array {
2354 + $measurement = $this->measured_performance();
2355 + $vitals = $measurement['core_web_vitals'] ?? null;
2356 +
2357 + if (!is_array($vitals)) {
2358 + return null;
2359 + }
2360 +
2361 + $scores = [];
2362 + $statuses = [];
2363 + $failing = [];
2364 +
2365 + // The three Google ranks on. fcp is diagnostic and not a Core Web Vital.
2366 + foreach (['lcp', 'inp', 'cls'] as $metric) {
2367 + $data = $vitals[$metric] ?? null;
2368 +
2369 + if (!is_array($data) || !isset($data['value'], $data['score']) || $data['value'] === null) {
2370 + continue;
2371 + }
2372 +
2373 + $scores[] = (float) $data['score'];
2374 + $statuses[$metric] = $data['status'] ?? 'unknown';
2375 +
2376 + if (($data['status'] ?? '') !== 'good') {
2377 + $failing[] = strtoupper($metric);
2378 + }
2379 + }
2380 +
2381 + if (empty($scores)) {
2382 + return null;
2383 + }
2384 +
2385 + return [
2386 + 'average' => array_sum($scores) / count($scores),
2387 + 'statuses' => $statuses,
2388 + 'failing' => $failing,
2389 + ];
2390 + }
2391 +
2392 + /**
1754 2393 * Score internal linking - declining importance (1 point)
1755 2394 *
1756 2395 * @param array $content_data Content analysis data
1757 2396 * @return array Scoring result
@@ -1781,9 +2420,9 @@
1781 2420 $suggestions = [];
1782 2421
1783 2422 // Meta description check
1784 2423 $meta_desc = $metadata['description'] ?? '';
1785 - if (!empty($meta_desc) && mb_strlen($meta_desc) >= 120 && mb_strlen($meta_desc) <= 160) {
2424 + if (!empty($meta_desc) && mb_strlen($meta_desc) >= self::DESCRIPTION_OPTIMAL_MIN && mb_strlen($meta_desc) <= self::DESCRIPTION_OPTIMAL_MAX) {
1786 2425 $score += 0.5;
1787 2426 } else {
1788 2427 $suggestions[] = 'Add a compelling meta description (120-160 characters)';
1789 2428 }
@@ -1814,19 +2453,30 @@
1814 2453 */
1815 2454 private function get_grade_from_score($score): string {
1816 2455 $score = (int) $score; // Ensure it's an integer
1817 2456
1818 - if ($score >= 95) return 'A+';
1819 - if ($score >= 90) return 'A';
1820 - if ($score >= 85) return 'A-';
1821 - if ($score >= 80) return 'B+';
1822 - if ($score >= 75) return 'B';
1823 - if ($score >= 70) return 'B-';
1824 - if ($score >= 65) return 'C+';
1825 - if ($score >= 60) return 'C';
1826 - if ($score >= 55) return 'C-';
1827 - if ($score >= 45) return 'D+';
1828 - if ($score >= 35) return 'D';
2457 + if ($score >= 95) { return 'A+';
2458 + }
2459 + if ($score >= 90) { return 'A';
2460 + }
2461 + if ($score >= 85) { return 'A-';
2462 + }
2463 + if ($score >= 80) { return 'B+';
2464 + }
2465 + if ($score >= 75) { return 'B';
2466 + }
2467 + if ($score >= 70) { return 'B-';
2468 + }
2469 + if ($score >= 65) { return 'C+';
2470 + }
2471 + if ($score >= 60) { return 'C';
2472 + }
2473 + if ($score >= 55) { return 'C-';
2474 + }
2475 + if ($score >= 45) { return 'D+';
2476 + }
2477 + if ($score >= 35) { return 'D';
2478 + }
1829 2479 return 'F';
1830 2480 }
1831 2481
1832 2482 /**
@@ -1854,11 +2504,12 @@
1854 2504
1855 2505 // Extract headings from content
1856 2506 $headings = $this->extract_headings($content);
1857 2507
1858 - // Count words using JavaScript-compatible method
1859 - $plain_text = wp_strip_all_tags($content);
1860 - $word_count = $this->calculate_word_count_js_style($plain_text);
2508 + // Count the markup, not wp_strip_all_tags() output: stripping deletes
2509 + // tags without a space, so "five</p><p>six" would already be one word
2510 + // before the counter saw it.
2511 + $word_count = $this->calculate_word_count_js_style($content);
1861 2512
1862 2513 // Calculate readability
1863 2514 $readability_score = $this->calculate_readability_score($content);
1864 2515
@@ -1885,27 +2536,66 @@
1885 2536 'images' => $images,
1886 2537 'url' => $url,
1887 2538 'slug' => $post->post_name,
1888 2539 'post_modified' => $post->post_modified,
1889 - 'schema_present' => $this->detect_schema_present($content) || $this->thinkrank_global_schema_active($post->post_type),
2540 + 'schema_present' => $this->detect_schema_present($content)
2541 + || $this->thinkrank_global_schema_active($post->post_type)
2542 + || $this->thinkrank_deployed_schema_active($post),
1890 2543 ];
1891 2544 }
1892 2545
1893 2546 /**
1894 - * Calculate word count using JavaScript-compatible method
1895 - * Matches the logic in contentAnalysis.js for consistency
2547 + * Count the words a reader reads in HTML or text.
1896 2548 *
1897 - * @param string $text Text to count words in
2549 + * The same extraction and counting as the thin content report
2550 + * ({@see \ThinkRank\SEO\Word_Count_Index::reading_text()} and
2551 + * {@see \ThinkRank\SEO\Word_Count_Index::count_words()}), so the editor
2552 + * score and the report give one number for one page. Splitting on
2553 + * whitespace counted shortcode syntax as words: a WPBakery page with 216
2554 + * words of prose scored a word count of 325 while the report said 216
2555 + * (#893 fixed the report only). It also counted tokens of punctuation
2556 + * alone, such as a full stop after a link.
2557 + *
2558 + * Always words, whatever the locale's unit, because every threshold that
2559 + * reads this value (content length, long paragraphs, headings per 300
2560 + * words, readability, keyword density) is in words.
2561 + *
2562 + * contentAnalysis.js calculateWordCount() applies the same two rules in
2563 + * the editor.
2564 + *
2565 + * @param string $text HTML or text to count words in.
1898 2566 * @return int Word count
1899 2567 */
1900 2568 private function calculate_word_count_js_style(string $text): int {
1901 - if (empty($text)) {
2569 + if ('' === $text) {
1902 2570 return 0;
1903 2571 }
1904 2572
1905 - // Match JavaScript: trim, split by whitespace, filter empty
1906 - $words = preg_split('/\s+/', trim($text), -1, PREG_SPLIT_NO_EMPTY);
1907 - return count($words);
2573 + return \ThinkRank\SEO\Word_Count_Index::count_words(self::reading_text_of($text));
2574 + }
2575 +
2576 + /**
2577 + * The text a reader reads in HTML or text, as the word count sees it.
2578 + *
2579 + * Every check that divides by the word count (readability, keyword
2580 + * density, topic relevance) reads its numerator from this same text, so
2581 + * shortcode syntax is never on one side of a ratio and not the other.
2582 + *
2583 + * @since 2.14.2
2584 + *
2585 + * @param string $content HTML or text.
2586 + * @return string Plain text, whitespace collapsed.
2587 + */
2588 + private static function reading_text_of(string $content): string {
2589 + if ('' === $content) {
2590 + return '';
2591 + }
2592 +
2593 + if (!class_exists('\ThinkRank\SEO\Word_Count_Index')) {
2594 + require_once THINKRANK_PLUGIN_DIR . 'includes/seo/class-word-count-index.php';
2595 + }
2596 +
2597 + return \ThinkRank\SEO\Word_Count_Index::reading_text($content);
1908 2598 }
1909 2599
1910 2600 /**
1911 2601 * Get existing score data for a post from database