PluginProbe
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO / 2.14.0
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO v2.14.0
2.14.0 2.13.0 2.12.0 2.11.0 2.10.0 2.9.0 2.8.0 2.7.0 2.6.0 2.5.0 2.4.0 2.3.0 2.2.0 2.1.1 2.1.0 2.0.2 2.0.1 2.0.0 1.32.0 1.31.0 1.30.0 1.29.0 1.28.0 1.27.0 1.26.0 All 55 releases
thinkrank / includes / ai / class-seo-score-calculator.php

class-seo-score-calculator.php in ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO 2.14.0, at includes/ai/class-seo-score-calculator.php

2,579 lines 99.4 KB
No matching file
Up and down to move Enter to open Esc to close
Raw Download Zip
1 <?php
2 /**
3 * SEO Score Calculator
4 *
5 * Advanced SEO scoring system based on 2025 Google ranking factors
6 * Implements AI-driven content analysis, searcher engagement signals, and current SEO best practices
7 *
8 * @package ThinkRank\AI
9 * @since 1.0.0
10 */
11
12 declare(strict_types=1);
13
14 namespace ThinkRank\AI;
15
16 use ThinkRank\Core\Database;
17
18 // Prevent direct access
19 if (!defined('ABSPATH')) {
20 exit;
21 }
22
23 /**
24 * SEO Score Calculator Class
25 *
26 * Implements 2025 SEO scoring algorithm based on:
27 * - Google's Q1 2025 algorithm updates (First Page Sage research)
28 * - Satisfying content as #1 ranking factor (23%)
29 * - Searcher engagement and intent satisfaction (12%)
30 * - Mobile Experience Score (MES) and Core Web Vitals 2.0
31 * - Content freshness and niche expertise signals
32 *
33 * @since 1.0.0
34 */
35 class SEOScoreCalculator {
36
37 /**
38 * Database instance
39 *
40 * @var Database
41 */
42 private Database $database;
43
44 /**
45 * Memoised collected performance measurement, and whether it was resolved.
46 *
47 * Two factors read it and both may be asked for on every post in a list, so
48 * the lookup happens once per calculator. `null` is a real answer here — the
49 * separate flag keeps "not looked up yet" distinct from "nothing measured".
50 *
51 * @since 2.3.1
52 * @var array|null
53 */
54 private ?array $measured_performance = null;
55
56 /**
57 * @since 2.3.1
58 * @var bool
59 */
60 private bool $measured_performance_resolved = false;
61
62 /**
63 * Length bands the editor scores against, in characters.
64 *
65 * Public so every surface that judges a title or description — the editor
66 * score and the Bulk Snippets problem filter — reads one set of numbers.
67 * Before these existed the bands were literals inside the scoring methods,
68 * and a second screen would have had to copy them and drift (#727).
69 *
70 * @since 2.8.0
71 */
72 public const TITLE_OPTIMAL_MIN = 35;
73 public const TITLE_OPTIMAL_MAX = 60;
74 public const DESCRIPTION_OPTIMAL_MIN = 120;
75 public const DESCRIPTION_OPTIMAL_MAX = 160;
76
77 /**
78 * 2025 SEO scoring factors (Q1 2025 Google Algorithm)
79 * Based on First Page Sage research and Google's latest updates
80 *
81 * @var array
82 */
83 private array $scoring_factors = [
84 'satisfying_content' => 23, // #1: Consistent publication of satisfying content
85 'title_optimization' => 14, // #2: Keyword in meta title (looser matching)
86 'niche_expertise' => 13, // #3: Hub & spoke content clusters
87 'searcher_engagement' => 12, // #4: Dwell time, bounce rate, pages/session
88 'backlink_authority' => 13, // #5: Quality backlinks (declining but important)
89 'content_freshness' => 6, // #6: Quarterly content updates
90 'mobile_experience' => 5, // #7: Mobile Experience Score (MES) - NEW 2025
91 'trustworthiness' => 4, // #8: E-E-A-T verification
92 'link_diversity' => 3, // #9: Link distribution across multiple pages
93 'core_web_vitals' => 3, // #10: Page speed + Interaction Readiness
94 'site_security' => 2, // #11: SSL certificate
95 'internal_linking' => 1, // #12: Declining importance
96 'technical_factors' => 1, // #13: Meta descriptions, schema, etc.
97 ];
98
99 /**
100 * Constructor
101 *
102 * @param Database $database Database instance
103 */
104 public function __construct(Database $database) {
105 $this->database = $database;
106 }
107
108 /**
109 * Calculate comprehensive modern SEO score.
110 *
111 * Supports multiple focus keywords: when `$options['target_keywords']` holds
112 * more than one keyword the score is computed independently for each and the
113 * HIGHEST overall score is returned as the final result, with per-keyword
114 * results (`keyword_results`) and OR-combined per-check matches
115 * (`keyword_checks`) attached. A single keyword (or the legacy
116 * `target_keyword` option) falls through to the single-keyword path.
117 *
118 * @param array $content_data Content analysis data
119 * @param array $metadata Post metadata
120 * @param array $options Additional options
121 * @return array Complete scoring result
122 */
123 public function calculate_score(array $content_data, array $metadata, array $options = []): array {
124 $keywords = $this->resolve_target_keywords($options, $metadata);
125
126 if (count($keywords) > 1) {
127 return $this->calculate_score_multi($content_data, $metadata, $keywords, $options);
128 }
129
130 $options['target_keyword'] = $keywords[0] ?? '';
131 $result = $this->compute_score($content_data, $metadata, $options);
132
133 // Expose the keyword surface uniformly so consumers can rely on it
134 // regardless of how many keywords were supplied.
135 $result['target_keywords'] = $keywords;
136 if (!empty($keywords)) {
137 $result['keyword_results'] = [[
138 'keyword' => $keywords[0],
139 'overall_score' => $result['overall_score'],
140 'grade' => $result['grade'],
141 ]];
142 $result['keyword_checks'] = $this->analyze_keyword_checks($content_data, $metadata, $keywords);
143 }
144 $result['keywords'] = $this->keyword_placements($content_data, $metadata, $keywords);
145
146 return $result;
147 }
148
149 /**
150 * Resolve the target keyword list from the options array, falling back to
151 * the focus keywords carried on the post's metadata.
152 *
153 * Accepts `target_keywords` (array) or the legacy `target_keyword` (string),
154 * trims, drops empties and removes case-insensitive duplicates.
155 *
156 * Options win over metadata on purpose: the editor scores unsaved keyword
157 * edits by passing them explicitly, and that live value must beat whatever
158 * is currently persisted. The metadata fallback applies only when the
159 * caller mentions no keyword option AT ALL — a caller that passes an empty
160 * keyword is deliberately clearing it (the editor does exactly this when
161 * the field is emptied), so the stored value must not resurrect it.
162 *
163 * Without the fallback, every caller that hands over
164 * `Metabox_Manager::get_post_metadata()` (the metabox and the MCP scoring
165 * abilities) silently scored as if no keyword were set.
166 *
167 * @param array $options Scoring options.
168 * @param array $metadata Post metadata (may carry focus_keyword(s)).
169 * @return string[] Normalized keyword list.
170 */
171 private function resolve_target_keywords(array $options, array $metadata = []): array {
172 $raw = [];
173 if (!empty($options['target_keywords']) && is_array($options['target_keywords'])) {
174 $raw = $options['target_keywords'];
175 } elseif (isset($options['target_keyword']) && $options['target_keyword'] !== '') {
176 $raw = [$options['target_keyword']];
177 } elseif (!$this->options_mention_keywords($options)) {
178 if (!empty($metadata['focus_keywords']) && is_array($metadata['focus_keywords'])) {
179 $raw = $metadata['focus_keywords'];
180 } elseif (isset($metadata['focus_keyword']) && is_string($metadata['focus_keyword']) && $metadata['focus_keyword'] !== '') {
181 $raw = [$metadata['focus_keyword']];
182 }
183 }
184
185 $seen = [];
186 $keywords = [];
187 foreach ($raw as $keyword) {
188 $keyword = trim((string) $keyword);
189 if ($keyword === '') {
190 continue;
191 }
192 $key = strtolower($keyword);
193 if (isset($seen[$key])) {
194 continue;
195 }
196 $seen[$key] = true;
197 $keywords[] = $keyword;
198 }
199
200 return $keywords;
201 }
202
203 /**
204 * Whether the caller said anything about keywords — including saying
205 * "none". Distinguishes an intentional clear (score with no keyword) from
206 * silence (fall back to the post's stored focus keywords).
207 *
208 * @param array $options Scoring options.
209 * @return bool True when a keyword option key is present.
210 */
211 private function options_mention_keywords(array $options): bool {
212 return array_key_exists('target_keywords', $options)
213 || array_key_exists('target_keyword', $options);
214 }
215
216 /**
217 * Score each keyword independently and return the highest-scoring result.
218 *
219 * @param array $content_data Content analysis data.
220 * @param array $metadata Post metadata.
221 * @param string[] $keywords Target keywords (already normalized, 2+).
222 * @param array $options Additional options.
223 * @return array Best scoring result, with per-keyword data attached.
224 */
225 private function calculate_score_multi(array $content_data, array $metadata, array $keywords, array $options): array {
226 $per_keyword = [];
227 $best = null;
228 $best_keyword = $keywords[0];
229
230 foreach ($keywords as $keyword) {
231 $opts = $options;
232 unset($opts['target_keywords']);
233 $opts['target_keyword'] = $keyword;
234
235 $result = $this->compute_score($content_data, $metadata, $opts);
236
237 $per_keyword[] = [
238 'keyword' => $keyword,
239 'overall_score' => $result['overall_score'],
240 'grade' => $result['grade'],
241 'score_breakdown' => $result['score_breakdown'],
242 ];
243
244 if ($best === null || $result['overall_score'] > $best['overall_score']) {
245 $best = $result;
246 $best_keyword = $keyword;
247 }
248 }
249
250 // Final score = highest individual keyword score. Retain per-keyword
251 // results and OR-combined checks so the UI can show both.
252 $best['target_keyword'] = $best_keyword;
253 $best['target_keywords'] = $keywords;
254 $best['keyword_results'] = $per_keyword;
255 $best['keyword_checks'] = $this->analyze_keyword_checks($content_data, $metadata, $keywords);
256 $best['keywords'] = $this->keyword_placements($content_data, $metadata, $keywords);
257
258 return $best;
259 }
260
261 /**
262 * Evaluate per-location keyword checks across ALL focus keywords.
263 *
264 * Each check (title, meta description, content, image alt, slug) passes when
265 * ANY of the focus keywords matches that location.
266 *
267 * @param array $content_data Content analysis data.
268 * @param array $metadata Post metadata.
269 * @param string[] $keywords Target keywords.
270 * @return array<string,array{passed:bool,matched_keywords:string[]}>
271 */
272 private function analyze_keyword_checks(array $content_data, array $metadata, array $keywords): array {
273 $title = self::lower((string) ($metadata['title'] ?? $content_data['title'] ?? ''));
274 $description = self::lower((string) ($metadata['description'] ?? ''));
275 $content = self::lower(self::plain_text((string) ($content_data['content'] ?? '')));
276
277 $alts = '';
278 foreach ((array) ($content_data['images'] ?? []) as $image) {
279 $alts .= ' ' . self::lower((string) ($image['alt'] ?? ''));
280 }
281
282 // Build a searchable slug haystack from the post's OWN slug — never the
283 // full URL path. The path carries ancestors, category bases and date
284 // segments, so a child of /clinical-trials/ reported "keyword in slug"
285 // for a page actually slugged `contact-us`. It also breaks the other
286 // way: an unpublished post has no pretty permalink (get_permalink()
287 // returns ?p=123), so the path held no slug at all and every draft
288 // scored "no match" until it was published. Hyphens/underscores become
289 // spaces so multi-word keywords can match.
290 $slug = self::lower(self::slug_haystack($content_data));
291
292 $haystacks = [
293 'title' => trim($title),
294 'meta_description' => trim($description),
295 'content' => trim($content),
296 'image_alt' => trim($alts),
297 'slug' => trim($slug),
298 ];
299
300 $checks = [];
301 foreach ($haystacks as $location => $haystack) {
302 $matched = [];
303 foreach ($keywords as $keyword) {
304 if ($this->keyword_matches($haystack, self::lower(trim($keyword)))) {
305 $matched[] = $keyword;
306 }
307 }
308 $checks[$location] = [
309 'passed' => !empty($matched),
310 'matched_keywords' => $matched,
311 ];
312 }
313
314 return $checks;
315 }
316
317 /**
318 * The keyword placements the editor draws one gauge segment for, in the
319 * order a reader meets them (#729).
320 *
321 * @since 2.11.0
322 * @var string[]
323 */
324 public const PLACEMENTS = ['title', 'meta_description', 'slug', 'first_paragraph', 'subheading', 'content', 'image_alt'];
325
326 /**
327 * Characters of plain text read as the opening when the content has no
328 * paragraph tag.
329 *
330 * @since 2.11.0
331 */
332 private const OPENING_CHARS = 300;
333
334 /**
335 * Where each focus keyword is placed, keyword by keyword (#729).
336 *
337 * analyze_keyword_checks() answers "does ANY keyword appear here" for
338 * five places; this answers "where does THIS keyword appear" for seven,
339 * so the editor can show each keyword's own gauge. Same matcher, so a
340 * keyword counts as a word (not a fragment) and a keyword in a script
341 * written without spaces (Thai, Chinese, Japanese) still matches.
342 *
343 * `where` names the heading or alt text that matched, so the editor can
344 * say which one.
345 *
346 * @since 2.11.0
347 *
348 * @param array $content_data Content analysis data.
349 * @param array $metadata Post metadata (title, description).
350 * @param string[] $keywords Focus keywords.
351 * @return array<int, array{keyword: string, passed: int, total: int, placements: array<string, array{passed: bool, where: string}>}>
352 */
353 public function keyword_placements(array $content_data, array $metadata, array $keywords): array {
354 $html = (string) ($content_data['content'] ?? '');
355 $plain = self::lower(self::plain_text($html));
356
357 $headings = [];
358 $source = isset($content_data['headings']) && is_array($content_data['headings']) ? $content_data['headings'] : $this->extract_headings($html);
359 foreach ($source as $heading) {
360 $text = trim((string) ($heading['text'] ?? ''));
361 if ((int) ($heading['level'] ?? 0) >= 2 && '' !== $text) {
362 $headings[] = $text;
363 }
364 }
365
366 $alts = [];
367 foreach ((array) ($content_data['images'] ?? []) as $image) {
368 $alt = trim((string) ($image['alt'] ?? ''));
369 if ('' !== $alt) {
370 $alts[] = $alt;
371 }
372 }
373
374 $single = [
375 'title' => self::lower((string) ($metadata['title'] ?? $content_data['title'] ?? '')),
376 'meta_description' => self::lower((string) ($metadata['description'] ?? '')),
377 'slug' => self::lower(self::slug_haystack($content_data)),
378 'first_paragraph' => self::lower(self::opening($html)),
379 'content' => $plain,
380 ];
381
382 $out = [];
383 foreach ($keywords as $keyword) {
384 $needle = self::lower(trim((string) $keyword));
385 $placements = [];
386
387 foreach (self::PLACEMENTS as $placement) {
388 if ('subheading' === $placement || 'image_alt' === $placement) {
389 $where = '';
390 foreach ('subheading' === $placement ? $headings : $alts as $text) {
391 if ($this->keyword_matches(self::lower($text), $needle)) {
392 $where = $text;
393 break;
394 }
395 }
396 $placements[$placement] = ['passed' => '' !== $where, 'where' => $where];
397 continue;
398 }
399
400 $placements[$placement] = ['passed' => $this->keyword_matches($single[$placement], $needle), 'where' => ''];
401 }
402
403 $out[] = [
404 'keyword' => (string) $keyword,
405 'passed' => count(array_filter(array_column($placements, 'passed'))),
406 'total' => count(self::PLACEMENTS),
407 'placements' => $placements,
408 ];
409 }
410
411 return $out;
412 }
413
414 /**
415 * The post's own slug as searchable text: hyphens and underscores become
416 * spaces so a multi-word keyword can match, and a slug WordPress
417 * percent-encoded (Thai, Cyrillic, Chinese) is decoded, or it could never
418 * match a keyword typed in that script.
419 *
420 * Never the full URL path: the path carries ancestors, category bases and
421 * date segments, so a child of /clinical-trials/ reported "keyword in slug"
422 * for a page actually slugged `contact-us`. An unpublished post has no
423 * pretty permalink either, so the path held no slug at all.
424 *
425 * @param array $content_data Content analysis data.
426 * @return string
427 */
428 private static function slug_haystack(array $content_data): string {
429 $slug = (string) ($content_data['slug'] ?? '');
430 if ('' === $slug) {
431 // Draft with no slug assigned yet: score what WordPress would
432 // generate from the title, which is what the editor shows as the
433 // proposed URL — so the check reads the same before and after
434 // publishing instead of flipping.
435 $slug = sanitize_title((string) ($content_data['title'] ?? ''));
436 }
437
438 return trim(str_replace(['-', '_'], ' ', rawurldecode($slug)));
439 }
440
441 /**
442 * The opening of the content: its first paragraph with text, or the first
443 * few hundred characters when it has none. Counted in characters, not
444 * words, so a language written without spaces is not read as one word.
445 *
446 * @param string $html Content HTML.
447 * @return string Plain text.
448 */
449 private static function opening(string $html): string {
450 if (preg_match_all('/<p\b[^>]*>(.*?)<\/p>/isu', $html, $matches)) {
451 foreach ($matches[1] as $paragraph) {
452 $text = self::collapse_whitespace(wp_strip_all_tags($paragraph));
453 if ('' !== $text) {
454 return $text;
455 }
456 }
457 }
458
459 $text = self::plain_text($html);
460
461 return function_exists('mb_substr') ? mb_substr($text, 0, self::OPENING_CHARS) : substr($text, 0, self::OPENING_CHARS);
462 }
463
464 /**
465 * Content as plain text, with a space where each tag was. wp_strip_all_tags()
466 * alone joins neighbouring blocks — "…coffee grinder</h3><p>A good…" became
467 * "coffee grinderA good" — so a keyword at the end of a heading or a
468 * paragraph was no longer a word and did not match.
469 *
470 * @param string $html Content HTML.
471 * @return string
472 */
473 private static function plain_text(string $html): string {
474 $spaced = preg_replace('/<[^>]+>/', ' $0 ', $html);
475
476 return self::collapse_whitespace(wp_strip_all_tags(null === $spaced ? $html : (string) $spaced));
477 }
478
479 /**
480 * Runs of whitespace down to one space.
481 *
482 * The `/u` pass is the one that understands a multibyte space, but
483 * preg_replace() answers null on bytes that are not valid UTF-8 rather than
484 * throwing — and casting that null to a string blanked the haystack, so a
485 * post carrying one mojibake byte (a Latin-1 paste, an old import) reported
486 * every keyword as missing from its body, its opening, and every
487 * subheading. The gauge said 0/7 and told the author to add a keyword that
488 * was already there.
489 *
490 * Falls back to the byte-wise collapse, which is what this did before the
491 * multibyte work added the modifier. Same reasoning keyword_matches()
492 * already records for its own PCRE failure: a pattern PCRE refuses must not
493 * be reported as a confident "no match".
494 *
495 * @param string $text Text to collapse.
496 * @return string
497 */
498 private static function collapse_whitespace(string $text): string {
499 $collapsed = preg_replace('/\s+/u', ' ', $text);
500
501 if (null === $collapsed) {
502 $collapsed = preg_replace('/\s+/', ' ', $text);
503 }
504
505 return trim(null === $collapsed ? $text : (string) $collapsed);
506 }
507
508 /**
509 * Lowercase in any script. strtolower() only folds ASCII, so "Кофе" never
510 * matched "кофе".
511 *
512 * @param string $text Text.
513 * @return string
514 */
515 private static function lower(string $text): string {
516 return function_exists('mb_strtolower') ? mb_strtolower($text, 'UTF-8') : strtolower($text);
517 }
518
519 /**
520 * Scripts written without spaces between words.
521 *
522 * @since 2.1.0
523 * @var string
524 */
525 private const SCRIPTIO_CONTINUA = '/[\p{Han}\p{Hiragana}\p{Katakana}\p{Thai}\p{Lao}\p{Khmer}\p{Myanmar}]/u';
526
527 /**
528 * Whether a keyword appears in a haystack as a word rather than as a
529 * fragment of a longer one.
530 *
531 * The five keyword checks used a plain strpos(), so any substring hit
532 * counted: "test coronavirus" matched "la|test coronavirus|news", "art"
533 * matched "start", "cat" matched "category". The panel then confidently
534 * reported a keyword placement that does not exist (#416). Same class of
535 * problem #71 fixed in the Image SEO rewriter, and the same remedy.
536 *
537 * Both arguments are expected lowercased already.
538 *
539 * @since 2.1.0
540 *
541 * @param string $haystack Text to search.
542 * @param string $needle Keyword, lowercased and trimmed.
543 * @return bool
544 */
545 private function keyword_matches(string $haystack, string $needle): bool {
546 if ($needle === '' || $haystack === '') {
547 return false;
548 }
549
550 if (!$this->supports_word_boundaries($needle)) {
551 return strpos($haystack, $needle) !== false;
552 }
553
554 $matched = preg_match('/\b' . preg_quote($needle, '/') . '\b/u', $haystack);
555
556 // PCRE refusing the pattern — invalid UTF-8 in the keyword, a
557 // backtrack limit — must not be reported as a confident "no match".
558 // Fall back to the behaviour this replaced rather than invent a
559 // negative the user cannot explain.
560 if ($matched === false) {
561 return strpos($haystack, $needle) !== false;
562 }
563
564 return $matched === 1;
565 }
566
567 /**
568 * Whether \b can express "this keyword, as a word" for this keyword.
569 *
570 * It asserts a transition between a word and a non-word character, which
571 * only means something where words are separated. Two cases where it is
572 * not, both verified against PCRE rather than assumed:
573 *
574 * - the keyword's own edges are not word characters ("c++", "#seo"), so
575 * no boundary can assert there and a real match is lost;
576 * - scripts written without spaces, where the neighbouring characters
577 * are word characters too — "冠状�
578 毒" inside "最新冠状�
579 毒新闻" is a
580 * legitimate match that \b never sees.
581 *
582 * Accented Latin and Cyrillic need no special handling: PHP's /u modifier
583 * turns on Unicode character properties, so "café" correctly does not
584 * match "cafés" and "коронавирус" does not match "коронавирусный".
585 *
586 * @since 2.1.0
587 *
588 * @param string $needle Keyword, lowercased and trimmed.
589 * @return bool
590 */
591 private function supports_word_boundaries(string $needle): bool {
592 if (preg_match(self::SCRIPTIO_CONTINUA, $needle)) {
593 return false;
594 }
595
596 return preg_match('/^\w/u', $needle) === 1 && preg_match('/\w$/u', $needle) === 1;
597 }
598
599 /**
600 * Compute the SEO score for a single target keyword.
601 *
602 * @param array $content_data Content analysis data
603 * @param array $metadata Post metadata
604 * @param array $options Additional options (expects scalar target_keyword)
605 * @return array Complete scoring result
606 *
607 * @throws \Exception On failure.
608 */
609 private function compute_score(array $content_data, array $metadata, array $options = []): array {
610 $scores = [];
611 $suggestions = [];
612 $total_score = 0;
613
614 try {
615 // 1. Satisfying Content (23 points) - #1 factor in 2025
616 $satisfying_result = $this->score_satisfying_content($content_data, $options['target_keyword'] ?? '', $metadata);
617 $scores['satisfying_content'] = $satisfying_result;
618 $total_score += $satisfying_result['score'];
619 $suggestions = array_merge($suggestions, $satisfying_result['suggestions']);
620 } catch (\Exception $e) {
621 throw $e;
622 }
623
624 try {
625 // 2. Title Optimization (14 points) - Looser keyword matching in 2025
626 $title_result = $this->score_2025_title_optimization($metadata['title'] ?? '', $options['target_keyword'] ?? '');
627 $scores['title_optimization'] = $title_result;
628 $total_score += $title_result['score'];
629 $suggestions = array_merge($suggestions, $title_result['suggestions']);
630 } catch (\Exception $e) {
631 throw $e;
632 }
633
634 try {
635 // 3. Niche Expertise (13 points) - Hub & spoke content clusters
636 $expertise_result = $this->score_niche_expertise($content_data, $options['target_keyword'] ?? '');
637 $scores['niche_expertise'] = $expertise_result;
638 $total_score += $expertise_result['score'];
639 $suggestions = array_merge($suggestions, $expertise_result['suggestions']);
640 } catch (\Exception $e) {
641 throw $e;
642 }
643
644 try {
645 // 4. Searcher Engagement (12 points) - Dwell time, bounce rate, pages/session
646 $engagement_result = $this->score_searcher_engagement($content_data);
647 $scores['searcher_engagement'] = $engagement_result;
648 $total_score += $engagement_result['score'];
649 $suggestions = array_merge($suggestions, $engagement_result['suggestions']);
650 } catch (\Exception $e) {
651 throw $e;
652 }
653
654 try {
655 // 5. Backlink Authority (13 points) - Quality backlinks
656 $backlink_result = $this->score_backlink_authority($content_data);
657 $scores['backlink_authority'] = $backlink_result;
658 $total_score += $backlink_result['score'];
659 $suggestions = array_merge($suggestions, $backlink_result['suggestions']);
660 } catch (\Exception $e) {
661 throw $e;
662 }
663
664 // 6. Content Freshness (6 points) - Quarterly updates priority
665 $freshness_result = $this->score_content_freshness($content_data);
666 $scores['content_freshness'] = $freshness_result;
667 $total_score += $freshness_result['score'];
668 $suggestions = array_merge($suggestions, $freshness_result['suggestions']);
669
670 // 7. Mobile Experience (5 points) - NEW: Mobile Experience Score (MES)
671 $mobile_result = $this->score_mobile_experience($content_data);
672 $scores['mobile_experience'] = $mobile_result;
673 $total_score += $mobile_result['score'];
674 $suggestions = array_merge($suggestions, $mobile_result['suggestions']);
675
676 // 8. Trustworthiness (4 points) - E-E-A-T verification
677 $trust_result = $this->score_trustworthiness($content_data, $metadata);
678 $scores['trustworthiness'] = $trust_result;
679 $total_score += $trust_result['score'];
680 $suggestions = array_merge($suggestions, $trust_result['suggestions']);
681
682 // 9. Link Diversity (3 points) - Multiple pages with backlinks
683 $diversity_result = $this->score_link_diversity($content_data);
684 $scores['link_diversity'] = $diversity_result;
685 $total_score += $diversity_result['score'];
686 $suggestions = array_merge($suggestions, $diversity_result['suggestions']);
687
688 // 10. Core Web Vitals (3 points) - Interaction Readiness + CLS 2.0
689 $vitals_result = $this->score_core_web_vitals($content_data);
690 $scores['core_web_vitals'] = $vitals_result;
691 $total_score += $vitals_result['score'];
692 $suggestions = array_merge($suggestions, $vitals_result['suggestions']);
693
694 // 11. Site Security (2 points) - SSL certificate
695 $security_result = $this->score_site_security($content_data);
696 $scores['site_security'] = $security_result;
697 $total_score += $security_result['score'];
698 $suggestions = array_merge($suggestions, $security_result['suggestions']);
699
700 // 12. Internal Linking (1 point) - Declining importance
701 $internal_result = $this->score_internal_linking($content_data);
702 $scores['internal_linking'] = $internal_result;
703 $total_score += $internal_result['score'];
704 $suggestions = array_merge($suggestions, $internal_result['suggestions']);
705
706 // 13. Technical Factors (1 point) - Meta descriptions, schema, etc.
707 $technical_result = $this->score_technical_factors($content_data, $metadata, $options);
708 $scores['technical_factors'] = $technical_result;
709 $total_score += $technical_result['score'];
710 $suggestions = array_merge($suggestions, $technical_result['suggestions']);
711
712 try {
713 $prioritized_suggestions = $this->prioritize_suggestions($suggestions, $scores);
714 $grade = $this->get_grade_from_score($total_score);
715
716 return [
717 'overall_score' => min(100, $total_score),
718 'score_breakdown' => $scores,
719 'suggestions' => $prioritized_suggestions,
720 'grade' => $grade,
721 // Readability + content-quality labels so persisted scores
722 // (e.g. bulk-analyzed on import) populate the post-list columns
723 // without a manual re-analyze. The REST endpoint still overrides
724 // these with live-editor values when the metabox provides them.
725 'readability_score' => $this->format_readability_label($content_data),
726 'content_quality' => $this->derive_content_quality($content_data),
727 'calculated_at' => current_time('mysql'),
728 'algorithm_version' => '2025.2',
729 'algorithm_source' => 'First Page Sage Q1 2025 Research',
730 'factors_count' => count($scores),
731 ];
732 } catch (\Exception $e) {
733 throw $e;
734 }
735 }
736
737 /**
738 * Score satisfying content - #1 factor in 2025 (23 points)
739 * Google tests content to see if it satisfies search intent
740 *
741 * @param array $content_data Content analysis data
742 * @param string $target_keyword Target keyword
743 * @param array $metadata Post metadata (title, description)
744 * @return array Scoring result
745 */
746 private function score_satisfying_content(array $content_data, string $target_keyword, array $metadata = []): array {
747 $score = 0;
748 $max_score = $this->scoring_factors['satisfying_content'];
749 $suggestions = [];
750
751 $content = $content_data['content'] ?? '';
752 $word_count = $content_data['word_count'] ?? 0;
753 $meta_description = (string) ($metadata['description'] ?? '');
754
755 // Content depth and comprehensiveness (8 points). Tiers softened so a
756 // genuinely useful post is not capped the way the old 2000-word gate did
757 // (Rank Math awards full content credit well below 2000 words).
758 if ($word_count >= 1500) {
759 $score += 8;
760 } elseif ($word_count >= 1000) {
761 $score += 7;
762 $suggestions[] = 'Consider expanding content to 1500+ words for more comprehensive coverage';
763 } elseif ($word_count >= 600) {
764 $score += 5;
765 $suggestions[] = 'Content is adequate - aim for 1000+ words for stronger topic depth';
766 } elseif ($word_count >= 300) {
767 $score += 3;
768 $suggestions[] = 'Content is thin - aim for 600+ words minimum';
769 } else {
770 $score++;
771 $suggestions[] = 'Content too shallow - Google prioritizes comprehensive, satisfying content';
772 }
773
774 // Keyword presence & placement (8 points) - deterministic, replaces the
775 // old literal-phrase intent heuristic ("what is"/"because"). Measures
776 // signals the editor actually controls: body (3), first paragraph (3),
777 // meta description (2) - the last mirrors Rank Math's "keyword in meta
778 // description" basic-SEO check.
779 if (empty($target_keyword)) {
780 $score += 4; // Benefit of the doubt when no focus keyword is set.
781 $suggestions[] = 'Set a focus keyword so content relevance can be measured';
782 } else {
783 if ($this->keyword_in_content($content, $target_keyword)) {
784 $score += 3;
785 } else {
786 $suggestions[] = "Use the focus keyword '{$target_keyword}' in the body content";
787 }
788 if ($this->keyword_in_first_paragraph($content, $target_keyword)) {
789 $score += 3;
790 } else {
791 $suggestions[] = "Mention '{$target_keyword}' near the start of the content (first paragraph)";
792 }
793 if ($this->keyword_in_meta($meta_description, $target_keyword)) {
794 $score += 2;
795 } else {
796 $suggestions[] = "Include the focus keyword '{$target_keyword}' in the meta description";
797 }
798 }
799
800 // Content structure & value (7 points) - reuses the deterministic
801 // content-quality signal (length, paragraph length, subheading
802 // distribution) instead of the noisy sentence-length variety heuristic.
803 $quality = $this->derive_content_quality_score($content_data); // 0-100
804 $score += (int) round(($quality / 100) * 7);
805
806 if ($quality < 60) {
807 $suggestions[] = 'Improve content structure - break up long paragraphs and add subheadings';
808 }
809
810 return [
811 'score' => $score,
812 'max_score' => $max_score,
813 'suggestions' => $suggestions,
814 'details' => [
815 'word_count' => $word_count,
816 'keyword_in_content' => !empty($target_keyword) && $this->keyword_in_content($content, $target_keyword),
817 'keyword_in_first_paragraph' => !empty($target_keyword) && $this->keyword_in_first_paragraph($content, $target_keyword),
818 'keyword_in_meta_description' => !empty($target_keyword) && $this->keyword_in_meta($meta_description, $target_keyword),
819 'content_quality_score' => $quality,
820 'content_depth' => $this->assess_content_depth_2025($word_count),
821 ]
822 ];
823 }
824
825 /**
826 * Assess content depth for 2025 standards
827 *
828 * @param int $word_count Word count
829 * @return string Depth assessment
830 */
831 private function assess_content_depth_2025(int $word_count): string {
832 if ($word_count >= 3000) { return 'Comprehensive';
833 }
834 if ($word_count >= 2000) { return 'Detailed';
835 }
836 if ($word_count >= 1200) { return 'Adequate';
837 }
838 if ($word_count >= 800) { return 'Basic';
839 }
840 return 'Insufficient';
841 }
842
843 /**
844 * Score 2025 title optimization with looser keyword matching
845 *
846 * @param string $title Post title
847 * @param string $target_keyword Target keyword
848 * @return array Scoring result
849 */
850 private function score_2025_title_optimization(string $title, string $target_keyword): array {
851 $score = 0;
852 $max_score = $this->scoring_factors['title_optimization'];
853 $suggestions = [];
854
855 if (empty($title)) {
856 $suggestions[] = 'Add a compelling, click-worthy title that matches search intent';
857 return ['score' => 0, 'max_score' => $max_score, 'suggestions' => $suggestions];
858 }
859
860 $title_length = mb_strlen($title);
861
862 // 2025 length optimization (6 points). 60 characters is the recommended
863 // maximum for best SERP visibility before Google truncates the title.
864 if ($title_length >= self::TITLE_OPTIMAL_MIN && $title_length <= self::TITLE_OPTIMAL_MAX) {
865 $score += 6;
866 } elseif ($title_length >= 25 && $title_length <= 75) {
867 $score += 4;
868 $suggestions[] = 'Optimize title length to 35-60 characters for better SERP visibility';
869 } else {
870 $score++;
871 $suggestions[] = $title_length < 25 ?
872 'Title too short - aim for 35-60 characters' :
873 'Title too long - risk truncation in search results';
874 }
875
876 // Looser keyword matching (6 points) - 2025 update
877 if (!empty($target_keyword)) {
878 $title_lower = strtolower($title);
879 $keyword_lower = strtolower($target_keyword);
880
881 // Exact match
882 if (strpos($title_lower, $keyword_lower) !== false) {
883 $score += 6;
884 } else {
885 // Check for semantic variations (2025 improvement)
886 $semantic_match = $this->check_semantic_keyword_match($title, $target_keyword);
887 if ($semantic_match) {
888 $score += 5; // Almost full credit for semantic match
889 $suggestions[] = 'Good semantic keyword usage - Google now recognizes keyword variations';
890 } else {
891 // Check for partial keyword match
892 $keyword_parts = explode(' ', $keyword_lower);
893 $partial_matches = 0;
894 foreach ($keyword_parts as $part) {
895 if (strpos($title_lower, $part) !== false) {
896 $partial_matches++;
897 }
898 }
899
900 if ($partial_matches > 0) {
901 $score += round(($partial_matches / count($keyword_parts)) * 4);
902 $suggestions[] = "Include more parts of target keyword '{$target_keyword}' in title";
903 } else {
904 $suggestions[] = "Include target keyword '{$target_keyword}' or related terms in title";
905 }
906 }
907 }
908 } else {
909 $score += 2; // Partial credit
910 $suggestions[] = 'Set a target keyword to optimize title effectiveness';
911 }
912
913 // Title readability (2 points) - mirrors Rank Math's title checks for a
914 // number/power word (drives CTR) and emotional sentiment.
915 $has_number = (bool) preg_match('/\d/', $title);
916 $has_power_word = $this->title_has_power_word($title);
917 $has_sentiment = $this->title_has_sentiment_word($title);
918
919 if ($has_number || $has_power_word) {
920 $score++;
921 } else {
922 $suggestions[] = 'Add a number or a power word to the title to boost click-through rate';
923 }
924 if ($has_sentiment) {
925 $score++;
926 } else {
927 $suggestions[] = 'Use an emotional/sentiment word in the title to make it more compelling';
928 }
929
930 return [
931 'score' => $score,
932 'max_score' => $max_score,
933 'suggestions' => $suggestions,
934 'details' => [
935 'title_length' => $title_length,
936 'optimal_range' => '35-60 characters',
937 'keyword_present' => !empty($target_keyword) && strpos(strtolower($title), strtolower($target_keyword)) !== false,
938 'semantic_match' => !empty($target_keyword) ? $this->check_semantic_keyword_match($title, $target_keyword) : false,
939 'has_number_or_power_word' => $has_number || $has_power_word,
940 'has_sentiment_word' => $has_sentiment,
941 ]
942 ];
943 }
944
945 // Placeholder methods for remaining 2025 factors
946
947 private function score_niche_expertise(array $content_data, string $target_keyword): array {
948 $max_score = $this->scoring_factors['niche_expertise'];
949 $suggestions = [];
950
951 // Without a focus keyword we cannot measure topical coverage; award
952 // partial credit rather than capping the ceiling with a placeholder.
953 if (empty($target_keyword)) {
954 return [
955 'score' => 7,
956 'max_score' => $max_score,
957 'suggestions' => ['Set a focus keyword and use it in subheadings and the URL for stronger topical signals'],
958 'details' => ['expertise_level' => 'Unmeasured (no focus keyword)'],
959 ];
960 }
961
962 $score = 0;
963 $content = (string) ($content_data['content'] ?? '');
964 $headings = (array) ($content_data['headings'] ?? []);
965 $images = (array) ($content_data['images'] ?? []);
966 $slug = strtolower((string) ($content_data['slug'] ?? ''));
967
968 // Keyword in a subheading (4 points).
969 if ($this->keyword_in_subheadings($headings, $target_keyword)) {
970 $score += 4;
971 } else {
972 $suggestions[] = "Include '{$target_keyword}' in at least one subheading (H2-H6)";
973 }
974
975 // Keyword density in a healthy band (4 points). Rank Math treats
976 // ~0.5%-2.5% as optimal; reward in-band, partial when present but thin.
977 $density = $this->keyword_density($content, $target_keyword);
978 if ($density >= 0.5 && $density <= 2.5) {
979 $score += 4;
980 } elseif ($density > 0) {
981 $score += 2;
982 $suggestions[] = $density > 2.5
983 ? 'Keyword density is high - reduce repetition to avoid over-optimization'
984 : 'Keyword density is low - use the focus keyword a little more often';
985 } else {
986 $suggestions[] = "Use the focus keyword '{$target_keyword}' in the content";
987 }
988
989 // URL optimization (3 points): keyword in slug (2) + a reasonably short
990 // URL (1). Rank Math flags overly long URLs, so reward concise slugs.
991 $keyword_slug = str_replace(' ', '-', strtolower($target_keyword));
992 $keyword_in_slug = $slug !== '' && (strpos($slug, $keyword_slug) !== false || strpos(str_replace('-', '', $slug), str_replace('-', '', $keyword_slug)) !== false);
993 if ($keyword_in_slug) {
994 $score += 2;
995 } else {
996 $suggestions[] = 'Include the focus keyword in the URL slug';
997 }
998
999 // A slug under ~75 chars keeps the URL clean and fully visible in SERPs.
1000 $slug_length = strlen($slug);
1001 if ($slug === '' || $slug_length <= 75) {
1002 $score++;
1003 } else {
1004 $suggestions[] = 'Shorten the URL slug - long URLs are harder to read and share';
1005 }
1006
1007 // Keyword in image alt text (2 points) - mirrors Rank Math's
1008 // "keyword in image alt" check. When the post has no images the check
1009 // does not apply, so award the points (benefit of the doubt) rather
1010 // than capping the ceiling for legitimately image-less posts.
1011 if (empty($images)) {
1012 $score += 2;
1013 $suggestions[] = 'Add a relevant image with the focus keyword in its alt text';
1014 } elseif ($this->keyword_in_alt($images, $target_keyword)) {
1015 $score += 2;
1016 } else {
1017 $suggestions[] = 'Include the focus keyword in at least one image alt attribute';
1018 }
1019
1020 return [
1021 'score' => $score,
1022 'max_score' => $max_score,
1023 'suggestions' => $suggestions,
1024 'details' => [
1025 'keyword_in_subheading' => $this->keyword_in_subheadings($headings, $target_keyword),
1026 'keyword_density' => round($density, 2),
1027 'keyword_in_slug' => $keyword_in_slug,
1028 'slug_length' => $slug_length,
1029 'keyword_in_image_alt' => $this->keyword_in_alt($images, $target_keyword),
1030 ],
1031 ];
1032 }
1033
1034 private function score_searcher_engagement(array $content_data): array {
1035 $max_score = $this->scoring_factors['searcher_engagement'];
1036
1037 // Engagement (dwell time / bounce) is off-page, so estimate it from the
1038 // on-page signals that drive it: readability (half) + content structure
1039 // (half). This raises the old readability-only floor.
1040 $readability = (float) ($content_data['readability_score'] ?? 50);
1041 $structure = (float) $this->derive_content_quality_score($content_data);
1042
1043 $readability_pts = ($readability / 100) * ($max_score / 2);
1044 $structure_pts = ($structure / 100) * ($max_score / 2);
1045 $score = (int) round($readability_pts + $structure_pts);
1046
1047 return [
1048 'score' => $score,
1049 'max_score' => $max_score,
1050 'suggestions' => $score < ($max_score * 0.7)
1051 ? ['Improve readability and structure (shorter sentences, subheadings, shorter paragraphs)']
1052 : [],
1053 'details' => ['engagement_estimate' => round(($score / $max_score) * 100, 1) . '%'],
1054 ];
1055 }
1056
1057 private function score_backlink_authority(array $content_data): array {
1058 $max_score = $this->scoring_factors['backlink_authority'];
1059 $external_links = (int) ($content_data['external_links'] ?? 0);
1060 $dofollow_links = (int) ($content_data['external_dofollow_links'] ?? 0);
1061
1062 // Backlinks are off-page and cannot be read from post content. We use
1063 // the only on-page proxy available - whether the content cites external
1064 // sources - and avoid hard-penalizing posts for something outside the
1065 // editor's control (the old external_links*3 formula needed 5 outbound
1066 // links just to reach full marks, dragging nearly every post down).
1067 // Dofollow links pass equity, so they earn full credit (Rank Math's
1068 // "external dofollow link" check); nofollow-only citations earn less.
1069 $suggestions = [];
1070 if ($dofollow_links >= 2) {
1071 $score = $max_score;
1072 } elseif ($dofollow_links === 1) {
1073 $score = (int) round($max_score * 0.85);
1074 } elseif ($external_links > 0) {
1075 // Cites sources but every external link is nofollow.
1076 $score = (int) round($max_score * 0.75);
1077 $suggestions[] = 'Add at least one dofollow link to an authoritative external source';
1078 } else {
1079 $score = (int) round($max_score * 0.6);
1080 $suggestions[] = 'Cite authoritative external sources, and build quality backlinks to this page';
1081 }
1082
1083 return [
1084 'score' => $score,
1085 'max_score' => $max_score,
1086 'suggestions' => $suggestions,
1087 'details' => [
1088 'external_links' => $external_links,
1089 'external_dofollow_links' => $dofollow_links,
1090 ],
1091 ];
1092 }
1093 private function score_trustworthiness(array $content_data, array $metadata): array {
1094 // E-E-A-T is an off-page/site-wide signal we cannot reliably measure
1095 // from a single post. Award full credit (benefit of the doubt) rather
1096 // than a fixed partial that silently caps every post's ceiling.
1097 return [
1098 'score' => $this->scoring_factors['trustworthiness'],
1099 'max_score' => $this->scoring_factors['trustworthiness'],
1100 'suggestions' => ['Add author credentials, citations, and contact information to reinforce trustworthiness'],
1101 'details' => ['trust_level' => 'Assumed adequate'],
1102 ];
1103 }
1104
1105 private function score_link_diversity(array $content_data): array {
1106 // Off-page link distribution; not measurable per post. Full credit.
1107 return [
1108 'score' => $this->scoring_factors['link_diversity'],
1109 'max_score' => $this->scoring_factors['link_diversity'],
1110 'suggestions' => [],
1111 'details' => ['diversity_level' => 'Assumed adequate'],
1112 ];
1113 }
1114 private function score_site_security(array $content_data): array {
1115 $max_score = $this->scoring_factors['site_security'];
1116 $ssl = function_exists('is_ssl') ? is_ssl() : true;
1117
1118 return [
1119 'score' => $ssl ? $max_score : 0,
1120 'max_score' => $max_score,
1121 'suggestions' => $ssl ? [] : ['Serve the site over HTTPS (install an SSL certificate)'],
1122 'details' => ['ssl_enabled' => $ssl, 'security_level' => $ssl ? 'Good' : 'Insecure'],
1123 ];
1124 }
1125 private function check_semantic_keyword_match(string $text, string $keyword): bool {
1126 // Simple semantic matching - can be enhanced with AI/NLP
1127 $keyword_parts = explode(' ', strtolower($keyword));
1128 $text_lower = strtolower($text);
1129
1130 $matches = 0;
1131 foreach ($keyword_parts as $part) {
1132 if (strpos($text_lower, $part) !== false) {
1133 $matches++;
1134 }
1135 }
1136
1137 // Consider it a semantic match if 70% of keyword parts are present
1138 return ($matches / count($keyword_parts)) >= 0.7;
1139 }
1140 private function prioritize_suggestions(array $suggestions, array $scores = []): array {
1141 // Map each suggestion back to the factor that emitted it, so priority
1142 // can rank by the points the factor actually lost instead of keyword-
1143 // matching the advice text — which sorted a 2-point title tweak above
1144 // a 6-point thin-content loss and contradicted the row's own impact
1145 // tag (#408).
1146 $by_text = [];
1147 foreach ($scores as $factor => $result) {
1148 if (!is_array($result) || empty($result['suggestions']) || !is_array($result['suggestions'])) {
1149 continue;
1150 }
1151 $lost = max(0, (float) ($result['max_score'] ?? 0) - (float) ($result['score'] ?? 0));
1152 foreach ($result['suggestions'] as $text) {
1153 if (is_string($text) && !isset($by_text[$text])) {
1154 $by_text[$text] = ['factor' => (string) $factor, 'lost' => $lost];
1155 }
1156 }
1157 }
1158
1159 $prioritized = [];
1160
1161 foreach ($suggestions as $suggestion) {
1162 $origin = $by_text[$suggestion] ?? null;
1163
1164 // A factor already at full marks loses nothing to this advice —
1165 // it was occupying list positions (sometimes at "High") while
1166 // recovering zero points. Dropped rather than sorted last.
1167 if (null !== $origin && $origin['lost'] <= 0) {
1168 continue;
1169 }
1170
1171 if (null !== $origin) {
1172 $priority = $origin['lost'] >= 4 ? 'High' : ($origin['lost'] >= 2 ? 'Medium' : 'Low');
1173 } else {
1174 // No factor attached (defensive: a filter-added or legacy
1175 // suggestion) — the old keyword map is the fallback.
1176 $priority = $this->determine_suggestion_priority($suggestion);
1177 }
1178
1179 $prioritized[] = [
1180 'text' => $suggestion,
1181 'priority' => $priority,
1182 'impact' => $this->estimate_impact($suggestion),
1183 'effort' => $this->estimate_effort($suggestion),
1184 'factor' => $origin['factor'] ?? null,
1185 'points_recoverable' => $origin['lost'] ?? null,
1186 ];
1187 }
1188
1189 // Biggest recoverable loss first; keyword-mapped stragglers (no
1190 // factor) sort within their priority band after the measured rows.
1191 usort($prioritized, function($a, $b) {
1192 $al = $a['points_recoverable'] ?? -1;
1193 $bl = $b['points_recoverable'] ?? -1;
1194 if ($al !== $bl) {
1195 return $bl <=> $al;
1196 }
1197 $priority_order = ['High' => 3, 'Medium' => 2, 'Low' => 1];
1198 return $priority_order[$b['priority']] - $priority_order[$a['priority']];
1199 });
1200
1201 return $prioritized;
1202 }
1203
1204 /**
1205 * Determine suggestion priority based on content
1206 *
1207 * @param string $suggestion Suggestion text
1208 * @return string Priority level
1209 */
1210 private function determine_suggestion_priority(string $suggestion): string {
1211 $high_priority_keywords = ['title', 'keyword', 'content quality', 'heading'];
1212 $medium_priority_keywords = ['meta description', 'internal link', 'readability'];
1213
1214 $suggestion_lower = strtolower($suggestion);
1215
1216 foreach ($high_priority_keywords as $keyword) {
1217 if (strpos($suggestion_lower, $keyword) !== false) {
1218 return 'High';
1219 }
1220 }
1221
1222 foreach ($medium_priority_keywords as $keyword) {
1223 if (strpos($suggestion_lower, $keyword) !== false) {
1224 return 'Medium';
1225 }
1226 }
1227
1228 return 'Low';
1229 }
1230
1231 /**
1232 * Estimate impact of implementing suggestion
1233 *
1234 * @param string $suggestion Suggestion text
1235 * @return string Impact level
1236 */
1237 private function estimate_impact(string $suggestion): string {
1238 // Simple heuristic - can be enhanced with ML
1239 if (strpos(strtolower($suggestion), 'title') !== false) { return 'High';
1240 }
1241 if (strpos(strtolower($suggestion), 'content') !== false) { return 'High';
1242 }
1243 if (strpos(strtolower($suggestion), 'keyword') !== false) { return 'Medium';
1244 }
1245 return 'Low';
1246 }
1247
1248 /**
1249 * Estimate effort required to implement suggestion
1250 *
1251 * @param string $suggestion Suggestion text
1252 * @return string Effort level
1253 */
1254 private function estimate_effort(string $suggestion): string {
1255 // Simple heuristic - can be enhanced with ML
1256 if (strpos(strtolower($suggestion), 'rewrite') !== false) { return 'High';
1257 }
1258 if (strpos(strtolower($suggestion), 'add') !== false) { return 'Medium';
1259 }
1260 if (strpos(strtolower($suggestion), 'optimize') !== false) { return 'Medium';
1261 }
1262 return 'Low';
1263 }
1264 private function calculate_topic_relevance(string $content, string $target_keyword): float {
1265 if (empty($content) || empty($target_keyword)) {
1266 return 0.0;
1267 }
1268
1269 $content_lower = strtolower(wp_strip_all_tags($content));
1270 $keyword_lower = strtolower($target_keyword);
1271
1272 // Calculate keyword and semantic term frequency
1273 $keyword_count = substr_count($content_lower, $keyword_lower);
1274 $word_count = $this->calculate_word_count_js_style($content_lower);
1275
1276 if ($word_count === 0) {
1277 return 0.0;
1278 }
1279
1280 // Base relevance from keyword presence
1281 $keyword_density = ($keyword_count / $word_count) * 100;
1282 $base_relevance = min(1.0, $keyword_density / 2.0); // Optimal around 1-2%
1283
1284 // Boost for semantic variations
1285 $semantic_boost = $this->calculate_semantic_boost($content_lower, $keyword_lower);
1286
1287 return min(1.0, $base_relevance + $semantic_boost);
1288 }
1289
1290 /**
1291 * Calculate semantic boost for related terms
1292 *
1293 * @param string $content Content text (lowercase)
1294 * @param string $keyword Target keyword (lowercase)
1295 * @return float Semantic boost (0-0.3)
1296 */
1297 private function calculate_semantic_boost(string $content, string $keyword): float {
1298 // Simple semantic term detection - can be enhanced with NLP
1299 $semantic_terms = $this->get_semantic_terms($keyword);
1300 $boost = 0.0;
1301
1302 foreach ($semantic_terms as $term) {
1303 if (strpos($content, $term) !== false) {
1304 $boost += 0.05; // Small boost per semantic term
1305 }
1306 }
1307
1308 return min(0.3, $boost); // Cap at 30% boost
1309 }
1310
1311 /**
1312 * Get semantic terms for a keyword
1313 *
1314 * @param string $keyword Target keyword
1315 * @return array Semantic terms
1316 */
1317 private function get_semantic_terms(string $keyword): array {
1318 // Simple semantic term generation - can be enhanced with AI/NLP
1319 $terms = [];
1320
1321 // Add plural/singular variations
1322 if (substr($keyword, -1) === 's') {
1323 $terms[] = rtrim($keyword, 's');
1324 } else {
1325 $terms[] = $keyword . 's';
1326 }
1327
1328 // Add common related terms based on keyword
1329 $keyword_lower = strtolower($keyword);
1330
1331 // SEO-related terms
1332 if (strpos($keyword_lower, 'seo') !== false) {
1333 $terms = array_merge($terms, ['optimization', 'search engine', 'ranking', 'visibility']);
1334 }
1335
1336 // WordPress-related terms
1337 if (strpos($keyword_lower, 'wordpress') !== false) { // phpcs:ignore WordPress.WP.CapitalPDangit.MisspelledInText -- lowercase on purpose: the haystack is strtolower()ed.
1338 $terms = array_merge($terms, ['wp', 'plugin', 'theme', 'cms']);
1339 }
1340
1341 return $terms;
1342 }
1343
1344 /**
1345 * Keyword density (%) of the target keyword across the plain-text body.
1346 *
1347 * @param string $content Raw/HTML content.
1348 * @param string $target_keyword Target keyword.
1349 * @return float Density percentage (0 when no keyword/content).
1350 */
1351 private function keyword_density(string $content, string $target_keyword): float {
1352 if (empty($content) || empty($target_keyword)) {
1353 return 0.0;
1354 }
1355 $plain = strtolower(wp_strip_all_tags($content));
1356 $word_count = $this->calculate_word_count_js_style($plain);
1357 if ($word_count === 0) {
1358 return 0.0;
1359 }
1360 $occurrences = substr_count($plain, strtolower($target_keyword));
1361 return ($occurrences / $word_count) * 100;
1362 }
1363
1364 /**
1365 * Whether the target keyword (or a semantic variation) appears in the body.
1366 *
1367 * @param string $content Raw/HTML content.
1368 * @param string $target_keyword Target keyword.
1369 * @return bool
1370 */
1371 private function keyword_in_content(string $content, string $target_keyword): bool {
1372 if (empty($content) || empty($target_keyword)) {
1373 return false;
1374 }
1375 $plain = strtolower(wp_strip_all_tags($content));
1376 if (strpos($plain, strtolower($target_keyword)) !== false) {
1377 return true;
1378 }
1379 return $this->check_semantic_keyword_match($plain, $target_keyword);
1380 }
1381
1382 /**
1383 * Whether the keyword appears early (first paragraph / first ~10% of words).
1384 *
1385 * @param string $content Raw/HTML content.
1386 * @param string $target_keyword Target keyword.
1387 * @return bool
1388 */
1389 private function keyword_in_first_paragraph(string $content, string $target_keyword): bool {
1390 if (empty($content) || empty($target_keyword)) {
1391 return false;
1392 }
1393 $plain = strtolower(wp_strip_all_tags($content));
1394 $words = preg_split('/\s+/', trim($plain), -1, PREG_SPLIT_NO_EMPTY) ?: [];
1395 $window = array_slice($words, 0, max(50, (int) ceil(count($words) * 0.1)));
1396 return strpos(implode(' ', $window), strtolower($target_keyword)) !== false;
1397 }
1398
1399 /**
1400 * Whether the keyword appears in any subheading (H2–H6).
1401 *
1402 * @param array $headings Extracted headings (each with 'level' + 'text').
1403 * @param string $target_keyword Target keyword.
1404 * @return bool
1405 */
1406 private function keyword_in_subheadings(array $headings, string $target_keyword): bool {
1407 if (empty($target_keyword)) {
1408 return false;
1409 }
1410 $keyword_lower = strtolower($target_keyword);
1411 foreach ($headings as $heading) {
1412 if ((int) ($heading['level'] ?? 0) < 2) {
1413 continue;
1414 }
1415 $text = strtolower((string) ($heading['text'] ?? ''));
1416 if ($text !== '' && strpos($text, $keyword_lower) !== false) {
1417 return true;
1418 }
1419 }
1420 return false;
1421 }
1422
1423 /**
1424 * Days elapsed since a MySQL datetime string, or null when unparseable.
1425 *
1426 * @param string $datetime MySQL datetime (e.g. post_modified).
1427 * @return int|null
1428 */
1429 private function days_since(string $datetime): ?int {
1430 $datetime = trim($datetime);
1431 if ($datetime === '' || strpos($datetime, '0000-00-00') === 0) {
1432 return null;
1433 }
1434 $ts = strtotime($datetime);
1435 if ($ts === false) {
1436 return null;
1437 }
1438 // strtotime() returns a real unix timestamp, so this must compare against
1439 // one: current_time('timestamp') is offset by the site timezone and made
1440 // every "days ago" figure wrong by that offset.
1441 $now = time();
1442 return (int) floor(($now - $ts) / 86400);
1443 }
1444
1445 /**
1446 * Whether the keyword appears in the meta description.
1447 *
1448 * @param string $meta_description Meta description text.
1449 * @param string $target_keyword Target keyword.
1450 * @return bool
1451 */
1452 private function keyword_in_meta(string $meta_description, string $target_keyword): bool {
1453 if ($meta_description === '' || $target_keyword === '') {
1454 return false;
1455 }
1456 return strpos(strtolower($meta_description), strtolower($target_keyword)) !== false;
1457 }
1458
1459 /**
1460 * Whether the keyword appears in any image alt text.
1461 *
1462 * @param array $images Images (each with an 'alt' key).
1463 * @param string $target_keyword Target keyword.
1464 * @return bool
1465 */
1466 private function keyword_in_alt(array $images, string $target_keyword): bool {
1467 if ($target_keyword === '') {
1468 return false;
1469 }
1470 $keyword_lower = strtolower($target_keyword);
1471 foreach ($images as $image) {
1472 $alt = strtolower((string) ($image['alt'] ?? ''));
1473 if ($alt !== '' && strpos($alt, $keyword_lower) !== false) {
1474 return true;
1475 }
1476 }
1477 return false;
1478 }
1479
1480 /**
1481 * Whether the title contains a common power word (CTR booster).
1482 *
1483 * @param string $title Post title.
1484 * @return bool
1485 */
1486 private function title_has_power_word(string $title): bool {
1487 $title_lower = strtolower($title);
1488 foreach (self::get_title_power_words() as $word) {
1489 if (strpos($title_lower, $word) !== false) {
1490 return true;
1491 }
1492 }
1493 return false;
1494 }
1495
1496 /**
1497 * The power words rewarded by the title check. Single source of truth so the
1498 * AI title improver can require the generated title to actually contain one.
1499 *
1500 * @return string[]
1501 */
1502 public static function get_title_power_words(): array {
1503 return [
1504 'ultimate', 'essential', 'complete', 'proven', 'guide', 'best', 'top',
1505 'free', 'easy', 'simple', 'quick', 'fast', 'powerful', 'secret', 'expert',
1506 'effective', 'amazing', 'incredible', 'exclusive', 'definitive', 'step-by-step',
1507 ];
1508 }
1509
1510 /**
1511 * The emotion/sentiment words rewarded by the title check. Single source of
1512 * truth shared with the AI title improver so a generated title satisfies the
1513 * same validation.
1514 *
1515 * @return string[]
1516 */
1517 public static function get_title_sentiment_words(): array {
1518 return [
1519 // Positive
1520 'great', 'good', 'better', 'awesome', 'love', 'win', 'boost', 'improve',
1521 'success', 'smart', 'brilliant', 'perfect', 'happy', 'beautiful',
1522 // Negative (drives clicks too)
1523 'avoid', 'mistake', 'worst', 'stop', 'never', 'bad', 'wrong', 'fail',
1524 'danger', 'warning', 'painful', 'ugly',
1525 ];
1526 }
1527
1528 /**
1529 * Whether the title carries an emotional/sentiment word (positive or
1530 * negative), which Rank Math rewards for higher engagement.
1531 *
1532 * @param string $title Post title.
1533 * @return bool
1534 */
1535 private function title_has_sentiment_word(string $title): bool {
1536 $title_lower = strtolower($title);
1537 foreach (self::get_title_sentiment_words() as $word) {
1538 if (strpos($title_lower, $word) !== false) {
1539 return true;
1540 }
1541 }
1542 return false;
1543 }
1544
1545 /**
1546 * Assess content depth based on word count
1547 *
1548 * @param int $word_count Word count
1549 * @return string Depth assessment
1550 */
1551 private function assess_content_depth(int $word_count): string {
1552 if ($word_count >= 2000) { return 'Comprehensive';
1553 }
1554 if ($word_count >= 1000) { return 'Detailed';
1555 }
1556 if ($word_count >= 500) { return 'Moderate';
1557 }
1558 if ($word_count >= 300) { return 'Basic';
1559 }
1560 return 'Insufficient';
1561 }
1562 private function score_content_freshness(array $content_data): array {
1563 $max_score = $this->scoring_factors['content_freshness'];
1564 $suggestions = [];
1565
1566 // Derive freshness from the real last-modified date when available
1567 // (live editing has no stored date yet -> treat as fresh).
1568 $days = $this->days_since((string) ($content_data['post_modified'] ?? ''));
1569
1570 if ($days === null || $days <= 180) {
1571 $score = $max_score;
1572 $status = 'Current';
1573 } elseif ($days <= 365) {
1574 $score = (int) round($max_score * 0.66);
1575 $status = 'Aging';
1576 $suggestions[] = 'Content is 6-12 months old - review and refresh it for better freshness signals';
1577 } else {
1578 $score = (int) round($max_score * 0.33);
1579 $status = 'Stale';
1580 $suggestions[] = 'Content is over a year old - update it to maintain freshness signals';
1581 }
1582
1583 return [
1584 'score' => $score,
1585 'max_score' => $max_score,
1586 'suggestions' => $suggestions,
1587 'details' => [
1588 'freshness_status' => $status,
1589 'days_since_modified' => $days,
1590 ],
1591 ];
1592 }
1593 /**
1594 * Resolve a post's content into something worth analyzing.
1595 *
1596 * Delegates to Builder_Content, which knows where each page builder keeps
1597 * its text. Kept as the historical entry point for existing callers.
1598 *
1599 * @since 1.23.0
1600 *
1601 * @param \WP_Post $post Post being analyzed.
1602 * @return string Content to analyze.
1603 */
1604 public static function resolve_analyzable_content(\WP_Post $post): string {
1605 if (!class_exists('\ThinkRank\SEO\Builder_Content')) {
1606 require_once THINKRANK_PLUGIN_DIR . 'includes/seo/class-builder-content.php';
1607 }
1608
1609 return \ThinkRank\SEO\Builder_Content::resolve($post);
1610 }
1611
1612 /**
1613 * Resolve editor-supplied live content into something worth analyzing.
1614 *
1615 * @since 1.23.0
1616 *
1617 * @param string $live_content Markup supplied by the editor.
1618 * @param \WP_Post $post Post the markup belongs to.
1619 * @return string Content to analyze.
1620 */
1621 public static function resolve_live_content(string $live_content, \WP_Post $post): string {
1622 if (!class_exists('\ThinkRank\SEO\Builder_Content')) {
1623 require_once THINKRANK_PLUGIN_DIR . 'includes/seo/class-builder-content.php';
1624 }
1625
1626 // Bind the live markup to the post the render makes current. Since #862
1627 // the resolver runs setup_postdata() on this post, so a shortcode that
1628 // builds its output from the current post's content — get_the_content(),
1629 // get_post()->post_content, as a table of contents or a reading-time
1630 // shortcode does — otherwise read the last saved body while the unsaved
1631 // markup rendered around it, and lagged a save behind (#864).
1632 //
1633 // A clone, not the caller's object: the post is current only for the
1634 // duration of the render and the caller's $post must come back
1635 // unchanged. `thinkrank_analyzable_content` receives the clone too,
1636 // which is what makes $post->post_content there agree with the markup
1637 // being analyzed on the live path.
1638 $bound = clone $post;
1639 $bound->post_content = $live_content;
1640
1641 return \ThinkRank\SEO\Builder_Content::resolve_markup($live_content, $bound);
1642 }
1643
1644 public function analyze_post_content(int $post_id): array {
1645 $post = get_post($post_id);
1646 if (!$post) {
1647 return [];
1648 }
1649
1650 $content = self::resolve_analyzable_content($post);
1651 $title = $post->post_title;
1652
1653 // Extract headings from content
1654 $headings = $this->extract_headings($content);
1655
1656 // Count words using JavaScript-compatible method
1657 $plain_text = wp_strip_all_tags($content);
1658 $word_count = $this->calculate_word_count_js_style($plain_text);
1659
1660 // Calculate readability
1661 $readability_score = $this->calculate_readability_score($content);
1662
1663 // Count links
1664 $internal_links = $this->count_internal_links($content);
1665 $external_links = $this->count_external_links($content);
1666 $external_dofollow_links = $this->count_external_dofollow_links($content);
1667
1668 // Analyze images
1669 $images = $this->analyze_images($content);
1670
1671 // Get URL
1672 $url = get_permalink($post_id);
1673
1674 return [
1675 'content' => $content,
1676 'title' => $title,
1677 'headings' => $headings,
1678 'word_count' => $word_count,
1679 'readability_score' => $readability_score,
1680 'internal_links' => $internal_links,
1681 'external_links' => $external_links,
1682 'external_dofollow_links' => $external_dofollow_links,
1683 'images' => $images,
1684 'url' => $url,
1685 'slug' => $post->post_name,
1686 'post_modified' => $post->post_modified,
1687 'schema_present' => $this->detect_schema_present($content)
1688 || $this->thinkrank_global_schema_active($post->post_type)
1689 || $this->thinkrank_deployed_schema_active($post),
1690 ];
1691 }
1692
1693 /**
1694 * Build the human-readable readability label (mirrors the editor's
1695 * calculateReadabilityScore: "<level> (<flesch>)") from analyzed content.
1696 *
1697 * @param array $content_data Output of analyze_post_content()/analyze_live_content()
1698 * @return string Readability label, e.g. "Standard (62)"
1699 */
1700 private function format_readability_label(array $content_data): string {
1701 if ((int) ($content_data['word_count'] ?? 0) === 0) {
1702 return 'No content';
1703 }
1704
1705 $rounded = (int) round((float) ($content_data['readability_score'] ?? 0));
1706
1707 if ($rounded >= 90) {
1708 $level = 'Very Easy';
1709 } elseif ($rounded >= 80) {
1710 $level = 'Easy';
1711 } elseif ($rounded >= 70) {
1712 $level = 'Fairly Easy';
1713 } elseif ($rounded >= 60) {
1714 $level = 'Standard';
1715 } elseif ($rounded >= 50) {
1716 $level = 'Fairly Difficult';
1717 } elseif ($rounded >= 30) {
1718 $level = 'Difficult';
1719 } else {
1720 $level = 'Very Difficult';
1721 }
1722
1723 return "{$level} ({$rounded})";
1724 }
1725
1726 /**
1727 * Derive the content-quality label (mirrors the editor's
1728 * calculateContentQuality: word count + long-paragraph + subheading scoring)
1729 * so persisted scores carry a non-null quality value.
1730 *
1731 * @param array $content_data Output of analyze_post_content()/analyze_live_content()
1732 * @return string One of: No content, Good, OK, Needs improvement
1733 */
1734 private function derive_content_quality(array $content_data): string {
1735 $word_count = (int) ($content_data['word_count'] ?? 0);
1736 if ($word_count === 0) {
1737 return 'No content';
1738 }
1739
1740 $final = $this->derive_content_quality_score($content_data);
1741
1742 if ($final >= 80) {
1743 return 'Good';
1744 }
1745 if ($final >= 50) {
1746 return 'OK';
1747 }
1748
1749 return 'Needs improvement';
1750 }
1751
1752 /**
1753 * Numeric content-quality score (0-100): word count + long-paragraph +
1754 * subheading distribution. Shared by the quality label and the
1755 * satisfying-content factor so both stay in sync.
1756 *
1757 * @param array $content_data Output of analyze_post_content()/analyze_live_content()
1758 * @return int Quality score 0-100.
1759 */
1760 private function derive_content_quality_score(array $content_data): int {
1761 $word_count = (int) ($content_data['word_count'] ?? 0);
1762 if ($word_count === 0) {
1763 return 0;
1764 }
1765
1766 $content = (string) ($content_data['content'] ?? '');
1767 $score = 0;
1768
1769 // 1. Word count (industry standard: 300+ words).
1770 if ($word_count >= 600) {
1771 $score += 100;
1772 } elseif ($word_count >= 300) {
1773 $score += 50;
1774 }
1775
1776 // 2. Long paragraphs (flag paragraphs over 150 words).
1777 $long_paragraphs = 0;
1778 if (preg_match_all('/<p[^>]*>(.*?)<\/p>/is', $content, $matches)) {
1779 foreach ($matches[1] as $paragraph) {
1780 if ($this->calculate_word_count_js_style(wp_strip_all_tags($paragraph)) > 150) {
1781 $long_paragraphs++;
1782 }
1783 }
1784 }
1785 if ($long_paragraphs === 0) {
1786 $score += 100;
1787 } elseif ($long_paragraphs <= 2) {
1788 $score += 50;
1789 }
1790
1791 // 3. Subheading distribution (H2–H6, ~one per 300 words).
1792 $subheadings = 0;
1793 foreach ((array) ($content_data['headings'] ?? []) as $heading) {
1794 if ((int) ($heading['level'] ?? 0) >= 2) {
1795 $subheadings++;
1796 }
1797 }
1798 $expected = (int) floor($word_count / 300);
1799 if ($subheadings > 0 && $subheadings >= $expected) {
1800 $score += 100;
1801 } elseif ($subheadings > 0) {
1802 $score += 50;
1803 }
1804
1805 return (int) round($score / 3);
1806 }
1807
1808 /**
1809 * Extract headings from content
1810 *
1811 * @param string $content Content HTML
1812 * @return array Array of headings with levels
1813 */
1814 private function extract_headings(string $content): array {
1815 $headings = [];
1816
1817 // Match H1-H6 tags
1818 if (preg_match_all('/<h([1-6])[^>]*>(.*?)<\/h[1-6]>/i', $content, $matches, PREG_SET_ORDER)) {
1819 foreach ($matches as $match) {
1820 $headings[] = [
1821 'level' => (int)$match[1],
1822 'text' => wp_strip_all_tags($match[2]),
1823 ];
1824 }
1825 }
1826
1827 return $headings;
1828 }
1829
1830 /**
1831 * Calculate readability score using Flesch Reading Ease
1832 *
1833 * @param string $content Content text
1834 * @return float Readability score
1835 */
1836 private function calculate_readability_score(string $content): float {
1837 $text = wp_strip_all_tags($content);
1838
1839 if (empty($text)) {
1840 return 0;
1841 }
1842
1843 // Count sentences (approximate)
1844 $sentences = preg_split('/[.!?]+/', $text, -1, PREG_SPLIT_NO_EMPTY);
1845 $sentence_count = count($sentences);
1846
1847 // Count words
1848 $word_count = $this->calculate_word_count_js_style(wp_strip_all_tags($text));
1849
1850 // Count syllables (approximate)
1851 $syllable_count = $this->count_syllables($text);
1852
1853 if ($sentence_count === 0 || $word_count === 0) {
1854 return 0;
1855 }
1856
1857 // Flesch Reading Ease formula
1858 $score = 206.835 - (1.015 * ($word_count / $sentence_count)) - (84.6 * ($syllable_count / $word_count));
1859
1860 return max(0, min(100, $score));
1861 }
1862
1863 /**
1864 * Count syllables in text (approximate)
1865 *
1866 * @param string $text Text to analyze
1867 * @return int Syllable count
1868 */
1869 private function count_syllables(string $text): int {
1870 $words = preg_split('/\s+/', trim(strtolower(wp_strip_all_tags($text))), -1, PREG_SPLIT_NO_EMPTY);
1871 $syllables = 0;
1872
1873 foreach ($words as $word) {
1874 $word = preg_replace('/[^a-z]/', '', $word);
1875 if ($word === '') {
1876 continue;
1877 }
1878
1879 $groups = preg_match_all('/[aeiouy]+/', $word);
1880
1881 // Standard Flesch heuristic: a trailing silent e does not form a
1882 // syllable ("make", "time", "these") — but only when a consonant
1883 // precedes it (a vowel+e ending like "movie" already shares its
1884 // group) and never for consonant-le ("table"), which does count.
1885 // Without this the counter inflated syllables/word by ~0.2-0.3 on
1886 // ordinary prose, driving raw Flesch negative and the UI to a
1887 // clamped "Very Difficult (0)" (#407).
1888 if ($groups > 1 && preg_match('/[^aeiouy]e$/', $word) && !str_ends_with($word, 'le')) {
1889 $groups--;
1890 }
1891
1892 $syllables += max(1, $groups);
1893 }
1894
1895 return $syllables;
1896 }
1897
1898 /**
1899 * Count internal links in content
1900 *
1901 * @param string $content Content HTML
1902 * @return int Internal link count
1903 */
1904 private function count_internal_links(string $content): int {
1905 $site_url = get_site_url();
1906 $count = 0;
1907
1908 if (preg_match_all('/<a[^>]+href=["\']([^"\']+)["\'][^>]*>/i', $content, $matches)) {
1909 foreach ($matches[1] as $url) {
1910 if (strpos($url, $site_url) !== false || strpos($url, '/') === 0) {
1911 $count++;
1912 }
1913 }
1914 }
1915
1916 return $count;
1917 }
1918
1919 /**
1920 * Count external links in content
1921 *
1922 * @param string $content Content HTML
1923 * @return int External link count
1924 */
1925 private function count_external_links(string $content): int {
1926 $site_url = get_site_url();
1927 $count = 0;
1928
1929 if (preg_match_all('/<a[^>]+href=["\']([^"\']+)["\'][^>]*>/i', $content, $matches)) {
1930 foreach ($matches[1] as $url) {
1931 if (strpos($url, 'http') === 0 && strpos($url, $site_url) === false) {
1932 $count++;
1933 }
1934 }
1935 }
1936
1937 return $count;
1938 }
1939
1940 /**
1941 * Count external links that pass link equity (not rel="nofollow").
1942 * Mirrors Rank Math's "external dofollow link" check.
1943 *
1944 * @param string $content Content HTML
1945 * @return int External dofollow link count
1946 */
1947 private function count_external_dofollow_links(string $content): int {
1948 $site_url = get_site_url();
1949 $count = 0;
1950
1951 if (preg_match_all('/<a\b[^>]*>/i', $content, $matches)) {
1952 foreach ($matches[0] as $tag) {
1953 if (!preg_match('/href=["\']([^"\']+)["\']/i', $tag, $href)) {
1954 continue;
1955 }
1956 $url = $href[1];
1957 $is_external = strpos($url, 'http') === 0 && strpos($url, $site_url) === false;
1958 if (!$is_external) {
1959 continue;
1960 }
1961 if (preg_match('/rel=["\'][^"\']*\bnofollow\b[^"\']*["\']/i', $tag)) {
1962 continue;
1963 }
1964 $count++;
1965 }
1966 }
1967
1968 return $count;
1969 }
1970
1971 /**
1972 * Analyze images in content
1973 *
1974 * @param string $content Content HTML
1975 * @return array Image analysis data
1976 */
1977 /**
1978 * Detect structured data embedded directly in the content (JSON-LD script
1979 * blocks or microdata attributes). Site-wide schema injected at render time
1980 * is a separate feature and intentionally out of scope here.
1981 *
1982 * @param string $content Raw/HTML content.
1983 * @return bool
1984 */
1985 private function detect_schema_present(string $content): bool {
1986 if ($content === '') {
1987 return false;
1988 }
1989 return stripos($content, 'application/ld+json') !== false
1990 || stripos($content, 'itemscope') !== false
1991 || stripos($content, 'itemtype') !== false;
1992 }
1993
1994 /**
1995 * Whether ThinkRank's Global SEO schema output is active for a post type.
1996 *
1997 * ThinkRank injects JSON-LD at render time (wp_head) when a schema type is
1998 * configured for the post type, so a post can have valid structured data
1999 * even when none is embedded in the post body. The score credits this so the
2000 * "add structured data" suggestion reflects ThinkRank's own schema engine.
2001 *
2002 * @param string $post_type Post type slug.
2003 * @return bool
2004 */
2005 private function thinkrank_global_schema_active(string $post_type): bool {
2006 if ($post_type === '') {
2007 return false;
2008 }
2009 $all_settings = get_option('thinkrank_global_seo_settings', []);
2010 return !empty($all_settings[$post_type]['schema_type']);
2011 }
2012
2013 /**
2014 * Whether the Schema Manager has an active deployed schema for this post.
2015 *
2016 * Per-post schema deployed from the editor's Schema tab is stored in the
2017 * Schema Manager's own table and emitted at wp_head by
2018 * Frontend\SEO_Manager::output_site_schema_markup(). Neither
2019 * detect_schema_present() (body scan) nor thinkrank_global_schema_active()
2020 * (post-type option) sees it, so without this the score reported "no
2021 * structured data" for posts that do emit it.
2022 *
2023 * Mirrors the context_type whitelist the emitter and the metabox both use, so
2024 * the lookup targets the same row the front end reads.
2025 *
2026 * @param \WP_Post $post Post being scored.
2027 * @return bool
2028 */
2029 private function thinkrank_deployed_schema_active(\WP_Post $post): bool {
2030 if (!class_exists('ThinkRank\\SEO\\Schema_Management_System')) {
2031 $manager_file = THINKRANK_PLUGIN_DIR . 'includes/seo/class-schema-management-system.php';
2032 if (!file_exists($manager_file)) {
2033 return false;
2034 }
2035 require_once $manager_file;
2036 }
2037
2038 $context_type = in_array($post->post_type, ['site', 'post', 'page', 'product'], true)
2039 ? $post->post_type
2040 : 'post';
2041
2042 try {
2043 $manager = new \ThinkRank\SEO\Schema_Management_System();
2044 return !empty($manager->get_deployed_schemas($context_type, (int) $post->ID));
2045 } catch (\Throwable $e) {
2046 return false;
2047 }
2048 }
2049
2050 private function analyze_images(string $content): array {
2051 $images = [];
2052
2053 if (preg_match_all('/<img[^>]+>/i', $content, $matches)) {
2054 foreach ($matches[0] as $img_tag) {
2055 $alt = '';
2056 if (preg_match('/alt=["\']([^"\']*)["\']/', $img_tag, $alt_match)) {
2057 $alt = $alt_match[1];
2058 }
2059
2060 $src = '';
2061 if (preg_match('/src=["\']([^"\']*)["\']/', $img_tag, $src_match)) {
2062 $src = $src_match[1];
2063 }
2064
2065 $images[] = [
2066 'src' => $src,
2067 'alt' => $alt,
2068 ];
2069 }
2070 }
2071
2072 return $images;
2073 }
2074
2075 /**
2076 * Save SEO score to database
2077 *
2078 * @param int $post_id Post ID
2079 * @param int $user_id User ID
2080 * @param array $score_data Score data
2081 * @return int|false Score ID or false on failure
2082 */
2083 public function save_score(int $post_id, int $user_id, array $score_data) {
2084 global $wpdb;
2085
2086 $table_name = $wpdb->prefix . 'thinkrank_seo_scores';
2087
2088 // Prepare data for insertion
2089 $insert_data = [
2090 'post_id' => $post_id,
2091 'user_id' => $user_id,
2092 'overall_score' => $score_data['overall_score'],
2093 'score_breakdown' => wp_json_encode($score_data['score_breakdown']),
2094 'suggestions' => wp_json_encode($score_data['suggestions']),
2095 'grade' => $score_data['grade'],
2096 'algorithm_version' => $score_data['algorithm_version'] ?? '2024.1',
2097 'calculated_at' => $score_data['calculated_at'],
2098 'created_at' => current_time('mysql'),
2099 ];
2100
2101 $format = [
2102 '%d', '%d', '%d', '%s', '%s', '%s', '%s', '%s', '%s'
2103 ];
2104
2105 // Add readability_score if provided
2106 if (isset($score_data['readability_score'])) {
2107 $insert_data['readability_score'] = $score_data['readability_score'];
2108 $format[] = '%s';
2109 }
2110
2111 // Add content_quality if provided
2112 if (isset($score_data['content_quality'])) {
2113 $insert_data['content_quality'] = $score_data['content_quality'];
2114 $format[] = '%s';
2115 }
2116
2117 // phpcs:ignore WordPress.DB.DirectDatabaseQuery.DirectQuery,WordPress.DB.DirectDatabaseQuery.NoCaching, PluginCheck.Security.DirectDB.UnescapedDBParameter -- SEO score storage requires direct database access
2118 $result = $wpdb->insert(
2119 $table_name,
2120 $insert_data,
2121 $format
2122 );
2123
2124 return $result ? $wpdb->insert_id : false;
2125 }
2126
2127 /**
2128 * Get score history for a post
2129 *
2130 * @param int $post_id Post ID
2131 * @param int $limit Number of scores to retrieve
2132 * @return array Score history
2133 */
2134 public function get_score_history(int $post_id, int $limit = 10): array {
2135 global $wpdb;
2136
2137 // Get table name and escape it properly (table names cannot be parameterized)
2138 $table_name = esc_sql($wpdb->prefix . 'thinkrank_seo_scores');
2139
2140 // phpcs:ignore WordPress.DB.DirectDatabaseQuery.DirectQuery,WordPress.DB.DirectDatabaseQuery.NoCaching, PluginCheck.Security.DirectDB.UnescapedDBParameter -- SEO score history requires direct database access
2141 $results = $wpdb->get_results(
2142 $wpdb->prepare(
2143 // phpcs:ignore WordPress.DB.PreparedSQL.InterpolatedNotPrepared, PluginCheck.Security.DirectDB.UnescapedDBParameter -- Table name is properly escaped using esc_sql()
2144 "SELECT * FROM `{$table_name}` WHERE post_id = %d ORDER BY created_at DESC LIMIT %d",
2145 $post_id,
2146 $limit
2147 ),
2148 ARRAY_A
2149 );
2150
2151 // Decode JSON fields
2152 foreach ($results as &$result) {
2153 $result['score_breakdown'] = json_decode($result['score_breakdown'], true);
2154 $result['suggestions'] = json_decode($result['suggestions'], true);
2155 }
2156
2157 return $results ?: [];
2158 }
2159
2160 /**
2161 * Get latest score for a post
2162 *
2163 * @param int $post_id Post ID
2164 * @return array|null Latest score data
2165 */
2166 public function get_latest_score(int $post_id): ?array {
2167 $history = $this->get_score_history($post_id, 1);
2168 return !empty($history) ? $history[0] : null;
2169 }
2170
2171 /**
2172 * Score mobile experience - NEW 2025 factor (5 points)
2173 *
2174 * @param array $content_data Content analysis data
2175 * @return array Scoring result
2176 */
2177 private function score_mobile_experience(array $content_data): array {
2178 $max = $this->scoring_factors['mobile_experience'];
2179
2180 // Mobile experience is theme/site-level, not controlled by post content
2181 // — but the plugin already measures it. When a mobile Lighthouse score
2182 // has been collected, score against it; the "benefit of the doubt" below
2183 // is for sites nobody has measured, not for sites measured as slow.
2184 $performance_score = $this->measured_performance_score();
2185
2186 if ($performance_score === null) {
2187 return [
2188 'score' => $max,
2189 'max_score' => $max,
2190 'suggestions' => ['Ensure mobile-first design and fast loading on mobile devices'],
2191 'details' => ['mobile_score' => 'Assumed adequate', 'measured' => false],
2192 ];
2193 }
2194
2195 $score = (int) round($max * $performance_score / 100);
2196
2197 return [
2198 'score' => $score,
2199 'max_score' => $max,
2200 'suggestions' => $score < $max
2201 ? ['Improve mobile page speed: the last PageSpeed run scored ' . $performance_score . '/100 on mobile']
2202 : [],
2203 'details' => [
2204 'mobile_score' => $performance_score,
2205 'measured' => true,
2206 'source' => 'pagespeed_mobile',
2207 ],
2208 ];
2209 }
2210
2211 /**
2212 * The last collected mobile Lighthouse score, or null when unmeasured.
2213 *
2214 * Memoised per instance: compute_score() asks twice, and a post-list screen
2215 * scores a page of posts at a time.
2216 *
2217 * Every failure — no performance module, no collected row, an unreadable
2218 * table — resolves to null, which the callers read as "not measured" and
2219 * answer with the full-credit fallback. A site is never penalised for
2220 * ThinkRank being unable to look.
2221 *
2222 * @since 2.3.1
2223 * @return int|null Score 0-100, or null when nothing has been collected.
2224 */
2225 private function measured_performance_score(): ?int {
2226 $measurement = $this->measured_performance();
2227
2228 if ($measurement === null || !isset($measurement['performance_score'])) {
2229 return null;
2230 }
2231
2232 $score = $measurement['performance_score'];
2233
2234 if (!is_numeric($score)) {
2235 return null;
2236 }
2237
2238 return (int) round(max(0, min(100, (float) $score)));
2239 }
2240
2241 /**
2242 * The last collected mobile measurement, or null when there is none.
2243 *
2244 * @since 2.3.1
2245 * @return array|null { core_web_vitals: array, performance_score: float|null }
2246 */
2247 private function measured_performance(): ?array {
2248 if ($this->measured_performance_resolved) {
2249 return $this->measured_performance;
2250 }
2251
2252 $this->measured_performance_resolved = true;
2253
2254 if (!class_exists('ThinkRank\\SEO\\Performance_Monitoring_Manager')) {
2255 return null;
2256 }
2257
2258 try {
2259 $manager = new \ThinkRank\SEO\Performance_Monitoring_Manager();
2260 // Mobile deliberately: Google indexes mobile-first, and it is the
2261 // device the mobile_experience factor is named after.
2262 $this->measured_performance = $manager->get_stored_performance_measurement('mobile');
2263 } catch (\Throwable $e) {
2264 $this->measured_performance = null;
2265 }
2266
2267 return $this->measured_performance;
2268 }
2269
2270 /**
2271 * Score core web vitals - 2025 version (3 points)
2272 *
2273 * @param array $content_data Content analysis data
2274 * @return array Scoring result
2275 */
2276 private function score_core_web_vitals(array $content_data): array {
2277 $max = $this->scoring_factors['core_web_vitals'];
2278
2279 // Not derivable from post content — but it is measured, and the audit
2280 // stores LCP, INP and CLS with a rating each. Score against those when
2281 // they exist; fall back to the benefit of the doubt when they do not.
2282 $rated = $this->measured_vitals_score();
2283
2284 if ($rated === null) {
2285 return [
2286 'score' => $max,
2287 'max_score' => $max,
2288 'suggestions' => ['Optimize Core Web Vitals: LCP, INP, and CLS for better user experience'],
2289 'details' => ['vitals_status' => 'Assumed adequate', 'measured' => false],
2290 ];
2291 }
2292
2293 $score = (int) round($max * $rated['average'] / 100);
2294
2295 return [
2296 'score' => $score,
2297 'max_score' => $max,
2298 // Gated on the measurement, not the rounded score: two good metrics
2299 // and one needing improvement averages 88.33, which rounds to the
2300 // full 3 of 3 and used to swallow the suggestion naming the metric
2301 // that is actually failing.
2302 'suggestions' => !empty($rated['failing'])
2303 ? ['Optimize Core Web Vitals: ' . implode(', ', $rated['failing']) . ' below target on mobile']
2304 : [],
2305 'details' => [
2306 'vitals_status' => $rated['statuses'],
2307 'measured' => true,
2308 'source' => 'pagespeed_mobile',
2309 ],
2310 ];
2311 }
2312
2313 /**
2314 * Rate the collected Core Web Vitals, or null when none were measured.
2315 *
2316 * Reuses the per-metric score the performance module already assigns
2317 * (good 100, needs improvement 65, poor 30) rather than inventing a second
2318 * scale, so the SEO score and the performance card cannot disagree about
2319 * whether a metric is healthy.
2320 *
2321 * Metrics with no stored value — fcp is not always collected — are skipped
2322 * rather than counted as failures.
2323 *
2324 * @since 2.3.1
2325 * @return array|null { average: float, statuses: array, failing: string[] }
2326 */
2327 private function measured_vitals_score(): ?array {
2328 $measurement = $this->measured_performance();
2329 $vitals = $measurement['core_web_vitals'] ?? null;
2330
2331 if (!is_array($vitals)) {
2332 return null;
2333 }
2334
2335 $scores = [];
2336 $statuses = [];
2337 $failing = [];
2338
2339 // The three Google ranks on. fcp is diagnostic and not a Core Web Vital.
2340 foreach (['lcp', 'inp', 'cls'] as $metric) {
2341 $data = $vitals[$metric] ?? null;
2342
2343 if (!is_array($data) || !isset($data['value'], $data['score']) || $data['value'] === null) {
2344 continue;
2345 }
2346
2347 $scores[] = (float) $data['score'];
2348 $statuses[$metric] = $data['status'] ?? 'unknown';
2349
2350 if (($data['status'] ?? '') !== 'good') {
2351 $failing[] = strtoupper($metric);
2352 }
2353 }
2354
2355 if (empty($scores)) {
2356 return null;
2357 }
2358
2359 return [
2360 'average' => array_sum($scores) / count($scores),
2361 'statuses' => $statuses,
2362 'failing' => $failing,
2363 ];
2364 }
2365
2366 /**
2367 * Score internal linking - declining importance (1 point)
2368 *
2369 * @param array $content_data Content analysis data
2370 * @return array Scoring result
2371 */
2372 private function score_internal_linking(array $content_data): array {
2373 $internal_links = $content_data['internal_links'] ?? 0;
2374 $score = $internal_links > 0 ? 1 : 0;
2375
2376 return [
2377 'score' => $score,
2378 'max_score' => $this->scoring_factors['internal_linking'],
2379 'suggestions' => $score === 0 ? ['Add relevant internal links to other pages on your site'] : [],
2380 'details' => ['internal_links_count' => $internal_links]
2381 ];
2382 }
2383
2384 /**
2385 * Score technical factors (1 point)
2386 *
2387 * @param array $content_data Content analysis data
2388 * @param array $metadata Post metadata
2389 * @param array $options Additional options
2390 * @return array Scoring result
2391 */
2392 private function score_technical_factors(array $content_data, array $metadata, array $options): array {
2393 $score = 0;
2394 $suggestions = [];
2395
2396 // Meta description check
2397 $meta_desc = $metadata['description'] ?? '';
2398 if (!empty($meta_desc) && mb_strlen($meta_desc) >= self::DESCRIPTION_OPTIMAL_MIN && mb_strlen($meta_desc) <= self::DESCRIPTION_OPTIMAL_MAX) {
2399 $score += 0.5;
2400 } else {
2401 $suggestions[] = 'Add a compelling meta description (120-160 characters)';
2402 }
2403
2404 // Schema markup check (simplified)
2405 if (!empty($content_data['schema_present'])) {
2406 $score += 0.5;
2407 } else {
2408 $suggestions[] = 'Consider adding structured data (schema markup)';
2409 }
2410
2411 return [
2412 'score' => $score,
2413 'max_score' => $this->scoring_factors['technical_factors'],
2414 'suggestions' => $suggestions,
2415 'details' => [
2416 'meta_description_length' => mb_strlen($meta_desc),
2417 'schema_present' => !empty($content_data['schema_present'])
2418 ]
2419 ];
2420 }
2421
2422 /**
2423 * Get grade from score
2424 *
2425 * @param mixed $score Numeric score
2426 * @return string Letter grade
2427 */
2428 private function get_grade_from_score($score): string {
2429 $score = (int) $score; // Ensure it's an integer
2430
2431 if ($score >= 95) { return 'A+';
2432 }
2433 if ($score >= 90) { return 'A';
2434 }
2435 if ($score >= 85) { return 'A-';
2436 }
2437 if ($score >= 80) { return 'B+';
2438 }
2439 if ($score >= 75) { return 'B';
2440 }
2441 if ($score >= 70) { return 'B-';
2442 }
2443 if ($score >= 65) { return 'C+';
2444 }
2445 if ($score >= 60) { return 'C';
2446 }
2447 if ($score >= 55) { return 'C-';
2448 }
2449 if ($score >= 45) { return 'D+';
2450 }
2451 if ($score >= 35) { return 'D';
2452 }
2453 return 'F';
2454 }
2455
2456 /**
2457 * Analyze live content from editor (not saved to database yet)
2458 * Same as analyze_post_content but uses provided content instead of saved content
2459 *
2460 * @param string $live_content Live content from editor
2461 * @param int $post_id Post ID for metadata
2462 * @return array Content analysis data
2463 */
2464 public function analyze_live_content(string $live_content, int $post_id): array {
2465 $post = get_post($post_id);
2466 if (!$post) {
2467 return [];
2468 }
2469
2470 // Resolve the live string the same way stored content is resolved. On a
2471 // builder page the editor hands over raw builder markup (the block
2472 // editor cannot render blocks it has no client-side registration for),
2473 // which analyzed as-is reads as zero words — the reason a Divi page
2474 // could show a correct saved score beside a live panel still claiming
2475 // "No content".
2476 $content = self::resolve_live_content($live_content, $post);
2477 $title = $post->post_title;
2478
2479 // Extract headings from content
2480 $headings = $this->extract_headings($content);
2481
2482 // Count words using JavaScript-compatible method
2483 $plain_text = wp_strip_all_tags($content);
2484 $word_count = $this->calculate_word_count_js_style($plain_text);
2485
2486 // Calculate readability
2487 $readability_score = $this->calculate_readability_score($content);
2488
2489 // Count links
2490 $internal_links = $this->count_internal_links($content);
2491 $external_links = $this->count_external_links($content);
2492 $external_dofollow_links = $this->count_external_dofollow_links($content);
2493
2494 // Analyze images
2495 $images = $this->analyze_images($content);
2496
2497 // Get URL
2498 $url = get_permalink($post_id);
2499
2500 return [
2501 'content' => $content,
2502 'title' => $title,
2503 'headings' => $headings,
2504 'word_count' => $word_count,
2505 'readability_score' => $readability_score,
2506 'internal_links' => $internal_links,
2507 'external_links' => $external_links,
2508 'external_dofollow_links' => $external_dofollow_links,
2509 'images' => $images,
2510 'url' => $url,
2511 'slug' => $post->post_name,
2512 'post_modified' => $post->post_modified,
2513 'schema_present' => $this->detect_schema_present($content)
2514 || $this->thinkrank_global_schema_active($post->post_type)
2515 || $this->thinkrank_deployed_schema_active($post),
2516 ];
2517 }
2518
2519 /**
2520 * Calculate word count using JavaScript-compatible method
2521 * Matches the logic in contentAnalysis.js for consistency
2522 *
2523 * @param string $text Text to count words in
2524 * @return int Word count
2525 */
2526 private function calculate_word_count_js_style(string $text): int {
2527 if (empty($text)) {
2528 return 0;
2529 }
2530
2531 // Match JavaScript: trim, split by whitespace, filter empty
2532 $words = preg_split('/\s+/', trim($text), -1, PREG_SPLIT_NO_EMPTY);
2533 return count($words);
2534 }
2535
2536 /**
2537 * Get existing score data for a post from database
2538 *
2539 * @param int $post_id Post ID
2540 * @return array|null Existing score data or null if not found
2541 */
2542 public function get_existing_score_data(int $post_id): ?array {
2543 global $wpdb;
2544
2545 // Get table name and escape it properly (table names cannot be parameterized)
2546 $table_name = esc_sql($wpdb->prefix . 'thinkrank_seo_scores');
2547
2548 // Get the most recent score for this post
2549 // phpcs:ignore WordPress.DB.DirectDatabaseQuery.DirectQuery,WordPress.DB.DirectDatabaseQuery.NoCaching, PluginCheck.Security.DirectDB.UnescapedDBParameter -- SEO score retrieval requires direct database access
2550 $result = $wpdb->get_row($wpdb->prepare(
2551 // phpcs:ignore WordPress.DB.PreparedSQL.InterpolatedNotPrepared, PluginCheck.Security.DirectDB.UnescapedDBParameter -- Table name is properly escaped using esc_sql()
2552 "SELECT * FROM `{$table_name}`
2553 WHERE post_id = %d
2554 ORDER BY calculated_at DESC
2555 LIMIT 1",
2556 $post_id
2557 ), ARRAY_A);
2558
2559 if (!$result) {
2560 return null;
2561 }
2562
2563 // Decode JSON data (stored with json_encode)
2564 $score_breakdown = json_decode($result['score_breakdown'], true);
2565 $suggestions = json_decode($result['suggestions'], true);
2566
2567 // Format the data to match the expected structure
2568 return [
2569 'overall_score' => (int) $result['overall_score'],
2570 'grade' => $result['grade'],
2571 'score_breakdown' => $score_breakdown,
2572 'suggestions' => $suggestions ?: [],
2573 'target_keyword' => null, // Not stored in database, will be provided by frontend
2574 'calculated_at' => $result['calculated_at'],
2575 'score_id' => $result['id']
2576 ];
2577 }
2578 }
2579