PluginProbe
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO / 2.11.0
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO v2.11.0
2.11.0 2.10.0 2.9.0 2.8.0 2.7.0 2.6.0 2.5.0 2.4.0 2.3.0 2.2.0 2.1.1 2.1.0 2.0.2 2.0.1 2.0.0 1.32.0 1.31.0 1.30.0 1.29.0 1.28.0 1.27.0 1.26.0 1.25.0 trunk 1.0.0 All 52 releases
thinkrank / includes / ai / class-seo-score-calculator.php

class-seo-score-calculator.php in ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO 2.11.0, at includes/ai/class-seo-score-calculator.php

2,564 lines 98.5 KB
No matching file
Up and down to move Enter to open Esc to close
Raw Download Zip
1 <?php
2 /**
3 * SEO Score Calculator
4 *
5 * Advanced SEO scoring system based on 2025 Google ranking factors
6 * Implements AI-driven content analysis, searcher engagement signals, and current SEO best practices
7 *
8 * @package ThinkRank\AI
9 * @since 1.0.0
10 */
11
12 declare(strict_types=1);
13
14 namespace ThinkRank\AI;
15
16 use ThinkRank\Core\Database;
17
18 // Prevent direct access
19 if (!defined('ABSPATH')) {
20 exit;
21 }
22
23 /**
24 * SEO Score Calculator Class
25 *
26 * Implements 2025 SEO scoring algorithm based on:
27 * - Google's Q1 2025 algorithm updates (First Page Sage research)
28 * - Satisfying content as #1 ranking factor (23%)
29 * - Searcher engagement and intent satisfaction (12%)
30 * - Mobile Experience Score (MES) and Core Web Vitals 2.0
31 * - Content freshness and niche expertise signals
32 *
33 * @since 1.0.0
34 */
35 class SEOScoreCalculator {
36
37 /**
38 * Database instance
39 *
40 * @var Database
41 */
42 private Database $database;
43
44 /**
45 * Memoised collected performance measurement, and whether it was resolved.
46 *
47 * Two factors read it and both may be asked for on every post in a list, so
48 * the lookup happens once per calculator. `null` is a real answer here — the
49 * separate flag keeps "not looked up yet" distinct from "nothing measured".
50 *
51 * @since 2.3.1
52 * @var array|null
53 */
54 private ?array $measured_performance = null;
55
56 /**
57 * @since 2.3.1
58 * @var bool
59 */
60 private bool $measured_performance_resolved = false;
61
62 /**
63 * Length bands the editor scores against, in characters.
64 *
65 * Public so every surface that judges a title or description — the editor
66 * score and the Bulk Snippets problem filter — reads one set of numbers.
67 * Before these existed the bands were literals inside the scoring methods,
68 * and a second screen would have had to copy them and drift (#727).
69 *
70 * @since 2.8.0
71 */
72 public const TITLE_OPTIMAL_MIN = 35;
73 public const TITLE_OPTIMAL_MAX = 60;
74 public const DESCRIPTION_OPTIMAL_MIN = 120;
75 public const DESCRIPTION_OPTIMAL_MAX = 160;
76
77 /**
78 * 2025 SEO scoring factors (Q1 2025 Google Algorithm)
79 * Based on First Page Sage research and Google's latest updates
80 *
81 * @var array
82 */
83 private array $scoring_factors = [
84 'satisfying_content' => 23, // #1: Consistent publication of satisfying content
85 'title_optimization' => 14, // #2: Keyword in meta title (looser matching)
86 'niche_expertise' => 13, // #3: Hub & spoke content clusters
87 'searcher_engagement' => 12, // #4: Dwell time, bounce rate, pages/session
88 'backlink_authority' => 13, // #5: Quality backlinks (declining but important)
89 'content_freshness' => 6, // #6: Quarterly content updates
90 'mobile_experience' => 5, // #7: Mobile Experience Score (MES) - NEW 2025
91 'trustworthiness' => 4, // #8: E-E-A-T verification
92 'link_diversity' => 3, // #9: Link distribution across multiple pages
93 'core_web_vitals' => 3, // #10: Page speed + Interaction Readiness
94 'site_security' => 2, // #11: SSL certificate
95 'internal_linking' => 1, // #12: Declining importance
96 'technical_factors' => 1, // #13: Meta descriptions, schema, etc.
97 ];
98
99 /**
100 * Constructor
101 *
102 * @param Database $database Database instance
103 */
104 public function __construct(Database $database) {
105 $this->database = $database;
106 }
107
108 /**
109 * Calculate comprehensive modern SEO score.
110 *
111 * Supports multiple focus keywords: when `$options['target_keywords']` holds
112 * more than one keyword the score is computed independently for each and the
113 * HIGHEST overall score is returned as the final result, with per-keyword
114 * results (`keyword_results`) and OR-combined per-check matches
115 * (`keyword_checks`) attached. A single keyword (or the legacy
116 * `target_keyword` option) falls through to the single-keyword path.
117 *
118 * @param array $content_data Content analysis data
119 * @param array $metadata Post metadata
120 * @param array $options Additional options
121 * @return array Complete scoring result
122 */
123 public function calculate_score(array $content_data, array $metadata, array $options = []): array {
124 $keywords = $this->resolve_target_keywords($options, $metadata);
125
126 if (count($keywords) > 1) {
127 return $this->calculate_score_multi($content_data, $metadata, $keywords, $options);
128 }
129
130 $options['target_keyword'] = $keywords[0] ?? '';
131 $result = $this->compute_score($content_data, $metadata, $options);
132
133 // Expose the keyword surface uniformly so consumers can rely on it
134 // regardless of how many keywords were supplied.
135 $result['target_keywords'] = $keywords;
136 if (!empty($keywords)) {
137 $result['keyword_results'] = [[
138 'keyword' => $keywords[0],
139 'overall_score' => $result['overall_score'],
140 'grade' => $result['grade'],
141 ]];
142 $result['keyword_checks'] = $this->analyze_keyword_checks($content_data, $metadata, $keywords);
143 }
144 $result['keywords'] = $this->keyword_placements($content_data, $metadata, $keywords);
145
146 return $result;
147 }
148
149 /**
150 * Resolve the target keyword list from the options array, falling back to
151 * the focus keywords carried on the post's metadata.
152 *
153 * Accepts `target_keywords` (array) or the legacy `target_keyword` (string),
154 * trims, drops empties and removes case-insensitive duplicates.
155 *
156 * Options win over metadata on purpose: the editor scores unsaved keyword
157 * edits by passing them explicitly, and that live value must beat whatever
158 * is currently persisted. The metadata fallback applies only when the
159 * caller mentions no keyword option AT ALL — a caller that passes an empty
160 * keyword is deliberately clearing it (the editor does exactly this when
161 * the field is emptied), so the stored value must not resurrect it.
162 *
163 * Without the fallback, every caller that hands over
164 * `Metabox_Manager::get_post_metadata()` (the metabox and the MCP scoring
165 * abilities) silently scored as if no keyword were set.
166 *
167 * @param array $options Scoring options.
168 * @param array $metadata Post metadata (may carry focus_keyword(s)).
169 * @return string[] Normalized keyword list.
170 */
171 private function resolve_target_keywords(array $options, array $metadata = []): array {
172 $raw = [];
173 if (!empty($options['target_keywords']) && is_array($options['target_keywords'])) {
174 $raw = $options['target_keywords'];
175 } elseif (isset($options['target_keyword']) && $options['target_keyword'] !== '') {
176 $raw = [$options['target_keyword']];
177 } elseif (!$this->options_mention_keywords($options)) {
178 if (!empty($metadata['focus_keywords']) && is_array($metadata['focus_keywords'])) {
179 $raw = $metadata['focus_keywords'];
180 } elseif (isset($metadata['focus_keyword']) && is_string($metadata['focus_keyword']) && $metadata['focus_keyword'] !== '') {
181 $raw = [$metadata['focus_keyword']];
182 }
183 }
184
185 $seen = [];
186 $keywords = [];
187 foreach ($raw as $keyword) {
188 $keyword = trim((string) $keyword);
189 if ($keyword === '') {
190 continue;
191 }
192 $key = strtolower($keyword);
193 if (isset($seen[$key])) {
194 continue;
195 }
196 $seen[$key] = true;
197 $keywords[] = $keyword;
198 }
199
200 return $keywords;
201 }
202
203 /**
204 * Whether the caller said anything about keywords — including saying
205 * "none". Distinguishes an intentional clear (score with no keyword) from
206 * silence (fall back to the post's stored focus keywords).
207 *
208 * @param array $options Scoring options.
209 * @return bool True when a keyword option key is present.
210 */
211 private function options_mention_keywords(array $options): bool {
212 return array_key_exists('target_keywords', $options)
213 || array_key_exists('target_keyword', $options);
214 }
215
216 /**
217 * Score each keyword independently and return the highest-scoring result.
218 *
219 * @param array $content_data Content analysis data.
220 * @param array $metadata Post metadata.
221 * @param string[] $keywords Target keywords (already normalized, 2+).
222 * @param array $options Additional options.
223 * @return array Best scoring result, with per-keyword data attached.
224 */
225 private function calculate_score_multi(array $content_data, array $metadata, array $keywords, array $options): array {
226 $per_keyword = [];
227 $best = null;
228 $best_keyword = $keywords[0];
229
230 foreach ($keywords as $keyword) {
231 $opts = $options;
232 unset($opts['target_keywords']);
233 $opts['target_keyword'] = $keyword;
234
235 $result = $this->compute_score($content_data, $metadata, $opts);
236
237 $per_keyword[] = [
238 'keyword' => $keyword,
239 'overall_score' => $result['overall_score'],
240 'grade' => $result['grade'],
241 'score_breakdown' => $result['score_breakdown'],
242 ];
243
244 if ($best === null || $result['overall_score'] > $best['overall_score']) {
245 $best = $result;
246 $best_keyword = $keyword;
247 }
248 }
249
250 // Final score = highest individual keyword score. Retain per-keyword
251 // results and OR-combined checks so the UI can show both.
252 $best['target_keyword'] = $best_keyword;
253 $best['target_keywords'] = $keywords;
254 $best['keyword_results'] = $per_keyword;
255 $best['keyword_checks'] = $this->analyze_keyword_checks($content_data, $metadata, $keywords);
256 $best['keywords'] = $this->keyword_placements($content_data, $metadata, $keywords);
257
258 return $best;
259 }
260
261 /**
262 * Evaluate per-location keyword checks across ALL focus keywords.
263 *
264 * Each check (title, meta description, content, image alt, slug) passes when
265 * ANY of the focus keywords matches that location.
266 *
267 * @param array $content_data Content analysis data.
268 * @param array $metadata Post metadata.
269 * @param string[] $keywords Target keywords.
270 * @return array<string,array{passed:bool,matched_keywords:string[]}>
271 */
272 private function analyze_keyword_checks(array $content_data, array $metadata, array $keywords): array {
273 $title = self::lower((string) ($metadata['title'] ?? $content_data['title'] ?? ''));
274 $description = self::lower((string) ($metadata['description'] ?? ''));
275 $content = self::lower(self::plain_text((string) ($content_data['content'] ?? '')));
276
277 $alts = '';
278 foreach ((array) ($content_data['images'] ?? []) as $image) {
279 $alts .= ' ' . self::lower((string) ($image['alt'] ?? ''));
280 }
281
282 // Build a searchable slug haystack from the post's OWN slug — never the
283 // full URL path. The path carries ancestors, category bases and date
284 // segments, so a child of /clinical-trials/ reported "keyword in slug"
285 // for a page actually slugged `contact-us`. It also breaks the other
286 // way: an unpublished post has no pretty permalink (get_permalink()
287 // returns ?p=123), so the path held no slug at all and every draft
288 // scored "no match" until it was published. Hyphens/underscores become
289 // spaces so multi-word keywords can match.
290 $slug = self::lower(self::slug_haystack($content_data));
291
292 $haystacks = [
293 'title' => trim($title),
294 'meta_description' => trim($description),
295 'content' => trim($content),
296 'image_alt' => trim($alts),
297 'slug' => trim($slug),
298 ];
299
300 $checks = [];
301 foreach ($haystacks as $location => $haystack) {
302 $matched = [];
303 foreach ($keywords as $keyword) {
304 if ($this->keyword_matches($haystack, self::lower(trim($keyword)))) {
305 $matched[] = $keyword;
306 }
307 }
308 $checks[$location] = [
309 'passed' => !empty($matched),
310 'matched_keywords' => $matched,
311 ];
312 }
313
314 return $checks;
315 }
316
317 /**
318 * The keyword placements the editor draws one gauge segment for, in the
319 * order a reader meets them (#729).
320 *
321 * @since 2.11.0
322 * @var string[]
323 */
324 public const PLACEMENTS = ['title', 'meta_description', 'slug', 'first_paragraph', 'subheading', 'content', 'image_alt'];
325
326 /**
327 * Characters of plain text read as the opening when the content has no
328 * paragraph tag.
329 *
330 * @since 2.11.0
331 */
332 private const OPENING_CHARS = 300;
333
334 /**
335 * Where each focus keyword is placed, keyword by keyword (#729).
336 *
337 * analyze_keyword_checks() answers "does ANY keyword appear here" for
338 * five places; this answers "where does THIS keyword appear" for seven,
339 * so the editor can show each keyword's own gauge. Same matcher, so a
340 * keyword counts as a word (not a fragment) and a keyword in a script
341 * written without spaces (Thai, Chinese, Japanese) still matches.
342 *
343 * `where` names the heading or alt text that matched, so the editor can
344 * say which one.
345 *
346 * @since 2.11.0
347 *
348 * @param array $content_data Content analysis data.
349 * @param array $metadata Post metadata (title, description).
350 * @param string[] $keywords Focus keywords.
351 * @return array<int, array{keyword: string, passed: int, total: int, placements: array<string, array{passed: bool, where: string}>}>
352 */
353 public function keyword_placements(array $content_data, array $metadata, array $keywords): array {
354 $html = (string) ($content_data['content'] ?? '');
355 $plain = self::lower(self::plain_text($html));
356
357 $headings = [];
358 $source = isset($content_data['headings']) && is_array($content_data['headings']) ? $content_data['headings'] : $this->extract_headings($html);
359 foreach ($source as $heading) {
360 $text = trim((string) ($heading['text'] ?? ''));
361 if ((int) ($heading['level'] ?? 0) >= 2 && '' !== $text) {
362 $headings[] = $text;
363 }
364 }
365
366 $alts = [];
367 foreach ((array) ($content_data['images'] ?? []) as $image) {
368 $alt = trim((string) ($image['alt'] ?? ''));
369 if ('' !== $alt) {
370 $alts[] = $alt;
371 }
372 }
373
374 $single = [
375 'title' => self::lower((string) ($metadata['title'] ?? $content_data['title'] ?? '')),
376 'meta_description' => self::lower((string) ($metadata['description'] ?? '')),
377 'slug' => self::lower(self::slug_haystack($content_data)),
378 'first_paragraph' => self::lower(self::opening($html)),
379 'content' => $plain,
380 ];
381
382 $out = [];
383 foreach ($keywords as $keyword) {
384 $needle = self::lower(trim((string) $keyword));
385 $placements = [];
386
387 foreach (self::PLACEMENTS as $placement) {
388 if ('subheading' === $placement || 'image_alt' === $placement) {
389 $where = '';
390 foreach ('subheading' === $placement ? $headings : $alts as $text) {
391 if ($this->keyword_matches(self::lower($text), $needle)) {
392 $where = $text;
393 break;
394 }
395 }
396 $placements[$placement] = ['passed' => '' !== $where, 'where' => $where];
397 continue;
398 }
399
400 $placements[$placement] = ['passed' => $this->keyword_matches($single[$placement], $needle), 'where' => ''];
401 }
402
403 $out[] = [
404 'keyword' => (string) $keyword,
405 'passed' => count(array_filter(array_column($placements, 'passed'))),
406 'total' => count(self::PLACEMENTS),
407 'placements' => $placements,
408 ];
409 }
410
411 return $out;
412 }
413
414 /**
415 * The post's own slug as searchable text: hyphens and underscores become
416 * spaces so a multi-word keyword can match, and a slug WordPress
417 * percent-encoded (Thai, Cyrillic, Chinese) is decoded, or it could never
418 * match a keyword typed in that script.
419 *
420 * Never the full URL path: the path carries ancestors, category bases and
421 * date segments, so a child of /clinical-trials/ reported "keyword in slug"
422 * for a page actually slugged `contact-us`. An unpublished post has no
423 * pretty permalink either, so the path held no slug at all.
424 *
425 * @param array $content_data Content analysis data.
426 * @return string
427 */
428 private static function slug_haystack(array $content_data): string {
429 $slug = (string) ($content_data['slug'] ?? '');
430 if ('' === $slug) {
431 // Draft with no slug assigned yet: score what WordPress would
432 // generate from the title, which is what the editor shows as the
433 // proposed URL — so the check reads the same before and after
434 // publishing instead of flipping.
435 $slug = sanitize_title((string) ($content_data['title'] ?? ''));
436 }
437
438 return trim(str_replace(['-', '_'], ' ', rawurldecode($slug)));
439 }
440
441 /**
442 * The opening of the content: its first paragraph with text, or the first
443 * few hundred characters when it has none. Counted in characters, not
444 * words, so a language written without spaces is not read as one word.
445 *
446 * @param string $html Content HTML.
447 * @return string Plain text.
448 */
449 private static function opening(string $html): string {
450 if (preg_match_all('/<p\b[^>]*>(.*?)<\/p>/isu', $html, $matches)) {
451 foreach ($matches[1] as $paragraph) {
452 $text = self::collapse_whitespace(wp_strip_all_tags($paragraph));
453 if ('' !== $text) {
454 return $text;
455 }
456 }
457 }
458
459 $text = self::plain_text($html);
460
461 return function_exists('mb_substr') ? mb_substr($text, 0, self::OPENING_CHARS) : substr($text, 0, self::OPENING_CHARS);
462 }
463
464 /**
465 * Content as plain text, with a space where each tag was. wp_strip_all_tags()
466 * alone joins neighbouring blocks — "…coffee grinder</h3><p>A good…" became
467 * "coffee grinderA good" — so a keyword at the end of a heading or a
468 * paragraph was no longer a word and did not match.
469 *
470 * @param string $html Content HTML.
471 * @return string
472 */
473 private static function plain_text(string $html): string {
474 $spaced = preg_replace('/<[^>]+>/', ' $0 ', $html);
475
476 return self::collapse_whitespace(wp_strip_all_tags(null === $spaced ? $html : (string) $spaced));
477 }
478
479 /**
480 * Runs of whitespace down to one space.
481 *
482 * The `/u` pass is the one that understands a multibyte space, but
483 * preg_replace() answers null on bytes that are not valid UTF-8 rather than
484 * throwing — and casting that null to a string blanked the haystack, so a
485 * post carrying one mojibake byte (a Latin-1 paste, an old import) reported
486 * every keyword as missing from its body, its opening, and every
487 * subheading. The gauge said 0/7 and told the author to add a keyword that
488 * was already there.
489 *
490 * Falls back to the byte-wise collapse, which is what this did before the
491 * multibyte work added the modifier. Same reasoning keyword_matches()
492 * already records for its own PCRE failure: a pattern PCRE refuses must not
493 * be reported as a confident "no match".
494 *
495 * @param string $text Text to collapse.
496 * @return string
497 */
498 private static function collapse_whitespace(string $text): string {
499 $collapsed = preg_replace('/\s+/u', ' ', $text);
500
501 if (null === $collapsed) {
502 $collapsed = preg_replace('/\s+/', ' ', $text);
503 }
504
505 return trim(null === $collapsed ? $text : (string) $collapsed);
506 }
507
508 /**
509 * Lowercase in any script. strtolower() only folds ASCII, so "Кофе" never
510 * matched "кофе".
511 *
512 * @param string $text Text.
513 * @return string
514 */
515 private static function lower(string $text): string {
516 return function_exists('mb_strtolower') ? mb_strtolower($text, 'UTF-8') : strtolower($text);
517 }
518
519 /**
520 * Scripts written without spaces between words.
521 *
522 * @since 2.1.0
523 * @var string
524 */
525 private const SCRIPTIO_CONTINUA = '/[\p{Han}\p{Hiragana}\p{Katakana}\p{Thai}\p{Lao}\p{Khmer}\p{Myanmar}]/u';
526
527 /**
528 * Whether a keyword appears in a haystack as a word rather than as a
529 * fragment of a longer one.
530 *
531 * The five keyword checks used a plain strpos(), so any substring hit
532 * counted: "test coronavirus" matched "la|test coronavirus|news", "art"
533 * matched "start", "cat" matched "category". The panel then confidently
534 * reported a keyword placement that does not exist (#416). Same class of
535 * problem #71 fixed in the Image SEO rewriter, and the same remedy.
536 *
537 * Both arguments are expected lowercased already.
538 *
539 * @since 2.1.0
540 *
541 * @param string $haystack Text to search.
542 * @param string $needle Keyword, lowercased and trimmed.
543 * @return bool
544 */
545 private function keyword_matches(string $haystack, string $needle): bool {
546 if ($needle === '' || $haystack === '') {
547 return false;
548 }
549
550 if (!$this->supports_word_boundaries($needle)) {
551 return strpos($haystack, $needle) !== false;
552 }
553
554 $matched = preg_match('/\b' . preg_quote($needle, '/') . '\b/u', $haystack);
555
556 // PCRE refusing the pattern — invalid UTF-8 in the keyword, a
557 // backtrack limit — must not be reported as a confident "no match".
558 // Fall back to the behaviour this replaced rather than invent a
559 // negative the user cannot explain.
560 if ($matched === false) {
561 return strpos($haystack, $needle) !== false;
562 }
563
564 return $matched === 1;
565 }
566
567 /**
568 * Whether \b can express "this keyword, as a word" for this keyword.
569 *
570 * It asserts a transition between a word and a non-word character, which
571 * only means something where words are separated. Two cases where it is
572 * not, both verified against PCRE rather than assumed:
573 *
574 * - the keyword's own edges are not word characters ("c++", "#seo"), so
575 * no boundary can assert there and a real match is lost;
576 * - scripts written without spaces, where the neighbouring characters
577 * are word characters too — "冠状�
578 毒" inside "最新冠状�
579 毒新闻" is a
580 * legitimate match that \b never sees.
581 *
582 * Accented Latin and Cyrillic need no special handling: PHP's /u modifier
583 * turns on Unicode character properties, so "café" correctly does not
584 * match "cafés" and "коронавирус" does not match "коронавирусный".
585 *
586 * @since 2.1.0
587 *
588 * @param string $needle Keyword, lowercased and trimmed.
589 * @return bool
590 */
591 private function supports_word_boundaries(string $needle): bool {
592 if (preg_match(self::SCRIPTIO_CONTINUA, $needle)) {
593 return false;
594 }
595
596 return preg_match('/^\w/u', $needle) === 1 && preg_match('/\w$/u', $needle) === 1;
597 }
598
599 /**
600 * Compute the SEO score for a single target keyword.
601 *
602 * @param array $content_data Content analysis data
603 * @param array $metadata Post metadata
604 * @param array $options Additional options (expects scalar target_keyword)
605 * @return array Complete scoring result
606 *
607 * @throws \Exception On failure.
608 */
609 private function compute_score(array $content_data, array $metadata, array $options = []): array {
610 $scores = [];
611 $suggestions = [];
612 $total_score = 0;
613
614 try {
615 // 1. Satisfying Content (23 points) - #1 factor in 2025
616 $satisfying_result = $this->score_satisfying_content($content_data, $options['target_keyword'] ?? '', $metadata);
617 $scores['satisfying_content'] = $satisfying_result;
618 $total_score += $satisfying_result['score'];
619 $suggestions = array_merge($suggestions, $satisfying_result['suggestions']);
620 } catch (\Exception $e) {
621 throw $e;
622 }
623
624 try {
625 // 2. Title Optimization (14 points) - Looser keyword matching in 2025
626 $title_result = $this->score_2025_title_optimization($metadata['title'] ?? '', $options['target_keyword'] ?? '');
627 $scores['title_optimization'] = $title_result;
628 $total_score += $title_result['score'];
629 $suggestions = array_merge($suggestions, $title_result['suggestions']);
630 } catch (\Exception $e) {
631 throw $e;
632 }
633
634 try {
635 // 3. Niche Expertise (13 points) - Hub & spoke content clusters
636 $expertise_result = $this->score_niche_expertise($content_data, $options['target_keyword'] ?? '');
637 $scores['niche_expertise'] = $expertise_result;
638 $total_score += $expertise_result['score'];
639 $suggestions = array_merge($suggestions, $expertise_result['suggestions']);
640 } catch (\Exception $e) {
641 throw $e;
642 }
643
644 try {
645 // 4. Searcher Engagement (12 points) - Dwell time, bounce rate, pages/session
646 $engagement_result = $this->score_searcher_engagement($content_data);
647 $scores['searcher_engagement'] = $engagement_result;
648 $total_score += $engagement_result['score'];
649 $suggestions = array_merge($suggestions, $engagement_result['suggestions']);
650 } catch (\Exception $e) {
651 throw $e;
652 }
653
654 try {
655 // 5. Backlink Authority (13 points) - Quality backlinks
656 $backlink_result = $this->score_backlink_authority($content_data);
657 $scores['backlink_authority'] = $backlink_result;
658 $total_score += $backlink_result['score'];
659 $suggestions = array_merge($suggestions, $backlink_result['suggestions']);
660 } catch (\Exception $e) {
661 throw $e;
662 }
663
664 // 6. Content Freshness (6 points) - Quarterly updates priority
665 $freshness_result = $this->score_content_freshness($content_data);
666 $scores['content_freshness'] = $freshness_result;
667 $total_score += $freshness_result['score'];
668 $suggestions = array_merge($suggestions, $freshness_result['suggestions']);
669
670 // 7. Mobile Experience (5 points) - NEW: Mobile Experience Score (MES)
671 $mobile_result = $this->score_mobile_experience($content_data);
672 $scores['mobile_experience'] = $mobile_result;
673 $total_score += $mobile_result['score'];
674 $suggestions = array_merge($suggestions, $mobile_result['suggestions']);
675
676 // 8. Trustworthiness (4 points) - E-E-A-T verification
677 $trust_result = $this->score_trustworthiness($content_data, $metadata);
678 $scores['trustworthiness'] = $trust_result;
679 $total_score += $trust_result['score'];
680 $suggestions = array_merge($suggestions, $trust_result['suggestions']);
681
682 // 9. Link Diversity (3 points) - Multiple pages with backlinks
683 $diversity_result = $this->score_link_diversity($content_data);
684 $scores['link_diversity'] = $diversity_result;
685 $total_score += $diversity_result['score'];
686 $suggestions = array_merge($suggestions, $diversity_result['suggestions']);
687
688 // 10. Core Web Vitals (3 points) - Interaction Readiness + CLS 2.0
689 $vitals_result = $this->score_core_web_vitals($content_data);
690 $scores['core_web_vitals'] = $vitals_result;
691 $total_score += $vitals_result['score'];
692 $suggestions = array_merge($suggestions, $vitals_result['suggestions']);
693
694 // 11. Site Security (2 points) - SSL certificate
695 $security_result = $this->score_site_security($content_data);
696 $scores['site_security'] = $security_result;
697 $total_score += $security_result['score'];
698 $suggestions = array_merge($suggestions, $security_result['suggestions']);
699
700 // 12. Internal Linking (1 point) - Declining importance
701 $internal_result = $this->score_internal_linking($content_data);
702 $scores['internal_linking'] = $internal_result;
703 $total_score += $internal_result['score'];
704 $suggestions = array_merge($suggestions, $internal_result['suggestions']);
705
706 // 13. Technical Factors (1 point) - Meta descriptions, schema, etc.
707 $technical_result = $this->score_technical_factors($content_data, $metadata, $options);
708 $scores['technical_factors'] = $technical_result;
709 $total_score += $technical_result['score'];
710 $suggestions = array_merge($suggestions, $technical_result['suggestions']);
711
712 try {
713 $prioritized_suggestions = $this->prioritize_suggestions($suggestions, $scores);
714 $grade = $this->get_grade_from_score($total_score);
715
716 return [
717 'overall_score' => min(100, $total_score),
718 'score_breakdown' => $scores,
719 'suggestions' => $prioritized_suggestions,
720 'grade' => $grade,
721 // Readability + content-quality labels so persisted scores
722 // (e.g. bulk-analyzed on import) populate the post-list columns
723 // without a manual re-analyze. The REST endpoint still overrides
724 // these with live-editor values when the metabox provides them.
725 'readability_score' => $this->format_readability_label($content_data),
726 'content_quality' => $this->derive_content_quality($content_data),
727 'calculated_at' => current_time('mysql'),
728 'algorithm_version' => '2025.2',
729 'algorithm_source' => 'First Page Sage Q1 2025 Research',
730 'factors_count' => count($scores),
731 ];
732 } catch (\Exception $e) {
733 throw $e;
734 }
735 }
736
737 /**
738 * Score satisfying content - #1 factor in 2025 (23 points)
739 * Google tests content to see if it satisfies search intent
740 *
741 * @param array $content_data Content analysis data
742 * @param string $target_keyword Target keyword
743 * @param array $metadata Post metadata (title, description)
744 * @return array Scoring result
745 */
746 private function score_satisfying_content(array $content_data, string $target_keyword, array $metadata = []): array {
747 $score = 0;
748 $max_score = $this->scoring_factors['satisfying_content'];
749 $suggestions = [];
750
751 $content = $content_data['content'] ?? '';
752 $word_count = $content_data['word_count'] ?? 0;
753 $meta_description = (string) ($metadata['description'] ?? '');
754
755 // Content depth and comprehensiveness (8 points). Tiers softened so a
756 // genuinely useful post is not capped the way the old 2000-word gate did
757 // (Rank Math awards full content credit well below 2000 words).
758 if ($word_count >= 1500) {
759 $score += 8;
760 } elseif ($word_count >= 1000) {
761 $score += 7;
762 $suggestions[] = 'Consider expanding content to 1500+ words for more comprehensive coverage';
763 } elseif ($word_count >= 600) {
764 $score += 5;
765 $suggestions[] = 'Content is adequate - aim for 1000+ words for stronger topic depth';
766 } elseif ($word_count >= 300) {
767 $score += 3;
768 $suggestions[] = 'Content is thin - aim for 600+ words minimum';
769 } else {
770 $score++;
771 $suggestions[] = 'Content too shallow - Google prioritizes comprehensive, satisfying content';
772 }
773
774 // Keyword presence & placement (8 points) - deterministic, replaces the
775 // old literal-phrase intent heuristic ("what is"/"because"). Measures
776 // signals the editor actually controls: body (3), first paragraph (3),
777 // meta description (2) - the last mirrors Rank Math's "keyword in meta
778 // description" basic-SEO check.
779 if (empty($target_keyword)) {
780 $score += 4; // Benefit of the doubt when no focus keyword is set.
781 $suggestions[] = 'Set a focus keyword so content relevance can be measured';
782 } else {
783 if ($this->keyword_in_content($content, $target_keyword)) {
784 $score += 3;
785 } else {
786 $suggestions[] = "Use the focus keyword '{$target_keyword}' in the body content";
787 }
788 if ($this->keyword_in_first_paragraph($content, $target_keyword)) {
789 $score += 3;
790 } else {
791 $suggestions[] = "Mention '{$target_keyword}' near the start of the content (first paragraph)";
792 }
793 if ($this->keyword_in_meta($meta_description, $target_keyword)) {
794 $score += 2;
795 } else {
796 $suggestions[] = "Include the focus keyword '{$target_keyword}' in the meta description";
797 }
798 }
799
800 // Content structure & value (7 points) - reuses the deterministic
801 // content-quality signal (length, paragraph length, subheading
802 // distribution) instead of the noisy sentence-length variety heuristic.
803 $quality = $this->derive_content_quality_score($content_data); // 0-100
804 $score += (int) round(($quality / 100) * 7);
805
806 if ($quality < 60) {
807 $suggestions[] = 'Improve content structure - break up long paragraphs and add subheadings';
808 }
809
810 return [
811 'score' => $score,
812 'max_score' => $max_score,
813 'suggestions' => $suggestions,
814 'details' => [
815 'word_count' => $word_count,
816 'keyword_in_content' => !empty($target_keyword) && $this->keyword_in_content($content, $target_keyword),
817 'keyword_in_first_paragraph' => !empty($target_keyword) && $this->keyword_in_first_paragraph($content, $target_keyword),
818 'keyword_in_meta_description' => !empty($target_keyword) && $this->keyword_in_meta($meta_description, $target_keyword),
819 'content_quality_score' => $quality,
820 'content_depth' => $this->assess_content_depth_2025($word_count),
821 ]
822 ];
823 }
824
825 /**
826 * Assess content depth for 2025 standards
827 *
828 * @param int $word_count Word count
829 * @return string Depth assessment
830 */
831 private function assess_content_depth_2025(int $word_count): string {
832 if ($word_count >= 3000) { return 'Comprehensive';
833 }
834 if ($word_count >= 2000) { return 'Detailed';
835 }
836 if ($word_count >= 1200) { return 'Adequate';
837 }
838 if ($word_count >= 800) { return 'Basic';
839 }
840 return 'Insufficient';
841 }
842
843 /**
844 * Score 2025 title optimization with looser keyword matching
845 *
846 * @param string $title Post title
847 * @param string $target_keyword Target keyword
848 * @return array Scoring result
849 */
850 private function score_2025_title_optimization(string $title, string $target_keyword): array {
851 $score = 0;
852 $max_score = $this->scoring_factors['title_optimization'];
853 $suggestions = [];
854
855 if (empty($title)) {
856 $suggestions[] = 'Add a compelling, click-worthy title that matches search intent';
857 return ['score' => 0, 'max_score' => $max_score, 'suggestions' => $suggestions];
858 }
859
860 $title_length = mb_strlen($title);
861
862 // 2025 length optimization (6 points). 60 characters is the recommended
863 // maximum for best SERP visibility before Google truncates the title.
864 if ($title_length >= self::TITLE_OPTIMAL_MIN && $title_length <= self::TITLE_OPTIMAL_MAX) {
865 $score += 6;
866 } elseif ($title_length >= 25 && $title_length <= 75) {
867 $score += 4;
868 $suggestions[] = 'Optimize title length to 35-60 characters for better SERP visibility';
869 } else {
870 $score++;
871 $suggestions[] = $title_length < 25 ?
872 'Title too short - aim for 35-60 characters' :
873 'Title too long - risk truncation in search results';
874 }
875
876 // Looser keyword matching (6 points) - 2025 update
877 if (!empty($target_keyword)) {
878 $title_lower = strtolower($title);
879 $keyword_lower = strtolower($target_keyword);
880
881 // Exact match
882 if (strpos($title_lower, $keyword_lower) !== false) {
883 $score += 6;
884 } else {
885 // Check for semantic variations (2025 improvement)
886 $semantic_match = $this->check_semantic_keyword_match($title, $target_keyword);
887 if ($semantic_match) {
888 $score += 5; // Almost full credit for semantic match
889 $suggestions[] = 'Good semantic keyword usage - Google now recognizes keyword variations';
890 } else {
891 // Check for partial keyword match
892 $keyword_parts = explode(' ', $keyword_lower);
893 $partial_matches = 0;
894 foreach ($keyword_parts as $part) {
895 if (strpos($title_lower, $part) !== false) {
896 $partial_matches++;
897 }
898 }
899
900 if ($partial_matches > 0) {
901 $score += round(($partial_matches / count($keyword_parts)) * 4);
902 $suggestions[] = "Include more parts of target keyword '{$target_keyword}' in title";
903 } else {
904 $suggestions[] = "Include target keyword '{$target_keyword}' or related terms in title";
905 }
906 }
907 }
908 } else {
909 $score += 2; // Partial credit
910 $suggestions[] = 'Set a target keyword to optimize title effectiveness';
911 }
912
913 // Title readability (2 points) - mirrors Rank Math's title checks for a
914 // number/power word (drives CTR) and emotional sentiment.
915 $has_number = (bool) preg_match('/\d/', $title);
916 $has_power_word = $this->title_has_power_word($title);
917 $has_sentiment = $this->title_has_sentiment_word($title);
918
919 if ($has_number || $has_power_word) {
920 $score++;
921 } else {
922 $suggestions[] = 'Add a number or a power word to the title to boost click-through rate';
923 }
924 if ($has_sentiment) {
925 $score++;
926 } else {
927 $suggestions[] = 'Use an emotional/sentiment word in the title to make it more compelling';
928 }
929
930 return [
931 'score' => $score,
932 'max_score' => $max_score,
933 'suggestions' => $suggestions,
934 'details' => [
935 'title_length' => $title_length,
936 'optimal_range' => '35-60 characters',
937 'keyword_present' => !empty($target_keyword) && strpos(strtolower($title), strtolower($target_keyword)) !== false,
938 'semantic_match' => !empty($target_keyword) ? $this->check_semantic_keyword_match($title, $target_keyword) : false,
939 'has_number_or_power_word' => $has_number || $has_power_word,
940 'has_sentiment_word' => $has_sentiment,
941 ]
942 ];
943 }
944
945 // Placeholder methods for remaining 2025 factors
946
947 private function score_niche_expertise(array $content_data, string $target_keyword): array {
948 $max_score = $this->scoring_factors['niche_expertise'];
949 $suggestions = [];
950
951 // Without a focus keyword we cannot measure topical coverage; award
952 // partial credit rather than capping the ceiling with a placeholder.
953 if (empty($target_keyword)) {
954 return [
955 'score' => 7,
956 'max_score' => $max_score,
957 'suggestions' => ['Set a focus keyword and use it in subheadings and the URL for stronger topical signals'],
958 'details' => ['expertise_level' => 'Unmeasured (no focus keyword)'],
959 ];
960 }
961
962 $score = 0;
963 $content = (string) ($content_data['content'] ?? '');
964 $headings = (array) ($content_data['headings'] ?? []);
965 $images = (array) ($content_data['images'] ?? []);
966 $slug = strtolower((string) ($content_data['slug'] ?? ''));
967
968 // Keyword in a subheading (4 points).
969 if ($this->keyword_in_subheadings($headings, $target_keyword)) {
970 $score += 4;
971 } else {
972 $suggestions[] = "Include '{$target_keyword}' in at least one subheading (H2-H6)";
973 }
974
975 // Keyword density in a healthy band (4 points). Rank Math treats
976 // ~0.5%-2.5% as optimal; reward in-band, partial when present but thin.
977 $density = $this->keyword_density($content, $target_keyword);
978 if ($density >= 0.5 && $density <= 2.5) {
979 $score += 4;
980 } elseif ($density > 0) {
981 $score += 2;
982 $suggestions[] = $density > 2.5
983 ? 'Keyword density is high - reduce repetition to avoid over-optimization'
984 : 'Keyword density is low - use the focus keyword a little more often';
985 } else {
986 $suggestions[] = "Use the focus keyword '{$target_keyword}' in the content";
987 }
988
989 // URL optimization (3 points): keyword in slug (2) + a reasonably short
990 // URL (1). Rank Math flags overly long URLs, so reward concise slugs.
991 $keyword_slug = str_replace(' ', '-', strtolower($target_keyword));
992 $keyword_in_slug = $slug !== '' && (strpos($slug, $keyword_slug) !== false || strpos(str_replace('-', '', $slug), str_replace('-', '', $keyword_slug)) !== false);
993 if ($keyword_in_slug) {
994 $score += 2;
995 } else {
996 $suggestions[] = 'Include the focus keyword in the URL slug';
997 }
998
999 // A slug under ~75 chars keeps the URL clean and fully visible in SERPs.
1000 $slug_length = strlen($slug);
1001 if ($slug === '' || $slug_length <= 75) {
1002 $score++;
1003 } else {
1004 $suggestions[] = 'Shorten the URL slug - long URLs are harder to read and share';
1005 }
1006
1007 // Keyword in image alt text (2 points) - mirrors Rank Math's
1008 // "keyword in image alt" check. When the post has no images the check
1009 // does not apply, so award the points (benefit of the doubt) rather
1010 // than capping the ceiling for legitimately image-less posts.
1011 if (empty($images)) {
1012 $score += 2;
1013 $suggestions[] = 'Add a relevant image with the focus keyword in its alt text';
1014 } elseif ($this->keyword_in_alt($images, $target_keyword)) {
1015 $score += 2;
1016 } else {
1017 $suggestions[] = 'Include the focus keyword in at least one image alt attribute';
1018 }
1019
1020 return [
1021 'score' => $score,
1022 'max_score' => $max_score,
1023 'suggestions' => $suggestions,
1024 'details' => [
1025 'keyword_in_subheading' => $this->keyword_in_subheadings($headings, $target_keyword),
1026 'keyword_density' => round($density, 2),
1027 'keyword_in_slug' => $keyword_in_slug,
1028 'slug_length' => $slug_length,
1029 'keyword_in_image_alt' => $this->keyword_in_alt($images, $target_keyword),
1030 ],
1031 ];
1032 }
1033
1034 private function score_searcher_engagement(array $content_data): array {
1035 $max_score = $this->scoring_factors['searcher_engagement'];
1036
1037 // Engagement (dwell time / bounce) is off-page, so estimate it from the
1038 // on-page signals that drive it: readability (half) + content structure
1039 // (half). This raises the old readability-only floor.
1040 $readability = (float) ($content_data['readability_score'] ?? 50);
1041 $structure = (float) $this->derive_content_quality_score($content_data);
1042
1043 $readability_pts = ($readability / 100) * ($max_score / 2);
1044 $structure_pts = ($structure / 100) * ($max_score / 2);
1045 $score = (int) round($readability_pts + $structure_pts);
1046
1047 return [
1048 'score' => $score,
1049 'max_score' => $max_score,
1050 'suggestions' => $score < ($max_score * 0.7)
1051 ? ['Improve readability and structure (shorter sentences, subheadings, shorter paragraphs)']
1052 : [],
1053 'details' => ['engagement_estimate' => round(($score / $max_score) * 100, 1) . '%'],
1054 ];
1055 }
1056
1057 private function score_backlink_authority(array $content_data): array {
1058 $max_score = $this->scoring_factors['backlink_authority'];
1059 $external_links = (int) ($content_data['external_links'] ?? 0);
1060 $dofollow_links = (int) ($content_data['external_dofollow_links'] ?? 0);
1061
1062 // Backlinks are off-page and cannot be read from post content. We use
1063 // the only on-page proxy available - whether the content cites external
1064 // sources - and avoid hard-penalizing posts for something outside the
1065 // editor's control (the old external_links*3 formula needed 5 outbound
1066 // links just to reach full marks, dragging nearly every post down).
1067 // Dofollow links pass equity, so they earn full credit (Rank Math's
1068 // "external dofollow link" check); nofollow-only citations earn less.
1069 $suggestions = [];
1070 if ($dofollow_links >= 2) {
1071 $score = $max_score;
1072 } elseif ($dofollow_links === 1) {
1073 $score = (int) round($max_score * 0.85);
1074 } elseif ($external_links > 0) {
1075 // Cites sources but every external link is nofollow.
1076 $score = (int) round($max_score * 0.75);
1077 $suggestions[] = 'Add at least one dofollow link to an authoritative external source';
1078 } else {
1079 $score = (int) round($max_score * 0.6);
1080 $suggestions[] = 'Cite authoritative external sources, and build quality backlinks to this page';
1081 }
1082
1083 return [
1084 'score' => $score,
1085 'max_score' => $max_score,
1086 'suggestions' => $suggestions,
1087 'details' => [
1088 'external_links' => $external_links,
1089 'external_dofollow_links' => $dofollow_links,
1090 ],
1091 ];
1092 }
1093 private function score_trustworthiness(array $content_data, array $metadata): array {
1094 // E-E-A-T is an off-page/site-wide signal we cannot reliably measure
1095 // from a single post. Award full credit (benefit of the doubt) rather
1096 // than a fixed partial that silently caps every post's ceiling.
1097 return [
1098 'score' => $this->scoring_factors['trustworthiness'],
1099 'max_score' => $this->scoring_factors['trustworthiness'],
1100 'suggestions' => ['Add author credentials, citations, and contact information to reinforce trustworthiness'],
1101 'details' => ['trust_level' => 'Assumed adequate'],
1102 ];
1103 }
1104
1105 private function score_link_diversity(array $content_data): array {
1106 // Off-page link distribution; not measurable per post. Full credit.
1107 return [
1108 'score' => $this->scoring_factors['link_diversity'],
1109 'max_score' => $this->scoring_factors['link_diversity'],
1110 'suggestions' => [],
1111 'details' => ['diversity_level' => 'Assumed adequate'],
1112 ];
1113 }
1114 private function score_site_security(array $content_data): array {
1115 $max_score = $this->scoring_factors['site_security'];
1116 $ssl = function_exists('is_ssl') ? is_ssl() : true;
1117
1118 return [
1119 'score' => $ssl ? $max_score : 0,
1120 'max_score' => $max_score,
1121 'suggestions' => $ssl ? [] : ['Serve the site over HTTPS (install an SSL certificate)'],
1122 'details' => ['ssl_enabled' => $ssl, 'security_level' => $ssl ? 'Good' : 'Insecure'],
1123 ];
1124 }
1125 private function check_semantic_keyword_match(string $text, string $keyword): bool {
1126 // Simple semantic matching - can be enhanced with AI/NLP
1127 $keyword_parts = explode(' ', strtolower($keyword));
1128 $text_lower = strtolower($text);
1129
1130 $matches = 0;
1131 foreach ($keyword_parts as $part) {
1132 if (strpos($text_lower, $part) !== false) {
1133 $matches++;
1134 }
1135 }
1136
1137 // Consider it a semantic match if 70% of keyword parts are present
1138 return ($matches / count($keyword_parts)) >= 0.7;
1139 }
1140 private function prioritize_suggestions(array $suggestions, array $scores = []): array {
1141 // Map each suggestion back to the factor that emitted it, so priority
1142 // can rank by the points the factor actually lost instead of keyword-
1143 // matching the advice text — which sorted a 2-point title tweak above
1144 // a 6-point thin-content loss and contradicted the row's own impact
1145 // tag (#408).
1146 $by_text = [];
1147 foreach ($scores as $factor => $result) {
1148 if (!is_array($result) || empty($result['suggestions']) || !is_array($result['suggestions'])) {
1149 continue;
1150 }
1151 $lost = max(0, (float) ($result['max_score'] ?? 0) - (float) ($result['score'] ?? 0));
1152 foreach ($result['suggestions'] as $text) {
1153 if (is_string($text) && !isset($by_text[$text])) {
1154 $by_text[$text] = ['factor' => (string) $factor, 'lost' => $lost];
1155 }
1156 }
1157 }
1158
1159 $prioritized = [];
1160
1161 foreach ($suggestions as $suggestion) {
1162 $origin = $by_text[$suggestion] ?? null;
1163
1164 // A factor already at full marks loses nothing to this advice —
1165 // it was occupying list positions (sometimes at "High") while
1166 // recovering zero points. Dropped rather than sorted last.
1167 if (null !== $origin && $origin['lost'] <= 0) {
1168 continue;
1169 }
1170
1171 if (null !== $origin) {
1172 $priority = $origin['lost'] >= 4 ? 'High' : ($origin['lost'] >= 2 ? 'Medium' : 'Low');
1173 } else {
1174 // No factor attached (defensive: a filter-added or legacy
1175 // suggestion) — the old keyword map is the fallback.
1176 $priority = $this->determine_suggestion_priority($suggestion);
1177 }
1178
1179 $prioritized[] = [
1180 'text' => $suggestion,
1181 'priority' => $priority,
1182 'impact' => $this->estimate_impact($suggestion),
1183 'effort' => $this->estimate_effort($suggestion),
1184 'factor' => $origin['factor'] ?? null,
1185 'points_recoverable' => $origin['lost'] ?? null,
1186 ];
1187 }
1188
1189 // Biggest recoverable loss first; keyword-mapped stragglers (no
1190 // factor) sort within their priority band after the measured rows.
1191 usort($prioritized, function($a, $b) {
1192 $al = $a['points_recoverable'] ?? -1;
1193 $bl = $b['points_recoverable'] ?? -1;
1194 if ($al !== $bl) {
1195 return $bl <=> $al;
1196 }
1197 $priority_order = ['High' => 3, 'Medium' => 2, 'Low' => 1];
1198 return $priority_order[$b['priority']] - $priority_order[$a['priority']];
1199 });
1200
1201 return $prioritized;
1202 }
1203
1204 /**
1205 * Determine suggestion priority based on content
1206 *
1207 * @param string $suggestion Suggestion text
1208 * @return string Priority level
1209 */
1210 private function determine_suggestion_priority(string $suggestion): string {
1211 $high_priority_keywords = ['title', 'keyword', 'content quality', 'heading'];
1212 $medium_priority_keywords = ['meta description', 'internal link', 'readability'];
1213
1214 $suggestion_lower = strtolower($suggestion);
1215
1216 foreach ($high_priority_keywords as $keyword) {
1217 if (strpos($suggestion_lower, $keyword) !== false) {
1218 return 'High';
1219 }
1220 }
1221
1222 foreach ($medium_priority_keywords as $keyword) {
1223 if (strpos($suggestion_lower, $keyword) !== false) {
1224 return 'Medium';
1225 }
1226 }
1227
1228 return 'Low';
1229 }
1230
1231 /**
1232 * Estimate impact of implementing suggestion
1233 *
1234 * @param string $suggestion Suggestion text
1235 * @return string Impact level
1236 */
1237 private function estimate_impact(string $suggestion): string {
1238 // Simple heuristic - can be enhanced with ML
1239 if (strpos(strtolower($suggestion), 'title') !== false) { return 'High';
1240 }
1241 if (strpos(strtolower($suggestion), 'content') !== false) { return 'High';
1242 }
1243 if (strpos(strtolower($suggestion), 'keyword') !== false) { return 'Medium';
1244 }
1245 return 'Low';
1246 }
1247
1248 /**
1249 * Estimate effort required to implement suggestion
1250 *
1251 * @param string $suggestion Suggestion text
1252 * @return string Effort level
1253 */
1254 private function estimate_effort(string $suggestion): string {
1255 // Simple heuristic - can be enhanced with ML
1256 if (strpos(strtolower($suggestion), 'rewrite') !== false) { return 'High';
1257 }
1258 if (strpos(strtolower($suggestion), 'add') !== false) { return 'Medium';
1259 }
1260 if (strpos(strtolower($suggestion), 'optimize') !== false) { return 'Medium';
1261 }
1262 return 'Low';
1263 }
1264 private function calculate_topic_relevance(string $content, string $target_keyword): float {
1265 if (empty($content) || empty($target_keyword)) {
1266 return 0.0;
1267 }
1268
1269 $content_lower = strtolower(wp_strip_all_tags($content));
1270 $keyword_lower = strtolower($target_keyword);
1271
1272 // Calculate keyword and semantic term frequency
1273 $keyword_count = substr_count($content_lower, $keyword_lower);
1274 $word_count = $this->calculate_word_count_js_style($content_lower);
1275
1276 if ($word_count === 0) {
1277 return 0.0;
1278 }
1279
1280 // Base relevance from keyword presence
1281 $keyword_density = ($keyword_count / $word_count) * 100;
1282 $base_relevance = min(1.0, $keyword_density / 2.0); // Optimal around 1-2%
1283
1284 // Boost for semantic variations
1285 $semantic_boost = $this->calculate_semantic_boost($content_lower, $keyword_lower);
1286
1287 return min(1.0, $base_relevance + $semantic_boost);
1288 }
1289
1290 /**
1291 * Calculate semantic boost for related terms
1292 *
1293 * @param string $content Content text (lowercase)
1294 * @param string $keyword Target keyword (lowercase)
1295 * @return float Semantic boost (0-0.3)
1296 */
1297 private function calculate_semantic_boost(string $content, string $keyword): float {
1298 // Simple semantic term detection - can be enhanced with NLP
1299 $semantic_terms = $this->get_semantic_terms($keyword);
1300 $boost = 0.0;
1301
1302 foreach ($semantic_terms as $term) {
1303 if (strpos($content, $term) !== false) {
1304 $boost += 0.05; // Small boost per semantic term
1305 }
1306 }
1307
1308 return min(0.3, $boost); // Cap at 30% boost
1309 }
1310
1311 /**
1312 * Get semantic terms for a keyword
1313 *
1314 * @param string $keyword Target keyword
1315 * @return array Semantic terms
1316 */
1317 private function get_semantic_terms(string $keyword): array {
1318 // Simple semantic term generation - can be enhanced with AI/NLP
1319 $terms = [];
1320
1321 // Add plural/singular variations
1322 if (substr($keyword, -1) === 's') {
1323 $terms[] = rtrim($keyword, 's');
1324 } else {
1325 $terms[] = $keyword . 's';
1326 }
1327
1328 // Add common related terms based on keyword
1329 $keyword_lower = strtolower($keyword);
1330
1331 // SEO-related terms
1332 if (strpos($keyword_lower, 'seo') !== false) {
1333 $terms = array_merge($terms, ['optimization', 'search engine', 'ranking', 'visibility']);
1334 }
1335
1336 // WordPress-related terms
1337 if (strpos($keyword_lower, 'wordpress') !== false) { // phpcs:ignore WordPress.WP.CapitalPDangit.MisspelledInText -- lowercase on purpose: the haystack is strtolower()ed.
1338 $terms = array_merge($terms, ['wp', 'plugin', 'theme', 'cms']);
1339 }
1340
1341 return $terms;
1342 }
1343
1344 /**
1345 * Keyword density (%) of the target keyword across the plain-text body.
1346 *
1347 * @param string $content Raw/HTML content.
1348 * @param string $target_keyword Target keyword.
1349 * @return float Density percentage (0 when no keyword/content).
1350 */
1351 private function keyword_density(string $content, string $target_keyword): float {
1352 if (empty($content) || empty($target_keyword)) {
1353 return 0.0;
1354 }
1355 $plain = strtolower(wp_strip_all_tags($content));
1356 $word_count = $this->calculate_word_count_js_style($plain);
1357 if ($word_count === 0) {
1358 return 0.0;
1359 }
1360 $occurrences = substr_count($plain, strtolower($target_keyword));
1361 return ($occurrences / $word_count) * 100;
1362 }
1363
1364 /**
1365 * Whether the target keyword (or a semantic variation) appears in the body.
1366 *
1367 * @param string $content Raw/HTML content.
1368 * @param string $target_keyword Target keyword.
1369 * @return bool
1370 */
1371 private function keyword_in_content(string $content, string $target_keyword): bool {
1372 if (empty($content) || empty($target_keyword)) {
1373 return false;
1374 }
1375 $plain = strtolower(wp_strip_all_tags($content));
1376 if (strpos($plain, strtolower($target_keyword)) !== false) {
1377 return true;
1378 }
1379 return $this->check_semantic_keyword_match($plain, $target_keyword);
1380 }
1381
1382 /**
1383 * Whether the keyword appears early (first paragraph / first ~10% of words).
1384 *
1385 * @param string $content Raw/HTML content.
1386 * @param string $target_keyword Target keyword.
1387 * @return bool
1388 */
1389 private function keyword_in_first_paragraph(string $content, string $target_keyword): bool {
1390 if (empty($content) || empty($target_keyword)) {
1391 return false;
1392 }
1393 $plain = strtolower(wp_strip_all_tags($content));
1394 $words = preg_split('/\s+/', trim($plain), -1, PREG_SPLIT_NO_EMPTY) ?: [];
1395 $window = array_slice($words, 0, max(50, (int) ceil(count($words) * 0.1)));
1396 return strpos(implode(' ', $window), strtolower($target_keyword)) !== false;
1397 }
1398
1399 /**
1400 * Whether the keyword appears in any subheading (H2–H6).
1401 *
1402 * @param array $headings Extracted headings (each with 'level' + 'text').
1403 * @param string $target_keyword Target keyword.
1404 * @return bool
1405 */
1406 private function keyword_in_subheadings(array $headings, string $target_keyword): bool {
1407 if (empty($target_keyword)) {
1408 return false;
1409 }
1410 $keyword_lower = strtolower($target_keyword);
1411 foreach ($headings as $heading) {
1412 if ((int) ($heading['level'] ?? 0) < 2) {
1413 continue;
1414 }
1415 $text = strtolower((string) ($heading['text'] ?? ''));
1416 if ($text !== '' && strpos($text, $keyword_lower) !== false) {
1417 return true;
1418 }
1419 }
1420 return false;
1421 }
1422
1423 /**
1424 * Days elapsed since a MySQL datetime string, or null when unparseable.
1425 *
1426 * @param string $datetime MySQL datetime (e.g. post_modified).
1427 * @return int|null
1428 */
1429 private function days_since(string $datetime): ?int {
1430 $datetime = trim($datetime);
1431 if ($datetime === '' || strpos($datetime, '0000-00-00') === 0) {
1432 return null;
1433 }
1434 $ts = strtotime($datetime);
1435 if ($ts === false) {
1436 return null;
1437 }
1438 // strtotime() returns a real unix timestamp, so this must compare against
1439 // one: current_time('timestamp') is offset by the site timezone and made
1440 // every "days ago" figure wrong by that offset.
1441 $now = time();
1442 return (int) floor(($now - $ts) / 86400);
1443 }
1444
1445 /**
1446 * Whether the keyword appears in the meta description.
1447 *
1448 * @param string $meta_description Meta description text.
1449 * @param string $target_keyword Target keyword.
1450 * @return bool
1451 */
1452 private function keyword_in_meta(string $meta_description, string $target_keyword): bool {
1453 if ($meta_description === '' || $target_keyword === '') {
1454 return false;
1455 }
1456 return strpos(strtolower($meta_description), strtolower($target_keyword)) !== false;
1457 }
1458
1459 /**
1460 * Whether the keyword appears in any image alt text.
1461 *
1462 * @param array $images Images (each with an 'alt' key).
1463 * @param string $target_keyword Target keyword.
1464 * @return bool
1465 */
1466 private function keyword_in_alt(array $images, string $target_keyword): bool {
1467 if ($target_keyword === '') {
1468 return false;
1469 }
1470 $keyword_lower = strtolower($target_keyword);
1471 foreach ($images as $image) {
1472 $alt = strtolower((string) ($image['alt'] ?? ''));
1473 if ($alt !== '' && strpos($alt, $keyword_lower) !== false) {
1474 return true;
1475 }
1476 }
1477 return false;
1478 }
1479
1480 /**
1481 * Whether the title contains a common power word (CTR booster).
1482 *
1483 * @param string $title Post title.
1484 * @return bool
1485 */
1486 private function title_has_power_word(string $title): bool {
1487 $title_lower = strtolower($title);
1488 foreach (self::get_title_power_words() as $word) {
1489 if (strpos($title_lower, $word) !== false) {
1490 return true;
1491 }
1492 }
1493 return false;
1494 }
1495
1496 /**
1497 * The power words rewarded by the title check. Single source of truth so the
1498 * AI title improver can require the generated title to actually contain one.
1499 *
1500 * @return string[]
1501 */
1502 public static function get_title_power_words(): array {
1503 return [
1504 'ultimate', 'essential', 'complete', 'proven', 'guide', 'best', 'top',
1505 'free', 'easy', 'simple', 'quick', 'fast', 'powerful', 'secret', 'expert',
1506 'effective', 'amazing', 'incredible', 'exclusive', 'definitive', 'step-by-step',
1507 ];
1508 }
1509
1510 /**
1511 * The emotion/sentiment words rewarded by the title check. Single source of
1512 * truth shared with the AI title improver so a generated title satisfies the
1513 * same validation.
1514 *
1515 * @return string[]
1516 */
1517 public static function get_title_sentiment_words(): array {
1518 return [
1519 // Positive
1520 'great', 'good', 'better', 'awesome', 'love', 'win', 'boost', 'improve',
1521 'success', 'smart', 'brilliant', 'perfect', 'happy', 'beautiful',
1522 // Negative (drives clicks too)
1523 'avoid', 'mistake', 'worst', 'stop', 'never', 'bad', 'wrong', 'fail',
1524 'danger', 'warning', 'painful', 'ugly',
1525 ];
1526 }
1527
1528 /**
1529 * Whether the title carries an emotional/sentiment word (positive or
1530 * negative), which Rank Math rewards for higher engagement.
1531 *
1532 * @param string $title Post title.
1533 * @return bool
1534 */
1535 private function title_has_sentiment_word(string $title): bool {
1536 $title_lower = strtolower($title);
1537 foreach (self::get_title_sentiment_words() as $word) {
1538 if (strpos($title_lower, $word) !== false) {
1539 return true;
1540 }
1541 }
1542 return false;
1543 }
1544
1545 /**
1546 * Assess content depth based on word count
1547 *
1548 * @param int $word_count Word count
1549 * @return string Depth assessment
1550 */
1551 private function assess_content_depth(int $word_count): string {
1552 if ($word_count >= 2000) { return 'Comprehensive';
1553 }
1554 if ($word_count >= 1000) { return 'Detailed';
1555 }
1556 if ($word_count >= 500) { return 'Moderate';
1557 }
1558 if ($word_count >= 300) { return 'Basic';
1559 }
1560 return 'Insufficient';
1561 }
1562 private function score_content_freshness(array $content_data): array {
1563 $max_score = $this->scoring_factors['content_freshness'];
1564 $suggestions = [];
1565
1566 // Derive freshness from the real last-modified date when available
1567 // (live editing has no stored date yet -> treat as fresh).
1568 $days = $this->days_since((string) ($content_data['post_modified'] ?? ''));
1569
1570 if ($days === null || $days <= 180) {
1571 $score = $max_score;
1572 $status = 'Current';
1573 } elseif ($days <= 365) {
1574 $score = (int) round($max_score * 0.66);
1575 $status = 'Aging';
1576 $suggestions[] = 'Content is 6-12 months old - review and refresh it for better freshness signals';
1577 } else {
1578 $score = (int) round($max_score * 0.33);
1579 $status = 'Stale';
1580 $suggestions[] = 'Content is over a year old - update it to maintain freshness signals';
1581 }
1582
1583 return [
1584 'score' => $score,
1585 'max_score' => $max_score,
1586 'suggestions' => $suggestions,
1587 'details' => [
1588 'freshness_status' => $status,
1589 'days_since_modified' => $days,
1590 ],
1591 ];
1592 }
1593 /**
1594 * Resolve a post's content into something worth analyzing.
1595 *
1596 * Delegates to Builder_Content, which knows where each page builder keeps
1597 * its text. Kept as the historical entry point for existing callers.
1598 *
1599 * @since 1.23.0
1600 *
1601 * @param \WP_Post $post Post being analyzed.
1602 * @return string Content to analyze.
1603 */
1604 public static function resolve_analyzable_content(\WP_Post $post): string {
1605 if (!class_exists('\ThinkRank\SEO\Builder_Content')) {
1606 require_once THINKRANK_PLUGIN_DIR . 'includes/seo/class-builder-content.php';
1607 }
1608
1609 return \ThinkRank\SEO\Builder_Content::resolve($post);
1610 }
1611
1612 /**
1613 * Resolve editor-supplied live content into something worth analyzing.
1614 *
1615 * @since 1.23.0
1616 *
1617 * @param string $live_content Markup supplied by the editor.
1618 * @param \WP_Post $post Post the markup belongs to.
1619 * @return string Content to analyze.
1620 */
1621 public static function resolve_live_content(string $live_content, \WP_Post $post): string {
1622 if (!class_exists('\ThinkRank\SEO\Builder_Content')) {
1623 require_once THINKRANK_PLUGIN_DIR . 'includes/seo/class-builder-content.php';
1624 }
1625
1626 return \ThinkRank\SEO\Builder_Content::resolve_markup($live_content, $post);
1627 }
1628
1629 public function analyze_post_content(int $post_id): array {
1630 $post = get_post($post_id);
1631 if (!$post) {
1632 return [];
1633 }
1634
1635 $content = self::resolve_analyzable_content($post);
1636 $title = $post->post_title;
1637
1638 // Extract headings from content
1639 $headings = $this->extract_headings($content);
1640
1641 // Count words using JavaScript-compatible method
1642 $plain_text = wp_strip_all_tags($content);
1643 $word_count = $this->calculate_word_count_js_style($plain_text);
1644
1645 // Calculate readability
1646 $readability_score = $this->calculate_readability_score($content);
1647
1648 // Count links
1649 $internal_links = $this->count_internal_links($content);
1650 $external_links = $this->count_external_links($content);
1651 $external_dofollow_links = $this->count_external_dofollow_links($content);
1652
1653 // Analyze images
1654 $images = $this->analyze_images($content);
1655
1656 // Get URL
1657 $url = get_permalink($post_id);
1658
1659 return [
1660 'content' => $content,
1661 'title' => $title,
1662 'headings' => $headings,
1663 'word_count' => $word_count,
1664 'readability_score' => $readability_score,
1665 'internal_links' => $internal_links,
1666 'external_links' => $external_links,
1667 'external_dofollow_links' => $external_dofollow_links,
1668 'images' => $images,
1669 'url' => $url,
1670 'slug' => $post->post_name,
1671 'post_modified' => $post->post_modified,
1672 'schema_present' => $this->detect_schema_present($content)
1673 || $this->thinkrank_global_schema_active($post->post_type)
1674 || $this->thinkrank_deployed_schema_active($post),
1675 ];
1676 }
1677
1678 /**
1679 * Build the human-readable readability label (mirrors the editor's
1680 * calculateReadabilityScore: "<level> (<flesch>)") from analyzed content.
1681 *
1682 * @param array $content_data Output of analyze_post_content()/analyze_live_content()
1683 * @return string Readability label, e.g. "Standard (62)"
1684 */
1685 private function format_readability_label(array $content_data): string {
1686 if ((int) ($content_data['word_count'] ?? 0) === 0) {
1687 return 'No content';
1688 }
1689
1690 $rounded = (int) round((float) ($content_data['readability_score'] ?? 0));
1691
1692 if ($rounded >= 90) {
1693 $level = 'Very Easy';
1694 } elseif ($rounded >= 80) {
1695 $level = 'Easy';
1696 } elseif ($rounded >= 70) {
1697 $level = 'Fairly Easy';
1698 } elseif ($rounded >= 60) {
1699 $level = 'Standard';
1700 } elseif ($rounded >= 50) {
1701 $level = 'Fairly Difficult';
1702 } elseif ($rounded >= 30) {
1703 $level = 'Difficult';
1704 } else {
1705 $level = 'Very Difficult';
1706 }
1707
1708 return "{$level} ({$rounded})";
1709 }
1710
1711 /**
1712 * Derive the content-quality label (mirrors the editor's
1713 * calculateContentQuality: word count + long-paragraph + subheading scoring)
1714 * so persisted scores carry a non-null quality value.
1715 *
1716 * @param array $content_data Output of analyze_post_content()/analyze_live_content()
1717 * @return string One of: No content, Good, OK, Needs improvement
1718 */
1719 private function derive_content_quality(array $content_data): string {
1720 $word_count = (int) ($content_data['word_count'] ?? 0);
1721 if ($word_count === 0) {
1722 return 'No content';
1723 }
1724
1725 $final = $this->derive_content_quality_score($content_data);
1726
1727 if ($final >= 80) {
1728 return 'Good';
1729 }
1730 if ($final >= 50) {
1731 return 'OK';
1732 }
1733
1734 return 'Needs improvement';
1735 }
1736
1737 /**
1738 * Numeric content-quality score (0-100): word count + long-paragraph +
1739 * subheading distribution. Shared by the quality label and the
1740 * satisfying-content factor so both stay in sync.
1741 *
1742 * @param array $content_data Output of analyze_post_content()/analyze_live_content()
1743 * @return int Quality score 0-100.
1744 */
1745 private function derive_content_quality_score(array $content_data): int {
1746 $word_count = (int) ($content_data['word_count'] ?? 0);
1747 if ($word_count === 0) {
1748 return 0;
1749 }
1750
1751 $content = (string) ($content_data['content'] ?? '');
1752 $score = 0;
1753
1754 // 1. Word count (industry standard: 300+ words).
1755 if ($word_count >= 600) {
1756 $score += 100;
1757 } elseif ($word_count >= 300) {
1758 $score += 50;
1759 }
1760
1761 // 2. Long paragraphs (flag paragraphs over 150 words).
1762 $long_paragraphs = 0;
1763 if (preg_match_all('/<p[^>]*>(.*?)<\/p>/is', $content, $matches)) {
1764 foreach ($matches[1] as $paragraph) {
1765 if ($this->calculate_word_count_js_style(wp_strip_all_tags($paragraph)) > 150) {
1766 $long_paragraphs++;
1767 }
1768 }
1769 }
1770 if ($long_paragraphs === 0) {
1771 $score += 100;
1772 } elseif ($long_paragraphs <= 2) {
1773 $score += 50;
1774 }
1775
1776 // 3. Subheading distribution (H2–H6, ~one per 300 words).
1777 $subheadings = 0;
1778 foreach ((array) ($content_data['headings'] ?? []) as $heading) {
1779 if ((int) ($heading['level'] ?? 0) >= 2) {
1780 $subheadings++;
1781 }
1782 }
1783 $expected = (int) floor($word_count / 300);
1784 if ($subheadings > 0 && $subheadings >= $expected) {
1785 $score += 100;
1786 } elseif ($subheadings > 0) {
1787 $score += 50;
1788 }
1789
1790 return (int) round($score / 3);
1791 }
1792
1793 /**
1794 * Extract headings from content
1795 *
1796 * @param string $content Content HTML
1797 * @return array Array of headings with levels
1798 */
1799 private function extract_headings(string $content): array {
1800 $headings = [];
1801
1802 // Match H1-H6 tags
1803 if (preg_match_all('/<h([1-6])[^>]*>(.*?)<\/h[1-6]>/i', $content, $matches, PREG_SET_ORDER)) {
1804 foreach ($matches as $match) {
1805 $headings[] = [
1806 'level' => (int)$match[1],
1807 'text' => wp_strip_all_tags($match[2]),
1808 ];
1809 }
1810 }
1811
1812 return $headings;
1813 }
1814
1815 /**
1816 * Calculate readability score using Flesch Reading Ease
1817 *
1818 * @param string $content Content text
1819 * @return float Readability score
1820 */
1821 private function calculate_readability_score(string $content): float {
1822 $text = wp_strip_all_tags($content);
1823
1824 if (empty($text)) {
1825 return 0;
1826 }
1827
1828 // Count sentences (approximate)
1829 $sentences = preg_split('/[.!?]+/', $text, -1, PREG_SPLIT_NO_EMPTY);
1830 $sentence_count = count($sentences);
1831
1832 // Count words
1833 $word_count = $this->calculate_word_count_js_style(wp_strip_all_tags($text));
1834
1835 // Count syllables (approximate)
1836 $syllable_count = $this->count_syllables($text);
1837
1838 if ($sentence_count === 0 || $word_count === 0) {
1839 return 0;
1840 }
1841
1842 // Flesch Reading Ease formula
1843 $score = 206.835 - (1.015 * ($word_count / $sentence_count)) - (84.6 * ($syllable_count / $word_count));
1844
1845 return max(0, min(100, $score));
1846 }
1847
1848 /**
1849 * Count syllables in text (approximate)
1850 *
1851 * @param string $text Text to analyze
1852 * @return int Syllable count
1853 */
1854 private function count_syllables(string $text): int {
1855 $words = preg_split('/\s+/', trim(strtolower(wp_strip_all_tags($text))), -1, PREG_SPLIT_NO_EMPTY);
1856 $syllables = 0;
1857
1858 foreach ($words as $word) {
1859 $word = preg_replace('/[^a-z]/', '', $word);
1860 if ($word === '') {
1861 continue;
1862 }
1863
1864 $groups = preg_match_all('/[aeiouy]+/', $word);
1865
1866 // Standard Flesch heuristic: a trailing silent e does not form a
1867 // syllable ("make", "time", "these") — but only when a consonant
1868 // precedes it (a vowel+e ending like "movie" already shares its
1869 // group) and never for consonant-le ("table"), which does count.
1870 // Without this the counter inflated syllables/word by ~0.2-0.3 on
1871 // ordinary prose, driving raw Flesch negative and the UI to a
1872 // clamped "Very Difficult (0)" (#407).
1873 if ($groups > 1 && preg_match('/[^aeiouy]e$/', $word) && !str_ends_with($word, 'le')) {
1874 $groups--;
1875 }
1876
1877 $syllables += max(1, $groups);
1878 }
1879
1880 return $syllables;
1881 }
1882
1883 /**
1884 * Count internal links in content
1885 *
1886 * @param string $content Content HTML
1887 * @return int Internal link count
1888 */
1889 private function count_internal_links(string $content): int {
1890 $site_url = get_site_url();
1891 $count = 0;
1892
1893 if (preg_match_all('/<a[^>]+href=["\']([^"\']+)["\'][^>]*>/i', $content, $matches)) {
1894 foreach ($matches[1] as $url) {
1895 if (strpos($url, $site_url) !== false || strpos($url, '/') === 0) {
1896 $count++;
1897 }
1898 }
1899 }
1900
1901 return $count;
1902 }
1903
1904 /**
1905 * Count external links in content
1906 *
1907 * @param string $content Content HTML
1908 * @return int External link count
1909 */
1910 private function count_external_links(string $content): int {
1911 $site_url = get_site_url();
1912 $count = 0;
1913
1914 if (preg_match_all('/<a[^>]+href=["\']([^"\']+)["\'][^>]*>/i', $content, $matches)) {
1915 foreach ($matches[1] as $url) {
1916 if (strpos($url, 'http') === 0 && strpos($url, $site_url) === false) {
1917 $count++;
1918 }
1919 }
1920 }
1921
1922 return $count;
1923 }
1924
1925 /**
1926 * Count external links that pass link equity (not rel="nofollow").
1927 * Mirrors Rank Math's "external dofollow link" check.
1928 *
1929 * @param string $content Content HTML
1930 * @return int External dofollow link count
1931 */
1932 private function count_external_dofollow_links(string $content): int {
1933 $site_url = get_site_url();
1934 $count = 0;
1935
1936 if (preg_match_all('/<a\b[^>]*>/i', $content, $matches)) {
1937 foreach ($matches[0] as $tag) {
1938 if (!preg_match('/href=["\']([^"\']+)["\']/i', $tag, $href)) {
1939 continue;
1940 }
1941 $url = $href[1];
1942 $is_external = strpos($url, 'http') === 0 && strpos($url, $site_url) === false;
1943 if (!$is_external) {
1944 continue;
1945 }
1946 if (preg_match('/rel=["\'][^"\']*\bnofollow\b[^"\']*["\']/i', $tag)) {
1947 continue;
1948 }
1949 $count++;
1950 }
1951 }
1952
1953 return $count;
1954 }
1955
1956 /**
1957 * Analyze images in content
1958 *
1959 * @param string $content Content HTML
1960 * @return array Image analysis data
1961 */
1962 /**
1963 * Detect structured data embedded directly in the content (JSON-LD script
1964 * blocks or microdata attributes). Site-wide schema injected at render time
1965 * is a separate feature and intentionally out of scope here.
1966 *
1967 * @param string $content Raw/HTML content.
1968 * @return bool
1969 */
1970 private function detect_schema_present(string $content): bool {
1971 if ($content === '') {
1972 return false;
1973 }
1974 return stripos($content, 'application/ld+json') !== false
1975 || stripos($content, 'itemscope') !== false
1976 || stripos($content, 'itemtype') !== false;
1977 }
1978
1979 /**
1980 * Whether ThinkRank's Global SEO schema output is active for a post type.
1981 *
1982 * ThinkRank injects JSON-LD at render time (wp_head) when a schema type is
1983 * configured for the post type, so a post can have valid structured data
1984 * even when none is embedded in the post body. The score credits this so the
1985 * "add structured data" suggestion reflects ThinkRank's own schema engine.
1986 *
1987 * @param string $post_type Post type slug.
1988 * @return bool
1989 */
1990 private function thinkrank_global_schema_active(string $post_type): bool {
1991 if ($post_type === '') {
1992 return false;
1993 }
1994 $all_settings = get_option('thinkrank_global_seo_settings', []);
1995 return !empty($all_settings[$post_type]['schema_type']);
1996 }
1997
1998 /**
1999 * Whether the Schema Manager has an active deployed schema for this post.
2000 *
2001 * Per-post schema deployed from the editor's Schema tab is stored in the
2002 * Schema Manager's own table and emitted at wp_head by
2003 * Frontend\SEO_Manager::output_site_schema_markup(). Neither
2004 * detect_schema_present() (body scan) nor thinkrank_global_schema_active()
2005 * (post-type option) sees it, so without this the score reported "no
2006 * structured data" for posts that do emit it.
2007 *
2008 * Mirrors the context_type whitelist the emitter and the metabox both use, so
2009 * the lookup targets the same row the front end reads.
2010 *
2011 * @param \WP_Post $post Post being scored.
2012 * @return bool
2013 */
2014 private function thinkrank_deployed_schema_active(\WP_Post $post): bool {
2015 if (!class_exists('ThinkRank\\SEO\\Schema_Management_System')) {
2016 $manager_file = THINKRANK_PLUGIN_DIR . 'includes/seo/class-schema-management-system.php';
2017 if (!file_exists($manager_file)) {
2018 return false;
2019 }
2020 require_once $manager_file;
2021 }
2022
2023 $context_type = in_array($post->post_type, ['site', 'post', 'page', 'product'], true)
2024 ? $post->post_type
2025 : 'post';
2026
2027 try {
2028 $manager = new \ThinkRank\SEO\Schema_Management_System();
2029 return !empty($manager->get_deployed_schemas($context_type, (int) $post->ID));
2030 } catch (\Throwable $e) {
2031 return false;
2032 }
2033 }
2034
2035 private function analyze_images(string $content): array {
2036 $images = [];
2037
2038 if (preg_match_all('/<img[^>]+>/i', $content, $matches)) {
2039 foreach ($matches[0] as $img_tag) {
2040 $alt = '';
2041 if (preg_match('/alt=["\']([^"\']*)["\']/', $img_tag, $alt_match)) {
2042 $alt = $alt_match[1];
2043 }
2044
2045 $src = '';
2046 if (preg_match('/src=["\']([^"\']*)["\']/', $img_tag, $src_match)) {
2047 $src = $src_match[1];
2048 }
2049
2050 $images[] = [
2051 'src' => $src,
2052 'alt' => $alt,
2053 ];
2054 }
2055 }
2056
2057 return $images;
2058 }
2059
2060 /**
2061 * Save SEO score to database
2062 *
2063 * @param int $post_id Post ID
2064 * @param int $user_id User ID
2065 * @param array $score_data Score data
2066 * @return int|false Score ID or false on failure
2067 */
2068 public function save_score(int $post_id, int $user_id, array $score_data) {
2069 global $wpdb;
2070
2071 $table_name = $wpdb->prefix . 'thinkrank_seo_scores';
2072
2073 // Prepare data for insertion
2074 $insert_data = [
2075 'post_id' => $post_id,
2076 'user_id' => $user_id,
2077 'overall_score' => $score_data['overall_score'],
2078 'score_breakdown' => wp_json_encode($score_data['score_breakdown']),
2079 'suggestions' => wp_json_encode($score_data['suggestions']),
2080 'grade' => $score_data['grade'],
2081 'algorithm_version' => $score_data['algorithm_version'] ?? '2024.1',
2082 'calculated_at' => $score_data['calculated_at'],
2083 'created_at' => current_time('mysql'),
2084 ];
2085
2086 $format = [
2087 '%d', '%d', '%d', '%s', '%s', '%s', '%s', '%s', '%s'
2088 ];
2089
2090 // Add readability_score if provided
2091 if (isset($score_data['readability_score'])) {
2092 $insert_data['readability_score'] = $score_data['readability_score'];
2093 $format[] = '%s';
2094 }
2095
2096 // Add content_quality if provided
2097 if (isset($score_data['content_quality'])) {
2098 $insert_data['content_quality'] = $score_data['content_quality'];
2099 $format[] = '%s';
2100 }
2101
2102 // phpcs:ignore WordPress.DB.DirectDatabaseQuery.DirectQuery,WordPress.DB.DirectDatabaseQuery.NoCaching, PluginCheck.Security.DirectDB.UnescapedDBParameter -- SEO score storage requires direct database access
2103 $result = $wpdb->insert(
2104 $table_name,
2105 $insert_data,
2106 $format
2107 );
2108
2109 return $result ? $wpdb->insert_id : false;
2110 }
2111
2112 /**
2113 * Get score history for a post
2114 *
2115 * @param int $post_id Post ID
2116 * @param int $limit Number of scores to retrieve
2117 * @return array Score history
2118 */
2119 public function get_score_history(int $post_id, int $limit = 10): array {
2120 global $wpdb;
2121
2122 // Get table name and escape it properly (table names cannot be parameterized)
2123 $table_name = esc_sql($wpdb->prefix . 'thinkrank_seo_scores');
2124
2125 // phpcs:ignore WordPress.DB.DirectDatabaseQuery.DirectQuery,WordPress.DB.DirectDatabaseQuery.NoCaching, PluginCheck.Security.DirectDB.UnescapedDBParameter -- SEO score history requires direct database access
2126 $results = $wpdb->get_results(
2127 $wpdb->prepare(
2128 // phpcs:ignore WordPress.DB.PreparedSQL.InterpolatedNotPrepared, PluginCheck.Security.DirectDB.UnescapedDBParameter -- Table name is properly escaped using esc_sql()
2129 "SELECT * FROM `{$table_name}` WHERE post_id = %d ORDER BY created_at DESC LIMIT %d",
2130 $post_id,
2131 $limit
2132 ),
2133 ARRAY_A
2134 );
2135
2136 // Decode JSON fields
2137 foreach ($results as &$result) {
2138 $result['score_breakdown'] = json_decode($result['score_breakdown'], true);
2139 $result['suggestions'] = json_decode($result['suggestions'], true);
2140 }
2141
2142 return $results ?: [];
2143 }
2144
2145 /**
2146 * Get latest score for a post
2147 *
2148 * @param int $post_id Post ID
2149 * @return array|null Latest score data
2150 */
2151 public function get_latest_score(int $post_id): ?array {
2152 $history = $this->get_score_history($post_id, 1);
2153 return !empty($history) ? $history[0] : null;
2154 }
2155
2156 /**
2157 * Score mobile experience - NEW 2025 factor (5 points)
2158 *
2159 * @param array $content_data Content analysis data
2160 * @return array Scoring result
2161 */
2162 private function score_mobile_experience(array $content_data): array {
2163 $max = $this->scoring_factors['mobile_experience'];
2164
2165 // Mobile experience is theme/site-level, not controlled by post content
2166 // — but the plugin already measures it. When a mobile Lighthouse score
2167 // has been collected, score against it; the "benefit of the doubt" below
2168 // is for sites nobody has measured, not for sites measured as slow.
2169 $performance_score = $this->measured_performance_score();
2170
2171 if ($performance_score === null) {
2172 return [
2173 'score' => $max,
2174 'max_score' => $max,
2175 'suggestions' => ['Ensure mobile-first design and fast loading on mobile devices'],
2176 'details' => ['mobile_score' => 'Assumed adequate', 'measured' => false],
2177 ];
2178 }
2179
2180 $score = (int) round($max * $performance_score / 100);
2181
2182 return [
2183 'score' => $score,
2184 'max_score' => $max,
2185 'suggestions' => $score < $max
2186 ? ['Improve mobile page speed: the last PageSpeed run scored ' . $performance_score . '/100 on mobile']
2187 : [],
2188 'details' => [
2189 'mobile_score' => $performance_score,
2190 'measured' => true,
2191 'source' => 'pagespeed_mobile',
2192 ],
2193 ];
2194 }
2195
2196 /**
2197 * The last collected mobile Lighthouse score, or null when unmeasured.
2198 *
2199 * Memoised per instance: compute_score() asks twice, and a post-list screen
2200 * scores a page of posts at a time.
2201 *
2202 * Every failure — no performance module, no collected row, an unreadable
2203 * table — resolves to null, which the callers read as "not measured" and
2204 * answer with the full-credit fallback. A site is never penalised for
2205 * ThinkRank being unable to look.
2206 *
2207 * @since 2.3.1
2208 * @return int|null Score 0-100, or null when nothing has been collected.
2209 */
2210 private function measured_performance_score(): ?int {
2211 $measurement = $this->measured_performance();
2212
2213 if ($measurement === null || !isset($measurement['performance_score'])) {
2214 return null;
2215 }
2216
2217 $score = $measurement['performance_score'];
2218
2219 if (!is_numeric($score)) {
2220 return null;
2221 }
2222
2223 return (int) round(max(0, min(100, (float) $score)));
2224 }
2225
2226 /**
2227 * The last collected mobile measurement, or null when there is none.
2228 *
2229 * @since 2.3.1
2230 * @return array|null { core_web_vitals: array, performance_score: float|null }
2231 */
2232 private function measured_performance(): ?array {
2233 if ($this->measured_performance_resolved) {
2234 return $this->measured_performance;
2235 }
2236
2237 $this->measured_performance_resolved = true;
2238
2239 if (!class_exists('ThinkRank\\SEO\\Performance_Monitoring_Manager')) {
2240 return null;
2241 }
2242
2243 try {
2244 $manager = new \ThinkRank\SEO\Performance_Monitoring_Manager();
2245 // Mobile deliberately: Google indexes mobile-first, and it is the
2246 // device the mobile_experience factor is named after.
2247 $this->measured_performance = $manager->get_stored_performance_measurement('mobile');
2248 } catch (\Throwable $e) {
2249 $this->measured_performance = null;
2250 }
2251
2252 return $this->measured_performance;
2253 }
2254
2255 /**
2256 * Score core web vitals - 2025 version (3 points)
2257 *
2258 * @param array $content_data Content analysis data
2259 * @return array Scoring result
2260 */
2261 private function score_core_web_vitals(array $content_data): array {
2262 $max = $this->scoring_factors['core_web_vitals'];
2263
2264 // Not derivable from post content — but it is measured, and the audit
2265 // stores LCP, INP and CLS with a rating each. Score against those when
2266 // they exist; fall back to the benefit of the doubt when they do not.
2267 $rated = $this->measured_vitals_score();
2268
2269 if ($rated === null) {
2270 return [
2271 'score' => $max,
2272 'max_score' => $max,
2273 'suggestions' => ['Optimize Core Web Vitals: LCP, INP, and CLS for better user experience'],
2274 'details' => ['vitals_status' => 'Assumed adequate', 'measured' => false],
2275 ];
2276 }
2277
2278 $score = (int) round($max * $rated['average'] / 100);
2279
2280 return [
2281 'score' => $score,
2282 'max_score' => $max,
2283 // Gated on the measurement, not the rounded score: two good metrics
2284 // and one needing improvement averages 88.33, which rounds to the
2285 // full 3 of 3 and used to swallow the suggestion naming the metric
2286 // that is actually failing.
2287 'suggestions' => !empty($rated['failing'])
2288 ? ['Optimize Core Web Vitals: ' . implode(', ', $rated['failing']) . ' below target on mobile']
2289 : [],
2290 'details' => [
2291 'vitals_status' => $rated['statuses'],
2292 'measured' => true,
2293 'source' => 'pagespeed_mobile',
2294 ],
2295 ];
2296 }
2297
2298 /**
2299 * Rate the collected Core Web Vitals, or null when none were measured.
2300 *
2301 * Reuses the per-metric score the performance module already assigns
2302 * (good 100, needs improvement 65, poor 30) rather than inventing a second
2303 * scale, so the SEO score and the performance card cannot disagree about
2304 * whether a metric is healthy.
2305 *
2306 * Metrics with no stored value — fcp is not always collected — are skipped
2307 * rather than counted as failures.
2308 *
2309 * @since 2.3.1
2310 * @return array|null { average: float, statuses: array, failing: string[] }
2311 */
2312 private function measured_vitals_score(): ?array {
2313 $measurement = $this->measured_performance();
2314 $vitals = $measurement['core_web_vitals'] ?? null;
2315
2316 if (!is_array($vitals)) {
2317 return null;
2318 }
2319
2320 $scores = [];
2321 $statuses = [];
2322 $failing = [];
2323
2324 // The three Google ranks on. fcp is diagnostic and not a Core Web Vital.
2325 foreach (['lcp', 'inp', 'cls'] as $metric) {
2326 $data = $vitals[$metric] ?? null;
2327
2328 if (!is_array($data) || !isset($data['value'], $data['score']) || $data['value'] === null) {
2329 continue;
2330 }
2331
2332 $scores[] = (float) $data['score'];
2333 $statuses[$metric] = $data['status'] ?? 'unknown';
2334
2335 if (($data['status'] ?? '') !== 'good') {
2336 $failing[] = strtoupper($metric);
2337 }
2338 }
2339
2340 if (empty($scores)) {
2341 return null;
2342 }
2343
2344 return [
2345 'average' => array_sum($scores) / count($scores),
2346 'statuses' => $statuses,
2347 'failing' => $failing,
2348 ];
2349 }
2350
2351 /**
2352 * Score internal linking - declining importance (1 point)
2353 *
2354 * @param array $content_data Content analysis data
2355 * @return array Scoring result
2356 */
2357 private function score_internal_linking(array $content_data): array {
2358 $internal_links = $content_data['internal_links'] ?? 0;
2359 $score = $internal_links > 0 ? 1 : 0;
2360
2361 return [
2362 'score' => $score,
2363 'max_score' => $this->scoring_factors['internal_linking'],
2364 'suggestions' => $score === 0 ? ['Add relevant internal links to other pages on your site'] : [],
2365 'details' => ['internal_links_count' => $internal_links]
2366 ];
2367 }
2368
2369 /**
2370 * Score technical factors (1 point)
2371 *
2372 * @param array $content_data Content analysis data
2373 * @param array $metadata Post metadata
2374 * @param array $options Additional options
2375 * @return array Scoring result
2376 */
2377 private function score_technical_factors(array $content_data, array $metadata, array $options): array {
2378 $score = 0;
2379 $suggestions = [];
2380
2381 // Meta description check
2382 $meta_desc = $metadata['description'] ?? '';
2383 if (!empty($meta_desc) && mb_strlen($meta_desc) >= self::DESCRIPTION_OPTIMAL_MIN && mb_strlen($meta_desc) <= self::DESCRIPTION_OPTIMAL_MAX) {
2384 $score += 0.5;
2385 } else {
2386 $suggestions[] = 'Add a compelling meta description (120-160 characters)';
2387 }
2388
2389 // Schema markup check (simplified)
2390 if (!empty($content_data['schema_present'])) {
2391 $score += 0.5;
2392 } else {
2393 $suggestions[] = 'Consider adding structured data (schema markup)';
2394 }
2395
2396 return [
2397 'score' => $score,
2398 'max_score' => $this->scoring_factors['technical_factors'],
2399 'suggestions' => $suggestions,
2400 'details' => [
2401 'meta_description_length' => mb_strlen($meta_desc),
2402 'schema_present' => !empty($content_data['schema_present'])
2403 ]
2404 ];
2405 }
2406
2407 /**
2408 * Get grade from score
2409 *
2410 * @param mixed $score Numeric score
2411 * @return string Letter grade
2412 */
2413 private function get_grade_from_score($score): string {
2414 $score = (int) $score; // Ensure it's an integer
2415
2416 if ($score >= 95) { return 'A+';
2417 }
2418 if ($score >= 90) { return 'A';
2419 }
2420 if ($score >= 85) { return 'A-';
2421 }
2422 if ($score >= 80) { return 'B+';
2423 }
2424 if ($score >= 75) { return 'B';
2425 }
2426 if ($score >= 70) { return 'B-';
2427 }
2428 if ($score >= 65) { return 'C+';
2429 }
2430 if ($score >= 60) { return 'C';
2431 }
2432 if ($score >= 55) { return 'C-';
2433 }
2434 if ($score >= 45) { return 'D+';
2435 }
2436 if ($score >= 35) { return 'D';
2437 }
2438 return 'F';
2439 }
2440
2441 /**
2442 * Analyze live content from editor (not saved to database yet)
2443 * Same as analyze_post_content but uses provided content instead of saved content
2444 *
2445 * @param string $live_content Live content from editor
2446 * @param int $post_id Post ID for metadata
2447 * @return array Content analysis data
2448 */
2449 public function analyze_live_content(string $live_content, int $post_id): array {
2450 $post = get_post($post_id);
2451 if (!$post) {
2452 return [];
2453 }
2454
2455 // Resolve the live string the same way stored content is resolved. On a
2456 // builder page the editor hands over raw builder markup (the block
2457 // editor cannot render blocks it has no client-side registration for),
2458 // which analyzed as-is reads as zero words — the reason a Divi page
2459 // could show a correct saved score beside a live panel still claiming
2460 // "No content".
2461 $content = self::resolve_live_content($live_content, $post);
2462 $title = $post->post_title;
2463
2464 // Extract headings from content
2465 $headings = $this->extract_headings($content);
2466
2467 // Count words using JavaScript-compatible method
2468 $plain_text = wp_strip_all_tags($content);
2469 $word_count = $this->calculate_word_count_js_style($plain_text);
2470
2471 // Calculate readability
2472 $readability_score = $this->calculate_readability_score($content);
2473
2474 // Count links
2475 $internal_links = $this->count_internal_links($content);
2476 $external_links = $this->count_external_links($content);
2477 $external_dofollow_links = $this->count_external_dofollow_links($content);
2478
2479 // Analyze images
2480 $images = $this->analyze_images($content);
2481
2482 // Get URL
2483 $url = get_permalink($post_id);
2484
2485 return [
2486 'content' => $content,
2487 'title' => $title,
2488 'headings' => $headings,
2489 'word_count' => $word_count,
2490 'readability_score' => $readability_score,
2491 'internal_links' => $internal_links,
2492 'external_links' => $external_links,
2493 'external_dofollow_links' => $external_dofollow_links,
2494 'images' => $images,
2495 'url' => $url,
2496 'slug' => $post->post_name,
2497 'post_modified' => $post->post_modified,
2498 'schema_present' => $this->detect_schema_present($content)
2499 || $this->thinkrank_global_schema_active($post->post_type)
2500 || $this->thinkrank_deployed_schema_active($post),
2501 ];
2502 }
2503
2504 /**
2505 * Calculate word count using JavaScript-compatible method
2506 * Matches the logic in contentAnalysis.js for consistency
2507 *
2508 * @param string $text Text to count words in
2509 * @return int Word count
2510 */
2511 private function calculate_word_count_js_style(string $text): int {
2512 if (empty($text)) {
2513 return 0;
2514 }
2515
2516 // Match JavaScript: trim, split by whitespace, filter empty
2517 $words = preg_split('/\s+/', trim($text), -1, PREG_SPLIT_NO_EMPTY);
2518 return count($words);
2519 }
2520
2521 /**
2522 * Get existing score data for a post from database
2523 *
2524 * @param int $post_id Post ID
2525 * @return array|null Existing score data or null if not found
2526 */
2527 public function get_existing_score_data(int $post_id): ?array {
2528 global $wpdb;
2529
2530 // Get table name and escape it properly (table names cannot be parameterized)
2531 $table_name = esc_sql($wpdb->prefix . 'thinkrank_seo_scores');
2532
2533 // Get the most recent score for this post
2534 // phpcs:ignore WordPress.DB.DirectDatabaseQuery.DirectQuery,WordPress.DB.DirectDatabaseQuery.NoCaching, PluginCheck.Security.DirectDB.UnescapedDBParameter -- SEO score retrieval requires direct database access
2535 $result = $wpdb->get_row($wpdb->prepare(
2536 // phpcs:ignore WordPress.DB.PreparedSQL.InterpolatedNotPrepared, PluginCheck.Security.DirectDB.UnescapedDBParameter -- Table name is properly escaped using esc_sql()
2537 "SELECT * FROM `{$table_name}`
2538 WHERE post_id = %d
2539 ORDER BY calculated_at DESC
2540 LIMIT 1",
2541 $post_id
2542 ), ARRAY_A);
2543
2544 if (!$result) {
2545 return null;
2546 }
2547
2548 // Decode JSON data (stored with json_encode)
2549 $score_breakdown = json_decode($result['score_breakdown'], true);
2550 $suggestions = json_decode($result['suggestions'], true);
2551
2552 // Format the data to match the expected structure
2553 return [
2554 'overall_score' => (int) $result['overall_score'],
2555 'grade' => $result['grade'],
2556 'score_breakdown' => $score_breakdown,
2557 'suggestions' => $suggestions ?: [],
2558 'target_keyword' => null, // Not stored in database, will be provided by frontend
2559 'calculated_at' => $result['calculated_at'],
2560 'score_id' => $result['id']
2561 ];
2562 }
2563 }
2564