| @@ -59,8 +59,23 @@ | ||
| 59 | 59 | */ |
| 60 | 60 | private bool $measured_performance_resolved = false; |
| 61 | 61 | |
| 62 | 62 | /** |
| 63 | + * Length bands the editor scores against, in characters. | |
| 64 | + * | |
| 65 | + * Public so every surface that judges a title or description — the editor | |
| 66 | + * score and the Bulk Snippets problem filter — reads one set of numbers. | |
| 67 | + * Before these existed the bands were literals inside the scoring methods, | |
| 68 | + * and a second screen would have had to copy them and drift (#727). | |
| 69 | + * | |
| 70 | + * @since 2.8.0 | |
| 71 | + */ | |
| 72 | + public const TITLE_OPTIMAL_MIN = 35; | |
| 73 | + public const TITLE_OPTIMAL_MAX = 60; | |
| 74 | + public const DESCRIPTION_OPTIMAL_MIN = 120; | |
| 75 | + public const DESCRIPTION_OPTIMAL_MAX = 160; | |
| 76 | + | |
| 77 | + /** | |
| 63 | 78 | * 2025 SEO scoring factors (Q1 2025 Google Algorithm) |
| 64 | 79 | * Based on First Page Sage research and Google's latest updates |
| 65 | 80 | * |
| 66 | 81 | * @var array |
| @@ -125,8 +140,9 @@ | ||
| 125 | 140 | 'grade' => $result['grade'], |
| 126 | 141 | ]]; |
| 127 | 142 | $result['keyword_checks'] = $this->analyze_keyword_checks($content_data, $metadata, $keywords); |
| 128 | 143 | } |
| 144 | + $result['keywords'] = $this->keyword_placements($content_data, $metadata, $keywords); | |
| 129 | 145 | |
| 130 | 146 | return $result; |
| 131 | 147 | } |
| 132 | 148 | |
| @@ -236,8 +252,9 @@ | ||
| 236 | 252 | $best['target_keyword'] = $best_keyword; |
| 237 | 253 | $best['target_keywords'] = $keywords; |
| 238 | 254 | $best['keyword_results'] = $per_keyword; |
| 239 | 255 | $best['keyword_checks'] = $this->analyze_keyword_checks($content_data, $metadata, $keywords); |
| 256 | + $best['keywords'] = $this->keyword_placements($content_data, $metadata, $keywords); | |
| 240 | 257 | |
| 241 | 258 | return $best; |
| 242 | 259 | } |
| 243 | 260 | |
| @@ -252,15 +269,15 @@ | ||
| 252 | 269 | * @param string[] $keywords Target keywords. |
| 253 | 270 | * @return array<string,array{passed:bool,matched_keywords:string[]}> |
| 254 | 271 | */ |
| 255 | 272 | private function analyze_keyword_checks(array $content_data, array $metadata, array $keywords): array { |
| 256 | - $title = strtolower((string) ($metadata['title'] ?? $content_data['title'] ?? '')); | |
| 257 | - $description = strtolower((string) ($metadata['description'] ?? '')); | |
| 258 | - $content = strtolower(wp_strip_all_tags((string) ($content_data['content'] ?? ''))); | |
| 273 | + $title = self::lower((string) ($metadata['title'] ?? $content_data['title'] ?? '')); | |
| 274 | + $description = self::lower((string) ($metadata['description'] ?? '')); | |
| 275 | + $content = self::lower(self::plain_text((string) ($content_data['content'] ?? ''))); | |
| 259 | 276 | |
| 260 | 277 | $alts = ''; |
| 261 | 278 | foreach ((array) ($content_data['images'] ?? []) as $image) { |
| 262 | - $alts .= ' ' . strtolower((string) ($image['alt'] ?? '')); | |
| 279 | + $alts .= ' ' . self::lower((string) ($image['alt'] ?? '')); | |
| 263 | 280 | } |
| 264 | 281 | |
| 265 | 282 | // Build a searchable slug haystack from the post's OWN slug — never the |
| 266 | 283 | // full URL path. The path carries ancestors, category bases and date |
| @@ -269,17 +286,9 @@ | ||
| 269 | 286 | // way: an unpublished post has no pretty permalink (get_permalink() |
| 270 | 287 | // returns ?p=123), so the path held no slug at all and every draft |
| 271 | 288 | // scored "no match" until it was published. Hyphens/underscores become |
| 272 | 289 | // spaces so multi-word keywords can match. |
| 273 | - $slug_source = (string) ($content_data['slug'] ?? ''); | |
| 274 | - if ($slug_source === '') { | |
| 275 | - // Draft with no slug assigned yet: score what WordPress would | |
| 276 | - // generate from the title, which is what the editor shows as the | |
| 277 | - // proposed URL — so the check reads the same before and after | |
| 278 | - // publishing instead of flipping. | |
| 279 | - $slug_source = sanitize_title((string) ($content_data['title'] ?? '')); | |
| 280 | - } | |
| 281 | - $slug = strtolower(str_replace(['-', '_'], ' ', $slug_source)); | |
| 290 | + $slug = self::lower(self::slug_haystack($content_data)); | |
| 282 | 291 | |
| 283 | 292 | $haystacks = [ |
| 284 | 293 | 'title' => trim($title), |
| 285 | 294 | 'meta_description' => trim($description), |
| @@ -291,9 +300,9 @@ | ||
| 291 | 300 | $checks = []; |
| 292 | 301 | foreach ($haystacks as $location => $haystack) { |
| 293 | 302 | $matched = []; |
| 294 | 303 | foreach ($keywords as $keyword) { |
| 295 | - if ($this->keyword_matches($haystack, strtolower(trim($keyword)))) { | |
| 304 | + if ($this->keyword_matches($haystack, self::lower(trim($keyword)))) { | |
| 296 | 305 | $matched[] = $keyword; |
| 297 | 306 | } |
| 298 | 307 | } |
| 299 | 308 | $checks[$location] = [ |
| @@ -305,8 +314,210 @@ | ||
| 305 | 314 | return $checks; |
| 306 | 315 | } |
| 307 | 316 | |
| 308 | 317 | /** |
| 318 | + * The keyword placements the editor draws one gauge segment for, in the | |
| 319 | + * order a reader meets them (#729). | |
| 320 | + * | |
| 321 | + * @since 2.11.0 | |
| 322 | + * @var string[] | |
| 323 | + */ | |
| 324 | + public const PLACEMENTS = ['title', 'meta_description', 'slug', 'first_paragraph', 'subheading', 'content', 'image_alt']; | |
| 325 | + | |
| 326 | + /** | |
| 327 | + * Characters of plain text read as the opening when the content has no | |
| 328 | + * paragraph tag. | |
| 329 | + * | |
| 330 | + * @since 2.11.0 | |
| 331 | + */ | |
| 332 | + private const OPENING_CHARS = 300; | |
| 333 | + | |
| 334 | + /** | |
| 335 | + * Where each focus keyword is placed, keyword by keyword (#729). | |
| 336 | + * | |
| 337 | + * analyze_keyword_checks() answers "does ANY keyword appear here" for | |
| 338 | + * five places; this answers "where does THIS keyword appear" for seven, | |
| 339 | + * so the editor can show each keyword's own gauge. Same matcher, so a | |
| 340 | + * keyword counts as a word (not a fragment) and a keyword in a script | |
| 341 | + * written without spaces (Thai, Chinese, Japanese) still matches. | |
| 342 | + * | |
| 343 | + * `where` names the heading or alt text that matched, so the editor can | |
| 344 | + * say which one. | |
| 345 | + * | |
| 346 | + * @since 2.11.0 | |
| 347 | + * | |
| 348 | + * @param array $content_data Content analysis data. | |
| 349 | + * @param array $metadata Post metadata (title, description). | |
| 350 | + * @param string[] $keywords Focus keywords. | |
| 351 | + * @return array<int, array{keyword: string, passed: int, total: int, placements: array<string, array{passed: bool, where: string}>}> | |
| 352 | + */ | |
| 353 | + public function keyword_placements(array $content_data, array $metadata, array $keywords): array { | |
| 354 | + $html = (string) ($content_data['content'] ?? ''); | |
| 355 | + $plain = self::lower(self::plain_text($html)); | |
| 356 | + | |
| 357 | + $headings = []; | |
| 358 | + $source = isset($content_data['headings']) && is_array($content_data['headings']) ? $content_data['headings'] : $this->extract_headings($html); | |
| 359 | + foreach ($source as $heading) { | |
| 360 | + $text = trim((string) ($heading['text'] ?? '')); | |
| 361 | + if ((int) ($heading['level'] ?? 0) >= 2 && '' !== $text) { | |
| 362 | + $headings[] = $text; | |
| 363 | + } | |
| 364 | + } | |
| 365 | + | |
| 366 | + $alts = []; | |
| 367 | + foreach ((array) ($content_data['images'] ?? []) as $image) { | |
| 368 | + $alt = trim((string) ($image['alt'] ?? '')); | |
| 369 | + if ('' !== $alt) { | |
| 370 | + $alts[] = $alt; | |
| 371 | + } | |
| 372 | + } | |
| 373 | + | |
| 374 | + $single = [ | |
| 375 | + 'title' => self::lower((string) ($metadata['title'] ?? $content_data['title'] ?? '')), | |
| 376 | + 'meta_description' => self::lower((string) ($metadata['description'] ?? '')), | |
| 377 | + 'slug' => self::lower(self::slug_haystack($content_data)), | |
| 378 | + 'first_paragraph' => self::lower(self::opening($html)), | |
| 379 | + 'content' => $plain, | |
| 380 | + ]; | |
| 381 | + | |
| 382 | + $out = []; | |
| 383 | + foreach ($keywords as $keyword) { | |
| 384 | + $needle = self::lower(trim((string) $keyword)); | |
| 385 | + $placements = []; | |
| 386 | + | |
| 387 | + foreach (self::PLACEMENTS as $placement) { | |
| 388 | + if ('subheading' === $placement || 'image_alt' === $placement) { | |
| 389 | + $where = ''; | |
| 390 | + foreach ('subheading' === $placement ? $headings : $alts as $text) { | |
| 391 | + if ($this->keyword_matches(self::lower($text), $needle)) { | |
| 392 | + $where = $text; | |
| 393 | + break; | |
| 394 | + } | |
| 395 | + } | |
| 396 | + $placements[$placement] = ['passed' => '' !== $where, 'where' => $where]; | |
| 397 | + continue; | |
| 398 | + } | |
| 399 | + | |
| 400 | + $placements[$placement] = ['passed' => $this->keyword_matches($single[$placement], $needle), 'where' => '']; | |
| 401 | + } | |
| 402 | + | |
| 403 | + $out[] = [ | |
| 404 | + 'keyword' => (string) $keyword, | |
| 405 | + 'passed' => count(array_filter(array_column($placements, 'passed'))), | |
| 406 | + 'total' => count(self::PLACEMENTS), | |
| 407 | + 'placements' => $placements, | |
| 408 | + ]; | |
| 409 | + } | |
| 410 | + | |
| 411 | + return $out; | |
| 412 | + } | |
| 413 | + | |
| 414 | + /** | |
| 415 | + * The post's own slug as searchable text: hyphens and underscores become | |
| 416 | + * spaces so a multi-word keyword can match, and a slug WordPress | |
| 417 | + * percent-encoded (Thai, Cyrillic, Chinese) is decoded, or it could never | |
| 418 | + * match a keyword typed in that script. | |
| 419 | + * | |
| 420 | + * Never the full URL path: the path carries ancestors, category bases and | |
| 421 | + * date segments, so a child of /clinical-trials/ reported "keyword in slug" | |
| 422 | + * for a page actually slugged `contact-us`. An unpublished post has no | |
| 423 | + * pretty permalink either, so the path held no slug at all. | |
| 424 | + * | |
| 425 | + * @param array $content_data Content analysis data. | |
| 426 | + * @return string | |
| 427 | + */ | |
| 428 | + private static function slug_haystack(array $content_data): string { | |
| 429 | + $slug = (string) ($content_data['slug'] ?? ''); | |
| 430 | + if ('' === $slug) { | |
| 431 | + // Draft with no slug assigned yet: score what WordPress would | |
| 432 | + // generate from the title, which is what the editor shows as the | |
| 433 | + // proposed URL — so the check reads the same before and after | |
| 434 | + // publishing instead of flipping. | |
| 435 | + $slug = sanitize_title((string) ($content_data['title'] ?? '')); | |
| 436 | + } | |
| 437 | + | |
| 438 | + return trim(str_replace(['-', '_'], ' ', rawurldecode($slug))); | |
| 439 | + } | |
| 440 | + | |
| 441 | + /** | |
| 442 | + * The opening of the content: its first paragraph with text, or the first | |
| 443 | + * few hundred characters when it has none. Counted in characters, not | |
| 444 | + * words, so a language written without spaces is not read as one word. | |
| 445 | + * | |
| 446 | + * @param string $html Content HTML. | |
| 447 | + * @return string Plain text. | |
| 448 | + */ | |
| 449 | + private static function opening(string $html): string { | |
| 450 | + if (preg_match_all('/<p\b[^>]*>(.*?)<\/p>/isu', $html, $matches)) { | |
| 451 | + foreach ($matches[1] as $paragraph) { | |
| 452 | + $text = self::collapse_whitespace(wp_strip_all_tags($paragraph)); | |
| 453 | + if ('' !== $text) { | |
| 454 | + return $text; | |
| 455 | + } | |
| 456 | + } | |
| 457 | + } | |
| 458 | + | |
| 459 | + $text = self::plain_text($html); | |
| 460 | + | |
| 461 | + return function_exists('mb_substr') ? mb_substr($text, 0, self::OPENING_CHARS) : substr($text, 0, self::OPENING_CHARS); | |
| 462 | + } | |
| 463 | + | |
| 464 | + /** | |
| 465 | + * Content as plain text, with a space where each tag was. wp_strip_all_tags() | |
| 466 | + * alone joins neighbouring blocks — "…coffee grinder</h3><p>A good…" became | |
| 467 | + * "coffee grinderA good" — so a keyword at the end of a heading or a | |
| 468 | + * paragraph was no longer a word and did not match. | |
| 469 | + * | |
| 470 | + * @param string $html Content HTML. | |
| 471 | + * @return string | |
| 472 | + */ | |
| 473 | + private static function plain_text(string $html): string { | |
| 474 | + $spaced = preg_replace('/<[^>]+>/', ' $0 ', $html); | |
| 475 | + | |
| 476 | + return self::collapse_whitespace(wp_strip_all_tags(null === $spaced ? $html : (string) $spaced)); | |
| 477 | + } | |
| 478 | + | |
| 479 | + /** | |
| 480 | + * Runs of whitespace down to one space. | |
| 481 | + * | |
| 482 | + * The `/u` pass is the one that understands a multibyte space, but | |
| 483 | + * preg_replace() answers null on bytes that are not valid UTF-8 rather than | |
| 484 | + * throwing — and casting that null to a string blanked the haystack, so a | |
| 485 | + * post carrying one mojibake byte (a Latin-1 paste, an old import) reported | |
| 486 | + * every keyword as missing from its body, its opening, and every | |
| 487 | + * subheading. The gauge said 0/7 and told the author to add a keyword that | |
| 488 | + * was already there. | |
| 489 | + * | |
| 490 | + * Falls back to the byte-wise collapse, which is what this did before the | |
| 491 | + * multibyte work added the modifier. Same reasoning keyword_matches() | |
| 492 | + * already records for its own PCRE failure: a pattern PCRE refuses must not | |
| 493 | + * be reported as a confident "no match". | |
| 494 | + * | |
| 495 | + * @param string $text Text to collapse. | |
| 496 | + * @return string | |
| 497 | + */ | |
| 498 | + private static function collapse_whitespace(string $text): string { | |
| 499 | + $collapsed = preg_replace('/\s+/u', ' ', $text); | |
| 500 | + | |
| 501 | + if (null === $collapsed) { | |
| 502 | + $collapsed = preg_replace('/\s+/', ' ', $text); | |
| 503 | + } | |
| 504 | + | |
| 505 | + return trim(null === $collapsed ? $text : (string) $collapsed); | |
| 506 | + } | |
| 507 | + | |
| 508 | + /** | |
| 509 | + * Lowercase in any script. strtolower() only folds ASCII, so "Кофе" never | |
| 510 | + * matched "кофе". | |
| 511 | + * | |
| 512 | + * @param string $text Text. | |
| 513 | + * @return string | |
| 514 | + */ | |
| 515 | + private static function lower(string $text): string { | |
| 516 | + return function_exists('mb_strtolower') ? mb_strtolower($text, 'UTF-8') : strtolower($text); | |
| 517 | + } | |
| 518 | + | |
| 519 | + /** | |
| 309 | 520 | * Scripts written without spaces between words. |
| 310 | 521 | * |
| 311 | 522 | * @since 2.1.0 |
| 312 | 523 | * @var string |
| @@ -647,9 +858,9 @@ | ||
| 647 | 858 | $title_length = mb_strlen($title); |
| 648 | 859 | |
| 649 | 860 | // 2025 length optimization (6 points). 60 characters is the recommended |
| 650 | 861 | // maximum for best SERP visibility before Google truncates the title. |
| 651 | - if ($title_length >= 35 && $title_length <= 60) { | |
| 862 | + if ($title_length >= self::TITLE_OPTIMAL_MIN && $title_length <= self::TITLE_OPTIMAL_MAX) { | |
| 652 | 863 | $score += 6; |
| 653 | 864 | } elseif ($title_length >= 25 && $title_length <= 75) { |
| 654 | 865 | $score += 4; |
| 655 | 866 | $suggestions[] = 'Optimize title length to 35-60 characters for better SERP visibility'; |
| @@ -1052,14 +1263,16 @@ | ||
| 1052 | 1263 | if (empty($content) || empty($target_keyword)) { |
| 1053 | 1264 | return 0.0; |
| 1054 | 1265 | } |
| 1055 | 1266 | |
| 1056 | - $content_lower = strtolower(wp_strip_all_tags($content)); | |
| 1267 | + // Occurrences and the word count both come from the reading text, so | |
| 1268 | + // shortcode syntax is in neither. | |
| 1269 | + $content_lower = strtolower(self::reading_text_of($content)); | |
| 1057 | 1270 | $keyword_lower = strtolower($target_keyword); |
| 1058 | 1271 | |
| 1059 | 1272 | // Calculate keyword and semantic term frequency |
| 1060 | 1273 | $keyword_count = substr_count($content_lower, $keyword_lower); |
| 1061 | - $word_count = $this->calculate_word_count_js_style($content_lower); | |
| 1274 | + $word_count = \ThinkRank\SEO\Word_Count_Index::count_words($content_lower); | |
| 1062 | 1275 | |
| 1063 | 1276 | if ($word_count === 0) { |
| 1064 | 1277 | return 0.0; |
| 1065 | 1278 | } |
| @@ -1138,10 +1351,13 @@ | ||
| 1138 | 1351 | private function keyword_density(string $content, string $target_keyword): float { |
| 1139 | 1352 | if (empty($content) || empty($target_keyword)) { |
| 1140 | 1353 | return 0.0; |
| 1141 | 1354 | } |
| 1142 | - $plain = strtolower(wp_strip_all_tags($content)); | |
| 1143 | - $word_count = $this->calculate_word_count_js_style($plain); | |
| 1355 | + // Numerator and denominator from the same reading text. Counting the | |
| 1356 | + // keyword in wp_strip_all_tags() output found it inside shortcode | |
| 1357 | + // attributes the word count no longer includes. | |
| 1358 | + $plain = strtolower(self::reading_text_of($content)); | |
| 1359 | + $word_count = \ThinkRank\SEO\Word_Count_Index::count_words($plain); | |
| 1144 | 1360 | if ($word_count === 0) { |
| 1145 | 1361 | return 0.0; |
| 1146 | 1362 | } |
| 1147 | 1363 | $occurrences = substr_count($plain, strtolower($target_keyword)); |
| @@ -1176,10 +1392,10 @@ | ||
| 1176 | 1392 | private function keyword_in_first_paragraph(string $content, string $target_keyword): bool { |
| 1177 | 1393 | if (empty($content) || empty($target_keyword)) { |
| 1178 | 1394 | return false; |
| 1179 | 1395 | } |
| 1180 | - $plain = strtolower(wp_strip_all_tags($content)); | |
| 1181 | - $words = preg_split('/\s+/', trim($plain), -1, PREG_SPLIT_NO_EMPTY) ?: []; | |
| 1396 | + $plain = strtolower(self::reading_text_of($content)); | |
| 1397 | + $words = preg_split('/\s+/', $plain, -1, PREG_SPLIT_NO_EMPTY) ?: []; | |
| 1182 | 1398 | $window = array_slice($words, 0, max(50, (int) ceil(count($words) * 0.1))); |
| 1183 | 1399 | return strpos(implode(' ', $window), strtolower($target_keyword)) !== false; |
| 1184 | 1400 | } |
| 1185 | 1401 | |
| @@ -1409,9 +1625,24 @@ | ||
| 1409 | 1625 | if (!class_exists('\ThinkRank\SEO\Builder_Content')) { |
| 1410 | 1626 | require_once THINKRANK_PLUGIN_DIR . 'includes/seo/class-builder-content.php'; |
| 1411 | 1627 | } |
| 1412 | 1628 | |
| 1413 | - return \ThinkRank\SEO\Builder_Content::resolve_markup($live_content, $post); | |
| 1629 | + // Bind the live markup to the post the render makes current. Since #862 | |
| 1630 | + // the resolver runs setup_postdata() on this post, so a shortcode that | |
| 1631 | + // builds its output from the current post's content — get_the_content(), | |
| 1632 | + // get_post()->post_content, as a table of contents or a reading-time | |
| 1633 | + // shortcode does — otherwise read the last saved body while the unsaved | |
| 1634 | + // markup rendered around it, and lagged a save behind (#864). | |
| 1635 | + // | |
| 1636 | + // A clone, not the caller's object: the post is current only for the | |
| 1637 | + // duration of the render and the caller's $post must come back | |
| 1638 | + // unchanged. `thinkrank_analyzable_content` receives the clone too, | |
| 1639 | + // which is what makes $post->post_content there agree with the markup | |
| 1640 | + // being analyzed on the live path. | |
| 1641 | + $bound = clone $post; | |
| 1642 | + $bound->post_content = $live_content; | |
| 1643 | + | |
| 1644 | + return \ThinkRank\SEO\Builder_Content::resolve_markup($live_content, $bound); | |
| 1414 | 1645 | } |
| 1415 | 1646 | |
| 1416 | 1647 | public function analyze_post_content(int $post_id): array { |
| 1417 | 1648 | $post = get_post($post_id); |
| @@ -1424,11 +1655,12 @@ | ||
| 1424 | 1655 | |
| 1425 | 1656 | // Extract headings from content |
| 1426 | 1657 | $headings = $this->extract_headings($content); |
| 1427 | 1658 | |
| 1428 | - // Count words using JavaScript-compatible method | |
| 1429 | - $plain_text = wp_strip_all_tags($content); | |
| 1430 | - $word_count = $this->calculate_word_count_js_style($plain_text); | |
| 1659 | + // Count the markup, not wp_strip_all_tags() output: stripping deletes | |
| 1660 | + // tags without a space, so "five</p><p>six" would already be one word | |
| 1661 | + // before the counter saw it. | |
| 1662 | + $word_count = $this->calculate_word_count_js_style($content); | |
| 1431 | 1663 | |
| 1432 | 1664 | // Calculate readability |
| 1433 | 1665 | $readability_score = $this->calculate_readability_score($content); |
| 1434 | 1666 | |
| @@ -1548,9 +1780,9 @@ | ||
| 1548 | 1780 | // 2. Long paragraphs (flag paragraphs over 150 words). |
| 1549 | 1781 | $long_paragraphs = 0; |
| 1550 | 1782 | if (preg_match_all('/<p[^>]*>(.*?)<\/p>/is', $content, $matches)) { |
| 1551 | 1783 | foreach ($matches[1] as $paragraph) { |
| 1552 | - if ($this->calculate_word_count_js_style(wp_strip_all_tags($paragraph)) > 150) { | |
| 1784 | + if ($this->calculate_word_count_js_style($paragraph) > 150) { | |
| 1553 | 1785 | $long_paragraphs++; |
| 1554 | 1786 | } |
| 1555 | 1787 | } |
| 1556 | 1788 | } |
| @@ -1605,22 +1837,22 @@ | ||
| 1605 | 1837 | * @param string $content Content text |
| 1606 | 1838 | * @return float Readability score |
| 1607 | 1839 | */ |
| 1608 | 1840 | private function calculate_readability_score(string $content): float { |
| 1609 | - $text = wp_strip_all_tags($content); | |
| 1841 | + // Sentences, words and syllables all come from one reading text. The | |
| 1842 | + // word count drops shortcode syntax and punctuation-only tokens; when | |
| 1843 | + // sentences and syllables were still read off wp_strip_all_tags() | |
| 1844 | + // output, "[vc_column width="1/2"]" added syllables (and, with a "." in | |
| 1845 | + // an attribute, sentences) to a word count that did not include it, | |
| 1846 | + // and a WPBakery page's Flesch fell from 65 to 46. | |
| 1847 | + $text = self::reading_text_of($content); | |
| 1610 | 1848 | |
| 1611 | - if (empty($text)) { | |
| 1849 | + if ('' === $text) { | |
| 1612 | 1850 | return 0; |
| 1613 | 1851 | } |
| 1614 | 1852 | |
| 1615 | - // Count sentences (approximate) | |
| 1616 | - $sentences = preg_split('/[.!?]+/', $text, -1, PREG_SPLIT_NO_EMPTY); | |
| 1617 | - $sentence_count = count($sentences); | |
| 1618 | - | |
| 1619 | - // Count words | |
| 1620 | - $word_count = $this->calculate_word_count_js_style(wp_strip_all_tags($text)); | |
| 1621 | - | |
| 1622 | - // Count syllables (approximate) | |
| 1853 | + $sentence_count = self::count_sentences($text); | |
| 1854 | + $word_count = \ThinkRank\SEO\Word_Count_Index::count_words($text); | |
| 1623 | 1855 | $syllable_count = $this->count_syllables($text); |
| 1624 | 1856 | |
| 1625 | 1857 | if ($sentence_count === 0 || $word_count === 0) { |
| 1626 | 1858 | return 0; |
| @@ -1632,15 +1864,37 @@ | ||
| 1632 | 1864 | return max(0, min(100, $score)); |
| 1633 | 1865 | } |
| 1634 | 1866 | |
| 1635 | 1867 | /** |
| 1868 | + * Count the sentences in reading text. | |
| 1869 | + * | |
| 1870 | + * A sentence is a run of text between terminal punctuation that holds at | |
| 1871 | + * least one letter or digit. A fragment with no word in it (the space | |
| 1872 | + * after the final full stop, a stray "!" between two "?") is not a | |
| 1873 | + * sentence. contentAnalysis.js countSentences() applies the same rule. | |
| 1874 | + * | |
| 1875 | + * @since 2.14.2 | |
| 1876 | + * | |
| 1877 | + * @param string $text Text as returned by {@see self::reading_text_of()}. | |
| 1878 | + * @return int | |
| 1879 | + */ | |
| 1880 | + private static function count_sentences(string $text): int { | |
| 1881 | + $fragments = preg_split('/[.!?]+/', $text, -1, PREG_SPLIT_NO_EMPTY) ?: []; | |
| 1882 | + | |
| 1883 | + return count(preg_grep('/[\p{L}\p{N}]/u', $fragments) ?: []); | |
| 1884 | + } | |
| 1885 | + | |
| 1886 | + /** | |
| 1636 | 1887 | * Count syllables in text (approximate) |
| 1637 | 1888 | * |
| 1889 | + * Expects reading text ({@see self::reading_text_of()}); tags are not | |
| 1890 | + * stripped here, so a decoded "<" in prose is not read as a tag. | |
| 1891 | + * | |
| 1638 | 1892 | * @param string $text Text to analyze |
| 1639 | 1893 | * @return int Syllable count |
| 1640 | 1894 | */ |
| 1641 | 1895 | private function count_syllables(string $text): int { |
| 1642 | - $words = preg_split('/\s+/', trim(strtolower(wp_strip_all_tags($text))), -1, PREG_SPLIT_NO_EMPTY); | |
| 1896 | + $words = preg_split('/\s+/', trim(strtolower($text)), -1, PREG_SPLIT_NO_EMPTY) ?: []; | |
| 1643 | 1897 | $syllables = 0; |
| 1644 | 1898 | |
| 1645 | 1899 | foreach ($words as $word) { |
| 1646 | 1900 | $word = preg_replace('/[^a-z]/', '', $word); |
| @@ -2166,9 +2420,9 @@ | ||
| 2166 | 2420 | $suggestions = []; |
| 2167 | 2421 | |
| 2168 | 2422 | // Meta description check |
| 2169 | 2423 | $meta_desc = $metadata['description'] ?? ''; |
| 2170 | - if (!empty($meta_desc) && mb_strlen($meta_desc) >= 120 && mb_strlen($meta_desc) <= 160) { | |
| 2424 | + if (!empty($meta_desc) && mb_strlen($meta_desc) >= self::DESCRIPTION_OPTIMAL_MIN && mb_strlen($meta_desc) <= self::DESCRIPTION_OPTIMAL_MAX) { | |
| 2171 | 2425 | $score += 0.5; |
| 2172 | 2426 | } else { |
| 2173 | 2427 | $suggestions[] = 'Add a compelling meta description (120-160 characters)'; |
| 2174 | 2428 | } |
| @@ -2250,11 +2504,12 @@ | ||
| 2250 | 2504 | |
| 2251 | 2505 | // Extract headings from content |
| 2252 | 2506 | $headings = $this->extract_headings($content); |
| 2253 | 2507 | |
| 2254 | - // Count words using JavaScript-compatible method | |
| 2255 | - $plain_text = wp_strip_all_tags($content); | |
| 2256 | - $word_count = $this->calculate_word_count_js_style($plain_text); | |
| 2508 | + // Count the markup, not wp_strip_all_tags() output: stripping deletes | |
| 2509 | + // tags without a space, so "five</p><p>six" would already be one word | |
| 2510 | + // before the counter saw it. | |
| 2511 | + $word_count = $this->calculate_word_count_js_style($content); | |
| 2257 | 2512 | |
| 2258 | 2513 | // Calculate readability |
| 2259 | 2514 | $readability_score = $this->calculate_readability_score($content); |
| 2260 | 2515 | |
| @@ -2288,22 +2543,59 @@ | ||
| 2288 | 2543 | ]; |
| 2289 | 2544 | } |
| 2290 | 2545 | |
| 2291 | 2546 | /** |
| 2292 | - * Calculate word count using JavaScript-compatible method | |
| 2293 | - * Matches the logic in contentAnalysis.js for consistency | |
| 2547 | + * Count the words a reader reads in HTML or text. | |
| 2294 | 2548 | * |
| 2295 | - * @param string $text Text to count words in | |
| 2549 | + * The same extraction and counting as the thin content report | |
| 2550 | + * ({@see \ThinkRank\SEO\Word_Count_Index::reading_text()} and | |
| 2551 | + * {@see \ThinkRank\SEO\Word_Count_Index::count_words()}), so the editor | |
| 2552 | + * score and the report give one number for one page. Splitting on | |
| 2553 | + * whitespace counted shortcode syntax as words: a WPBakery page with 216 | |
| 2554 | + * words of prose scored a word count of 325 while the report said 216 | |
| 2555 | + * (#893 fixed the report only). It also counted tokens of punctuation | |
| 2556 | + * alone, such as a full stop after a link. | |
| 2557 | + * | |
| 2558 | + * Always words, whatever the locale's unit, because every threshold that | |
| 2559 | + * reads this value (content length, long paragraphs, headings per 300 | |
| 2560 | + * words, readability, keyword density) is in words. | |
| 2561 | + * | |
| 2562 | + * contentAnalysis.js calculateWordCount() applies the same two rules in | |
| 2563 | + * the editor. | |
| 2564 | + * | |
| 2565 | + * @param string $text HTML or text to count words in. | |
| 2296 | 2566 | * @return int Word count |
| 2297 | 2567 | */ |
| 2298 | 2568 | private function calculate_word_count_js_style(string $text): int { |
| 2299 | - if (empty($text)) { | |
| 2569 | + if ('' === $text) { | |
| 2300 | 2570 | return 0; |
| 2301 | 2571 | } |
| 2302 | 2572 | |
| 2303 | - // Match JavaScript: trim, split by whitespace, filter empty | |
| 2304 | - $words = preg_split('/\s+/', trim($text), -1, PREG_SPLIT_NO_EMPTY); | |
| 2305 | - return count($words); | |
| 2573 | + return \ThinkRank\SEO\Word_Count_Index::count_words(self::reading_text_of($text)); | |
| 2574 | + } | |
| 2575 | + | |
| 2576 | + /** | |
| 2577 | + * The text a reader reads in HTML or text, as the word count sees it. | |
| 2578 | + * | |
| 2579 | + * Every check that divides by the word count (readability, keyword | |
| 2580 | + * density, topic relevance) reads its numerator from this same text, so | |
| 2581 | + * shortcode syntax is never on one side of a ratio and not the other. | |
| 2582 | + * | |
| 2583 | + * @since 2.14.2 | |
| 2584 | + * | |
| 2585 | + * @param string $content HTML or text. | |
| 2586 | + * @return string Plain text, whitespace collapsed. | |
| 2587 | + */ | |
| 2588 | + private static function reading_text_of(string $content): string { | |
| 2589 | + if ('' === $content) { | |
| 2590 | + return ''; | |
| 2591 | + } | |
| 2592 | + | |
| 2593 | + if (!class_exists('\ThinkRank\SEO\Word_Count_Index')) { | |
| 2594 | + require_once THINKRANK_PLUGIN_DIR . 'includes/seo/class-word-count-index.php'; | |
| 2595 | + } | |
| 2596 | + | |
| 2597 | + return \ThinkRank\SEO\Word_Count_Index::reading_text($content); | |
| 2306 | 2598 | } |
| 2307 | 2599 | |
| 2308 | 2600 | /** |
| 2309 | 2601 | * Get existing score data for a post from database |