| @@ -1263,16 +1263,14 @@ | ||
| 1263 | 1263 | if (empty($content) || empty($target_keyword)) { |
| 1264 | 1264 | return 0.0; |
| 1265 | 1265 | } |
| 1266 | 1266 | |
| 1267 | - // Occurrences and the word count both come from the reading text, so | |
| 1268 | - // shortcode syntax is in neither. | |
| 1269 | - $content_lower = strtolower(self::reading_text_of($content)); | |
| 1267 | + $content_lower = strtolower(wp_strip_all_tags($content)); | |
| 1270 | 1268 | $keyword_lower = strtolower($target_keyword); |
| 1271 | 1269 | |
| 1272 | 1270 | // Calculate keyword and semantic term frequency |
| 1273 | 1271 | $keyword_count = substr_count($content_lower, $keyword_lower); |
| 1274 | - $word_count = \ThinkRank\SEO\Word_Count_Index::count_words($content_lower); | |
| 1272 | + $word_count = $this->calculate_word_count_js_style($content_lower); | |
| 1275 | 1273 | |
| 1276 | 1274 | if ($word_count === 0) { |
| 1277 | 1275 | return 0.0; |
| 1278 | 1276 | } |
| @@ -1351,13 +1349,10 @@ | ||
| 1351 | 1349 | private function keyword_density(string $content, string $target_keyword): float { |
| 1352 | 1350 | if (empty($content) || empty($target_keyword)) { |
| 1353 | 1351 | return 0.0; |
| 1354 | 1352 | } |
| 1355 | - // Numerator and denominator from the same reading text. Counting the | |
| 1356 | - // keyword in wp_strip_all_tags() output found it inside shortcode | |
| 1357 | - // attributes the word count no longer includes. | |
| 1358 | - $plain = strtolower(self::reading_text_of($content)); | |
| 1359 | - $word_count = \ThinkRank\SEO\Word_Count_Index::count_words($plain); | |
| 1353 | + $plain = strtolower(wp_strip_all_tags($content)); | |
| 1354 | + $word_count = $this->calculate_word_count_js_style($plain); | |
| 1360 | 1355 | if ($word_count === 0) { |
| 1361 | 1356 | return 0.0; |
| 1362 | 1357 | } |
| 1363 | 1358 | $occurrences = substr_count($plain, strtolower($target_keyword)); |
| @@ -1392,10 +1387,10 @@ | ||
| 1392 | 1387 | private function keyword_in_first_paragraph(string $content, string $target_keyword): bool { |
| 1393 | 1388 | if (empty($content) || empty($target_keyword)) { |
| 1394 | 1389 | return false; |
| 1395 | 1390 | } |
| 1396 | - $plain = strtolower(self::reading_text_of($content)); | |
| 1397 | - $words = preg_split('/\s+/', $plain, -1, PREG_SPLIT_NO_EMPTY) ?: []; | |
| 1391 | + $plain = strtolower(wp_strip_all_tags($content)); | |
| 1392 | + $words = preg_split('/\s+/', trim($plain), -1, PREG_SPLIT_NO_EMPTY) ?: []; | |
| 1398 | 1393 | $window = array_slice($words, 0, max(50, (int) ceil(count($words) * 0.1))); |
| 1399 | 1394 | return strpos(implode(' ', $window), strtolower($target_keyword)) !== false; |
| 1400 | 1395 | } |
| 1401 | 1396 | |
| @@ -1655,12 +1650,11 @@ | ||
| 1655 | 1650 | |
| 1656 | 1651 | // Extract headings from content |
| 1657 | 1652 | $headings = $this->extract_headings($content); |
| 1658 | 1653 | |
| 1659 | - // Count the markup, not wp_strip_all_tags() output: stripping deletes | |
| 1660 | - // tags without a space, so "five</p><p>six" would already be one word | |
| 1661 | - // before the counter saw it. | |
| 1662 | - $word_count = $this->calculate_word_count_js_style($content); | |
| 1654 | + // Count words using JavaScript-compatible method | |
| 1655 | + $plain_text = wp_strip_all_tags($content); | |
| 1656 | + $word_count = $this->calculate_word_count_js_style($plain_text); | |
| 1663 | 1657 | |
| 1664 | 1658 | // Calculate readability |
| 1665 | 1659 | $readability_score = $this->calculate_readability_score($content); |
| 1666 | 1660 | |
| @@ -1780,9 +1774,9 @@ | ||
| 1780 | 1774 | // 2. Long paragraphs (flag paragraphs over 150 words). |
| 1781 | 1775 | $long_paragraphs = 0; |
| 1782 | 1776 | if (preg_match_all('/<p[^>]*>(.*?)<\/p>/is', $content, $matches)) { |
| 1783 | 1777 | foreach ($matches[1] as $paragraph) { |
| 1784 | - if ($this->calculate_word_count_js_style($paragraph) > 150) { | |
| 1778 | + if ($this->calculate_word_count_js_style(wp_strip_all_tags($paragraph)) > 150) { | |
| 1785 | 1779 | $long_paragraphs++; |
| 1786 | 1780 | } |
| 1787 | 1781 | } |
| 1788 | 1782 | } |
| @@ -1837,22 +1831,22 @@ | ||
| 1837 | 1831 | * @param string $content Content text |
| 1838 | 1832 | * @return float Readability score |
| 1839 | 1833 | */ |
| 1840 | 1834 | private function calculate_readability_score(string $content): float { |
| 1841 | - // Sentences, words and syllables all come from one reading text. The | |
| 1842 | - // word count drops shortcode syntax and punctuation-only tokens; when | |
| 1843 | - // sentences and syllables were still read off wp_strip_all_tags() | |
| 1844 | - // output, "[vc_column width="1/2"]" added syllables (and, with a "." in | |
| 1845 | - // an attribute, sentences) to a word count that did not include it, | |
| 1846 | - // and a WPBakery page's Flesch fell from 65 to 46. | |
| 1847 | - $text = self::reading_text_of($content); | |
| 1835 | + $text = wp_strip_all_tags($content); | |
| 1848 | 1836 | |
| 1849 | - if ('' === $text) { | |
| 1837 | + if (empty($text)) { | |
| 1850 | 1838 | return 0; |
| 1851 | 1839 | } |
| 1852 | 1840 | |
| 1853 | - $sentence_count = self::count_sentences($text); | |
| 1854 | - $word_count = \ThinkRank\SEO\Word_Count_Index::count_words($text); | |
| 1841 | + // Count sentences (approximate) | |
| 1842 | + $sentences = preg_split('/[.!?]+/', $text, -1, PREG_SPLIT_NO_EMPTY); | |
| 1843 | + $sentence_count = count($sentences); | |
| 1844 | + | |
| 1845 | + // Count words | |
| 1846 | + $word_count = $this->calculate_word_count_js_style(wp_strip_all_tags($text)); | |
| 1847 | + | |
| 1848 | + // Count syllables (approximate) | |
| 1855 | 1849 | $syllable_count = $this->count_syllables($text); |
| 1856 | 1850 | |
| 1857 | 1851 | if ($sentence_count === 0 || $word_count === 0) { |
| 1858 | 1852 | return 0; |
| @@ -1864,37 +1858,15 @@ | ||
| 1864 | 1858 | return max(0, min(100, $score)); |
| 1865 | 1859 | } |
| 1866 | 1860 | |
| 1867 | 1861 | /** |
| 1868 | - * Count the sentences in reading text. | |
| 1869 | - * | |
| 1870 | - * A sentence is a run of text between terminal punctuation that holds at | |
| 1871 | - * least one letter or digit. A fragment with no word in it (the space | |
| 1872 | - * after the final full stop, a stray "!" between two "?") is not a | |
| 1873 | - * sentence. contentAnalysis.js countSentences() applies the same rule. | |
| 1874 | - * | |
| 1875 | - * @since 2.14.2 | |
| 1876 | - * | |
| 1877 | - * @param string $text Text as returned by {@see self::reading_text_of()}. | |
| 1878 | - * @return int | |
| 1879 | - */ | |
| 1880 | - private static function count_sentences(string $text): int { | |
| 1881 | - $fragments = preg_split('/[.!?]+/', $text, -1, PREG_SPLIT_NO_EMPTY) ?: []; | |
| 1882 | - | |
| 1883 | - return count(preg_grep('/[\p{L}\p{N}]/u', $fragments) ?: []); | |
| 1884 | - } | |
| 1885 | - | |
| 1886 | - /** | |
| 1887 | 1862 | * Count syllables in text (approximate) |
| 1888 | 1863 | * |
| 1889 | - * Expects reading text ({@see self::reading_text_of()}); tags are not | |
| 1890 | - * stripped here, so a decoded "<" in prose is not read as a tag. | |
| 1891 | - * | |
| 1892 | 1864 | * @param string $text Text to analyze |
| 1893 | 1865 | * @return int Syllable count |
| 1894 | 1866 | */ |
| 1895 | 1867 | private function count_syllables(string $text): int { |
| 1896 | - $words = preg_split('/\s+/', trim(strtolower($text)), -1, PREG_SPLIT_NO_EMPTY) ?: []; | |
| 1868 | + $words = preg_split('/\s+/', trim(strtolower(wp_strip_all_tags($text))), -1, PREG_SPLIT_NO_EMPTY); | |
| 1897 | 1869 | $syllables = 0; |
| 1898 | 1870 | |
| 1899 | 1871 | foreach ($words as $word) { |
| 1900 | 1872 | $word = preg_replace('/[^a-z]/', '', $word); |
| @@ -2504,12 +2476,11 @@ | ||
| 2504 | 2476 | |
| 2505 | 2477 | // Extract headings from content |
| 2506 | 2478 | $headings = $this->extract_headings($content); |
| 2507 | 2479 | |
| 2508 | - // Count the markup, not wp_strip_all_tags() output: stripping deletes | |
| 2509 | - // tags without a space, so "five</p><p>six" would already be one word | |
| 2510 | - // before the counter saw it. | |
| 2511 | - $word_count = $this->calculate_word_count_js_style($content); | |
| 2480 | + // Count words using JavaScript-compatible method | |
| 2481 | + $plain_text = wp_strip_all_tags($content); | |
| 2482 | + $word_count = $this->calculate_word_count_js_style($plain_text); | |
| 2512 | 2483 | |
| 2513 | 2484 | // Calculate readability |
| 2514 | 2485 | $readability_score = $this->calculate_readability_score($content); |
| 2515 | 2486 | |
| @@ -2543,59 +2514,22 @@ | ||
| 2543 | 2514 | ]; |
| 2544 | 2515 | } |
| 2545 | 2516 | |
| 2546 | 2517 | /** |
| 2547 | - * Count the words a reader reads in HTML or text. | |
| 2518 | + * Calculate word count using JavaScript-compatible method | |
| 2519 | + * Matches the logic in contentAnalysis.js for consistency | |
| 2548 | 2520 | * |
| 2549 | - * The same extraction and counting as the thin content report | |
| 2550 | - * ({@see \ThinkRank\SEO\Word_Count_Index::reading_text()} and | |
| 2551 | - * {@see \ThinkRank\SEO\Word_Count_Index::count_words()}), so the editor | |
| 2552 | - * score and the report give one number for one page. Splitting on | |
| 2553 | - * whitespace counted shortcode syntax as words: a WPBakery page with 216 | |
| 2554 | - * words of prose scored a word count of 325 while the report said 216 | |
| 2555 | - * (#893 fixed the report only). It also counted tokens of punctuation | |
| 2556 | - * alone, such as a full stop after a link. | |
| 2557 | - * | |
| 2558 | - * Always words, whatever the locale's unit, because every threshold that | |
| 2559 | - * reads this value (content length, long paragraphs, headings per 300 | |
| 2560 | - * words, readability, keyword density) is in words. | |
| 2561 | - * | |
| 2562 | - * contentAnalysis.js calculateWordCount() applies the same two rules in | |
| 2563 | - * the editor. | |
| 2564 | - * | |
| 2565 | - * @param string $text HTML or text to count words in. | |
| 2521 | + * @param string $text Text to count words in | |
| 2566 | 2522 | * @return int Word count |
| 2567 | 2523 | */ |
| 2568 | 2524 | private function calculate_word_count_js_style(string $text): int { |
| 2569 | - if ('' === $text) { | |
| 2525 | + if (empty($text)) { | |
| 2570 | 2526 | return 0; |
| 2571 | 2527 | } |
| 2572 | 2528 | |
| 2573 | - return \ThinkRank\SEO\Word_Count_Index::count_words(self::reading_text_of($text)); | |
| 2574 | - } | |
| 2575 | - | |
| 2576 | - /** | |
| 2577 | - * The text a reader reads in HTML or text, as the word count sees it. | |
| 2578 | - * | |
| 2579 | - * Every check that divides by the word count (readability, keyword | |
| 2580 | - * density, topic relevance) reads its numerator from this same text, so | |
| 2581 | - * shortcode syntax is never on one side of a ratio and not the other. | |
| 2582 | - * | |
| 2583 | - * @since 2.14.2 | |
| 2584 | - * | |
| 2585 | - * @param string $content HTML or text. | |
| 2586 | - * @return string Plain text, whitespace collapsed. | |
| 2587 | - */ | |
| 2588 | - private static function reading_text_of(string $content): string { | |
| 2589 | - if ('' === $content) { | |
| 2590 | - return ''; | |
| 2591 | - } | |
| 2592 | - | |
| 2593 | - if (!class_exists('\ThinkRank\SEO\Word_Count_Index')) { | |
| 2594 | - require_once THINKRANK_PLUGIN_DIR . 'includes/seo/class-word-count-index.php'; | |
| 2595 | - } | |
| 2596 | - | |
| 2597 | - return \ThinkRank\SEO\Word_Count_Index::reading_text($content); | |
| 2529 | + // Match JavaScript: trim, split by whitespace, filter empty | |
| 2530 | + $words = preg_split('/\s+/', trim($text), -1, PREG_SPLIT_NO_EMPTY); | |
| 2531 | + return count($words); | |
| 2598 | 2532 | } |
| 2599 | 2533 | |
| 2600 | 2534 | /** |
| 2601 | 2535 | * Get existing score data for a post from database |