| @@ -1263,14 +1263,16 @@ | ||
| 1263 | 1263 | if (empty($content) || empty($target_keyword)) { |
| 1264 | 1264 | return 0.0; |
| 1265 | 1265 | } |
| 1266 | 1266 | |
| 1267 | - $content_lower = strtolower(wp_strip_all_tags($content)); | |
| 1267 | + // Occurrences and the word count both come from the reading text, so | |
| 1268 | + // shortcode syntax is in neither. | |
| 1269 | + $content_lower = strtolower(self::reading_text_of($content)); | |
| 1268 | 1270 | $keyword_lower = strtolower($target_keyword); |
| 1269 | 1271 | |
| 1270 | 1272 | // Calculate keyword and semantic term frequency |
| 1271 | 1273 | $keyword_count = substr_count($content_lower, $keyword_lower); |
| 1272 | - $word_count = $this->calculate_word_count_js_style($content_lower); | |
| 1274 | + $word_count = \ThinkRank\SEO\Word_Count_Index::count_words($content_lower); | |
| 1273 | 1275 | |
| 1274 | 1276 | if ($word_count === 0) { |
| 1275 | 1277 | return 0.0; |
| 1276 | 1278 | } |
| @@ -1349,10 +1351,13 @@ | ||
| 1349 | 1351 | private function keyword_density(string $content, string $target_keyword): float { |
| 1350 | 1352 | if (empty($content) || empty($target_keyword)) { |
| 1351 | 1353 | return 0.0; |
| 1352 | 1354 | } |
| 1353 | - $plain = strtolower(wp_strip_all_tags($content)); | |
| 1354 | - $word_count = $this->calculate_word_count_js_style($plain); | |
| 1355 | + // Numerator and denominator from the same reading text. Counting the | |
| 1356 | + // keyword in wp_strip_all_tags() output found it inside shortcode | |
| 1357 | + // attributes the word count no longer includes. | |
| 1358 | + $plain = strtolower(self::reading_text_of($content)); | |
| 1359 | + $word_count = \ThinkRank\SEO\Word_Count_Index::count_words($plain); | |
| 1355 | 1360 | if ($word_count === 0) { |
| 1356 | 1361 | return 0.0; |
| 1357 | 1362 | } |
| 1358 | 1363 | $occurrences = substr_count($plain, strtolower($target_keyword)); |
| @@ -1387,10 +1392,10 @@ | ||
| 1387 | 1392 | private function keyword_in_first_paragraph(string $content, string $target_keyword): bool { |
| 1388 | 1393 | if (empty($content) || empty($target_keyword)) { |
| 1389 | 1394 | return false; |
| 1390 | 1395 | } |
| 1391 | - $plain = strtolower(wp_strip_all_tags($content)); | |
| 1392 | - $words = preg_split('/\s+/', trim($plain), -1, PREG_SPLIT_NO_EMPTY) ?: []; | |
| 1396 | + $plain = strtolower(self::reading_text_of($content)); | |
| 1397 | + $words = preg_split('/\s+/', $plain, -1, PREG_SPLIT_NO_EMPTY) ?: []; | |
| 1393 | 1398 | $window = array_slice($words, 0, max(50, (int) ceil(count($words) * 0.1))); |
| 1394 | 1399 | return strpos(implode(' ', $window), strtolower($target_keyword)) !== false; |
| 1395 | 1400 | } |
| 1396 | 1401 | |
| @@ -1650,11 +1655,12 @@ | ||
| 1650 | 1655 | |
| 1651 | 1656 | // Extract headings from content |
| 1652 | 1657 | $headings = $this->extract_headings($content); |
| 1653 | 1658 | |
| 1654 | - // Count words using JavaScript-compatible method | |
| 1655 | - $plain_text = wp_strip_all_tags($content); | |
| 1656 | - $word_count = $this->calculate_word_count_js_style($plain_text); | |
| 1659 | + // Count the markup, not wp_strip_all_tags() output: stripping deletes | |
| 1660 | + // tags without a space, so "five</p><p>six" would already be one word | |
| 1661 | + // before the counter saw it. | |
| 1662 | + $word_count = $this->calculate_word_count_js_style($content); | |
| 1657 | 1663 | |
| 1658 | 1664 | // Calculate readability |
| 1659 | 1665 | $readability_score = $this->calculate_readability_score($content); |
| 1660 | 1666 | |
| @@ -1774,9 +1780,9 @@ | ||
| 1774 | 1780 | // 2. Long paragraphs (flag paragraphs over 150 words). |
| 1775 | 1781 | $long_paragraphs = 0; |
| 1776 | 1782 | if (preg_match_all('/<p[^>]*>(.*?)<\/p>/is', $content, $matches)) { |
| 1777 | 1783 | foreach ($matches[1] as $paragraph) { |
| 1778 | - if ($this->calculate_word_count_js_style(wp_strip_all_tags($paragraph)) > 150) { | |
| 1784 | + if ($this->calculate_word_count_js_style($paragraph) > 150) { | |
| 1779 | 1785 | $long_paragraphs++; |
| 1780 | 1786 | } |
| 1781 | 1787 | } |
| 1782 | 1788 | } |
| @@ -1831,22 +1837,22 @@ | ||
| 1831 | 1837 | * @param string $content Content text |
| 1832 | 1838 | * @return float Readability score |
| 1833 | 1839 | */ |
| 1834 | 1840 | private function calculate_readability_score(string $content): float { |
| 1835 | - $text = wp_strip_all_tags($content); | |
| 1841 | + // Sentences, words and syllables all come from one reading text. The | |
| 1842 | + // word count drops shortcode syntax and punctuation-only tokens; when | |
| 1843 | + // sentences and syllables were still read off wp_strip_all_tags() | |
| 1844 | + // output, "[vc_column width="1/2"]" added syllables (and, with a "." in | |
| 1845 | + // an attribute, sentences) to a word count that did not include it, | |
| 1846 | + // and a WPBakery page's Flesch fell from 65 to 46. | |
| 1847 | + $text = self::reading_text_of($content); | |
| 1836 | 1848 | |
| 1837 | - if (empty($text)) { | |
| 1849 | + if ('' === $text) { | |
| 1838 | 1850 | return 0; |
| 1839 | 1851 | } |
| 1840 | 1852 | |
| 1841 | - // Count sentences (approximate) | |
| 1842 | - $sentences = preg_split('/[.!?]+/', $text, -1, PREG_SPLIT_NO_EMPTY); | |
| 1843 | - $sentence_count = count($sentences); | |
| 1844 | - | |
| 1845 | - // Count words | |
| 1846 | - $word_count = $this->calculate_word_count_js_style(wp_strip_all_tags($text)); | |
| 1847 | - | |
| 1848 | - // Count syllables (approximate) | |
| 1853 | + $sentence_count = self::count_sentences($text); | |
| 1854 | + $word_count = \ThinkRank\SEO\Word_Count_Index::count_words($text); | |
| 1849 | 1855 | $syllable_count = $this->count_syllables($text); |
| 1850 | 1856 | |
| 1851 | 1857 | if ($sentence_count === 0 || $word_count === 0) { |
| 1852 | 1858 | return 0; |
| @@ -1858,15 +1864,37 @@ | ||
| 1858 | 1864 | return max(0, min(100, $score)); |
| 1859 | 1865 | } |
| 1860 | 1866 | |
| 1861 | 1867 | /** |
| 1868 | + * Count the sentences in reading text. | |
| 1869 | + * | |
| 1870 | + * A sentence is a run of text between terminal punctuation that holds at | |
| 1871 | + * least one letter or digit. A fragment with no word in it (the space | |
| 1872 | + * after the final full stop, a stray "!" between two "?") is not a | |
| 1873 | + * sentence. contentAnalysis.js countSentences() applies the same rule. | |
| 1874 | + * | |
| 1875 | + * @since 2.14.2 | |
| 1876 | + * | |
| 1877 | + * @param string $text Text as returned by {@see self::reading_text_of()}. | |
| 1878 | + * @return int | |
| 1879 | + */ | |
| 1880 | + private static function count_sentences(string $text): int { | |
| 1881 | + $fragments = preg_split('/[.!?]+/', $text, -1, PREG_SPLIT_NO_EMPTY) ?: []; | |
| 1882 | + | |
| 1883 | + return count(preg_grep('/[\p{L}\p{N}]/u', $fragments) ?: []); | |
| 1884 | + } | |
| 1885 | + | |
| 1886 | + /** | |
| 1862 | 1887 | * Count syllables in text (approximate) |
| 1863 | 1888 | * |
| 1889 | + * Expects reading text ({@see self::reading_text_of()}); tags are not | |
| 1890 | + * stripped here, so a decoded "<" in prose is not read as a tag. | |
| 1891 | + * | |
| 1864 | 1892 | * @param string $text Text to analyze |
| 1865 | 1893 | * @return int Syllable count |
| 1866 | 1894 | */ |
| 1867 | 1895 | private function count_syllables(string $text): int { |
| 1868 | - $words = preg_split('/\s+/', trim(strtolower(wp_strip_all_tags($text))), -1, PREG_SPLIT_NO_EMPTY); | |
| 1896 | + $words = preg_split('/\s+/', trim(strtolower($text)), -1, PREG_SPLIT_NO_EMPTY) ?: []; | |
| 1869 | 1897 | $syllables = 0; |
| 1870 | 1898 | |
| 1871 | 1899 | foreach ($words as $word) { |
| 1872 | 1900 | $word = preg_replace('/[^a-z]/', '', $word); |
| @@ -2476,11 +2504,12 @@ | ||
| 2476 | 2504 | |
| 2477 | 2505 | // Extract headings from content |
| 2478 | 2506 | $headings = $this->extract_headings($content); |
| 2479 | 2507 | |
| 2480 | - // Count words using JavaScript-compatible method | |
| 2481 | - $plain_text = wp_strip_all_tags($content); | |
| 2482 | - $word_count = $this->calculate_word_count_js_style($plain_text); | |
| 2508 | + // Count the markup, not wp_strip_all_tags() output: stripping deletes | |
| 2509 | + // tags without a space, so "five</p><p>six" would already be one word | |
| 2510 | + // before the counter saw it. | |
| 2511 | + $word_count = $this->calculate_word_count_js_style($content); | |
| 2483 | 2512 | |
| 2484 | 2513 | // Calculate readability |
| 2485 | 2514 | $readability_score = $this->calculate_readability_score($content); |
| 2486 | 2515 | |
| @@ -2514,22 +2543,59 @@ | ||
| 2514 | 2543 | ]; |
| 2515 | 2544 | } |
| 2516 | 2545 | |
| 2517 | 2546 | /** |
| 2518 | - * Calculate word count using JavaScript-compatible method | |
| 2519 | - * Matches the logic in contentAnalysis.js for consistency | |
| 2547 | + * Count the words a reader reads in HTML or text. | |
| 2520 | 2548 | * |
| 2521 | - * @param string $text Text to count words in | |
| 2549 | + * The same extraction and counting as the thin content report | |
| 2550 | + * ({@see \ThinkRank\SEO\Word_Count_Index::reading_text()} and | |
| 2551 | + * {@see \ThinkRank\SEO\Word_Count_Index::count_words()}), so the editor | |
| 2552 | + * score and the report give one number for one page. Splitting on | |
| 2553 | + * whitespace counted shortcode syntax as words: a WPBakery page with 216 | |
| 2554 | + * words of prose scored a word count of 325 while the report said 216 | |
| 2555 | + * (#893 fixed the report only). It also counted tokens of punctuation | |
| 2556 | + * alone, such as a full stop after a link. | |
| 2557 | + * | |
| 2558 | + * Always words, whatever the locale's unit, because every threshold that | |
| 2559 | + * reads this value (content length, long paragraphs, headings per 300 | |
| 2560 | + * words, readability, keyword density) is in words. | |
| 2561 | + * | |
| 2562 | + * contentAnalysis.js calculateWordCount() applies the same two rules in | |
| 2563 | + * the editor. | |
| 2564 | + * | |
| 2565 | + * @param string $text HTML or text to count words in. | |
| 2522 | 2566 | * @return int Word count |
| 2523 | 2567 | */ |
| 2524 | 2568 | private function calculate_word_count_js_style(string $text): int { |
| 2525 | - if (empty($text)) { | |
| 2569 | + if ('' === $text) { | |
| 2526 | 2570 | return 0; |
| 2527 | 2571 | } |
| 2528 | 2572 | |
| 2529 | - // Match JavaScript: trim, split by whitespace, filter empty | |
| 2530 | - $words = preg_split('/\s+/', trim($text), -1, PREG_SPLIT_NO_EMPTY); | |
| 2531 | - return count($words); | |
| 2573 | + return \ThinkRank\SEO\Word_Count_Index::count_words(self::reading_text_of($text)); | |
| 2574 | + } | |
| 2575 | + | |
| 2576 | + /** | |
| 2577 | + * The text a reader reads in HTML or text, as the word count sees it. | |
| 2578 | + * | |
| 2579 | + * Every check that divides by the word count (readability, keyword | |
| 2580 | + * density, topic relevance) reads its numerator from this same text, so | |
| 2581 | + * shortcode syntax is never on one side of a ratio and not the other. | |
| 2582 | + * | |
| 2583 | + * @since 2.14.2 | |
| 2584 | + * | |
| 2585 | + * @param string $content HTML or text. | |
| 2586 | + * @return string Plain text, whitespace collapsed. | |
| 2587 | + */ | |
| 2588 | + private static function reading_text_of(string $content): string { | |
| 2589 | + if ('' === $content) { | |
| 2590 | + return ''; | |
| 2591 | + } | |
| 2592 | + | |
| 2593 | + if (!class_exists('\ThinkRank\SEO\Word_Count_Index')) { | |
| 2594 | + require_once THINKRANK_PLUGIN_DIR . 'includes/seo/class-word-count-index.php'; | |
| 2595 | + } | |
| 2596 | + | |
| 2597 | + return \ThinkRank\SEO\Word_Count_Index::reading_text($content); | |
| 2532 | 2598 | } |
| 2533 | 2599 | |
| 2534 | 2600 | /** |
| 2535 | 2601 | * Get existing score data for a post from database |