| @@ -1263,14 +1263,16 @@ | ||
| 1263 | 1263 | if (empty($content) || empty($target_keyword)) { |
| 1264 | 1264 | return 0.0; |
| 1265 | 1265 | } |
| 1266 | 1266 | |
| 1267 | - $content_lower = strtolower(wp_strip_all_tags($content)); | |
| 1267 | + // Occurrences and the word count both come from the reading text, so | |
| 1268 | + // shortcode syntax is in neither. | |
| 1269 | + $content_lower = strtolower(self::reading_text_of($content)); | |
| 1268 | 1270 | $keyword_lower = strtolower($target_keyword); |
| 1269 | 1271 | |
| 1270 | 1272 | // Calculate keyword and semantic term frequency |
| 1271 | 1273 | $keyword_count = substr_count($content_lower, $keyword_lower); |
| 1272 | - $word_count = $this->calculate_word_count_js_style($content_lower); | |
| 1274 | + $word_count = \ThinkRank\SEO\Word_Count_Index::count_words($content_lower); | |
| 1273 | 1275 | |
| 1274 | 1276 | if ($word_count === 0) { |
| 1275 | 1277 | return 0.0; |
| 1276 | 1278 | } |
| @@ -1349,10 +1351,13 @@ | ||
| 1349 | 1351 | private function keyword_density(string $content, string $target_keyword): float { |
| 1350 | 1352 | if (empty($content) || empty($target_keyword)) { |
| 1351 | 1353 | return 0.0; |
| 1352 | 1354 | } |
| 1353 | - $plain = strtolower(wp_strip_all_tags($content)); | |
| 1354 | - $word_count = $this->calculate_word_count_js_style($plain); | |
| 1355 | + // Numerator and denominator from the same reading text. Counting the | |
| 1356 | + // keyword in wp_strip_all_tags() output found it inside shortcode | |
| 1357 | + // attributes the word count no longer includes. | |
| 1358 | + $plain = strtolower(self::reading_text_of($content)); | |
| 1359 | + $word_count = \ThinkRank\SEO\Word_Count_Index::count_words($plain); | |
| 1355 | 1360 | if ($word_count === 0) { |
| 1356 | 1361 | return 0.0; |
| 1357 | 1362 | } |
| 1358 | 1363 | $occurrences = substr_count($plain, strtolower($target_keyword)); |
| @@ -1387,10 +1392,10 @@ | ||
| 1387 | 1392 | private function keyword_in_first_paragraph(string $content, string $target_keyword): bool { |
| 1388 | 1393 | if (empty($content) || empty($target_keyword)) { |
| 1389 | 1394 | return false; |
| 1390 | 1395 | } |
| 1391 | - $plain = strtolower(wp_strip_all_tags($content)); | |
| 1392 | - $words = preg_split('/\s+/', trim($plain), -1, PREG_SPLIT_NO_EMPTY) ?: []; | |
| 1396 | + $plain = strtolower(self::reading_text_of($content)); | |
| 1397 | + $words = preg_split('/\s+/', $plain, -1, PREG_SPLIT_NO_EMPTY) ?: []; | |
| 1393 | 1398 | $window = array_slice($words, 0, max(50, (int) ceil(count($words) * 0.1))); |
| 1394 | 1399 | return strpos(implode(' ', $window), strtolower($target_keyword)) !== false; |
| 1395 | 1400 | } |
| 1396 | 1401 | |
| @@ -1620,9 +1625,24 @@ | ||
| 1620 | 1625 | if (!class_exists('\ThinkRank\SEO\Builder_Content')) { |
| 1621 | 1626 | require_once THINKRANK_PLUGIN_DIR . 'includes/seo/class-builder-content.php'; |
| 1622 | 1627 | } |
| 1623 | 1628 | |
| 1624 | - return \ThinkRank\SEO\Builder_Content::resolve_markup($live_content, $post); | |
| 1629 | + // Bind the live markup to the post the render makes current. Since #862 | |
| 1630 | + // the resolver runs setup_postdata() on this post, so a shortcode that | |
| 1631 | + // builds its output from the current post's content — get_the_content(), | |
| 1632 | + // get_post()->post_content, as a table of contents or a reading-time | |
| 1633 | + // shortcode does — otherwise read the last saved body while the unsaved | |
| 1634 | + // markup rendered around it, and lagged a save behind (#864). | |
| 1635 | + // | |
| 1636 | + // A clone, not the caller's object: the post is current only for the | |
| 1637 | + // duration of the render and the caller's $post must come back | |
| 1638 | + // unchanged. `thinkrank_analyzable_content` receives the clone too, | |
| 1639 | + // which is what makes $post->post_content there agree with the markup | |
| 1640 | + // being analyzed on the live path. | |
| 1641 | + $bound = clone $post; | |
| 1642 | + $bound->post_content = $live_content; | |
| 1643 | + | |
| 1644 | + return \ThinkRank\SEO\Builder_Content::resolve_markup($live_content, $bound); | |
| 1625 | 1645 | } |
| 1626 | 1646 | |
| 1627 | 1647 | public function analyze_post_content(int $post_id): array { |
| 1628 | 1648 | $post = get_post($post_id); |
| @@ -1635,11 +1655,12 @@ | ||
| 1635 | 1655 | |
| 1636 | 1656 | // Extract headings from content |
| 1637 | 1657 | $headings = $this->extract_headings($content); |
| 1638 | 1658 | |
| 1639 | - // Count words using JavaScript-compatible method | |
| 1640 | - $plain_text = wp_strip_all_tags($content); | |
| 1641 | - $word_count = $this->calculate_word_count_js_style($plain_text); | |
| 1659 | + // Count the markup, not wp_strip_all_tags() output: stripping deletes | |
| 1660 | + // tags without a space, so "five</p><p>six" would already be one word | |
| 1661 | + // before the counter saw it. | |
| 1662 | + $word_count = $this->calculate_word_count_js_style($content); | |
| 1642 | 1663 | |
| 1643 | 1664 | // Calculate readability |
| 1644 | 1665 | $readability_score = $this->calculate_readability_score($content); |
| 1645 | 1666 | |
| @@ -1759,9 +1780,9 @@ | ||
| 1759 | 1780 | // 2. Long paragraphs (flag paragraphs over 150 words). |
| 1760 | 1781 | $long_paragraphs = 0; |
| 1761 | 1782 | if (preg_match_all('/<p[^>]*>(.*?)<\/p>/is', $content, $matches)) { |
| 1762 | 1783 | foreach ($matches[1] as $paragraph) { |
| 1763 | - if ($this->calculate_word_count_js_style(wp_strip_all_tags($paragraph)) > 150) { | |
| 1784 | + if ($this->calculate_word_count_js_style($paragraph) > 150) { | |
| 1764 | 1785 | $long_paragraphs++; |
| 1765 | 1786 | } |
| 1766 | 1787 | } |
| 1767 | 1788 | } |
| @@ -1816,22 +1837,22 @@ | ||
| 1816 | 1837 | * @param string $content Content text |
| 1817 | 1838 | * @return float Readability score |
| 1818 | 1839 | */ |
| 1819 | 1840 | private function calculate_readability_score(string $content): float { |
| 1820 | - $text = wp_strip_all_tags($content); | |
| 1841 | + // Sentences, words and syllables all come from one reading text. The | |
| 1842 | + // word count drops shortcode syntax and punctuation-only tokens; when | |
| 1843 | + // sentences and syllables were still read off wp_strip_all_tags() | |
| 1844 | + // output, "[vc_column width="1/2"]" added syllables (and, with a "." in | |
| 1845 | + // an attribute, sentences) to a word count that did not include it, | |
| 1846 | + // and a WPBakery page's Flesch fell from 65 to 46. | |
| 1847 | + $text = self::reading_text_of($content); | |
| 1821 | 1848 | |
| 1822 | - if (empty($text)) { | |
| 1849 | + if ('' === $text) { | |
| 1823 | 1850 | return 0; |
| 1824 | 1851 | } |
| 1825 | 1852 | |
| 1826 | - // Count sentences (approximate) | |
| 1827 | - $sentences = preg_split('/[.!?]+/', $text, -1, PREG_SPLIT_NO_EMPTY); | |
| 1828 | - $sentence_count = count($sentences); | |
| 1829 | - | |
| 1830 | - // Count words | |
| 1831 | - $word_count = $this->calculate_word_count_js_style(wp_strip_all_tags($text)); | |
| 1832 | - | |
| 1833 | - // Count syllables (approximate) | |
| 1853 | + $sentence_count = self::count_sentences($text); | |
| 1854 | + $word_count = \ThinkRank\SEO\Word_Count_Index::count_words($text); | |
| 1834 | 1855 | $syllable_count = $this->count_syllables($text); |
| 1835 | 1856 | |
| 1836 | 1857 | if ($sentence_count === 0 || $word_count === 0) { |
| 1837 | 1858 | return 0; |
| @@ -1843,15 +1864,37 @@ | ||
| 1843 | 1864 | return max(0, min(100, $score)); |
| 1844 | 1865 | } |
| 1845 | 1866 | |
| 1846 | 1867 | /** |
| 1868 | + * Count the sentences in reading text. | |
| 1869 | + * | |
| 1870 | + * A sentence is a run of text between terminal punctuation that holds at | |
| 1871 | + * least one letter or digit. A fragment with no word in it (the space | |
| 1872 | + * after the final full stop, a stray "!" between two "?") is not a | |
| 1873 | + * sentence. contentAnalysis.js countSentences() applies the same rule. | |
| 1874 | + * | |
| 1875 | + * @since 2.14.2 | |
| 1876 | + * | |
| 1877 | + * @param string $text Text as returned by {@see self::reading_text_of()}. | |
| 1878 | + * @return int | |
| 1879 | + */ | |
| 1880 | + private static function count_sentences(string $text): int { | |
| 1881 | + $fragments = preg_split('/[.!?]+/', $text, -1, PREG_SPLIT_NO_EMPTY) ?: []; | |
| 1882 | + | |
| 1883 | + return count(preg_grep('/[\p{L}\p{N}]/u', $fragments) ?: []); | |
| 1884 | + } | |
| 1885 | + | |
| 1886 | + /** | |
| 1847 | 1887 | * Count syllables in text (approximate) |
| 1848 | 1888 | * |
| 1889 | + * Expects reading text ({@see self::reading_text_of()}); tags are not | |
| 1890 | + * stripped here, so a decoded "<" in prose is not read as a tag. | |
| 1891 | + * | |
| 1849 | 1892 | * @param string $text Text to analyze |
| 1850 | 1893 | * @return int Syllable count |
| 1851 | 1894 | */ |
| 1852 | 1895 | private function count_syllables(string $text): int { |
| 1853 | - $words = preg_split('/\s+/', trim(strtolower(wp_strip_all_tags($text))), -1, PREG_SPLIT_NO_EMPTY); | |
| 1896 | + $words = preg_split('/\s+/', trim(strtolower($text)), -1, PREG_SPLIT_NO_EMPTY) ?: []; | |
| 1854 | 1897 | $syllables = 0; |
| 1855 | 1898 | |
| 1856 | 1899 | foreach ($words as $word) { |
| 1857 | 1900 | $word = preg_replace('/[^a-z]/', '', $word); |
| @@ -2461,11 +2504,12 @@ | ||
| 2461 | 2504 | |
| 2462 | 2505 | // Extract headings from content |
| 2463 | 2506 | $headings = $this->extract_headings($content); |
| 2464 | 2507 | |
| 2465 | - // Count words using JavaScript-compatible method | |
| 2466 | - $plain_text = wp_strip_all_tags($content); | |
| 2467 | - $word_count = $this->calculate_word_count_js_style($plain_text); | |
| 2508 | + // Count the markup, not wp_strip_all_tags() output: stripping deletes | |
| 2509 | + // tags without a space, so "five</p><p>six" would already be one word | |
| 2510 | + // before the counter saw it. | |
| 2511 | + $word_count = $this->calculate_word_count_js_style($content); | |
| 2468 | 2512 | |
| 2469 | 2513 | // Calculate readability |
| 2470 | 2514 | $readability_score = $this->calculate_readability_score($content); |
| 2471 | 2515 | |
| @@ -2499,22 +2543,59 @@ | ||
| 2499 | 2543 | ]; |
| 2500 | 2544 | } |
| 2501 | 2545 | |
| 2502 | 2546 | /** |
| 2503 | - * Calculate word count using JavaScript-compatible method | |
| 2504 | - * Matches the logic in contentAnalysis.js for consistency | |
| 2547 | + * Count the words a reader reads in HTML or text. | |
| 2505 | 2548 | * |
| 2506 | - * @param string $text Text to count words in | |
| 2549 | + * The same extraction and counting as the thin content report | |
| 2550 | + * ({@see \ThinkRank\SEO\Word_Count_Index::reading_text()} and | |
| 2551 | + * {@see \ThinkRank\SEO\Word_Count_Index::count_words()}), so the editor | |
| 2552 | + * score and the report give one number for one page. Splitting on | |
| 2553 | + * whitespace counted shortcode syntax as words: a WPBakery page with 216 | |
| 2554 | + * words of prose scored a word count of 325 while the report said 216 | |
| 2555 | + * (#893 fixed the report only). It also counted tokens of punctuation | |
| 2556 | + * alone, such as a full stop after a link. | |
| 2557 | + * | |
| 2558 | + * Always words, whatever the locale's unit, because every threshold that | |
| 2559 | + * reads this value (content length, long paragraphs, headings per 300 | |
| 2560 | + * words, readability, keyword density) is in words. | |
| 2561 | + * | |
| 2562 | + * contentAnalysis.js calculateWordCount() applies the same two rules in | |
| 2563 | + * the editor. | |
| 2564 | + * | |
| 2565 | + * @param string $text HTML or text to count words in. | |
| 2507 | 2566 | * @return int Word count |
| 2508 | 2567 | */ |
| 2509 | 2568 | private function calculate_word_count_js_style(string $text): int { |
| 2510 | - if (empty($text)) { | |
| 2569 | + if ('' === $text) { | |
| 2511 | 2570 | return 0; |
| 2512 | 2571 | } |
| 2513 | 2572 | |
| 2514 | - // Match JavaScript: trim, split by whitespace, filter empty | |
| 2515 | - $words = preg_split('/\s+/', trim($text), -1, PREG_SPLIT_NO_EMPTY); | |
| 2516 | - return count($words); | |
| 2573 | + return \ThinkRank\SEO\Word_Count_Index::count_words(self::reading_text_of($text)); | |
| 2574 | + } | |
| 2575 | + | |
| 2576 | + /** | |
| 2577 | + * The text a reader reads in HTML or text, as the word count sees it. | |
| 2578 | + * | |
| 2579 | + * Every check that divides by the word count (readability, keyword | |
| 2580 | + * density, topic relevance) reads its numerator from this same text, so | |
| 2581 | + * shortcode syntax is never on one side of a ratio and not the other. | |
| 2582 | + * | |
| 2583 | + * @since 2.14.2 | |
| 2584 | + * | |
| 2585 | + * @param string $content HTML or text. | |
| 2586 | + * @return string Plain text, whitespace collapsed. | |
| 2587 | + */ | |
| 2588 | + private static function reading_text_of(string $content): string { | |
| 2589 | + if ('' === $content) { | |
| 2590 | + return ''; | |
| 2591 | + } | |
| 2592 | + | |
| 2593 | + if (!class_exists('\ThinkRank\SEO\Word_Count_Index')) { | |
| 2594 | + require_once THINKRANK_PLUGIN_DIR . 'includes/seo/class-word-count-index.php'; | |
| 2595 | + } | |
| 2596 | + | |
| 2597 | + return \ThinkRank\SEO\Word_Count_Index::reading_text($content); | |
| 2517 | 2598 | } |
| 2518 | 2599 | |
| 2519 | 2600 | /** |
| 2520 | 2601 | * Get existing score data for a post from database |