| @@ -62,19 +62,10 @@ | ||
| 62 | 62 | public const META_KEY = '_thinkrank_word_count'; |
| 63 | 63 | |
| 64 | 64 | /** |
| 65 | 65 | * Counting rules version. Bump to retire every stored entry. |
| 66 | - * | |
| 67 | - * Version 3 stopped counting shortcode syntax as words and stopped merging | |
| 68 | - * two words that only a tag separated (#893), so every version 2 entry is | |
| 69 | - * recounted once. | |
| 70 | - * | |
| 71 | - * Version 4 stopped counting tokens with no letter or digit in them. The | |
| 72 | - * tag-to-space change in version 3 left the punctuation after an inline | |
| 73 | - * tag ("<a>link</a>.") standing alone, where it was counted as a word, so | |
| 74 | - * every version 3 entry is recounted once. | |
| 75 | 66 | */ |
| 76 | - public const VERSION = 4; | |
| 67 | + public const VERSION = 2; | |
| 77 | 68 | |
| 78 | 69 | /** |
| 79 | 70 | * Short codes for the counting unit, as stored in an entry. |
| 80 | 71 | * |
| @@ -543,9 +534,17 @@ | ||
| 543 | 534 | * @param string $content HTML or text. |
| 544 | 535 | * @return int |
| 545 | 536 | */ |
| 546 | 537 | public static function count_text(string $content): int { |
| 547 | - $text = self::reading_text($content); | |
| 538 | + // Scripts and styles carry no reading matter, and a page builder's | |
| 539 | + // output can hold a great deal of both. Counting them would make an | |
| 540 | + // empty page look substantial, which is the failure that matters here. | |
| 541 | + $text = preg_replace('#<(script|style)\b[^>]*>.*?</\1>#is', ' ', $content); | |
| 542 | + $text = wp_strip_all_tags((string) $text); | |
| 543 | + $text = html_entity_decode($text, ENT_QUOTES | ENT_HTML5, 'UTF-8'); | |
| 544 | + // Non-breaking spaces are spaces to a reader. | |
| 545 | + $text = str_replace(["\xc2\xa0", "\xe2\x80\x8b"], ' ', $text); | |
| 546 | + $text = trim((string) preg_replace('/\s+/u', ' ', $text)); | |
| 548 | 547 | |
| 549 | 548 | if ('' === $text) { |
| 550 | 549 | return 0; |
| 551 | 550 | } |
| @@ -557,99 +556,10 @@ | ||
| 557 | 556 | case 'characters_excluding_spaces': |
| 558 | 557 | return mb_strlen(str_replace(' ', '', $text)); |
| 559 | 558 | |
| 560 | 559 | default: |
| 561 | - return self::count_words($text); | |
| 560 | + return count(preg_split('/\s+/u', $text, -1, PREG_SPLIT_NO_EMPTY) ?: []); | |
| 562 | 561 | } |
| 563 | - } | |
| 564 | - | |
| 565 | - /** | |
| 566 | - * Count the words in a line of reading text. | |
| 567 | - * | |
| 568 | - * A word is a whitespace-separated token holding at least one letter or | |
| 569 | - * digit, in any script. A token of punctuation or symbols alone is not | |
| 570 | - * one: reading_text() turns every tag into a space, so the full stop in | |
| 571 | - * "<a>link</a>." and the "৳" in WooCommerce's | |
| 572 | - * "<span>৳</span>100" stand on their own, and a dash set between spaces | |
| 573 | - * ("one — two") does the same in plain prose. Core's JS word counter drops | |
| 574 | - * punctuation too. Letters and digits are matched by Unicode property, so | |
| 575 | - * Bengali, Arabic or Cyrillic words still count; a combining mark is part | |
| 576 | - * of the letter before it, so a Bengali word keeps its vowel signs. | |
| 577 | - * | |
| 578 | - * Only the words unit uses this. The character units count every | |
| 579 | - * character, as core does, which is how CJK locales are counted. | |
| 580 | - * | |
| 581 | - * @since 2.14.2 | |
| 582 | - * | |
| 583 | - * @param string $text Text as returned by {@see self::reading_text()}. | |
| 584 | - * @return int | |
| 585 | - */ | |
| 586 | - public static function count_words(string $text): int { | |
| 587 | - $tokens = preg_split('/\s+/u', $text, -1, PREG_SPLIT_NO_EMPTY); | |
| 588 | - | |
| 589 | - if (!is_array($tokens)) { | |
| 590 | - return 0; | |
| 591 | - } | |
| 592 | - | |
| 593 | - return count(preg_grep('/[\p{L}\p{N}]/u', $tokens) ?: []); | |
| 594 | - } | |
| 595 | - | |
| 596 | - /** | |
| 597 | - * The text a reader reads in a piece of stored content, as one line. | |
| 598 | - * | |
| 599 | - * Everything that is markup rather than reading matter is removed: | |
| 600 | - * | |
| 601 | - * - **Scripts and styles.** A page builder's output can hold a great deal | |
| 602 | - * of both. Counting them would make an empty page look substantial, | |
| 603 | - * which is the failure that matters here. | |
| 604 | - * - **Shortcode syntax** (#893). `[vc_column width="1/2"]` is not three | |
| 605 | - * words, and a WPBakery or Divi classic page is mostly made of it, so | |
| 606 | - * counting it reported a 216-word page as 325 and called it not thin. | |
| 607 | - * The shortcodes are stripped, not rendered: `do_shortcode()` would | |
| 608 | - * count what they output more accurately, but it means executing every | |
| 609 | - * shortcode on the site inside a batch count, with whatever side effects | |
| 610 | - * each one has (#860, #864). A shortcode that renders real prose is | |
| 611 | - * undercounted, which is the safe direction for a thin content report. | |
| 612 | - * Anything shortcode-shaped is stripped, registered or not, because a | |
| 613 | - * builder's shortcodes are often not registered when the count runs. | |
| 614 | - * Bracketed prose that looks like one ("see [note 4]") goes with it; | |
| 615 | - * "[1]" and "[...]" do not, since a shortcode name starts with a letter. | |
| 616 | - * - **Tags**, replaced with a space rather than deleted, the way core's | |
| 617 | - * own word counter does, so `<p>five</p><p>six</p>` stays two words. | |
| 618 | - * | |
| 619 | - * @since 2.14.2 | |
| 620 | - * | |
| 621 | - * @param string $content HTML or text. | |
| 622 | - * @return string Plain text with whitespace collapsed to single spaces. | |
| 623 | - */ | |
| 624 | - public static function reading_text(string $content): string { | |
| 625 | - // A `/u` pattern answers null on bytes that are not valid UTF-8, and | |
| 626 | - // the string casts below turned that into "": one Latin-1 byte from an | |
| 627 | - // old import made the whole page read as empty, a word count of 0 in | |
| 628 | - // both this report and the SEO score. Replace the bad bytes instead. | |
| 629 | - if ('' !== $content && 1 !== preg_match('//u', $content) && function_exists('mb_scrub')) { | |
| 630 | - $content = mb_scrub($content, 'UTF-8'); | |
| 631 | - } | |
| 632 | - | |
| 633 | - $text = (string) preg_replace('#<(script|style)\b[^>]*>.*?</\1>#is', ' ', $content); | |
| 634 | - | |
| 635 | - // Before tags: an attribute value may hold a ">", which would end a | |
| 636 | - // tag match early. WordPress does not allow "[" or "]" inside a | |
| 637 | - // shortcode's attributes, so neither is crossed; that also keeps a | |
| 638 | - // stray "[" in prose from swallowing the text after it. The optional | |
| 639 | - // outer brackets take the escaped form "[[name]]" whole, rather than | |
| 640 | - // leaving two stray brackets to be counted as words. | |
| 641 | - $text = (string) preg_replace('/\[?\[\/?[A-Za-z][\w-]*[^\[\]]*\]\]?/u', ' ', $text); | |
| 642 | - | |
| 643 | - $text = (string) preg_replace('/<!--.*?-->/s', ' ', $text); | |
| 644 | - $text = (string) preg_replace('#</?[A-Za-z][^>]*>#', ' ', $text); | |
| 645 | - // Backstop for anything malformed the patterns above did not take. | |
| 646 | - $text = wp_strip_all_tags($text); | |
| 647 | - $text = html_entity_decode($text, ENT_QUOTES | ENT_HTML5, 'UTF-8'); | |
| 648 | - // Non-breaking spaces are spaces to a reader. | |
| 649 | - $text = str_replace(["\xc2\xa0", "\xe2\x80\x8b"], ' ', $text); | |
| 650 | - | |
| 651 | - return trim((string) preg_replace('/\s+/u', ' ', $text)); | |
| 652 | 562 | } |
| 653 | 563 | |
| 654 | 564 | /** |
| 655 | 565 | * The unit the site counts in. |