| @@ -62,10 +62,19 @@ | ||
| 62 | 62 | public const META_KEY = '_thinkrank_word_count'; |
| 63 | 63 | |
| 64 | 64 | /** |
| 65 | 65 | * Counting rules version. Bump to retire every stored entry. |
| 66 | + * | |
| 67 | + * Version 3 stopped counting shortcode syntax as words and stopped merging | |
| 68 | + * two words that only a tag separated (#893), so every version 2 entry is | |
| 69 | + * recounted once. | |
| 70 | + * | |
| 71 | + * Version 4 stopped counting tokens with no letter or digit in them. The | |
| 72 | + * tag-to-space change in version 3 left the punctuation after an inline | |
| 73 | + * tag ("<a>link</a>.") standing alone, where it was counted as a word, so | |
| 74 | + * every version 3 entry is recounted once. | |
| 66 | 75 | */ |
| 67 | - public const VERSION = 2; | |
| 76 | + public const VERSION = 4; | |
| 68 | 77 | |
| 69 | 78 | /** |
| 70 | 79 | * Short codes for the counting unit, as stored in an entry. |
| 71 | 80 | * |
| @@ -534,17 +543,9 @@ | ||
| 534 | 543 | * @param string $content HTML or text. |
| 535 | 544 | * @return int |
| 536 | 545 | */ |
| 537 | 546 | public static function count_text(string $content): int { |
| 538 | - // Scripts and styles carry no reading matter, and a page builder's | |
| 539 | - // output can hold a great deal of both. Counting them would make an | |
| 540 | - // empty page look substantial, which is the failure that matters here. | |
| 541 | - $text = preg_replace('#<(script|style)\b[^>]*>.*?</\1>#is', ' ', $content); | |
| 542 | - $text = wp_strip_all_tags((string) $text); | |
| 543 | - $text = html_entity_decode($text, ENT_QUOTES | ENT_HTML5, 'UTF-8'); | |
| 544 | - // Non-breaking spaces are spaces to a reader. | |
| 545 | - $text = str_replace(["\xc2\xa0", "\xe2\x80\x8b"], ' ', $text); | |
| 546 | - $text = trim((string) preg_replace('/\s+/u', ' ', $text)); | |
| 547 | + $text = self::reading_text($content); | |
| 547 | 548 | |
| 548 | 549 | if ('' === $text) { |
| 549 | 550 | return 0; |
| 550 | 551 | } |
| @@ -556,10 +557,99 @@ | ||
| 556 | 557 | case 'characters_excluding_spaces': |
| 557 | 558 | return mb_strlen(str_replace(' ', '', $text)); |
| 558 | 559 | |
| 559 | 560 | default: |
| 560 | - return count(preg_split('/\s+/u', $text, -1, PREG_SPLIT_NO_EMPTY) ?: []); | |
| 561 | + return self::count_words($text); | |
| 561 | 562 | } |
| 563 | + } | |
| 564 | + | |
| 565 | + /** | |
| 566 | + * Count the words in a line of reading text. | |
| 567 | + * | |
| 568 | + * A word is a whitespace-separated token holding at least one letter or | |
| 569 | + * digit, in any script. A token of punctuation or symbols alone is not | |
| 570 | + * one: reading_text() turns every tag into a space, so the full stop in | |
| 571 | + * "<a>link</a>." and the "৳" in WooCommerce's | |
| 572 | + * "<span>৳</span>100" stand on their own, and a dash set between spaces | |
| 573 | + * ("one — two") does the same in plain prose. Core's JS word counter drops | |
| 574 | + * punctuation too. Letters and digits are matched by Unicode property, so | |
| 575 | + * Bengali, Arabic or Cyrillic words still count; a combining mark is part | |
| 576 | + * of the letter before it, so a Bengali word keeps its vowel signs. | |
| 577 | + * | |
| 578 | + * Only the words unit uses this. The character units count every | |
| 579 | + * character, as core does, which is how CJK locales are counted. | |
| 580 | + * | |
| 581 | + * @since 2.14.2 | |
| 582 | + * | |
| 583 | + * @param string $text Text as returned by {@see self::reading_text()}. | |
| 584 | + * @return int | |
| 585 | + */ | |
| 586 | + public static function count_words(string $text): int { | |
| 587 | + $tokens = preg_split('/\s+/u', $text, -1, PREG_SPLIT_NO_EMPTY); | |
| 588 | + | |
| 589 | + if (!is_array($tokens)) { | |
| 590 | + return 0; | |
| 591 | + } | |
| 592 | + | |
| 593 | + return count(preg_grep('/[\p{L}\p{N}]/u', $tokens) ?: []); | |
| 594 | + } | |
| 595 | + | |
| 596 | + /** | |
| 597 | + * The text a reader reads in a piece of stored content, as one line. | |
| 598 | + * | |
| 599 | + * Everything that is markup rather than reading matter is removed: | |
| 600 | + * | |
| 601 | + * - **Scripts and styles.** A page builder's output can hold a great deal | |
| 602 | + * of both. Counting them would make an empty page look substantial, | |
| 603 | + * which is the failure that matters here. | |
| 604 | + * - **Shortcode syntax** (#893). `[vc_column width="1/2"]` is not three | |
| 605 | + * words, and a WPBakery or Divi classic page is mostly made of it, so | |
| 606 | + * counting it reported a 216-word page as 325 and called it not thin. | |
| 607 | + * The shortcodes are stripped, not rendered: `do_shortcode()` would | |
| 608 | + * count what they output more accurately, but it means executing every | |
| 609 | + * shortcode on the site inside a batch count, with whatever side effects | |
| 610 | + * each one has (#860, #864). A shortcode that renders real prose is | |
| 611 | + * undercounted, which is the safe direction for a thin content report. | |
| 612 | + * Anything shortcode-shaped is stripped, registered or not, because a | |
| 613 | + * builder's shortcodes are often not registered when the count runs. | |
| 614 | + * Bracketed prose that looks like one ("see [note 4]") goes with it; | |
| 615 | + * "[1]" and "[...]" do not, since a shortcode name starts with a letter. | |
| 616 | + * - **Tags**, replaced with a space rather than deleted, the way core's | |
| 617 | + * own word counter does, so `<p>five</p><p>six</p>` stays two words. | |
| 618 | + * | |
| 619 | + * @since 2.14.2 | |
| 620 | + * | |
| 621 | + * @param string $content HTML or text. | |
| 622 | + * @return string Plain text with whitespace collapsed to single spaces. | |
| 623 | + */ | |
| 624 | + public static function reading_text(string $content): string { | |
| 625 | + // A `/u` pattern answers null on bytes that are not valid UTF-8, and | |
| 626 | + // the string casts below turned that into "": one Latin-1 byte from an | |
| 627 | + // old import made the whole page read as empty, a word count of 0 in | |
| 628 | + // both this report and the SEO score. Replace the bad bytes instead. | |
| 629 | + if ('' !== $content && 1 !== preg_match('//u', $content) && function_exists('mb_scrub')) { | |
| 630 | + $content = mb_scrub($content, 'UTF-8'); | |
| 631 | + } | |
| 632 | + | |
| 633 | + $text = (string) preg_replace('#<(script|style)\b[^>]*>.*?</\1>#is', ' ', $content); | |
| 634 | + | |
| 635 | + // Before tags: an attribute value may hold a ">", which would end a | |
| 636 | + // tag match early. WordPress does not allow "[" or "]" inside a | |
| 637 | + // shortcode's attributes, so neither is crossed; that also keeps a | |
| 638 | + // stray "[" in prose from swallowing the text after it. The optional | |
| 639 | + // outer brackets take the escaped form "[[name]]" whole, rather than | |
| 640 | + // leaving two stray brackets to be counted as words. | |
| 641 | + $text = (string) preg_replace('/\[?\[\/?[A-Za-z][\w-]*[^\[\]]*\]\]?/u', ' ', $text); | |
| 642 | + | |
| 643 | + $text = (string) preg_replace('/<!--.*?-->/s', ' ', $text); | |
| 644 | + $text = (string) preg_replace('#</?[A-Za-z][^>]*>#', ' ', $text); | |
| 645 | + // Backstop for anything malformed the patterns above did not take. | |
| 646 | + $text = wp_strip_all_tags($text); | |
| 647 | + $text = html_entity_decode($text, ENT_QUOTES | ENT_HTML5, 'UTF-8'); | |
| 648 | + // Non-breaking spaces are spaces to a reader. | |
| 649 | + $text = str_replace(["\xc2\xa0", "\xe2\x80\x8b"], ' ', $text); | |
| 650 | + | |
| 651 | + return trim((string) preg_replace('/\s+/u', ' ', $text)); | |
| 562 | 652 | } |
| 563 | 653 | |
| 564 | 654 | /** |
| 565 | 655 | * The unit the site counts in. |