PluginProbe
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO / 2.14.1
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO v2.14.1
2.14.2 2.14.1 2.14.0 2.13.0 2.12.0 2.11.0 2.10.0 2.9.0 2.8.0 2.7.0 2.6.0 2.5.0 2.4.0 2.3.0 2.2.0 2.1.1 2.1.0 2.0.2 2.0.1 2.0.0 1.32.0 1.31.0 1.30.0 1.29.0 1.28.0 All 57 releases
← All changes | includes/seo/class-word-count-index.php +11 -101 2.14.2 → 2.14.1 View file →
@@ -62,19 +62,10 @@
62 62 public const META_KEY = '_thinkrank_word_count';
63 63
64 64 /**
65 65 * Counting rules version. Bump to retire every stored entry.
66 - *
67 - * Version 3 stopped counting shortcode syntax as words and stopped merging
68 - * two words that only a tag separated (#893), so every version 2 entry is
69 - * recounted once.
70 - *
71 - * Version 4 stopped counting tokens with no letter or digit in them. The
72 - * tag-to-space change in version 3 left the punctuation after an inline
73 - * tag ("<a>link</a>.") standing alone, where it was counted as a word, so
74 - * every version 3 entry is recounted once.
75 66 */
76 - public const VERSION = 4;
67 + public const VERSION = 2;
77 68
78 69 /**
79 70 * Short codes for the counting unit, as stored in an entry.
80 71 *
@@ -543,9 +534,17 @@
543 534 * @param string $content HTML or text.
544 535 * @return int
545 536 */
546 537 public static function count_text(string $content): int {
547 - $text = self::reading_text($content);
538 + // Scripts and styles carry no reading matter, and a page builder's
539 + // output can hold a great deal of both. Counting them would make an
540 + // empty page look substantial, which is the failure that matters here.
541 + $text = preg_replace('#<(script|style)\b[^>]*>.*?</\1>#is', ' ', $content);
542 + $text = wp_strip_all_tags((string) $text);
543 + $text = html_entity_decode($text, ENT_QUOTES | ENT_HTML5, 'UTF-8');
544 + // Non-breaking spaces are spaces to a reader.
545 + $text = str_replace(["\xc2\xa0", "\xe2\x80\x8b"], ' ', $text);
546 + $text = trim((string) preg_replace('/\s+/u', ' ', $text));
548 547
549 548 if ('' === $text) {
550 549 return 0;
551 550 }
@@ -557,99 +556,10 @@
557 556 case 'characters_excluding_spaces':
558 557 return mb_strlen(str_replace(' ', '', $text));
559 558
560 559 default:
561 - return self::count_words($text);
560 + return count(preg_split('/\s+/u', $text, -1, PREG_SPLIT_NO_EMPTY) ?: []);
562 561 }
563 - }
564 -
565 - /**
566 - * Count the words in a line of reading text.
567 - *
568 - * A word is a whitespace-separated token holding at least one letter or
569 - * digit, in any script. A token of punctuation or symbols alone is not
570 - * one: reading_text() turns every tag into a space, so the full stop in
571 - * "<a>link</a>." and the "৳" in WooCommerce's
572 - * "<span>৳</span>100" stand on their own, and a dash set between spaces
573 - * ("one — two") does the same in plain prose. Core's JS word counter drops
574 - * punctuation too. Letters and digits are matched by Unicode property, so
575 - * Bengali, Arabic or Cyrillic words still count; a combining mark is part
576 - * of the letter before it, so a Bengali word keeps its vowel signs.
577 - *
578 - * Only the words unit uses this. The character units count every
579 - * character, as core does, which is how CJK locales are counted.
580 - *
581 - * @since 2.14.2
582 - *
583 - * @param string $text Text as returned by {@see self::reading_text()}.
584 - * @return int
585 - */
586 - public static function count_words(string $text): int {
587 - $tokens = preg_split('/\s+/u', $text, -1, PREG_SPLIT_NO_EMPTY);
588 -
589 - if (!is_array($tokens)) {
590 - return 0;
591 - }
592 -
593 - return count(preg_grep('/[\p{L}\p{N}]/u', $tokens) ?: []);
594 - }
595 -
596 - /**
597 - * The text a reader reads in a piece of stored content, as one line.
598 - *
599 - * Everything that is markup rather than reading matter is removed:
600 - *
601 - * - **Scripts and styles.** A page builder's output can hold a great deal
602 - * of both. Counting them would make an empty page look substantial,
603 - * which is the failure that matters here.
604 - * - **Shortcode syntax** (#893). `[vc_column width="1/2"]` is not three
605 - * words, and a WPBakery or Divi classic page is mostly made of it, so
606 - * counting it reported a 216-word page as 325 and called it not thin.
607 - * The shortcodes are stripped, not rendered: `do_shortcode()` would
608 - * count what they output more accurately, but it means executing every
609 - * shortcode on the site inside a batch count, with whatever side effects
610 - * each one has (#860, #864). A shortcode that renders real prose is
611 - * undercounted, which is the safe direction for a thin content report.
612 - * Anything shortcode-shaped is stripped, registered or not, because a
613 - * builder's shortcodes are often not registered when the count runs.
614 - * Bracketed prose that looks like one ("see [note 4]") goes with it;
615 - * "[1]" and "[...]" do not, since a shortcode name starts with a letter.
616 - * - **Tags**, replaced with a space rather than deleted, the way core's
617 - * own word counter does, so `<p>five</p><p>six</p>` stays two words.
618 - *
619 - * @since 2.14.2
620 - *
621 - * @param string $content HTML or text.
622 - * @return string Plain text with whitespace collapsed to single spaces.
623 - */
624 - public static function reading_text(string $content): string {
625 - // A `/u` pattern answers null on bytes that are not valid UTF-8, and
626 - // the string casts below turned that into "": one Latin-1 byte from an
627 - // old import made the whole page read as empty, a word count of 0 in
628 - // both this report and the SEO score. Replace the bad bytes instead.
629 - if ('' !== $content && 1 !== preg_match('//u', $content) && function_exists('mb_scrub')) {
630 - $content = mb_scrub($content, 'UTF-8');
631 - }
632 -
633 - $text = (string) preg_replace('#<(script|style)\b[^>]*>.*?</\1>#is', ' ', $content);
634 -
635 - // Before tags: an attribute value may hold a ">", which would end a
636 - // tag match early. WordPress does not allow "[" or "]" inside a
637 - // shortcode's attributes, so neither is crossed; that also keeps a
638 - // stray "[" in prose from swallowing the text after it. The optional
639 - // outer brackets take the escaped form "[[name]]" whole, rather than
640 - // leaving two stray brackets to be counted as words.
641 - $text = (string) preg_replace('/\[?\[\/?[A-Za-z][\w-]*[^\[\]]*\]\]?/u', ' ', $text);
642 -
643 - $text = (string) preg_replace('/<!--.*?-->/s', ' ', $text);
644 - $text = (string) preg_replace('#</?[A-Za-z][^>]*>#', ' ', $text);
645 - // Backstop for anything malformed the patterns above did not take.
646 - $text = wp_strip_all_tags($text);
647 - $text = html_entity_decode($text, ENT_QUOTES | ENT_HTML5, 'UTF-8');
648 - // Non-breaking spaces are spaces to a reader.
649 - $text = str_replace(["\xc2\xa0", "\xe2\x80\x8b"], ' ', $text);
650 -
651 - return trim((string) preg_replace('/\s+/u', ' ', $text));
652 562 }
653 563
654 564 /**
655 565 * The unit the site counts in.