PluginProbe
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO / 2.14.2
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO v2.14.2
2.14.2 2.14.1 2.14.0 2.13.0 2.12.0 2.11.0 2.10.0 2.9.0 2.8.0 2.7.0 2.6.0 2.5.0 2.4.0 2.3.0 2.2.0 2.1.1 2.1.0 2.0.2 2.0.1 2.0.0 1.32.0 1.31.0 1.30.0 1.29.0 1.28.0 All 57 releases
← All changes | includes/seo/class-word-count-index.php +101 -11 2.14.1 → 2.14.2 View file →
@@ -62,10 +62,19 @@
62 62 public const META_KEY = '_thinkrank_word_count';
63 63
64 64 /**
65 65 * Counting rules version. Bump to retire every stored entry.
66 + *
67 + * Version 3 stopped counting shortcode syntax as words and stopped merging
68 + * two words that only a tag separated (#893), so every version 2 entry is
69 + * recounted once.
70 + *
71 + * Version 4 stopped counting tokens with no letter or digit in them. The
72 + * tag-to-space change in version 3 left the punctuation after an inline
73 + * tag ("<a>link</a>.") standing alone, where it was counted as a word, so
74 + * every version 3 entry is recounted once.
66 75 */
67 - public const VERSION = 2;
76 + public const VERSION = 4;
68 77
69 78 /**
70 79 * Short codes for the counting unit, as stored in an entry.
71 80 *
@@ -534,17 +543,9 @@
534 543 * @param string $content HTML or text.
535 544 * @return int
536 545 */
537 546 public static function count_text(string $content): int {
538 - // Scripts and styles carry no reading matter, and a page builder's
539 - // output can hold a great deal of both. Counting them would make an
540 - // empty page look substantial, which is the failure that matters here.
541 - $text = preg_replace('#<(script|style)\b[^>]*>.*?</\1>#is', ' ', $content);
542 - $text = wp_strip_all_tags((string) $text);
543 - $text = html_entity_decode($text, ENT_QUOTES | ENT_HTML5, 'UTF-8');
544 - // Non-breaking spaces are spaces to a reader.
545 - $text = str_replace(["\xc2\xa0", "\xe2\x80\x8b"], ' ', $text);
546 - $text = trim((string) preg_replace('/\s+/u', ' ', $text));
547 + $text = self::reading_text($content);
547 548
548 549 if ('' === $text) {
549 550 return 0;
550 551 }
@@ -556,10 +557,99 @@
556 557 case 'characters_excluding_spaces':
557 558 return mb_strlen(str_replace(' ', '', $text));
558 559
559 560 default:
560 - return count(preg_split('/\s+/u', $text, -1, PREG_SPLIT_NO_EMPTY) ?: []);
561 + return self::count_words($text);
561 562 }
563 + }
564 +
565 + /**
566 + * Count the words in a line of reading text.
567 + *
568 + * A word is a whitespace-separated token holding at least one letter or
569 + * digit, in any script. A token of punctuation or symbols alone is not
570 + * one: reading_text() turns every tag into a space, so the full stop in
571 + * "<a>link</a>." and the "৳" in WooCommerce's
572 + * "<span>৳</span>100" stand on their own, and a dash set between spaces
573 + * ("one — two") does the same in plain prose. Core's JS word counter drops
574 + * punctuation too. Letters and digits are matched by Unicode property, so
575 + * Bengali, Arabic or Cyrillic words still count; a combining mark is part
576 + * of the letter before it, so a Bengali word keeps its vowel signs.
577 + *
578 + * Only the words unit uses this. The character units count every
579 + * character, as core does, which is how CJK locales are counted.
580 + *
581 + * @since 2.14.2
582 + *
583 + * @param string $text Text as returned by {@see self::reading_text()}.
584 + * @return int
585 + */
586 + public static function count_words(string $text): int {
587 + $tokens = preg_split('/\s+/u', $text, -1, PREG_SPLIT_NO_EMPTY);
588 +
589 + if (!is_array($tokens)) {
590 + return 0;
591 + }
592 +
593 + return count(preg_grep('/[\p{L}\p{N}]/u', $tokens) ?: []);
594 + }
595 +
596 + /**
597 + * The text a reader reads in a piece of stored content, as one line.
598 + *
599 + * Everything that is markup rather than reading matter is removed:
600 + *
601 + * - **Scripts and styles.** A page builder's output can hold a great deal
602 + * of both. Counting them would make an empty page look substantial,
603 + * which is the failure that matters here.
604 + * - **Shortcode syntax** (#893). `[vc_column width="1/2"]` is not three
605 + * words, and a WPBakery or Divi classic page is mostly made of it, so
606 + * counting it reported a 216-word page as 325 and called it not thin.
607 + * The shortcodes are stripped, not rendered: `do_shortcode()` would
608 + * count what they output more accurately, but it means executing every
609 + * shortcode on the site inside a batch count, with whatever side effects
610 + * each one has (#860, #864). A shortcode that renders real prose is
611 + * undercounted, which is the safe direction for a thin content report.
612 + * Anything shortcode-shaped is stripped, registered or not, because a
613 + * builder's shortcodes are often not registered when the count runs.
614 + * Bracketed prose that looks like one ("see [note 4]") goes with it;
615 + * "[1]" and "[...]" do not, since a shortcode name starts with a letter.
616 + * - **Tags**, replaced with a space rather than deleted, the way core's
617 + * own word counter does, so `<p>five</p><p>six</p>` stays two words.
618 + *
619 + * @since 2.14.2
620 + *
621 + * @param string $content HTML or text.
622 + * @return string Plain text with whitespace collapsed to single spaces.
623 + */
624 + public static function reading_text(string $content): string {
625 + // A `/u` pattern answers null on bytes that are not valid UTF-8, and
626 + // the string casts below turned that into "": one Latin-1 byte from an
627 + // old import made the whole page read as empty, a word count of 0 in
628 + // both this report and the SEO score. Replace the bad bytes instead.
629 + if ('' !== $content && 1 !== preg_match('//u', $content) && function_exists('mb_scrub')) {
630 + $content = mb_scrub($content, 'UTF-8');
631 + }
632 +
633 + $text = (string) preg_replace('#<(script|style)\b[^>]*>.*?</\1>#is', ' ', $content);
634 +
635 + // Before tags: an attribute value may hold a ">", which would end a
636 + // tag match early. WordPress does not allow "[" or "]" inside a
637 + // shortcode's attributes, so neither is crossed; that also keeps a
638 + // stray "[" in prose from swallowing the text after it. The optional
639 + // outer brackets take the escaped form "[[name]]" whole, rather than
640 + // leaving two stray brackets to be counted as words.
641 + $text = (string) preg_replace('/\[?\[\/?[A-Za-z][\w-]*[^\[\]]*\]\]?/u', ' ', $text);
642 +
643 + $text = (string) preg_replace('/<!--.*?-->/s', ' ', $text);
644 + $text = (string) preg_replace('#</?[A-Za-z][^>]*>#', ' ', $text);
645 + // Backstop for anything malformed the patterns above did not take.
646 + $text = wp_strip_all_tags($text);
647 + $text = html_entity_decode($text, ENT_QUOTES | ENT_HTML5, 'UTF-8');
648 + // Non-breaking spaces are spaces to a reader.
649 + $text = str_replace(["\xc2\xa0", "\xe2\x80\x8b"], ' ', $text);
650 +
651 + return trim((string) preg_replace('/\s+/u', ' ', $text));
562 652 }
563 653
564 654 /**
565 655 * The unit the site counts in.