PluginProbe
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO / 2.10.0
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO v2.10.0
2.10.0 2.9.0 2.8.0 2.7.0 2.6.0 2.5.0 2.4.0 2.3.0 2.2.0 2.1.1 2.1.0 2.0.2 2.0.1 2.0.0 1.32.0 1.31.0 1.30.0 1.29.0 1.28.0 1.27.0 1.26.0 1.25.0 trunk 1.0.0 1.0.1 All 51 releases
← All changes | includes/seo/class-snippet-issues.php +114 -32 2.9.0 → 2.10.0 View file →
@@ -10,8 +10,9 @@
10 10
11 11 namespace ThinkRank\SEO;
12 12
13 13 use ThinkRank\AI\SEOScoreCalculator;
14 +use ThinkRank\Core\Seo_Text;
14 15
15 16 // Prevent direct access
16 17 if (!defined('ABSPATH')) {
17 18 exit;
@@ -49,8 +50,9 @@
49 50 public const DESCRIPTION_TOO_LONG = 'description_too_long';
50 51 public const NO_FOCUS_KEYWORD = 'no_focus_keyword';
51 52 public const NOINDEX = 'noindex';
52 53 public const DUPLICATE_TITLE = 'duplicate_title';
54 + public const DUPLICATE_DESCRIPTION = 'duplicate_description';
53 55
54 56 /**
55 57 * Every issue, in the order the filter chips show them.
56 58 *
@@ -68,17 +70,35 @@
68 70 self::DESCRIPTION_TOO_LONG,
69 71 self::NO_FOCUS_KEYWORD,
70 72 self::NOINDEX,
71 73 self::DUPLICATE_TITLE,
74 + self::DUPLICATE_DESCRIPTION,
72 75 ];
73 76 }
74 77
75 78 /**
79 + * Issues decided by comparing a post against the rest of the site, rather
80 + * than by looking at the post alone.
81 + *
82 + * These carry no bit in the stored flags: the index stores a grouping key
83 + * for each and the duplicates are found by grouping at query time, so
84 + * fixing one of two duplicates clears the other without a write to it.
85 + *
86 + * @since 2.10.0
87 + *
88 + * @return string[]
89 + */
90 + public static function grouped(): array {
91 + return [self::DUPLICATE_TITLE, self::DUPLICATE_DESCRIPTION];
92 + }
93 +
94 + /**
76 95 * Bit per issue that depends on the post alone.
77 96 *
78 - * Duplicate title is not here: it depends on the other posts, so the index
79 - * stores a key for it and finds duplicates by grouping at query time. The
80 - * order is part of the stored format — append, never reorder.
97 + * The grouped issues ({@see self::grouped()}) are not here. They sit last
98 + * in {@see self::all()} so that every other issue keeps the bit it has
99 + * always had — the order is part of the stored format, so append, never
100 + * reorder.
81 101 *
82 102 * @since 2.8.0
83 103 *
84 104 * @return array<string,int> Issue => bit.
@@ -83,12 +103,14 @@
83 103 *
84 104 * @return array<string,int> Issue => bit.
85 105 */
86 106 public static function flag_bits(): array {
107 + $grouped = self::grouped();
108 +
87 109 $bits = [];
88 110 $position = 0;
89 111 foreach (self::all() as $issue) {
90 - if (self::DUPLICATE_TITLE === $issue) {
112 + if (in_array($issue, $grouped, true)) {
91 113 continue;
92 114 }
93 115 $bits[$issue] = 1 << $position;
94 116 $position++;
@@ -97,9 +119,9 @@
97 119 return $bits;
98 120 }
99 121
100 122 /**
101 - * Encode issues as a bitmask. Duplicate title is ignored (see flag_bits()).
123 + * Encode issues as a bitmask. Grouped issues are ignored (see flag_bits()).
102 124 *
103 125 * @since 2.8.0
104 126 *
105 127 * @param string[] $issues Issue slugs.
@@ -117,9 +139,9 @@
117 139
118 140 /**
119 141 * Load everything the rules need for one post.
120 142 *
121 - * Duplicate status is not included — it needs the rest of the post type.
143 + * Duplicate status is not included — it needs the rest of the site.
122 144 *
123 145 * @since 2.8.0
124 146 *
125 147 * @param \WP_Post $post Post.
@@ -153,9 +175,10 @@
153 175 * effective_title?: string,
154 176 * effective_description?: string,
155 177 * focus_keyword?: string,
156 178 * noindex?: bool,
157 - * duplicate_title?: bool
179 + * duplicate_title?: bool,
180 + * duplicate_description?: bool
158 181 * } $row Already-loaded snippet values.
159 182 * @return string[] Issue slugs, in {@see self::all()} order.
160 183 */
161 184 public static function evaluate(array $row): array {
@@ -172,9 +195,11 @@
172 195 // Lengths are what Google sees, so they are measured on the effective
173 196 // value — a post inheriting a 90-character template title has a title
174 197 // that is too long even though it has no title of its own. A value
175 198 // that renders to nothing is reported as empty above, not as short.
176 - $title_length = mb_strlen(trim((string) ($row['effective_title'] ?? '')));
199 + // Measured decoded, as the editor's counter measures it: a texturized
200 + // `&#038;` is one character on the results page, not six.
201 + $title_length = mb_strlen(trim(Seo_Text::as_displayed((string) ($row['effective_title'] ?? ''))));
177 202 if ($title_length > 0 && $title_length < SEOScoreCalculator::TITLE_OPTIMAL_MIN) {
178 203 $issues[] = self::TITLE_TOO_SHORT;
179 204 } elseif ($title_length > SEOScoreCalculator::TITLE_OPTIMAL_MAX) {
180 205 $issues[] = self::TITLE_TOO_LONG;
@@ -179,9 +204,9 @@
179 204 } elseif ($title_length > SEOScoreCalculator::TITLE_OPTIMAL_MAX) {
180 205 $issues[] = self::TITLE_TOO_LONG;
181 206 }
182 207
183 - $description_length = mb_strlen(trim((string) ($row['effective_description'] ?? '')));
208 + $description_length = mb_strlen(trim(Seo_Text::as_displayed((string) ($row['effective_description'] ?? ''))));
184 209 if ($description_length > 0 && $description_length < SEOScoreCalculator::DESCRIPTION_OPTIMAL_MIN) {
185 210 $issues[] = self::DESCRIPTION_TOO_SHORT;
186 211 } elseif ($description_length > SEOScoreCalculator::DESCRIPTION_OPTIMAL_MAX) {
187 212 $issues[] = self::DESCRIPTION_TOO_LONG;
@@ -198,8 +223,12 @@
198 223 if (!empty($row['duplicate_title'])) {
199 224 $issues[] = self::DUPLICATE_TITLE;
200 225 }
201 226
227 + if (!empty($row['duplicate_description'])) {
228 + $issues[] = self::DUPLICATE_DESCRIPTION;
229 + }
230 +
202 231 return $issues;
203 232 }
204 233
205 234 /**
@@ -204,40 +233,93 @@
204 233
205 234 /**
206 235 * The key two posts share when they render the same title.
207 236 *
208 - * Duplicate detection has to work without resolving every title on the
209 - * site, so it compares the *inputs* that produce a title rather than the
210 - * output, case-insensitively (a search engine does not care about case):
237 + * This is the *rendered* title, normalized. Until 2.10.0 it was a hash of
238 + * the inputs that produce a title instead — the stored custom title, or
239 + * the post's own title when the post type's template was going to supply
240 + * the rest. That was sound only while duplicates were grouped inside one
241 + * post type, which is what Bulk Snippets did:
211 242 *
212 - * - a custom title with no variable tags is the title itself;
213 - * - a custom title with tags renders from the tags plus the post's own
214 - * title, so both are the key;
215 - * - no custom title means the post type's template, which is the same for
216 - * every post of the type, so the post's title is the key.
243 + * - A post and a page can share a post title and still render different
244 + * titles, because each post type has its own template. Grouping on the
245 + * post title alone reports them as duplicates when they are not (#564).
246 + * - A template carrying any tag that varies per post beyond `%title%` —
247 + * `%category%`, `%date%`, `%author%` — breaks the same assumption inside
248 + * a single post type, since two posts with one post title between them
249 + * render different titles.
217 250 *
218 - * Equal inputs give equal titles. The reverse is not guaranteed — a custom
219 - * "Foo – Site" and a template that happens to render the same string are
220 - * not matched — which errs toward missing a duplicate, never toward
221 - * inventing one.
251 + * Both disappear when the key is what the page actually renders, and the
252 + * reverse direction improves too: a hand-written title that happens to
253 + * match what another page's template renders is now matched, where the old
254 + * key could not see it.
222 255 *
256 + * There is no cost to resolving, because the caller has already resolved
257 + * it: {@see self::snapshot()} computes `effective_title` for the length
258 + * rules, and the index builds both keys from that one snapshot.
259 + *
223 260 * @since 2.8.0
261 + * @since 2.10.0 Keyed on the rendered title rather than on its inputs.
224 262 *
225 - * @param string $raw_title Stored `_thinkrank_seo_title` (may be empty).
226 - * @param string $post_title The post's own title.
263 + * @param string $effective_title The title the page renders.
227 264 * @return string Grouping key, or '' when there is nothing to compare.
228 265 */
229 - public static function duplicate_key(string $raw_title, string $post_title): string {
230 - $raw_title = mb_strtolower(trim($raw_title));
231 - $post_title = mb_strtolower(trim($post_title));
266 + public static function duplicate_key(string $effective_title): string {
267 + $normalized = self::normalize_for_comparison($effective_title);
232 268
233 - if ('' !== $raw_title) {
234 - return false === strpos($raw_title, '%')
235 - ? 'custom:' . $raw_title
236 - : 'tagged:' . $raw_title . '|' . $post_title;
237 - }
269 + return '' !== $normalized ? 'title:' . $normalized : '';
270 + }
238 271
239 - return '' !== $post_title ? 'template:' . $post_title : '';
272 + /**
273 + * The key two posts share when they render the same meta description.
274 + *
275 + * The description side has never had an input-comparison option: the
276 + * default template is `%excerpt%`, which differs for every post and is not
277 + * recoverable from a short stored string, so two posts inheriting the same
278 + * template collide only when their excerpts do. The rendered value is the
279 + * only thing worth comparing — which is now also true of the title, so the
280 + * two keys are built the same way.
281 + *
282 + * A description that renders to nothing returns '' and never groups —
283 + * "every post without a description" is the empty-description issue, not a
284 + * duplicate.
285 + *
286 + * @since 2.10.0
287 + *
288 + * @param string $effective_description The description the page renders.
289 + * @return string Grouping key, or '' when there is nothing to compare.
290 + */
291 + public static function description_duplicate_key(string $effective_description): string {
292 + $normalized = self::normalize_for_comparison($effective_description);
293 +
294 + return '' !== $normalized ? 'description:' . $normalized : '';
295 + }
296 +
297 + /**
298 + * Reduce a rendered value to what a search engine would see as the same
299 + * string: entities decoded, case folded, and runs of whitespace collapsed
300 + * to one space.
301 + *
302 + * The whitespace half matters more than it looks. An excerpt rebuilt from
303 + * post content can differ from a hand-written copy of it by a line break
304 + * alone, and the two snippets are identical on a results page.
305 + *
306 + * Decoding comes first. A template title reaches here through
307 + * get_the_title(), which texturizes "Foo & Bar" into `Foo &#038; Bar`,
308 + * while the same words typed as an SEO title arrive as `Foo &amp; Bar` or
309 + * a bare `&`. All three print the same `<title>`, and keying the encoded
310 + * forms kept them in separate groups. A decoded `&nbsp;` is a no-break
311 + * space, which the `/u` whitespace class then collapses like any other.
312 + * Changing this changes every stored key, which is why
313 + * {@see Snippet_Index::KEY_FORMAT} moved with it.
314 + *
315 + * @since 2.10.0
316 + *
317 + * @param string $value Rendered title or description.
318 + * @return string Normalized value, '' when it renders to nothing.
319 + */
320 + private static function normalize_for_comparison(string $value): string {
321 + return mb_strtolower(trim((string) preg_replace('/\s+/u', ' ', Seo_Text::as_displayed($value))));
240 322 }
241 323
242 324 /**
243 325 * Mark which rows share a duplicate key with another row.