PluginProbe
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO / 2.14.2
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO v2.14.2
2.14.2 2.14.1 2.14.0 2.13.0 2.12.0 2.11.0 2.10.0 2.9.0 2.8.0 2.7.0 2.6.0 2.5.0 2.4.0 2.3.0 2.2.0 2.1.1 2.1.0 2.0.2 2.0.1 2.0.0 1.32.0 1.31.0 1.30.0 1.29.0 1.28.0 All 57 releases
← All changes | includes/seo/class-seo-analyzer.php +1831 -35 1.29.0 → 2.14.2 View file →
@@ -41,8 +41,63 @@
41 41 * How long a computed analysis stays cached (seconds).
42 42 */
43 43 private const CACHE_TTL = HOUR_IN_SECONDS;
44 44
45 + /**
46 + * Names ThinkRank's FAQ / How-To producers share across Elementor widgets,
47 + * Bricks elements and Beaver Builder modules.
48 + */
49 + private const ANSWER_FAQ_NAME = 'thinkrank-faq';
50 + private const ANSWER_HOWTO_NAME = 'thinkrank-howto';
51 +
52 + /**
53 + * Element names page builders give their own accordion / FAQ widgets.
54 + *
55 + * Detection used to recognise ThinkRank's surfaces only, so an FAQ built
56 + * with the builder's own accordion was invisible — and the check then
57 + * advised adding FAQ content to a site that already had it (#686). Used
58 + * for the "is there Q&A content here?" question only; whether ThinkRank
59 + * *emits* schema for it is a separate question, still answered by
60 + * ThinkRank's own element names.
61 + *
62 + * @since 2.7.0
63 + * @var string[]
64 + */
65 + private const GENERIC_QA_ELEMENT_MARKERS = [
66 + 'accordion', // Elementor, Bricks, Beaver Builder, Breakdance, Oxygen
67 + 'toggle', // Elementor
68 + 'faq', // Widely used in third-party add-on element names
69 + ];
70 +
71 + /**
72 + * Builder meta keys that never describe the published page.
73 + *
74 + * `_fl_builder_draft` holds Beaver Builder changes that were never
75 + * published, so a FAQ that exists only there is not on the page (#945).
76 + *
77 + * @since 2.14.2
78 + * @var string[]
79 + */
80 + private const UNPUBLISHED_BUILDER_META_KEYS = ['_fl_builder_draft'];
81 +
82 + /**
83 + * Builder layouts that only count while their builder renders the post.
84 + *
85 + * Elementor and Beaver Builder both keep their stored layout after the
86 + * author switches the page back to the block editor, so the layout alone
87 + * does not mean the builder renders it. Each key maps to the builder's own
88 + * flag and the value that means "on": Elementor's `_elementor_edit_mode`
89 + * is `builder`, Beaver's `_fl_builder_enabled` is any non-empty value
90 + * (null here). The same tests as FAQ_Content::builder() (#945).
91 + *
92 + * @since 2.14.2
93 + * @var array<string,array{0:string,1:string|null}>
94 + */
95 + private const FLAGGED_BUILDER_LAYOUTS = [
96 + '_elementor_data' => ['_elementor_edit_mode', 'builder'],
97 + '_fl_builder_data' => ['_fl_builder_enabled', null],
98 + ];
99 +
45 100 // Check result statuses.
46 101 public const PASSED = 'passed';
47 102 public const WARNING = 'warning';
48 103 public const FAILED = 'failed';
@@ -58,12 +113,57 @@
58 113 'advanced' => __('Advanced SEO', 'thinkrank'),
59 114 'content' => __('Content', 'thinkrank'),
60 115 'performance' => __('Performance & Technical', 'thinkrank'),
61 116 'security' => __('Security', 'thinkrank'),
117 + // Generative/answer engine optimization. Spelled out because the
118 + // acronym alone reads as geography, and ucfirst()'d "Geo" — what a
119 + // filter-registered category falls back to — reads as nothing.
120 + 'geo' => __('AI Search (GEO)', 'thinkrank'),
62 121 ];
63 122 }
64 123
65 124 /**
125 + * WordPress options whose value the analyzer reports on directly.
126 + *
127 + * @since 2.2.0
128 + * @var string[]
129 + */
130 + private const WATCHED_OPTIONS = [
131 + 'blog_public',
132 + 'permalink_structure',
133 + 'blogname',
134 + 'blogdescription',
135 + // The published llms.txt document. Publishing or clearing it flips a
136 + // GEO check, and it is written as an option rather than through the
137 + // settings manager, so the settings-saved hook never sees it.
138 + 'thinkrank_llms_txt_content',
139 + ];
140 +
141 + /**
142 + * Register cache invalidation.
143 + *
144 + * The analysis is cached for an hour, and until now only the image alt-text
145 + * bulk writer ever busted it — so changing any other setting the audit
146 + * reports on left the screen confidently wrong for up to 60 minutes. The
147 + * audit's whole job is to describe the site's current configuration, so it
148 + * invalidates on every write it could possibly be reading.
149 + *
150 + * @since 2.2.0
151 + * @return void
152 + */
153 + public function init(): void {
154 + foreach (self::WATCHED_OPTIONS as $option) {
155 + add_action("update_option_{$option}", [$this, 'flush_cache']);
156 + add_action("add_option_{$option}", [$this, 'flush_cache']);
157 + }
158 +
159 + // Any ThinkRank settings category can feed a check (sitemap, schema,
160 + // image SEO today; more later). Flushing on all of them is cheaper than
161 + // a list that silently rots as checks are added.
162 + add_action('thinkrank_seo_settings_saved', [$this, 'flush_cache']);
163 + }
164 +
165 + /**
66 166 * Return the cached analysis, computing (and caching) it when missing or
67 167 * when a fresh run is forced.
68 168 *
69 169 * @param bool $force When true, ignore and overwrite the cached result.
@@ -160,9 +260,9 @@
160 260 unset($data);
161 261
162 262 $overall = $total_weight > 0 ? (int) round(($earned / $total_weight) * 100) : 0;
163 263
164 - return [
264 + $result = [
165 265 'overall_score' => $overall,
166 266 'grade' => $this->score_to_grade($overall),
167 267 'summary' => $summary,
168 268 'categories' => array_values($categories),
@@ -168,8 +268,28 @@
168 268 'categories' => array_values($categories),
169 269 'checks' => $checks,
170 270 'generated_at' => gmdate('c'),
171 271 ];
272 +
273 + /**
274 + * Fires after a site audit has been computed.
275 + *
276 + * Every path that produces a fresh analysis passes through here — the
277 + * REST run route, the one-click fixer's re-run, and a cold cache — so a
278 + * listener sees every run exactly once and never sees a cache hit.
279 + * ThinkRank Pro uses this to persist a dated snapshot for the score
280 + * trend and the run comparison.
281 + *
282 + * The payload is the analysis as returned to the caller; a listener
283 + * must treat it as read-only.
284 + *
285 + * @since 2.5.0
286 + *
287 + * @param array $result The completed analysis (see the return docblock).
288 + */
289 + do_action('thinkrank_seo_analysis_completed', $result);
290 +
291 + return $result;
172 292 }
173 293
174 294 /**
175 295 * Map a 0–100 score to a letter grade.
@@ -218,17 +338,34 @@
218 338 if (!is_array($outcome) || empty($outcome['status'])) {
219 339 continue;
220 340 }
221 341
342 + $id = (string) ($def['id'] ?? '');
343 + $status = (string) $outcome['status'];
344 +
345 + // Only offer a fix on a finding that still needs one — a passing
346 + // check with a Fix button reads as "did this even work?".
347 + $fixable = self::PASSED !== $status && SEO_Analyzer_Fixer::can_fix($id);
348 + $fix = $fixable ? (SEO_Analyzer_Fixer::fixable()[$id] ?? []) : [];
349 +
350 + // The list can be capped below the number of items the finding is
351 + // about, so the total travels with it.
352 + $affected = $this->normalize_affected_posts($outcome['affected_posts'] ?? []);
353 +
222 354 $results[] = [
223 - 'id' => (string) ($def['id'] ?? ''),
224 - 'category' => (string) ($def['category'] ?? 'basic'),
225 - 'weight' => isset($def['weight']) ? (float) $def['weight'] : 1.0,
226 - 'label' => (string) ($outcome['label'] ?? $def['label'] ?? ''),
227 - 'status' => (string) $outcome['status'],
228 - 'message' => (string) ($outcome['message'] ?? ''),
229 - 'how_to_fix' => (string) ($outcome['how_to_fix'] ?? ''),
230 - 'value' => $outcome['value'] ?? null,
355 + 'id' => $id,
356 + 'category' => (string) ($def['category'] ?? 'basic'),
357 + 'weight' => isset($def['weight']) ? (float) $def['weight'] : 1.0,
358 + 'label' => (string) ($outcome['label'] ?? $def['label'] ?? ''),
359 + 'status' => $status,
360 + 'message' => (string) ($outcome['message'] ?? ''),
361 + 'how_to_fix' => (string) ($outcome['how_to_fix'] ?? ''),
362 + 'value' => $outcome['value'] ?? null,
363 + 'affected_posts' => $affected,
364 + 'affected_total' => max(count($affected), absint($outcome['affected_total'] ?? 0)),
365 + 'can_auto_fix' => $fixable,
366 + 'fix_label' => (string) ($fix['label'] ?? ''),
367 + 'fix_warning' => (string) ($fix['warning'] ?? ''),
231 368 ];
232 369 }
233 370
234 371 return $results;
@@ -234,8 +371,43 @@
234 371 return $results;
235 372 }
236 373
237 374 /**
375 + * Reduce a check's list of posts to fix to a known, escaped shape.
376 + *
377 + * The list reaches the audit UI as links, and a check registered through
378 + * `thinkrank_seo_analyzer_checks` can put anything in it, so every entry is
379 + * rebuilt here rather than passed through.
380 + *
381 + * @since 2.6.0
382 + * @param mixed $posts Raw `affected_posts` from a check result.
383 + * @return array<int,array{id: int, title: string, type: string, edit_url: string, url: string}>
384 + */
385 + private function normalize_affected_posts($posts): array {
386 + if (!is_array($posts)) {
387 + return [];
388 + }
389 +
390 + $out = [];
391 +
392 + foreach ($posts as $post) {
393 + if (!is_array($post) || empty($post['id'])) {
394 + continue;
395 + }
396 +
397 + $out[] = [
398 + 'id' => absint($post['id']),
399 + 'title' => sanitize_text_field((string) ($post['title'] ?? '')),
400 + 'type' => sanitize_text_field((string) ($post['type'] ?? '')),
401 + 'edit_url' => esc_url_raw((string) ($post['edit_url'] ?? '')),
402 + 'url' => esc_url_raw((string) ($post['url'] ?? '')),
403 + ];
404 + }
405 +
406 + return $out;
407 + }
408 +
409 + /**
238 410 * The registry of checks: id, category, weight, and the callback that
239 411 * evaluates it. Filterable so Pro/add-ons can register additional checks.
240 412 *
241 413 * @return array<int,array>
@@ -263,8 +435,21 @@
263 435 // Security
264 436 ['id' => 'https', 'category' => 'security', 'weight' => 3, 'callback' => [$this, 'check_https']],
265 437 ['id' => 'file_editing', 'category' => 'security', 'weight' => 2, 'callback' => [$this, 'check_file_editing']],
266 438 ['id' => 'debug_display', 'category' => 'security', 'weight' => 1, 'callback' => [$this, 'check_debug_display']],
439 +
440 + // GEO / AEO — AI answer-engine readiness. The three configuration
441 + // checks carry the weight of a normal check; the five sampled
442 + // content ones are half that, so the category as a whole sits
443 + // beside the existing ones rather than dominating the score.
444 + ['id' => 'ai_crawler_access', 'category' => 'geo', 'weight' => 1.5, 'callback' => [$this, 'check_ai_crawler_access']],
445 + ['id' => 'llms_txt', 'category' => 'geo', 'weight' => 1.5, 'callback' => [$this, 'check_llms_txt']],
446 + ['id' => 'answer_ready_schema', 'category' => 'geo', 'weight' => 1.5, 'callback' => [$this, 'check_answer_ready_schema']],
447 + ['id' => 'direct_answer', 'category' => 'geo', 'weight' => 0.5, 'callback' => [$this, 'check_direct_answer']],
448 + ['id' => 'question_headings', 'category' => 'geo', 'weight' => 0.5, 'callback' => [$this, 'check_question_headings']],
449 + ['id' => 'structured_content', 'category' => 'geo', 'weight' => 0.5, 'callback' => [$this, 'check_structured_content']],
450 + ['id' => 'content_depth', 'category' => 'geo', 'weight' => 0.5, 'callback' => [$this, 'check_content_depth']],
451 + ['id' => 'content_freshness', 'category' => 'geo', 'weight' => 0.5, 'callback' => [$this, 'check_content_freshness']],
267 452 ];
268 453
269 454 /**
270 455 * Filter the Site SEO Analyzer check registry.
@@ -269,9 +454,9 @@
269 454 /**
270 455 * Filter the Site SEO Analyzer check registry.
271 456 *
272 457 * Each entry is an array with keys: id, category (basic|advanced|
273 - * content|performance|security), weight (float), and callback (callable
458 + * content|performance|security|geo), weight (float), and callback (callable
274 459 * returning ['status' => passed|warning|failed, 'label', 'message',
275 460 * 'how_to_fix']).
276 461 *
277 462 * @since 1.18.0
@@ -319,10 +504,11 @@
319 504 * @return array
320 505 */
321 506 public function check_tagline(): array {
322 507 $tagline = trim((string) get_bloginfo('description'));
323 - $is_default = strtolower($tagline) === strtolower('Just another WordPress site');
324 508
509 + $is_default = $this->is_default_tagline($tagline);
510 +
325 511 if ($tagline === '' || $is_default) {
326 512 return [
327 513 'label' => __('Tagline is customized', 'thinkrank'),
328 514 'status' => self::WARNING,
@@ -339,8 +525,53 @@
339 525 ];
340 526 }
341 527
342 528 /**
529 + * Whether a tagline is still WordPress' shipped default.
530 + *
531 + * The installer writes the TRANSLATED default into blogdescription, so an
532 + * English-only literal silently passed an untouched tagline on every
533 + * non-English install. The string lives in core's `admin-{locale}.mo`,
534 + * which a REST request (how this analyzer runs) does not load — so the
535 + * catalogue is loaded on demand for the comparison when the site is not
536 + * running in English.
537 + *
538 + * @since 2.2.0
539 + * @param string $tagline Trimmed tagline.
540 + * @return bool
541 + */
542 + private function is_default_tagline(string $tagline): bool {
543 + $candidates = ['Just another WordPress site'];
544 +
545 + $locale = get_locale();
546 + if ('en_US' !== $locale) {
547 + // phpcs:ignore WordPress.WP.I18n.TextDomainMismatch,WordPress.WP.I18n.LowLevelTranslationFunction -- core's own string in the `default` domain, read at runtime.
548 + $translated = translate('Just another WordPress site', 'default');
549 +
550 + if ($translated === 'Just another WordPress site') {
551 + // Not in the loaded catalogue — pull in the admin one, which is
552 + // where core ships this string, then ask again.
553 + $mofile = WP_LANG_DIR . '/admin-' . $locale . '.mo';
554 + if (is_readable($mofile)) {
555 + load_textdomain('default', $mofile, $locale);
556 + // phpcs:ignore WordPress.WP.I18n.TextDomainMismatch,WordPress.WP.I18n.LowLevelTranslationFunction -- as above.
557 + $translated = translate('Just another WordPress site', 'default');
558 + }
559 + }
560 +
561 + $candidates[] = $translated;
562 + }
563 +
564 + foreach ($candidates as $candidate) {
565 + if (strtolower($tagline) === strtolower($candidate)) {
566 + return true;
567 + }
568 + }
569 +
570 + return false;
571 + }
572 +
573 + /**
343 574 * "Discourage search engines from indexing this site" must be OFF.
344 575 *
345 576 * @return array
346 577 */
@@ -349,9 +580,9 @@
349 580 if (!get_option('blog_public')) {
350 581 return [
351 582 'label' => __('Site is visible to search engines', 'thinkrank'),
352 583 'status' => self::FAILED,
353 - 'message' => __('Your site is telling search engines not to index it — it will not appear in search results.', 'thinkrank'),
584 + 'message' => __('Your site is telling search engines not to index it, so it will not appear in search results.', 'thinkrank'),
354 585 'how_to_fix' => __('Untick "Discourage search engines from indexing this site" under Settings → Reading.', 'thinkrank'),
355 586 ];
356 587 }
357 588
@@ -399,9 +630,12 @@
399 630 public function check_sitemap(): array {
400 631 $enabled = true;
401 632 try {
402 633 $generator = new Sitemap_Generator();
403 - $data = $generator->get_output_data('global', null);
634 + // 'site' is the stored context; 'global' is unsupported and
635 + // returns DEFAULTS (enabled=true), which made this check unable
636 + // to fail no matter what the user configured.
637 + $data = $generator->get_output_data('site', null);
404 638 $enabled = !empty($data['enabled']);
405 639 } catch (\Throwable $e) {
406 640 // Fall back to "enabled" — the default state — on any lookup error.
407 641 $enabled = true;
@@ -430,9 +664,9 @@
430 664 */
431 665 public function check_schema(): array {
432 666 $label = __('Structured data configured', 'thinkrank');
433 667
434 - if ($this->schema_is_output()) {
668 + if ($this->schema_is_configured()) {
435 669 return [
436 670 'label' => $label,
437 671 'status' => self::PASSED,
438 672 'message' => __('Structured data is configured for your content.', 'thinkrank'),
@@ -438,17 +672,90 @@
438 672 'message' => __('Structured data is configured for your content.', 'thinkrank'),
439 673 ];
440 674 }
441 675
676 + // Nothing is configured, but ThinkRank still emits JSON-LD from its
677 + // built-in per-post-type defaults. Saying "no schema" there would be
678 + // false; the actionable point is that nobody has reviewed it.
679 + if ($this->schema_is_output()) {
680 + return [
681 + 'label' => $label,
682 + 'status' => self::WARNING,
683 + 'message' => __('Structured data is running on ThinkRank\'s built-in defaults. Reviewing the schema type for each post type gives you control over how rich results appear.', 'thinkrank'),
684 + 'how_to_fix' => __('Choose a schema type for each post type under Essential SEO → Bulk SEO Optimization.', 'thinkrank'),
685 + ];
686 + }
687 +
442 688 return [
443 689 'label' => $label,
444 - 'status' => self::WARNING,
445 - 'message' => __('No schema/structured data is configured. Schema powers rich results in search.', 'thinkrank'),
446 - 'how_to_fix' => __('Choose a default schema type for each post type under Essential SEO → Bulk SEO Optimization.', 'thinkrank'),
690 + 'status' => self::FAILED,
691 + 'message' => __('No schema/structured data is configured or emitted. Schema powers rich results in search.', 'thinkrank'),
692 + 'how_to_fix' => __('Turn on automatic structured data, or choose a schema type for each post type under Essential SEO → Bulk SEO Optimization.', 'thinkrank'),
447 693 ];
448 694 }
449 695
450 696 /**
697 + * Whether the user has EXPLICITLY configured structured data.
698 + *
699 + * Distinct from schema_is_output(): the Global SEO layer falls back to a
700 + * built-in schema type for every public post type, so "something is
701 + * emitted" is true on every site and made this check impossible to fail
702 + * (its weight was earned unconditionally and its one-click fix was
703 + * unreachable). This asks the question the check's copy actually claims to
704 + * answer.
705 + *
706 + * Both layers must be read WITHOUT their defaults, or the same trap closes
707 + * again one level down: get_settings() merges the context defaults under
708 + * the saved rows, and Schema_Settings_Config's 'site' defaults set both
709 + * enabled_schema_types and auto_generate_schema — so an untouched site came
710 + * back looking configured and this method still could not return false
711 + * (#586). get_stored_settings() answers with only what was actually saved.
712 + *
713 + * @since 2.2.0
714 + * @return bool
715 + */
716 + private function schema_is_configured(): bool {
717 + // 1) Schema Management System — an explicit opt-in. Read the SAVED rows
718 + // only; the defaults-merged view is truthy on every site.
719 + if (class_exists('ThinkRank\\SEO\\Schema_Management_System')) {
720 + $settings = (new Schema_Management_System())->get_stored_settings('site', null);
721 + // The master switch gates these the same way it gates
722 + // schema_is_output(). Saving the settings form persists the whole
723 + // payload, so turning the feature off stores enabled = '0' while
724 + // auto_generate_schema stays '1' — and reading past the switch then
725 + // reported "structured data is configured" for a site emitting
726 + // none, with the one-click fix withheld. array_key_exists rather
727 + // than a bare !empty so an untouched site, where 'enabled' was
728 + // never saved at all, still falls through to the Global SEO layer
729 + // below instead of short-circuiting to false.
730 + $master_on = is_array($settings)
731 + && (!array_key_exists('enabled', $settings) || !empty($settings['enabled']));
732 +
733 + if ($master_on) {
734 + if (!empty($settings['enabled_schema_types']) && is_array($settings['enabled_schema_types'])) {
735 + return true;
736 + }
737 + if (!empty($settings['auto_generate_schema'])) {
738 + return true;
739 + }
740 + }
741 + }
742 +
743 + // 2) A saved per-post-type schema_type in the Global SEO layer. The
744 + // built-in default deliberately does not count here.
745 + if (class_exists('ThinkRank\\Frontend\\Global_SEO_Schema_Output')) {
746 + $output = new \ThinkRank\Frontend\Global_SEO_Schema_Output();
747 + foreach (get_post_types(['public' => true], 'names') as $post_type) {
748 + if ($output->has_explicit_schema_type((string) $post_type)) {
749 + return true;
750 + }
751 + }
752 + }
753 +
754 + return false;
755 + }
756 +
757 + /**
451 758 * Whether ThinkRank actually emits structured data for this site.
452 759 *
453 760 * The audit must reflect what is rendered, not a single legacy option.
454 761 * ThinkRank outputs schema from two current sources, so this check consults
@@ -467,10 +774,24 @@
467 774 */
468 775 private function schema_is_output(): bool {
469 776 // 1) Newer Schema Management System (thinkrank_seo_settings table).
470 777 if (class_exists('ThinkRank\\SEO\\Schema_Management_System')) {
471 - $settings = (new Schema_Management_System())->get_settings('schema_management_system');
472 - if (is_array($settings)) {
778 + // 'site' is the context type; 'schema_management_system' is the manager
779 + // NAME, which get_settings() rejects as an unsupported context and
780 + // answers with bare defaults — where auto_generate_schema is true, so
781 + // this always returned true and never read the site's real settings (#473).
782 + //
783 + // This one KEEPS the defaults-merged view on purpose, unlike
784 + // schema_is_configured() (#586). The question here is "does JSON-LD
785 + // reach the page?", and an untouched site answers yes: the 'site'
786 + // defaults leave the system enabled with auto_generate_schema on, so
787 + // the merged value is the emitted behaviour, not a mask over it.
788 + $settings = (new Schema_Management_System())->get_settings('site', null);
789 + // The master switch gates everything below it: with 'enabled' off,
790 + // get_output_data() reports the feature as off and nothing is
791 + // emitted, so reading auto_generate_schema past it told the audit
792 + // schema was on the page when it was not (the same shape as #461).
793 + if (is_array($settings) && !empty($settings['enabled'])) {
473 794 if (!empty($settings['enabled_schema_types']) && is_array($settings['enabled_schema_types'])) {
474 795 return true;
475 796 }
476 797 if (!empty($settings['auto_generate_schema'])) {
@@ -509,8 +830,17 @@
509 830 private const COVERAGE_PASS = 90;
510 831 private const COVERAGE_WARN = 50;
511 832
512 833 /**
834 + * Most items a finding names when it is not drawn from the bounded sample.
835 + *
836 + * The image check counts the whole media library, which can run to tens of
837 + * thousands of rows; a list that long would bloat the cached analysis and
838 + * every stored audit snapshot while telling the user nothing more.
839 + */
840 + private const AFFECTED_LIMIT = 100;
841 +
842 + /**
513 843 * Recent published posts/pages should have meta descriptions.
514 844 *
515 845 * Samples the most recent CONTENT_SAMPLE_SIZE published posts/pages so the
516 846 * check stays fast on large sites.
@@ -519,18 +849,9 @@
519 849 */
520 850 public function check_meta_descriptions(): array {
521 851 $label = __('Posts have meta descriptions', 'thinkrank');
522 852
523 - $post_ids = get_posts([
524 - 'post_type' => ['post', 'page'],
525 - 'post_status' => 'publish',
526 - 'posts_per_page' => self::CONTENT_SAMPLE_SIZE,
527 - 'orderby' => 'date',
528 - 'order' => 'DESC',
529 - 'fields' => 'ids',
530 - 'no_found_rows' => true,
531 - 'suppress_filters' => false,
532 - ]);
853 + $post_ids = $this->sample_post_ids();
533 854
534 855 $total = count($post_ids);
535 856 if (0 === $total) {
536 857 return [
@@ -546,14 +867,22 @@
546 867 // only the custom post-meta produced false negatives — posts that output
547 868 // a valid description via the pattern fallback were wrongly reported as
548 869 // missing. The sample is bounded (CONTENT_SAMPLE_SIZE) so the per-post
549 870 // resolution stays cheap, and the whole analysis is cached for an hour.
871 + // 'fields' => 'ids' skips WP_Query's meta priming, so the first
872 + // get_post_meta() below would issue a query per post. Warm the whole
873 + // sample once instead — 100 posts went from ~200 queries to a handful.
874 + _prime_post_caches($post_ids, false, true);
875 +
550 876 $with_description = 0;
877 + $without = [];
551 878 foreach ($post_ids as $post_id) {
552 879 $custom = (string) get_post_meta($post_id, '_thinkrank_meta_description', true);
553 880 $resolved = '' !== $custom ? $custom : Pattern_Resolver::description((int) $post_id);
554 881 if ('' !== trim($resolved)) {
555 882 $with_description++;
883 + } else {
884 + $without[] = (int) $post_id;
556 885 }
557 886 }
558 887
559 888 $coverage = (int) round(($with_description / $total) * 100);
@@ -574,10 +903,11 @@
574 903 'label' => $label,
575 904 'status' => $coverage >= self::COVERAGE_WARN ? self::WARNING : self::FAILED,
576 905 /* translators: 1: posts missing a meta description, 2: sampled posts. */
577 906 'message' => sprintf(__('%1$d of your %2$d most recent posts are missing a meta description. Search engines fall back to arbitrary page text for their snippets.', 'thinkrank'), $missing, $total),
578 - 'how_to_fix' => __('Add meta descriptions in the ThinkRank SEO panel when editing a post — or use Bulk SEO Optimization to generate them with AI.', 'thinkrank'),
579 - 'value' => $value,
907 + 'how_to_fix' => __('Add meta descriptions in the ThinkRank SEO panel when editing a post, or use Bulk SEO Optimization to generate them with AI.', 'thinkrank'),
908 + 'value' => $value,
909 + 'affected_posts' => $this->affected_posts($without),
580 910 ];
581 911 }
582 912
583 913 /**
@@ -592,9 +922,11 @@
592 922 global $wpdb;
593 923 // phpcs:ignore WordPress.DB.DirectDatabaseQuery.DirectQuery, WordPress.DB.DirectDatabaseQuery.NoCaching -- two indexed COUNTs; results are cached at the analysis level
594 924 $total = (int) $wpdb->get_var(
595 925 "SELECT COUNT(*) FROM {$wpdb->posts}
596 - WHERE post_type = 'attachment' AND post_mime_type LIKE 'image/%'"
926 + WHERE post_type = 'attachment'
927 + AND post_mime_type LIKE 'image/%'
928 + AND post_status != 'trash'"
597 929 );
598 930
599 931 if (0 === $total) {
600 932 return [
@@ -610,9 +942,11 @@
610 942 INNER JOIN {$wpdb->postmeta} pm
611 943 ON pm.post_id = p.ID
612 944 AND pm.meta_key = '_wp_attachment_image_alt'
613 945 AND pm.meta_value != ''
614 - WHERE p.post_type = 'attachment' AND p.post_mime_type LIKE 'image/%'"
946 + WHERE p.post_type = 'attachment'
947 + AND p.post_mime_type LIKE 'image/%'
948 + AND p.post_status != 'trash'"
615 949 );
616 950
617 951 $coverage = (int) round(($with_alt / $total) * 100);
618 952 $missing = $total - $with_alt;
@@ -632,13 +966,50 @@
632 966 'label' => $label,
633 967 'status' => $coverage >= self::COVERAGE_WARN ? self::WARNING : self::FAILED,
634 968 /* translators: 1: images missing alt text, 2: total images. */
635 969 'message' => sprintf(__('%1$d of your %2$d images are missing alt text. Alt text drives image search rankings and is an accessibility requirement.', 'thinkrank'), $missing, $total),
636 - 'how_to_fix' => __('Under Essential SEO → Image SEO, turn on "Save alt text to the Media Library" and run "Fill missing alt text" to populate them from your format, or add alt text manually in the Media Library.', 'thinkrank'),
637 - 'value' => $value,
970 + 'how_to_fix' => __('Under Essential SEO → Image SEO, turn on "Save alt text to the Media Library" and run "Fill missing alt text" to populate them from your format, or add alt text manually in the Media Library.', 'thinkrank'),
971 + 'value' => $value,
972 + 'affected_posts' => $this->affected_posts($this->image_ids_without_alt()),
973 + 'affected_total' => $missing,
638 974 ];
639 975 }
640 976
977 + /**
978 + * The most recent images with no alt text, capped at AFFECTED_LIMIT.
979 + *
980 + * Mirrors the counts above: an image counts as having alt text when any
981 + * `_wp_attachment_image_alt` row for it is non-empty.
982 + *
983 + * @since 2.6.0
984 + * @return int[]
985 + */
986 + private function image_ids_without_alt(): array {
987 + global $wpdb;
988 +
989 + // phpcs:ignore WordPress.DB.DirectDatabaseQuery.DirectQuery, WordPress.DB.DirectDatabaseQuery.NoCaching -- bounded id list for a cached analysis; no core API filters on a missing meta value
990 + $ids = $wpdb->get_col(
991 + $wpdb->prepare(
992 + "SELECT p.ID FROM {$wpdb->posts} p
993 + WHERE p.post_type = 'attachment'
994 + AND p.post_mime_type LIKE %s
995 + AND p.post_status != 'trash'
996 + AND NOT EXISTS (
997 + SELECT 1 FROM {$wpdb->postmeta} pm
998 + WHERE pm.post_id = p.ID
999 + AND pm.meta_key = '_wp_attachment_image_alt'
1000 + AND pm.meta_value != ''
1001 + )
1002 + ORDER BY p.post_date DESC
1003 + LIMIT %d",
1004 + $wpdb->esc_like('image/') . '%',
1005 + self::AFFECTED_LIMIT
1006 + )
1007 + );
1008 +
1009 + return array_map('intval', (array) $ids);
1010 + }
1011 +
641 1012 // ─────────────────────────────────────────────────────────────────────
642 1013 // Performance & Technical checks
643 1014 // ─────────────────────────────────────────────────────────────────────
644 1015
@@ -780,6 +1151,1431 @@
780 1151 'label' => __('Errors not shown publicly', 'thinkrank'),
781 1152 'status' => self::PASSED,
782 1153 'message' => __('PHP errors are not displayed to visitors.', 'thinkrank'),
783 1154 ];
1155 + }
1156 + // ─────────────────────────────────────────────────────────────────────
1157 + // GEO / AEO checks (AI answer-engine readiness)
1158 + // ─────────────────────────────────────────────────────────────────────
1159 +
1160 + /**
1161 + * The crawlers that decide whether a site can be CITED by an AI answer
1162 + * engine, as opposed to the ones that only collect training data.
1163 + *
1164 + * Blocking a training crawler (GPTBot, ClaudeBot, CCBot…) is a legitimate
1165 + * editorial choice and is deliberately not marked down here: it costs the
1166 + * site nothing in ChatGPT, Claude, Perplexity or AI Overviews. Blocking
1167 + * the agents below is what makes a site invisible to those answers, so
1168 + * they are the only ones this check reports on.
1169 + *
1170 + * Public because the one-click fix writes exactly this set to `allow`;
1171 + * a second copy in the fixer is how the check and its remedy drift into
1172 + * disagreeing about which crawlers matter.
1173 + *
1174 + * @since 2.5.0
1175 + * @var string[]
1176 + */
1177 + public const GEO_ANSWER_AGENTS = [
1178 + 'oai-searchbot',
1179 + 'chatgpt-user',
1180 + 'perplexitybot',
1181 + 'perplexity-user',
1182 + 'claude-searchbot',
1183 + 'claude-user',
1184 + 'google-extended',
1185 + 'mistral-user',
1186 + ];
1187 +
1188 + /**
1189 + * Word-count window for an opening passage that reads as a direct answer.
1190 + *
1191 + * The AEO convention is a self-contained 40–60 word answer directly under
1192 + * the title. The window is widened at both ends so ordinary good writing
1193 + * passes: below the floor there is no answer to quote, and well above the
1194 + * ceiling the passage is a preamble an engine has to summarize rather than
1195 + * a sentence it can lift.
1196 + */
1197 + private const GEO_ANSWER_MIN_WORDS = 20;
1198 + private const GEO_ANSWER_MAX_WORDS = 120;
1199 +
1200 + /**
1201 + * Words below which a page has too little substance to be cited.
1202 + */
1203 + private const GEO_DEPTH_MIN_WORDS = 300;
1204 +
1205 + /**
1206 + * How long a page can go unrevised before it reads as stale to an engine
1207 + * that prefers recent sources.
1208 + */
1209 + private const GEO_FRESHNESS_MAX_AGE = 365 * DAY_IN_SECONDS;
1210 +
1211 + /**
1212 + * The content sample the GEO checks share, memoized for one analysis.
1213 + *
1214 + * Five checks read the same 100 posts. Sampling once turns five queries
1215 + * (and five cache-priming passes) into one, and guarantees the five
1216 + * results describe the same set of posts.
1217 + *
1218 + * @since 2.5.0
1219 + * @var array<int,array{id: int, content: string, text: string, modified: int}>|null
1220 + */
1221 + private $content_sample = null;
1222 +
1223 + /**
1224 + * The sampled post ids, memoized so the Content and GEO categories run one
1225 + * query between them rather than one each over the same CONTENT_SAMPLE_SIZE
1226 + * posts. Both want the same slice — the most recent published posts and
1227 + * pages — so two queries only guaranteed they could disagree after a
1228 + * publish mid-analysis.
1229 + *
1230 + * @since 2.5.0
1231 + * @var int[]|null
1232 + */
1233 + private $sample_post_ids = null;
1234 +
1235 + /**
1236 + * Memoised answer to "does this site publish Q&A content?".
1237 + *
1238 + * @since 2.7.0
1239 + * @var bool|null
1240 + */
1241 + private $site_has_qa_content = null;
1242 +
1243 + /**
1244 + * The sampled post ids, fetched once and shared by every check that reads
1245 + * the same slice.
1246 + *
1247 + * @since 2.5.0
1248 + * @return int[]
1249 + */
1250 + private function sample_post_ids(): array {
1251 + if (null !== $this->sample_post_ids) {
1252 + return $this->sample_post_ids;
1253 + }
1254 +
1255 + $this->sample_post_ids = get_posts([
1256 + 'post_type' => ['post', 'page'],
1257 + 'post_status' => 'publish',
1258 + 'posts_per_page' => self::CONTENT_SAMPLE_SIZE,
1259 + 'orderby' => 'date',
1260 + 'order' => 'DESC',
1261 + 'fields' => 'ids',
1262 + 'no_found_rows' => true,
1263 + 'suppress_filters' => false,
1264 + ]);
1265 +
1266 + return $this->sample_post_ids;
1267 + }
1268 +
1269 + /**
1270 + * The most recent published posts/pages, as raw content, its plain-text
1271 + * rendering and the modified time.
1272 + *
1273 + * Raw `post_content` on purpose: running `the_content` over 100 posts in a
1274 + * REST request would fire every shortcode and block renderer on the site.
1275 + * The structural signals these checks look for (headings, lists, tables,
1276 + * the opening passage) survive in the stored markup.
1277 + *
1278 + * Except on a builder page, where they do not: the words and headings are
1279 + * in builder meta and `post_content` is empty, or, on a Bricks page, holds
1280 + * blocks Bricks never renders. Those rows read from builder_content(), so
1281 + * a builder-built site is graded on its pages rather than skipped (#892).
1282 + *
1283 + * @since 2.5.0
1284 + * @return array<int,array{id: int, content: string, text: string, modified: int}>
1285 + */
1286 + private function get_content_sample(): array {
1287 + if (null !== $this->content_sample) {
1288 + return $this->content_sample;
1289 + }
1290 +
1291 + $post_ids = $this->sample_post_ids();
1292 +
1293 + // Meta too: builder_content() asks every sampled post whether Bricks
1294 + // owns it, and one query beats a hundred.
1295 + if (function_exists('_prime_post_caches')) {
1296 + _prime_post_caches($post_ids, false, true);
1297 + }
1298 +
1299 + $sample = [];
1300 +
1301 + foreach ($post_ids as $post_id) {
1302 + $post = get_post($post_id);
1303 + if (!$post) {
1304 + continue;
1305 + }
1306 +
1307 + $modified = isset($post->post_modified_gmt) ? strtotime((string) $post->post_modified_gmt . ' UTC') : false;
1308 +
1309 + $content = $this->builder_content($post) ?? (string) $post->post_content;
1310 +
1311 + $sample[] = [
1312 + 'id' => (int) $post->ID,
1313 + 'content' => $content,
1314 + 'text' => $this->content_to_text($content),
1315 + 'modified' => is_int($modified) ? $modified : 0,
1316 + ];
1317 + }
1318 +
1319 + $this->content_sample = $sample;
1320 +
1321 + return $sample;
1322 + }
1323 +
1324 + /**
1325 + * A sampled post's content from its page builder, when that is where it is.
1326 + *
1327 + * Null for every post whose `post_content` is what the visitor reads, so
1328 + * those keep the raw-markup path and never pay for rendering. A post with
1329 + * an empty `post_content`, or a Bricks page that discards it, resolves
1330 + * through Builder_Content, the extractor every other server-side scoring
1331 + * path reads through (#617). With nothing in `post_content` to render, the
1332 + * resolution only reads stored builder meta: no shortcode or block
1333 + * renderer runs.
1334 + *
1335 + * Guarded like builder_meta_keys(), so a partial checkout degrades to the
1336 + * raw markup instead of fataling mid-audit.
1337 + *
1338 + * @since 2.15.0
1339 + *
1340 + * @param \WP_Post $post Sampled post.
1341 + * @return string|null Resolved content, or null to use `post_content`.
1342 + */
1343 + private function builder_content(\WP_Post $post): ?string {
1344 + if ([] === self::builder_meta_keys()) {
1345 + return null;
1346 + }
1347 +
1348 + if ('' !== trim((string) $post->post_content)
1349 + && !Builder_Content::bricks_supersedes_post_content((int) $post->ID)
1350 + ) {
1351 + return null;
1352 + }
1353 +
1354 + try {
1355 + return Builder_Content::resolve($post);
1356 + } catch (\Throwable $e) {
1357 + return null;
1358 + }
1359 + }
1360 +
1361 + /**
1362 + * Shared shape for the content-sampled GEO checks: count how many posts in
1363 + * the sample satisfy a predicate and grade it on the coverage thresholds
1364 + * the Content category already uses.
1365 + *
1366 + * @since 2.5.0
1367 + *
1368 + * @param string $label Check label.
1369 + * @param callable $predicate Receives one sample row, returns bool.
1370 + * @param string $pass_text sprintf template: 1 = matching, 2 = sampled.
1371 + * @param string $fail_text sprintf template: 1 = missing, 2 = sampled.
1372 + * @param string $how_to_fix Advice shown on a warning/fail.
1373 + * @param string $empty_text Message when the site has no content yet.
1374 + * @return array
1375 + */
1376 + private function coverage_check(
1377 + string $label,
1378 + callable $predicate,
1379 + string $pass_text,
1380 + string $fail_text,
1381 + string $how_to_fix,
1382 + string $empty_text
1383 + ): array {
1384 + // A page whose body is only a shortcode or a commerce block (Checkout,
1385 + // My account, Shop) leaves no extractable text, so every prose-shaped
1386 + // question here answers "no" for it and would drag the category down
1387 + // over content nobody wants quoted in an AI answer. Skipping them is
1388 + // deliberate: this grades the pages that could be cited. Builder pages
1389 + // are not in this group: get_content_sample() reads their builder
1390 + // storage, so they have text and are graded (#892).
1391 + $sampled = $this->get_content_sample();
1392 + $sample = [];
1393 + $textless = [];
1394 + foreach ($sampled as $row) {
1395 + if ('' !== $row['text']) {
1396 + $sample[] = $row;
1397 + } else {
1398 + $textless[] = $row['id'];
1399 + }
1400 + }
1401 +
1402 + $total = count($sample);
1403 +
1404 + // Nothing published: nothing to report.
1405 + if ([] === $sampled) {
1406 + return [
1407 + 'label' => $label,
1408 + 'status' => self::PASSED,
1409 + 'message' => $empty_text,
1410 + ];
1411 + }
1412 +
1413 + // Pages exist, but none yielded text. That is a sample this check
1414 + // could not read, not an empty site, and it used to pass with "No
1415 + // published content to check yet", awarding the weight for content
1416 + // nobody examined (#892). Say so instead.
1417 + if (0 === $total) {
1418 + return [
1419 + 'label' => $label,
1420 + 'status' => self::WARNING,
1421 + 'message' => sprintf(
1422 + /* translators: %d: number of sampled pages. */
1423 + _n(
1424 + 'ThinkRank found no readable text on the %d page it sampled, so this could not be checked.',
1425 + 'ThinkRank found no readable text on any of the %d pages it sampled, so this could not be checked.',
1426 + count($textless),
1427 + 'thinkrank'
1428 + ),
1429 + count($textless)
1430 + ),
1431 + 'how_to_fix' => __('Pages built only from shortcodes, or with a page builder ThinkRank does not support, have no text it can read. Publish pages with written content, or check that your page builder is supported.', 'thinkrank'),
1432 + 'value' => sprintf('0/%d', count($textless)),
1433 + 'affected_posts' => $this->affected_posts($textless),
1434 + ];
1435 + }
1436 +
1437 + $matching = 0;
1438 + $failing = [];
1439 + foreach ($sample as $row) {
1440 + if ($predicate($row)) {
1441 + $matching++;
1442 + } else {
1443 + $failing[] = $row['id'];
1444 + }
1445 + }
1446 +
1447 + $coverage = (int) round(($matching / $total) * 100);
1448 + $value = sprintf('%d/%d', $matching, $total);
1449 +
1450 + if ($coverage >= self::COVERAGE_PASS) {
1451 + return [
1452 + 'label' => $label,
1453 + 'status' => self::PASSED,
1454 + 'message' => sprintf($pass_text, $matching, $total),
1455 + 'value' => $value,
1456 + ];
1457 + }
1458 +
1459 + // A count alone leaves the user to guess which pages it means, so the
1460 + // finding names them.
1461 + return [
1462 + 'label' => $label,
1463 + 'status' => $coverage >= self::COVERAGE_WARN ? self::WARNING : self::FAILED,
1464 + 'message' => sprintf($fail_text, $total - $matching, $total),
1465 + 'how_to_fix' => $how_to_fix,
1466 + 'value' => $value,
1467 + 'affected_posts' => $this->affected_posts($failing),
1468 + ];
1469 + }
1470 +
1471 + /**
1472 + * The posts a finding is about, with where to fix and where to view each.
1473 + *
1474 + * The edit link is built rather than taken from get_edit_post_link(), which
1475 + * returns nothing without a user who can edit — and the analysis is cached
1476 + * and can be computed outside a request.
1477 + *
1478 + * @since 2.6.0
1479 + * @param int[] $post_ids Post ids, in sample order.
1480 + * @return array<int,array{id: int, title: string, type: string, edit_url: string, url: string}>
1481 + */
1482 + private function affected_posts(array $post_ids): array {
1483 + $posts = [];
1484 +
1485 + if ($post_ids && function_exists('_prime_post_caches')) {
1486 + _prime_post_caches($post_ids, false, false);
1487 + }
1488 +
1489 + foreach ($post_ids as $post_id) {
1490 + $post = get_post($post_id);
1491 + if (!$post) {
1492 + continue;
1493 + }
1494 +
1495 + $type_object = get_post_type_object($post->post_type);
1496 + $title = html_entity_decode((string) get_the_title($post), ENT_QUOTES, 'UTF-8');
1497 +
1498 + // An attachment's permalink is its attachment page, which core
1499 + // redirects to the file on most sites; link the file itself.
1500 + $url = 'attachment' === $post->post_type && function_exists('wp_get_attachment_url')
1501 + ? (string) wp_get_attachment_url((int) $post->ID)
1502 + : (string) get_permalink($post);
1503 +
1504 + $posts[] = [
1505 + 'id' => (int) $post->ID,
1506 + 'title' => '' !== trim($title) ? $title : __('(no title)', 'thinkrank'),
1507 + 'type' => $type_object ? (string) $type_object->labels->singular_name : (string) $post->post_type,
1508 + 'edit_url' => admin_url('post.php?post=' . (int) $post->ID . '&action=edit'),
1509 + 'url' => $url,
1510 + ];
1511 + }
1512 +
1513 + return $posts;
1514 + }
1515 +
1516 + /**
1517 + * The plain text of a post's content, with markup, blocks and shortcodes
1518 + * reduced to the words a language model would actually read.
1519 + *
1520 + * @since 2.5.0
1521 + * @param string $content Raw post content.
1522 + * @return string
1523 + */
1524 + private function content_to_text(string $content): string {
1525 + // Block delimiters are HTML comments, so strip_tags leaves their
1526 + // attribute JSON behind as text and inflates every word count.
1527 + $text = preg_replace('/<!--.*?-->/s', ' ', $content);
1528 + $text = preg_replace('/\[[^\]]*\]/', ' ', (string) $text);
1529 + $text = wp_strip_all_tags((string) $text);
1530 +
1531 + // `/u` makes preg_replace() return NULL on content that is not valid
1532 + // UTF-8 — a latin1 install, a raw SQL import, a migrated dump — and
1533 + // trim(null) is a TypeError under strict_types. run_checks() catches
1534 + // Throwable and continues, so the check did not fail: it DISAPPEARED
1535 + // from the audit, taking its weight with it and pushing the overall
1536 + // score UP because a failing check had been removed rather than scored.
1537 + // Fall back to the byte-wise collapse when the Unicode pass cannot read
1538 + // the bytes; a slightly coarser word split is worth far more than a
1539 + // check that silently deletes itself.
1540 + $collapsed = preg_replace('/\s+/u', ' ', (string) $text);
1541 +
1542 + if (null === $collapsed) {
1543 + $collapsed = preg_replace('/\s+/', ' ', (string) $text);
1544 + }
1545 +
1546 + return trim((string) $collapsed);
1547 + }
1548 +
1549 + /**
1550 + * Word count of a string of plain text.
1551 + *
1552 + * @since 2.5.0
1553 + * @param string $text Plain text.
1554 + * @return int
1555 + */
1556 + private function word_count(string $text): int {
1557 + if ('' === $text) {
1558 + return 0;
1559 + }
1560 +
1561 + return count(preg_split('/\s+/u', $text, -1, PREG_SPLIT_NO_EMPTY) ?: []);
1562 + }
1563 +
1564 + /**
1565 + * Whether AI answer engines are allowed to reach this site.
1566 + *
1567 + * Reads the robots.txt that is ACTUALLY served — physical file, custom
1568 + * body or generated defaults — rather than the per-agent rule map alone,
1569 + * because a hand-written `Disallow: /` for GPTBot blocks it just as
1570 + * effectively as the toggle does, and a `User-agent: *` block reaches
1571 + * every crawler on the list at once.
1572 + *
1573 + * @since 2.5.0
1574 + * @return array
1575 + */
1576 + public function check_ai_crawler_access(): array {
1577 + $label = __('AI answer engines can crawl your site', 'thinkrank');
1578 + $blocked = $this->blocked_answer_agents();
1579 +
1580 + if (empty($blocked)) {
1581 + return [
1582 + 'label' => $label,
1583 + 'status' => self::PASSED,
1584 + 'message' => __('ChatGPT, Claude, Perplexity and Google AI Overviews can all reach your content and cite it.', 'thinkrank'),
1585 + ];
1586 + }
1587 +
1588 + $names = implode(', ', $blocked);
1589 + // Every one of them blocked is a different situation from one stray
1590 + // rule: the site cannot appear in AI answers at all. Counted against
1591 + // the agents the registry actually knows, not the slug list: a slug
1592 + // leaving AI_Crawlers can never be collected as blocked, so comparing
1593 + // with the list would make this branch quietly unreachable.
1594 + $known = count($this->known_answer_agents());
1595 + $all = $known > 0 && count($blocked) >= $known;
1596 +
1597 + return [
1598 + 'label' => $label,
1599 + 'status' => $all ? self::FAILED : self::WARNING,
1600 + 'message' => $all
1601 + /* translators: %s: comma-separated crawler names. */
1602 + ? sprintf(__('Your robots.txt blocks every AI answer engine (%s), so your pages cannot be cited in AI answers at all.', 'thinkrank'), $names)
1603 + /* translators: %s: comma-separated crawler names. */
1604 + : sprintf(__('Your robots.txt blocks these AI answer engines: %s. Their assistants cannot read or cite your pages.', 'thinkrank'), $names),
1605 + 'how_to_fix' => __('Under Essential SEO → Crawling & AI Indexing → Robots.txt, set the answer-engine crawlers to Allow. Blocking the training-only crawlers (GPTBot, ClaudeBot, CCBot) is a separate choice and does not cost you citations.', 'thinkrank'),
1606 + 'value' => $names,
1607 + ];
1608 + }
1609 +
1610 + /**
1611 + * The answer-engine crawlers the served robots.txt disallows entirely.
1612 + *
1613 + * Public so the one-click fix can re-ask the question after writing the
1614 + * rules, rather than reporting a success the served file contradicts.
1615 + *
1616 + * @since 2.5.0
1617 + * @return string[] Crawler labels, empty when all are allowed.
1618 + */
1619 + public function blocked_answer_agents(): array {
1620 + if (!class_exists('ThinkRank\\SEO\\Site_Identity_Manager') || !class_exists('ThinkRank\\SEO\\AI_Crawlers')) {
1621 + return [];
1622 + }
1623 +
1624 + $effective = (new Site_Identity_Manager())->get_effective_robots_txt();
1625 + $groups = $this->parse_robots_disallow_all((string) ($effective['content'] ?? ''));
1626 +
1627 + if (empty($groups)) {
1628 + return [];
1629 + }
1630 +
1631 + $blocked = [];
1632 +
1633 + foreach ($this->known_answer_agents() as $agent) {
1634 + $token = strtolower((string) $agent['token']);
1635 + // A group naming the agent wins over the wildcard group, which is
1636 + // how robots.txt precedence works: the most specific matching
1637 + // group is the only one that applies.
1638 + $denied = array_key_exists($token, $groups) ? $groups[$token] : ($groups['*'] ?? false);
1639 +
1640 + if ($denied) {
1641 + $blocked[] = $agent['label'];
1642 + }
1643 + }
1644 +
1645 + return $blocked;
1646 + }
1647 +
1648 + /**
1649 + * The answer-engine crawlers the AI_Crawlers registry actually knows.
1650 + *
1651 + * The slug list is ThinkRank's editorial position on which crawlers decide
1652 + * whether a site can be cited; the registry is what carries their tokens
1653 + * and labels. Everything that counts answer engines counts these, so a slug
1654 + * that ever leaves the registry drops out of the check and its totals
1655 + * together instead of skewing one against the other.
1656 + *
1657 + * @since 2.5.0
1658 + * @return array<string,array> Registry entries keyed by slug.
1659 + */
1660 + private function known_answer_agents(): array {
1661 + if (!class_exists('ThinkRank\\SEO\\AI_Crawlers')) {
1662 + return [];
1663 + }
1664 +
1665 + $agents = AI_Crawlers::all();
1666 + $known = [];
1667 +
1668 + foreach (self::GEO_ANSWER_AGENTS as $slug) {
1669 + if (isset($agents[$slug]['token'], $agents[$slug]['label'])) {
1670 + $known[$slug] = $agents[$slug];
1671 + }
1672 + }
1673 +
1674 + return $known;
1675 + }
1676 +
1677 + /**
1678 + * Map a robots.txt body to `user-agent => disallows everything`.
1679 + *
1680 + * Consecutive `User-agent:` lines open one shared group, so the agents
1681 + * listed above a `Disallow: /` all inherit it. An agent that appears with
1682 + * narrower rules is recorded as false rather than omitted — otherwise it
1683 + * would fall through to the wildcard group it is meant to override.
1684 + *
1685 + * @since 2.5.0
1686 + * @param string $body Robots.txt content.
1687 + * @return array<string,bool>
1688 + */
1689 + private function parse_robots_disallow_all(string $body): array {
1690 + $groups = [];
1691 + $current = [];
1692 + // Whether the next `User-agent:` continues this group or opens a new
1693 + // one. Rules close a group; another agent line before any rule does not.
1694 + $collecting = true;
1695 +
1696 + foreach (preg_split('/\R/', $body) ?: [] as $line) {
1697 + $line = trim((string) preg_replace('/#.*/', '', $line));
1698 +
1699 + if ('' === $line || false === strpos($line, ':')) {
1700 + continue;
1701 + }
1702 +
1703 + [$field, $value] = array_map('trim', explode(':', $line, 2));
1704 + $field = strtolower($field);
1705 +
1706 + if ('user-agent' === $field) {
1707 + if (!$collecting) {
1708 + $current = [];
1709 + $collecting = true;
1710 + }
1711 +
1712 + $agent = strtolower($value);
1713 + if ('' !== $agent) {
1714 + $current[] = $agent;
1715 + if (!isset($groups[$agent])) {
1716 + $groups[$agent] = false;
1717 + }
1718 + }
1719 +
1720 + continue;
1721 + }
1722 +
1723 + $collecting = false;
1724 +
1725 + if ('disallow' === $field && self::disallow_blocks_everything($value)) {
1726 + foreach ($current as $agent) {
1727 + $groups[$agent] = true;
1728 + }
1729 + }
1730 + }
1731 +
1732 + return $groups;
1733 + }
1734 +
1735 + /**
1736 + * Whether a `Disallow:` value matches every URL on the site.
1737 + *
1738 + * `/` is the canonical full block; `/*` is the same instruction written
1739 + * for a wildcard-aware crawler, and every answer engine on the list is
1740 + * one. A value is a path pattern in which `*` matches any run of
1741 + * characters, so a bare `*` (and `**`, `/**`) matches every path too, and
1742 + * Google's parser treats it that way. Recognising only `/` and `/*`
1743 + * reported a site that blocked everyone with `Disallow: *` as reachable
1744 + * by every answer engine (#890).
1745 + *
1746 + * A trailing `$` anchors the pattern to the end of the URL. After a `*`
1747 + * it changes nothing, since "any run of characters, then the end" still
1748 + * matches every path, so `*$` and `/*$` are full blocks as well. Without
1749 + * a `*` it narrows the match: `/$` is the home page only, and `/*.pdf$`
1750 + * is PDFs only, so neither counts.
1751 + *
1752 + * An empty value means "allow everything" and must never count as a
1753 + * block, so it is rejected here rather than left to the caller.
1754 + *
1755 + * `Allow:` lines are deliberately not weighed against this: a full
1756 + * robots.txt evaluator with longest-match precedence is a separate change.
1757 + *
1758 + * @since 2.14.2
1759 + * @param string $value Trimmed `Disallow:` value, comment already removed.
1760 + * @return bool
1761 + */
1762 + private static function disallow_blocks_everything(string $value): bool {
1763 + if ('' === $value) {
1764 + return false;
1765 + }
1766 +
1767 + // Optional leading slash, then either nothing (`/`) or a run of `*`
1768 + // with an optional end anchor (`*`, `/*`, `/*$`, `**$`). A `$` with
1769 + // no `*` before it never matches here, so `/$` stays a partial block.
1770 + return 1 === preg_match('#^/?(?:\*+\$?)?$#', $value);
1771 + }
1772 +
1773 + /**
1774 + * /llms.txt should be published — it is the one file whose entire purpose
1775 + * is telling an AI assistant what this site is and what to read.
1776 + *
1777 + * @since 2.5.0
1778 + * @return array
1779 + */
1780 + public function check_llms_txt(): array {
1781 + $label = __('llms.txt is published', 'thinkrank');
1782 +
1783 + if (!class_exists('ThinkRank\\SEO\\LLMs_Txt_Manager')) {
1784 + return [
1785 + 'label' => $label,
1786 + 'status' => self::WARNING,
1787 + 'message' => __('The llms.txt module is unavailable, so this could not be checked.', 'thinkrank'),
1788 + ];
1789 + }
1790 +
1791 + $manager = new LLMs_Txt_Manager();
1792 +
1793 + if ($manager->is_published()) {
1794 + return [
1795 + 'label' => $label,
1796 + 'status' => self::PASSED,
1797 + 'message' => __('Your llms.txt is published, so AI assistants have a summary of your site and its key pages.', 'thinkrank'),
1798 + 'value' => home_url('/llms.txt'),
1799 + ];
1800 + }
1801 +
1802 + return [
1803 + 'label' => $label,
1804 + 'status' => self::FAILED,
1805 + 'message' => __('No llms.txt is published. It is the file AI assistants read to learn what your site is about and which pages matter.', 'thinkrank'),
1806 + 'how_to_fix' => __('Fill in the fields under Essential SEO → Crawling & AI Indexing → LLMs.txt and publish it.', 'thinkrank'),
1807 + ];
1808 + }
1809 +
1810 + /**
1811 + * Answer-shaped schema — FAQ, HowTo or Q&A — is what lets an engine lift a
1812 + * question and its answer as a pair instead of guessing at prose.
1813 + *
1814 + * @since 2.5.0
1815 + * @return array
1816 + */
1817 + public function check_answer_ready_schema(): array {
1818 + $label = __('Answer-ready structured data', 'thinkrank');
1819 +
1820 + if ($this->has_answer_schema()) {
1821 + return [
1822 + 'label' => $label,
1823 + 'status' => self::PASSED,
1824 + 'message' => __('Your site publishes FAQ, How-To or Q&A structured data, which AI answers can quote question-and-answer pairs from directly.', 'thinkrank'),
1825 + ];
1826 + }
1827 +
1828 + // Only advise FAQ/HowTo markup where there is question-and-answer
1829 + // content to mark up. This used to be assigned unconditionally and
1830 + // reused by both branches below, so a site of ordinary articles was
1831 + // told to enable FAQPage — and a user who follows that gets FAQPage
1832 + // schema with an empty or invented mainEntity, which is the failure
1833 + // #494 documents. The check is allowed to report the gap; it is not
1834 + // allowed to recommend manufacturing the content (#686).
1835 + $how_to_fix = $this->site_has_qa_content()
1836 + ? __('Add a ThinkRank FAQ or How-To block, widget or element to the pages that answer questions, or enable the FAQPage / HowTo schema types under Essential SEO → Schema.', 'thinkrank')
1837 + : __('Only mark up question-and-answer content you already publish. If this site does not answer discrete questions, answer-shaped schema does not apply and there is nothing to fix here.', 'thinkrank');
1838 +
1839 + // Article/WebPage schema still tells an engine what the page is; the
1840 + // gap is the answer pairing, not structured data as a whole.
1841 + if ($this->schema_is_output()) {
1842 + return [
1843 + 'label' => $label,
1844 + 'status' => self::WARNING,
1845 + 'message' => __('Your pages publish Article or WebPage structured data, but no FAQ, How-To or Q&A schema. Those are the types AI answers quote from.', 'thinkrank'),
1846 + 'how_to_fix' => $how_to_fix,
1847 + ];
1848 + }
1849 +
1850 + return [
1851 + 'label' => $label,
1852 + 'status' => self::FAILED,
1853 + 'message' => __('Your pages publish no structured data at all, so an AI assistant has to infer what each page is from its prose.', 'thinkrank'),
1854 + 'how_to_fix' => $how_to_fix,
1855 + ];
1856 + }
1857 +
1858 + /**
1859 + * Whether the site plausibly publishes question-and-answer content.
1860 + *
1861 + * Deliberately a different question from has_answer_schema(). That one asks
1862 + * "does ThinkRank emit answer markup?", which is what the check reports on.
1863 + * This one asks "is there anything here that answer markup would describe?",
1864 + * which is what decides whether recommending it is sound advice — and it has
1865 + * to see content ThinkRank did not author, because an FAQ built with a page
1866 + * builder's own accordion is still an FAQ (#686).
1867 + *
1868 + * Any one signal is enough; all of them are bounded.
1869 + *
1870 + * @since 2.7.0
1871 + * @return bool
1872 + */
1873 + private function site_has_qa_content(): bool {
1874 + if (null !== $this->site_has_qa_content) {
1875 + return $this->site_has_qa_content;
1876 + }
1877 +
1878 + $this->site_has_qa_content = $this->qa_content_markers_exist()
1879 + || $this->sample_has_question_headings();
1880 +
1881 + return $this->site_has_qa_content;
1882 + }
1883 +
1884 + /**
1885 + * Q&A markers in stored content or builder trees, site-wide.
1886 + *
1887 + * Two bounded queries rather than a walk over the content sample: the
1888 + * sample is the 100 most recent posts and pages, so a site whose only FAQ
1889 + * lives in a custom post type, or further back than that, answered "no
1890 + * answer content" while publishing exactly that (#686).
1891 + *
1892 + * @since 2.7.0
1893 + * @return bool
1894 + */
1895 + private function qa_content_markers_exist(): bool {
1896 + global $wpdb;
1897 +
1898 + // Block markup and core's details/summary block, in any public type.
1899 + $content_markers = ['wp:thinkrank/faq', 'wp:thinkrank/howto', 'wp:details'];
1900 + $clauses = [];
1901 + $values = [];
1902 +
1903 + foreach ($content_markers as $marker) {
1904 + $clauses[] = 'p.post_content LIKE %s';
1905 + $values[] = '%' . $wpdb->esc_like($marker) . '%';
1906 + }
1907 +
1908 + // The OR list is one '%s' per entry in a fixed class-level marker list,
1909 + // so its length varies but its content never comes from input; every
1910 + // value goes through prepare(). phpcs cannot see that, and this is the
1911 + // usual variable-length-IN exemption.
1912 + // phpcs:disable WordPress.DB.PreparedSQL.NotPrepared, WordPress.DB.PreparedSQL.InterpolatedNotPrepared, WordPress.DB.PreparedSQLPlaceholders.UnfinishedPrepare, WordPress.DB.DirectDatabaseQuery.DirectQuery, WordPress.DB.DirectDatabaseQuery.NoCaching
1913 + $found = (int) $wpdb->get_var(
1914 + $wpdb->prepare(
1915 + "SELECT 1 FROM {$wpdb->posts} p
1916 + WHERE p.post_status = 'publish'
1917 + AND (" . implode(' OR ', $clauses) . ')
1918 + LIMIT 1',
1919 + ...$values
1920 + )
1921 + );
1922 + // phpcs:enable
1923 +
1924 + if (1 === $found) {
1925 + return true;
1926 + }
1927 +
1928 + // Builder trees: ThinkRank's own elements, and the builders' generic
1929 + // accordion/toggle/FAQ elements, which are what a non-ThinkRank FAQ
1930 + // is actually built from. Only layouts the visitor is served count.
1931 + $meta_keys = self::rendered_builder_meta_keys();
1932 +
1933 + if (empty($meta_keys)) {
1934 + return false;
1935 + }
1936 +
1937 + $markers = array_merge(
1938 + [self::ANSWER_FAQ_NAME, self::ANSWER_HOWTO_NAME],
1939 + self::GENERIC_QA_ELEMENT_MARKERS
1940 + );
1941 +
1942 + $key_placeholders = implode(', ', array_fill(0, count($meta_keys), '%s'));
1943 + $marker_clauses = [];
1944 + $marker_values = [];
1945 +
1946 + foreach ($markers as $marker) {
1947 + $marker_clauses[] = 'pm.meta_value LIKE %s';
1948 + $marker_values[] = '%' . $wpdb->esc_like($marker) . '%';
1949 + }
1950 +
1951 + [$gate_sql, $gate_values] = self::builder_layout_renders_sql('pm');
1952 +
1953 + // Same exemption: both placeholder runs are sized from fixed lists —
1954 + // the builder meta keys and the marker list — and every value is
1955 + // prepared.
1956 + // phpcs:disable WordPress.DB.PreparedSQL.NotPrepared, WordPress.DB.PreparedSQL.InterpolatedNotPrepared, WordPress.DB.PreparedSQLPlaceholders.UnfinishedPrepare, WordPress.DB.DirectDatabaseQuery.DirectQuery, WordPress.DB.DirectDatabaseQuery.NoCaching
1957 + $found = (int) $wpdb->get_var(
1958 + $wpdb->prepare(
1959 + "SELECT 1 FROM {$wpdb->postmeta} pm
1960 + INNER JOIN {$wpdb->posts} p ON p.ID = pm.post_id
1961 + WHERE p.post_status = 'publish'
1962 + AND pm.meta_key IN ({$key_placeholders})
1963 + AND (" . implode(' OR ', $marker_clauses) . ")
1964 + {$gate_sql}
1965 + LIMIT 1",
1966 + ...array_merge($meta_keys, $marker_values, $gate_values)
1967 + )
1968 + );
1969 + // phpcs:enable
1970 +
1971 + return 1 === $found;
1972 + }
1973 +
1974 + /**
1975 + * Whether the sampled content asks questions in its headings.
1976 + *
1977 + * The weakest signal and the last one tried: prose that poses questions is
1978 + * content answer markup could describe, even where nothing has been marked
1979 + * up yet. Runs over the existing sample, so it costs nothing extra.
1980 + *
1981 + * @since 2.7.0
1982 + * @return bool
1983 + */
1984 + private function sample_has_question_headings(): bool {
1985 + foreach ($this->get_content_sample() as $row) {
1986 + $content = (string) ($row['content'] ?? '');
1987 +
1988 + if ('' === $content) {
1989 + continue;
1990 + }
1991 +
1992 + if (preg_match('/<h[2-4][^>]*>\s*[^<]*\?\s*<\/h[2-4]>/i', $content)) {
1993 + return true;
1994 + }
1995 + }
1996 +
1997 + return false;
1998 + }
1999 +
2000 + /**
2001 + * Whether the site publishes FAQ / HowTo / Q&A structured data.
2002 + *
2003 + * Three sources, because three things emit it: the Schema Management
2004 + * System's enabled types, a per-post-type schema_type in Global SEO, and
2005 + * ThinkRank's own FAQ/HowTo blocks, which output their schema from the
2006 + * block itself with nothing to configure.
2007 + *
2008 + * @since 2.5.0
2009 + * @return bool
2010 + */
2011 + private function has_answer_schema(): bool {
2012 + $answer_types = ['faqpage', 'faq', 'howto', 'qapage'];
2013 +
2014 + if (class_exists('ThinkRank\\SEO\\Schema_Management_System')) {
2015 + $settings = (new Schema_Management_System())->get_settings('site', null);
2016 +
2017 + if (is_array($settings) && !empty($settings['enabled'])) {
2018 + $enabled = $settings['enabled_schema_types'] ?? [];
2019 + if (is_array($enabled)) {
2020 + foreach ($enabled as $type) {
2021 + if (in_array(strtolower((string) $type), $answer_types, true)) {
2022 + return true;
2023 + }
2024 + }
2025 + }
2026 + }
2027 + }
2028 +
2029 + $global = get_option('thinkrank_global_seo_settings', []);
2030 + if (is_array($global)) {
2031 + foreach ($global as $per_type) {
2032 + $type = is_array($per_type) ? strtolower((string) ($per_type['schema_type'] ?? '')) : '';
2033 + if ('' !== $type && in_array($type, $answer_types, true)) {
2034 + return true;
2035 + }
2036 + }
2037 + }
2038 +
2039 + // The blocks carry their own schema, so a single page using one is a
2040 + // true positive even with every schema setting untouched.
2041 + foreach ($this->get_content_sample() as $row) {
2042 + if ($this->has_answer_content((int) ($row['id'] ?? 0), (string) $row['content'])) {
2043 + return true;
2044 + }
2045 + }
2046 +
2047 + // ...and again beyond the sample, which is the 100 most recent posts
2048 + // and pages. A site whose only FAQ lives in a custom post type, or
2049 + // simply further back than that, reported no answer schema while
2050 + // ThinkRank was publishing exactly that (#686). Added alongside the
2051 + // walk above rather than replacing it: the sample is already loaded and
2052 + // resolves Bricks trees, so it stays the primary source and this only
2053 + // extends the reach.
2054 + foreach ($this->answer_content_candidates() as $post_id => $content) {
2055 + if ($this->has_answer_content($post_id, $content)) {
2056 + return true;
2057 + }
2058 + }
2059 +
2060 + return false;
2061 + }
2062 +
2063 + /**
2064 + * Posts that might carry a ThinkRank FAQ / How-To, from anywhere on the site.
2065 + *
2066 + * Narrowed in SQL to posts whose content or builder meta names one of
2067 + * ThinkRank's answer surfaces, so the per-post confirmation below runs over
2068 + * a handful of rows rather than the whole site.
2069 + *
2070 + * @since 2.7.0
2071 + * @return array<int,string> Post id => raw content.
2072 + */
2073 + private function answer_content_candidates(): array {
2074 + global $wpdb;
2075 +
2076 + $markers = [self::ANSWER_FAQ_NAME, self::ANSWER_HOWTO_NAME, 'wp:thinkrank/faq', 'wp:thinkrank/howto'];
2077 +
2078 + $content_clauses = [];
2079 + $content_values = [];
2080 + foreach ($markers as $marker) {
2081 + $content_clauses[] = 'p.post_content LIKE %s';
2082 + $content_values[] = '%' . $wpdb->esc_like($marker) . '%';
2083 + }
2084 +
2085 + $meta_keys = self::rendered_builder_meta_keys();
2086 + $meta_sql = '';
2087 + $values = $content_values;
2088 +
2089 + if (!empty($meta_keys)) {
2090 + $meta_clauses = [];
2091 + $meta_values = [];
2092 + foreach ($markers as $marker) {
2093 + $meta_clauses[] = 'pm.meta_value LIKE %s';
2094 + $meta_values[] = '%' . $wpdb->esc_like($marker) . '%';
2095 + }
2096 +
2097 + [$gate_sql, $gate_values] = self::builder_layout_renders_sql('pm');
2098 +
2099 + $key_placeholders = implode(', ', array_fill(0, count($meta_keys), '%s'));
2100 + $meta_sql = " OR EXISTS (
2101 + SELECT 1 FROM {$wpdb->postmeta} pm
2102 + WHERE pm.post_id = p.ID
2103 + AND pm.meta_key IN ({$key_placeholders})
2104 + AND (" . implode(' OR ', $meta_clauses) . ")
2105 + {$gate_sql}
2106 + )";
2107 + $values = array_merge($content_values, $meta_keys, $meta_values, $gate_values);
2108 + }
2109 +
2110 + $values[] = self::CONTENT_SAMPLE_SIZE;
2111 +
2112 + // Same exemption as qa_content_markers_exist(): the clause lists are
2113 + // sized from fixed marker and meta-key lists, and every value is
2114 + // prepared.
2115 + // phpcs:disable WordPress.DB.PreparedSQL.NotPrepared, WordPress.DB.PreparedSQL.InterpolatedNotPrepared, WordPress.DB.PreparedSQLPlaceholders.UnfinishedPrepare, WordPress.DB.DirectDatabaseQuery.DirectQuery, WordPress.DB.DirectDatabaseQuery.NoCaching
2116 + $rows = $wpdb->get_results(
2117 + $wpdb->prepare(
2118 + "SELECT p.ID, p.post_content FROM {$wpdb->posts} p
2119 + WHERE p.post_status = 'publish'
2120 + AND ((" . implode(' OR ', $content_clauses) . ")
2121 + {$meta_sql})
2122 + ORDER BY p.post_date DESC
2123 + LIMIT %d",
2124 + ...$values
2125 + ),
2126 + ARRAY_A
2127 + );
2128 + // phpcs:enable
2129 +
2130 + $candidates = [];
2131 + foreach ((array) $rows as $row) {
2132 + $candidates[(int) $row['ID']] = (string) $row['post_content'];
2133 + }
2134 +
2135 + return $candidates;
2136 + }
2137 +
2138 + /**
2139 + * Whether one post carries a ThinkRank FAQ or How-To that emits schema.
2140 + *
2141 + * Four surfaces, because `Schema_Graph` collects from four: the Gutenberg
2142 + * block in `post_content`, the Elementor widget, the Bricks element and the
2143 + * Beaver Builder module — the last three living in postmeta. Reading only
2144 + * the block would tell a site whose FAQs are built in a page builder that it
2145 + * publishes no FAQ schema while the graph is publishing exactly that.
2146 + *
2147 + * @since 2.5.0
2148 + * @param int $post_id Post to inspect; 0 skips the builder surfaces.
2149 + * @param string $content Raw post content.
2150 + * @return bool
2151 + */
2152 + private function has_answer_content(int $post_id, string $content): bool {
2153 + if (false !== strpos($content, 'wp:thinkrank/faq')
2154 + || false !== strpos($content, 'wp:thinkrank/howto')) {
2155 + return true;
2156 + }
2157 +
2158 + if ($post_id <= 0) {
2159 + return false;
2160 + }
2161 +
2162 + // Elementor stores its tree as JSON, so the widget name appears verbatim.
2163 + // Read only while Elementor renders the post: the tree survives
2164 + // switching the page back to the block editor (#945).
2165 + $elementor = self::builder_layout_renders($post_id, '_elementor_data')
2166 + ? get_post_meta($post_id, '_elementor_data', true)
2167 + : '';
2168 + if (is_string($elementor)
2169 + && (false !== strpos($elementor, '"' . self::ANSWER_FAQ_NAME . '"')
2170 + || false !== strpos($elementor, '"' . self::ANSWER_HOWTO_NAME . '"'))) {
2171 + return true;
2172 + }
2173 +
2174 + // Bricks: the tree it will actually render, resolved the same way
2175 + // Schema_Graph resolves it, so templates and components are covered.
2176 + if ($this->bricks_has_answer_element($post_id)) {
2177 + return true;
2178 + }
2179 +
2180 + // Beaver Builder keeps its layout in postmeta as a map of node objects.
2181 + // Read only while Beaver renders the post, for the same reason (#945).
2182 + $layout = self::builder_layout_renders($post_id, '_fl_builder_data')
2183 + ? get_post_meta($post_id, '_fl_builder_data', true)
2184 + : [];
2185 + if (is_array($layout)) {
2186 + foreach ($layout as $node) {
2187 + $settings = is_object($node) ? ($node->settings ?? null) : ($node['settings'] ?? null);
2188 + $settings = is_object($settings) ? get_object_vars($settings) : $settings;
2189 + $type = is_array($settings) ? (string) ($settings['type'] ?? '') : '';
2190 +
2191 + if (self::ANSWER_FAQ_NAME === $type || self::ANSWER_HOWTO_NAME === $type) {
2192 + return true;
2193 + }
2194 + }
2195 + }
2196 +
2197 + // Oxygen and Breakdance were missing entirely, so a ThinkRank FAQ
2198 + // element placed inside one of those pages reported no answer schema
2199 + // while the page was publishing exactly that (#686). Builder_Content
2200 + // already knows every key involved — Oxygen 6 is Breakdance under the
2201 + // hood, and older releases used two other keys — so ask it rather than
2202 + // keeping a second list that can drift.
2203 + foreach (self::rendered_builder_meta_keys() as $meta_key) {
2204 + if ('_fl_builder_data' === $meta_key || '_elementor_data' === $meta_key) {
2205 + continue; // Handled above, in their own storage shapes.
2206 + }
2207 +
2208 + $stored = get_post_meta($post_id, $meta_key, true);
2209 +
2210 + if (self::blob_names_answer_element($stored)) {
2211 + return true;
2212 + }
2213 + }
2214 +
2215 + return false;
2216 + }
2217 +
2218 + /**
2219 + * Builder meta keys, or [] when Builder_Content is unavailable.
2220 + *
2221 + * Mirrors the defensive load in bricks_has_answer_element(): the analyzer
2222 + * must degrade to "no answer content found" on a partial checkout rather
2223 + * than fatal mid-audit.
2224 + *
2225 + * @since 2.7.0
2226 + * @return string[]
2227 + */
2228 + private static function builder_meta_keys(): array {
2229 + if (!class_exists('ThinkRank\\SEO\\Builder_Content')) {
2230 + $file = THINKRANK_PLUGIN_DIR . 'includes/seo/class-builder-content.php';
2231 + if (!file_exists($file)) {
2232 + return [];
2233 + }
2234 + require_once $file;
2235 + }
2236 +
2237 + return Builder_Content::builder_meta_keys();
2238 + }
2239 +
2240 + /**
2241 + * Builder meta keys that can hold the published page.
2242 + *
2243 + * {@see builder_meta_keys()} without the unpublished draft keys, which
2244 + * Builder_Content lists because the word-count index watches them, not
2245 + * because they are ever served (#945).
2246 + *
2247 + * @since 2.14.2
2248 + * @return string[]
2249 + */
2250 + private static function rendered_builder_meta_keys(): array {
2251 + return array_values(array_diff(self::builder_meta_keys(), self::UNPUBLISHED_BUILDER_META_KEYS));
2252 + }
2253 +
2254 + /**
2255 + * Whether a builder layout key holds what the visitor is served.
2256 + *
2257 + * The answer-content readers' guard for #945. Builder_Content gains the
2258 + * same guards for its resolver in #910, and once that lands this can ask
2259 + * it instead; until then it reads the builders' own flags, exactly as
2260 + * FAQ_Content::builder() does.
2261 + *
2262 + * @since 2.14.2
2263 + * @param int $post_id Post being inspected.
2264 + * @param string $meta_key Builder meta key about to be read.
2265 + * @return bool
2266 + */
2267 + private static function builder_layout_renders(int $post_id, string $meta_key): bool {
2268 + if (in_array($meta_key, self::UNPUBLISHED_BUILDER_META_KEYS, true)) {
2269 + return false;
2270 + }
2271 +
2272 + if (!isset(self::FLAGGED_BUILDER_LAYOUTS[$meta_key])) {
2273 + return true;
2274 + }
2275 +
2276 + [$flag, $on] = self::FLAGGED_BUILDER_LAYOUTS[$meta_key];
2277 + $value = get_post_meta($post_id, $flag, true);
2278 +
2279 + return null === $on ? !empty($value) : $on === (string) $value;
2280 + }
2281 +
2282 + /**
2283 + * {@see builder_layout_renders()} as an SQL condition on a postmeta row.
2284 + *
2285 + * The site-wide queries match markers in SQL with nothing confirming each
2286 + * row in PHP afterwards, so the flag test has to happen there too. Rows
2287 + * under any other key pass untouched. `!empty()` in PHP is false for '' and
2288 + * '0', which is what the "any non-empty value" branch excludes.
2289 + *
2290 + * @since 2.14.2
2291 + * @param string $alias Alias of the postmeta row being tested.
2292 + * @return array{0:string,1:array<int,string>} ` AND (...)` clause and its values.
2293 + */
2294 + private static function builder_layout_renders_sql(string $alias): array {
2295 + global $wpdb;
2296 +
2297 + $flagged = array_keys(self::FLAGGED_BUILDER_LAYOUTS);
2298 + $clauses = [$alias . '.meta_key NOT IN (' . implode(', ', array_fill(0, count($flagged), '%s')) . ')'];
2299 + $values = $flagged;
2300 +
2301 + foreach (self::FLAGGED_BUILDER_LAYOUTS as $meta_key => [$flag, $on]) {
2302 + $test = null === $on ? "flag.meta_value NOT IN ('', '0')" : 'flag.meta_value = %s';
2303 + $clauses[] = "({$alias}.meta_key = %s AND EXISTS (
2304 + SELECT 1 FROM {$wpdb->postmeta} flag
2305 + WHERE flag.post_id = {$alias}.post_id
2306 + AND flag.meta_key = %s
2307 + AND {$test}
2308 + ))";
2309 + $values[] = $meta_key;
2310 + $values[] = $flag;
2311 +
2312 + if (null !== $on) {
2313 + $values[] = $on;
2314 + }
2315 + }
2316 +
2317 + return [' AND (' . implode(' OR ', $clauses) . ')', $values];
2318 + }
2319 +
2320 + /**
2321 + * Whether a stored builder blob names a ThinkRank FAQ or How-To.
2322 + *
2323 + * The blob is a JSON tree for Breakdance/Oxygen 6 and a shortcode string
2324 + * for Oxygen classic, so this matches on the element name appearing in the
2325 + * serialized form rather than parsing each dialect.
2326 + *
2327 + * @since 2.7.0
2328 + * @param mixed $stored Raw meta value.
2329 + * @return bool
2330 + */
2331 + private static function blob_names_answer_element($stored): bool {
2332 + if (is_array($stored)) {
2333 + $stored = wp_json_encode($stored);
2334 + }
2335 +
2336 + if (!is_string($stored) || '' === $stored) {
2337 + return false;
2338 + }
2339 +
2340 + return false !== strpos($stored, self::ANSWER_FAQ_NAME)
2341 + || false !== strpos($stored, self::ANSWER_HOWTO_NAME);
2342 + }
2343 +
2344 + /**
2345 + * Whether a Bricks-rendered post holds a ThinkRank FAQ or How-To element.
2346 + *
2347 + * @since 2.5.0
2348 + * @param int $post_id Post to inspect.
2349 + * @return bool
2350 + */
2351 + private function bricks_has_answer_element(int $post_id): bool {
2352 + if (!class_exists('ThinkRank\\SEO\\Builder_Content')) {
2353 + $file = THINKRANK_PLUGIN_DIR . 'includes/seo/class-builder-content.php';
2354 + if (!file_exists($file)) {
2355 + return false;
2356 + }
2357 + require_once $file;
2358 + }
2359 +
2360 + foreach (Builder_Content::bricks_tree($post_id) as $element) {
2361 + $name = is_array($element) ? (string) ($element['name'] ?? '') : '';
2362 +
2363 + if (self::ANSWER_FAQ_NAME === $name || self::ANSWER_HOWTO_NAME === $name) {
2364 + return true;
2365 + }
2366 + }
2367 +
2368 + return false;
2369 + }
2370 +
2371 + /**
2372 + * Pages should open with a short, self-contained answer an engine can lift.
2373 + *
2374 + * @since 2.5.0
2375 + * @return array
2376 + */
2377 + public function check_direct_answer(): array {
2378 + return $this->coverage_check(
2379 + __('Pages open with a direct answer', 'thinkrank'),
2380 + function (array $row): bool {
2381 + $words = $this->word_count($this->opening_passage($row['content']));
2382 +
2383 + return $words >= self::GEO_ANSWER_MIN_WORDS && $words <= self::GEO_ANSWER_MAX_WORDS;
2384 + },
2385 + /* translators: 1: matching posts, 2: sampled posts. */
2386 + __('%1$d of your %2$d most recent pages open with a concise, quotable answer.', 'thinkrank'),
2387 + /* translators: 1: posts without one, 2: sampled posts. */
2388 + __('%1$d of your %2$d most recent pages do not open with a concise answer. AI assistants quote the first self-contained passage they find, and a long preamble gives them nothing to lift.', 'thinkrank'),
2389 + __('Open each page with a 40–60 word paragraph that answers its title directly, before any background or introduction.', 'thinkrank'),
2390 + __('No published content to check yet.', 'thinkrank')
2391 + );
2392 + }
2393 +
2394 + /**
2395 + * The first block of prose in a post, before any heading.
2396 + *
2397 + * @since 2.5.0
2398 + * @param string $content Raw post content.
2399 + * @return string Plain text of the opening passage.
2400 + */
2401 + private function opening_passage(string $content): string {
2402 + // Cut at the first heading: everything above it is the intro, and an
2403 + // intro that runs past a heading is not a direct answer either way.
2404 + $chunks = preg_split('/<h[1-6][^>]*>/i', $content, 3) ?: [];
2405 + $intro = (string) ($chunks[0] ?? $content);
2406 +
2407 + $passage = $this->first_paragraph_text($intro);
2408 + if ('' !== $passage) {
2409 + return $passage;
2410 + }
2411 +
2412 + $text = $this->content_to_text($intro);
2413 +
2414 + // A page that opens with its heading has nothing above it, so the chunk
2415 + // read so far is empty and the answer sits directly under that heading.
2416 + // Scoring it as a missing opening would fail exactly the pages written
2417 + // as "question, then answer" — the shape this check exists to reward.
2418 + if ('' === trim($text) && isset($chunks[1])) {
2419 + // Drop the heading's own text, which the split left at the head of
2420 + // the next chunk, so a long heading cannot pose as the answer.
2421 + $after = preg_replace('/^.*?<\/h[1-6]>/is', '', (string) $chunks[1], 1);
2422 + $after = null === $after ? (string) $chunks[1] : $after;
2423 +
2424 + $passage = $this->first_paragraph_text($after);
2425 + if ('' !== $passage) {
2426 + return $passage;
2427 + }
2428 +
2429 + return $this->content_to_text($after);
2430 + }
2431 +
2432 + return $text;
2433 + }
2434 +
2435 + /**
2436 + * The first paragraph of a chunk of content with enough words to be a
2437 + * passage rather than a caption or a stray line.
2438 + *
2439 + * Only the first one counts — a three-paragraph intro is exactly the
2440 + * preamble the direct-answer check is looking for.
2441 + *
2442 + * @since 2.5.0
2443 + * @param string $chunk Raw content fragment.
2444 + * @return string Plain text, or '' when the chunk holds no paragraph.
2445 + */
2446 + private function first_paragraph_text(string $chunk): string {
2447 + $paragraphs = preg_split('/<\/p>|\R{2,}/', $chunk) ?: [];
2448 +
2449 + foreach ($paragraphs as $paragraph) {
2450 + $candidate = $this->content_to_text((string) $paragraph);
2451 + if ($this->word_count($candidate) >= 5) {
2452 + return $candidate;
2453 + }
2454 + }
2455 +
2456 + return '';
2457 + }
2458 +
2459 + /**
2460 + * Question-shaped H2/H3 headings map a page onto the questions people
2461 + * actually ask an assistant.
2462 + *
2463 + * @since 2.5.0
2464 + * @return array
2465 + */
2466 + public function check_question_headings(): array {
2467 + return $this->coverage_check(
2468 + __('Headings phrased as questions', 'thinkrank'),
2469 + function (array $row): bool {
2470 + return $this->has_question_heading($row['content']);
2471 + },
2472 + /* translators: 1: matching posts, 2: sampled posts. */
2473 + __('%1$d of your %2$d most recent pages use question-style headings.', 'thinkrank'),
2474 + /* translators: 1: posts without one, 2: sampled posts. */
2475 + __('%1$d of your %2$d most recent pages have no question-style heading. Assistants match a user\'s question against your headings first.', 'thinkrank'),
2476 + __('Phrase at least one H2 or H3 per page as the question it answers: "How does X work?" rather than "Overview".', 'thinkrank'),
2477 + __('No published content to check yet.', 'thinkrank')
2478 + );
2479 + }
2480 +
2481 + /**
2482 + * Whether any H2/H3 in the content reads as a question.
2483 + *
2484 + * @since 2.5.0
2485 + * @param string $content Raw post content.
2486 + * @return bool
2487 + */
2488 + private function has_question_heading(string $content): bool {
2489 + if (!preg_match_all('/<h[23][^>]*>(.*?)<\/h[23]>/is', $content, $matches)) {
2490 + return false;
2491 + }
2492 +
2493 + $starters = ['what', 'why', 'how', 'when', 'where', 'who', 'which', 'can', 'do', 'does', 'is', 'are', 'should', 'will'];
2494 +
2495 + foreach ($matches[1] as $heading) {
2496 + $text = $this->content_to_text((string) $heading);
2497 +
2498 + if ('' === $text) {
2499 + continue;
2500 + }
2501 +
2502 + if ('?' === substr($text, -1)) {
2503 + return true;
2504 + }
2505 +
2506 + // A question mark is the reliable signal, but plenty of good
2507 + // question headings drop it ("How image search works").
2508 + $first = strtolower((string) strtok($text, " \t\n"));
2509 + if (in_array($first, $starters, true)) {
2510 + return true;
2511 + }
2512 + }
2513 +
2514 + return false;
2515 + }
2516 +
2517 + /**
2518 + * Lists and tables are the shapes an engine extracts most reliably.
2519 + *
2520 + * @since 2.5.0
2521 + * @return array
2522 + */
2523 + public function check_structured_content(): array {
2524 + return $this->coverage_check(
2525 + __('Content uses lists or tables', 'thinkrank'),
2526 + static function (array $row): bool {
2527 + return (bool) preg_match('/<(ul|ol|table)[\s>]/i', $row['content']);
2528 + },
2529 + /* translators: 1: matching posts, 2: sampled posts. */
2530 + __('%1$d of your %2$d most recent pages present information in lists or tables.', 'thinkrank'),
2531 + /* translators: 1: posts without one, 2: sampled posts. */
2532 + __('%1$d of your %2$d most recent pages are unbroken prose. Steps, comparisons and specifications are extracted far more reliably from a list or table.', 'thinkrank'),
2533 + __('Break steps, comparisons and specifications out of the paragraphs into list or table blocks.', 'thinkrank'),
2534 + __('No published content to check yet.', 'thinkrank')
2535 + );
2536 + }
2537 +
2538 + /**
2539 + * Thin pages are not cited: there is nothing in them worth quoting.
2540 + *
2541 + * @since 2.5.0
2542 + * @return array
2543 + */
2544 + public function check_content_depth(): array {
2545 + return $this->coverage_check(
2546 + __('Pages have enough depth to cite', 'thinkrank'),
2547 + function (array $row): bool {
2548 + return $this->word_count($row['text']) >= self::GEO_DEPTH_MIN_WORDS;
2549 + },
2550 + /* translators: 1: matching posts, 2: sampled posts. */
2551 + __('%1$d of your %2$d most recent pages have enough substance for an assistant to cite.', 'thinkrank'),
2552 + /* translators: 1: thin posts, 2: sampled posts. */
2553 + __('%1$d of your %2$d most recent pages are under 300 words. An assistant with a choice of sources rarely quotes the thinnest one.', 'thinkrank'),
2554 + __('Expand thin pages so each one answers its topic completely, or merge them into a page that does.', 'thinkrank'),
2555 + __('No published content to check yet.', 'thinkrank')
2556 + );
2557 + }
2558 +
2559 + /**
2560 + * Answer engines strongly prefer recently-revised sources.
2561 + *
2562 + * @since 2.5.0
2563 + * @return array
2564 + */
2565 + public function check_content_freshness(): array {
2566 + $cutoff = time() - self::GEO_FRESHNESS_MAX_AGE;
2567 +
2568 + return $this->coverage_check(
2569 + __('Content has been updated recently', 'thinkrank'),
2570 + static function (array $row) use ($cutoff): bool {
2571 + return $row['modified'] > 0 && $row['modified'] >= $cutoff;
2572 + },
2573 + /* translators: 1: matching posts, 2: sampled posts. */
2574 + __('%1$d of your %2$d most recent pages were revised within the last year.', 'thinkrank'),
2575 + /* translators: 1: stale posts, 2: sampled posts. */
2576 + __('%1$d of your %2$d most recent pages have not been revised in over a year. AI answers favour sources that look current.', 'thinkrank'),
2577 + __('Review your most important pages, update what has changed, and save them so their modified date reflects the revision.', 'thinkrank'),
2578 + __('No published content to check yet.', 'thinkrank')
2579 + );
784 2580 }
785 2581 }