PluginProbe
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO / 2.11.0
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO v2.11.0
2.11.0 2.10.0 2.9.0 2.8.0 2.7.0 2.6.0 2.5.0 2.4.0 2.3.0 2.2.0 2.1.1 2.1.0 2.0.2 2.0.1 2.0.0 1.32.0 1.31.0 1.30.0 1.29.0 1.28.0 1.27.0 1.26.0 1.25.0 trunk 1.0.0 All 52 releases
← All changes | includes/ai/class-content-brief-generator.php +259 -23 1.29.0 → 2.11.0 View file →
@@ -83,9 +83,11 @@
83 83
84 84 /**
85 85 * AI client instance
86 86 *
87 - * @var OpenAI_Client|Claude_Client
87 + * Null when the generator was built for storage-only work.
88 + *
89 + * @var OpenAI_Client|Claude_Client|null
88 90 */
89 91 private $ai_client;
90 92
91 93 /**
@@ -92,18 +94,31 @@
92 94 * Constructor
93 95 *
94 96 * @param Settings|null $settings Settings instance
95 97 * @param OpenAI_Client|Claude_Client|null $ai_client AI client instance
98 + * @param bool $require_ai_client Whether a provider client is required. Pass
99 + * false for storage-only use (list/export/
100 + * delete), which never calls a provider.
96 101 */
97 - public function __construct(?Settings $settings = null, $ai_client = null) {
102 + public function __construct(?Settings $settings = null, $ai_client = null, bool $require_ai_client = true) {
98 103 $this->settings = $settings ?? Settings::instance();
99 104
100 105 if ($ai_client) {
101 106 $this->ai_client = $ai_client;
102 - } else {
103 - // Fallback to creating own client for backward compatibility
104 - $this->init_ai_client();
107 +
108 + return;
105 109 }
110 +
111 + // Read-only callers (listing, exporting and deleting saved briefs) only
112 + // touch the database and never reach a provider. Constructing a client
113 + // for them turns "no API key configured" — the default state of a fresh
114 + // install — into a hard failure, so let them opt out.
115 + if (!$require_ai_client) {
116 + return;
117 + }
118 +
119 + // Fallback to creating own client for backward compatibility
120 + $this->init_ai_client();
106 121 }
107 122
108 123 /**
109 124 * Initialize AI client based on available API keys
@@ -112,9 +127,9 @@
112 127 *
113 128 * @throws \Exception On failure.
114 129 */
115 130 private function init_ai_client(): void {
116 - $provider = $this->settings->get('ai_provider', 'openai');
131 + $provider = $this->settings->get('ai_provider', Settings::AI_PROVIDER_NONE);
117 132
118 133 if ($provider === 'openai') {
119 134 $api_key = $this->settings->get('openai_api_key');
120 135 if ($api_key) {
@@ -138,11 +153,31 @@
138 153 if ($api_key) {
139 154 $model = $this->settings->get('openrouter_model', Settings::DEFAULT_OPENROUTER_MODEL);
140 155 $this->ai_client = new OpenRouter_Client($api_key, $model, self::AI_REQUEST_TIMEOUT);
141 156 }
157 + } elseif ($provider === 'openai_compatible') {
158 + // Same client as OpenAI, different host — and the key is optional,
159 + // so the URL and model id are what gate it (#721). The user's own
160 + // timeout applies: a local model writing a brief on CPU is slow,
161 + // and the setting exists for exactly that.
162 + $base_url = (string) $this->settings->get('openai_compatible_base_url', '');
163 + $model = trim((string) $this->settings->get('openai_compatible_model', ''));
164 + if ('' !== $base_url && '' !== $model) {
165 + $this->ai_client = new OpenAI_Client(
166 + (string) $this->settings->get('openai_compatible_api_key', ''),
167 + $model,
168 + (int) $this->settings->get('openai_compatible_timeout', Settings::DEFAULT_OPENAI_COMPATIBLE_TIMEOUT),
169 + $base_url
170 + );
171 + $this->ai_client->set_json_mode((bool) $this->settings->get('openai_compatible_json_mode', false));
172 + }
142 173 }
143 174
144 175 if (!$this->ai_client) {
176 + if ('openai_compatible' === $provider) {
177 + throw new \Exception('Please set the base URL and model id for your OpenAI-compatible endpoint in ThinkRank settings.');
178 + }
179 +
145 180 throw new \Exception('Please configure your AI provider API key in ThinkRank settings.');
146 181 }
147 182 }
148 183
@@ -157,9 +192,15 @@
157 192 return $this->ai_client->get_model();
158 193 }
159 194
160 195 // Fallback to settings
161 - $provider = $this->settings->get('ai_provider', 'openai');
196 + $provider = $this->settings->get('ai_provider', Settings::AI_PROVIDER_NONE);
197 + if (Settings::AI_PROVIDER_NONE === $provider) {
198 + // No provider chosen, so there is no model to name. Reporting the
199 + // OpenAI default here would attribute output to a provider the site
200 + // never selected (#572).
201 + return '';
202 + }
162 203 if ($provider === 'claude') {
163 204 return $this->settings->get('claude_model', Settings::DEFAULT_CLAUDE_MODEL);
164 205 } elseif ($provider === 'gemini') {
165 206 return $this->settings->get('gemini_model', Settings::DEFAULT_GEMINI_MODEL);
@@ -164,8 +205,10 @@
164 205 } elseif ($provider === 'gemini') {
165 206 return $this->settings->get('gemini_model', Settings::DEFAULT_GEMINI_MODEL);
166 207 } elseif ($provider === 'openrouter') {
167 208 return $this->settings->get('openrouter_model', Settings::DEFAULT_OPENROUTER_MODEL);
209 + } elseif ($provider === 'openai_compatible') {
210 + return (string) $this->settings->get('openai_compatible_model', '');
168 211 } else {
169 212 return $this->settings->get('openai_model', Settings::DEFAULT_OPENAI_MODEL);
170 213 }
171 214 }
@@ -178,9 +221,9 @@
178 221 * costliest configuration, where billed reasoning tokens (drawn from the
179 222 * same budget) are spent before any visible output (issue #286).
180 223 *
181 224 * A brief is a structured planning task, so 'low' is a provisional middle
182 - * ground — Brand Visibility uses 'minimal' for quick consumer-style answers.
225 + * ground between 'minimal' and the model's default.
183 226 * The level is filterable so a site can trade latency for more reasoning;
184 227 * returning '' opts out entirely and lets the model use its default effort.
185 228 * Only the GPT-5 family consumes this — o1/o3, gpt-4o and the non-OpenAI
186 229 * clients ignore an unrecognised option key.
@@ -212,9 +255,9 @@
212 255 *
213 256 * @return string Current provider name
214 257 */
215 258 private function get_current_provider(): string {
216 - return $this->settings->get('ai_provider', 'openai');
259 + return $this->settings->get('ai_provider', Settings::AI_PROVIDER_NONE);
217 260 }
218 261
219 262 /**
220 263 * Extract token usage from AI response
@@ -224,9 +267,9 @@
224 267 */
225 268 private function extract_token_usage(array $ai_response): int {
226 269 $provider = $this->get_current_provider();
227 270
228 - if ($provider === 'openai' || $provider === 'openrouter') {
271 + if ($provider === 'openai' || $provider === 'openrouter' || $provider === 'openai_compatible') {
229 272 // OpenAI-compatible format: response['usage']['total_tokens']
230 273 return (int) ($ai_response['usage']['total_tokens'] ?? 0);
231 274 } elseif ($provider === 'claude') {
232 275 // Claude format: response['usage']['input_tokens'] + response['usage']['output_tokens']
@@ -326,8 +369,11 @@
326 369 // max_completion_tokens internally. Temperature is intentionally
327 370 // omitted: every client defaults it to 0.7, and reasoning models
328 371 // reject it outright, so passing it here was misleading no-op.
329 372 'max_tokens' => $max_tokens,
373 + // The brief is one JSON object. Only a compatible endpoint
374 + // with JSON mode on reads this; every other client ignores it.
375 + 'json_object' => true,
330 376 ];
331 377 if ('' !== $reasoning_effort) {
332 378 $completion_options['reasoning_effort'] = $reasoning_effort;
333 379 }
@@ -339,9 +385,9 @@
339 385 // truncation) BEFORE attempting text extraction. Otherwise a
340 386 // refusal — which OpenAI returns as HTTP 200 with content=null —
341 387 // slips past every isset() branch and gets serialized into the
342 388 // brief body instead of being reported to the user.
343 - $this->guard_against_non_answer($ai_response);
389 + $this->guard_against_non_answer($ai_response, $max_tokens);
344 390
345 391 // Extract text content from AI response
346 392 $ai_text = '';
347 393
@@ -489,12 +535,13 @@
489 535 * don't catch them here they fall through to the "unexpected format" path
490 536 * (or, historically, were serialized into the brief body). All messages
491 537 * start with "The AI " so the outer catch passes them through unchanged.
492 538 *
493 - * @param mixed $ai_response Raw response from the AI client.
539 + * @param mixed $ai_response Raw response from the AI client.
540 + * @param int $requested_tokens The max_tokens this request asked for; 0 when unknown.
494 541 * @throws \Exception If the response is a refusal, policy block, or truncation.
495 542 */
496 - private function guard_against_non_answer($ai_response): void {
543 + private function guard_against_non_answer($ai_response, int $requested_tokens = 0): void {
497 544 if (!is_array($ai_response)) {
498 545 return;
499 546 }
500 547
@@ -506,18 +553,30 @@
506 553 $message = $ai_response['choices'][0]['message'];
507 554 $finish = (string) ($ai_response['choices'][0]['finish_reason'] ?? '');
508 555
509 556 if (!empty($message['refusal'])) {
510 - throw new \Exception(sprintf(
557 + throw new \Exception(esc_html(sprintf(
511 558 'The AI declined to generate this brief: %s',
512 559 (string) $message['refusal']
513 - ));
560 + )));
514 561 }
515 562 if ('content_filter' === $finish) {
516 563 throw new \Exception('The AI blocked this request under its content policy. Try a different topic or less sensitive keywords.');
517 564 }
565 + // A self-hosted server can stop short of max_tokens because the
566 + // prompt and the answer together filled its context window
567 + // (Ollama loads models at 4096 by default). A bigger output budget
568 + // cannot fix that, so say what can.
569 + $completion_tokens = (int) ($ai_response['usage']['completion_tokens'] ?? 0);
570 + if ('length' === $finish && $requested_tokens > 0 && $completion_tokens > 0 && $completion_tokens < $requested_tokens) {
571 + throw new \Exception(esc_html(sprintf(
572 + 'The AI stopped after %1$d tokens, short of the %2$d allowed, because the server ran out of context window before finishing the brief. Raise the context length on your AI server (for Ollama, set OLLAMA_CONTEXT_LENGTH to 16384 or more) and try again.',
573 + $completion_tokens,
574 + $requested_tokens
575 + )));
576 + }
518 577 if ('length' === $finish) {
519 - throw new \Exception('The AI stopped at its output token limit before finishing the brief. Try a shorter content length or fewer competitor URLs.');
578 + throw new \Exception('The AI stopped at its output token limit before finishing the brief. Try fewer competitor URLs, or a model with a larger output limit. A shorter content length will not help: it asks for a smaller budget, not a smaller answer.');
520 579 }
521 580 }
522 581
523 582 // --- Claude (Messages) ---
@@ -526,9 +585,9 @@
526 585 if ('refusal' === $stop_reason) {
527 586 throw new \Exception('The AI declined to generate this brief for this topic. Try a different topic or less sensitive keywords.');
528 587 }
529 588 if ('max_tokens' === $stop_reason) {
530 - throw new \Exception('The AI stopped at its output token limit before finishing the brief. Try a shorter content length or fewer competitor URLs.');
589 + throw new \Exception('The AI stopped at its output token limit before finishing the brief. Try fewer competitor URLs, or a model with a larger output limit. A shorter content length will not help: it asks for a smaller budget, not a smaller answer.');
531 590 }
532 591 }
533 592
534 593 // --- Gemini ---
@@ -536,12 +595,12 @@
536 595 // promptFeedback.blockReason; a candidate can also finish on SAFETY or
537 596 // PROHIBITED_CONTENT, or be truncated at MAX_TOKENS.
538 597 $block_reason = (string) ($ai_response['promptFeedback']['blockReason'] ?? '');
539 598 if ('' !== $block_reason) {
540 - throw new \Exception(sprintf(
599 + throw new \Exception(esc_html(sprintf(
541 600 'The AI blocked this request under its content policy (%s). Try a different topic or less sensitive keywords.',
542 601 $block_reason
543 - ));
602 + )));
544 603 }
545 604 $gemini_finish = (string) ($ai_response['candidates'][0]['finishReason'] ?? '');
546 605 if (in_array($gemini_finish, ['SAFETY', 'PROHIBITED_CONTENT'], true)) {
547 606 throw new \Exception('The AI blocked this request under its content policy. Try a different topic or less sensitive keywords.');
@@ -546,9 +605,9 @@
546 605 if (in_array($gemini_finish, ['SAFETY', 'PROHIBITED_CONTENT'], true)) {
547 606 throw new \Exception('The AI blocked this request under its content policy. Try a different topic or less sensitive keywords.');
548 607 }
549 608 if ('MAX_TOKENS' === $gemini_finish) {
550 - throw new \Exception('The AI stopped at its output token limit before finishing the brief. Try a shorter content length or fewer competitor URLs.');
609 + throw new \Exception('The AI stopped at its output token limit before finishing the brief. Try fewer competitor URLs, or a model with a larger output limit. A shorter content length will not help: it asks for a smaller budget, not a smaller answer.');
551 610 }
552 611 }
553 612
554 613 /**
@@ -642,9 +701,9 @@
642 701 'title' => $json_data['title_suggestions'] ?? [],
643 702 'meta_description' => $json_data['meta_descriptions'][0] ?? '',
644 703 'meta_descriptions' => $json_data['meta_descriptions'] ?? [],
645 704 'url_slugs' => $json_data['url_slugs'] ?? [],
646 - 'outline' => $json_data['outline'] ?? [],
705 + 'outline' => self::strip_outline_level_labels($json_data['outline'] ?? []),
647 706 'seo_recommendations' => [
648 707 'title_suggestions' => $json_data['title_suggestions'] ?? [],
649 708 'meta_description' => $json_data['meta_descriptions'][0] ?? '',
650 709 'meta_descriptions' => $json_data['meta_descriptions'] ?? [],
@@ -670,9 +729,9 @@
670 729 ],
671 730 'competitor_gaps' => $json_data['competitor_analysis']['content_gaps'] ?? [],
672 731 'call_to_actions' => $json_data['call_to_actions'] ?? [],
673 732 'writing_guidelines' => $json_data['writing_guidelines'] ?? [],
674 - 'content_body' => $json_data['content_body'] ?? '',
733 + 'content_body' => self::strip_heading_level_labels((string) ($json_data['content_body'] ?? '')),
675 734 'estimated_word_count' => $this->get_word_count_estimate($original_params['content_length'] ?? 'medium'),
676 735 'raw_response' => '', // Will be retrieved from ai_usage table
677 736 'generation_params' => $original_params,
678 737 'parsing_status' => 'success',
@@ -680,8 +739,118 @@
680 739 ];
681 740 }
682 741
683 742 /**
743 + * Remove a leading level label from a heading string.
744 + *
745 + * The prompt's own JSON example labelled outline headings with their level
746 + * (`"heading": "H1: Main Title"` next to a separate `"level": 1`), so the
747 + * model often carried the convention into the drafted article and Pro's
748 + * "Insert into post" wrote `<h2>H2: Real Heading</h2>` into published
749 + * content. The prompt no longer does that, but a prompt change never fully
750 + * binds a model — so the label is stripped here too (#410).
751 + *
752 + * Covers the label forms a model actually emits: `H2:`, `h3:`, `H2 -`,
753 + * `H4.`, `H2)` and the en/em dash variants, optionally wrapped in markdown
754 + * emphasis (`**H2:**`). The delimiter is anchored directly after the digit
755 + * so `H10:` — a plausible heading in a numbered list — is left alone, and
756 + * only a leading label is matched so body copy that mentions a level
757 + * survives. Trailing emphasis is consumed only when the same marker opened
758 + * the label, so `H2: *emphasised start*` keeps its asterisks.
759 + *
760 + * @since 2.0.1
761 + *
762 + * @param string $heading Heading text.
763 + * @return string Heading without its level prefix.
764 + */
765 + public static function strip_level_label(string $heading): string {
766 + // En dash and em dash as raw UTF-8 bytes, so the pattern needs no /u
767 + // modifier and cannot blank a heading that is not valid UTF-8.
768 + $delimiter = '(?:[:.)\-]|\xe2\x80\x93|\xe2\x80\x94)';
769 + $emphasis = '(\*{1,3}|_{1,3})';
770 +
771 + $pattern = '/^\s*(?:'
772 + . $emphasis . '\s*[Hh][1-6]\s*' . $delimiter . '\s*\1'
773 + . '|[Hh][1-6]\s*' . $delimiter
774 + . ')\s*/';
775 +
776 + return (string) preg_replace($pattern, '', $heading);
777 + }
778 +
779 + /**
780 + * Strip a level label from a heading's inner HTML.
781 + *
782 + * A model drafting publish-ready HTML often wraps the heading text in an
783 + * inline tag (`<h2><strong>H2: Real Heading</strong></h2>`). That pushes a
784 + * `<` in front of the label, so the leading run of inline opening tags is
785 + * set aside and re-attached around the cleaned text.
786 + *
787 + * @since 2.0.1
788 + *
789 + * @param string $inner Heading inner HTML.
790 + * @return string Inner HTML without the level prefix.
791 + */
792 + private static function strip_inner_level_label(string $inner): string {
793 + $prefix = '';
794 +
795 + if (preg_match('/^(\s*(?:<(?:strong|em|b|i|span|mark|code|u)\b[^>]*>\s*)+)(.*)$/is', $inner, $parts)) {
796 + $prefix = $parts[1];
797 + $inner = $parts[2];
798 + }
799 +
800 + return $prefix . self::strip_level_label($inner);
801 + }
802 +
803 + /**
804 + * Strip level labels from every heading in an outline.
805 + *
806 + * @since 2.0.1
807 + *
808 + * @param mixed $outline Outline as returned by the model.
809 + * @return array Outline with clean headings.
810 + */
811 + public static function strip_outline_level_labels($outline): array {
812 + if (!is_array($outline)) {
813 + return [];
814 + }
815 +
816 + foreach ($outline as $index => $section) {
817 + if (is_array($section) && isset($section['heading']) && is_string($section['heading'])) {
818 + $outline[$index]['heading'] = self::strip_level_label($section['heading']);
819 + } elseif (is_string($section)) {
820 + $outline[$index] = self::strip_level_label($section);
821 + }
822 + }
823 +
824 + return $outline;
825 + }
826 +
827 + /**
828 + * Strip level labels from the heading text inside drafted HTML.
829 + *
830 + * This is the path that reaches published post content, so it is the one
831 + * that matters most. Only the text directly inside an <h1>-<h6> is touched.
832 + *
833 + * @since 2.0.1
834 + *
835 + * @param string $html Drafted article body.
836 + * @return string Body with clean headings.
837 + */
838 + public static function strip_heading_level_labels(string $html): string {
839 + if ('' === $html || false === stripos($html, '<h')) {
840 + return $html;
841 + }
842 +
843 + return (string) preg_replace_callback(
844 + '/(<h([1-6])\b[^>]*>)(.*?)(<\/h\2>)/is',
845 + static function (array $parts): string {
846 + return $parts[1] . self::strip_inner_level_label($parts[3]) . $parts[4];
847 + },
848 + $html
849 + );
850 + }
851 +
852 + /**
684 853 * Create error response when JSON parsing fails
685 854 *
686 855 * @param string $ai_response Raw AI response
687 856 * @param array $original_params Original generation parameters
@@ -1144,8 +1313,20 @@
1144 1313 if (false === $result) {
1145 1314 throw new \Exception('Failed to save content brief to database.');
1146 1315 }
1147 1316
1317 + /**
1318 + * Fires after a content brief is persisted.
1319 + *
1320 + * Analytics listens to drop its cached overview so the brief counts
1321 + * on the Usages page are not stale for a TTL.
1322 + *
1323 + * @since 2.2.1
1324 + *
1325 + * @param int $brief_id Row id of the stored brief.
1326 + */
1327 + do_action('thinkrank_content_brief_created', (int) $wpdb->insert_id);
1328 +
1148 1329 return $wpdb->insert_id;
1149 1330 }
1150 1331
1151 1332 /**
@@ -1187,8 +1368,54 @@
1187 1368 return is_string($rec) ? $rec : 'Image recommendation';
1188 1369 }, $brief_data['visual_content']['image_recommendations']);
1189 1370 }
1190 1371
1372 + return $this->sanitize_brief_output($brief_data);
1373 + }
1374 +
1375 + /**
1376 + * Strip untrusted markup out of brief fields before they leave the server.
1377 + *
1378 + * Brief content crosses a trust boundary: it is assembled by an external AI
1379 + * provider from prompts that can include text fetched from competitor URLs.
1380 + * It was previously copied out of the decoded JSON verbatim and rendered in
1381 + * the admin SPA through dangerouslySetInnerHTML, so a malicious or
1382 + * prompt-injected response could execute script in the admin origin (#365).
1383 + *
1384 + * Runs on the read path as well as generation, so briefs stored before this
1385 + * fix are sanitized when they are loaded.
1386 + *
1387 + * @since 1.32.0
1388 + *
1389 + * @param array $brief_data Brief data to sanitize.
1390 + * @return array Sanitized brief data.
1391 + */
1392 + private function sanitize_brief_output(array $brief_data): array {
1393 + foreach ($brief_data as $key => $value) {
1394 + // The raw provider response is debug output shown as plain text, and
1395 + // the generation params are our own values — leave both intact.
1396 + if ('raw_response' === $key || 'generation_params' === $key) {
1397 + continue;
1398 + }
1399 +
1400 + if ('content_body' === $key && is_string($value)) {
1401 + // Deliberately HTML: it is the drafted article and is rendered as
1402 + // markup. wp_kses_post() keeps normal post formatting while
1403 + // dropping script/style/iframe, event-handler attributes and
1404 + // javascript: URLs.
1405 + $brief_data[$key] = wp_kses_post($value);
1406 + continue;
1407 + }
1408 +
1409 + if (is_array($value)) {
1410 + $brief_data[$key] = $this->sanitize_brief_output($value);
1411 + } elseif (is_string($value)) {
1412 + // Every other field is plain text (headings, keywords, guidance).
1413 + // Markdown emphasis markers are preserved; HTML tags are not.
1414 + $brief_data[$key] = wp_strip_all_tags($value);
1415 + }
1416 + }
1417 +
1191 1418 return $brief_data;
1192 1419 }
1193 1420
1194 1421 /**
@@ -1821,9 +2048,9 @@
1821 2048 [
1822 2049 'user_id' => $user_id,
1823 2050 'action' => $action,
1824 2051 'tokens_used' => $tokens_used,
1825 - 'provider' => $this->settings->get('ai_provider', 'openai'),
2052 + 'provider' => $this->settings->get('ai_provider', Settings::AI_PROVIDER_NONE),
1826 2053 'post_id' => $post_id,
1827 2054 'metadata' => !empty($metadata) ? wp_json_encode($metadata) : null,
1828 2055 'created_at' => current_time('mysql'),
1829 2056 ],
@@ -1828,8 +2055,17 @@
1828 2055 'created_at' => current_time('mysql'),
1829 2056 ],
1830 2057 ['%d', '%s', '%d', '%s', '%d', '%s', '%s']
1831 2058 );
2059 +
2060 + /**
2061 + * Fires after an AI usage row is recorded.
2062 + *
2063 + * @since 2.2.1
2064 + *
2065 + * @param int $user_id User the usage was recorded against.
2066 + */
2067 + do_action('thinkrank_ai_usage_logged', $user_id);
1832 2068
1833 2069 return $wpdb->insert_id;
1834 2070 }
1835 2071