PluginProbe
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO / 2.10.0
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO v2.10.0
2.10.0 2.9.0 2.8.0 2.7.0 2.6.0 2.5.0 2.4.0 2.3.0 2.2.0 2.1.1 2.1.0 2.0.2 2.0.1 2.0.0 1.32.0 1.31.0 1.30.0 1.29.0 1.28.0 1.27.0 1.26.0 1.25.0 trunk 1.0.0 1.0.1 All 51 releases
thinkrank / includes / ai / class-content-brief-generator.php

class-content-brief-generator.php in ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO 2.10.0, at includes/ai/class-content-brief-generator.php

2,144 lines 87.1 KB
No matching file
Up and down to move Enter to open Esc to close
Raw Download Zip
1 <?php
2 /**
3 * Content Brief Generator
4 *
5 * Handles AI-powered content brief generation with competitor analysis
6 *
7 * @package ThinkRank
8 * @subpackage AI
9 * @since 1.0.0
10 */
11
12 namespace ThinkRank\AI;
13
14 use ThinkRank\Core\Settings;
15 use ThinkRank\AI\OpenAI_Client;
16 use ThinkRank\AI\Claude_Client;
17 use ThinkRank\AI\OpenRouter_Client;
18
19 // Prevent direct access
20 if (!defined('ABSPATH')) {
21 exit;
22 }
23
24 /**
25 * Content Brief Generator class
26 */
27 class Content_Brief_Generator {
28
29 /**
30 * AI request timeout (seconds) for brief generation.
31 *
32 * Content briefs request a very large completion (~0.9–0.95 of the model's
33 * max tokens) from reasoning models, which routinely take 40–90s — far
34 * longer than the AI clients' 30s default. Without this the HTTP call is
35 * aborted with cURL error 28 and the brief never completes. PHP execution
36 * time is covered by each client's raise_request_time_limit() (timeout+45).
37 */
38 private const AI_REQUEST_TIMEOUT = 120;
39
40 /**
41 * AI request timeout (seconds) for OpenAI specifically.
42 *
43 * OpenAI's reasoning models (GPT-5 / o-series) burn reasoning tokens before
44 * emitting any content, and briefs request ~95% of the model's completion
45 * limit — so the call frequently runs past the 120s the other providers
46 * need. PHP execution time is covered by raise_request_time_limit()
47 * (timeout+45); the web server's own read timeout still caps the maximum.
48 */
49 private const OPENAI_REQUEST_TIMEOUT = 300;
50
51 /**
52 * Minimum output-token budget for a content brief.
53 *
54 * A shorter length tier must never starve the structured JSON — plus any
55 * reasoning/thinking tokens, which are drawn from the same budget — to the
56 * point of truncating mid-response (the failure #165 fixed on Gemini). This
57 * floor is only a safety net for small-ceiling models; it never exceeds the
58 * model-aware base budget. See scale_tokens_for_length().
59 */
60 private const MIN_BRIEF_TOKENS = 2000;
61
62 /**
63 * Content-length → budget multipliers, applied to the model-aware base.
64 *
65 * NOTE: provisional starting points (issue #287). They make Short/Medium/
66 * Long request measurably different budgets, but the exact figures should
67 * be validated against recorded completion-token usage for a real brief on
68 * each provider before being treated as final. Unknown lengths fall back to
69 * the 'medium' tier (see scale_tokens_for_length()).
70 */
71 private const LENGTH_TOKEN_MULTIPLIERS = [
72 'short' => 0.6,
73 'medium' => 0.8,
74 'long' => 1.0,
75 ];
76
77 /**
78 * Settings instance
79 *
80 * @var Settings
81 */
82 private Settings $settings;
83
84 /**
85 * AI client instance
86 *
87 * Null when the generator was built for storage-only work.
88 *
89 * @var OpenAI_Client|Claude_Client|null
90 */
91 private $ai_client;
92
93 /**
94 * Constructor
95 *
96 * @param Settings|null $settings Settings instance
97 * @param OpenAI_Client|Claude_Client|null $ai_client AI client instance
98 * @param bool $require_ai_client Whether a provider client is required. Pass
99 * false for storage-only use (list/export/
100 * delete), which never calls a provider.
101 */
102 public function __construct(?Settings $settings = null, $ai_client = null, bool $require_ai_client = true) {
103 $this->settings = $settings ?? Settings::instance();
104
105 if ($ai_client) {
106 $this->ai_client = $ai_client;
107
108 return;
109 }
110
111 // Read-only callers (listing, exporting and deleting saved briefs) only
112 // touch the database and never reach a provider. Constructing a client
113 // for them turns "no API key configured" — the default state of a fresh
114 // install — into a hard failure, so let them opt out.
115 if (!$require_ai_client) {
116 return;
117 }
118
119 // Fallback to creating own client for backward compatibility
120 $this->init_ai_client();
121 }
122
123 /**
124 * Initialize AI client based on available API keys
125 *
126 * @return void
127 *
128 * @throws \Exception On failure.
129 */
130 private function init_ai_client(): void {
131 $provider = $this->settings->get('ai_provider', Settings::AI_PROVIDER_NONE);
132
133 if ($provider === 'openai') {
134 $api_key = $this->settings->get('openai_api_key');
135 if ($api_key) {
136 $model = $this->settings->get('openai_model', Settings::DEFAULT_OPENAI_MODEL);
137 $this->ai_client = new OpenAI_Client($api_key, $model, self::OPENAI_REQUEST_TIMEOUT);
138 }
139 } elseif ($provider === 'claude') {
140 $api_key = $this->settings->get('claude_api_key');
141 if ($api_key) {
142 $model = $this->settings->get('claude_model', Settings::DEFAULT_CLAUDE_MODEL);
143 $this->ai_client = new Claude_Client($api_key, $model, self::AI_REQUEST_TIMEOUT);
144 }
145 } elseif ($provider === 'gemini') {
146 $api_key = $this->settings->get('gemini_api_key');
147 if ($api_key) {
148 $model = $this->settings->get('gemini_model', Settings::DEFAULT_GEMINI_MODEL);
149 $this->ai_client = new Gemini_Client($api_key, $model, self::AI_REQUEST_TIMEOUT);
150 }
151 } elseif ($provider === 'openrouter') {
152 $api_key = $this->settings->get('openrouter_api_key');
153 if ($api_key) {
154 $model = $this->settings->get('openrouter_model', Settings::DEFAULT_OPENROUTER_MODEL);
155 $this->ai_client = new OpenRouter_Client($api_key, $model, self::AI_REQUEST_TIMEOUT);
156 }
157 } elseif ($provider === 'openai_compatible') {
158 // Same client as OpenAI, different host — and the key is optional,
159 // so the URL and model id are what gate it (#721). The user's own
160 // timeout applies: a local model writing a brief on CPU is slow,
161 // and the setting exists for exactly that.
162 $base_url = (string) $this->settings->get('openai_compatible_base_url', '');
163 $model = trim((string) $this->settings->get('openai_compatible_model', ''));
164 if ('' !== $base_url && '' !== $model) {
165 $this->ai_client = new OpenAI_Client(
166 (string) $this->settings->get('openai_compatible_api_key', ''),
167 $model,
168 (int) $this->settings->get('openai_compatible_timeout', Settings::DEFAULT_OPENAI_COMPATIBLE_TIMEOUT),
169 $base_url
170 );
171 $this->ai_client->set_json_mode((bool) $this->settings->get('openai_compatible_json_mode', false));
172 }
173 }
174
175 if (!$this->ai_client) {
176 if ('openai_compatible' === $provider) {
177 throw new \Exception('Please set the base URL and model id for your OpenAI-compatible endpoint in ThinkRank settings.');
178 }
179
180 throw new \Exception('Please configure your AI provider API key in ThinkRank settings.');
181 }
182 }
183
184 /**
185 * Get current AI model being used
186 *
187 * @return string Current model name
188 */
189 private function get_current_model(): string {
190 // Try to get model from the actual AI client if available
191 if ($this->ai_client && method_exists($this->ai_client, 'get_model')) {
192 return $this->ai_client->get_model();
193 }
194
195 // Fallback to settings
196 $provider = $this->settings->get('ai_provider', Settings::AI_PROVIDER_NONE);
197 if (Settings::AI_PROVIDER_NONE === $provider) {
198 // No provider chosen, so there is no model to name. Reporting the
199 // OpenAI default here would attribute output to a provider the site
200 // never selected (#572).
201 return '';
202 }
203 if ($provider === 'claude') {
204 return $this->settings->get('claude_model', Settings::DEFAULT_CLAUDE_MODEL);
205 } elseif ($provider === 'gemini') {
206 return $this->settings->get('gemini_model', Settings::DEFAULT_GEMINI_MODEL);
207 } elseif ($provider === 'openrouter') {
208 return $this->settings->get('openrouter_model', Settings::DEFAULT_OPENROUTER_MODEL);
209 } elseif ($provider === 'openai_compatible') {
210 return (string) $this->settings->get('openai_compatible_model', '');
211 } else {
212 return $this->settings->get('openai_model', Settings::DEFAULT_OPENAI_MODEL);
213 }
214 }
215
216 /**
217 * Resolve the reasoning-effort level for a content-brief request.
218 *
219 * Without an explicit level, GPT-5 models run at their default (maximum)
220 * reasoning effort against a ~95% completion budget — the slowest and
221 * costliest configuration, where billed reasoning tokens (drawn from the
222 * same budget) are spent before any visible output (issue #286).
223 *
224 * A brief is a structured planning task, so 'low' is a provisional middle
225 * ground between 'minimal' and the model's default.
226 * The level is filterable so a site can trade latency for more reasoning;
227 * returning '' opts out entirely and lets the model use its default effort.
228 * Only the GPT-5 family consumes this — o1/o3, gpt-4o and the non-OpenAI
229 * clients ignore an unrecognised option key.
230 *
231 * @param string $model The resolved model ID (passed to the filter).
232 * @param array $params The brief generation parameters (passed to the filter).
233 * @return string One of 'minimal' | 'low' | 'medium' | 'high', or '' to opt out.
234 */
235 private function resolve_reasoning_effort(string $model, array $params): string {
236 /**
237 * Filter the reasoning-effort level used for content-brief generation.
238 *
239 * @param string $effort The default level ('low'). Return '' to opt out.
240 * @param string $model The resolved model ID for this request.
241 * @param array $params The brief generation parameters.
242 */
243 $effort = (string) apply_filters('thinkrank_content_brief_reasoning_effort', 'low', $model, $params);
244
245 // Only values OpenAI accepts may reach the request body ('' opts out).
246 // An unrecognised filter return (e.g. 'turbo') would otherwise be sent
247 // verbatim and fail the whole brief with a 400, so degrade to the
248 // documented default instead.
249 $allowed = ['', 'minimal', 'low', 'medium', 'high'];
250 return in_array($effort, $allowed, true) ? $effort : 'low';
251 }
252
253 /**
254 * Get current AI provider
255 *
256 * @return string Current provider name
257 */
258 private function get_current_provider(): string {
259 return $this->settings->get('ai_provider', Settings::AI_PROVIDER_NONE);
260 }
261
262 /**
263 * Extract token usage from AI response
264 *
265 * @param array $ai_response AI response data
266 * @return int Number of tokens used
267 */
268 private function extract_token_usage(array $ai_response): int {
269 $provider = $this->get_current_provider();
270
271 if ($provider === 'openai' || $provider === 'openrouter' || $provider === 'openai_compatible') {
272 // OpenAI-compatible format: response['usage']['total_tokens']
273 return (int) ($ai_response['usage']['total_tokens'] ?? 0);
274 } elseif ($provider === 'claude') {
275 // Claude format: response['usage']['input_tokens'] + response['usage']['output_tokens']
276 $input_tokens = (int) ($ai_response['usage']['input_tokens'] ?? 0);
277 $output_tokens = (int) ($ai_response['usage']['output_tokens'] ?? 0);
278 return $input_tokens + $output_tokens;
279 } elseif ($provider === 'gemini') {
280 // Gemini format: response['usageMetadata']['totalTokenCount']
281 return (int) ($ai_response['usageMetadata']['totalTokenCount'] ?? 0);
282 }
283
284 // Fallback: return 0 if provider not recognized or no usage data
285 return 0;
286 }
287
288 /**
289 * Extract actual model used from AI response
290 *
291 * @param array $ai_response AI response data
292 * @return string|null Actual model used or null if not found
293 */
294 private function extract_model_from_response(array $ai_response): ?string {
295 // OpenAI format: response['model']
296 if (isset($ai_response['model'])) {
297 return $ai_response['model'];
298 }
299
300 // Claude format: response['model']
301 if (isset($ai_response['model'])) {
302 return $ai_response['model'];
303 }
304
305 // Gemini doesn't include model in response, fallback to client model
306 return null;
307 }
308
309 /**
310 * Generate content brief
311 *
312 * @param array $params Brief generation parameters
313 * @return array Generated brief data
314 * @throws \Exception If generation fails
315 */
316 public function generate_brief(array $params): array {
317 // Validate required parameters
318 $this->validate_brief_params($params);
319
320 // Extract parameters
321 $target_keywords = $params['target_keywords'] ?? [];
322 $content_type = $params['content_type'] ?? 'blog_post';
323 $target_audience = $params['target_audience'] ?? 'general';
324 $content_length = $params['content_length'] ?? 'medium';
325 $tone = $params['tone'] ?? 'professional';
326 $competitor_urls = $params['competitor_urls'] ?? [];
327 $additional_context = $params['additional_context'] ?? '';
328 // Write the brief in the site (or related post's) language rather than
329 // defaulting to English on non-English sites (issue #234).
330 $language = \ThinkRank\AI\Language_Resolver::resolve((int) ($params['post_id'] ?? 0));
331
332 // Analyze competitor URLs if provided
333 $competitor_analysis = '';
334 if (!empty($competitor_urls)) {
335 $competitor_analysis = $this->analyze_competitor_urls($competitor_urls);
336 }
337
338 // Build AI prompt using shared Prompt Builder
339 $prompt_builder = $this->get_prompt_builder();
340 $prompt = $prompt_builder->build_content_brief_prompt(
341 $target_keywords,
342 $content_type,
343 $target_audience,
344 $content_length,
345 $tone,
346 $competitor_analysis,
347 $additional_context,
348 $this->get_current_provider(),
349 $language
350 );
351
352 try {
353 // Get the model-aware budget for a comprehensive brief, then scale
354 // it to the requested content length so Short/Medium/Long actually
355 // request different budgets (issue #287). Every client (OpenAI,
356 // Claude, Gemini, OpenRouter) implements get_recommended_tokens(),
357 // so there is no model-blind fallback.
358 $base_tokens = (int) $this->ai_client->get_recommended_tokens('content_brief');
359 $max_tokens = $this->scale_tokens_for_length($base_tokens, $content_length);
360
361 // Bound hidden reasoning on the GPT-5 family (issue #286). See
362 // resolve_reasoning_effort(). Only the GPT-5 family reads this;
363 // o1/o3, gpt-4o and the non-OpenAI clients ignore the option, and
364 // an empty string opts out (model default effort).
365 $reasoning_effort = $this->resolve_reasoning_effort($this->get_current_model(), $params);
366
367 $completion_options = [
368 // For GPT‑5 family the client translates max_tokens to
369 // max_completion_tokens internally. Temperature is intentionally
370 // omitted: every client defaults it to 0.7, and reasoning models
371 // reject it outright, so passing it here was misleading no-op.
372 'max_tokens' => $max_tokens,
373 // The brief is one JSON object. Only a compatible endpoint
374 // with JSON mode on reads this; every other client ignores it.
375 'json_object' => true,
376 ];
377 if ('' !== $reasoning_effort) {
378 $completion_options['reasoning_effort'] = $reasoning_effort;
379 }
380
381 // Generate brief using AI
382 $ai_response = $this->ai_client->generate_completion($prompt, $completion_options);
383
384 // Detect a provider-side non-answer (refusal, policy block, or
385 // truncation) BEFORE attempting text extraction. Otherwise a
386 // refusal — which OpenAI returns as HTTP 200 with content=null —
387 // slips past every isset() branch and gets serialized into the
388 // brief body instead of being reported to the user.
389 $this->guard_against_non_answer($ai_response, $max_tokens);
390
391 // Extract text content from AI response
392 $ai_text = '';
393
394 // Handle OpenAI response format
395 if (isset($ai_response['choices'][0]['message']['content'])) {
396 $content_field = $ai_response['choices'][0]['message']['content'];
397 if (is_string($content_field)) {
398 $ai_text = $content_field;
399 } elseif (is_array($content_field)) {
400 // Concatenate text parts from array-based content (Chat Completions multimodal)
401 $parts = array_map(function($part) {
402 if (is_array($part)) {
403 return $part['text'] ?? '';
404 }
405 return is_string($part) ? $part : '';
406 }, $content_field);
407 $ai_text = trim(implode("\n", array_filter($parts)));
408 }
409 }
410 // Handle Claude response format
411 elseif (isset($ai_response['content'][0]['text'])) {
412 $ai_text = $ai_response['content'][0]['text'];
413 }
414 // Handle Gemini response format
415 elseif (isset($ai_response['candidates'][0]['content']['parts'][0]['text'])) {
416 $ai_text = $ai_response['candidates'][0]['content']['parts'][0]['text'];
417 }
418 // Handle direct content field
419 elseif (isset($ai_response['content']) && is_string($ai_response['content'])) {
420 $ai_text = $ai_response['content'];
421 }
422 // Handle direct string response
423 elseif (is_string($ai_response)) {
424 $ai_text = $ai_response;
425 }
426 // No known provider shape matched and guard_against_non_answer()
427 // found nothing it recognised. Never serialize the raw envelope
428 // into the brief body — that turns a clear failure into a saved,
429 // meaningless brief. Log the shape for diagnostics and fail.
430 else {
431 if (defined('WP_DEBUG') && WP_DEBUG) {
432 $shape = is_array($ai_response) ? implode(', ', array_keys($ai_response)) : gettype($ai_response);
433 // phpcs:ignore WordPress.PHP.DevelopmentFunctions.error_log_error_log -- Debug logging only when WP_DEBUG is enabled.
434 error_log('[ThinkRank] Content brief: unrecognised AI response shape. Top-level keys: ' . $shape);
435 }
436 throw new \Exception('The AI returned a response in an unexpected format. Please try again.');
437 }
438
439 // Ensure we have actual text content
440 if (empty(trim($ai_text))) {
441 throw new \Exception('AI response was empty or contained no text content.');
442 }
443
444 // Extract token usage for analytics tracking
445 $tokens_used = $this->extract_token_usage($ai_response);
446
447 // Parse and structure the response
448 $brief_data = $this->parse_ai_response($ai_text, $params);
449
450 // Extract actual model from response before using it
451 $actual_model = $this->extract_model_from_response($ai_response);
452
453 // Add generation metadata (use actual model from response if available)
454 $brief_data['generation_meta'] = [
455 'provider' => $this->get_current_provider(),
456 'model' => $actual_model ?: $this->get_current_model(),
457 'generated_at' => current_time('mysql'),
458 'version' => '1.0'
459 ];
460
461 // Save brief to database
462 $brief_id = $this->save_brief($brief_data);
463 $brief_data['id'] = $brief_id;
464
465 // Log AI usage for analytics tracking (including raw response and actual model used)
466 $usage_id = $this->log_ai_usage(get_current_user_id(), 'Content Brief', $tokens_used, $brief_id, $ai_text, $actual_model);
467
468 // Set raw response for immediate display
469 $brief_data['raw_response'] = $ai_text;
470
471 // Apply normalization for React compatibility
472 $brief_data = $this->normalize_brief_data($brief_data);
473
474 return $brief_data;
475
476 } catch (\Exception $e) {
477 // Provide more specific error messages
478 $error_message = $e->getMessage();
479
480 // Messages we authored for the user (refusals, policy blocks,
481 // token-limit truncation, unexpected shape) all start with "The AI "
482 // and are already actionable. Pass them through verbatim instead of
483 // flattening them via the substring matching below — e.g. so a
484 // refusal is not rewritten into generic "empty content" advice.
485 if (strpos($error_message, 'The AI ') === 0) {
486 throw new \Exception(esc_html($error_message));
487 }
488
489 if (strpos($error_message, 'API key') !== false) {
490 throw new \Exception('API key configuration error. Please check your AI provider settings.');
491 } elseif (strpos($error_message, 'Invalid AI response format') !== false) {
492 throw new \Exception('AI service returned an unexpected response format. Please try again.');
493 } elseif (strpos($error_message, 'empty') !== false) {
494 throw new \Exception('AI service returned empty content. Please try again with different parameters.');
495 } else {
496 throw new \Exception('Failed to generate content brief: ' . esc_html($error_message));
497 }
498 }
499 }
500
501 /**
502 * Scale the model-aware brief budget to the requested content length.
503 *
504 * get_recommended_tokens('content_brief') returns the budget for a full,
505 * comprehensive (Long) brief, already capped at the model's completion
506 * ceiling. Shorter tiers request proportionally less so that choosing Short
507 * is genuinely faster and cheaper (issue #287), while every tier stays at or
508 * below the base and at or above MIN_BRIEF_TOKENS so it cannot truncate.
509 *
510 * @param int $base_tokens Model-aware budget for a comprehensive brief.
511 * @param string $content_length One of 'short' | 'medium' | 'long'.
512 * @return int Scaled max_tokens, clamped to [floor, base_tokens].
513 */
514 private function scale_tokens_for_length(int $base_tokens, string $content_length): int {
515 // Unknown/missing length falls back to the medium tier — never to 0 or
516 // to the raw ceiling.
517 $multiplier = self::LENGTH_TOKEN_MULTIPLIERS[$content_length]
518 ?? self::LENGTH_TOKEN_MULTIPLIERS['medium'];
519
520 $scaled = (int) round($base_tokens * $multiplier);
521
522 // The floor can never exceed the base itself, so a model with a tiny
523 // ceiling still yields a sane, in-range value.
524 $floor = (int) min($base_tokens, self::MIN_BRIEF_TOKENS);
525
526 return max($floor, min($scaled, $base_tokens));
527 }
528
529 /**
530 * Detect a provider-side non-answer and fail with the real reason.
531 *
532 * A refusal, content-policy block, or token-limit truncation is not a
533 * usable brief. Each provider signals these differently, and none of the
534 * signals set the content field the extraction chain looks for — so if we
535 * don't catch them here they fall through to the "unexpected format" path
536 * (or, historically, were serialized into the brief body). All messages
537 * start with "The AI " so the outer catch passes them through unchanged.
538 *
539 * @param mixed $ai_response Raw response from the AI client.
540 * @param int $requested_tokens The max_tokens this request asked for; 0 when unknown.
541 * @throws \Exception If the response is a refusal, policy block, or truncation.
542 */
543 private function guard_against_non_answer($ai_response, int $requested_tokens = 0): void {
544 if (!is_array($ai_response)) {
545 return;
546 }
547
548 // --- OpenAI (Chat Completions) ---
549 // A structured refusal is HTTP 200 with message.content=null and the
550 // stated reason carried in message.refusal. finish_reason distinguishes
551 // a policy block from a truncated completion.
552 if (isset($ai_response['choices'][0]['message'])) {
553 $message = $ai_response['choices'][0]['message'];
554 $finish = (string) ($ai_response['choices'][0]['finish_reason'] ?? '');
555
556 if (!empty($message['refusal'])) {
557 throw new \Exception(esc_html(sprintf(
558 'The AI declined to generate this brief: %s',
559 (string) $message['refusal']
560 )));
561 }
562 if ('content_filter' === $finish) {
563 throw new \Exception('The AI blocked this request under its content policy. Try a different topic or less sensitive keywords.');
564 }
565 // A self-hosted server can stop short of max_tokens because the
566 // prompt and the answer together filled its context window
567 // (Ollama loads models at 4096 by default). A bigger output budget
568 // cannot fix that, so say what can.
569 $completion_tokens = (int) ($ai_response['usage']['completion_tokens'] ?? 0);
570 if ('length' === $finish && $requested_tokens > 0 && $completion_tokens > 0 && $completion_tokens < $requested_tokens) {
571 throw new \Exception(esc_html(sprintf(
572 'The AI stopped after %1$d tokens, short of the %2$d allowed, because the server ran out of context window before finishing the brief. Raise the context length on your AI server (for Ollama, set OLLAMA_CONTEXT_LENGTH to 16384 or more) and try again.',
573 $completion_tokens,
574 $requested_tokens
575 )));
576 }
577 if ('length' === $finish) {
578 throw new \Exception('The AI stopped at its output token limit before finishing the brief. Try fewer competitor URLs, or a model with a larger output limit. A shorter content length will not help: it asks for a smaller budget, not a smaller answer.');
579 }
580 }
581
582 // --- Claude (Messages) ---
583 if (isset($ai_response['stop_reason'])) {
584 $stop_reason = (string) $ai_response['stop_reason'];
585 if ('refusal' === $stop_reason) {
586 throw new \Exception('The AI declined to generate this brief for this topic. Try a different topic or less sensitive keywords.');
587 }
588 if ('max_tokens' === $stop_reason) {
589 throw new \Exception('The AI stopped at its output token limit before finishing the brief. Try fewer competitor URLs, or a model with a larger output limit. A shorter content length will not help: it asks for a smaller budget, not a smaller answer.');
590 }
591 }
592
593 // --- Gemini ---
594 // A prompt rejected outright returns no candidate at all, only
595 // promptFeedback.blockReason; a candidate can also finish on SAFETY or
596 // PROHIBITED_CONTENT, or be truncated at MAX_TOKENS.
597 $block_reason = (string) ($ai_response['promptFeedback']['blockReason'] ?? '');
598 if ('' !== $block_reason) {
599 throw new \Exception(esc_html(sprintf(
600 'The AI blocked this request under its content policy (%s). Try a different topic or less sensitive keywords.',
601 $block_reason
602 )));
603 }
604 $gemini_finish = (string) ($ai_response['candidates'][0]['finishReason'] ?? '');
605 if (in_array($gemini_finish, ['SAFETY', 'PROHIBITED_CONTENT'], true)) {
606 throw new \Exception('The AI blocked this request under its content policy. Try a different topic or less sensitive keywords.');
607 }
608 if ('MAX_TOKENS' === $gemini_finish) {
609 throw new \Exception('The AI stopped at its output token limit before finishing the brief. Try fewer competitor URLs, or a model with a larger output limit. A shorter content length will not help: it asks for a smaller budget, not a smaller answer.');
610 }
611 }
612
613 /**
614 * Validate brief generation parameters
615 *
616 * @param array $params Parameters to validate
617 * @throws \Exception If validation fails
618 */
619 private function validate_brief_params(array $params): void {
620 if (empty($params['target_keywords']) || !is_array($params['target_keywords'])) {
621 throw new \Exception('Target keywords are required and must be an array.');
622 }
623
624 $valid_content_types = ['blog_post', 'product_page', 'landing_page', 'tutorial'];
625 if (!empty($params['content_type']) && !in_array($params['content_type'], $valid_content_types, true)) {
626 throw new \Exception('Invalid content type specified.');
627 }
628
629 $valid_lengths = ['short', 'medium', 'long'];
630 if (!empty($params['content_length']) && !in_array($params['content_length'], $valid_lengths, true)) {
631 throw new \Exception('Invalid content length specified.');
632 }
633
634 $valid_tones = ['professional', 'casual', 'technical', 'friendly'];
635 if (!empty($params['tone']) && !in_array($params['tone'], $valid_tones, true)) {
636 throw new \Exception('Invalid tone specified.');
637 }
638 }
639
640 /**
641 * Parse AI response into structured data
642 *
643 * @param string $ai_response Raw AI response
644 * @param array $original_params Original generation parameters
645 * @return array Structured brief data
646 */
647 private function parse_ai_response(string $ai_response, array $original_params): array {
648 $json_data = $this->parse_json_response($ai_response);
649
650 if (null === $json_data) {
651 // JSON parsing failed - return error structure
652 return $this->create_parsing_error_response($ai_response, $original_params);
653 }
654
655 return $this->structure_json_data($json_data, $original_params);
656 }
657
658 /**
659 * Parse JSON response from AI
660 *
661 * @param string $ai_response Raw AI response
662 * @return array|null Parsed JSON data or null if parsing fails
663 */
664 private function parse_json_response(string $ai_response): ?array {
665 // Clean the response - remove any text before/after JSON
666 $ai_response = trim($ai_response);
667
668 // Handle markdown code blocks (```json ... ```)
669 if (preg_match('/```(?:json)?\s*\n?(.*?)\n?```/s', $ai_response, $matches)) {
670 $json_string = trim($matches[1]);
671 } else {
672 // Find JSON object boundaries
673 $start = strpos($ai_response, '{');
674 $end = strrpos($ai_response, '}');
675
676 if (false === $start || false === $end || $start >= $end) {
677 return null;
678 }
679
680 $json_string = substr($ai_response, $start, $end - $start + 1);
681 }
682
683 $json_data = json_decode($json_string, true);
684
685 if (json_last_error() !== JSON_ERROR_NONE) {
686 return null;
687 }
688
689 return $json_data;
690 }
691
692 /**
693 * Structure JSON data into expected format
694 *
695 * @param array $json_data Parsed JSON data
696 * @param array $original_params Original generation parameters
697 * @return array Structured brief data
698 */
699 private function structure_json_data(array $json_data, array $original_params): array {
700 return [
701 'title' => $json_data['title_suggestions'] ?? [],
702 'meta_description' => $json_data['meta_descriptions'][0] ?? '',
703 'meta_descriptions' => $json_data['meta_descriptions'] ?? [],
704 'url_slugs' => $json_data['url_slugs'] ?? [],
705 'outline' => self::strip_outline_level_labels($json_data['outline'] ?? []),
706 'seo_recommendations' => [
707 'title_suggestions' => $json_data['title_suggestions'] ?? [],
708 'meta_description' => $json_data['meta_descriptions'][0] ?? '',
709 'meta_descriptions' => $json_data['meta_descriptions'] ?? [],
710 'url_slugs' => $json_data['url_slugs'] ?? [],
711 'focus_keyword_analysis' => $this->normalize_focus_keyword_analysis($json_data['focus_keyword_analysis'] ?? []),
712 'internal_links' => $json_data['internal_linking'] ?? [],
713 'related_keywords' => $json_data['related_keywords'] ?? [],
714 'long_tail_keywords' => []
715 ],
716 'social_media' => $json_data['social_media'] ?? [
717 'open_graph' => ['title' => '', 'description' => ''],
718 'twitter_card' => ['title' => '', 'description' => '']
719 ],
720 'schema_markup' => $json_data['schema_markup'] ?? [
721 'recommended_types' => [],
722 'key_properties' => [],
723 'faq_questions' => []
724 ],
725 'visual_content' => $json_data['visual_content'] ?? [
726 'image_recommendations' => [],
727 'alt_text_suggestions' => [],
728 'infographic_opportunities' => []
729 ],
730 'competitor_gaps' => $json_data['competitor_analysis']['content_gaps'] ?? [],
731 'call_to_actions' => $json_data['call_to_actions'] ?? [],
732 'writing_guidelines' => $json_data['writing_guidelines'] ?? [],
733 'content_body' => self::strip_heading_level_labels((string) ($json_data['content_body'] ?? '')),
734 'estimated_word_count' => $this->get_word_count_estimate($original_params['content_length'] ?? 'medium'),
735 'raw_response' => '', // Will be retrieved from ai_usage table
736 'generation_params' => $original_params,
737 'parsing_status' => 'success',
738 'created_at' => current_time('mysql')
739 ];
740 }
741
742 /**
743 * Remove a leading level label from a heading string.
744 *
745 * The prompt's own JSON example labelled outline headings with their level
746 * (`"heading": "H1: Main Title"` next to a separate `"level": 1`), so the
747 * model often carried the convention into the drafted article and Pro's
748 * "Insert into post" wrote `<h2>H2: Real Heading</h2>` into published
749 * content. The prompt no longer does that, but a prompt change never fully
750 * binds a model — so the label is stripped here too (#410).
751 *
752 * Covers the label forms a model actually emits: `H2:`, `h3:`, `H2 -`,
753 * `H4.`, `H2)` and the en/em dash variants, optionally wrapped in markdown
754 * emphasis (`**H2:**`). The delimiter is anchored directly after the digit
755 * so `H10:` — a plausible heading in a numbered list — is left alone, and
756 * only a leading label is matched so body copy that mentions a level
757 * survives. Trailing emphasis is consumed only when the same marker opened
758 * the label, so `H2: *emphasised start*` keeps its asterisks.
759 *
760 * @since 2.0.1
761 *
762 * @param string $heading Heading text.
763 * @return string Heading without its level prefix.
764 */
765 public static function strip_level_label(string $heading): string {
766 // En dash and em dash as raw UTF-8 bytes, so the pattern needs no /u
767 // modifier and cannot blank a heading that is not valid UTF-8.
768 $delimiter = '(?:[:.)\-]|\xe2\x80\x93|\xe2\x80\x94)';
769 $emphasis = '(\*{1,3}|_{1,3})';
770
771 $pattern = '/^\s*(?:'
772 . $emphasis . '\s*[Hh][1-6]\s*' . $delimiter . '\s*\1'
773 . '|[Hh][1-6]\s*' . $delimiter
774 . ')\s*/';
775
776 return (string) preg_replace($pattern, '', $heading);
777 }
778
779 /**
780 * Strip a level label from a heading's inner HTML.
781 *
782 * A model drafting publish-ready HTML often wraps the heading text in an
783 * inline tag (`<h2><strong>H2: Real Heading</strong></h2>`). That pushes a
784 * `<` in front of the label, so the leading run of inline opening tags is
785 * set aside and re-attached around the cleaned text.
786 *
787 * @since 2.0.1
788 *
789 * @param string $inner Heading inner HTML.
790 * @return string Inner HTML without the level prefix.
791 */
792 private static function strip_inner_level_label(string $inner): string {
793 $prefix = '';
794
795 if (preg_match('/^(\s*(?:<(?:strong|em|b|i|span|mark|code|u)\b[^>]*>\s*)+)(.*)$/is', $inner, $parts)) {
796 $prefix = $parts[1];
797 $inner = $parts[2];
798 }
799
800 return $prefix . self::strip_level_label($inner);
801 }
802
803 /**
804 * Strip level labels from every heading in an outline.
805 *
806 * @since 2.0.1
807 *
808 * @param mixed $outline Outline as returned by the model.
809 * @return array Outline with clean headings.
810 */
811 public static function strip_outline_level_labels($outline): array {
812 if (!is_array($outline)) {
813 return [];
814 }
815
816 foreach ($outline as $index => $section) {
817 if (is_array($section) && isset($section['heading']) && is_string($section['heading'])) {
818 $outline[$index]['heading'] = self::strip_level_label($section['heading']);
819 } elseif (is_string($section)) {
820 $outline[$index] = self::strip_level_label($section);
821 }
822 }
823
824 return $outline;
825 }
826
827 /**
828 * Strip level labels from the heading text inside drafted HTML.
829 *
830 * This is the path that reaches published post content, so it is the one
831 * that matters most. Only the text directly inside an <h1>-<h6> is touched.
832 *
833 * @since 2.0.1
834 *
835 * @param string $html Drafted article body.
836 * @return string Body with clean headings.
837 */
838 public static function strip_heading_level_labels(string $html): string {
839 if ('' === $html || false === stripos($html, '<h')) {
840 return $html;
841 }
842
843 return (string) preg_replace_callback(
844 '/(<h([1-6])\b[^>]*>)(.*?)(<\/h\2>)/is',
845 static function (array $parts): string {
846 return $parts[1] . self::strip_inner_level_label($parts[3]) . $parts[4];
847 },
848 $html
849 );
850 }
851
852 /**
853 * Create error response when JSON parsing fails
854 *
855 * @param string $ai_response Raw AI response
856 * @param array $original_params Original generation parameters
857 * @return array Error response structure
858 */
859 private function create_parsing_error_response(string $ai_response, array $original_params): array {
860 return [
861 'title' => ['Error: Unable to parse AI response'],
862 'meta_description' => 'AI response could not be parsed as valid JSON.',
863 'meta_descriptions' => ['AI response could not be parsed as valid JSON.'],
864 'url_slugs' => ['error-parsing-response'],
865 'outline' => [],
866 'seo_recommendations' => [
867 'title_suggestions' => ['Error: Unable to parse AI response'],
868 'meta_description' => 'AI response could not be parsed as valid JSON.',
869 'meta_descriptions' => ['AI response could not be parsed as valid JSON.'],
870 'url_slugs' => ['error-parsing-response'],
871 'focus_keyword_analysis' => [
872 'primary_placement' => [],
873 'secondary_integration' => [],
874 'density_guidelines' => []
875 ],
876 'internal_links' => [],
877 'related_keywords' => [],
878 'long_tail_keywords' => []
879 ],
880 'social_media' => [
881 'open_graph' => ['title' => 'Error', 'description' => 'Parsing failed'],
882 'twitter_card' => ['title' => 'Error', 'description' => 'Parsing failed']
883 ],
884 'schema_markup' => [
885 'recommended_types' => [],
886 'key_properties' => [],
887 'faq_questions' => []
888 ],
889 'visual_content' => [
890 'image_recommendations' => [],
891 'alt_text_suggestions' => [],
892 'infographic_opportunities' => []
893 ],
894 'competitor_gaps' => [],
895 'call_to_actions' => [],
896 'writing_guidelines' => [],
897 'content_body' => '',
898 'estimated_word_count' => $this->get_word_count_estimate($original_params['content_length'] ?? 'medium'),
899 'raw_response' => $ai_response, // Store raw response in error case
900 'generation_params' => $original_params,
901 'parsing_status' => 'failed',
902 'created_at' => current_time('mysql')
903 ];
904 }
905
906 /**
907 * Normalize focus keyword analysis to ensure proper array structure
908 *
909 * @param array $focus_keyword_analysis Raw focus keyword analysis data
910 * @return array Normalized focus keyword analysis
911 */
912 private function normalize_focus_keyword_analysis(array $focus_keyword_analysis): array {
913 $normalized = [
914 'primary_placement' => [],
915 'secondary_integration' => [],
916 'density_guidelines' => []
917 ];
918
919 // Normalize primary_placement
920 if (isset($focus_keyword_analysis['primary_placement'])) {
921 if (is_array($focus_keyword_analysis['primary_placement'])) {
922 $normalized['primary_placement'] = $focus_keyword_analysis['primary_placement'];
923 } elseif (is_string($focus_keyword_analysis['primary_placement'])) {
924 // Convert string to array by splitting on common delimiters
925 $normalized['primary_placement'] = array_filter(array_map('trim', preg_split('/[,;]/', $focus_keyword_analysis['primary_placement'])));
926 }
927 }
928
929 // Normalize secondary_integration - this is the problematic field
930 if (isset($focus_keyword_analysis['secondary_integration'])) {
931 if (is_array($focus_keyword_analysis['secondary_integration'])) {
932 $normalized['secondary_integration'] = $focus_keyword_analysis['secondary_integration'];
933 } elseif (is_string($focus_keyword_analysis['secondary_integration'])) {
934 // Convert string to array - split by sentences or use as single item
935 $text = trim($focus_keyword_analysis['secondary_integration']);
936 if (!empty($text)) {
937 // Split by sentences if it contains periods, otherwise use as single item
938 if (strpos($text, '.') !== false) {
939 $sentences = array_filter(array_map('trim', explode('.', $text)));
940 $normalized['secondary_integration'] = array_map(function($sentence) {
941 return $sentence . (substr($sentence, -1) !== '.' ? '.' : '');
942 }, $sentences);
943 } else {
944 $normalized['secondary_integration'] = [$text];
945 }
946 }
947 }
948 }
949
950 // Normalize density_guidelines
951 if (isset($focus_keyword_analysis['density_guidelines'])) {
952 if (is_array($focus_keyword_analysis['density_guidelines'])) {
953 $normalized['density_guidelines'] = $focus_keyword_analysis['density_guidelines'];
954 } elseif (is_string($focus_keyword_analysis['density_guidelines'])) {
955 // Convert string to array by splitting on common delimiters
956 $guidelines = array_filter(array_map('trim', preg_split('/[,;]/', $focus_keyword_analysis['density_guidelines'])));
957 $normalized['density_guidelines'] = $guidelines ?: [$focus_keyword_analysis['density_guidelines']];
958 }
959 }
960
961 return $normalized;
962 }
963 private function analyze_competitor_urls(array $urls): string {
964 $analysis_results = [];
965 $failed_urls = [];
966
967 // Limit to first 3 URLs to prevent timeout
968 $urls = array_slice($urls, 0, 3);
969
970 foreach ($urls as $url) {
971 $url = trim($url);
972 if (empty($url) || !filter_var($url, FILTER_VALIDATE_URL)) {
973 $failed_urls[] = $url . " (invalid URL)";
974 continue;
975 }
976
977 // SSRF guard: only fetch public http/https hosts. Blocks loopback,
978 // link-local (cloud metadata), private and reserved ranges before any
979 // request is made.
980 if (!$this->is_safe_public_url($url)) {
981 $failed_urls[] = $url . " (blocked: non-public host)";
982 continue;
983 }
984
985 $content_data = $this->scrape_competitor_content($url);
986 if ($content_data) {
987 $analysis_results[] = $this->format_competitor_analysis($url, $content_data);
988 } else {
989 $failed_urls[] = $url . " (failed to scrape)";
990 }
991 }
992
993 $result = "";
994
995 if (!empty($analysis_results)) {
996 $result .= implode("\n\n", $analysis_results);
997 }
998
999 if (!empty($failed_urls)) {
1000 $result .= "\n\nNote: The following URLs could not be analyzed:\n";
1001 $result .= "- " . implode("\n- ", $failed_urls);
1002 }
1003
1004 if (empty($analysis_results)) {
1005 return "No competitor URLs could be successfully analyzed. Please ensure URLs are accessible and valid.";
1006 }
1007
1008 return $result;
1009 }
1010
1011 /**
1012 * Whether a competitor URL is safe to fetch: an http/https URL whose host
1013 * resolves only to public IP addresses.
1014 *
1015 * Prevents SSRF — a low-privilege user could otherwise point competitor
1016 * scraping at loopback, link-local (e.g. 169.254.169.254 cloud metadata),
1017 * CGNAT, private or reserved addresses to probe internal services. The
1018 * block list lives in {@see \ThinkRank\Core\Url_Safety} so this and the
1019 * schema importer can never drift apart.
1020 *
1021 * @param string $url URL to validate.
1022 * @return bool True when safe to fetch.
1023 */
1024 private function is_safe_public_url(string $url): bool {
1025 return \ThinkRank\Core\Url_Safety::is_safe_public_url($url);
1026 }
1027
1028 /**
1029 * Scrape content from a competitor URL
1030 *
1031 * @param string $url The URL to scrape
1032 * @return array|null Content data or null if failed
1033 */
1034 private function scrape_competitor_content(string $url): ?array {
1035 // Url_Safety::safe_remote_get() follows redirects manually and re-checks
1036 // the resolved host on every hop, so a target that redirects to (or
1037 // rebinds onto) an internal address after the pre-flight check is
1038 // refused rather than fetched.
1039 $response = \ThinkRank\Core\Url_Safety::safe_remote_get($url, [
1040 'timeout' => 8, // Reduced from 15 to 8 seconds
1041 'user-agent' => 'Mozilla/5.0 (compatible; ThinkRank SEO Bot)',
1042 'headers' => [
1043 'Accept' => 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
1044 'Accept-Language' => 'en-US,en;q=0.5',
1045 ]
1046 ]);
1047
1048 if (is_wp_error($response)) {
1049 return null;
1050 }
1051
1052 $status_code = wp_remote_retrieve_response_code($response);
1053 if ($status_code !== 200) {
1054 return null;
1055 }
1056
1057 $html = wp_remote_retrieve_body($response);
1058 if (empty($html)) {
1059 return null;
1060 }
1061
1062 return $this->parse_html_content($html, $url);
1063 }
1064
1065 /**
1066 * Parse HTML content and extract key SEO elements
1067 *
1068 * @param string $html HTML content
1069 * @param string $url Original URL for context
1070 * @return array Parsed content data
1071 */
1072 private function parse_html_content(string $html, string $url): array {
1073 // Create DOMDocument to parse HTML
1074 $dom = new \DOMDocument();
1075
1076 // Suppress warnings for malformed HTML
1077 libxml_use_internal_errors(true);
1078 $dom->loadHTML('<?xml encoding="UTF-8">' . $html);
1079 libxml_clear_errors();
1080
1081 $xpath = new \DOMXPath($dom);
1082
1083 // Extract title
1084 $title_nodes = $xpath->query('//title');
1085 $title = $title_nodes->length > 0 ? trim($title_nodes->item(0)->textContent) : '';
1086
1087 // Extract meta description
1088 $meta_desc_nodes = $xpath->query('//meta[@name="description"]/@content');
1089 $meta_description = $meta_desc_nodes->length > 0 ? trim($meta_desc_nodes->item(0)->textContent) : '';
1090
1091 // Extract headings (H1-H6)
1092 $headings = [];
1093 for ($i = 1; $i <= 6; $i++) {
1094 $heading_nodes = $xpath->query("//h{$i}");
1095 foreach ($heading_nodes as $node) {
1096 $text = trim($node->textContent);
1097 if (!empty($text)) {
1098 $headings["h{$i}"][] = $text;
1099 }
1100 }
1101 }
1102
1103 // Extract body text and calculate word count
1104 $body_nodes = $xpath->query('//body');
1105 $body_text = '';
1106 if ($body_nodes->length > 0) {
1107 $body_text = $this->extract_clean_text($body_nodes->item(0));
1108 }
1109
1110 $word_count = str_word_count($body_text);
1111
1112 // Extract meta keywords if present
1113 $meta_keywords_nodes = $xpath->query('//meta[@name="keywords"]/@content');
1114 $meta_keywords = $meta_keywords_nodes->length > 0 ? trim($meta_keywords_nodes->item(0)->textContent) : '';
1115
1116 // Extract internal links count
1117 $internal_links = $xpath->query('//a[starts-with(@href, "/") or contains(@href, "' . wp_parse_url($url, PHP_URL_HOST) . '")]');
1118 $internal_link_count = $internal_links->length;
1119
1120 // Extract external links count
1121 $external_links = $xpath->query('//a[starts-with(@href, "http") and not(contains(@href, "' . wp_parse_url($url, PHP_URL_HOST) . '"))]');
1122 $external_link_count = $external_links->length;
1123
1124 // Extract images count and alt text analysis
1125 $images = $xpath->query('//img');
1126 $image_count = $images->length;
1127 $images_with_alt = $xpath->query('//img[@alt and @alt!=""]');
1128 $images_with_alt_count = $images_with_alt->length;
1129
1130 // Extract schema markup
1131 $schema_scripts = $xpath->query('//script[@type="application/ld+json"]');
1132 $has_schema = $schema_scripts->length > 0;
1133
1134 // Extract last modified date if available
1135 $last_modified_nodes = $xpath->query('//meta[@name="last-modified"]/@content | //meta[@property="article:modified_time"]/@content');
1136 $last_modified = $last_modified_nodes->length > 0 ? $last_modified_nodes->item(0)->textContent : '';
1137
1138 // Calculate readability metrics
1139 $readability_score = $this->calculate_readability_score($body_text);
1140
1141 // Extract keyword density for target keywords (if provided)
1142 $keyword_density = $this->analyze_keyword_density($body_text, $title);
1143
1144 // Detect content freshness indicators
1145 $freshness_indicators = $this->detect_freshness_indicators($html, $body_text);
1146
1147 return [
1148 'url' => $url,
1149 'title' => $title,
1150 'meta_description' => $meta_description,
1151 'meta_keywords' => $meta_keywords,
1152 'headings' => $headings,
1153 'word_count' => $word_count,
1154 'internal_links' => $internal_link_count,
1155 'external_links' => $external_link_count,
1156 'images' => [
1157 'total' => $image_count,
1158 'with_alt' => $images_with_alt_count,
1159 'alt_ratio' => $image_count > 0 ? round(($images_with_alt_count / $image_count) * 100, 1) : 0
1160 ],
1161 'seo' => [
1162 'has_schema' => $has_schema,
1163 'title_length' => strlen($title),
1164 'meta_desc_length' => strlen($meta_description),
1165 'title_score' => $this->score_title_seo($title),
1166 'meta_desc_score' => $this->score_meta_description($meta_description)
1167 ],
1168 'content_quality' => [
1169 'readability_score' => $readability_score,
1170 'keyword_density' => $keyword_density,
1171 'freshness_indicators' => $freshness_indicators,
1172 'content_depth' => $this->assess_content_depth($headings, $word_count)
1173 ],
1174 'last_modified' => $last_modified,
1175 'content_preview' => substr($body_text, 0, 500) . '...',
1176 'analysis_timestamp' => current_time('mysql')
1177 ];
1178 }
1179
1180 /**
1181 * Extract clean text from DOM node, removing scripts and styles
1182 *
1183 * @param \DOMNode $node DOM node to extract text from
1184 * @return string Clean text content
1185 */
1186 private function extract_clean_text(\DOMNode $node): string {
1187 // Remove script and style elements
1188 $xpath = new \DOMXPath($node->ownerDocument);
1189 $scripts = $xpath->query('.//script | .//style', $node);
1190
1191 foreach ($scripts as $script) {
1192 $script->parentNode->removeChild($script);
1193 }
1194
1195 // Get text content and clean it up
1196 $text = $node->textContent;
1197
1198 // Remove extra whitespace and normalize
1199 $text = preg_replace('/\s+/', ' ', $text);
1200 $text = trim($text);
1201
1202 return $text;
1203 }
1204
1205 /**
1206 * Format competitor analysis for AI prompt
1207 *
1208 * @param string $url Competitor URL
1209 * @param array $content_data Parsed content data
1210 * @return string Formatted analysis
1211 */
1212 private function format_competitor_analysis(string $url, array $content_data): string {
1213 $analysis = "=== COMPETITOR ANALYSIS ===\n";
1214 $analysis .= "URL: {$url}\n";
1215 $analysis .= "Title: {$content_data['title']} (Length: {$content_data['seo']['title_length']} chars, Score: {$content_data['seo']['title_score']['grade']})\n";
1216
1217 if (!empty($content_data['meta_description'])) {
1218 $analysis .= "Meta Description: {$content_data['meta_description']} (Length: {$content_data['seo']['meta_desc_length']} chars, Score: {$content_data['seo']['meta_desc_score']['grade']})\n";
1219 }
1220
1221 $analysis .= "\nCONTENT METRICS:\n";
1222 $analysis .= "- Word Count: {$content_data['word_count']} words\n";
1223 $analysis .= "- Content Depth: {$content_data['content_quality']['content_depth']['level']} (Score: {$content_data['content_quality']['content_depth']['score']}/100)\n";
1224 $analysis .= "- Readability: {$content_data['content_quality']['readability_score']['level']} (Score: {$content_data['content_quality']['readability_score']['score']}/100)\n";
1225 $analysis .= "- Internal Links: {$content_data['internal_links']}\n";
1226 $analysis .= "- External Links: {$content_data['external_links']}\n";
1227 $analysis .= "- Images: {$content_data['images']['total']} total, {$content_data['images']['with_alt']} with alt text ({$content_data['images']['alt_ratio']}%)\n";
1228
1229 // Add heading structure
1230 if (!empty($content_data['headings'])) {
1231 $analysis .= "\nCONTENT STRUCTURE:\n";
1232 foreach ($content_data['headings'] as $level => $headings) {
1233 $analysis .= "- " . strtoupper($level) . " ({count}): " . implode(', ', array_slice($headings, 0, 3));
1234 if (count($headings) > 3) {
1235 $analysis .= "... (+" . (count($headings) - 3) . " more)";
1236 }
1237 $analysis .= "\n";
1238 }
1239 }
1240
1241 // Add SEO features
1242 $analysis .= "\nSEO FEATURES:\n";
1243 $analysis .= "- Schema Markup: " . ($content_data['seo']['has_schema'] ? 'Yes' : 'No') . "\n";
1244 if (!empty($content_data['meta_keywords'])) {
1245 $analysis .= "- Meta Keywords: {$content_data['meta_keywords']}\n";
1246 }
1247
1248 // Add content quality insights
1249 if (!empty($content_data['content_quality']['keyword_density']['top_keywords'])) {
1250 $analysis .= "\nTOP KEYWORDS:\n";
1251 foreach (array_slice($content_data['content_quality']['keyword_density']['top_keywords'], 0, 5) as $kw) {
1252 $analysis .= "- {$kw['keyword']}: {$kw['count']} times ({$kw['density']}%)\n";
1253 }
1254 }
1255
1256 // Add freshness indicators
1257 if (!empty($content_data['content_quality']['freshness_indicators'])) {
1258 $analysis .= "\nCONTENT FRESHNESS:\n";
1259 foreach ($content_data['content_quality']['freshness_indicators'] as $indicator) {
1260 $analysis .= "- {$indicator}\n";
1261 }
1262 }
1263
1264 $analysis .= "\n" . str_repeat("=", 50) . "\n";
1265
1266 return $analysis;
1267 }
1268
1269 /**
1270 * Get word count estimate based on content length
1271 *
1272 * @param string $content_length Content length setting
1273 * @return int Estimated word count
1274 */
1275 private function get_word_count_estimate(string $content_length): int {
1276 $estimates = [
1277 'short' => 650,
1278 'medium' => 1250,
1279 'long' => 2500
1280 ];
1281
1282 return $estimates[$content_length] ?? 1250;
1283 }
1284 private function save_brief(array $brief_data): int {
1285 global $wpdb;
1286
1287 $table_name = $wpdb->prefix . 'thinkrank_content_briefs';
1288
1289 // Prepare data for insertion
1290 $insert_data = [
1291 'user_id' => get_current_user_id(),
1292 'title' => $brief_data['title'][0] ?? 'Untitled Brief',
1293 'target_keywords' => wp_json_encode($brief_data['generation_params']['target_keywords'] ?? []),
1294 'content_type' => $brief_data['generation_params']['content_type'] ?? 'blog_post',
1295 'brief_data' => wp_json_encode($brief_data),
1296 'created_at' => current_time('mysql'),
1297 'updated_at' => current_time('mysql')
1298 ];
1299
1300 $insert_format = [
1301 '%d', // user_id
1302 '%s', // title
1303 '%s', // target_keywords
1304 '%s', // content_type
1305 '%s', // brief_data
1306 '%s', // created_at
1307 '%s' // updated_at
1308 ];
1309
1310 // phpcs:ignore WordPress.DB.DirectDatabaseQuery.DirectQuery,WordPress.DB.DirectDatabaseQuery.NoCaching, PluginCheck.Security.DirectDB.UnescapedDBParameter -- Content brief storage requires direct database access
1311 $result = $wpdb->insert($table_name, $insert_data, $insert_format);
1312
1313 if (false === $result) {
1314 throw new \Exception('Failed to save content brief to database.');
1315 }
1316
1317 /**
1318 * Fires after a content brief is persisted.
1319 *
1320 * Analytics listens to drop its cached overview so the brief counts
1321 * on the Usages page are not stale for a TTL.
1322 *
1323 * @since 2.2.1
1324 *
1325 * @param int $brief_id Row id of the stored brief.
1326 */
1327 do_action('thinkrank_content_brief_created', (int) $wpdb->insert_id);
1328
1329 return $wpdb->insert_id;
1330 }
1331
1332 /**
1333 * Normalize brief data for React compatibility
1334 *
1335 * @param array $brief_data Brief data to normalize
1336 * @return array Normalized brief data
1337 */
1338 private function normalize_brief_data(array $brief_data): array {
1339 // Normalize focus_keyword_analysis
1340 if (isset($brief_data['seo_recommendations']['focus_keyword_analysis'])) {
1341 $brief_data['seo_recommendations']['focus_keyword_analysis'] =
1342 $this->normalize_focus_keyword_analysis($brief_data['seo_recommendations']['focus_keyword_analysis']);
1343 }
1344
1345 // Normalize call_to_actions (convert objects to strings)
1346 if (isset($brief_data['call_to_actions']) && is_array($brief_data['call_to_actions'])) {
1347 $brief_data['call_to_actions'] = array_map(function($cta) {
1348 if (is_array($cta) && isset($cta['text'])) {
1349 return $cta['text'] . (isset($cta['placement']) ? ' (' . $cta['placement'] . ')' : '');
1350 }
1351 return is_string($cta) ? $cta : '';
1352 }, $brief_data['call_to_actions']);
1353 }
1354
1355 // Normalize visual content image_recommendations (convert objects to strings)
1356 if (isset($brief_data['visual_content']['image_recommendations']) && is_array($brief_data['visual_content']['image_recommendations'])) {
1357 $brief_data['visual_content']['image_recommendations'] = array_map(function($rec) {
1358 if (is_array($rec)) {
1359 $text = '';
1360 if (isset($rec['type'])) { $text .= $rec['type'] . ': ';
1361 }
1362 if (isset($rec['description'])) { $text .= $rec['description'];
1363 }
1364 if (isset($rec['alt_text'])) { $text .= ' (Alt: ' . $rec['alt_text'] . ')';
1365 }
1366 return $text ?: 'Image recommendation';
1367 }
1368 return is_string($rec) ? $rec : 'Image recommendation';
1369 }, $brief_data['visual_content']['image_recommendations']);
1370 }
1371
1372 return $this->sanitize_brief_output($brief_data);
1373 }
1374
1375 /**
1376 * Strip untrusted markup out of brief fields before they leave the server.
1377 *
1378 * Brief content crosses a trust boundary: it is assembled by an external AI
1379 * provider from prompts that can include text fetched from competitor URLs.
1380 * It was previously copied out of the decoded JSON verbatim and rendered in
1381 * the admin SPA through dangerouslySetInnerHTML, so a malicious or
1382 * prompt-injected response could execute script in the admin origin (#365).
1383 *
1384 * Runs on the read path as well as generation, so briefs stored before this
1385 * fix are sanitized when they are loaded.
1386 *
1387 * @since 1.32.0
1388 *
1389 * @param array $brief_data Brief data to sanitize.
1390 * @return array Sanitized brief data.
1391 */
1392 private function sanitize_brief_output(array $brief_data): array {
1393 foreach ($brief_data as $key => $value) {
1394 // The raw provider response is debug output shown as plain text, and
1395 // the generation params are our own values — leave both intact.
1396 if ('raw_response' === $key || 'generation_params' === $key) {
1397 continue;
1398 }
1399
1400 if ('content_body' === $key && is_string($value)) {
1401 // Deliberately HTML: it is the drafted article and is rendered as
1402 // markup. wp_kses_post() keeps normal post formatting while
1403 // dropping script/style/iframe, event-handler attributes and
1404 // javascript: URLs.
1405 $brief_data[$key] = wp_kses_post($value);
1406 continue;
1407 }
1408
1409 if (is_array($value)) {
1410 $brief_data[$key] = $this->sanitize_brief_output($value);
1411 } elseif (is_string($value)) {
1412 // Every other field is plain text (headings, keywords, guidance).
1413 // Markdown emphasis markers are preserved; HTML tags are not.
1414 $brief_data[$key] = wp_strip_all_tags($value);
1415 }
1416 }
1417
1418 return $brief_data;
1419 }
1420
1421 /**
1422 * Get saved briefs for current user
1423 *
1424 * @param int $limit Number of briefs to retrieve
1425 * @param int $offset Offset for pagination
1426 * @return array Array of saved briefs
1427 */
1428 public function get_user_briefs(int $limit = 10, int $offset = 0): array {
1429 global $wpdb;
1430
1431 // Get table name and escape it properly (table names cannot be parameterized)
1432 $table_name = esc_sql($wpdb->prefix . 'thinkrank_content_briefs');
1433 $user_id = get_current_user_id();
1434
1435 // phpcs:ignore WordPress.DB.DirectDatabaseQuery.DirectQuery,WordPress.DB.DirectDatabaseQuery.NoCaching, PluginCheck.Security.DirectDB.UnescapedDBParameter -- Content brief retrieval requires direct database access
1436 $results = $wpdb->get_results(
1437 $wpdb->prepare(
1438 // phpcs:ignore WordPress.DB.PreparedSQL.InterpolatedNotPrepared, PluginCheck.Security.DirectDB.UnescapedDBParameter -- Table name is properly escaped using esc_sql()
1439 "SELECT * FROM `{$table_name}` WHERE user_id = %d ORDER BY created_at DESC LIMIT %d OFFSET %d",
1440 $user_id,
1441 $limit,
1442 $offset
1443 ),
1444 ARRAY_A
1445 );
1446
1447 // $wpdb->get_results() returns null on a DB error; this method's return
1448 // type is : array, so normalize before iterating/returning.
1449 if (!is_array($results)) {
1450 return [];
1451 }
1452
1453 // Decode JSON data and normalize for React compatibility
1454 foreach ($results as &$brief) {
1455 $brief = $this->hydrate_brief_row($brief);
1456 }
1457 unset($brief);
1458
1459 return $results;
1460 }
1461
1462 /**
1463 * Get a single saved brief by id, scoped to the current user.
1464 *
1465 * @param int $brief_id Brief ID.
1466 * @return array|null Hydrated brief, or null if it doesn't exist or does not
1467 * belong to the current user.
1468 */
1469 public function get_brief(int $brief_id): ?array {
1470 global $wpdb;
1471
1472 // Table names cannot be parameterized; escape it.
1473 $table_name = esc_sql($wpdb->prefix . 'thinkrank_content_briefs');
1474 $user_id = get_current_user_id();
1475
1476 // phpcs:ignore WordPress.DB.DirectDatabaseQuery.DirectQuery,WordPress.DB.DirectDatabaseQuery.NoCaching, PluginCheck.Security.DirectDB.UnescapedDBParameter -- Content brief retrieval requires direct database access
1477 $brief = $wpdb->get_row(
1478 $wpdb->prepare(
1479 // phpcs:ignore WordPress.DB.PreparedSQL.InterpolatedNotPrepared, PluginCheck.Security.DirectDB.UnescapedDBParameter -- Table name is properly escaped using esc_sql()
1480 "SELECT * FROM `{$table_name}` WHERE id = %d AND user_id = %d LIMIT 1",
1481 $brief_id,
1482 $user_id
1483 ),
1484 ARRAY_A
1485 );
1486
1487 if (!$brief) {
1488 return null;
1489 }
1490
1491 return $this->hydrate_brief_row($brief);
1492 }
1493
1494 /**
1495 * Decode + normalize a raw content-brief DB row for API/React consumption.
1496 *
1497 * @param array $brief Raw database row.
1498 * @return array Hydrated brief.
1499 */
1500 private function hydrate_brief_row(array $brief): array {
1501 $brief['target_keywords'] = json_decode($brief['target_keywords'], true);
1502 $brief['brief_data'] = json_decode($brief['brief_data'], true);
1503
1504 // Cast: the row comes from $wpdb, which returns every column as a
1505 // string, and both helpers declare an int parameter.
1506 $brief_id = (int) $brief['id'];
1507
1508 // Retrieve raw response from ai_usage table
1509 $brief['brief_data']['raw_response'] = $this->get_raw_response_for_brief($brief_id);
1510
1511 // Update model with actual model used (if available in ai_usage table)
1512 $actual_model = $this->get_actual_model_for_brief($brief_id);
1513 if ($actual_model && isset($brief['brief_data']['generation_meta'])) {
1514 $brief['brief_data']['generation_meta']['model'] = $actual_model;
1515 }
1516
1517 // Apply normalization to existing briefs to ensure React compatibility
1518 $brief['brief_data'] = $this->normalize_brief_data($brief['brief_data']);
1519
1520 return $brief;
1521 }
1522
1523 /**
1524 * Delete brief
1525 *
1526 * @param int $brief_id Brief ID to delete
1527 * @return bool Success status
1528 */
1529 public function delete_brief(int $brief_id): bool {
1530 global $wpdb;
1531
1532 $table_name = $wpdb->prefix . 'thinkrank_content_briefs';
1533 $user_id = get_current_user_id();
1534
1535 // phpcs:ignore WordPress.DB.DirectDatabaseQuery.DirectQuery,WordPress.DB.DirectDatabaseQuery.NoCaching, PluginCheck.Security.DirectDB.UnescapedDBParameter -- Content brief deletion requires direct database access
1536 $result = $wpdb->delete(
1537 $table_name,
1538 [
1539 'id' => $brief_id,
1540 'user_id' => $user_id
1541 ],
1542 ['%d', '%d']
1543 );
1544
1545 return $result !== false;
1546 }
1547
1548 /**
1549 * Calculate readability score using Flesch Reading Ease
1550 *
1551 * @param string $text Text to analyze
1552 * @return array Readability metrics
1553 */
1554 private function calculate_readability_score(string $text): array {
1555 if (empty($text)) {
1556 return ['score' => 0, 'level' => 'Unknown', 'grade' => 'N/A'];
1557 }
1558
1559 // Count sentences (approximate)
1560 $sentences = preg_split('/[.!?]+/', $text);
1561 $sentence_count = count(array_filter($sentences, function($s) { return trim($s) !== '';
1562 }));
1563
1564 // Count words
1565 $word_count = str_word_count($text);
1566
1567 // Count syllables (approximate)
1568 $syllable_count = $this->count_syllables($text);
1569
1570 if ($sentence_count === 0 || $word_count === 0) {
1571 return ['score' => 0, 'level' => 'Unknown', 'grade' => 'N/A'];
1572 }
1573
1574 // Flesch Reading Ease formula
1575 $avg_sentence_length = $word_count / $sentence_count;
1576 $avg_syllables_per_word = $syllable_count / $word_count;
1577
1578 $flesch_score = 206.835 - (1.015 * $avg_sentence_length) - (84.6 * $avg_syllables_per_word);
1579 $flesch_score = max(0, min(100, $flesch_score)); // Clamp between 0-100
1580
1581 // Determine reading level
1582 if ($flesch_score >= 90) {
1583 $level = 'Very Easy';
1584 $grade = '5th grade';
1585 } elseif ($flesch_score >= 80) {
1586 $level = 'Easy';
1587 $grade = '6th grade';
1588 } elseif ($flesch_score >= 70) {
1589 $level = 'Fairly Easy';
1590 $grade = '7th grade';
1591 } elseif ($flesch_score >= 60) {
1592 $level = 'Standard';
1593 $grade = '8th-9th grade';
1594 } elseif ($flesch_score >= 50) {
1595 $level = 'Fairly Difficult';
1596 $grade = '10th-12th grade';
1597 } elseif ($flesch_score >= 30) {
1598 $level = 'Difficult';
1599 $grade = 'College level';
1600 } else {
1601 $level = 'Very Difficult';
1602 $grade = 'Graduate level';
1603 }
1604
1605 return [
1606 'score' => round($flesch_score, 1),
1607 'level' => $level,
1608 'grade' => $grade
1609 ];
1610 }
1611
1612 /**
1613 * Count syllables in text (approximate)
1614 *
1615 * @param string $text Text to analyze
1616 * @return int Syllable count
1617 */
1618 private function count_syllables(string $text): int {
1619 $words = str_word_count(strtolower($text), 1);
1620 $syllable_count = 0;
1621
1622 foreach ($words as $word) {
1623 $syllable_count += $this->count_word_syllables($word);
1624 }
1625
1626 return max(1, $syllable_count); // At least 1 syllable
1627 }
1628
1629 /**
1630 * Count syllables in a single word
1631 *
1632 * @param string $word Word to analyze
1633 * @return int Syllable count
1634 */
1635 private function count_word_syllables(string $word): int {
1636 $word = strtolower($word);
1637 $vowels = 'aeiouy';
1638 $syllable_count = 0;
1639 $previous_was_vowel = false;
1640
1641 for ($i = 0, $len = strlen($word); $i < $len; $i++) {
1642 $is_vowel = strpos($vowels, $word[$i]) !== false;
1643 if ($is_vowel && !$previous_was_vowel) {
1644 $syllable_count++;
1645 }
1646 $previous_was_vowel = $is_vowel;
1647 }
1648
1649 // Handle silent 'e'
1650 if (substr($word, -1) === 'e' && $syllable_count > 1) {
1651 $syllable_count--;
1652 }
1653
1654 return max(1, $syllable_count);
1655 }
1656
1657 /**
1658 * Analyze keyword density in content
1659 *
1660 * @param string $text Content text
1661 * @param string $title Page title
1662 * @return array Keyword analysis
1663 */
1664 private function analyze_keyword_density(string $text, string $title): array {
1665 $combined_text = strtolower($title . ' ' . $text);
1666 $words = str_word_count($combined_text, 1);
1667 $total_words = count($words);
1668
1669 if ($total_words === 0) {
1670 return ['top_keywords' => [], 'total_words' => 0];
1671 }
1672
1673 // Count word frequency
1674 $word_counts = array_count_values($words);
1675
1676 // Filter out common stop words
1677 $stop_words = ['the', 'a', 'an', 'and', 'or', 'but', 'in', 'on', 'at', 'to', 'for', 'of', 'with', 'by', 'is', 'are', 'was', 'were', 'be', 'been', 'have', 'has', 'had', 'do', 'does', 'did', 'will', 'would', 'could', 'should', 'may', 'might', 'must', 'can', 'this', 'that', 'these', 'those', 'i', 'you', 'he', 'she', 'it', 'we', 'they', 'me', 'him', 'her', 'us', 'them'];
1678
1679 foreach ($stop_words as $stop_word) {
1680 unset($word_counts[$stop_word]);
1681 }
1682
1683 // Filter out single characters and numbers
1684 $word_counts = array_filter($word_counts, function($count, $word) {
1685 return strlen($word) > 2 && !is_numeric($word) && $count > 1;
1686 }, ARRAY_FILTER_USE_BOTH);
1687
1688 // Sort by frequency
1689 arsort($word_counts);
1690
1691 // Calculate density and format results
1692 $top_keywords = [];
1693 foreach (array_slice($word_counts, 0, 10, true) as $word => $count) {
1694 $density = round(($count / $total_words) * 100, 2);
1695 $top_keywords[] = [
1696 'keyword' => $word,
1697 'count' => $count,
1698 'density' => $density
1699 ];
1700 }
1701
1702 return [
1703 'top_keywords' => $top_keywords,
1704 'total_words' => $total_words
1705 ];
1706 }
1707
1708 /**
1709 * Detect content freshness indicators
1710 *
1711 * @param string $html Full HTML content
1712 * @param string $text Body text
1713 * @return array Freshness indicators
1714 */
1715 private function detect_freshness_indicators(string $html, string $text): array {
1716 $indicators = [];
1717
1718 // Check for date patterns in content
1719 if (preg_match('/\b(updated|revised|modified|published).*?(\d{4}|\d{1,2}\/\d{1,2}\/\d{2,4})/i', $text)) {
1720 $indicators[] = 'Contains recent update dates';
1721 }
1722
1723 // Check for current year references
1724 $current_year = gmdate('Y');
1725 if (strpos($text, $current_year) !== false) {
1726 $indicators[] = "References current year ({$current_year})";
1727 }
1728
1729 // Check for "latest", "new", "recent" keywords
1730 if (preg_match('/\b(latest|newest|recent|updated|current|modern|today)\b/i', $text)) {
1731 $indicators[] = 'Uses freshness keywords';
1732 }
1733
1734 // Check for structured data with dates
1735 if (preg_match('/"dateModified"|"datePublished"/i', $html)) {
1736 $indicators[] = 'Has structured date metadata';
1737 }
1738
1739 return $indicators;
1740 }
1741
1742 /**
1743 * Score title for SEO effectiveness
1744 *
1745 * @param string $title Page title
1746 * @return array Title scoring
1747 */
1748 private function score_title_seo(string $title): array {
1749 $score = 0;
1750 $max_score = 100;
1751 $feedback = [];
1752
1753 // Length check (optimal: 50-60 characters)
1754 $length = strlen($title);
1755 if ($length >= 50 && $length <= 60) {
1756 $score += 25;
1757 $feedback[] = 'Good length (50-60 chars)';
1758 } elseif ($length >= 40 && $length <= 70) {
1759 $score += 15;
1760 $feedback[] = 'Acceptable length';
1761 } else {
1762 $feedback[] = $length < 40 ? 'Too short (under 40 chars)' : 'Too long (over 70 chars)';
1763 }
1764
1765 // Word count (optimal: 5-9 words)
1766 $word_count = str_word_count($title);
1767 if ($word_count >= 5 && $word_count <= 9) {
1768 $score += 20;
1769 $feedback[] = 'Good word count';
1770 } elseif ($word_count >= 3 && $word_count <= 12) {
1771 $score += 10;
1772 $feedback[] = 'Acceptable word count';
1773 } else {
1774 $feedback[] = $word_count < 3 ? 'Too few words' : 'Too many words';
1775 }
1776
1777 // Check for power words
1778 $power_words = ['ultimate', 'complete', 'guide', 'best', 'top', 'essential', 'proven', 'expert', 'advanced', 'beginner'];
1779 $has_power_words = false;
1780 foreach ($power_words as $power_word) {
1781 if (stripos($title, $power_word) !== false) {
1782 $has_power_words = true;
1783 break;
1784 }
1785 }
1786 if ($has_power_words) {
1787 $score += 15;
1788 $feedback[] = 'Contains power words';
1789 }
1790
1791 // Check for numbers
1792 if (preg_match('/\d+/', $title)) {
1793 $score += 10;
1794 $feedback[] = 'Contains numbers';
1795 }
1796
1797 // Check for emotional triggers
1798 $emotional_words = ['amazing', 'incredible', 'shocking', 'secret', 'revealed', 'proven', 'guaranteed'];
1799 $has_emotional_words = false;
1800 foreach ($emotional_words as $emotional_word) {
1801 if (stripos($title, $emotional_word) !== false) {
1802 $has_emotional_words = true;
1803 break;
1804 }
1805 }
1806 if ($has_emotional_words) {
1807 $score += 10;
1808 $feedback[] = 'Contains emotional triggers';
1809 }
1810
1811 // Uniqueness check (avoid generic titles)
1812 $generic_patterns = ['untitled', 'new page', 'home', 'welcome'];
1813 $is_generic = false;
1814 foreach ($generic_patterns as $pattern) {
1815 if (stripos($title, $pattern) !== false) {
1816 $is_generic = true;
1817 break;
1818 }
1819 }
1820 if (!$is_generic) {
1821 $score += 20;
1822 $feedback[] = 'Appears unique';
1823 } else {
1824 $feedback[] = 'Appears generic';
1825 }
1826
1827 return [
1828 'score' => min($score, $max_score),
1829 'max_score' => $max_score,
1830 'grade' => $this->get_grade_from_score($score),
1831 'feedback' => $feedback
1832 ];
1833 }
1834
1835 /**
1836 * Score meta description for SEO effectiveness
1837 *
1838 * @param string $meta_desc Meta description
1839 * @return array Meta description scoring
1840 */
1841 private function score_meta_description(string $meta_desc): array {
1842 $score = 0;
1843 $max_score = 100;
1844 $feedback = [];
1845
1846 if (empty($meta_desc)) {
1847 return [
1848 'score' => 0,
1849 'max_score' => $max_score,
1850 'grade' => 'F',
1851 'feedback' => ['No meta description found']
1852 ];
1853 }
1854
1855 // Length check (optimal: 150-160 characters)
1856 $length = strlen($meta_desc);
1857 if ($length >= 150 && $length <= 160) {
1858 $score += 30;
1859 $feedback[] = 'Optimal length (150-160 chars)';
1860 } elseif ($length >= 120 && $length <= 170) {
1861 $score += 20;
1862 $feedback[] = 'Good length';
1863 } elseif ($length >= 100 && $length <= 180) {
1864 $score += 10;
1865 $feedback[] = 'Acceptable length';
1866 } else {
1867 $feedback[] = $length < 100 ? 'Too short (under 100 chars)' : 'Too long (over 180 chars)';
1868 }
1869
1870 // Check for call-to-action
1871 $cta_words = ['learn', 'discover', 'find out', 'get', 'download', 'try', 'start', 'join', 'sign up', 'contact', 'buy', 'shop'];
1872 $has_cta = false;
1873 foreach ($cta_words as $cta_word) {
1874 if (stripos($meta_desc, $cta_word) !== false) {
1875 $has_cta = true;
1876 break;
1877 }
1878 }
1879 if ($has_cta) {
1880 $score += 20;
1881 $feedback[] = 'Contains call-to-action';
1882 }
1883
1884 // Check for unique selling proposition
1885 $usp_words = ['best', 'top', 'leading', 'expert', 'professional', 'trusted', 'proven', 'award-winning'];
1886 $has_usp = false;
1887 foreach ($usp_words as $usp_word) {
1888 if (stripos($meta_desc, $usp_word) !== false) {
1889 $has_usp = true;
1890 break;
1891 }
1892 }
1893 if ($has_usp) {
1894 $score += 15;
1895 $feedback[] = 'Contains unique selling proposition';
1896 }
1897
1898 // Check for benefits/value proposition
1899 $benefit_words = ['save', 'improve', 'increase', 'boost', 'enhance', 'optimize', 'maximize', 'reduce', 'eliminate'];
1900 $has_benefits = false;
1901 foreach ($benefit_words as $benefit_word) {
1902 if (stripos($meta_desc, $benefit_word) !== false) {
1903 $has_benefits = true;
1904 break;
1905 }
1906 }
1907 if ($has_benefits) {
1908 $score += 15;
1909 $feedback[] = 'Highlights benefits';
1910 }
1911
1912 // Readability check
1913 $sentences = preg_split('/[.!?]+/', $meta_desc);
1914 $sentence_count = count(array_filter($sentences, function($s) { return trim($s) !== '';
1915 }));
1916 if ($sentence_count >= 1 && $sentence_count <= 3) {
1917 $score += 20;
1918 $feedback[] = 'Good sentence structure';
1919 } else {
1920 $feedback[] = $sentence_count === 0 ? 'No clear sentences' : 'Too many sentences';
1921 }
1922
1923 return [
1924 'score' => min($score, $max_score),
1925 'max_score' => $max_score,
1926 'grade' => $this->get_grade_from_score($score),
1927 'feedback' => $feedback
1928 ];
1929 }
1930
1931 /**
1932 * Assess content depth based on structure and length
1933 *
1934 * @param array $headings Heading structure
1935 * @param int $word_count Word count
1936 * @return array Content depth assessment
1937 */
1938 private function assess_content_depth(array $headings, int $word_count): array {
1939 $depth_score = 0;
1940 $max_score = 100;
1941
1942 // Word count scoring (more words = more depth)
1943 if ($word_count >= 2000) {
1944 $depth_score += 40;
1945 } elseif ($word_count >= 1000) {
1946 $depth_score += 30;
1947 } elseif ($word_count >= 500) {
1948 $depth_score += 20;
1949 } elseif ($word_count >= 300) {
1950 $depth_score += 10;
1951 }
1952
1953 // Heading structure scoring
1954 $total_headings = 0;
1955 $heading_levels = 0;
1956 foreach ($headings as $level => $level_headings) {
1957 $total_headings += count($level_headings);
1958 $heading_levels++;
1959 }
1960
1961 if ($total_headings >= 10) {
1962 $depth_score += 25;
1963 } elseif ($total_headings >= 5) {
1964 $depth_score += 15;
1965 } elseif ($total_headings >= 3) {
1966 $depth_score += 10;
1967 }
1968
1969 // Heading hierarchy scoring
1970 if ($heading_levels >= 3) {
1971 $depth_score += 20;
1972 } elseif ($heading_levels >= 2) {
1973 $depth_score += 15;
1974 }
1975
1976 // Content structure bonus
1977 if (isset($headings['h1']) && isset($headings['h2'])) {
1978 $depth_score += 15;
1979 }
1980
1981 // Determine depth level
1982 if ($depth_score >= 80) {
1983 $level = 'Comprehensive';
1984 } elseif ($depth_score >= 60) {
1985 $level = 'Detailed';
1986 } elseif ($depth_score >= 40) {
1987 $level = 'Moderate';
1988 } elseif ($depth_score >= 20) {
1989 $level = 'Basic';
1990 } else {
1991 $level = 'Shallow';
1992 }
1993
1994 return [
1995 'score' => min($depth_score, $max_score),
1996 'level' => $level,
1997 'word_count' => $word_count,
1998 'total_headings' => $total_headings,
1999 'heading_levels' => $heading_levels
2000 ];
2001 }
2002
2003 /**
2004 * Convert numeric score to letter grade
2005 *
2006 * @param int $score Numeric score
2007 * @return string Letter grade
2008 */
2009 private function get_grade_from_score(int $score): string {
2010 if ($score >= 90) { return 'A';
2011 }
2012 if ($score >= 80) { return 'B';
2013 }
2014 if ($score >= 70) { return 'C';
2015 }
2016 if ($score >= 60) { return 'D';
2017 }
2018 return 'F';
2019 }
2020
2021 /**
2022 * Log AI usage for analytics
2023 *
2024 * @param int $user_id User ID
2025 * @param string $action Action performed
2026 * @param int $tokens_used Tokens consumed
2027 * @param int|null $post_id Related post/brief ID
2028 * @param string|null $raw_response Raw AI response for debugging
2029 * @param string|null $actual_model Actual model used (from response)
2030 * @return int Usage record ID
2031 */
2032 private function log_ai_usage(int $user_id, string $action, int $tokens_used, ?int $post_id = null, ?string $raw_response = null, ?string $actual_model = null): int {
2033 global $wpdb;
2034
2035 $table_name = $wpdb->prefix . 'thinkrank_ai_usage';
2036
2037 $metadata = [];
2038 if ($raw_response) {
2039 $metadata['raw_response'] = $raw_response;
2040 }
2041 if ($actual_model) {
2042 $metadata['actual_model'] = $actual_model;
2043 }
2044
2045 // phpcs:ignore WordPress.DB.DirectDatabaseQuery.DirectQuery,WordPress.DB.DirectDatabaseQuery.NoCaching, PluginCheck.Security.DirectDB.UnescapedDBParameter -- AI usage logging requires direct database access
2046 $wpdb->insert(
2047 $table_name,
2048 [
2049 'user_id' => $user_id,
2050 'action' => $action,
2051 'tokens_used' => $tokens_used,
2052 'provider' => $this->settings->get('ai_provider', Settings::AI_PROVIDER_NONE),
2053 'post_id' => $post_id,
2054 'metadata' => !empty($metadata) ? wp_json_encode($metadata) : null,
2055 'created_at' => current_time('mysql'),
2056 ],
2057 ['%d', '%s', '%d', '%s', '%d', '%s', '%s']
2058 );
2059
2060 /**
2061 * Fires after an AI usage row is recorded.
2062 *
2063 * @since 2.2.1
2064 *
2065 * @param int $user_id User the usage was recorded against.
2066 */
2067 do_action('thinkrank_ai_usage_logged', $user_id);
2068
2069 return $wpdb->insert_id;
2070 }
2071
2072 /**
2073 * Get raw AI response for a brief from ai_usage table
2074 *
2075 * @param int $brief_id Brief ID
2076 * @return string Raw AI response or empty string if not found
2077 */
2078 private function get_raw_response_for_brief(int $brief_id): string {
2079 global $wpdb;
2080
2081 $table_name = $wpdb->prefix . 'thinkrank_ai_usage';
2082
2083 // phpcs:ignore WordPress.DB.DirectDatabaseQuery.DirectQuery,WordPress.DB.DirectDatabaseQuery.NoCaching, PluginCheck.Security.DirectDB.UnescapedDBParameter -- AI usage retrieval requires direct database access
2084 $result = $wpdb->get_var(
2085 // phpcs:ignore WordPress.DB.PreparedSQL.InterpolatedNotPrepared, PluginCheck.Security.DirectDB.UnescapedDBParameter -- Table name is validated with WordPress prefix
2086 $wpdb->prepare(
2087 // phpcs:ignore WordPress.DB.PreparedSQL.InterpolatedNotPrepared, PluginCheck.Security.DirectDB.UnescapedDBParameter -- Table name is validated with WordPress prefix
2088 "SELECT metadata FROM `{$table_name}` WHERE post_id = %d AND action = 'content_brief' ORDER BY created_at DESC LIMIT 1",
2089 $brief_id
2090 )
2091 );
2092
2093 if ($result) {
2094 $metadata = json_decode($result, true);
2095 return $metadata['raw_response'] ?? '';
2096 }
2097
2098 return '';
2099 }
2100
2101 /**
2102 * Get actual model used for a brief from ai_usage table
2103 *
2104 * @param int $brief_id Brief ID
2105 * @return string|null Actual model used or null if not found
2106 */
2107 private function get_actual_model_for_brief(int $brief_id): ?string {
2108 global $wpdb;
2109
2110 $table_name = $wpdb->prefix . 'thinkrank_ai_usage';
2111
2112 // phpcs:ignore WordPress.DB.DirectDatabaseQuery.DirectQuery,WordPress.DB.DirectDatabaseQuery.NoCaching, PluginCheck.Security.DirectDB.UnescapedDBParameter -- AI usage retrieval requires direct database access
2113 $result = $wpdb->get_var(
2114 // phpcs:ignore WordPress.DB.PreparedSQL.InterpolatedNotPrepared, PluginCheck.Security.DirectDB.UnescapedDBParameter -- Table name is validated with WordPress prefix
2115 $wpdb->prepare(
2116 // phpcs:ignore WordPress.DB.PreparedSQL.InterpolatedNotPrepared, PluginCheck.Security.DirectDB.UnescapedDBParameter -- Table name is validated with WordPress prefix
2117 "SELECT metadata FROM `{$table_name}` WHERE post_id = %d AND action = 'content_brief' ORDER BY created_at DESC LIMIT 1",
2118 $brief_id
2119 )
2120 );
2121
2122 if ($result) {
2123 $metadata = json_decode($result, true);
2124 return $metadata['actual_model'] ?? null;
2125 }
2126
2127 return null;
2128 }
2129
2130 /**
2131 * Get Prompt Builder instance
2132 *
2133 * @since 1.0.0
2134 *
2135 * @return \ThinkRank\AI\Prompt_Builder Prompt Builder instance
2136 */
2137 private function get_prompt_builder(): \ThinkRank\AI\Prompt_Builder {
2138 if (!class_exists('ThinkRank\\AI\\Prompt_Builder')) {
2139 require_once THINKRANK_PLUGIN_DIR . 'includes/ai/class-prompt-builder.php';
2140 }
2141 return new \ThinkRank\AI\Prompt_Builder();
2142 }
2143 }
2144