PluginProbe
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO / 2.14.2
ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO v2.14.2
2.14.2 2.14.1 2.14.0 2.13.0 2.12.0 2.11.0 2.10.0 2.9.0 2.8.0 2.7.0 2.6.0 2.5.0 2.4.0 2.3.0 2.2.0 2.1.1 2.1.0 2.0.2 2.0.1 2.0.0 1.32.0 1.31.0 1.30.0 1.29.0 1.28.0 All 57 releases
thinkrank / includes / ai / class-content-brief-generator.php

class-content-brief-generator.php in ThinkRank AI SEO – AI SEO Plugin for WordPress: Schema, XML Sitemaps, Meta Tags, Search Console & Local SEO 2.14.2, at includes/ai/class-content-brief-generator.php

2,169 lines 88.1 KB
No matching file
Up and down to move Enter to open Esc to close
Raw Download Zip
1 <?php
2 /**
3 * Content Brief Generator
4 *
5 * Handles AI-powered content brief generation with competitor analysis
6 *
7 * @package ThinkRank
8 * @subpackage AI
9 * @since 1.0.0
10 */
11
12 namespace ThinkRank\AI;
13
14 use ThinkRank\Core\Settings;
15 use ThinkRank\AI\OpenAI_Client;
16 use ThinkRank\AI\Claude_Client;
17 use ThinkRank\AI\OpenRouter_Client;
18
19 // Prevent direct access
20 if (!defined('ABSPATH')) {
21 exit;
22 }
23
24 /**
25 * Content Brief Generator class
26 */
27 class Content_Brief_Generator {
28
29 /**
30 * AI request timeout (seconds) for brief generation.
31 *
32 * Content briefs request a very large completion (~0.9–0.95 of the model's
33 * max tokens) from reasoning models, which routinely take 40–90s — far
34 * longer than the AI clients' 30s default. Without this the HTTP call is
35 * aborted with cURL error 28 and the brief never completes. PHP execution
36 * time is covered by each client's raise_request_time_limit() (timeout+45).
37 */
38 private const AI_REQUEST_TIMEOUT = 120;
39
40 /**
41 * AI request timeout (seconds) for OpenAI specifically.
42 *
43 * OpenAI's reasoning models (GPT-5 / o-series) burn reasoning tokens before
44 * emitting any content, and briefs request ~95% of the model's completion
45 * limit — so the call frequently runs past the 120s the other providers
46 * need. PHP execution time is covered by raise_request_time_limit()
47 * (timeout+45); the web server's own read timeout still caps the maximum.
48 */
49 private const OPENAI_REQUEST_TIMEOUT = 300;
50
51 /**
52 * Minimum output-token budget for a content brief.
53 *
54 * A shorter length tier must never starve the structured JSON — plus any
55 * reasoning/thinking tokens, which are drawn from the same budget — to the
56 * point of truncating mid-response (the failure #165 fixed on Gemini). This
57 * floor is only a safety net for small-ceiling models; it never exceeds the
58 * model-aware base budget. See scale_tokens_for_length().
59 */
60 private const MIN_BRIEF_TOKENS = 2000;
61
62 /**
63 * Content-length → budget multipliers, applied to the model-aware base.
64 *
65 * NOTE: provisional starting points (issue #287). They make Short/Medium/
66 * Long request measurably different budgets, but the exact figures should
67 * be validated against recorded completion-token usage for a real brief on
68 * each provider before being treated as final. Unknown lengths fall back to
69 * the 'medium' tier (see scale_tokens_for_length()).
70 */
71 private const LENGTH_TOKEN_MULTIPLIERS = [
72 'short' => 0.6,
73 'medium' => 0.8,
74 'long' => 1.0,
75 ];
76
77 /**
78 * Settings instance
79 *
80 * @var Settings
81 */
82 private Settings $settings;
83
84 /**
85 * AI client instance
86 *
87 * Null when the generator was built for storage-only work.
88 *
89 * @var OpenAI_Client|Claude_Client|null
90 */
91 private $ai_client;
92
93 /**
94 * Constructor
95 *
96 * @param Settings|null $settings Settings instance
97 * @param OpenAI_Client|Claude_Client|null $ai_client AI client instance
98 * @param bool $require_ai_client Whether a provider client is required. Pass
99 * false for storage-only use (list/export/
100 * delete), which never calls a provider.
101 */
102 public function __construct(?Settings $settings = null, $ai_client = null, bool $require_ai_client = true) {
103 $this->settings = $settings ?? Settings::instance();
104
105 if ($ai_client) {
106 $this->ai_client = $ai_client;
107
108 return;
109 }
110
111 // Read-only callers (listing, exporting and deleting saved briefs) only
112 // touch the database and never reach a provider. Constructing a client
113 // for them turns "no API key configured" — the default state of a fresh
114 // install — into a hard failure, so let them opt out.
115 if (!$require_ai_client) {
116 return;
117 }
118
119 // Fallback to creating own client for backward compatibility
120 $this->init_ai_client();
121 }
122
123 /**
124 * Initialize AI client based on available API keys
125 *
126 * @return void
127 *
128 * @throws \Exception On failure.
129 */
130 private function init_ai_client(): void {
131 $provider = $this->settings->get('ai_provider', Settings::AI_PROVIDER_NONE);
132
133 if ($provider === 'openai') {
134 $api_key = $this->settings->get('openai_api_key');
135 if ($api_key) {
136 $model = $this->settings->get('openai_model', Settings::DEFAULT_OPENAI_MODEL);
137 $this->ai_client = new OpenAI_Client($api_key, $model, self::OPENAI_REQUEST_TIMEOUT);
138 }
139 } elseif ($provider === 'claude') {
140 $api_key = $this->settings->get('claude_api_key');
141 if ($api_key) {
142 $model = $this->settings->get('claude_model', Settings::DEFAULT_CLAUDE_MODEL);
143 $this->ai_client = new Claude_Client($api_key, $model, self::AI_REQUEST_TIMEOUT);
144 }
145 } elseif ($provider === 'gemini') {
146 $api_key = $this->settings->get('gemini_api_key');
147 if ($api_key) {
148 $model = $this->settings->get('gemini_model', Settings::DEFAULT_GEMINI_MODEL);
149 $this->ai_client = new Gemini_Client($api_key, $model, self::AI_REQUEST_TIMEOUT);
150 }
151 } elseif ($provider === 'openrouter') {
152 $api_key = $this->settings->get('openrouter_api_key');
153 if ($api_key) {
154 $model = $this->settings->get('openrouter_model', Settings::DEFAULT_OPENROUTER_MODEL);
155 $this->ai_client = new OpenRouter_Client($api_key, $model, self::AI_REQUEST_TIMEOUT);
156 }
157 } elseif ($provider === 'openai_compatible') {
158 // Same client as OpenAI, different host — and the key is optional,
159 // so the URL and model id are what gate it (#721). The user's own
160 // timeout applies: a local model writing a brief on CPU is slow,
161 // and the setting exists for exactly that.
162 $base_url = (string) $this->settings->get('openai_compatible_base_url', '');
163 $model = trim((string) $this->settings->get('openai_compatible_model', ''));
164 if ('' !== $base_url && '' !== $model) {
165 $this->ai_client = new OpenAI_Client(
166 (string) $this->settings->get('openai_compatible_api_key', ''),
167 $model,
168 (int) $this->settings->get('openai_compatible_timeout', Settings::DEFAULT_OPENAI_COMPATIBLE_TIMEOUT),
169 $base_url
170 );
171 $this->ai_client->set_json_mode((bool) $this->settings->get('openai_compatible_json_mode', false));
172 }
173 }
174
175 if (!$this->ai_client) {
176 if ('openai_compatible' === $provider) {
177 throw new \Exception('Please set the base URL and model id for your OpenAI-compatible endpoint in ThinkRank settings.');
178 }
179
180 throw new \Exception('Please configure your AI provider API key in ThinkRank settings.');
181 }
182 }
183
184 /**
185 * Get current AI model being used
186 *
187 * @return string Current model name
188 */
189 private function get_current_model(): string {
190 // Try to get model from the actual AI client if available
191 if ($this->ai_client && method_exists($this->ai_client, 'get_model')) {
192 return $this->ai_client->get_model();
193 }
194
195 // Fallback to settings
196 $provider = $this->settings->get('ai_provider', Settings::AI_PROVIDER_NONE);
197 if (Settings::AI_PROVIDER_NONE === $provider) {
198 // No provider chosen, so there is no model to name. Reporting the
199 // OpenAI default here would attribute output to a provider the site
200 // never selected (#572).
201 return '';
202 }
203 if ($provider === 'claude') {
204 return $this->settings->get('claude_model', Settings::DEFAULT_CLAUDE_MODEL);
205 } elseif ($provider === 'gemini') {
206 return $this->settings->get('gemini_model', Settings::DEFAULT_GEMINI_MODEL);
207 } elseif ($provider === 'openrouter') {
208 return $this->settings->get('openrouter_model', Settings::DEFAULT_OPENROUTER_MODEL);
209 } elseif ($provider === 'openai_compatible') {
210 return (string) $this->settings->get('openai_compatible_model', '');
211 } else {
212 return $this->settings->get('openai_model', Settings::DEFAULT_OPENAI_MODEL);
213 }
214 }
215
216 /**
217 * Resolve the reasoning-effort level for a content-brief request.
218 *
219 * Without an explicit level, GPT-5 models run at their default (maximum)
220 * reasoning effort against a ~95% completion budget — the slowest and
221 * costliest configuration, where billed reasoning tokens (drawn from the
222 * same budget) are spent before any visible output (issue #286).
223 *
224 * A brief is a structured planning task, so 'low' is a provisional middle
225 * ground between 'minimal' and the model's default.
226 * The level is filterable so a site can trade latency for more reasoning;
227 * returning '' opts out entirely and lets the model use its default effort.
228 * Only the GPT-5 family consumes this — o1/o3, gpt-4o and the non-OpenAI
229 * clients ignore an unrecognised option key.
230 *
231 * @param string $model The resolved model ID (passed to the filter).
232 * @param array $params The brief generation parameters (passed to the filter).
233 * @return string One of 'minimal' | 'low' | 'medium' | 'high', or '' to opt out.
234 */
235 private function resolve_reasoning_effort(string $model, array $params): string {
236 /**
237 * Filter the reasoning-effort level used for content-brief generation.
238 *
239 * @param string $effort The default level ('low'). Return '' to opt out.
240 * @param string $model The resolved model ID for this request.
241 * @param array $params The brief generation parameters.
242 */
243 $effort = (string) apply_filters('thinkrank_content_brief_reasoning_effort', 'low', $model, $params);
244
245 // Only values OpenAI accepts may reach the request body ('' opts out).
246 // An unrecognised filter return (e.g. 'turbo') would otherwise be sent
247 // verbatim and fail the whole brief with a 400, so degrade to the
248 // documented default instead.
249 $allowed = ['', 'minimal', 'low', 'medium', 'high'];
250 return in_array($effort, $allowed, true) ? $effort : 'low';
251 }
252
253 /**
254 * Get current AI provider
255 *
256 * @return string Current provider name
257 */
258 private function get_current_provider(): string {
259 return $this->settings->get('ai_provider', Settings::AI_PROVIDER_NONE);
260 }
261
262 /**
263 * Extract token usage from AI response
264 *
265 * @param array $ai_response AI response data
266 * @return int Number of tokens used
267 */
268 private function extract_token_usage(array $ai_response): int {
269 $provider = $this->get_current_provider();
270
271 if ($provider === 'openai' || $provider === 'openrouter' || $provider === 'openai_compatible') {
272 // OpenAI-compatible format: response['usage']['total_tokens']
273 return (int) ($ai_response['usage']['total_tokens'] ?? 0);
274 } elseif ($provider === 'claude') {
275 // Claude format: response['usage']['input_tokens'] + response['usage']['output_tokens']
276 $input_tokens = (int) ($ai_response['usage']['input_tokens'] ?? 0);
277 $output_tokens = (int) ($ai_response['usage']['output_tokens'] ?? 0);
278 return $input_tokens + $output_tokens;
279 } elseif ($provider === 'gemini') {
280 // Gemini format: response['usageMetadata']['totalTokenCount']
281 return (int) ($ai_response['usageMetadata']['totalTokenCount'] ?? 0);
282 }
283
284 // Fallback: return 0 if provider not recognized or no usage data
285 return 0;
286 }
287
288 /**
289 * Extract actual model used from AI response
290 *
291 * @param array $ai_response AI response data
292 * @return string|null Actual model used or null if not found
293 */
294 private function extract_model_from_response(array $ai_response): ?string {
295 // OpenAI format: response['model']
296 if (isset($ai_response['model'])) {
297 return $ai_response['model'];
298 }
299
300 // Claude format: response['model']
301 if (isset($ai_response['model'])) {
302 return $ai_response['model'];
303 }
304
305 // Gemini doesn't include model in response, fallback to client model
306 return null;
307 }
308
309 /**
310 * Generate content brief
311 *
312 * @param array $params Brief generation parameters
313 * @return array Generated brief data
314 * @throws \Exception If generation fails
315 */
316 public function generate_brief(array $params): array {
317 // Validate required parameters
318 $this->validate_brief_params($params);
319
320 // Extract parameters
321 $target_keywords = $params['target_keywords'] ?? [];
322 $content_type = $params['content_type'] ?? 'blog_post';
323 $target_audience = $params['target_audience'] ?? 'general';
324 $content_length = $params['content_length'] ?? 'medium';
325 $tone = $params['tone'] ?? 'professional';
326 $competitor_urls = $params['competitor_urls'] ?? [];
327 $additional_context = $params['additional_context'] ?? '';
328 // Write the brief in the site (or related post's) language rather than
329 // defaulting to English on non-English sites (issue #234).
330 $language = \ThinkRank\AI\Language_Resolver::resolve((int) ($params['post_id'] ?? 0));
331
332 // Analyze competitor URLs if provided
333 $competitor_analysis = '';
334 if (!empty($competitor_urls)) {
335 $competitor_analysis = $this->analyze_competitor_urls($competitor_urls);
336 }
337
338 // Build AI prompt using shared Prompt Builder
339 $prompt_builder = $this->get_prompt_builder();
340 $prompt = $prompt_builder->build_content_brief_prompt(
341 $target_keywords,
342 $content_type,
343 $target_audience,
344 $content_length,
345 $tone,
346 $competitor_analysis,
347 $additional_context,
348 $this->get_current_provider(),
349 $language
350 );
351
352 try {
353 // Get the model-aware budget for a comprehensive brief, then scale
354 // it to the requested content length so Short/Medium/Long actually
355 // request different budgets (issue #287). Every client (OpenAI,
356 // Claude, Gemini, OpenRouter) implements get_recommended_tokens(),
357 // so there is no model-blind fallback.
358 $base_tokens = (int) $this->ai_client->get_recommended_tokens('content_brief');
359 $max_tokens = $this->scale_tokens_for_length($base_tokens, $content_length);
360
361 // Bound hidden reasoning on the GPT-5 family (issue #286). See
362 // resolve_reasoning_effort(). Only the GPT-5 family reads this;
363 // o1/o3, gpt-4o and the non-OpenAI clients ignore the option, and
364 // an empty string opts out (model default effort).
365 $reasoning_effort = $this->resolve_reasoning_effort($this->get_current_model(), $params);
366
367 $completion_options = [
368 // For GPT‑5 family the client translates max_tokens to
369 // max_completion_tokens internally. Temperature is intentionally
370 // omitted: every client defaults it to 0.7, and reasoning models
371 // reject it outright, so passing it here was misleading no-op.
372 'max_tokens' => $max_tokens,
373 // The brief is one JSON object. Only a compatible endpoint
374 // with JSON mode on reads this; every other client ignores it.
375 'json_object' => true,
376 ];
377 if ('' !== $reasoning_effort) {
378 $completion_options['reasoning_effort'] = $reasoning_effort;
379 }
380
381 // Generate brief using AI
382 $ai_response = $this->ai_client->generate_completion($prompt, $completion_options);
383
384 // Detect a provider-side non-answer (refusal, policy block, or
385 // truncation) BEFORE attempting text extraction. Otherwise a
386 // refusal — which OpenAI returns as HTTP 200 with content=null —
387 // slips past every isset() branch and gets serialized into the
388 // brief body instead of being reported to the user.
389 $this->guard_against_non_answer($ai_response, $max_tokens);
390
391 // Extract text content from AI response
392 $ai_text = '';
393
394 // Handle OpenAI response format
395 if (isset($ai_response['choices'][0]['message']['content'])) {
396 $content_field = $ai_response['choices'][0]['message']['content'];
397 if (is_string($content_field)) {
398 $ai_text = $content_field;
399 } elseif (is_array($content_field)) {
400 // Concatenate text parts from array-based content (Chat Completions multimodal)
401 $parts = array_map(function($part) {
402 if (is_array($part)) {
403 return $part['text'] ?? '';
404 }
405 return is_string($part) ? $part : '';
406 }, $content_field);
407 $ai_text = trim(implode("\n", array_filter($parts)));
408 }
409 }
410 // Handle Claude response format
411 elseif (isset($ai_response['content'][0]['text'])) {
412 $ai_text = $ai_response['content'][0]['text'];
413 }
414 // Handle Gemini response format
415 elseif (isset($ai_response['candidates'][0]['content']['parts'][0]['text'])) {
416 $ai_text = $ai_response['candidates'][0]['content']['parts'][0]['text'];
417 }
418 // Handle direct content field
419 elseif (isset($ai_response['content']) && is_string($ai_response['content'])) {
420 $ai_text = $ai_response['content'];
421 }
422 // Handle direct string response
423 elseif (is_string($ai_response)) {
424 $ai_text = $ai_response;
425 }
426 // No known provider shape matched and guard_against_non_answer()
427 // found nothing it recognised. Never serialize the raw envelope
428 // into the brief body — that turns a clear failure into a saved,
429 // meaningless brief. Log the shape for diagnostics and fail.
430 else {
431 if (defined('WP_DEBUG') && WP_DEBUG) {
432 $shape = is_array($ai_response) ? implode(', ', array_keys($ai_response)) : gettype($ai_response);
433 // phpcs:ignore WordPress.PHP.DevelopmentFunctions.error_log_error_log -- Debug logging only when WP_DEBUG is enabled.
434 error_log('[ThinkRank] Content brief: unrecognised AI response shape. Top-level keys: ' . $shape);
435 }
436 throw new \Exception('The AI returned a response in an unexpected format. Please try again.');
437 }
438
439 // Ensure we have actual text content
440 if (empty(trim($ai_text))) {
441 throw new \Exception('AI response was empty or contained no text content.');
442 }
443
444 // Extract token usage for analytics tracking
445 $tokens_used = $this->extract_token_usage($ai_response);
446
447 // Parse and structure the response
448 $brief_data = $this->parse_ai_response($ai_text, $params);
449
450 // Extract actual model from response before using it
451 $actual_model = $this->extract_model_from_response($ai_response);
452
453 // Add generation metadata (use actual model from response if available)
454 $brief_data['generation_meta'] = [
455 'provider' => $this->get_current_provider(),
456 'model' => $actual_model ?: $this->get_current_model(),
457 'generated_at' => current_time('mysql'),
458 'version' => '1.0'
459 ];
460
461 // Save brief to database
462 $brief_id = $this->save_brief($brief_data);
463 $brief_data['id'] = $brief_id;
464
465 // Log AI usage for analytics tracking (including raw response and actual model used)
466 $usage_id = $this->log_ai_usage(get_current_user_id(), 'Content Brief', $tokens_used, $brief_id, $ai_text, $actual_model);
467
468 // Set raw response for immediate display
469 $brief_data['raw_response'] = $ai_text;
470
471 // Apply normalization for React compatibility
472 $brief_data = $this->normalize_brief_data($brief_data);
473
474 return $brief_data;
475
476 } catch (\Exception $e) {
477 // Provide more specific error messages
478 $error_message = $e->getMessage();
479
480 // Messages we authored for the user (refusals, policy blocks,
481 // token-limit truncation, unexpected shape) all start with "The AI "
482 // and are already actionable. Pass them through verbatim instead of
483 // flattening them via the substring matching below — e.g. so a
484 // refusal is not rewritten into generic "empty content" advice.
485 if (strpos($error_message, 'The AI ') === 0) {
486 throw new \Exception(esc_html($error_message));
487 }
488
489 if (strpos($error_message, 'API key') !== false) {
490 throw new \Exception('API key configuration error. Please check your AI provider settings.');
491 } elseif (strpos($error_message, 'Invalid AI response format') !== false) {
492 throw new \Exception('AI service returned an unexpected response format. Please try again.');
493 } elseif (strpos($error_message, 'empty') !== false) {
494 throw new \Exception('AI service returned empty content. Please try again with different parameters.');
495 } else {
496 throw new \Exception('Failed to generate content brief: ' . esc_html($error_message));
497 }
498 }
499 }
500
501 /**
502 * Scale the model-aware brief budget to the requested content length.
503 *
504 * get_recommended_tokens('content_brief') returns the budget for a full,
505 * comprehensive (Long) brief, already capped at the model's completion
506 * ceiling. Shorter tiers request proportionally less so that choosing Short
507 * is genuinely faster and cheaper (issue #287), while every tier stays at or
508 * below the base and at or above MIN_BRIEF_TOKENS so it cannot truncate.
509 *
510 * @param int $base_tokens Model-aware budget for a comprehensive brief.
511 * @param string $content_length One of 'short' | 'medium' | 'long'.
512 * @return int Scaled max_tokens, clamped to [floor, base_tokens].
513 */
514 private function scale_tokens_for_length(int $base_tokens, string $content_length): int {
515 // Unknown/missing length falls back to the medium tier — never to 0 or
516 // to the raw ceiling.
517 $multiplier = self::LENGTH_TOKEN_MULTIPLIERS[$content_length]
518 ?? self::LENGTH_TOKEN_MULTIPLIERS['medium'];
519
520 $scaled = (int) round($base_tokens * $multiplier);
521
522 // The floor can never exceed the base itself, so a model with a tiny
523 // ceiling still yields a sane, in-range value.
524 $floor = (int) min($base_tokens, self::MIN_BRIEF_TOKENS);
525
526 return max($floor, min($scaled, $base_tokens));
527 }
528
529 /**
530 * Detect a provider-side non-answer and fail with the real reason.
531 *
532 * A refusal, content-policy block, or token-limit truncation is not a
533 * usable brief. Each provider signals these differently, and none of the
534 * signals set the content field the extraction chain looks for — so if we
535 * don't catch them here they fall through to the "unexpected format" path
536 * (or, historically, were serialized into the brief body). All messages
537 * start with "The AI " so the outer catch passes them through unchanged.
538 *
539 * @param mixed $ai_response Raw response from the AI client.
540 * @param int $requested_tokens The max_tokens this request asked for; 0 when unknown.
541 * @throws \Exception If the response is a refusal, policy block, or truncation.
542 */
543 private function guard_against_non_answer($ai_response, int $requested_tokens = 0): void {
544 if (!is_array($ai_response)) {
545 return;
546 }
547
548 // --- OpenAI (Chat Completions) ---
549 // A structured refusal is HTTP 200 with message.content=null and the
550 // stated reason carried in message.refusal. finish_reason distinguishes
551 // a policy block from a truncated completion.
552 if (isset($ai_response['choices'][0]['message'])) {
553 $message = $ai_response['choices'][0]['message'];
554 $finish = (string) ($ai_response['choices'][0]['finish_reason'] ?? '');
555
556 if (!empty($message['refusal'])) {
557 throw new \Exception(esc_html(sprintf(
558 'The AI declined to generate this brief: %s',
559 (string) $message['refusal']
560 )));
561 }
562 if ('content_filter' === $finish) {
563 throw new \Exception('The AI blocked this request under its content policy. Try a different topic or less sensitive keywords.');
564 }
565 // A self-hosted server can stop short of max_tokens because the
566 // prompt and the answer together filled its context window
567 // (Ollama loads models at 4096 by default). A bigger output budget
568 // cannot fix that, so say what can.
569 $completion_tokens = (int) ($ai_response['usage']['completion_tokens'] ?? 0);
570 if ('length' === $finish && $requested_tokens > 0 && $completion_tokens > 0 && $completion_tokens < $requested_tokens) {
571 throw new \Exception(esc_html(sprintf(
572 'The AI stopped after %1$d tokens, short of the %2$d allowed, because the server ran out of context window before finishing the brief. Raise the context length on your AI server (for Ollama, set OLLAMA_CONTEXT_LENGTH to 16384 or more) and try again.',
573 $completion_tokens,
574 $requested_tokens
575 )));
576 }
577 if ('length' === $finish) {
578 throw new \Exception('The AI stopped at its output token limit before finishing the brief. Try fewer competitor URLs, or a model with a larger output limit. A shorter content length will not help: it asks for a smaller budget, not a smaller answer.');
579 }
580 }
581
582 // --- Claude (Messages) ---
583 if (isset($ai_response['stop_reason'])) {
584 $stop_reason = (string) $ai_response['stop_reason'];
585 if ('refusal' === $stop_reason) {
586 throw new \Exception('The AI declined to generate this brief for this topic. Try a different topic or less sensitive keywords.');
587 }
588 if ('max_tokens' === $stop_reason) {
589 throw new \Exception('The AI stopped at its output token limit before finishing the brief. Try fewer competitor URLs, or a model with a larger output limit. A shorter content length will not help: it asks for a smaller budget, not a smaller answer.');
590 }
591 }
592
593 // --- Gemini ---
594 // A prompt rejected outright returns no candidate at all, only
595 // promptFeedback.blockReason; a candidate can also finish on SAFETY or
596 // PROHIBITED_CONTENT, or be truncated at MAX_TOKENS.
597 $block_reason = (string) ($ai_response['promptFeedback']['blockReason'] ?? '');
598 if ('' !== $block_reason) {
599 throw new \Exception(esc_html(sprintf(
600 'The AI blocked this request under its content policy (%s). Try a different topic or less sensitive keywords.',
601 $block_reason
602 )));
603 }
604 $gemini_finish = (string) ($ai_response['candidates'][0]['finishReason'] ?? '');
605 if (in_array($gemini_finish, ['SAFETY', 'PROHIBITED_CONTENT'], true)) {
606 throw new \Exception('The AI blocked this request under its content policy. Try a different topic or less sensitive keywords.');
607 }
608 if ('MAX_TOKENS' === $gemini_finish) {
609 throw new \Exception('The AI stopped at its output token limit before finishing the brief. Try fewer competitor URLs, or a model with a larger output limit. A shorter content length will not help: it asks for a smaller budget, not a smaller answer.');
610 }
611 }
612
613 /**
614 * Validate brief generation parameters
615 *
616 * @param array $params Parameters to validate
617 * @throws \Exception If validation fails
618 */
619 private function validate_brief_params(array $params): void {
620 if (empty($params['target_keywords']) || !is_array($params['target_keywords'])) {
621 throw new \Exception('Target keywords are required and must be an array.');
622 }
623
624 $valid_content_types = ['blog_post', 'product_page', 'landing_page', 'tutorial'];
625 if (!empty($params['content_type']) && !in_array($params['content_type'], $valid_content_types, true)) {
626 throw new \Exception('Invalid content type specified.');
627 }
628
629 $valid_lengths = ['short', 'medium', 'long'];
630 if (!empty($params['content_length']) && !in_array($params['content_length'], $valid_lengths, true)) {
631 throw new \Exception('Invalid content length specified.');
632 }
633
634 $valid_tones = ['professional', 'casual', 'technical', 'friendly'];
635 if (!empty($params['tone']) && !in_array($params['tone'], $valid_tones, true)) {
636 throw new \Exception('Invalid tone specified.');
637 }
638 }
639
640 /**
641 * Parse AI response into structured data
642 *
643 * @param string $ai_response Raw AI response
644 * @param array $original_params Original generation parameters
645 * @return array Structured brief data
646 */
647 private function parse_ai_response(string $ai_response, array $original_params): array {
648 $json_data = $this->parse_json_response($ai_response);
649
650 if (null === $json_data) {
651 // JSON parsing failed - return error structure
652 return $this->create_parsing_error_response($ai_response, $original_params);
653 }
654
655 return $this->structure_json_data($json_data, $original_params);
656 }
657
658 /**
659 * Parse JSON response from AI
660 *
661 * @param string $ai_response Raw AI response
662 * @return array|null Parsed JSON data or null if parsing fails
663 */
664 private function parse_json_response(string $ai_response): ?array {
665 // Clean the response - remove any text before/after JSON
666 $ai_response = trim($ai_response);
667
668 // Handle markdown code blocks (```json ... ```)
669 if (preg_match('/```(?:json)?\s*\n?(.*?)\n?```/s', $ai_response, $matches)) {
670 $json_string = trim($matches[1]);
671 } else {
672 // Find JSON object boundaries
673 $start = strpos($ai_response, '{');
674 $end = strrpos($ai_response, '}');
675
676 if (false === $start || false === $end || $start >= $end) {
677 return null;
678 }
679
680 $json_string = substr($ai_response, $start, $end - $start + 1);
681 }
682
683 $json_data = json_decode($json_string, true);
684
685 if (json_last_error() !== JSON_ERROR_NONE) {
686 return null;
687 }
688
689 return $json_data;
690 }
691
692 /**
693 * Structure JSON data into expected format
694 *
695 * @param array $json_data Parsed JSON data
696 * @param array $original_params Original generation parameters
697 * @return array Structured brief data
698 */
699 private function structure_json_data(array $json_data, array $original_params): array {
700 return [
701 'title' => $json_data['title_suggestions'] ?? [],
702 'meta_description' => $json_data['meta_descriptions'][0] ?? '',
703 'meta_descriptions' => $json_data['meta_descriptions'] ?? [],
704 'url_slugs' => $json_data['url_slugs'] ?? [],
705 'outline' => self::strip_outline_level_labels($json_data['outline'] ?? []),
706 'seo_recommendations' => [
707 'title_suggestions' => $json_data['title_suggestions'] ?? [],
708 'meta_description' => $json_data['meta_descriptions'][0] ?? '',
709 'meta_descriptions' => $json_data['meta_descriptions'] ?? [],
710 'url_slugs' => $json_data['url_slugs'] ?? [],
711 'focus_keyword_analysis' => $this->normalize_focus_keyword_analysis($json_data['focus_keyword_analysis'] ?? []),
712 'internal_links' => $json_data['internal_linking'] ?? [],
713 'related_keywords' => $json_data['related_keywords'] ?? [],
714 'long_tail_keywords' => []
715 ],
716 'social_media' => $json_data['social_media'] ?? [
717 'open_graph' => ['title' => '', 'description' => ''],
718 'twitter_card' => ['title' => '', 'description' => '']
719 ],
720 'schema_markup' => $json_data['schema_markup'] ?? [
721 'recommended_types' => [],
722 'key_properties' => [],
723 'faq_questions' => []
724 ],
725 'visual_content' => $json_data['visual_content'] ?? [
726 'image_recommendations' => [],
727 'alt_text_suggestions' => [],
728 'infographic_opportunities' => []
729 ],
730 'competitor_gaps' => $json_data['competitor_analysis']['content_gaps'] ?? [],
731 'call_to_actions' => $json_data['call_to_actions'] ?? [],
732 'writing_guidelines' => $json_data['writing_guidelines'] ?? [],
733 'content_body' => self::strip_heading_level_labels((string) ($json_data['content_body'] ?? '')),
734 'estimated_word_count' => $this->get_word_count_estimate($original_params['content_length'] ?? 'medium'),
735 'raw_response' => '', // Will be retrieved from ai_usage table
736 'generation_params' => $original_params,
737 'parsing_status' => 'success',
738 'created_at' => current_time('mysql')
739 ];
740 }
741
742 /**
743 * Remove a leading level label from a heading string.
744 *
745 * The prompt's own JSON example labelled outline headings with their level
746 * (`"heading": "H1: Main Title"` next to a separate `"level": 1`), so the
747 * model often carried the convention into the drafted article and Pro's
748 * "Insert into post" wrote `<h2>H2: Real Heading</h2>` into published
749 * content. The prompt no longer does that, but a prompt change never fully
750 * binds a model — so the label is stripped here too (#410).
751 *
752 * Covers the label forms a model actually emits: `H2:`, `h3:`, `H2 -`,
753 * `H4.`, `H2)` and the en/em dash variants, optionally wrapped in markdown
754 * emphasis (`**H2:**`). The delimiter is anchored directly after the digit
755 * so `H10:` — a plausible heading in a numbered list — is left alone, and
756 * only a leading label is matched so body copy that mentions a level
757 * survives. Trailing emphasis is consumed only when the same marker opened
758 * the label, so `H2: *emphasised start*` keeps its asterisks.
759 *
760 * @since 2.0.1
761 *
762 * @param string $heading Heading text.
763 * @return string Heading without its level prefix.
764 */
765 public static function strip_level_label(string $heading): string {
766 // En dash and em dash as raw UTF-8 bytes, so the pattern needs no /u
767 // modifier and cannot blank a heading that is not valid UTF-8.
768 $delimiter = '(?:[:.)\-]|\xe2\x80\x93|\xe2\x80\x94)';
769 $emphasis = '(\*{1,3}|_{1,3})';
770
771 $pattern = '/^\s*(?:'
772 . $emphasis . '\s*[Hh][1-6]\s*' . $delimiter . '\s*\1'
773 . '|[Hh][1-6]\s*' . $delimiter
774 . ')\s*/';
775
776 return (string) preg_replace($pattern, '', $heading);
777 }
778
779 /**
780 * Strip a level label from a heading's inner HTML.
781 *
782 * A model drafting publish-ready HTML often wraps the heading text in an
783 * inline tag (`<h2><strong>H2: Real Heading</strong></h2>`). That pushes a
784 * `<` in front of the label, so the leading run of inline opening tags is
785 * set aside and re-attached around the cleaned text.
786 *
787 * @since 2.0.1
788 *
789 * @param string $inner Heading inner HTML.
790 * @return string Inner HTML without the level prefix.
791 */
792 private static function strip_inner_level_label(string $inner): string {
793 $prefix = '';
794
795 if (preg_match('/^(\s*(?:<(?:strong|em|b|i|span|mark|code|u)\b[^>]*>\s*)+)(.*)$/is', $inner, $parts)) {
796 $prefix = $parts[1];
797 $inner = $parts[2];
798 }
799
800 return $prefix . self::strip_level_label($inner);
801 }
802
803 /**
804 * Strip level labels from every heading in an outline.
805 *
806 * @since 2.0.1
807 *
808 * @param mixed $outline Outline as returned by the model.
809 * @return array Outline with clean headings.
810 */
811 public static function strip_outline_level_labels($outline): array {
812 if (!is_array($outline)) {
813 return [];
814 }
815
816 foreach ($outline as $index => $section) {
817 if (is_array($section) && isset($section['heading']) && is_string($section['heading'])) {
818 $outline[$index]['heading'] = self::strip_level_label($section['heading']);
819 } elseif (is_string($section)) {
820 $outline[$index] = self::strip_level_label($section);
821 }
822 }
823
824 return $outline;
825 }
826
827 /**
828 * Strip level labels from the heading text inside drafted HTML.
829 *
830 * This is the path that reaches published post content, so it is the one
831 * that matters most. Only the text directly inside an <h1>-<h6> is touched.
832 *
833 * @since 2.0.1
834 *
835 * @param string $html Drafted article body.
836 * @return string Body with clean headings.
837 */
838 public static function strip_heading_level_labels(string $html): string {
839 if ('' === $html || false === stripos($html, '<h')) {
840 return $html;
841 }
842
843 return (string) preg_replace_callback(
844 '/(<h([1-6])\b[^>]*>)(.*?)(<\/h\2>)/is',
845 static function (array $parts): string {
846 return $parts[1] . self::strip_inner_level_label($parts[3]) . $parts[4];
847 },
848 $html
849 );
850 }
851
852 /**
853 * Create error response when JSON parsing fails
854 *
855 * @param string $ai_response Raw AI response
856 * @param array $original_params Original generation parameters
857 * @return array Error response structure
858 */
859 private function create_parsing_error_response(string $ai_response, array $original_params): array {
860 return [
861 'title' => ['Error: Unable to parse AI response'],
862 'meta_description' => 'AI response could not be parsed as valid JSON.',
863 'meta_descriptions' => ['AI response could not be parsed as valid JSON.'],
864 'url_slugs' => ['error-parsing-response'],
865 'outline' => [],
866 'seo_recommendations' => [
867 'title_suggestions' => ['Error: Unable to parse AI response'],
868 'meta_description' => 'AI response could not be parsed as valid JSON.',
869 'meta_descriptions' => ['AI response could not be parsed as valid JSON.'],
870 'url_slugs' => ['error-parsing-response'],
871 'focus_keyword_analysis' => [
872 'primary_placement' => [],
873 'secondary_integration' => [],
874 'density_guidelines' => []
875 ],
876 'internal_links' => [],
877 'related_keywords' => [],
878 'long_tail_keywords' => []
879 ],
880 'social_media' => [
881 'open_graph' => ['title' => 'Error', 'description' => 'Parsing failed'],
882 'twitter_card' => ['title' => 'Error', 'description' => 'Parsing failed']
883 ],
884 'schema_markup' => [
885 'recommended_types' => [],
886 'key_properties' => [],
887 'faq_questions' => []
888 ],
889 'visual_content' => [
890 'image_recommendations' => [],
891 'alt_text_suggestions' => [],
892 'infographic_opportunities' => []
893 ],
894 'competitor_gaps' => [],
895 'call_to_actions' => [],
896 'writing_guidelines' => [],
897 'content_body' => '',
898 'estimated_word_count' => $this->get_word_count_estimate($original_params['content_length'] ?? 'medium'),
899 'raw_response' => $ai_response, // Store raw response in error case
900 'generation_params' => $original_params,
901 'parsing_status' => 'failed',
902 'created_at' => current_time('mysql')
903 ];
904 }
905
906 /**
907 * Normalize focus keyword analysis to ensure proper array structure
908 *
909 * @param array $focus_keyword_analysis Raw focus keyword analysis data
910 * @return array Normalized focus keyword analysis
911 */
912 private function normalize_focus_keyword_analysis(array $focus_keyword_analysis): array {
913 $normalized = [
914 'primary_placement' => [],
915 'secondary_integration' => [],
916 'density_guidelines' => []
917 ];
918
919 // Normalize primary_placement
920 if (isset($focus_keyword_analysis['primary_placement'])) {
921 if (is_array($focus_keyword_analysis['primary_placement'])) {
922 $normalized['primary_placement'] = $focus_keyword_analysis['primary_placement'];
923 } elseif (is_string($focus_keyword_analysis['primary_placement'])) {
924 // Convert string to array by splitting on common delimiters
925 $normalized['primary_placement'] = array_filter(array_map('trim', preg_split('/[,;]/', $focus_keyword_analysis['primary_placement'])));
926 }
927 }
928
929 // Normalize secondary_integration - this is the problematic field
930 if (isset($focus_keyword_analysis['secondary_integration'])) {
931 if (is_array($focus_keyword_analysis['secondary_integration'])) {
932 $normalized['secondary_integration'] = $focus_keyword_analysis['secondary_integration'];
933 } elseif (is_string($focus_keyword_analysis['secondary_integration'])) {
934 // Convert string to array - split by sentences or use as single item
935 $text = trim($focus_keyword_analysis['secondary_integration']);
936 if (!empty($text)) {
937 // Split by sentences if it contains periods, otherwise use as single item
938 if (strpos($text, '.') !== false) {
939 $sentences = array_filter(array_map('trim', explode('.', $text)));
940 $normalized['secondary_integration'] = array_map(function($sentence) {
941 return $sentence . (substr($sentence, -1) !== '.' ? '.' : '');
942 }, $sentences);
943 } else {
944 $normalized['secondary_integration'] = [$text];
945 }
946 }
947 }
948 }
949
950 // Normalize density_guidelines
951 if (isset($focus_keyword_analysis['density_guidelines'])) {
952 if (is_array($focus_keyword_analysis['density_guidelines'])) {
953 $normalized['density_guidelines'] = $focus_keyword_analysis['density_guidelines'];
954 } elseif (is_string($focus_keyword_analysis['density_guidelines'])) {
955 // Convert string to array by splitting on common delimiters
956 $guidelines = array_filter(array_map('trim', preg_split('/[,;]/', $focus_keyword_analysis['density_guidelines'])));
957 $normalized['density_guidelines'] = $guidelines ?: [$focus_keyword_analysis['density_guidelines']];
958 }
959 }
960
961 return $normalized;
962 }
963 /**
964 * The URL to check and fetch for a competitor URL the user entered, or
965 * null when it is not a valid URL.
966 *
967 * A competitor page on an internationalised domain, or with a Bengali or
968 * Arabic slug, is a valid URL, so the syntax check is Url_Validator's.
969 * What comes back is the ASCII form (punycode host, percent-encoded path),
970 * and both the SSRF guard and the fetch use it: the guard cannot resolve
971 * a Unicode host name, and it must check exactly the URL that is then
972 * requested.
973 *
974 * @since 2.14.2
975 *
976 * @param string $url Trimmed URL as entered.
977 * @return string|null ASCII URL, or null when invalid.
978 */
979 private function competitor_fetch_url(string $url): ?string {
980 if ('' === $url || !\ThinkRank\Core\Url_Validator::is_valid($url)) {
981 return null;
982 }
983
984 return \ThinkRank\Core\Url_Validator::to_ascii($url);
985 }
986
987 private function analyze_competitor_urls(array $urls): string {
988 $analysis_results = [];
989 $failed_urls = [];
990
991 // Limit to first 3 URLs to prevent timeout
992 $urls = array_slice($urls, 0, 3);
993
994 foreach ($urls as $url) {
995 $url = trim($url);
996 $fetch_url = $this->competitor_fetch_url($url);
997 if (null === $fetch_url) {
998 $failed_urls[] = $url . " (invalid URL)";
999 continue;
1000 }
1001
1002 // SSRF guard: only fetch public http/https hosts. Blocks loopback,
1003 // link-local (cloud metadata), private and reserved ranges before any
1004 // request is made.
1005 if (!$this->is_safe_public_url($fetch_url)) {
1006 $failed_urls[] = $url . " (blocked: non-public host)";
1007 continue;
1008 }
1009
1010 $content_data = $this->scrape_competitor_content($fetch_url);
1011 if ($content_data) {
1012 $analysis_results[] = $this->format_competitor_analysis($url, $content_data);
1013 } else {
1014 $failed_urls[] = $url . " (failed to scrape)";
1015 }
1016 }
1017
1018 $result = "";
1019
1020 if (!empty($analysis_results)) {
1021 $result .= implode("\n\n", $analysis_results);
1022 }
1023
1024 if (!empty($failed_urls)) {
1025 $result .= "\n\nNote: The following URLs could not be analyzed:\n";
1026 $result .= "- " . implode("\n- ", $failed_urls);
1027 }
1028
1029 if (empty($analysis_results)) {
1030 return "No competitor URLs could be successfully analyzed. Please ensure URLs are accessible and valid.";
1031 }
1032
1033 return $result;
1034 }
1035
1036 /**
1037 * Whether a competitor URL is safe to fetch: an http/https URL whose host
1038 * resolves only to public IP addresses.
1039 *
1040 * Prevents SSRF — a low-privilege user could otherwise point competitor
1041 * scraping at loopback, link-local (e.g. 169.254.169.254 cloud metadata),
1042 * CGNAT, private or reserved addresses to probe internal services. The
1043 * block list lives in {@see \ThinkRank\Core\Url_Safety} so this and the
1044 * schema importer can never drift apart.
1045 *
1046 * @param string $url URL to validate.
1047 * @return bool True when safe to fetch.
1048 */
1049 private function is_safe_public_url(string $url): bool {
1050 return \ThinkRank\Core\Url_Safety::is_safe_public_url($url);
1051 }
1052
1053 /**
1054 * Scrape content from a competitor URL
1055 *
1056 * @param string $url The URL to scrape
1057 * @return array|null Content data or null if failed
1058 */
1059 private function scrape_competitor_content(string $url): ?array {
1060 // Url_Safety::safe_remote_get() follows redirects manually and re-checks
1061 // the resolved host on every hop, so a target that redirects to (or
1062 // rebinds onto) an internal address after the pre-flight check is
1063 // refused rather than fetched.
1064 $response = \ThinkRank\Core\Url_Safety::safe_remote_get($url, [
1065 'timeout' => 8, // Reduced from 15 to 8 seconds
1066 'user-agent' => 'Mozilla/5.0 (compatible; ThinkRank SEO Bot)',
1067 'headers' => [
1068 'Accept' => 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
1069 'Accept-Language' => 'en-US,en;q=0.5',
1070 ]
1071 ]);
1072
1073 if (is_wp_error($response)) {
1074 return null;
1075 }
1076
1077 $status_code = wp_remote_retrieve_response_code($response);
1078 if ($status_code !== 200) {
1079 return null;
1080 }
1081
1082 $html = wp_remote_retrieve_body($response);
1083 if (empty($html)) {
1084 return null;
1085 }
1086
1087 return $this->parse_html_content($html, $url);
1088 }
1089
1090 /**
1091 * Parse HTML content and extract key SEO elements
1092 *
1093 * @param string $html HTML content
1094 * @param string $url Original URL for context
1095 * @return array Parsed content data
1096 */
1097 private function parse_html_content(string $html, string $url): array {
1098 // Create DOMDocument to parse HTML
1099 $dom = new \DOMDocument();
1100
1101 // Suppress warnings for malformed HTML
1102 libxml_use_internal_errors(true);
1103 $dom->loadHTML('<?xml encoding="UTF-8">' . $html);
1104 libxml_clear_errors();
1105
1106 $xpath = new \DOMXPath($dom);
1107
1108 // Extract title
1109 $title_nodes = $xpath->query('//title');
1110 $title = $title_nodes->length > 0 ? trim($title_nodes->item(0)->textContent) : '';
1111
1112 // Extract meta description
1113 $meta_desc_nodes = $xpath->query('//meta[@name="description"]/@content');
1114 $meta_description = $meta_desc_nodes->length > 0 ? trim($meta_desc_nodes->item(0)->textContent) : '';
1115
1116 // Extract headings (H1-H6)
1117 $headings = [];
1118 for ($i = 1; $i <= 6; $i++) {
1119 $heading_nodes = $xpath->query("//h{$i}");
1120 foreach ($heading_nodes as $node) {
1121 $text = trim($node->textContent);
1122 if (!empty($text)) {
1123 $headings["h{$i}"][] = $text;
1124 }
1125 }
1126 }
1127
1128 // Extract body text and calculate word count
1129 $body_nodes = $xpath->query('//body');
1130 $body_text = '';
1131 if ($body_nodes->length > 0) {
1132 $body_text = $this->extract_clean_text($body_nodes->item(0));
1133 }
1134
1135 $word_count = str_word_count($body_text);
1136
1137 // Extract meta keywords if present
1138 $meta_keywords_nodes = $xpath->query('//meta[@name="keywords"]/@content');
1139 $meta_keywords = $meta_keywords_nodes->length > 0 ? trim($meta_keywords_nodes->item(0)->textContent) : '';
1140
1141 // Extract internal links count
1142 $internal_links = $xpath->query('//a[starts-with(@href, "/") or contains(@href, "' . wp_parse_url($url, PHP_URL_HOST) . '")]');
1143 $internal_link_count = $internal_links->length;
1144
1145 // Extract external links count
1146 $external_links = $xpath->query('//a[starts-with(@href, "http") and not(contains(@href, "' . wp_parse_url($url, PHP_URL_HOST) . '"))]');
1147 $external_link_count = $external_links->length;
1148
1149 // Extract images count and alt text analysis
1150 $images = $xpath->query('//img');
1151 $image_count = $images->length;
1152 $images_with_alt = $xpath->query('//img[@alt and @alt!=""]');
1153 $images_with_alt_count = $images_with_alt->length;
1154
1155 // Extract schema markup
1156 $schema_scripts = $xpath->query('//script[@type="application/ld+json"]');
1157 $has_schema = $schema_scripts->length > 0;
1158
1159 // Extract last modified date if available
1160 $last_modified_nodes = $xpath->query('//meta[@name="last-modified"]/@content | //meta[@property="article:modified_time"]/@content');
1161 $last_modified = $last_modified_nodes->length > 0 ? $last_modified_nodes->item(0)->textContent : '';
1162
1163 // Calculate readability metrics
1164 $readability_score = $this->calculate_readability_score($body_text);
1165
1166 // Extract keyword density for target keywords (if provided)
1167 $keyword_density = $this->analyze_keyword_density($body_text, $title);
1168
1169 // Detect content freshness indicators
1170 $freshness_indicators = $this->detect_freshness_indicators($html, $body_text);
1171
1172 return [
1173 'url' => $url,
1174 'title' => $title,
1175 'meta_description' => $meta_description,
1176 'meta_keywords' => $meta_keywords,
1177 'headings' => $headings,
1178 'word_count' => $word_count,
1179 'internal_links' => $internal_link_count,
1180 'external_links' => $external_link_count,
1181 'images' => [
1182 'total' => $image_count,
1183 'with_alt' => $images_with_alt_count,
1184 'alt_ratio' => $image_count > 0 ? round(($images_with_alt_count / $image_count) * 100, 1) : 0
1185 ],
1186 'seo' => [
1187 'has_schema' => $has_schema,
1188 'title_length' => strlen($title),
1189 'meta_desc_length' => strlen($meta_description),
1190 'title_score' => $this->score_title_seo($title),
1191 'meta_desc_score' => $this->score_meta_description($meta_description)
1192 ],
1193 'content_quality' => [
1194 'readability_score' => $readability_score,
1195 'keyword_density' => $keyword_density,
1196 'freshness_indicators' => $freshness_indicators,
1197 'content_depth' => $this->assess_content_depth($headings, $word_count)
1198 ],
1199 'last_modified' => $last_modified,
1200 'content_preview' => substr($body_text, 0, 500) . '...',
1201 'analysis_timestamp' => current_time('mysql')
1202 ];
1203 }
1204
1205 /**
1206 * Extract clean text from DOM node, removing scripts and styles
1207 *
1208 * @param \DOMNode $node DOM node to extract text from
1209 * @return string Clean text content
1210 */
1211 private function extract_clean_text(\DOMNode $node): string {
1212 // Remove script and style elements
1213 $xpath = new \DOMXPath($node->ownerDocument);
1214 $scripts = $xpath->query('.//script | .//style', $node);
1215
1216 foreach ($scripts as $script) {
1217 $script->parentNode->removeChild($script);
1218 }
1219
1220 // Get text content and clean it up
1221 $text = $node->textContent;
1222
1223 // Remove extra whitespace and normalize
1224 $text = preg_replace('/\s+/', ' ', $text);
1225 $text = trim($text);
1226
1227 return $text;
1228 }
1229
1230 /**
1231 * Format competitor analysis for AI prompt
1232 *
1233 * @param string $url Competitor URL
1234 * @param array $content_data Parsed content data
1235 * @return string Formatted analysis
1236 */
1237 private function format_competitor_analysis(string $url, array $content_data): string {
1238 $analysis = "=== COMPETITOR ANALYSIS ===\n";
1239 $analysis .= "URL: {$url}\n";
1240 $analysis .= "Title: {$content_data['title']} (Length: {$content_data['seo']['title_length']} chars, Score: {$content_data['seo']['title_score']['grade']})\n";
1241
1242 if (!empty($content_data['meta_description'])) {
1243 $analysis .= "Meta Description: {$content_data['meta_description']} (Length: {$content_data['seo']['meta_desc_length']} chars, Score: {$content_data['seo']['meta_desc_score']['grade']})\n";
1244 }
1245
1246 $analysis .= "\nCONTENT METRICS:\n";
1247 $analysis .= "- Word Count: {$content_data['word_count']} words\n";
1248 $analysis .= "- Content Depth: {$content_data['content_quality']['content_depth']['level']} (Score: {$content_data['content_quality']['content_depth']['score']}/100)\n";
1249 $analysis .= "- Readability: {$content_data['content_quality']['readability_score']['level']} (Score: {$content_data['content_quality']['readability_score']['score']}/100)\n";
1250 $analysis .= "- Internal Links: {$content_data['internal_links']}\n";
1251 $analysis .= "- External Links: {$content_data['external_links']}\n";
1252 $analysis .= "- Images: {$content_data['images']['total']} total, {$content_data['images']['with_alt']} with alt text ({$content_data['images']['alt_ratio']}%)\n";
1253
1254 // Add heading structure
1255 if (!empty($content_data['headings'])) {
1256 $analysis .= "\nCONTENT STRUCTURE:\n";
1257 foreach ($content_data['headings'] as $level => $headings) {
1258 $analysis .= "- " . strtoupper($level) . " (" . count($headings) . "): " . implode(', ', array_slice($headings, 0, 3));
1259 if (count($headings) > 3) {
1260 $analysis .= "... (+" . (count($headings) - 3) . " more)";
1261 }
1262 $analysis .= "\n";
1263 }
1264 }
1265
1266 // Add SEO features
1267 $analysis .= "\nSEO FEATURES:\n";
1268 $analysis .= "- Schema Markup: " . ($content_data['seo']['has_schema'] ? 'Yes' : 'No') . "\n";
1269 if (!empty($content_data['meta_keywords'])) {
1270 $analysis .= "- Meta Keywords: {$content_data['meta_keywords']}\n";
1271 }
1272
1273 // Add content quality insights
1274 if (!empty($content_data['content_quality']['keyword_density']['top_keywords'])) {
1275 $analysis .= "\nTOP KEYWORDS:\n";
1276 foreach (array_slice($content_data['content_quality']['keyword_density']['top_keywords'], 0, 5) as $kw) {
1277 $analysis .= "- {$kw['keyword']}: {$kw['count']} times ({$kw['density']}%)\n";
1278 }
1279 }
1280
1281 // Add freshness indicators
1282 if (!empty($content_data['content_quality']['freshness_indicators'])) {
1283 $analysis .= "\nCONTENT FRESHNESS:\n";
1284 foreach ($content_data['content_quality']['freshness_indicators'] as $indicator) {
1285 $analysis .= "- {$indicator}\n";
1286 }
1287 }
1288
1289 $analysis .= "\n" . str_repeat("=", 50) . "\n";
1290
1291 return $analysis;
1292 }
1293
1294 /**
1295 * Get word count estimate based on content length
1296 *
1297 * @param string $content_length Content length setting
1298 * @return int Estimated word count
1299 */
1300 private function get_word_count_estimate(string $content_length): int {
1301 $estimates = [
1302 'short' => 650,
1303 'medium' => 1250,
1304 'long' => 2500
1305 ];
1306
1307 return $estimates[$content_length] ?? 1250;
1308 }
1309 private function save_brief(array $brief_data): int {
1310 global $wpdb;
1311
1312 $table_name = $wpdb->prefix . 'thinkrank_content_briefs';
1313
1314 // Prepare data for insertion
1315 $insert_data = [
1316 'user_id' => get_current_user_id(),
1317 'title' => $brief_data['title'][0] ?? 'Untitled Brief',
1318 'target_keywords' => wp_json_encode($brief_data['generation_params']['target_keywords'] ?? []),
1319 'content_type' => $brief_data['generation_params']['content_type'] ?? 'blog_post',
1320 'brief_data' => wp_json_encode($brief_data),
1321 'created_at' => current_time('mysql'),
1322 'updated_at' => current_time('mysql')
1323 ];
1324
1325 $insert_format = [
1326 '%d', // user_id
1327 '%s', // title
1328 '%s', // target_keywords
1329 '%s', // content_type
1330 '%s', // brief_data
1331 '%s', // created_at
1332 '%s' // updated_at
1333 ];
1334
1335 // phpcs:ignore WordPress.DB.DirectDatabaseQuery.DirectQuery,WordPress.DB.DirectDatabaseQuery.NoCaching, PluginCheck.Security.DirectDB.UnescapedDBParameter -- Content brief storage requires direct database access
1336 $result = $wpdb->insert($table_name, $insert_data, $insert_format);
1337
1338 if (false === $result) {
1339 throw new \Exception('Failed to save content brief to database.');
1340 }
1341
1342 /**
1343 * Fires after a content brief is persisted.
1344 *
1345 * Analytics listens to drop its cached overview so the brief counts
1346 * on the Usages page are not stale for a TTL.
1347 *
1348 * @since 2.2.1
1349 *
1350 * @param int $brief_id Row id of the stored brief.
1351 */
1352 do_action('thinkrank_content_brief_created', (int) $wpdb->insert_id);
1353
1354 return $wpdb->insert_id;
1355 }
1356
1357 /**
1358 * Normalize brief data for React compatibility
1359 *
1360 * @param array $brief_data Brief data to normalize
1361 * @return array Normalized brief data
1362 */
1363 private function normalize_brief_data(array $brief_data): array {
1364 // Normalize focus_keyword_analysis
1365 if (isset($brief_data['seo_recommendations']['focus_keyword_analysis'])) {
1366 $brief_data['seo_recommendations']['focus_keyword_analysis'] =
1367 $this->normalize_focus_keyword_analysis($brief_data['seo_recommendations']['focus_keyword_analysis']);
1368 }
1369
1370 // Normalize call_to_actions (convert objects to strings)
1371 if (isset($brief_data['call_to_actions']) && is_array($brief_data['call_to_actions'])) {
1372 $brief_data['call_to_actions'] = array_map(function($cta) {
1373 if (is_array($cta) && isset($cta['text'])) {
1374 return $cta['text'] . (isset($cta['placement']) ? ' (' . $cta['placement'] . ')' : '');
1375 }
1376 return is_string($cta) ? $cta : '';
1377 }, $brief_data['call_to_actions']);
1378 }
1379
1380 // Normalize visual content image_recommendations (convert objects to strings)
1381 if (isset($brief_data['visual_content']['image_recommendations']) && is_array($brief_data['visual_content']['image_recommendations'])) {
1382 $brief_data['visual_content']['image_recommendations'] = array_map(function($rec) {
1383 if (is_array($rec)) {
1384 $text = '';
1385 if (isset($rec['type'])) { $text .= $rec['type'] . ': ';
1386 }
1387 if (isset($rec['description'])) { $text .= $rec['description'];
1388 }
1389 if (isset($rec['alt_text'])) { $text .= ' (Alt: ' . $rec['alt_text'] . ')';
1390 }
1391 return $text ?: 'Image recommendation';
1392 }
1393 return is_string($rec) ? $rec : 'Image recommendation';
1394 }, $brief_data['visual_content']['image_recommendations']);
1395 }
1396
1397 return $this->sanitize_brief_output($brief_data);
1398 }
1399
1400 /**
1401 * Strip untrusted markup out of brief fields before they leave the server.
1402 *
1403 * Brief content crosses a trust boundary: it is assembled by an external AI
1404 * provider from prompts that can include text fetched from competitor URLs.
1405 * It was previously copied out of the decoded JSON verbatim and rendered in
1406 * the admin SPA through dangerouslySetInnerHTML, so a malicious or
1407 * prompt-injected response could execute script in the admin origin (#365).
1408 *
1409 * Runs on the read path as well as generation, so briefs stored before this
1410 * fix are sanitized when they are loaded.
1411 *
1412 * @since 1.32.0
1413 *
1414 * @param array $brief_data Brief data to sanitize.
1415 * @return array Sanitized brief data.
1416 */
1417 private function sanitize_brief_output(array $brief_data): array {
1418 foreach ($brief_data as $key => $value) {
1419 // The raw provider response is debug output shown as plain text, and
1420 // the generation params are our own values — leave both intact.
1421 if ('raw_response' === $key || 'generation_params' === $key) {
1422 continue;
1423 }
1424
1425 if ('content_body' === $key && is_string($value)) {
1426 // Deliberately HTML: it is the drafted article and is rendered as
1427 // markup. wp_kses_post() keeps normal post formatting while
1428 // dropping script/style/iframe, event-handler attributes and
1429 // javascript: URLs.
1430 $brief_data[$key] = wp_kses_post($value);
1431 continue;
1432 }
1433
1434 if (is_array($value)) {
1435 $brief_data[$key] = $this->sanitize_brief_output($value);
1436 } elseif (is_string($value)) {
1437 // Every other field is plain text (headings, keywords, guidance).
1438 // Markdown emphasis markers are preserved; HTML tags are not.
1439 $brief_data[$key] = wp_strip_all_tags($value);
1440 }
1441 }
1442
1443 return $brief_data;
1444 }
1445
1446 /**
1447 * Get saved briefs for current user
1448 *
1449 * @param int $limit Number of briefs to retrieve
1450 * @param int $offset Offset for pagination
1451 * @return array Array of saved briefs
1452 */
1453 public function get_user_briefs(int $limit = 10, int $offset = 0): array {
1454 global $wpdb;
1455
1456 // Get table name and escape it properly (table names cannot be parameterized)
1457 $table_name = esc_sql($wpdb->prefix . 'thinkrank_content_briefs');
1458 $user_id = get_current_user_id();
1459
1460 // phpcs:ignore WordPress.DB.DirectDatabaseQuery.DirectQuery,WordPress.DB.DirectDatabaseQuery.NoCaching, PluginCheck.Security.DirectDB.UnescapedDBParameter -- Content brief retrieval requires direct database access
1461 $results = $wpdb->get_results(
1462 $wpdb->prepare(
1463 // phpcs:ignore WordPress.DB.PreparedSQL.InterpolatedNotPrepared, PluginCheck.Security.DirectDB.UnescapedDBParameter -- Table name is properly escaped using esc_sql()
1464 "SELECT * FROM `{$table_name}` WHERE user_id = %d ORDER BY created_at DESC LIMIT %d OFFSET %d",
1465 $user_id,
1466 $limit,
1467 $offset
1468 ),
1469 ARRAY_A
1470 );
1471
1472 // $wpdb->get_results() returns null on a DB error; this method's return
1473 // type is : array, so normalize before iterating/returning.
1474 if (!is_array($results)) {
1475 return [];
1476 }
1477
1478 // Decode JSON data and normalize for React compatibility
1479 foreach ($results as &$brief) {
1480 $brief = $this->hydrate_brief_row($brief);
1481 }
1482 unset($brief);
1483
1484 return $results;
1485 }
1486
1487 /**
1488 * Get a single saved brief by id, scoped to the current user.
1489 *
1490 * @param int $brief_id Brief ID.
1491 * @return array|null Hydrated brief, or null if it doesn't exist or does not
1492 * belong to the current user.
1493 */
1494 public function get_brief(int $brief_id): ?array {
1495 global $wpdb;
1496
1497 // Table names cannot be parameterized; escape it.
1498 $table_name = esc_sql($wpdb->prefix . 'thinkrank_content_briefs');
1499 $user_id = get_current_user_id();
1500
1501 // phpcs:ignore WordPress.DB.DirectDatabaseQuery.DirectQuery,WordPress.DB.DirectDatabaseQuery.NoCaching, PluginCheck.Security.DirectDB.UnescapedDBParameter -- Content brief retrieval requires direct database access
1502 $brief = $wpdb->get_row(
1503 $wpdb->prepare(
1504 // phpcs:ignore WordPress.DB.PreparedSQL.InterpolatedNotPrepared, PluginCheck.Security.DirectDB.UnescapedDBParameter -- Table name is properly escaped using esc_sql()
1505 "SELECT * FROM `{$table_name}` WHERE id = %d AND user_id = %d LIMIT 1",
1506 $brief_id,
1507 $user_id
1508 ),
1509 ARRAY_A
1510 );
1511
1512 if (!$brief) {
1513 return null;
1514 }
1515
1516 return $this->hydrate_brief_row($brief);
1517 }
1518
1519 /**
1520 * Decode + normalize a raw content-brief DB row for API/React consumption.
1521 *
1522 * @param array $brief Raw database row.
1523 * @return array Hydrated brief.
1524 */
1525 private function hydrate_brief_row(array $brief): array {
1526 $brief['target_keywords'] = json_decode($brief['target_keywords'], true);
1527 $brief['brief_data'] = json_decode($brief['brief_data'], true);
1528
1529 // Cast: the row comes from $wpdb, which returns every column as a
1530 // string, and both helpers declare an int parameter.
1531 $brief_id = (int) $brief['id'];
1532
1533 // Retrieve raw response from ai_usage table
1534 $brief['brief_data']['raw_response'] = $this->get_raw_response_for_brief($brief_id);
1535
1536 // Update model with actual model used (if available in ai_usage table)
1537 $actual_model = $this->get_actual_model_for_brief($brief_id);
1538 if ($actual_model && isset($brief['brief_data']['generation_meta'])) {
1539 $brief['brief_data']['generation_meta']['model'] = $actual_model;
1540 }
1541
1542 // Apply normalization to existing briefs to ensure React compatibility
1543 $brief['brief_data'] = $this->normalize_brief_data($brief['brief_data']);
1544
1545 return $brief;
1546 }
1547
1548 /**
1549 * Delete brief
1550 *
1551 * @param int $brief_id Brief ID to delete
1552 * @return bool Success status
1553 */
1554 public function delete_brief(int $brief_id): bool {
1555 global $wpdb;
1556
1557 $table_name = $wpdb->prefix . 'thinkrank_content_briefs';
1558 $user_id = get_current_user_id();
1559
1560 // phpcs:ignore WordPress.DB.DirectDatabaseQuery.DirectQuery,WordPress.DB.DirectDatabaseQuery.NoCaching, PluginCheck.Security.DirectDB.UnescapedDBParameter -- Content brief deletion requires direct database access
1561 $result = $wpdb->delete(
1562 $table_name,
1563 [
1564 'id' => $brief_id,
1565 'user_id' => $user_id
1566 ],
1567 ['%d', '%d']
1568 );
1569
1570 return $result !== false;
1571 }
1572
1573 /**
1574 * Calculate readability score using Flesch Reading Ease
1575 *
1576 * @param string $text Text to analyze
1577 * @return array Readability metrics
1578 */
1579 private function calculate_readability_score(string $text): array {
1580 if (empty($text)) {
1581 return ['score' => 0, 'level' => 'Unknown', 'grade' => 'N/A'];
1582 }
1583
1584 // Count sentences (approximate)
1585 $sentences = preg_split('/[.!?]+/', $text);
1586 $sentence_count = count(array_filter($sentences, function($s) { return trim($s) !== '';
1587 }));
1588
1589 // Count words
1590 $word_count = str_word_count($text);
1591
1592 // Count syllables (approximate)
1593 $syllable_count = $this->count_syllables($text);
1594
1595 if ($sentence_count === 0 || $word_count === 0) {
1596 return ['score' => 0, 'level' => 'Unknown', 'grade' => 'N/A'];
1597 }
1598
1599 // Flesch Reading Ease formula
1600 $avg_sentence_length = $word_count / $sentence_count;
1601 $avg_syllables_per_word = $syllable_count / $word_count;
1602
1603 $flesch_score = 206.835 - (1.015 * $avg_sentence_length) - (84.6 * $avg_syllables_per_word);
1604 $flesch_score = max(0, min(100, $flesch_score)); // Clamp between 0-100
1605
1606 // Determine reading level
1607 if ($flesch_score >= 90) {
1608 $level = 'Very Easy';
1609 $grade = '5th grade';
1610 } elseif ($flesch_score >= 80) {
1611 $level = 'Easy';
1612 $grade = '6th grade';
1613 } elseif ($flesch_score >= 70) {
1614 $level = 'Fairly Easy';
1615 $grade = '7th grade';
1616 } elseif ($flesch_score >= 60) {
1617 $level = 'Standard';
1618 $grade = '8th-9th grade';
1619 } elseif ($flesch_score >= 50) {
1620 $level = 'Fairly Difficult';
1621 $grade = '10th-12th grade';
1622 } elseif ($flesch_score >= 30) {
1623 $level = 'Difficult';
1624 $grade = 'College level';
1625 } else {
1626 $level = 'Very Difficult';
1627 $grade = 'Graduate level';
1628 }
1629
1630 return [
1631 'score' => round($flesch_score, 1),
1632 'level' => $level,
1633 'grade' => $grade
1634 ];
1635 }
1636
1637 /**
1638 * Count syllables in text (approximate)
1639 *
1640 * @param string $text Text to analyze
1641 * @return int Syllable count
1642 */
1643 private function count_syllables(string $text): int {
1644 $words = str_word_count(strtolower($text), 1);
1645 $syllable_count = 0;
1646
1647 foreach ($words as $word) {
1648 $syllable_count += $this->count_word_syllables($word);
1649 }
1650
1651 return max(1, $syllable_count); // At least 1 syllable
1652 }
1653
1654 /**
1655 * Count syllables in a single word
1656 *
1657 * @param string $word Word to analyze
1658 * @return int Syllable count
1659 */
1660 private function count_word_syllables(string $word): int {
1661 $word = strtolower($word);
1662 $vowels = 'aeiouy';
1663 $syllable_count = 0;
1664 $previous_was_vowel = false;
1665
1666 for ($i = 0, $len = strlen($word); $i < $len; $i++) {
1667 $is_vowel = strpos($vowels, $word[$i]) !== false;
1668 if ($is_vowel && !$previous_was_vowel) {
1669 $syllable_count++;
1670 }
1671 $previous_was_vowel = $is_vowel;
1672 }
1673
1674 // Handle silent 'e'
1675 if (substr($word, -1) === 'e' && $syllable_count > 1) {
1676 $syllable_count--;
1677 }
1678
1679 return max(1, $syllable_count);
1680 }
1681
1682 /**
1683 * Analyze keyword density in content
1684 *
1685 * @param string $text Content text
1686 * @param string $title Page title
1687 * @return array Keyword analysis
1688 */
1689 private function analyze_keyword_density(string $text, string $title): array {
1690 $combined_text = strtolower($title . ' ' . $text);
1691 $words = str_word_count($combined_text, 1);
1692 $total_words = count($words);
1693
1694 if ($total_words === 0) {
1695 return ['top_keywords' => [], 'total_words' => 0];
1696 }
1697
1698 // Count word frequency
1699 $word_counts = array_count_values($words);
1700
1701 // Filter out common stop words
1702 $stop_words = ['the', 'a', 'an', 'and', 'or', 'but', 'in', 'on', 'at', 'to', 'for', 'of', 'with', 'by', 'is', 'are', 'was', 'were', 'be', 'been', 'have', 'has', 'had', 'do', 'does', 'did', 'will', 'would', 'could', 'should', 'may', 'might', 'must', 'can', 'this', 'that', 'these', 'those', 'i', 'you', 'he', 'she', 'it', 'we', 'they', 'me', 'him', 'her', 'us', 'them'];
1703
1704 foreach ($stop_words as $stop_word) {
1705 unset($word_counts[$stop_word]);
1706 }
1707
1708 // Filter out single characters and numbers
1709 $word_counts = array_filter($word_counts, function($count, $word) {
1710 return strlen($word) > 2 && !is_numeric($word) && $count > 1;
1711 }, ARRAY_FILTER_USE_BOTH);
1712
1713 // Sort by frequency
1714 arsort($word_counts);
1715
1716 // Calculate density and format results
1717 $top_keywords = [];
1718 foreach (array_slice($word_counts, 0, 10, true) as $word => $count) {
1719 $density = round(($count / $total_words) * 100, 2);
1720 $top_keywords[] = [
1721 'keyword' => $word,
1722 'count' => $count,
1723 'density' => $density
1724 ];
1725 }
1726
1727 return [
1728 'top_keywords' => $top_keywords,
1729 'total_words' => $total_words
1730 ];
1731 }
1732
1733 /**
1734 * Detect content freshness indicators
1735 *
1736 * @param string $html Full HTML content
1737 * @param string $text Body text
1738 * @return array Freshness indicators
1739 */
1740 private function detect_freshness_indicators(string $html, string $text): array {
1741 $indicators = [];
1742
1743 // Check for date patterns in content
1744 if (preg_match('/\b(updated|revised|modified|published).*?(\d{4}|\d{1,2}\/\d{1,2}\/\d{2,4})/i', $text)) {
1745 $indicators[] = 'Contains recent update dates';
1746 }
1747
1748 // Check for current year references
1749 $current_year = gmdate('Y');
1750 if (strpos($text, $current_year) !== false) {
1751 $indicators[] = "References current year ({$current_year})";
1752 }
1753
1754 // Check for "latest", "new", "recent" keywords
1755 if (preg_match('/\b(latest|newest|recent|updated|current|modern|today)\b/i', $text)) {
1756 $indicators[] = 'Uses freshness keywords';
1757 }
1758
1759 // Check for structured data with dates
1760 if (preg_match('/"dateModified"|"datePublished"/i', $html)) {
1761 $indicators[] = 'Has structured date metadata';
1762 }
1763
1764 return $indicators;
1765 }
1766
1767 /**
1768 * Score title for SEO effectiveness
1769 *
1770 * @param string $title Page title
1771 * @return array Title scoring
1772 */
1773 private function score_title_seo(string $title): array {
1774 $score = 0;
1775 $max_score = 100;
1776 $feedback = [];
1777
1778 // Length check (optimal: 50-60 characters)
1779 $length = strlen($title);
1780 if ($length >= 50 && $length <= 60) {
1781 $score += 25;
1782 $feedback[] = 'Good length (50-60 chars)';
1783 } elseif ($length >= 40 && $length <= 70) {
1784 $score += 15;
1785 $feedback[] = 'Acceptable length';
1786 } else {
1787 $feedback[] = $length < 40 ? 'Too short (under 40 chars)' : 'Too long (over 70 chars)';
1788 }
1789
1790 // Word count (optimal: 5-9 words)
1791 $word_count = str_word_count($title);
1792 if ($word_count >= 5 && $word_count <= 9) {
1793 $score += 20;
1794 $feedback[] = 'Good word count';
1795 } elseif ($word_count >= 3 && $word_count <= 12) {
1796 $score += 10;
1797 $feedback[] = 'Acceptable word count';
1798 } else {
1799 $feedback[] = $word_count < 3 ? 'Too few words' : 'Too many words';
1800 }
1801
1802 // Check for power words
1803 $power_words = ['ultimate', 'complete', 'guide', 'best', 'top', 'essential', 'proven', 'expert', 'advanced', 'beginner'];
1804 $has_power_words = false;
1805 foreach ($power_words as $power_word) {
1806 if (stripos($title, $power_word) !== false) {
1807 $has_power_words = true;
1808 break;
1809 }
1810 }
1811 if ($has_power_words) {
1812 $score += 15;
1813 $feedback[] = 'Contains power words';
1814 }
1815
1816 // Check for numbers
1817 if (preg_match('/\d+/', $title)) {
1818 $score += 10;
1819 $feedback[] = 'Contains numbers';
1820 }
1821
1822 // Check for emotional triggers
1823 $emotional_words = ['amazing', 'incredible', 'shocking', 'secret', 'revealed', 'proven', 'guaranteed'];
1824 $has_emotional_words = false;
1825 foreach ($emotional_words as $emotional_word) {
1826 if (stripos($title, $emotional_word) !== false) {
1827 $has_emotional_words = true;
1828 break;
1829 }
1830 }
1831 if ($has_emotional_words) {
1832 $score += 10;
1833 $feedback[] = 'Contains emotional triggers';
1834 }
1835
1836 // Uniqueness check (avoid generic titles)
1837 $generic_patterns = ['untitled', 'new page', 'home', 'welcome'];
1838 $is_generic = false;
1839 foreach ($generic_patterns as $pattern) {
1840 if (stripos($title, $pattern) !== false) {
1841 $is_generic = true;
1842 break;
1843 }
1844 }
1845 if (!$is_generic) {
1846 $score += 20;
1847 $feedback[] = 'Appears unique';
1848 } else {
1849 $feedback[] = 'Appears generic';
1850 }
1851
1852 return [
1853 'score' => min($score, $max_score),
1854 'max_score' => $max_score,
1855 'grade' => $this->get_grade_from_score($score),
1856 'feedback' => $feedback
1857 ];
1858 }
1859
1860 /**
1861 * Score meta description for SEO effectiveness
1862 *
1863 * @param string $meta_desc Meta description
1864 * @return array Meta description scoring
1865 */
1866 private function score_meta_description(string $meta_desc): array {
1867 $score = 0;
1868 $max_score = 100;
1869 $feedback = [];
1870
1871 if (empty($meta_desc)) {
1872 return [
1873 'score' => 0,
1874 'max_score' => $max_score,
1875 'grade' => 'F',
1876 'feedback' => ['No meta description found']
1877 ];
1878 }
1879
1880 // Length check (optimal: 150-160 characters)
1881 $length = strlen($meta_desc);
1882 if ($length >= 150 && $length <= 160) {
1883 $score += 30;
1884 $feedback[] = 'Optimal length (150-160 chars)';
1885 } elseif ($length >= 120 && $length <= 170) {
1886 $score += 20;
1887 $feedback[] = 'Good length';
1888 } elseif ($length >= 100 && $length <= 180) {
1889 $score += 10;
1890 $feedback[] = 'Acceptable length';
1891 } else {
1892 $feedback[] = $length < 100 ? 'Too short (under 100 chars)' : 'Too long (over 180 chars)';
1893 }
1894
1895 // Check for call-to-action
1896 $cta_words = ['learn', 'discover', 'find out', 'get', 'download', 'try', 'start', 'join', 'sign up', 'contact', 'buy', 'shop'];
1897 $has_cta = false;
1898 foreach ($cta_words as $cta_word) {
1899 if (stripos($meta_desc, $cta_word) !== false) {
1900 $has_cta = true;
1901 break;
1902 }
1903 }
1904 if ($has_cta) {
1905 $score += 20;
1906 $feedback[] = 'Contains call-to-action';
1907 }
1908
1909 // Check for unique selling proposition
1910 $usp_words = ['best', 'top', 'leading', 'expert', 'professional', 'trusted', 'proven', 'award-winning'];
1911 $has_usp = false;
1912 foreach ($usp_words as $usp_word) {
1913 if (stripos($meta_desc, $usp_word) !== false) {
1914 $has_usp = true;
1915 break;
1916 }
1917 }
1918 if ($has_usp) {
1919 $score += 15;
1920 $feedback[] = 'Contains unique selling proposition';
1921 }
1922
1923 // Check for benefits/value proposition
1924 $benefit_words = ['save', 'improve', 'increase', 'boost', 'enhance', 'optimize', 'maximize', 'reduce', 'eliminate'];
1925 $has_benefits = false;
1926 foreach ($benefit_words as $benefit_word) {
1927 if (stripos($meta_desc, $benefit_word) !== false) {
1928 $has_benefits = true;
1929 break;
1930 }
1931 }
1932 if ($has_benefits) {
1933 $score += 15;
1934 $feedback[] = 'Highlights benefits';
1935 }
1936
1937 // Readability check
1938 $sentences = preg_split('/[.!?]+/', $meta_desc);
1939 $sentence_count = count(array_filter($sentences, function($s) { return trim($s) !== '';
1940 }));
1941 if ($sentence_count >= 1 && $sentence_count <= 3) {
1942 $score += 20;
1943 $feedback[] = 'Good sentence structure';
1944 } else {
1945 $feedback[] = $sentence_count === 0 ? 'No clear sentences' : 'Too many sentences';
1946 }
1947
1948 return [
1949 'score' => min($score, $max_score),
1950 'max_score' => $max_score,
1951 'grade' => $this->get_grade_from_score($score),
1952 'feedback' => $feedback
1953 ];
1954 }
1955
1956 /**
1957 * Assess content depth based on structure and length
1958 *
1959 * @param array $headings Heading structure
1960 * @param int $word_count Word count
1961 * @return array Content depth assessment
1962 */
1963 private function assess_content_depth(array $headings, int $word_count): array {
1964 $depth_score = 0;
1965 $max_score = 100;
1966
1967 // Word count scoring (more words = more depth)
1968 if ($word_count >= 2000) {
1969 $depth_score += 40;
1970 } elseif ($word_count >= 1000) {
1971 $depth_score += 30;
1972 } elseif ($word_count >= 500) {
1973 $depth_score += 20;
1974 } elseif ($word_count >= 300) {
1975 $depth_score += 10;
1976 }
1977
1978 // Heading structure scoring
1979 $total_headings = 0;
1980 $heading_levels = 0;
1981 foreach ($headings as $level => $level_headings) {
1982 $total_headings += count($level_headings);
1983 $heading_levels++;
1984 }
1985
1986 if ($total_headings >= 10) {
1987 $depth_score += 25;
1988 } elseif ($total_headings >= 5) {
1989 $depth_score += 15;
1990 } elseif ($total_headings >= 3) {
1991 $depth_score += 10;
1992 }
1993
1994 // Heading hierarchy scoring
1995 if ($heading_levels >= 3) {
1996 $depth_score += 20;
1997 } elseif ($heading_levels >= 2) {
1998 $depth_score += 15;
1999 }
2000
2001 // Content structure bonus
2002 if (isset($headings['h1']) && isset($headings['h2'])) {
2003 $depth_score += 15;
2004 }
2005
2006 // Determine depth level
2007 if ($depth_score >= 80) {
2008 $level = 'Comprehensive';
2009 } elseif ($depth_score >= 60) {
2010 $level = 'Detailed';
2011 } elseif ($depth_score >= 40) {
2012 $level = 'Moderate';
2013 } elseif ($depth_score >= 20) {
2014 $level = 'Basic';
2015 } else {
2016 $level = 'Shallow';
2017 }
2018
2019 return [
2020 'score' => min($depth_score, $max_score),
2021 'level' => $level,
2022 'word_count' => $word_count,
2023 'total_headings' => $total_headings,
2024 'heading_levels' => $heading_levels
2025 ];
2026 }
2027
2028 /**
2029 * Convert numeric score to letter grade
2030 *
2031 * @param int $score Numeric score
2032 * @return string Letter grade
2033 */
2034 private function get_grade_from_score(int $score): string {
2035 if ($score >= 90) { return 'A';
2036 }
2037 if ($score >= 80) { return 'B';
2038 }
2039 if ($score >= 70) { return 'C';
2040 }
2041 if ($score >= 60) { return 'D';
2042 }
2043 return 'F';
2044 }
2045
2046 /**
2047 * Log AI usage for analytics
2048 *
2049 * @param int $user_id User ID
2050 * @param string $action Action performed
2051 * @param int $tokens_used Tokens consumed
2052 * @param int|null $post_id Related post/brief ID
2053 * @param string|null $raw_response Raw AI response for debugging
2054 * @param string|null $actual_model Actual model used (from response)
2055 * @return int Usage record ID
2056 */
2057 private function log_ai_usage(int $user_id, string $action, int $tokens_used, ?int $post_id = null, ?string $raw_response = null, ?string $actual_model = null): int {
2058 global $wpdb;
2059
2060 $table_name = $wpdb->prefix . 'thinkrank_ai_usage';
2061
2062 $metadata = [];
2063 if ($raw_response) {
2064 $metadata['raw_response'] = $raw_response;
2065 }
2066 if ($actual_model) {
2067 $metadata['actual_model'] = $actual_model;
2068 }
2069
2070 // phpcs:ignore WordPress.DB.DirectDatabaseQuery.DirectQuery,WordPress.DB.DirectDatabaseQuery.NoCaching, PluginCheck.Security.DirectDB.UnescapedDBParameter -- AI usage logging requires direct database access
2071 $wpdb->insert(
2072 $table_name,
2073 [
2074 'user_id' => $user_id,
2075 'action' => $action,
2076 'tokens_used' => $tokens_used,
2077 'provider' => $this->settings->get('ai_provider', Settings::AI_PROVIDER_NONE),
2078 'post_id' => $post_id,
2079 'metadata' => !empty($metadata) ? wp_json_encode($metadata) : null,
2080 'created_at' => current_time('mysql'),
2081 ],
2082 ['%d', '%s', '%d', '%s', '%d', '%s', '%s']
2083 );
2084
2085 /**
2086 * Fires after an AI usage row is recorded.
2087 *
2088 * @since 2.2.1
2089 *
2090 * @param int $user_id User the usage was recorded against.
2091 */
2092 do_action('thinkrank_ai_usage_logged', $user_id);
2093
2094 return $wpdb->insert_id;
2095 }
2096
2097 /**
2098 * Get raw AI response for a brief from ai_usage table
2099 *
2100 * @param int $brief_id Brief ID
2101 * @return string Raw AI response or empty string if not found
2102 */
2103 private function get_raw_response_for_brief(int $brief_id): string {
2104 global $wpdb;
2105
2106 $table_name = $wpdb->prefix . 'thinkrank_ai_usage';
2107
2108 // phpcs:ignore WordPress.DB.DirectDatabaseQuery.DirectQuery,WordPress.DB.DirectDatabaseQuery.NoCaching, PluginCheck.Security.DirectDB.UnescapedDBParameter -- AI usage retrieval requires direct database access
2109 $result = $wpdb->get_var(
2110 // phpcs:ignore WordPress.DB.PreparedSQL.InterpolatedNotPrepared, PluginCheck.Security.DirectDB.UnescapedDBParameter -- Table name is validated with WordPress prefix
2111 $wpdb->prepare(
2112 // phpcs:ignore WordPress.DB.PreparedSQL.InterpolatedNotPrepared, PluginCheck.Security.DirectDB.UnescapedDBParameter -- Table name is validated with WordPress prefix
2113 "SELECT metadata FROM `{$table_name}` WHERE post_id = %d AND action = 'content_brief' ORDER BY created_at DESC LIMIT 1",
2114 $brief_id
2115 )
2116 );
2117
2118 if ($result) {
2119 $metadata = json_decode($result, true);
2120 return $metadata['raw_response'] ?? '';
2121 }
2122
2123 return '';
2124 }
2125
2126 /**
2127 * Get actual model used for a brief from ai_usage table
2128 *
2129 * @param int $brief_id Brief ID
2130 * @return string|null Actual model used or null if not found
2131 */
2132 private function get_actual_model_for_brief(int $brief_id): ?string {
2133 global $wpdb;
2134
2135 $table_name = $wpdb->prefix . 'thinkrank_ai_usage';
2136
2137 // phpcs:ignore WordPress.DB.DirectDatabaseQuery.DirectQuery,WordPress.DB.DirectDatabaseQuery.NoCaching, PluginCheck.Security.DirectDB.UnescapedDBParameter -- AI usage retrieval requires direct database access
2138 $result = $wpdb->get_var(
2139 // phpcs:ignore WordPress.DB.PreparedSQL.InterpolatedNotPrepared, PluginCheck.Security.DirectDB.UnescapedDBParameter -- Table name is validated with WordPress prefix
2140 $wpdb->prepare(
2141 // phpcs:ignore WordPress.DB.PreparedSQL.InterpolatedNotPrepared, PluginCheck.Security.DirectDB.UnescapedDBParameter -- Table name is validated with WordPress prefix
2142 "SELECT metadata FROM `{$table_name}` WHERE post_id = %d AND action = 'content_brief' ORDER BY created_at DESC LIMIT 1",
2143 $brief_id
2144 )
2145 );
2146
2147 if ($result) {
2148 $metadata = json_decode($result, true);
2149 return $metadata['actual_model'] ?? null;
2150 }
2151
2152 return null;
2153 }
2154
2155 /**
2156 * Get Prompt Builder instance
2157 *
2158 * @since 1.0.0
2159 *
2160 * @return \ThinkRank\AI\Prompt_Builder Prompt Builder instance
2161 */
2162 private function get_prompt_builder(): \ThinkRank\AI\Prompt_Builder {
2163 if (!class_exists('ThinkRank\\AI\\Prompt_Builder')) {
2164 require_once THINKRANK_PLUGIN_DIR . 'includes/ai/class-prompt-builder.php';
2165 }
2166 return new \ThinkRank\AI\Prompt_Builder();
2167 }
2168 }
2169