PluginProbe
MxChat – AI Chatbot & Content Generation for WordPress / 3.2.8
MxChat – AI Chatbot & Content Generation for WordPress v3.2.8
3.2.21 3.2.20 3.2.19 3.2.18 3.2.17 3.2.16 3.2.15 3.2.14 3.2.12 3.2.13 3.2.11 3.2.10 3.2.9 3.2.8 3.2.7 3.2.6 3.2.5 3.2.4 3.2.3 3.2.2 3.2.1 2.0.3 2.0.4 2.0.5 2.0.6 All 152 releases
← All changes | admin/class-knowledge-manager.php +286 -7 3.2.33.2.8 View file →
@@ -602,12 +602,11 @@
602 602 array(
603 603 'article_content' => $article_content,
604 604 'embedding_vector' => $embedding_vector_serialized,
605 605 'source_url' => $article_url,
606 - 'embedding_model' => MxChat_Utils::get_active_embedding_model(),
607 606 ),
608 607 array('id' => $prompt_id),
609 - array('%s', '%s', '%s', '%s'),
608 + array('%s', '%s', '%s'),
610 609 array('%d')
611 610 );
612 611 if ($updated !== false) {
613 612 wp_send_json_success();
@@ -3483,9 +3482,22 @@
3483 3482 }
3484 3483
3485 3484 // Get bot_id from request
3486 3485 $bot_id = isset($_POST['bot_id']) ? sanitize_key($_POST['bot_id']) : 'default';
3487 -
3486 +
3487 + // ACF→PDF extraction is opt-in per import batch. Persist the last-used value so users
3488 + // don't re-check on every batch; the default is OFF for installs that haven't set it.
3489 + $extract_acf_pdfs = !empty($_POST['extract_acf_pdfs']) && $_POST['extract_acf_pdfs'] !== 'false';
3490 + $mxchat_options = get_option('mxchat_options', array());
3491 + if (!is_array($mxchat_options)) {
3492 + $mxchat_options = array();
3493 + }
3494 + $prior_default = !empty($mxchat_options['acf_pdf_extract_default']);
3495 + if ($prior_default !== $extract_acf_pdfs) {
3496 + $mxchat_options['acf_pdf_extract_default'] = $extract_acf_pdfs ? 1 : 0;
3497 + update_option('mxchat_options', $mxchat_options);
3498 + }
3499 +
3488 3500 // Process only ONE post at a time to avoid request size issues
3489 3501 $post_id = reset($post_ids);
3490 3502 $post = get_post($post_id);
3491 3503
@@ -3591,23 +3603,57 @@
3591 3603 }
3592 3604
3593 3605 // ADD ACF FIELDS SUPPORT
3594 3606 $acf_fields = $this->mxchat_get_acf_fields_for_post($post_id);
3607 + $pdf_extracted_count = 0;
3595 3608 if (!empty($acf_fields)) {
3596 3609 $acf_content_parts = array();
3597 -
3610 + $pdf_attachment_ids = array();
3611 +
3598 3612 foreach ($acf_fields as $field_name => $field_value) {
3599 3613 $formatted_value = $this->mxchat_format_acf_field_value($field_value, $field_name, $post_id);
3600 -
3614 +
3601 3615 if (!empty($formatted_value)) {
3602 3616 $field_label = ucwords(str_replace('_', ' ', $field_name));
3603 3617 $acf_content_parts[] = $field_label . ": " . $formatted_value;
3604 3618 }
3619 +
3620 + // Walk this field's value tree for any PDF attachment references and queue them for extraction.
3621 + // Only when the user opted into ACF→PDF extraction for this batch; otherwise the ACF text
3622 + // still lands in the KB but the heavier PDF parsing is skipped.
3623 + if ($extract_acf_pdfs) {
3624 + $this->mxchat_collect_pdf_attachment_ids_from_acf_value($field_value, $pdf_attachment_ids);
3625 + }
3605 3626 }
3606 -
3627 +
3607 3628 if (!empty($acf_content_parts)) {
3608 3629 $content .= "\n\n" . implode("\n", $acf_content_parts);
3609 3630 }
3631 +
3632 + // Extract text from each unique PDF found in ACF fields and append as a labeled section
3633 + if ($extract_acf_pdfs && !empty($pdf_attachment_ids)) {
3634 + $pdf_attachment_ids = array_unique(array_filter(array_map('intval', $pdf_attachment_ids)));
3635 + $pdf_sections = array();
3636 + foreach ($pdf_attachment_ids as $att_id) {
3637 + $pdf_text = $this->mxchat_extract_pdf_text_by_attachment_id($att_id);
3638 + if (!empty($pdf_text)) {
3639 + $pdf_title = get_the_title($att_id);
3640 + $pdf_url = wp_get_attachment_url($att_id);
3641 + $header = 'PDF Attachment';
3642 + if (!empty($pdf_title)) {
3643 + $header .= ': ' . $pdf_title;
3644 + }
3645 + if (!empty($pdf_url)) {
3646 + $header .= ' (' . $pdf_url . ')';
3647 + }
3648 + $pdf_sections[] = $header . "\n" . $pdf_text;
3649 + $pdf_extracted_count++;
3650 + }
3651 + }
3652 + if (!empty($pdf_sections)) {
3653 + $content .= "\n\nAttached PDFs:\n" . implode("\n\n", $pdf_sections);
3654 + }
3655 + }
3610 3656 }
3611 3657
3612 3658 // ADD CUSTOM POST META SUPPORT (whitelisted non-ACF meta fields)
3613 3659 $custom_meta = $this->mxchat_get_whitelisted_post_meta($post_id);
@@ -3737,8 +3783,9 @@
3737 3783 'title' => $post->post_title,
3738 3784 'operation_type' => $operation_type,
3739 3785 'vector_id' => $vector_id,
3740 3786 'acf_fields_found' => $acf_field_count,
3787 + 'pdf_extracted_count' => (int) $pdf_extracted_count,
3741 3788 'content_preview' => substr($content, 0, 100) . '...',
3742 3789 'bot_id' => $bot_id
3743 3790 ));
3744 3791 exit;
@@ -4153,9 +4200,20 @@
4153 4200
4154 4201 // Get bot-specific options
4155 4202 $bot_options = $this->get_bot_options($bot_id);
4156 4203 $options = !empty($bot_options) ? $bot_options : get_option('mxchat_options');
4157 -
4204 +
4205 + // Opt-in: when the custom provider is selected for embeddings, index through
4206 + // the same custom endpoint the query path uses so stored vectors and query
4207 + // vectors share a model. Returns the vector array on success, or an error
4208 + // string on failure (this function's existing failure contract).
4209 + if (isset($options['custom_provider_for_embeddings']) && $options['custom_provider_for_embeddings'] === 'on') {
4210 + if (!class_exists('MxChat_Utils')) {
4211 + require_once dirname(__FILE__) . '/../includes/class-mxchat-utils.php';
4212 + }
4213 + return MxChat_Utils::generate_embedding_custom($text, $options);
4214 + }
4215 +
4158 4216 $selected_model = $options['embedding_model'] ?? 'text-embedding-ada-002';
4159 4217 //error_log('[MXCHAT-EMBED] Selected embedding model for bot ' . $bot_id . ': ' . $selected_model);
4160 4218
4161 4219 // Determine provider and endpoint
@@ -5073,8 +5131,197 @@
5073 5131 return implode(', ', array_filter($text_parts));
5074 5132 }
5075 5133
5076 5134 /**
5135 + * Walk an ACF field value tree and collect attachment IDs for any value that
5136 + * resolves to a PDF in the WordPress media library. Handles the three shapes
5137 + * ACF returns for File/Image/URL fields (array with ID+url, integer attachment ID,
5138 + * plain URL string), and recurses through repeater/group/flexible content.
5139 + *
5140 + * @param mixed $value The ACF field value (any depth)
5141 + * @param array $out Accumulator (passed by reference) for attachment IDs
5142 + * @param int $depth Recursion guard
5143 + */
5144 +private function mxchat_collect_pdf_attachment_ids_from_acf_value($value, &$out, $depth = 0) {
5145 + if ($depth > 6) {
5146 + return; // prevent runaway recursion on circular/very-deep structures
5147 + }
5148 +
5149 + if (empty($value)) {
5150 + return;
5151 + }
5152 +
5153 + // Array shapes: ACF File/Image return value=array; repeaters/groups are arrays of arrays
5154 + if (is_array($value)) {
5155 + // Direct File/Image-style array (has 'url' and usually 'ID' + 'mime_type')
5156 + $looks_like_attachment = isset($value['url']) || isset($value['ID']) || isset($value['id']);
5157 + if ($looks_like_attachment) {
5158 + $att_id = 0;
5159 + if (!empty($value['ID']) && is_numeric($value['ID'])) {
5160 + $att_id = (int) $value['ID'];
5161 + } elseif (!empty($value['id']) && is_numeric($value['id'])) {
5162 + $att_id = (int) $value['id'];
5163 + } elseif (!empty($value['url']) && is_string($value['url'])) {
5164 + $att_id = (int) attachment_url_to_postid($value['url']);
5165 + }
5166 +
5167 + $is_pdf = false;
5168 + if (!empty($value['mime_type']) && $value['mime_type'] === 'application/pdf') {
5169 + $is_pdf = true;
5170 + } elseif (!empty($value['subtype']) && strtolower((string) $value['subtype']) === 'pdf') {
5171 + $is_pdf = true;
5172 + } elseif (!empty($value['url']) && is_string($value['url']) && $this->mxchat_url_looks_like_pdf($value['url'])) {
5173 + $is_pdf = true;
5174 + } elseif ($att_id && get_post_mime_type($att_id) === 'application/pdf') {
5175 + $is_pdf = true;
5176 + }
5177 +
5178 + if ($is_pdf && $att_id && get_post_mime_type($att_id) === 'application/pdf') {
5179 + $out[] = $att_id;
5180 + }
5181 + // An array node that represents one attachment doesn't contain other
5182 + // attachments inside it — done with this branch.
5183 + return;
5184 + }
5185 +
5186 + // Recurse: repeater rows, flexible-content layouts, groups, etc.
5187 + foreach ($value as $sub) {
5188 + $this->mxchat_collect_pdf_attachment_ids_from_acf_value($sub, $out, $depth + 1);
5189 + }
5190 + return;
5191 + }
5192 +
5193 + // Plain numeric attachment ID (ACF File field set to "Return: ID")
5194 + if (is_numeric($value)) {
5195 + $att_id = (int) $value;
5196 + if ($att_id > 0 && get_post_mime_type($att_id) === 'application/pdf') {
5197 + $out[] = $att_id;
5198 + }
5199 + return;
5200 + }
5201 +
5202 + // Plain string — URL pointing at a PDF (ACF File field set to "Return: URL", or a custom URL/text field)
5203 + if (is_string($value)) {
5204 + $trimmed = trim($value);
5205 + if ($trimmed !== '' && $this->mxchat_url_looks_like_pdf($trimmed)) {
5206 + $att_id = (int) attachment_url_to_postid($trimmed);
5207 + if ($att_id > 0 && get_post_mime_type($att_id) === 'application/pdf') {
5208 + $out[] = $att_id;
5209 + }
5210 + }
5211 + return;
5212 + }
5213 +}
5214 +
5215 +/**
5216 + * Heuristic: does this URL/string look like a PDF reference?
5217 + * Tolerates query strings and fragments (#page=2).
5218 + */
5219 +private function mxchat_url_looks_like_pdf($url) {
5220 + if (!is_string($url) || $url === '') {
5221 + return false;
5222 + }
5223 + // Strip query + fragment before checking extension
5224 + $path = preg_replace('/[?#].*$/', '', $url);
5225 + return (bool) preg_match('/\.pdf$/i', $path);
5226 +}
5227 +
5228 +/**
5229 + * Extract text from a PDF attachment by ID using the bundled Smalot parser.
5230 + * Reads the file directly from disk via get_attached_file (no HTTP fetch).
5231 + * Result is cached on the attachment as post_meta keyed by file mtime so we
5232 + * only parse the same PDF once unless the file changes on disk.
5233 + *
5234 + * @param int $attachment_id
5235 + * @return string Extracted plain text, or '' on failure.
5236 + */
5237 +private function mxchat_extract_pdf_text_by_attachment_id($attachment_id) {
5238 + $attachment_id = (int) $attachment_id;
5239 + if ($attachment_id <= 0) {
5240 + return '';
5241 + }
5242 + if (get_post_mime_type($attachment_id) !== 'application/pdf') {
5243 + return '';
5244 + }
5245 +
5246 + $pdf_path = get_attached_file($attachment_id);
5247 + if (empty($pdf_path) || !file_exists($pdf_path) || !is_readable($pdf_path)) {
5248 + return '';
5249 + }
5250 +
5251 + // Raw-file size cap. Parsing very large PDFs can OOM the request; skip with a log entry
5252 + // and let the rest of the ACF content land in the KB. Filterable for users who need it bigger.
5253 + $default_max_bytes = 25 * 1024 * 1024;
5254 + $max_bytes = (int) apply_filters('mxchat_acf_pdf_max_bytes', $default_max_bytes, $attachment_id, $pdf_path);
5255 + if ($max_bytes > 0) {
5256 + $file_size = @filesize($pdf_path);
5257 + if ($file_size !== false && $file_size > $max_bytes) {
5258 + error_log(sprintf(
5259 + '[mxchat] ACF PDF skipped (over size cap): attachment %d "%s" %d bytes > cap %d',
5260 + $attachment_id,
5261 + basename($pdf_path),
5262 + $file_size,
5263 + $max_bytes
5264 + ));
5265 + return '';
5266 + }
5267 + }
5268 +
5269 + $mtime = @filemtime($pdf_path);
5270 + $cache_meta_key = '_mxchat_acf_pdf_text_v1';
5271 + $cached = get_post_meta($attachment_id, $cache_meta_key, true);
5272 + if (is_array($cached) && isset($cached['mtime'], $cached['text']) && (int) $cached['mtime'] === (int) $mtime) {
5273 + return (string) $cached['text'];
5274 + }
5275 +
5276 + $text = '';
5277 + try {
5278 + if (function_exists('mxchat_load_pdf_parser')) {
5279 + mxchat_load_pdf_parser();
5280 + }
5281 + if (!class_exists('\\Smalot\\PdfParser\\Parser')) {
5282 + return '';
5283 + }
5284 + $parser = new \Smalot\PdfParser\Parser();
5285 + $pdf = $parser->parseFile($pdf_path);
5286 + $pages = $pdf->getPages();
5287 + $page_texts = array();
5288 + foreach ($pages as $page) {
5289 + $page_text = '';
5290 + try {
5291 + $page_text = $page->getText();
5292 + } catch (\Exception $e) {
5293 + $page_text = '';
5294 + }
5295 + if (!empty($page_text)) {
5296 + $page_texts[] = $page_text;
5297 + }
5298 + }
5299 + $text = trim(implode("\n\n", $page_texts));
5300 + } catch (\Exception $e) {
5301 + error_log('[mxchat] ACF PDF extraction failed for attachment ' . $attachment_id . ': ' . $e->getMessage());
5302 + return '';
5303 + } catch (\Throwable $e) {
5304 + error_log('[mxchat] ACF PDF extraction error for attachment ' . $attachment_id . ': ' . $e->getMessage());
5305 + return '';
5306 + }
5307 +
5308 + // Cap per-PDF text to avoid blowing up the embedding payload on enormous PDFs.
5309 + // The chunker downstream will still split this into multiple vectors.
5310 + $max_len = (int) apply_filters('mxchat_acf_pdf_text_max_length', 50000);
5311 + if ($max_len > 0 && strlen($text) > $max_len) {
5312 + $text = substr($text, 0, $max_len);
5313 + }
5314 +
5315 + update_post_meta($attachment_id, $cache_meta_key, array(
5316 + 'mtime' => (int) $mtime,
5317 + 'text' => $text,
5318 + ));
5319 +
5320 + return $text;
5321 +}
5322 +
5323 +/**
5077 5324 * Handle ACF save - fires after ACF fields are saved
5078 5325 * This ensures ACF field data is available when syncing to knowledge base
5079 5326 */
5080 5327 public function mxchat_handle_acf_save($post_id) {
@@ -5317,8 +5564,9 @@
5317 5564 // ADD ACF FIELDS SUPPORT (matches ajax_mxchat_process_selected_content behavior)
5318 5565 $acf_fields = $this->mxchat_get_acf_fields_for_post($post_id);
5319 5566 if (!empty($acf_fields)) {
5320 5567 $acf_content_parts = array();
5568 + $pdf_attachment_ids = array();
5321 5569
5322 5570 foreach ($acf_fields as $field_name => $field_value) {
5323 5571 $formatted_value = $this->mxchat_format_acf_field_value($field_value, $field_name, $post_id);
5324 5572 if (!empty($formatted_value)) {
@@ -5325,12 +5573,43 @@
5325 5573 // Convert field name to readable label
5326 5574 $field_label = ucwords(str_replace(['_', '-'], ' ', $field_name));
5327 5575 $acf_content_parts[] = $field_label . ": " . $formatted_value;
5328 5576 }
5577 +
5578 + $this->mxchat_collect_pdf_attachment_ids_from_acf_value($field_value, $pdf_attachment_ids);
5329 5579 }
5330 5580
5331 5581 if (!empty($acf_content_parts)) {
5332 5582 $final_content .= "\n\n" . implode("\n", $acf_content_parts);
5583 + }
5584 +
5585 + // Gate the auto-sync PDF-extraction loop behind an opt-in option.
5586 + // Mirrors the per-batch checkbox the manual content selector has; the
5587 + // 25 MB size cap lives in the shared extractor so it applies in both
5588 + // paths regardless. Default OFF — re-parsing every ACF PDF on every
5589 + // editor save is expensive and most sites don't want it.
5590 + $autosync_extract_acf_pdfs = get_option('mxchat_auto_sync_acf_pdfs', '0') === '1';
5591 + if ($autosync_extract_acf_pdfs && !empty($pdf_attachment_ids)) {
5592 + $pdf_attachment_ids = array_unique(array_filter(array_map('intval', $pdf_attachment_ids)));
5593 + $pdf_sections = array();
5594 + foreach ($pdf_attachment_ids as $att_id) {
5595 + $pdf_text = $this->mxchat_extract_pdf_text_by_attachment_id($att_id);
5596 + if (!empty($pdf_text)) {
5597 + $pdf_title = get_the_title($att_id);
5598 + $pdf_url = wp_get_attachment_url($att_id);
5599 + $header = 'PDF Attachment';
5600 + if (!empty($pdf_title)) {
5601 + $header .= ': ' . $pdf_title;
5602 + }
5603 + if (!empty($pdf_url)) {
5604 + $header .= ' (' . $pdf_url . ')';
5605 + }
5606 + $pdf_sections[] = $header . "\n" . $pdf_text;
5607 + }
5608 + }
5609 + if (!empty($pdf_sections)) {
5610 + $final_content .= "\n\nAttached PDFs:\n" . implode("\n\n", $pdf_sections);
5611 + }
5333 5612 }
5334 5613 }
5335 5614
5336 5615 // ADD CUSTOM POST META SUPPORT (whitelisted non-ACF meta fields)