PluginProbe
MxChat – AI Chatbot & Content Generation for WordPress / 3.1.4
MxChat – AI Chatbot & Content Generation for WordPress v3.1.4
3.2.21 3.2.20 3.2.19 3.2.18 3.2.17 3.2.16 3.2.15 3.2.14 3.2.12 3.2.13 3.2.11 3.2.10 3.2.9 3.2.8 3.2.7 3.2.6 3.2.5 3.2.4 3.2.3 3.2.2 3.2.1 2.0.3 2.0.4 2.0.5 2.0.6 All 152 releases
← All changes | admin/class-knowledge-manager.php +127 -525 3.2.63.1.4 View file →
@@ -59,8 +59,11 @@
59 59 add_action('wp_ajax_mxchat_paginate_entries', array($this, 'ajax_mxchat_paginate_entries'));
60 60 add_action('wp_ajax_mxchat_get_entry_content', array($this, 'ajax_mxchat_get_entry_content'));
61 61 add_action('wp_ajax_mxchat_save_entry_content', array($this, 'ajax_mxchat_save_entry_content'));
62 62
63 + // Hook for content deletion
64 + add_action('mxchat_delete_content', array($this, 'mxchat_delete_from_pinecone_by_url'), 10, 1);
65 +
63 66 // WordPress post management hooks
64 67 add_action('pre_post_update', array($this, 'mxchat_store_pre_update_status'), 10, 2);
65 68 add_action('post_updated', array($this, 'mxchat_handle_post_update'), 10, 3);
66 69 add_action('before_delete_post', array($this, 'mxchat_handle_post_delete'));
@@ -160,13 +163,9 @@
160 163 public function mxchat_is_pdf_url($url, $response) {
161 164 $content_type = wp_remote_retrieve_header($response, 'content-type');
162 165 $file_extension = strtolower(pathinfo($url, PATHINFO_EXTENSION));
163 166
164 - // Check Content-Disposition header for .pdf filename (Google Drive sends this)
165 - $disposition = wp_remote_retrieve_header($response, 'content-disposition');
166 - $has_pdf_disposition = ! empty($disposition) && stripos($disposition, '.pdf') !== false;
167 -
168 - return strpos($content_type, 'pdf') !== false || $file_extension === 'pdf' || $has_pdf_disposition;
167 + return strpos($content_type, 'pdf') !== false || $file_extension === 'pdf';
169 168 }
170 169
171 170
172 171 public function mxchat_handle_pdf_for_knowledge_base($pdf_url, $response, $bot_id = 'default') {
@@ -860,17 +859,10 @@
860 859 }
861 860
862 861 // For manual entries (no source_url or mxchat:// prefix), delete the old entry by ID first
863 862 // so submit_content_to_db creates a replacement instead of a duplicate
864 - // Also treat legacy mxchat.ai source URLs as manual — old bug assigned the site URL to manual entries
865 - $is_legacy_manual = !empty($source_url) && strpos($source_url, 'mxchat.ai') !== false && strpos($source_url, 'mxchat://') !== 0;
866 - if ( $entry_id > 0 && (empty($source_url) || strpos($source_url, 'mxchat://') === 0 || $is_legacy_manual) ) {
863 + if ( $entry_id > 0 && (empty($source_url) || strpos($source_url, 'mxchat://') === 0) ) {
867 864 $wpdb->delete( $table, array( 'id' => $entry_id ), array( '%d' ) );
868 - // Clear legacy URL so submit_content_to_db generates a unique mxchat:// identifier
869 - // instead of reusing the shared URL (which would mass-delete other entries with the same URL)
870 - if ( $is_legacy_manual ) {
871 - $source_url = '';
872 - }
873 865 }
874 866
875 867 // Use the existing submit_content_to_db which handles chunking, Pinecone, and WP DB
876 868 $vector_id = ! empty($source_url) ? md5($source_url) : md5('mxchat_manual_' . $entry_id);
@@ -947,22 +939,9 @@
947 939 exit;
948 940 }
949 941
950 942 $submitted_url = esc_url_raw($_POST['sitemap_url']);
951 -
952 - // Convert Google Drive sharing URLs to direct download URLs
953 - if ( strpos($submitted_url, 'drive.google.com') !== false ) {
954 - $file_id = '';
955 - if ( preg_match('/[?&]id=([a-zA-Z0-9_-]+)/', $submitted_url, $m) ) {
956 - $file_id = $m[1];
957 - } elseif ( preg_match('#/file/d/([a-zA-Z0-9_-]+)#', $submitted_url, $m) ) {
958 - $file_id = $m[1];
959 - }
960 - if ( ! empty($file_id) ) {
961 - $submitted_url = 'https://drive.google.com/uc?export=download&id=' . $file_id;
962 - }
963 - }
964 -
943 +
965 944 // Get bot_id from form submission
966 945 $bot_id = isset($_POST['bot_id']) ? sanitize_key($_POST['bot_id']) : 'default';
967 946
968 947 // Get bot-specific options and validate API key
@@ -990,18 +969,10 @@
990 969 wp_safe_redirect(esc_url(admin_url('admin.php?page=mxchat-prompts')));
991 970 exit;
992 971 }
993 972
994 - // Fetch URL — use browser-like headers so servers with bot protection don't block us
995 - $response = wp_remote_get($submitted_url, array(
996 - 'timeout' => 30,
997 - 'sslverify' => false,
998 - 'user-agent' => 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36',
999 - 'headers' => array(
1000 - 'Accept' => 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
1001 - 'Accept-Language' => 'en-US,en;q=0.9',
1002 - ),
1003 - ));
973 + // Fetch URL
974 + $response = wp_remote_get($submitted_url, array('timeout' => 30));
1004 975
1005 976 if (is_wp_error($response) || wp_remote_retrieve_response_code($response) !== 200) {
1006 977 $error_message = is_wp_error($response) ? $response->get_error_message() : 'HTTP Status: ' . wp_remote_retrieve_response_code($response);
1007 978 set_transient('mxchat_admin_notice_error',
@@ -2835,12 +2806,11 @@
2835 2806 foreach ($primary_indexes as $path => $source) {
2836 2807 $url = trailingslashit($site_url) . $path;
2837 2808
2838 2809 $response = wp_remote_head($url, array(
2839 - 'timeout' => 10,
2810 + 'timeout' => 3, // Short timeout
2840 2811 'sslverify' => false,
2841 - 'redirection' => 1,
2842 - 'user-agent' => 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36',
2812 + 'redirection' => 1
2843 2813 ));
2844 2814
2845 2815 if (!is_wp_error($response) && wp_remote_retrieve_response_code($response) === 200) {
2846 2816 // Found a sitemap index - parse it to get sub-sitemaps
@@ -2898,15 +2868,10 @@
2898 2868 private function parse_sitemap_index($url) {
2899 2869 $sub_sitemaps = array();
2900 2870
2901 2871 $response = wp_remote_get($url, array(
2902 - 'timeout' => 30,
2903 - 'sslverify' => false,
2904 - 'user-agent' => 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36',
2905 - 'headers' => array(
2906 - 'Accept' => 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
2907 - 'Accept-Language' => 'en-US,en;q=0.9',
2908 - ),
2872 + 'timeout' => 5,
2873 + 'sslverify' => false
2909 2874 ));
2910 2875
2911 2876 if (is_wp_error($response)) {
2912 2877 return $sub_sitemaps;
@@ -2957,15 +2922,10 @@
2957 2922 * Get URL count from a sitemap
2958 2923 */
2959 2924 private function get_sitemap_url_count($url) {
2960 2925 $response = wp_remote_get($url, array(
2961 - 'timeout' => 30,
2962 - 'sslverify' => false,
2963 - 'user-agent' => 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36',
2964 - 'headers' => array(
2965 - 'Accept' => 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
2966 - 'Accept-Language' => 'en-US,en;q=0.9',
2967 - ),
2926 + 'timeout' => 10,
2927 + 'sslverify' => false
2968 2928 ));
2969 2929
2970 2930 if (is_wp_error($response)) {
2971 2931 return 0;
@@ -2988,11 +2948,10 @@
2988 2948 $sitemaps = array();
2989 2949 $robots_url = trailingslashit($site_url) . 'robots.txt';
2990 2950
2991 2951 $response = wp_remote_get($robots_url, array(
2992 - 'timeout' => 15,
2993 - 'sslverify' => false,
2994 - 'user-agent' => 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36',
2952 + 'timeout' => 5,
2953 + 'sslverify' => false
2995 2954 ));
2996 2955
2997 2956 if (is_wp_error($response)) {
2998 2957 return $sitemaps;
@@ -3482,22 +3441,9 @@
3482 3441 }
3483 3442
3484 3443 // Get bot_id from request
3485 3444 $bot_id = isset($_POST['bot_id']) ? sanitize_key($_POST['bot_id']) : 'default';
3486 -
3487 - // ACF→PDF extraction is opt-in per import batch. Persist the last-used value so users
3488 - // don't re-check on every batch; the default is OFF for installs that haven't set it.
3489 - $extract_acf_pdfs = !empty($_POST['extract_acf_pdfs']) && $_POST['extract_acf_pdfs'] !== 'false';
3490 - $mxchat_options = get_option('mxchat_options', array());
3491 - if (!is_array($mxchat_options)) {
3492 - $mxchat_options = array();
3493 - }
3494 - $prior_default = !empty($mxchat_options['acf_pdf_extract_default']);
3495 - if ($prior_default !== $extract_acf_pdfs) {
3496 - $mxchat_options['acf_pdf_extract_default'] = $extract_acf_pdfs ? 1 : 0;
3497 - update_option('mxchat_options', $mxchat_options);
3498 - }
3499 -
3445 +
3500 3446 // Process only ONE post at a time to avoid request size issues
3501 3447 $post_id = reset($post_ids);
3502 3448 $post = get_post($post_id);
3503 3449
@@ -3603,57 +3549,23 @@
3603 3549 }
3604 3550
3605 3551 // ADD ACF FIELDS SUPPORT
3606 3552 $acf_fields = $this->mxchat_get_acf_fields_for_post($post_id);
3607 - $pdf_extracted_count = 0;
3608 3553 if (!empty($acf_fields)) {
3609 3554 $acf_content_parts = array();
3610 - $pdf_attachment_ids = array();
3611 -
3555 +
3612 3556 foreach ($acf_fields as $field_name => $field_value) {
3613 3557 $formatted_value = $this->mxchat_format_acf_field_value($field_value, $field_name, $post_id);
3614 -
3558 +
3615 3559 if (!empty($formatted_value)) {
3616 3560 $field_label = ucwords(str_replace('_', ' ', $field_name));
3617 3561 $acf_content_parts[] = $field_label . ": " . $formatted_value;
3618 3562 }
3619 -
3620 - // Walk this field's value tree for any PDF attachment references and queue them for extraction.
3621 - // Only when the user opted into ACF→PDF extraction for this batch; otherwise the ACF text
3622 - // still lands in the KB but the heavier PDF parsing is skipped.
3623 - if ($extract_acf_pdfs) {
3624 - $this->mxchat_collect_pdf_attachment_ids_from_acf_value($field_value, $pdf_attachment_ids);
3625 - }
3626 3563 }
3627 -
3564 +
3628 3565 if (!empty($acf_content_parts)) {
3629 3566 $content .= "\n\n" . implode("\n", $acf_content_parts);
3630 3567 }
3631 -
3632 - // Extract text from each unique PDF found in ACF fields and append as a labeled section
3633 - if ($extract_acf_pdfs && !empty($pdf_attachment_ids)) {
3634 - $pdf_attachment_ids = array_unique(array_filter(array_map('intval', $pdf_attachment_ids)));
3635 - $pdf_sections = array();
3636 - foreach ($pdf_attachment_ids as $att_id) {
3637 - $pdf_text = $this->mxchat_extract_pdf_text_by_attachment_id($att_id);
3638 - if (!empty($pdf_text)) {
3639 - $pdf_title = get_the_title($att_id);
3640 - $pdf_url = wp_get_attachment_url($att_id);
3641 - $header = 'PDF Attachment';
3642 - if (!empty($pdf_title)) {
3643 - $header .= ': ' . $pdf_title;
3644 - }
3645 - if (!empty($pdf_url)) {
3646 - $header .= ' (' . $pdf_url . ')';
3647 - }
3648 - $pdf_sections[] = $header . "\n" . $pdf_text;
3649 - $pdf_extracted_count++;
3650 - }
3651 - }
3652 - if (!empty($pdf_sections)) {
3653 - $content .= "\n\nAttached PDFs:\n" . implode("\n\n", $pdf_sections);
3654 - }
3655 - }
3656 3568 }
3657 3569
3658 3570 // ADD CUSTOM POST META SUPPORT (whitelisted non-ACF meta fields)
3659 3571 $custom_meta = $this->mxchat_get_whitelisted_post_meta($post_id);
@@ -3783,9 +3695,8 @@
3783 3695 'title' => $post->post_title,
3784 3696 'operation_type' => $operation_type,
3785 3697 'vector_id' => $vector_id,
3786 3698 'acf_fields_found' => $acf_field_count,
3787 - 'pdf_extracted_count' => (int) $pdf_extracted_count,
3788 3699 'content_preview' => substr($content, 0, 100) . '...',
3789 3700 'bot_id' => $bot_id
3790 3701 ));
3791 3702 exit;
@@ -5120,197 +5031,8 @@
5120 5031 return implode(', ', array_filter($text_parts));
5121 5032 }
5122 5033
5123 5034 /**
5124 - * Walk an ACF field value tree and collect attachment IDs for any value that
5125 - * resolves to a PDF in the WordPress media library. Handles the three shapes
5126 - * ACF returns for File/Image/URL fields (array with ID+url, integer attachment ID,
5127 - * plain URL string), and recurses through repeater/group/flexible content.
5128 - *
5129 - * @param mixed $value The ACF field value (any depth)
5130 - * @param array $out Accumulator (passed by reference) for attachment IDs
5131 - * @param int $depth Recursion guard
5132 - */
5133 -private function mxchat_collect_pdf_attachment_ids_from_acf_value($value, &$out, $depth = 0) {
5134 - if ($depth > 6) {
5135 - return; // prevent runaway recursion on circular/very-deep structures
5136 - }
5137 -
5138 - if (empty($value)) {
5139 - return;
5140 - }
5141 -
5142 - // Array shapes: ACF File/Image return value=array; repeaters/groups are arrays of arrays
5143 - if (is_array($value)) {
5144 - // Direct File/Image-style array (has 'url' and usually 'ID' + 'mime_type')
5145 - $looks_like_attachment = isset($value['url']) || isset($value['ID']) || isset($value['id']);
5146 - if ($looks_like_attachment) {
5147 - $att_id = 0;
5148 - if (!empty($value['ID']) && is_numeric($value['ID'])) {
5149 - $att_id = (int) $value['ID'];
5150 - } elseif (!empty($value['id']) && is_numeric($value['id'])) {
5151 - $att_id = (int) $value['id'];
5152 - } elseif (!empty($value['url']) && is_string($value['url'])) {
5153 - $att_id = (int) attachment_url_to_postid($value['url']);
5154 - }
5155 -
5156 - $is_pdf = false;
5157 - if (!empty($value['mime_type']) && $value['mime_type'] === 'application/pdf') {
5158 - $is_pdf = true;
5159 - } elseif (!empty($value['subtype']) && strtolower((string) $value['subtype']) === 'pdf') {
5160 - $is_pdf = true;
5161 - } elseif (!empty($value['url']) && is_string($value['url']) && $this->mxchat_url_looks_like_pdf($value['url'])) {
5162 - $is_pdf = true;
5163 - } elseif ($att_id && get_post_mime_type($att_id) === 'application/pdf') {
5164 - $is_pdf = true;
5165 - }
5166 -
5167 - if ($is_pdf && $att_id && get_post_mime_type($att_id) === 'application/pdf') {
5168 - $out[] = $att_id;
5169 - }
5170 - // An array node that represents one attachment doesn't contain other
5171 - // attachments inside it — done with this branch.
5172 - return;
5173 - }
5174 -
5175 - // Recurse: repeater rows, flexible-content layouts, groups, etc.
5176 - foreach ($value as $sub) {
5177 - $this->mxchat_collect_pdf_attachment_ids_from_acf_value($sub, $out, $depth + 1);
5178 - }
5179 - return;
5180 - }
5181 -
5182 - // Plain numeric attachment ID (ACF File field set to "Return: ID")
5183 - if (is_numeric($value)) {
5184 - $att_id = (int) $value;
5185 - if ($att_id > 0 && get_post_mime_type($att_id) === 'application/pdf') {
5186 - $out[] = $att_id;
5187 - }
5188 - return;
5189 - }
5190 -
5191 - // Plain string — URL pointing at a PDF (ACF File field set to "Return: URL", or a custom URL/text field)
5192 - if (is_string($value)) {
5193 - $trimmed = trim($value);
5194 - if ($trimmed !== '' && $this->mxchat_url_looks_like_pdf($trimmed)) {
5195 - $att_id = (int) attachment_url_to_postid($trimmed);
5196 - if ($att_id > 0 && get_post_mime_type($att_id) === 'application/pdf') {
5197 - $out[] = $att_id;
5198 - }
5199 - }
5200 - return;
5201 - }
5202 -}
5203 -
5204 -/**
5205 - * Heuristic: does this URL/string look like a PDF reference?
5206 - * Tolerates query strings and fragments (#page=2).
5207 - */
5208 -private function mxchat_url_looks_like_pdf($url) {
5209 - if (!is_string($url) || $url === '') {
5210 - return false;
5211 - }
5212 - // Strip query + fragment before checking extension
5213 - $path = preg_replace('/[?#].*$/', '', $url);
5214 - return (bool) preg_match('/\.pdf$/i', $path);
5215 -}
5216 -
5217 -/**
5218 - * Extract text from a PDF attachment by ID using the bundled Smalot parser.
5219 - * Reads the file directly from disk via get_attached_file (no HTTP fetch).
5220 - * Result is cached on the attachment as post_meta keyed by file mtime so we
5221 - * only parse the same PDF once unless the file changes on disk.
5222 - *
5223 - * @param int $attachment_id
5224 - * @return string Extracted plain text, or '' on failure.
5225 - */
5226 -private function mxchat_extract_pdf_text_by_attachment_id($attachment_id) {
5227 - $attachment_id = (int) $attachment_id;
5228 - if ($attachment_id <= 0) {
5229 - return '';
5230 - }
5231 - if (get_post_mime_type($attachment_id) !== 'application/pdf') {
5232 - return '';
5233 - }
5234 -
5235 - $pdf_path = get_attached_file($attachment_id);
5236 - if (empty($pdf_path) || !file_exists($pdf_path) || !is_readable($pdf_path)) {
5237 - return '';
5238 - }
5239 -
5240 - // Raw-file size cap. Parsing very large PDFs can OOM the request; skip with a log entry
5241 - // and let the rest of the ACF content land in the KB. Filterable for users who need it bigger.
5242 - $default_max_bytes = 25 * 1024 * 1024;
5243 - $max_bytes = (int) apply_filters('mxchat_acf_pdf_max_bytes', $default_max_bytes, $attachment_id, $pdf_path);
5244 - if ($max_bytes > 0) {
5245 - $file_size = @filesize($pdf_path);
5246 - if ($file_size !== false && $file_size > $max_bytes) {
5247 - error_log(sprintf(
5248 - '[mxchat] ACF PDF skipped (over size cap): attachment %d "%s" %d bytes > cap %d',
5249 - $attachment_id,
5250 - basename($pdf_path),
5251 - $file_size,
5252 - $max_bytes
5253 - ));
5254 - return '';
5255 - }
5256 - }
5257 -
5258 - $mtime = @filemtime($pdf_path);
5259 - $cache_meta_key = '_mxchat_acf_pdf_text_v1';
5260 - $cached = get_post_meta($attachment_id, $cache_meta_key, true);
5261 - if (is_array($cached) && isset($cached['mtime'], $cached['text']) && (int) $cached['mtime'] === (int) $mtime) {
5262 - return (string) $cached['text'];
5263 - }
5264 -
5265 - $text = '';
5266 - try {
5267 - if (function_exists('mxchat_load_pdf_parser')) {
5268 - mxchat_load_pdf_parser();
5269 - }
5270 - if (!class_exists('\\Smalot\\PdfParser\\Parser')) {
5271 - return '';
5272 - }
5273 - $parser = new \Smalot\PdfParser\Parser();
5274 - $pdf = $parser->parseFile($pdf_path);
5275 - $pages = $pdf->getPages();
5276 - $page_texts = array();
5277 - foreach ($pages as $page) {
5278 - $page_text = '';
5279 - try {
5280 - $page_text = $page->getText();
5281 - } catch (\Exception $e) {
5282 - $page_text = '';
5283 - }
5284 - if (!empty($page_text)) {
5285 - $page_texts[] = $page_text;
5286 - }
5287 - }
5288 - $text = trim(implode("\n\n", $page_texts));
5289 - } catch (\Exception $e) {
5290 - error_log('[mxchat] ACF PDF extraction failed for attachment ' . $attachment_id . ': ' . $e->getMessage());
5291 - return '';
5292 - } catch (\Throwable $e) {
5293 - error_log('[mxchat] ACF PDF extraction error for attachment ' . $attachment_id . ': ' . $e->getMessage());
5294 - return '';
5295 - }
5296 -
5297 - // Cap per-PDF text to avoid blowing up the embedding payload on enormous PDFs.
5298 - // The chunker downstream will still split this into multiple vectors.
5299 - $max_len = (int) apply_filters('mxchat_acf_pdf_text_max_length', 50000);
5300 - if ($max_len > 0 && strlen($text) > $max_len) {
5301 - $text = substr($text, 0, $max_len);
5302 - }
5303 -
5304 - update_post_meta($attachment_id, $cache_meta_key, array(
5305 - 'mtime' => (int) $mtime,
5306 - 'text' => $text,
5307 - ));
5308 -
5309 - return $text;
5310 -}
5311 -
5312 -/**
5313 5035 * Handle ACF save - fires after ACF fields are saved
5314 5036 * This ensures ACF field data is available when syncing to knowledge base
5315 5037 */
5316 5038 public function mxchat_handle_acf_save($post_id) {
@@ -5418,33 +5140,40 @@
5418 5140 // If the post was previously published but is now not published, remove from knowledge base
5419 5141 if ($previous_status === 'publish' && $post->post_status !== 'publish') {
5420 5142 // Use the stored URL from when it was published, or fall back to current permalink
5421 5143 $source_url = $previous_url ?: get_permalink($post_id);
5144 +
5145 + if ($source_url) {
5146 + // Check if Pinecone is enabled
5147 + $pinecone_options = get_option('mxchat_pinecone_addon_options', array());
5148 + $use_pinecone = ($pinecone_options['mxchat_use_pinecone'] ?? '0') === '1';
5422 5149
5423 - if ($source_url) {
5424 - // Chunk-aware deletion (routes to Pinecone or WP DB and removes base + all chunks)
5425 - MxChat_Utils::delete_chunks_for_url($source_url, 'default');
5150 + if ($use_pinecone && !empty($pinecone_options['mxchat_pinecone_api_key'])) {
5151 + // Delete from Pinecone
5152 + $this->mxchat_delete_from_pinecone_by_url($source_url, $pinecone_options);
5153 + } else {
5154 + // Delete from WordPress DB
5155 + global $wpdb;
5156 + $table_name = $wpdb->prefix . 'mxchat_system_prompt_content';
5157 +
5158 + $result = $wpdb->delete(
5159 + $table_name,
5160 + array('source_url' => $source_url),
5161 + array('%s')
5162 + );
5163 + }
5426 5164 }
5427 -
5165 +
5428 5166 // Clean up the transients and exit early
5429 5167 delete_transient($previous_status_key);
5430 5168 delete_transient($previous_url_key);
5431 5169 return;
5432 5170 }
5433 -
5434 - // Slug/permalink rename while still published: delete the old vectors before upserting new ones.
5435 - // Without this, md5(old_url) vectors (base + chunks) would be orphaned under the stale URL.
5436 - if ($post->post_status === 'publish' && !empty($previous_url)) {
5437 - $current_url = get_permalink($post_id);
5438 - if ($current_url && $current_url !== $previous_url) {
5439 - MxChat_Utils::delete_chunks_for_url($previous_url, 'default');
5440 - }
5441 - }
5442 -
5171 +
5443 5172 // Store the current status for next time (if this is an update)
5444 5173 if ($update) {
5445 5174 set_transient($previous_status_key, $post->post_status, DAY_IN_SECONDS);
5446 -
5175 +
5447 5176 // If the post is currently published, also store its URL
5448 5177 if ($post->post_status === 'publish') {
5449 5178 $current_url = get_permalink($post_id);
5450 5179 set_transient($previous_url_key, $current_url, DAY_IN_SECONDS);
@@ -5553,9 +5282,8 @@
5553 5282 // ADD ACF FIELDS SUPPORT (matches ajax_mxchat_process_selected_content behavior)
5554 5283 $acf_fields = $this->mxchat_get_acf_fields_for_post($post_id);
5555 5284 if (!empty($acf_fields)) {
5556 5285 $acf_content_parts = array();
5557 - $pdf_attachment_ids = array();
5558 5286
5559 5287 foreach ($acf_fields as $field_name => $field_value) {
5560 5288 $formatted_value = $this->mxchat_format_acf_field_value($field_value, $field_name, $post_id);
5561 5289 if (!empty($formatted_value)) {
@@ -5562,44 +5290,13 @@
5562 5290 // Convert field name to readable label
5563 5291 $field_label = ucwords(str_replace(['_', '-'], ' ', $field_name));
5564 5292 $acf_content_parts[] = $field_label . ": " . $formatted_value;
5565 5293 }
5566 -
5567 - $this->mxchat_collect_pdf_attachment_ids_from_acf_value($field_value, $pdf_attachment_ids);
5568 5294 }
5569 5295
5570 5296 if (!empty($acf_content_parts)) {
5571 5297 $final_content .= "\n\n" . implode("\n", $acf_content_parts);
5572 5298 }
5573 -
5574 - // Gate the auto-sync PDF-extraction loop behind an opt-in option.
5575 - // Mirrors the per-batch checkbox the manual content selector has; the
5576 - // 25 MB size cap lives in the shared extractor so it applies in both
5577 - // paths regardless. Default OFF — re-parsing every ACF PDF on every
5578 - // editor save is expensive and most sites don't want it.
5579 - $autosync_extract_acf_pdfs = get_option('mxchat_auto_sync_acf_pdfs', '0') === '1';
5580 - if ($autosync_extract_acf_pdfs && !empty($pdf_attachment_ids)) {
5581 - $pdf_attachment_ids = array_unique(array_filter(array_map('intval', $pdf_attachment_ids)));
5582 - $pdf_sections = array();
5583 - foreach ($pdf_attachment_ids as $att_id) {
5584 - $pdf_text = $this->mxchat_extract_pdf_text_by_attachment_id($att_id);
5585 - if (!empty($pdf_text)) {
5586 - $pdf_title = get_the_title($att_id);
5587 - $pdf_url = wp_get_attachment_url($att_id);
5588 - $header = 'PDF Attachment';
5589 - if (!empty($pdf_title)) {
5590 - $header .= ': ' . $pdf_title;
5591 - }
5592 - if (!empty($pdf_url)) {
5593 - $header .= ' (' . $pdf_url . ')';
5594 - }
5595 - $pdf_sections[] = $header . "\n" . $pdf_text;
5596 - }
5597 - }
5598 - if (!empty($pdf_sections)) {
5599 - $final_content .= "\n\nAttached PDFs:\n" . implode("\n\n", $pdf_sections);
5600 - }
5601 - }
5602 5299 }
5603 5300
5604 5301 // ADD CUSTOM POST META SUPPORT (whitelisted non-ACF meta fields)
5605 5302 $custom_meta = $this->mxchat_get_whitelisted_post_meta($post_id);
@@ -5706,14 +5403,12 @@
5706 5403 if (!$should_sync) {
5707 5404 return;
5708 5405 }
5709 5406
5710 - // Resolve the pre-trash URL. wp_trash_post renames the slug with "__trashed" before firing
5711 - // this hook, so get_permalink() here would return the trashed URL and md5() would miss the
5712 - // real vector IDs stored under the original URL.
5713 - $source_url = $this->mxchat_resolve_pre_trash_url($post_id);
5407 + // Get the URL before post is deleted
5408 + $source_url = get_permalink($post_id);
5714 5409 if (!$source_url) {
5715 - //error_log('MXChat: Failed to resolve source URL for post ' . $post_id);
5410 + //error_log('MXChat: Failed to get permalink for post ' . $post_id);
5716 5411 return;
5717 5412 }
5718 5413
5719 5414 // Use chunk-aware deletion (handles both chunked and non-chunked content)
@@ -5721,36 +5416,56 @@
5721 5416
5722 5417 if (is_wp_error($delete_result)) {
5723 5418 //error_log('MXChat: Chunk-aware deletion failed for URL: ' . $source_url . ' - ' . $delete_result->get_error_message());
5724 5419 }
5420 +}
5725 5421
5726 - delete_transient('mxchat_prev_url_' . $post_id);
5727 - delete_transient('mxchat_prev_status_' . $post_id);
5728 -}
5729 5422
5730 -/**
5731 - * Resolve the source URL for a post being trashed/deleted.
5732 - *
5733 - * Why: wp_trash_post appends "__trashed" to the slug before the wp_trash_post action fires, so
5734 - * get_permalink() returns a URL whose md5() won't match the vector IDs stored in Pinecone or
5735 - * the source_url rows in the WP DB. Prefer the URL captured by mxchat_store_pre_update_status
5736 - * (runs on pre_post_update, before the rename); fall back to stripping the __trashed suffix.
5737 - */
5738 -private function mxchat_resolve_pre_trash_url($post_id) {
5739 - $previous_url = get_transient('mxchat_prev_url_' . $post_id);
5740 - if (!empty($previous_url)) {
5741 - return $previous_url;
5742 - }
5423 + /**
5424 + * Deletes data from Pinecone using a source URL
5425 + */
5426 + public function mxchat_delete_from_pinecone_by_url($source_url, $pinecone_options) {
5427 + $host = $pinecone_options['mxchat_pinecone_host'] ?? '';
5428 + $api_key = $pinecone_options['mxchat_pinecone_api_key'] ?? '';
5743 5429
5744 - $current = get_permalink($post_id);
5745 - if (!$current) {
5746 - return '';
5747 - }
5748 - return preg_replace('#__trashed(/?)$#', '$1', $current);
5749 -}
5430 + if (empty($host) || empty($api_key)) {
5431 + //error_log('MXChat: Pinecone deletion failed - missing configuration');
5432 + return false;
5433 + }
5750 5434
5435 + $api_endpoint = "https://{$host}/vectors/delete";
5436 + $vector_id = md5($source_url);
5751 5437
5438 + $request_body = array(
5439 + 'ids' => array($vector_id)
5440 + );
5752 5441
5442 + $response = wp_remote_post($api_endpoint, array(
5443 + 'headers' => array(
5444 + 'Api-Key' => $api_key,
5445 + 'accept' => 'application/json',
5446 + 'content-type' => 'application/json'
5447 + ),
5448 + 'body' => wp_json_encode($request_body),
5449 + 'timeout' => 30
5450 + ));
5451 +
5452 + if (is_wp_error($response)) {
5453 + //error_log('MXChat: Pinecone deletion error - ' . $response->get_error_message());
5454 + return false;
5455 + }
5456 +
5457 + $response_code = wp_remote_retrieve_response_code($response);
5458 + if ($response_code !== 200) {
5459 + //error_log('MXChat: Pinecone deletion failed with status ' . $response_code);
5460 + return false;
5461 + }
5462 +
5463 + return true;
5464 + }
5465 +
5466 +
5467 +
5753 5468 public function mxchat_handle_product_change($post_id, $post, $update) {
5754 5469 if ($post->post_type !== 'product') {
5755 5470 return;
5756 5471 }
@@ -5902,18 +5617,28 @@
5902 5617 if (get_post_type($post_id) !== 'product') {
5903 5618 return;
5904 5619 }
5905 5620
5906 - $source_url = $this->mxchat_resolve_pre_trash_url($post_id);
5907 - if (!$source_url) {
5908 - return;
5909 - }
5621 + $source_url = get_permalink($post_id);
5910 5622
5911 - // Chunk-aware deletion (routes to Pinecone or WP DB and removes base + all chunks)
5912 - MxChat_Utils::delete_chunks_for_url($source_url, 'default');
5623 + // Check if Pinecone is enabled
5624 + $pinecone_options = get_option('mxchat_pinecone_addon_options', array());
5625 + $use_pinecone = ($pinecone_options['mxchat_use_pinecone'] ?? '0') === '1';
5913 5626
5914 - delete_transient('mxchat_prev_url_' . $post_id);
5915 - delete_transient('mxchat_prev_status_' . $post_id);
5627 + if ($use_pinecone && !empty($pinecone_options['mxchat_pinecone_api_key'])) {
5628 + // Delete from Pinecone
5629 + $this->mxchat_delete_from_pinecone_by_url($source_url, $pinecone_options);
5630 + } else {
5631 + // Delete from WordPress DB
5632 + global $wpdb;
5633 + $table_name = $wpdb->prefix . 'mxchat_system_prompt_content';
5634 +
5635 + $wpdb->delete(
5636 + $table_name,
5637 + array('source_url' => $source_url),
5638 + array('%s')
5639 + );
5640 + }
5916 5641 }
5917 5642
5918 5643 /**
5919 5644 * Handle individual Pinecone content deletion
@@ -7324,27 +7049,13 @@
7324 7049
7325 7050 $result = false;
7326 7051 $error_message = '';
7327 7052
7328 - // Read item directly from DB to get queue_id and preserve special chars in item_data
7329 - // (POST round-trip through JS mangles characters like apostrophes in URLs)
7330 - $db_item = $wpdb->get_row($wpdb->prepare(
7331 - "SELECT queue_id, item_data FROM $table_name WHERE id = %d",
7332 - $item_id
7333 - ));
7334 - $item_queue_id = $db_item ? $db_item->queue_id : '';
7335 - if ($db_item && !empty($db_item->item_data)) {
7336 - $db_data = json_decode($db_item->item_data, true);
7337 - if (is_array($db_data)) {
7338 - $item_data = $db_data;
7339 - }
7340 - }
7341 -
7342 7053 switch ($item_type) {
7343 7054 case 'url':
7344 - $result = $this->mxchat_process_queue_url($item_data, $bot_id, $item_queue_id);
7055 + $result = $this->mxchat_process_queue_url($item_data, $bot_id);
7345 7056 break;
7346 -
7057 +
7347 7058 case 'pdf_page':
7348 7059 $result = $this->mxchat_process_queue_pdf_page($item_data, $bot_id);
7349 7060 break;
7350 7061
@@ -7352,37 +7063,11 @@
7352 7063 throw new Exception('Unknown item type: ' . $item_type);
7353 7064 }
7354 7065
7355 7066 if (is_wp_error($result)) {
7356 - $error_code = $result->get_error_code();
7357 - // Content errors (empty page, sanitization) are permanent — retrying won't help
7358 - $permanent_codes = array('empty_page', 'empty_after_sanitization', 'no_api_key', 'page_not_found');
7359 - if (in_array($error_code, $permanent_codes)) {
7360 - // Mark as permanently failed — set attempts = max_attempts so it won't be retried
7361 - $current_item = $wpdb->get_row($wpdb->prepare(
7362 - "SELECT max_attempts FROM $table_name WHERE id = %d", $item_id
7363 - ));
7364 - $wpdb->update(
7365 - $table_name,
7366 - array(
7367 - 'status' => 'failed',
7368 - 'error_message' => $result->get_error_message(),
7369 - 'attempts' => $current_item ? $current_item->max_attempts : 3
7370 - ),
7371 - array('id' => $item_id),
7372 - array('%s', '%s', '%d'),
7373 - array('%d')
7374 - );
7375 - wp_send_json_error(array(
7376 - 'message' => $result->get_error_message(),
7377 - 'permanent_failure' => true,
7378 - 'item_id' => $item_id
7379 - ));
7380 - return;
7381 - }
7382 7067 throw new Exception($result->get_error_message());
7383 7068 }
7384 -
7069 +
7385 7070 if ($result === false) {
7386 7071 throw new Exception('Processing returned false - item may be empty or invalid');
7387 7072 }
7388 7073
@@ -7458,9 +7143,9 @@
7458 7143
7459 7144 /**
7460 7145 * Process a URL from the queue
7461 7146 */
7462 -private function mxchat_process_queue_url($item_data, $bot_id = 'default', $queue_id = '') {
7147 +private function mxchat_process_queue_url($item_data, $bot_id = 'default') {
7463 7148 $url = isset($item_data['url']) ? $item_data['url'] : '';
7464 7149
7465 7150 if (empty($url)) {
7466 7151 return new WP_Error('invalid_url', 'URL is empty');
@@ -7508,9 +7193,9 @@
7508 7193
7509 7194 // Fetch URL content (fallback for non-products or when WooCommerce extraction fails)
7510 7195 $is_likely_pdf = (strtolower(pathinfo(parse_url($url, PHP_URL_PATH) ?: '', PATHINFO_EXTENSION)) === 'pdf');
7511 7196 $response = wp_remote_get($url, array(
7512 - 'timeout' => $is_likely_pdf ? 120 : 30,
7197 + 'timeout' => $is_likely_pdf ? 60 : 30,
7513 7198 'redirection' => 5,
7514 7199 'user-agent' => 'MxChat/1.0'
7515 7200 ));
7516 7201
@@ -7519,14 +7204,14 @@
7519 7204 }
7520 7205
7521 7206 $response_code = wp_remote_retrieve_response_code($response);
7522 7207 if ($response_code !== 200) {
7523 - return new WP_Error('http_error', 'HTTP ' . $response_code . ' error for: ' . $url);
7208 + return new WP_Error('http_error', 'HTTP ' . $response_code . ' error');
7524 7209 }
7525 7210
7526 - // Check if URL is a PDF — expand into per-page queue items using the standard PDF pipeline
7211 + // Check if URL is a PDF — process through PDF pipeline instead of HTML
7527 7212 if ($this->mxchat_is_pdf_url($url, $response)) {
7528 - return $this->mxchat_expand_pdf_to_queue($url, $response, $bot_id, $queue_id);
7213 + return $this->mxchat_process_pdf_url_inline($url, $response, $api_key, $bot_id);
7529 7214 }
7530 7215
7531 7216 $html = wp_remote_retrieve_body($response);
7532 7217
@@ -7556,89 +7241,11 @@
7556 7241 return $result;
7557 7242 }
7558 7243
7559 7244 /**
7560 - * Expand a PDF URL into per-page queue items using the standard PDF pipeline.
7561 - * Called when a sitemap URL turns out to be a PDF — downloads, parses page count,
7562 - * and adds pdf_page items to the same queue so they process with full progress tracking.
7245 + * Process a PDF URL inline during sitemap queue processing.
7246 + * Downloads the PDF, extracts all pages, and submits each to the DB.
7563 7247 */
7564 -private function mxchat_expand_pdf_to_queue($pdf_url, $response, $bot_id = 'default', $queue_id = '') {
7565 - set_time_limit(120); // PDFs need extra time for download + parsing
7566 -
7567 - $upload_dir = wp_upload_dir();
7568 - $pdf_filename = sanitize_file_name('mxchat_kb_' . md5($pdf_url) . '.pdf');
7569 - $pdf_path = trailingslashit($upload_dir['path']) . $pdf_filename;
7570 -
7571 - $response_body = wp_remote_retrieve_body($response);
7572 - if (empty($response_body)) {
7573 - return new WP_Error('empty_pdf', 'Empty PDF response for: ' . $pdf_url);
7574 - }
7575 -
7576 - if (!wp_mkdir_p(dirname($pdf_path))) {
7577 - return new WP_Error('dir_error', 'Failed to create upload directory');
7578 - }
7579 -
7580 - file_put_contents($pdf_path, $response_body);
7581 -
7582 - if (!file_exists($pdf_path)) {
7583 - return new WP_Error('save_error', 'Failed to save PDF file');
7584 - }
7585 -
7586 - try {
7587 - $total_pages = $this->mxchat_validate_and_count_pdf_pages($pdf_path);
7588 -
7589 - if ($total_pages === false || $total_pages < 1) {
7590 - wp_delete_file($pdf_path);
7591 - return new WP_Error('no_pages', 'PDF has no pages: ' . $pdf_url);
7592 - }
7593 -
7594 - // Build per-page items identical to mxchat_handle_pdf_for_knowledge_base
7595 - $pages = array();
7596 - for ($i = 1; $i <= $total_pages; $i++) {
7597 - $pages[] = array(
7598 - 'pdf_path' => $pdf_path,
7599 - 'pdf_url' => $pdf_url,
7600 - 'page_number' => $i,
7601 - 'total_pages' => $total_pages
7602 - );
7603 - }
7604 -
7605 - // Add pdf_page items to the SAME queue so the JS picks them up automatically
7606 - if (!empty($queue_id)) {
7607 - $queued_count = $this->mxchat_add_to_queue($queue_id, 'pdf_page', $pages, $bot_id);
7608 - } else {
7609 - // Fallback: create a new PDF queue (shouldn't happen in sitemap flow)
7610 - $new_queue_id = 'pdf_' . md5($pdf_url . time());
7611 - $queued_count = $this->mxchat_add_to_queue($new_queue_id, 'pdf_page', $pages, $bot_id);
7612 - $this->mxchat_set_queue_meta($new_queue_id, 'source_url', $pdf_url);
7613 - $this->mxchat_set_queue_meta($new_queue_id, 'queue_type', 'pdf');
7614 - $this->mxchat_set_queue_meta($new_queue_id, 'total_items', $total_pages);
7615 - $this->mxchat_set_queue_meta($new_queue_id, 'bot_id', $bot_id);
7616 - $this->mxchat_set_queue_meta($new_queue_id, 'pdf_path', $pdf_path);
7617 - $this->mxchat_set_queue_meta($new_queue_id, 'created_at', current_time('mysql'));
7618 - }
7619 -
7620 - if ($queued_count === 0) {
7621 - wp_delete_file($pdf_path);
7622 - return new WP_Error('queue_error', 'Failed to add PDF pages to queue');
7623 - }
7624 -
7625 - // Return true so the original URL item is marked complete
7626 - // The new pdf_page items will be processed in subsequent batches
7627 - return true;
7628 -
7629 - } catch (Exception $e) {
7630 - if (file_exists($pdf_path)) {
7631 - wp_delete_file($pdf_path);
7632 - }
7633 - return new WP_Error('pdf_parse_error', 'Error parsing PDF: ' . $e->getMessage());
7634 - }
7635 -}
7636 -
7637 -/**
7638 - * Legacy: Process a PDF URL inline during sitemap queue processing.
7639 - * @deprecated Use mxchat_expand_pdf_to_queue instead — kept for reference only.
7640 - */
7641 7248 private function mxchat_process_pdf_url_inline($pdf_url, $response, $api_key, $bot_id = 'default') {
7642 7249 set_time_limit(120); // PDFs need more time — downloading + parsing all pages
7643 7250
7644 7251 $upload_dir = wp_upload_dir();
@@ -7672,24 +7279,21 @@
7672 7279 return new WP_Error('no_pages', 'PDF has no pages');
7673 7280 }
7674 7281
7675 7282 $processed = 0;
7676 - $skipped_pages = array();
7677 7283
7678 7284 for ($i = 0; $i < $total_pages; $i++) {
7679 - $page_num = $i + 1;
7680 7285 $text = $pages[$i]->getText();
7681 7286 if (empty($text)) {
7682 - $skipped_pages[] = 'Page ' . $page_num . ': No text could be extracted — page may contain only images, links, or non-standard encoding';
7683 7287 continue;
7684 7288 }
7685 7289
7686 7290 $sanitized = $this->mxchat_sanitize_content_for_api($text);
7687 7291 if (empty($sanitized)) {
7688 - $skipped_pages[] = 'Page ' . $page_num . ': Text was extracted but contained only special characters, control codes, or unsupported content';
7689 7292 continue;
7690 7293 }
7691 7294
7295 + $page_num = $i + 1;
7692 7296 $metadata = array(
7693 7297 'document_type' => 'pdf',
7694 7298 'total_pages' => $total_pages,
7695 7299 'current_page' => $page_num,
@@ -7713,12 +7317,8 @@
7713 7317
7714 7318 // Clean up the temp PDF file
7715 7319 wp_delete_file($pdf_path);
7716 7320
7717 - if (!empty($skipped_pages)) {
7718 - error_log('MxChat PDF: Skipped ' . count($skipped_pages) . ' of ' . $total_pages . ' pages: ' . implode('; ', $skipped_pages));
7719 - }
7720 -
7721 7321 return $processed > 0 ? true : false;
7722 7322
7723 7323 } catch (Exception $e) {
7724 7324 if (file_exists($pdf_path)) {
@@ -7883,17 +7483,18 @@
7883 7483 return new WP_Error('page_not_found', 'Page ' . $page_number . ' not found in PDF');
7884 7484 }
7885 7485
7886 7486 $text = $pages[$page_number - 1]->getText();
7887 -
7487 +
7888 7488 if (empty($text)) {
7889 - return new WP_Error('empty_page', 'Page ' . $page_number . ': No text could be extracted — page may contain only images, links, or non-standard encoding');
7489 + // Not an error - just an empty page
7490 + return false;
7890 7491 }
7891 -
7492 +
7892 7493 $sanitized = $this->mxchat_sanitize_content_for_api($text);
7893 -
7494 +
7894 7495 if (empty($sanitized)) {
7895 - return new WP_Error('empty_after_sanitization', 'Page ' . $page_number . ': Text was extracted but contained only special characters, control codes, or unsupported content that was removed during cleanup');
7496 + return false;
7896 7497 }
7897 7498
7898 7499 // Create metadata
7899 7500 $metadata = array(
@@ -7997,16 +7598,17 @@
7997 7598
7998 7599 // Calculate percentage
7999 7600 $percentage = $total > 0 ? round((($completed + $failed) / $total) * 100) : 0;
8000 7601
8001 - // Get failed items details (include all failed items, not just those that exhausted retries)
7602 + // Get failed items details
8002 7603 $failed_items = array();
8003 7604 if ($failed > 0) {
8004 7605 $failed_items = $wpdb->get_results($wpdb->prepare(
8005 - "SELECT item_type, item_data, error_message, attempts
8006 - FROM $table_name
8007 - WHERE queue_id = %s
7606 + "SELECT item_type, item_data, error_message, attempts
7607 + FROM $table_name
7608 + WHERE queue_id = %s
8008 7609 AND status = 'failed'
7610 + AND attempts >= max_attempts
8009 7611 ORDER BY id DESC
8010 7612 LIMIT 50",
8011 7613 $queue_id
8012 7614 ));