PluginProbe
MxChat – AI Chatbot & Content Generation for WordPress / 2.0.2
MxChat – AI Chatbot & Content Generation for WordPress v2.0.2
3.2.21 3.2.20 3.2.19 3.2.18 3.2.17 3.2.16 3.2.15 3.2.14 3.2.12 3.2.13 3.2.11 3.2.10 3.2.9 3.2.8 3.2.7 3.2.6 3.2.5 3.2.4 3.2.3 3.2.2 3.2.1 2.0.3 2.0.4 2.0.5 2.0.6 All 152 releases
mxchat-basic / includes / class-mxchat-word-handler.php

class-mxchat-word-handler.php in MxChat – AI Chatbot & Content Generation for WordPress 2.0.2, at includes/class-mxchat-word-handler.php

332 lines 11.2 KB
No matching file
Up and down to move Enter to open Esc to close
Raw Download Zip
1 <?php
2 /**
3 * Word document handler and processor for MXChat
4 * Can be directly bundled in WordPress plugins
5 */
6 class MXChat_Word_Handler {
7 private $temp_dir;
8 private $options;
9
10 public function __construct($options) {
11 $this->options = $options;
12 $this->temp_dir = wp_upload_dir()['path'];
13 }
14
15 /**
16 * Handle Word document upload and processing
17 */
18 public function mxchat_handle_word_upload() {
19 check_ajax_referer('mxchat_chat_nonce', 'nonce');
20
21 if (!isset($_FILES['word_file']) || !isset($_POST['session_id'])) {
22 wp_send_json_error(esc_html__('Missing required parameters.', 'mxchat'));
23 return;
24 }
25
26 $file = $_FILES['word_file'];
27 $session_id = sanitize_text_field($_POST['session_id']);
28 $original_filename = sanitize_text_field($file['name']);
29
30 // Check file type
31 $allowed_types = array(
32 'docx' => 'application/vnd.openxmlformats-officedocument.wordprocessingml.document'
33 );
34 $file_type = wp_check_filetype($file['name'], $allowed_types);
35
36 if (!$file_type['type']) {
37 wp_send_json_error(esc_html__('Invalid file type. Only .docx files are allowed.', 'mxchat'));
38 return;
39 }
40
41 // Generate unique filename
42 $word_filename = 'mxchat_word_' . $session_id . '_' . time() . '.docx';
43 $word_path = $this->temp_dir . '/' . $word_filename;
44
45 if (!move_uploaded_file($file['tmp_name'], $word_path)) {
46 wp_send_json_error(esc_html__('Failed to upload file.', 'mxchat'));
47 return;
48 }
49
50 $this->mxchat_clear_word_transients($session_id);
51
52 // Process the document
53 $embeddings = $this->mxchat_process_word_document($word_path);
54
55 if ($embeddings === false || empty($embeddings)) {
56 unlink($word_path);
57 $error_message = $this->options['word_intent_error_text'] ??
58 esc_html__('The uploaded document appears to be empty or contains unsupported content.', 'mxchat');
59 wp_send_json_error($error_message);
60 return;
61 }
62
63 // Store the embeddings and file information
64 set_transient('mxchat_word_url_' . $session_id, $word_path, HOUR_IN_SECONDS);
65 set_transient('mxchat_word_filename_' . $session_id, $original_filename, HOUR_IN_SECONDS);
66 set_transient('mxchat_word_embeddings_' . $session_id, $embeddings, HOUR_IN_SECONDS);
67 set_transient('mxchat_include_word_in_context_' . $session_id, true, HOUR_IN_SECONDS);
68
69 $success_message = $this->options['pdf_intent_success_text'] ??
70 __("I've processed the document. What questions do you have about it?", 'mxchat');
71
72 wp_send_json_success([
73 'message' => $success_message,
74 'filename' => $original_filename
75 ]);
76 }
77
78 /**
79 * Process Word document and generate embeddings
80 */
81 private function mxchat_process_word_document($file_path) {
82 // Get the maximum number of pages allowed from admin settings
83 $max_pages = isset($this->options['pdf_max_pages']) ? intval($this->options['pdf_max_pages']) : 69; // Use same setting as PDF
84
85 try {
86 $zip = new ZipArchive();
87 if ($zip->open($file_path) !== true) {
88 return false;
89 }
90
91 // Extract main document content
92 $content = $zip->getFromName('word/document.xml');
93 $zip->close();
94
95 if ($content === false) {
96 return false;
97 }
98
99 // Clean up the content
100 $text = $this->mxchat_clean_word_content($content);
101
102 // Count pages (roughly estimate based on paragraphs)
103 $paragraphs = explode("\n\n", $text);
104 $estimated_pages = ceil(count($paragraphs) / 3); // Assume ~3 paragraphs per page
105
106 if ($estimated_pages > $max_pages) {
107 return esc_html__('too_many_pages', 'mxchat');
108 }
109
110 // Split into chunks and continue processing...
111 $chunks = $this->mxchat_split_word_into_chunks($text, 1000);
112
113 $embeddings = [];
114 foreach ($chunks as $chunk_number => $chunk) {
115 if (empty(trim($chunk))) {
116 continue;
117 }
118
119 $embedding = $this->mxchat_generate_embedding_word(
120 esc_html__('Chunk ', 'mxchat') . ($chunk_number + 1) . ': ' . $chunk,
121 $this->options['api_key']
122 );
123
124 if ($embedding) {
125 $embeddings[] = [
126 'chunk_number' => $chunk_number + 1,
127 'embedding' => $embedding,
128 'text' => $chunk,
129 ];
130 }
131 }
132
133 return $embeddings;
134
135 } catch (Exception $e) {
136 return false;
137 }
138 }
139 /**
140 * Clean Word XML content
141 */
142 private function mxchat_clean_word_content($content) {
143 // Remove XML namespaces
144 $content = preg_replace('/xmlns[^=]*="[^"]*"/i', '', $content);
145
146 // Convert Word XML elements to text
147 $content = str_replace('</w:p>', "\n", $content);
148 $content = str_replace('</w:tr>', "\n", $content);
149
150 // Strip remaining XML tags
151 $content = strip_tags($content);
152
153 // Clean up whitespace
154 $content = preg_replace('/\s+/', ' ', $content);
155 $content = preg_replace('/\n\s*\n/', "\n\n", $content);
156
157 return trim($content);
158 }
159
160 /**
161 * Split text into manageable chunks
162 */
163 private function mxchat_split_word_into_chunks($text, $chunk_size) {
164 $chunks = [];
165 $paragraphs = explode("\n\n", $text);
166
167 $current_chunk = '';
168 foreach ($paragraphs as $paragraph) {
169 if (strlen($current_chunk) + strlen($paragraph) > $chunk_size) {
170 if (!empty($current_chunk)) {
171 $chunks[] = trim($current_chunk);
172 }
173 $current_chunk = $paragraph;
174 } else {
175 $current_chunk .= (!empty($current_chunk) ? "\n\n" : '') . $paragraph;
176 }
177 }
178
179 if (!empty($current_chunk)) {
180 $chunks[] = trim($current_chunk);
181 }
182
183 return $chunks;
184 }
185
186 /**
187 * Remove Word document and clean up transients
188 */
189 public function mxchat_handle_word_remove() {
190 check_ajax_referer('mxchat_chat_nonce', 'nonce');
191
192 if (empty($_POST['session_id'])) {
193 wp_send_json_error(esc_html__('Session ID missing.', 'mxchat'));
194 return;
195 }
196
197 $session_id = sanitize_text_field($_POST['session_id']);
198 $word_path = get_transient('mxchat_word_url_' . $session_id);
199
200 if ($word_path && file_exists($word_path)) {
201 unlink($word_path);
202 }
203
204 $this->mxchat_clear_word_transients($session_id);
205
206 wp_send_json_success([
207 'message' => esc_html__('Document removed successfully.', 'mxchat')
208 ]);
209 }
210
211 /**
212 * Clear all Word-related transients
213 */
214 private function mxchat_clear_word_transients($session_id) {
215 delete_transient('mxchat_word_url_' . $session_id);
216 delete_transient('mxchat_word_filename_' . $session_id);
217 delete_transient('mxchat_word_embeddings_' . $session_id);
218 delete_transient('mxchat_include_word_in_context_' . $session_id);
219 }
220
221 /**
222 * Find relevant chunks from the Word document
223 */
224 public function mxchat_find_relevant_word_chunks($query_embedding, $embeddings) {
225 $most_relevant = null;
226 $highest_similarity = -INF;
227
228 foreach ($embeddings as $chunk_data) {
229 $similarity = $this->mxchat_calculate_cosine_similarity_word($query_embedding, $chunk_data['embedding']);
230
231 if ($similarity > $highest_similarity) {
232 $highest_similarity = $similarity;
233 $most_relevant = $chunk_data['chunk_number'];
234 }
235 }
236
237 if (!is_null($most_relevant)) {
238 $chunk_numbers = range(
239 max(1, $most_relevant - 1),
240 min(count($embeddings), $most_relevant + 1)
241 );
242 return array_filter($embeddings, function ($chunk) use ($chunk_numbers) {
243 return in_array($chunk['chunk_number'], $chunk_numbers);
244 });
245 }
246
247 return [];
248 }
249
250 /**
251 * Handle Word document discussion similar to PDF discussion
252 */
253 public function mxchat_handle_word_discussion($message, $user_id, $session_id) {
254 // Get stored embeddings for the session
255 $embeddings = get_transient('mxchat_word_embeddings_' . $session_id);
256 $word_path = get_transient('mxchat_word_url_' . $session_id);
257
258 if (!$embeddings || !$word_path) {
259 $trigger_text = $this->options['word_intent_trigger_text'] ??
260 __("Please upload a Word document (.docx) that you'd like to discuss.", 'mxchat');
261 set_transient('mxchat_waiting_for_word_' . $session_id, true, HOUR_IN_SECONDS);
262 $this->fallbackResponse['text'] = $trigger_text;
263 return;
264 }
265
266 // Set context flag for including Word content in conversation
267 set_transient('mxchat_include_word_in_context_' . $session_id, true, HOUR_IN_SECONDS);
268 $this->fallbackResponse['text'] = ''; // Proceed without additional message
269 }
270
271
272 private function mxchat_generate_embedding_word($text, $api_key) {
273 $endpoint = 'https://api.openai.com/v1/embeddings';
274
275 $body = wp_json_encode([
276 'input' => $text,
277 'model' => 'text-embedding-ada-002'
278 ]);
279
280 $args = [
281 'body' => $body,
282 'headers' => [
283 'Content-Type' => 'application/json',
284 'Authorization' => 'Bearer ' . $api_key,
285 ],
286 'timeout' => 60,
287 'redirection' => 5,
288 'blocking' => true,
289 'httpversion' => '1.0',
290 'sslverify' => true,
291 ];
292
293 $response = wp_remote_post($endpoint, $args);
294
295 if (is_wp_error($response)) {
296 return null;
297 }
298
299 $response_body = json_decode(wp_remote_retrieve_body($response), true);
300
301 if (isset($response_body['data'][0]['embedding']) && is_array($response_body['data'][0]['embedding'])) {
302 return $response_body['data'][0]['embedding'];
303 } else {
304 return null;
305 }
306 }
307
308
309 private function mxchat_calculate_cosine_similarity_word($vectorA, $vectorB) {
310 if (!is_array($vectorA) || !is_array($vectorB) || empty($vectorA) || empty($vectorB)) {
311 return 0;
312 }
313
314 $dotProduct = array_sum(array_map(function ($a, $b) {
315 return $a * $b;
316 }, $vectorA, $vectorB));
317 $normA = sqrt(array_sum(array_map(function ($a) {
318 return $a * $a;
319 }, $vectorA)));
320 $normB = sqrt(array_sum(array_map(function ($b) {
321 return $b * $b;
322 }, $vectorB)));
323
324 if ($normA == 0 || $normB == 0) {
325 return 0;
326 }
327
328 return $dotProduct / ($normA * $normB);
329 }
330
331
332 }