| @@ -5,51 +5,8 @@ | ||
| 5 | 5 | |
| 6 | 6 | class MxChat_Utils { |
| 7 | 7 | |
| 8 | 8 | /** |
| 9 | - * Centralized embedding model registry. Single source of truth for dimensions | |
| 10 | - * and provider, so model-switch protection logic doesn't drift across files. | |
| 11 | - */ | |
| 12 | -public static function embedding_model_registry() { | |
| 13 | - return array( | |
| 14 | - 'text-embedding-ada-002' => array('dims' => 1536, 'provider' => 'openai', 'label' => 'Ada 2'), | |
| 15 | - 'text-embedding-3-small' => array('dims' => 1536, 'provider' => 'openai', 'label' => 'TE3 Small'), | |
| 16 | - 'text-embedding-3-large' => array('dims' => 3072, 'provider' => 'openai', 'label' => 'TE3 Large'), | |
| 17 | - 'voyage-3-large' => array('dims' => 2048, 'provider' => 'voyage', 'label' => 'Voyage-3 Large'), | |
| 18 | - 'gemini-embedding-001' => array('dims' => 1536, 'provider' => 'gemini', 'label' => 'Gemini Embedding'), | |
| 19 | - ); | |
| 20 | -} | |
| 21 | - | |
| 22 | -public static function embedding_model_dimensions($model) { | |
| 23 | - $registry = self::embedding_model_registry(); | |
| 24 | - return isset($registry[$model]) ? (int) $registry[$model]['dims'] : 0; | |
| 25 | -} | |
| 26 | - | |
| 27 | -public static function embedding_model_label($model) { | |
| 28 | - $registry = self::embedding_model_registry(); | |
| 29 | - return isset($registry[$model]) ? $registry[$model]['label'] : $model; | |
| 30 | -} | |
| 31 | - | |
| 32 | -/** | |
| 33 | - * Returns the model that was last used to actually write embeddings into the | |
| 34 | - * KB. Differs from the user-selected setting once a switch has happened but | |
| 35 | - * no re-embed has occurred yet — that's the mismatch state we warn about. | |
| 36 | - */ | |
| 37 | -public static function get_active_embedding_model() { | |
| 38 | - return get_option('mxchat_active_embedding_model', ''); | |
| 39 | -} | |
| 40 | - | |
| 41 | -/** | |
| 42 | - * Stamp the model that produced the most recent successful embedding. Called | |
| 43 | - * from generate_embedding() right after the API responds with a valid vector. | |
| 44 | - */ | |
| 45 | -public static function stamp_active_embedding_model($model) { | |
| 46 | - if (!empty($model) && $model !== self::get_active_embedding_model()) { | |
| 47 | - update_option('mxchat_active_embedding_model', $model, false); | |
| 48 | - } | |
| 49 | -} | |
| 50 | - | |
| 51 | -/** | |
| 52 | 9 | * UPDATED: Submit or update content (and its embedding) in the database. |
| 53 | 10 | * Stores in Pinecone if enabled, otherwise stores in WordPress DB. |
| 54 | 11 | * |
| 55 | 12 | * @param string $content The content to be embedded. |
| @@ -414,9 +371,9 @@ | ||
| 414 | 371 | 'source_url' => $url, // Can be empty for manual content |
| 415 | 372 | 'type' => $content_type, // Now supports: post, page, pdf, url, manual, product, etc. |
| 416 | 373 | 'last_updated' => time(), |
| 417 | 374 | 'created_at' => time(), // Add creation timestamp |
| 418 | - 'bot_id' => $bot_id, // Add bot identification | |
| 375 | + 'bot_id' => $bot_id // Add bot identification | |
| 419 | 376 | ); |
| 420 | 377 | |
| 421 | 378 | $vector_data = array( |
| 422 | 379 | 'id' => $vector_id, |
| @@ -564,9 +521,8 @@ | ||
| 564 | 521 | // Handle different response formats based on provider |
| 565 | 522 | if (strpos($selected_model, 'gemini-embedding') === 0) { |
| 566 | 523 | // Gemini API response format |
| 567 | 524 | if (isset($response_body['embedding']['values']) && is_array($response_body['embedding']['values'])) { |
| 568 | - self::stamp_active_embedding_model($selected_model); | |
| 569 | 525 | return $response_body['embedding']['values']; |
| 570 | 526 | } else { |
| 571 | 527 | //error_log('Invalid response received from Gemini embedding API for bot ' . $bot_id . ': ' . wp_json_encode($response_body)); |
| 572 | 528 | return null; |
| @@ -573,9 +529,8 @@ | ||
| 573 | 529 | } |
| 574 | 530 | } else { |
| 575 | 531 | // OpenAI/Voyage API response format |
| 576 | 532 | if (isset($response_body['data'][0]['embedding']) && is_array($response_body['data'][0]['embedding'])) { |
| 577 | - self::stamp_active_embedding_model($selected_model); | |
| 578 | 533 | return $response_body['data'][0]['embedding']; |
| 579 | 534 | } else { |
| 580 | 535 | //error_log('Invalid response received from embedding API for bot ' . $bot_id . ': ' . wp_json_encode($response_body)); |
| 581 | 536 | return null; |
| @@ -731,9 +686,9 @@ | ||
| 731 | 686 | 'total_chunks' => $chunk_metadata['total_chunks'], |
| 732 | 687 | 'parent_url_hash' => $chunk_metadata['parent_url_hash'], |
| 733 | 688 | 'last_updated' => time(), |
| 734 | 689 | 'created_at' => time(), |
| 735 | - 'bot_id' => $bot_id, | |
| 690 | + 'bot_id' => $bot_id | |
| 736 | 691 | ); |
| 737 | 692 | |
| 738 | 693 | $vector_data = array( |
| 739 | 694 | 'id' => $vector_id, |
| @@ -849,34 +804,31 @@ | ||
| 849 | 804 | |
| 850 | 805 | // Add the original single-vector ID (for non-chunked content) |
| 851 | 806 | $vectors_to_delete[] = $base_vector_id; |
| 852 | 807 | |
| 853 | - // Pinecone /vectors/list is a GET endpoint with query-string parameters; a POST here returns a | |
| 854 | - // non-200 silently and we end up only deleting the base vector, leaving chunks orphaned. | |
| 855 | - $query_params = array( | |
| 808 | + // Use Pinecone list API to find all chunk vectors with this prefix | |
| 809 | + $list_url = "https://{$host}/vectors/list"; | |
| 810 | + | |
| 811 | + $list_body = array( | |
| 856 | 812 | 'prefix' => $base_vector_id . '_chunk_', |
| 857 | - 'limit' => 100, | |
| 813 | + 'limit' => 100 | |
| 858 | 814 | ); |
| 815 | + | |
| 859 | 816 | if (!empty($namespace)) { |
| 860 | - $query_params['namespace'] = $namespace; | |
| 817 | + $list_body['namespace'] = $namespace; | |
| 861 | 818 | } |
| 862 | 819 | |
| 863 | - $list_url = "https://{$host}/vectors/list?" . http_build_query($query_params); | |
| 820 | + $list_response = wp_remote_post($list_url, array( | |
| 821 | + 'headers' => array( | |
| 822 | + 'Api-Key' => $api_key, | |
| 823 | + 'accept' => 'application/json', | |
| 824 | + 'content-type' => 'application/json' | |
| 825 | + ), | |
| 826 | + 'body' => wp_json_encode($list_body), | |
| 827 | + 'timeout' => 30 | |
| 828 | + )); | |
| 864 | 829 | |
| 865 | - // Paginate in case a URL has more than 100 chunks. | |
| 866 | - do { | |
| 867 | - $list_response = wp_remote_get($list_url, array( | |
| 868 | - 'headers' => array( | |
| 869 | - 'Api-Key' => $api_key, | |
| 870 | - 'accept' => 'application/json', | |
| 871 | - ), | |
| 872 | - 'timeout' => 30, | |
| 873 | - )); | |
| 874 | - | |
| 875 | - if (is_wp_error($list_response) || wp_remote_retrieve_response_code($list_response) !== 200) { | |
| 876 | - break; | |
| 877 | - } | |
| 878 | - | |
| 830 | + if (!is_wp_error($list_response)) { | |
| 879 | 831 | $list_data = json_decode(wp_remote_retrieve_body($list_response), true); |
| 880 | 832 | if (!empty($list_data['vectors'])) { |
| 881 | 833 | foreach ($list_data['vectors'] as $vector) { |
| 882 | 834 | if (isset($vector['id'])) { |
| @@ -883,17 +835,9 @@ | ||
| 883 | 835 | $vectors_to_delete[] = $vector['id']; |
| 884 | 836 | } |
| 885 | 837 | } |
| 886 | 838 | } |
| 887 | - | |
| 888 | - $next_token = $list_data['pagination']['next'] ?? ''; | |
| 889 | - if (empty($next_token)) { | |
| 890 | - break; | |
| 891 | - } | |
| 892 | - | |
| 893 | - $query_params['paginationToken'] = $next_token; | |
| 894 | - $list_url = "https://{$host}/vectors/list?" . http_build_query($query_params); | |
| 895 | - } while (true); | |
| 839 | + } | |
| 896 | 840 | |
| 897 | 841 | if (empty($vectors_to_delete)) { |
| 898 | 842 | //error_log('[MXCHAT-CHUNK-DELETE] No vectors found to delete'); |
| 899 | 843 | return true; |