| @@ -4,8 +4,9 @@ | ||
| 4 | 4 | |
| 5 | 5 | use WP_REST_Request; |
| 6 | 6 | use WPDeveloper\BetterDocs\Core\BaseAPI; |
| 7 | 7 | use WPDeveloper\BetterDocs\Utils\AIUsage; |
| 8 | +use WPDeveloper\BetterDocs\AI\ProviderFactory; | |
| 8 | 9 | |
| 9 | 10 | /** |
| 10 | 11 | * REST surface for the redesigned "Write with AI" modal. |
| 11 | 12 | * |
| @@ -25,8 +26,20 @@ | ||
| 25 | 26 | // 5 MB is generous for text/markdown/DOCX while capping abuse; the extracted |
| 26 | 27 | // text is still clipped to MAX_SOURCE_LENGTH before it reaches the model. |
| 27 | 28 | const MAX_UPLOAD_BYTES = 5242880; // 5 MB |
| 28 | 29 | |
| 30 | + // Recordings get their own, larger cap: 25 MB is OpenAI's transcription | |
| 31 | + // limit — roughly 25 minutes of mono MP3 — and there is no point accepting | |
| 32 | + // a file the provider will refuse. Gemini is capped lower still (see | |
| 33 | + // media_cap_for_platform): its media rides inline as base64, which inflates | |
| 34 | + // the payload by about a third. | |
| 35 | + // | |
| 36 | + // The cap sits 1 MB under that limit, not on it: the limit applies to the | |
| 37 | + // whole request, and a file of exactly 25 MB plus the multipart fields and | |
| 38 | + // boundaries would be refused after the full upload. | |
| 39 | + const MAX_MEDIA_BYTES = 25165824; // 24 MB | |
| 40 | + const MAX_MEDIA_BYTES_INLINE = 15728640; // 15 MB — inline-data platforms | |
| 41 | + | |
| 29 | 42 | public function register() { |
| 30 | 43 | $this->post( |
| 31 | 44 | '/write-with-ai', |
| 32 | 45 | array( $this, 'generate' ), |
| @@ -144,9 +157,9 @@ | ||
| 144 | 157 | |
| 145 | 158 | if ( empty( $write_ai->get_api_key() ) ) { |
| 146 | 159 | return $this->error( |
| 147 | 160 | 'missing_key', |
| 148 | - __( 'OpenAI API key is missing. Add one in BetterDocs settings.', 'betterdocs' ), | |
| 161 | + __( 'AI API key is missing. Add one in BetterDocs settings.', 'betterdocs' ), | |
| 149 | 162 | 400 |
| 150 | 163 | ); |
| 151 | 164 | } |
| 152 | 165 | |
| @@ -203,8 +216,9 @@ | ||
| 203 | 216 | $src_labels = array( |
| 204 | 217 | 'transcript' => __( 'support transcript', 'betterdocs' ), |
| 205 | 218 | 'forum' => __( 'forum thread', 'betterdocs' ), |
| 206 | 219 | 'notes' => __( 'raw notes', 'betterdocs' ), |
| 220 | + 'recording' => __( 'recording transcript', 'betterdocs' ), | |
| 207 | 221 | ); |
| 208 | 222 | $src_label = isset( $src_labels[ $src_type ] ) ? $src_labels[ $src_type ] : __( 'source material', 'betterdocs' ); |
| 209 | 223 | |
| 210 | 224 | // Light per-type framing: a one-line system hint steering how to treat |
| @@ -212,8 +226,12 @@ | ||
| 212 | 226 | $src_frames = array( |
| 213 | 227 | 'transcript' => __( 'The source below is a customer-support conversation. Focus on the user\'s problem and its resolution; ignore greetings and small talk.', 'betterdocs' ), |
| 214 | 228 | 'forum' => __( 'The source below is a forum discussion among multiple people. Treat the accepted or most-supported answer as authoritative and skip off-topic replies.', 'betterdocs' ), |
| 215 | 229 | 'notes' => __( 'The source below is rough notes. Expand them into clear, complete prose.', 'betterdocs' ), |
| 230 | + // Speech-to-text output reads nothing like written source: it | |
| 231 | + // has no punctuation discipline, keeps every "um", and may | |
| 232 | + // label speakers. Say so, or the model documents the filler. | |
| 233 | + 'recording' => __( 'The source below is a machine transcript of an audio or video recording. It may contain filler words, false starts, repetition and speaker labels — ignore those and document only the substance. The author has already reviewed it, so keep their spelling of names and product terms exactly as written.', 'betterdocs' ), | |
| 216 | 234 | ); |
| 217 | 235 | if ( isset( $src_frames[ $src_type ] ) ) { |
| 218 | 236 | array_unshift( $extra_system, array( 'role' => 'system', 'content' => $src_frames[ $src_type ] ) ); |
| 219 | 237 | } |
| @@ -225,10 +243,73 @@ | ||
| 225 | 243 | $src_label |
| 226 | 244 | ) |
| 227 | 245 | . "\n---\n" . $source . "\n---" ) |
| 228 | 246 | . $this->build_directives( $tone, $doc_size, $generate_title ); |
| 229 | - return $this->handle_doc( $write_ai, $post_id, $source_prompt, $keywords, $action, $doc_size, $extra_system ); | |
| 247 | + // A doc written from a recording's transcript counts as a recording | |
| 248 | + // in the insights, once — the transcribe step records nothing. | |
| 249 | + $usage_action = 'recording' === $src_type ? 'from-recording' : null; | |
| 250 | + return $this->handle_doc( $write_ai, $post_id, $source_prompt, $keywords, $action, $doc_size, $extra_system, null, $usage_action ); | |
| 230 | 251 | |
| 252 | + case 'transcribe-attachment': | |
| 253 | + // Step one of the recording flow: upload → transcript. No doc is | |
| 254 | + // written here. The transcript goes back to the modal so the | |
| 255 | + // author can fix mis-heard product names, and generation then | |
| 256 | + // runs through the ordinary `from-source` path with that text — | |
| 257 | + // which is why there is no media-shaped generation path at all. | |
| 258 | + $media = $this->read_uploaded_attachment( $request ); | |
| 259 | + if ( is_wp_error( $media ) ) { | |
| 260 | + return $this->error( $media->get_error_code() ?: 'ai_attachment_failed', $media->get_error_message(), 400 ); | |
| 261 | + } | |
| 262 | + | |
| 263 | + if ( ! isset( $media['kind'] ) || 'media' !== $media['kind'] ) { | |
| 264 | + return $this->error( 'ai_not_media', __( 'That file is not an audio or video recording.', 'betterdocs' ), 400 ); | |
| 265 | + } | |
| 266 | + | |
| 267 | + $transcript = $write_ai->transcribe( array( | |
| 268 | + 'path' => $media['path'], | |
| 269 | + 'filename' => $media['name'], | |
| 270 | + 'mime' => $media['mime'], | |
| 271 | + ) ); | |
| 272 | + | |
| 273 | + if ( is_wp_error( $transcript ) ) { | |
| 274 | + $code = $transcript->get_error_code() ?: 'ai_transcribe_failed'; | |
| 275 | + // A platform that cannot transcribe is the user's setting to | |
| 276 | + // change, not an upstream fault — 400, like ai_no_vision. | |
| 277 | + return $this->error( $code, $transcript->get_error_message(), 'ai_no_transcription' === $code ? 400 : 502 ); | |
| 278 | + } | |
| 279 | + | |
| 280 | + // Clip before returning, not after editing: the author should be | |
| 281 | + // correcting exactly the text the model will receive, rather than | |
| 282 | + // polishing a tail that gets silently cut on the way out. | |
| 283 | + // | |
| 284 | + // A clipped transcript says so in its own text: the author sees | |
| 285 | + // where it stops, and the marker travels with the text to the | |
| 286 | + // model, which would otherwise document half a recording as if it | |
| 287 | + // were the whole of it. | |
| 288 | + $transcript = wp_check_invalid_utf8( (string) $transcript, true ); | |
| 289 | + $clipped = strlen( $transcript ) > self::MAX_SOURCE_LENGTH; | |
| 290 | + | |
| 291 | + if ( $clipped ) { | |
| 292 | + $marker = "\n\n" . __( '[Transcript cut off here: the recording is longer than BetterDocs AI can use at once.]', 'betterdocs' ); | |
| 293 | + $transcript = rtrim( $this->clip( $transcript, self::MAX_SOURCE_LENGTH - strlen( $marker ) ) ) . $marker; | |
| 294 | + } | |
| 295 | + | |
| 296 | + if ( '' === trim( $transcript ) ) { | |
| 297 | + return $this->error( 'ai_no_speech', __( 'No speech was found in that recording.', 'betterdocs' ), 400 ); | |
| 298 | + } | |
| 299 | + | |
| 300 | + // No AIUsage::record() here. Nothing has been written yet — the doc | |
| 301 | + // is generated by the `from-source` call that follows, which records | |
| 302 | + // it (as a recording, via source_type). Counting both made every | |
| 303 | + // recording-based doc show up twice in the insights. | |
| 304 | + | |
| 305 | + return $this->success( array( | |
| 306 | + 'transcript' => $transcript, | |
| 307 | + 'clipped' => $clipped, | |
| 308 | + 'name' => $media['name'], | |
| 309 | + 'action' => $action, | |
| 310 | + ) ); | |
| 311 | + | |
| 231 | 312 | case 'from-attachment': |
| 232 | 313 | // Upload a file; extract its text server-side and treat it exactly |
| 233 | 314 | // like from-source (same "use only what it contains" contract and |
| 234 | 315 | // the same handle_doc → wp_kses_post output path). The file itself |
| @@ -238,8 +319,16 @@ | ||
| 238 | 319 | if ( is_wp_error( $extracted ) ) { |
| 239 | 320 | return $this->error( $extracted->get_error_code() ?: 'ai_attachment_failed', $extracted->get_error_message(), 400 ); |
| 240 | 321 | } |
| 241 | 322 | |
| 323 | + // A recording has no text to extract; it goes through | |
| 324 | + // `transcribe-attachment` first. Sent here directly (an older modal, | |
| 325 | + // or a hand-made request) it used to fall through to the text path | |
| 326 | + // and read an undefined `text` key. | |
| 327 | + if ( isset( $extracted['kind'] ) && 'media' === $extracted['kind'] ) { | |
| 328 | + return $this->error( 'ai_needs_transcript', __( 'Recordings are transcribed first. Use "Transcribe recording", then generate from the transcript.', 'betterdocs' ), 400 ); | |
| 329 | + } | |
| 330 | + | |
| 242 | 331 | // Image attachment → send the picture to a vision-capable model |
| 243 | 332 | // instead of extracting text (there is none). Same handle_doc |
| 244 | 333 | // output path (wp_kses_post), just a multimodal request. |
| 245 | 334 | if ( isset( $extracted['kind'] ) && 'image' === $extracted['kind'] ) { |
| @@ -374,9 +463,9 @@ | ||
| 374 | 463 | |
| 375 | 464 | /** |
| 376 | 465 | * Full-doc generation (generate-doc, expand-outline, from-source all land here). |
| 377 | 466 | */ |
| 378 | - protected function handle_doc( $write_ai, $post_id, $prompt, $keywords, $action, $doc_size = 'any', $extra_system = array(), $image = null ) { | |
| 467 | + protected function handle_doc( $write_ai, $post_id, $prompt, $keywords, $action, $doc_size = 'any', $extra_system = array(), $image = null, $usage_action = null ) { | |
| 379 | 468 | if ( '' === trim( $prompt ) ) { |
| 380 | 469 | return $this->error( 'ai_empty_prompt', __( 'Please provide a prompt for the AI.', 'betterdocs' ), 400 ); |
| 381 | 470 | } |
| 382 | 471 | |
| @@ -410,9 +499,11 @@ | ||
| 410 | 499 | // prompt asks the model to avoid these, but that is a soft constraint — this |
| 411 | 500 | // is the enforcement (a prompt-injected source/Git payload can't inject XSS). |
| 412 | 501 | $content = wp_kses_post( $content ); |
| 413 | 502 | |
| 414 | - AIUsage::record( 'write_with_ai', $post_id, $action ); | |
| 503 | + // `$usage_action` lets a caller count the generation under a different | |
| 504 | + // mode than the action it answers to (a recording arrives as from-source). | |
| 505 | + AIUsage::record( 'write_with_ai', $post_id, null !== $usage_action ? $usage_action : $action ); | |
| 415 | 506 | |
| 416 | 507 | return $this->success( array( 'content' => $content, 'action' => $action ) ); |
| 417 | 508 | } |
| 418 | 509 | |
| @@ -492,29 +583,21 @@ | ||
| 492 | 583 | if ( ! empty( $file['error'] ) || empty( $file['tmp_name'] ) || ! is_uploaded_file( $file['tmp_name'] ) ) { |
| 493 | 584 | return new \WP_Error( 'ai_upload_failed', __( 'The upload did not complete — please try again.', 'betterdocs' ) ); |
| 494 | 585 | } |
| 495 | 586 | |
| 496 | - if ( (int) $file['size'] > self::MAX_UPLOAD_BYTES ) { | |
| 497 | - return new \WP_Error( | |
| 498 | - 'ai_file_too_large', | |
| 499 | - sprintf( | |
| 500 | - /* translators: %s: maximum allowed size, e.g. "5 MB". */ | |
| 501 | - __( 'The file exceeds the %s limit.', 'betterdocs' ), | |
| 502 | - size_format( self::MAX_UPLOAD_BYTES ) | |
| 503 | - ) | |
| 504 | - ); | |
| 505 | - } | |
| 506 | - | |
| 507 | 587 | // Strict extension + MIME allow-list. wp_check_filetype() validates the |
| 508 | 588 | // name against exactly these types; anything else yields an empty ext. |
| 509 | - $allowed = array( | |
| 510 | - 'txt' => 'text/plain', | |
| 511 | - 'md|markdown' => 'text/markdown', | |
| 512 | - 'docx' => 'application/vnd.openxmlformats-officedocument.wordprocessingml.document', | |
| 513 | - 'pdf' => 'application/pdf', | |
| 514 | - 'png' => 'image/png', | |
| 515 | - 'jpg|jpeg' => 'image/jpeg', | |
| 516 | - 'webp' => 'image/webp', | |
| 589 | + $allowed = array_merge( | |
| 590 | + array( | |
| 591 | + 'txt' => 'text/plain', | |
| 592 | + 'md|markdown' => 'text/markdown', | |
| 593 | + 'docx' => 'application/vnd.openxmlformats-officedocument.wordprocessingml.document', | |
| 594 | + 'pdf' => 'application/pdf', | |
| 595 | + 'png' => 'image/png', | |
| 596 | + 'jpg|jpeg' => 'image/jpeg', | |
| 597 | + 'webp' => 'image/webp', | |
| 598 | + ), | |
| 599 | + self::media_mimes() | |
| 517 | 600 | ); |
| 518 | 601 | $check = wp_check_filetype( (string) $file['name'], $allowed ); |
| 519 | 602 | $ext = strtolower( (string) $check['ext'] ); |
| 520 | 603 | |
| @@ -519,13 +602,45 @@ | ||
| 519 | 602 | $ext = strtolower( (string) $check['ext'] ); |
| 520 | 603 | |
| 521 | 604 | $image_exts = array( 'png', 'jpg', 'jpeg', 'webp' ); |
| 522 | 605 | $text_exts = array( 'txt', 'md', 'markdown', 'docx', 'pdf' ); |
| 606 | + $media_exts = self::media_exts(); | |
| 523 | 607 | |
| 524 | - if ( ! in_array( $ext, array_merge( $text_exts, $image_exts ), true ) ) { | |
| 525 | - return new \WP_Error( 'ai_bad_filetype', __( 'Unsupported file type. Upload a .pdf, .docx, .txt, .md, or an image (.png, .jpg, .webp).', 'betterdocs' ) ); | |
| 608 | + if ( ! in_array( $ext, array_merge( $text_exts, $image_exts, $media_exts ), true ) ) { | |
| 609 | + // .mov and .avi are the two formats people actually try and that no | |
| 610 | + // provider accepts, so name the fix rather than listing types again. | |
| 611 | + $tried = strtolower( (string) pathinfo( (string) $file['name'], PATHINFO_EXTENSION ) ); | |
| 612 | + if ( in_array( $tried, array( 'mov', 'avi', 'wmv', 'mkv' ), true ) ) { | |
| 613 | + return new \WP_Error( | |
| 614 | + 'ai_bad_filetype', | |
| 615 | + sprintf( | |
| 616 | + /* translators: %s: the uploaded file's extension, e.g. "mov". */ | |
| 617 | + __( '.%s recordings are not supported. Export or convert it to MP4 and upload that.', 'betterdocs' ), | |
| 618 | + $tried | |
| 619 | + ) | |
| 620 | + ); | |
| 621 | + } | |
| 622 | + | |
| 623 | + return new \WP_Error( 'ai_bad_filetype', __( 'Unsupported file type. Upload a .pdf, .docx, .txt, .md, an image (.png, .jpg, .webp), or a recording (.mp3, .m4a, .wav, .flac, .ogg, .mp4, .mpeg, .webm).', 'betterdocs' ) ); | |
| 526 | 624 | } |
| 527 | 625 | |
| 626 | + // Size is capped per kind: a recording is legitimately much larger than a | |
| 627 | + // text file, but the cap can never exceed what the host will actually | |
| 628 | + // accept — PHP truncates a POST over post_max_size before we see it. | |
| 629 | + $is_media = in_array( $ext, $media_exts, true ); | |
| 630 | + $cap = self::upload_cap( $is_media ? 'media' : 'file' ); | |
| 631 | + | |
| 632 | + if ( (int) $file['size'] > $cap ) { | |
| 633 | + return new \WP_Error( | |
| 634 | + 'ai_file_too_large', | |
| 635 | + sprintf( | |
| 636 | + /* translators: %s: maximum allowed size, e.g. "25 MB". */ | |
| 637 | + __( 'The file exceeds the %s limit.', 'betterdocs' ), | |
| 638 | + size_format( $cap ) | |
| 639 | + ) | |
| 640 | + ); | |
| 641 | + } | |
| 642 | + | |
| 528 | 643 | // Image → send the picture itself to a vision model (there is no text to |
| 529 | 644 | // extract). Verify it is a real image by its bytes, not just its name, |
| 530 | 645 | // then hand back a base64 data URI for the multimodal request. |
| 531 | 646 | if ( in_array( $ext, $image_exts, true ) ) { |
| @@ -548,8 +663,41 @@ | ||
| 548 | 663 | 'data_uri' => 'data:' . $mime . ';base64,' . base64_encode( $raw ), |
| 549 | 664 | ); |
| 550 | 665 | } |
| 551 | 666 | |
| 667 | + // Recording → nothing to extract here; it goes to a speech-to-text model | |
| 668 | + // whole. Verify the bytes really are audio/video before handing a file to | |
| 669 | + // a paid endpoint: wp_check_filetype() above only read the *name*, so a | |
| 670 | + // renamed binary would otherwise sail through. | |
| 671 | + if ( $is_media ) { | |
| 672 | + $real = wp_check_filetype_and_ext( (string) $file['tmp_name'], (string) $file['name'], $allowed ); | |
| 673 | + $mime = ! empty( $real['type'] ) ? (string) $real['type'] : ''; | |
| 674 | + | |
| 675 | + if ( '' === $mime && function_exists( 'finfo_open' ) ) { | |
| 676 | + // wp_check_filetype_and_ext() only sniffs images and a short list | |
| 677 | + // of text formats; for A/V it hands back the name-based guess or | |
| 678 | + // nothing at all. finfo is the actual byte check. | |
| 679 | + $finfo = finfo_open( FILEINFO_MIME_TYPE ); | |
| 680 | + if ( $finfo ) { | |
| 681 | + $sniffed = finfo_file( $finfo, (string) $file['tmp_name'] ); | |
| 682 | + finfo_close( $finfo ); | |
| 683 | + $mime = is_string( $sniffed ) ? $sniffed : ''; | |
| 684 | + } | |
| 685 | + } | |
| 686 | + | |
| 687 | + if ( 0 !== strpos( $mime, 'audio/' ) && 0 !== strpos( $mime, 'video/' ) ) { | |
| 688 | + return new \WP_Error( 'ai_bad_media', __( 'That file is not a readable audio or video recording.', 'betterdocs' ) ); | |
| 689 | + } | |
| 690 | + | |
| 691 | + return array( | |
| 692 | + 'name' => sanitize_file_name( (string) $file['name'] ), | |
| 693 | + 'kind' => 'media', | |
| 694 | + 'mime' => $mime, | |
| 695 | + 'path' => (string) $file['tmp_name'], | |
| 696 | + 'size' => (int) $file['size'], | |
| 697 | + ); | |
| 698 | + } | |
| 699 | + | |
| 552 | 700 | $text = $this->extract_attachment_text( (string) $file['tmp_name'], $ext ); |
| 553 | 701 | if ( is_wp_error( $text ) ) { |
| 554 | 702 | return $text; |
| 555 | 703 | } |
| @@ -561,8 +709,86 @@ | ||
| 561 | 709 | ); |
| 562 | 710 | } |
| 563 | 711 | |
| 564 | 712 | /** |
| 713 | + * Extension → MIME map for the recording formats OpenAI's transcription | |
| 714 | + * endpoint accepts. The video containers are here on purpose: the endpoint | |
| 715 | + * reads their audio track, which is what lets this feature work without | |
| 716 | + * ffmpeg on the host. | |
| 717 | + * | |
| 718 | + * @since 4.9.4 | |
| 719 | + * | |
| 720 | + * @return array<string,string> | |
| 721 | + */ | |
| 722 | + public static function media_mimes() { | |
| 723 | + return array( | |
| 724 | + 'mp3|mpga' => 'audio/mpeg', | |
| 725 | + 'm4a' => 'audio/mp4', | |
| 726 | + 'wav' => 'audio/wav', | |
| 727 | + 'flac' => 'audio/flac', | |
| 728 | + 'ogg|oga' => 'audio/ogg', | |
| 729 | + 'mp4' => 'video/mp4', | |
| 730 | + 'mpeg|mpg' => 'video/mpeg', | |
| 731 | + 'webm' => 'video/webm', | |
| 732 | + ); | |
| 733 | + } | |
| 734 | + | |
| 735 | + /** | |
| 736 | + * Flat list of accepted recording extensions. | |
| 737 | + * | |
| 738 | + * @since 4.9.4 | |
| 739 | + * | |
| 740 | + * @return string[] | |
| 741 | + */ | |
| 742 | + public static function media_exts() { | |
| 743 | + $exts = array(); | |
| 744 | + foreach ( array_keys( self::media_mimes() ) as $group ) { | |
| 745 | + foreach ( explode( '|', $group ) as $ext ) { | |
| 746 | + $exts[] = $ext; | |
| 747 | + } | |
| 748 | + } | |
| 749 | + return $exts; | |
| 750 | + } | |
| 751 | + | |
| 752 | + /** | |
| 753 | + * Effective upload ceiling for a kind of attachment, in bytes. | |
| 754 | + * | |
| 755 | + * Always clamped to what the host will accept. A site with | |
| 756 | + * `upload_max_filesize = 8M` cannot receive 25 MB no matter what we allow — | |
| 757 | + * PHP discards the body and the request arrives empty — so advertising the | |
| 758 | + * higher number would just produce an unexplained failure. | |
| 759 | + * | |
| 760 | + * @since 4.9.4 | |
| 761 | + * | |
| 762 | + * @param string $kind `media` | `file` | |
| 763 | + * @return int | |
| 764 | + */ | |
| 765 | + public static function upload_cap( $kind = 'file' ) { | |
| 766 | + $cap = ( 'media' === $kind ) ? self::media_cap_for_platform() : self::MAX_UPLOAD_BYTES; | |
| 767 | + $host = (int) wp_max_upload_size(); | |
| 768 | + | |
| 769 | + return ( $host > 0 && $host < $cap ) ? $host : $cap; | |
| 770 | + } | |
| 771 | + | |
| 772 | + /** | |
| 773 | + * The recording cap for the configured platform, before the host clamp. | |
| 774 | + * | |
| 775 | + * Gemini carries the media inline as base64 inside the JSON request, which | |
| 776 | + * inflates it by roughly a third, so its practical ceiling is lower than | |
| 777 | + * OpenAI's, where the file is a real multipart part. | |
| 778 | + * | |
| 779 | + * @since 4.9.4 | |
| 780 | + * | |
| 781 | + * @return int | |
| 782 | + */ | |
| 783 | + public static function media_cap_for_platform() { | |
| 784 | + $factory = new ProviderFactory( betterdocs()->settings ); | |
| 785 | + $platform = $factory->active_platform(); | |
| 786 | + | |
| 787 | + return ( 'gemini' === $platform ) ? self::MAX_MEDIA_BYTES_INLINE : self::MAX_MEDIA_BYTES; | |
| 788 | + } | |
| 789 | + | |
| 790 | + /** | |
| 565 | 791 | * Extract plain text from a supported uploaded file. TXT/MD are read as-is; |
| 566 | 792 | * DOCX is unzipped natively (ZipArchive) and its document body flattened; |
| 567 | 793 | * PDF text is pulled natively from FlateDecode content streams — all without |
| 568 | 794 | * a third-party parser dependency. Images/scanned PDFs (no embedded text) are |
| @@ -789,9 +1015,20 @@ | ||
| 789 | 1015 | return ( $printable / $len ) >= 0.7; |
| 790 | 1016 | } |
| 791 | 1017 | |
| 792 | 1018 | protected function clip( $value, $max ) { |
| 793 | - return strlen( $value ) > $max ? substr( $value, 0, $max ) : $value; | |
| 1019 | + if ( strlen( $value ) <= $max ) { | |
| 1020 | + return $value; | |
| 1021 | + } | |
| 1022 | + | |
| 1023 | + // Byte-limited but never mid-character: a split multi-byte sequence is | |
| 1024 | + // invalid UTF-8, which wp_json_encode() refuses — the whole request | |
| 1025 | + // body would go out empty. | |
| 1026 | + if ( function_exists( 'mb_strcut' ) ) { | |
| 1027 | + return mb_strcut( $value, 0, $max, 'UTF-8' ); | |
| 1028 | + } | |
| 1029 | + | |
| 1030 | + return wp_check_invalid_utf8( substr( $value, 0, $max ), true ); | |
| 794 | 1031 | } |
| 795 | 1032 | |
| 796 | 1033 | /** |
| 797 | 1034 | * Coerce extracted PDF bytes to clean, valid UTF-8: drop C0/C1 control bytes |