| @@ -4,8 +4,9 @@ | ||
| 4 | 4 | |
| 5 | 5 | use WP_REST_Request; |
| 6 | 6 | use WPDeveloper\BetterDocs\Core\BaseAPI; |
| 7 | 7 | use WPDeveloper\BetterDocs\Utils\AIUsage; |
| 8 | +use WPDeveloper\BetterDocs\AI\ProviderFactory; | |
| 8 | 9 | |
| 9 | 10 | /** |
| 10 | 11 | * REST surface for the redesigned "Write with AI" modal. |
| 11 | 12 | * |
| @@ -20,8 +21,25 @@ | ||
| 20 | 21 | |
| 21 | 22 | const MAX_SOURCE_LENGTH = 12000; |
| 22 | 23 | const MAX_PROMPT_LENGTH = 4000; |
| 23 | 24 | |
| 25 | + // "From Attachment" source: server-side text extraction from an uploaded file. | |
| 26 | + // 5 MB is generous for text/markdown/DOCX while capping abuse; the extracted | |
| 27 | + // text is still clipped to MAX_SOURCE_LENGTH before it reaches the model. | |
| 28 | + const MAX_UPLOAD_BYTES = 5242880; // 5 MB | |
| 29 | + | |
| 30 | + // Recordings get their own, larger cap: 25 MB is OpenAI's transcription | |
| 31 | + // limit — roughly 25 minutes of mono MP3 — and there is no point accepting | |
| 32 | + // a file the provider will refuse. Gemini is capped lower still (see | |
| 33 | + // media_cap_for_platform): its media rides inline as base64, which inflates | |
| 34 | + // the payload by about a third. | |
| 35 | + // | |
| 36 | + // The cap sits 1 MB under that limit, not on it: the limit applies to the | |
| 37 | + // whole request, and a file of exactly 25 MB plus the multipart fields and | |
| 38 | + // boundaries would be refused after the full upload. | |
| 39 | + const MAX_MEDIA_BYTES = 25165824; // 24 MB | |
| 40 | + const MAX_MEDIA_BYTES_INLINE = 15728640; // 15 MB — inline-data platforms | |
| 41 | + | |
| 24 | 42 | public function register() { |
| 25 | 43 | $this->post( |
| 26 | 44 | '/write-with-ai', |
| 27 | 45 | array( $this, 'generate' ), |
| @@ -139,9 +157,9 @@ | ||
| 139 | 157 | |
| 140 | 158 | if ( empty( $write_ai->get_api_key() ) ) { |
| 141 | 159 | return $this->error( |
| 142 | 160 | 'missing_key', |
| 143 | - __( 'OpenAI API key is missing. Add one in BetterDocs settings.', 'betterdocs' ), | |
| 161 | + __( 'AI API key is missing. Add one in BetterDocs settings.', 'betterdocs' ), | |
| 144 | 162 | 400 |
| 145 | 163 | ); |
| 146 | 164 | } |
| 147 | 165 | |
| @@ -198,8 +216,9 @@ | ||
| 198 | 216 | $src_labels = array( |
| 199 | 217 | 'transcript' => __( 'support transcript', 'betterdocs' ), |
| 200 | 218 | 'forum' => __( 'forum thread', 'betterdocs' ), |
| 201 | 219 | 'notes' => __( 'raw notes', 'betterdocs' ), |
| 220 | + 'recording' => __( 'recording transcript', 'betterdocs' ), | |
| 202 | 221 | ); |
| 203 | 222 | $src_label = isset( $src_labels[ $src_type ] ) ? $src_labels[ $src_type ] : __( 'source material', 'betterdocs' ); |
| 204 | 223 | |
| 205 | 224 | // Light per-type framing: a one-line system hint steering how to treat |
| @@ -207,8 +226,12 @@ | ||
| 207 | 226 | $src_frames = array( |
| 208 | 227 | 'transcript' => __( 'The source below is a customer-support conversation. Focus on the user\'s problem and its resolution; ignore greetings and small talk.', 'betterdocs' ), |
| 209 | 228 | 'forum' => __( 'The source below is a forum discussion among multiple people. Treat the accepted or most-supported answer as authoritative and skip off-topic replies.', 'betterdocs' ), |
| 210 | 229 | 'notes' => __( 'The source below is rough notes. Expand them into clear, complete prose.', 'betterdocs' ), |
| 230 | + // Speech-to-text output reads nothing like written source: it | |
| 231 | + // has no punctuation discipline, keeps every "um", and may | |
| 232 | + // label speakers. Say so, or the model documents the filler. | |
| 233 | + 'recording' => __( 'The source below is a machine transcript of an audio or video recording. It may contain filler words, false starts, repetition and speaker labels — ignore those and document only the substance. The author has already reviewed it, so keep their spelling of names and product terms exactly as written.', 'betterdocs' ), | |
| 211 | 234 | ); |
| 212 | 235 | if ( isset( $src_frames[ $src_type ] ) ) { |
| 213 | 236 | array_unshift( $extra_system, array( 'role' => 'system', 'content' => $src_frames[ $src_type ] ) ); |
| 214 | 237 | } |
| @@ -220,10 +243,124 @@ | ||
| 220 | 243 | $src_label |
| 221 | 244 | ) |
| 222 | 245 | . "\n---\n" . $source . "\n---" ) |
| 223 | 246 | . $this->build_directives( $tone, $doc_size, $generate_title ); |
| 224 | - return $this->handle_doc( $write_ai, $post_id, $source_prompt, $keywords, $action, $doc_size, $extra_system ); | |
| 247 | + // A doc written from a recording's transcript counts as a recording | |
| 248 | + // in the insights, once — the transcribe step records nothing. | |
| 249 | + $usage_action = 'recording' === $src_type ? 'from-recording' : null; | |
| 250 | + return $this->handle_doc( $write_ai, $post_id, $source_prompt, $keywords, $action, $doc_size, $extra_system, null, $usage_action ); | |
| 225 | 251 | |
| 252 | + case 'transcribe-attachment': | |
| 253 | + // Step one of the recording flow: upload → transcript. No doc is | |
| 254 | + // written here. The transcript goes back to the modal so the | |
| 255 | + // author can fix mis-heard product names, and generation then | |
| 256 | + // runs through the ordinary `from-source` path with that text — | |
| 257 | + // which is why there is no media-shaped generation path at all. | |
| 258 | + $media = $this->read_uploaded_attachment( $request ); | |
| 259 | + if ( is_wp_error( $media ) ) { | |
| 260 | + return $this->error( $media->get_error_code() ?: 'ai_attachment_failed', $media->get_error_message(), 400 ); | |
| 261 | + } | |
| 262 | + | |
| 263 | + if ( ! isset( $media['kind'] ) || 'media' !== $media['kind'] ) { | |
| 264 | + return $this->error( 'ai_not_media', __( 'That file is not an audio or video recording.', 'betterdocs' ), 400 ); | |
| 265 | + } | |
| 266 | + | |
| 267 | + $transcript = $write_ai->transcribe( array( | |
| 268 | + 'path' => $media['path'], | |
| 269 | + 'filename' => $media['name'], | |
| 270 | + 'mime' => $media['mime'], | |
| 271 | + ) ); | |
| 272 | + | |
| 273 | + if ( is_wp_error( $transcript ) ) { | |
| 274 | + $code = $transcript->get_error_code() ?: 'ai_transcribe_failed'; | |
| 275 | + // A platform that cannot transcribe is the user's setting to | |
| 276 | + // change, not an upstream fault — 400, like ai_no_vision. | |
| 277 | + return $this->error( $code, $transcript->get_error_message(), 'ai_no_transcription' === $code ? 400 : 502 ); | |
| 278 | + } | |
| 279 | + | |
| 280 | + // Clip before returning, not after editing: the author should be | |
| 281 | + // correcting exactly the text the model will receive, rather than | |
| 282 | + // polishing a tail that gets silently cut on the way out. | |
| 283 | + // | |
| 284 | + // A clipped transcript says so in its own text: the author sees | |
| 285 | + // where it stops, and the marker travels with the text to the | |
| 286 | + // model, which would otherwise document half a recording as if it | |
| 287 | + // were the whole of it. | |
| 288 | + $transcript = wp_check_invalid_utf8( (string) $transcript, true ); | |
| 289 | + $clipped = strlen( $transcript ) > self::MAX_SOURCE_LENGTH; | |
| 290 | + | |
| 291 | + if ( $clipped ) { | |
| 292 | + $marker = "\n\n" . __( '[Transcript cut off here: the recording is longer than BetterDocs AI can use at once.]', 'betterdocs' ); | |
| 293 | + $transcript = rtrim( $this->clip( $transcript, self::MAX_SOURCE_LENGTH - strlen( $marker ) ) ) . $marker; | |
| 294 | + } | |
| 295 | + | |
| 296 | + if ( '' === trim( $transcript ) ) { | |
| 297 | + return $this->error( 'ai_no_speech', __( 'No speech was found in that recording.', 'betterdocs' ), 400 ); | |
| 298 | + } | |
| 299 | + | |
| 300 | + // No AIUsage::record() here. Nothing has been written yet — the doc | |
| 301 | + // is generated by the `from-source` call that follows, which records | |
| 302 | + // it (as a recording, via source_type). Counting both made every | |
| 303 | + // recording-based doc show up twice in the insights. | |
| 304 | + | |
| 305 | + return $this->success( array( | |
| 306 | + 'transcript' => $transcript, | |
| 307 | + 'clipped' => $clipped, | |
| 308 | + 'name' => $media['name'], | |
| 309 | + 'action' => $action, | |
| 310 | + ) ); | |
| 311 | + | |
| 312 | + case 'from-attachment': | |
| 313 | + // Upload a file; extract its text server-side and treat it exactly | |
| 314 | + // like from-source (same "use only what it contains" contract and | |
| 315 | + // the same handle_doc → wp_kses_post output path). The file itself | |
| 316 | + // is never stored or rendered — only its extracted text is used as | |
| 317 | + // grounded prompt context. | |
| 318 | + $extracted = $this->read_uploaded_attachment( $request ); | |
| 319 | + if ( is_wp_error( $extracted ) ) { | |
| 320 | + return $this->error( $extracted->get_error_code() ?: 'ai_attachment_failed', $extracted->get_error_message(), 400 ); | |
| 321 | + } | |
| 322 | + | |
| 323 | + // A recording has no text to extract; it goes through | |
| 324 | + // `transcribe-attachment` first. Sent here directly (an older modal, | |
| 325 | + // or a hand-made request) it used to fall through to the text path | |
| 326 | + // and read an undefined `text` key. | |
| 327 | + if ( isset( $extracted['kind'] ) && 'media' === $extracted['kind'] ) { | |
| 328 | + return $this->error( 'ai_needs_transcript', __( 'Recordings are transcribed first. Use "Transcribe recording", then generate from the transcript.', 'betterdocs' ), 400 ); | |
| 329 | + } | |
| 330 | + | |
| 331 | + // Image attachment → send the picture to a vision-capable model | |
| 332 | + // instead of extracting text (there is none). Same handle_doc | |
| 333 | + // output path (wp_kses_post), just a multimodal request. | |
| 334 | + if ( isset( $extracted['kind'] ) && 'image' === $extracted['kind'] ) { | |
| 335 | + $image_prompt = trim( $prompt . "\n\n" | |
| 336 | + . sprintf( | |
| 337 | + /* translators: %s: the uploaded image file name. */ | |
| 338 | + __( 'Read the attached image "%s" and turn what it shows — its text, tables, diagrams, UI or screenshots — into structured documentation. Describe only what is actually visible in the image; do not invent details:', 'betterdocs' ), | |
| 339 | + $extracted['name'] | |
| 340 | + ) ) | |
| 341 | + . $this->build_directives( $tone, $doc_size, $generate_title ); | |
| 342 | + | |
| 343 | + return $this->handle_doc( $write_ai, $post_id, $image_prompt, $keywords, $action, $doc_size, $extra_system, $extracted ); | |
| 344 | + } | |
| 345 | + | |
| 346 | + // Extracted file text is prompt-bound source (not rendered as HTML), | |
| 347 | + // so preserve angle brackets like from-source/from-git do. | |
| 348 | + $file_text = $this->clip( wp_check_invalid_utf8( (string) $extracted['text'], true ), self::MAX_SOURCE_LENGTH ); | |
| 349 | + if ( '' === trim( $file_text ) ) { | |
| 350 | + return $this->error( 'ai_empty_attachment', __( 'No readable text was found in that file.', 'betterdocs' ), 400 ); | |
| 351 | + } | |
| 352 | + | |
| 353 | + $file_prompt = trim( $prompt . "\n\n" | |
| 354 | + . sprintf( | |
| 355 | + /* translators: %s: the uploaded file name. */ | |
| 356 | + __( 'Turn the content of the uploaded file "%s" into structured documentation. Use only the information it contains; do not invent details:', 'betterdocs' ), | |
| 357 | + $extracted['name'] | |
| 358 | + ) | |
| 359 | + . "\n---\n" . $file_text . "\n---" ) | |
| 360 | + . $this->build_directives( $tone, $doc_size, $generate_title ); | |
| 361 | + return $this->handle_doc( $write_ai, $post_id, $file_prompt, $keywords, $action, $doc_size, $extra_system ); | |
| 362 | + | |
| 226 | 363 | case 'git-repos': |
| 227 | 364 | case 'git-items': |
| 228 | 365 | case 'git-contents': |
| 229 | 366 | // "Browse repository" data for the From Git tab. Read-only listing |
| @@ -326,9 +463,9 @@ | ||
| 326 | 463 | |
| 327 | 464 | /** |
| 328 | 465 | * Full-doc generation (generate-doc, expand-outline, from-source all land here). |
| 329 | 466 | */ |
| 330 | - protected function handle_doc( $write_ai, $post_id, $prompt, $keywords, $action, $doc_size = 'any', $extra_system = array() ) { | |
| 467 | + protected function handle_doc( $write_ai, $post_id, $prompt, $keywords, $action, $doc_size = 'any', $extra_system = array(), $image = null, $usage_action = null ) { | |
| 331 | 468 | if ( '' === trim( $prompt ) ) { |
| 332 | 469 | return $this->error( 'ai_empty_prompt', __( 'Please provide a prompt for the AI.', 'betterdocs' ), 400 ); |
| 333 | 470 | } |
| 334 | 471 | |
| @@ -334,9 +471,20 @@ | ||
| 334 | 471 | |
| 335 | 472 | // A "long" doc can outrun the default 2500-token cap; give it headroom. |
| 336 | 473 | $max_tokens = 'long' === $doc_size ? 4000 : null; |
| 337 | 474 | |
| 338 | - $content = $write_ai->generate_openai_response( $prompt, $keywords, $max_tokens, $extra_system ); | |
| 475 | + if ( null !== $image ) { | |
| 476 | + // Image attachment: send the picture to a vision model. Returns a | |
| 477 | + // WP_Error when the configured model can't read images (guard) — surface | |
| 478 | + // that as a 400 so the user knows to switch models, not a 502. | |
| 479 | + $content = $write_ai->generate_vision_response( $prompt, $image, $max_tokens, $extra_system ); | |
| 480 | + if ( is_wp_error( $content ) ) { | |
| 481 | + $code = $content->get_error_code() ?: 'ai_vision_failed'; | |
| 482 | + return $this->error( $code, $content->get_error_message(), 'ai_no_vision' === $code ? 400 : 502 ); | |
| 483 | + } | |
| 484 | + } else { | |
| 485 | + $content = $write_ai->generate_openai_response( $prompt, $keywords, $max_tokens, $extra_system ); | |
| 486 | + } | |
| 339 | 487 | |
| 340 | 488 | if ( ! is_string( $content ) || '' === trim( $content ) ) { |
| 341 | 489 | return $this->error( 'empty', __( 'The AI returned no content. Try again or rephrase your prompt.', 'betterdocs' ), 502 ); |
| 342 | 490 | } |
| @@ -351,9 +499,11 @@ | ||
| 351 | 499 | // prompt asks the model to avoid these, but that is a soft constraint — this |
| 352 | 500 | // is the enforcement (a prompt-injected source/Git payload can't inject XSS). |
| 353 | 501 | $content = wp_kses_post( $content ); |
| 354 | 502 | |
| 355 | - AIUsage::record( 'write_with_ai', $post_id, $action ); | |
| 503 | + // `$usage_action` lets a caller count the generation under a different | |
| 504 | + // mode than the action it answers to (a recording arrives as from-source). | |
| 505 | + AIUsage::record( 'write_with_ai', $post_id, null !== $usage_action ? $usage_action : $action ); | |
| 356 | 506 | |
| 357 | 507 | return $this->success( array( 'content' => $content, 'action' => $action ) ); |
| 358 | 508 | } |
| 359 | 509 | |
| @@ -410,10 +560,490 @@ | ||
| 410 | 560 | } |
| 411 | 561 | return implode( "\n", $lines ); |
| 412 | 562 | } |
| 413 | 563 | |
| 564 | + /** | |
| 565 | + * Validate the uploaded "From Attachment" file and return its extracted text. | |
| 566 | + * | |
| 567 | + * Security: enforces is_uploaded_file (a real HTTP upload, not an arbitrary | |
| 568 | + * server path), a byte cap, and a strict extension + MIME allow-list via | |
| 569 | + * wp_check_filetype(). The file is read for text only — never moved into the | |
| 570 | + * uploads dir, stored, or rendered — so there is no persisted attack surface. | |
| 571 | + * | |
| 572 | + * @param WP_REST_Request $request | |
| 573 | + * @return array{name:string,text:string}|\WP_Error | |
| 574 | + */ | |
| 575 | + protected function read_uploaded_attachment( WP_REST_Request $request ) { | |
| 576 | + $files = $request->get_file_params(); | |
| 577 | + if ( empty( $files['file'] ) || ! is_array( $files['file'] ) ) { | |
| 578 | + return new \WP_Error( 'ai_no_file', __( 'No file was received. Choose a file to write from.', 'betterdocs' ) ); | |
| 579 | + } | |
| 580 | + | |
| 581 | + $file = $files['file']; | |
| 582 | + | |
| 583 | + if ( ! empty( $file['error'] ) || empty( $file['tmp_name'] ) || ! is_uploaded_file( $file['tmp_name'] ) ) { | |
| 584 | + return new \WP_Error( 'ai_upload_failed', __( 'The upload did not complete — please try again.', 'betterdocs' ) ); | |
| 585 | + } | |
| 586 | + | |
| 587 | + // Strict extension + MIME allow-list. wp_check_filetype() validates the | |
| 588 | + // name against exactly these types; anything else yields an empty ext. | |
| 589 | + $allowed = array_merge( | |
| 590 | + array( | |
| 591 | + 'txt' => 'text/plain', | |
| 592 | + 'md|markdown' => 'text/markdown', | |
| 593 | + 'docx' => 'application/vnd.openxmlformats-officedocument.wordprocessingml.document', | |
| 594 | + 'pdf' => 'application/pdf', | |
| 595 | + 'png' => 'image/png', | |
| 596 | + 'jpg|jpeg' => 'image/jpeg', | |
| 597 | + 'webp' => 'image/webp', | |
| 598 | + ), | |
| 599 | + self::media_mimes() | |
| 600 | + ); | |
| 601 | + $check = wp_check_filetype( (string) $file['name'], $allowed ); | |
| 602 | + $ext = strtolower( (string) $check['ext'] ); | |
| 603 | + | |
| 604 | + $image_exts = array( 'png', 'jpg', 'jpeg', 'webp' ); | |
| 605 | + $text_exts = array( 'txt', 'md', 'markdown', 'docx', 'pdf' ); | |
| 606 | + $media_exts = self::media_exts(); | |
| 607 | + | |
| 608 | + if ( ! in_array( $ext, array_merge( $text_exts, $image_exts, $media_exts ), true ) ) { | |
| 609 | + // .mov and .avi are the two formats people actually try and that no | |
| 610 | + // provider accepts, so name the fix rather than listing types again. | |
| 611 | + $tried = strtolower( (string) pathinfo( (string) $file['name'], PATHINFO_EXTENSION ) ); | |
| 612 | + if ( in_array( $tried, array( 'mov', 'avi', 'wmv', 'mkv' ), true ) ) { | |
| 613 | + return new \WP_Error( | |
| 614 | + 'ai_bad_filetype', | |
| 615 | + sprintf( | |
| 616 | + /* translators: %s: the uploaded file's extension, e.g. "mov". */ | |
| 617 | + __( '.%s recordings are not supported. Export or convert it to MP4 and upload that.', 'betterdocs' ), | |
| 618 | + $tried | |
| 619 | + ) | |
| 620 | + ); | |
| 621 | + } | |
| 622 | + | |
| 623 | + return new \WP_Error( 'ai_bad_filetype', __( 'Unsupported file type. Upload a .pdf, .docx, .txt, .md, an image (.png, .jpg, .webp), or a recording (.mp3, .m4a, .wav, .flac, .ogg, .mp4, .mpeg, .webm).', 'betterdocs' ) ); | |
| 624 | + } | |
| 625 | + | |
| 626 | + // Size is capped per kind: a recording is legitimately much larger than a | |
| 627 | + // text file, but the cap can never exceed what the host will actually | |
| 628 | + // accept — PHP truncates a POST over post_max_size before we see it. | |
| 629 | + $is_media = in_array( $ext, $media_exts, true ); | |
| 630 | + $cap = self::upload_cap( $is_media ? 'media' : 'file' ); | |
| 631 | + | |
| 632 | + if ( (int) $file['size'] > $cap ) { | |
| 633 | + return new \WP_Error( | |
| 634 | + 'ai_file_too_large', | |
| 635 | + sprintf( | |
| 636 | + /* translators: %s: maximum allowed size, e.g. "25 MB". */ | |
| 637 | + __( 'The file exceeds the %s limit.', 'betterdocs' ), | |
| 638 | + size_format( $cap ) | |
| 639 | + ) | |
| 640 | + ); | |
| 641 | + } | |
| 642 | + | |
| 643 | + // Image → send the picture itself to a vision model (there is no text to | |
| 644 | + // extract). Verify it is a real image by its bytes, not just its name, | |
| 645 | + // then hand back a base64 data URI for the multimodal request. | |
| 646 | + if ( in_array( $ext, $image_exts, true ) ) { | |
| 647 | + $raw = file_get_contents( (string) $file['tmp_name'] ); // phpcs:ignore WordPressVIPMinimum.Performance.FetchingRemoteData.FileGetContentsUnknown -- local tmp upload. | |
| 648 | + if ( false === $raw || '' === $raw ) { | |
| 649 | + return new \WP_Error( 'ai_read_failed', __( 'Could not read the image file.', 'betterdocs' ) ); | |
| 650 | + } | |
| 651 | + | |
| 652 | + $info = @getimagesize( (string) $file['tmp_name'] ); | |
| 653 | + $mime = ( is_array( $info ) && ! empty( $info['mime'] ) ) ? (string) $info['mime'] : ''; | |
| 654 | + | |
| 655 | + if ( ! in_array( $mime, array( 'image/png', 'image/jpeg', 'image/webp' ), true ) ) { | |
| 656 | + return new \WP_Error( 'ai_bad_image', __( 'That file is not a valid PNG, JPG or WEBP image.', 'betterdocs' ) ); | |
| 657 | + } | |
| 658 | + | |
| 659 | + return array( | |
| 660 | + 'name' => sanitize_file_name( (string) $file['name'] ), | |
| 661 | + 'kind' => 'image', | |
| 662 | + 'mime' => $mime, | |
| 663 | + 'data_uri' => 'data:' . $mime . ';base64,' . base64_encode( $raw ), | |
| 664 | + ); | |
| 665 | + } | |
| 666 | + | |
| 667 | + // Recording → nothing to extract here; it goes to a speech-to-text model | |
| 668 | + // whole. Verify the bytes really are audio/video before handing a file to | |
| 669 | + // a paid endpoint: wp_check_filetype() above only read the *name*, so a | |
| 670 | + // renamed binary would otherwise sail through. | |
| 671 | + if ( $is_media ) { | |
| 672 | + $real = wp_check_filetype_and_ext( (string) $file['tmp_name'], (string) $file['name'], $allowed ); | |
| 673 | + $mime = ! empty( $real['type'] ) ? (string) $real['type'] : ''; | |
| 674 | + | |
| 675 | + if ( '' === $mime && function_exists( 'finfo_open' ) ) { | |
| 676 | + // wp_check_filetype_and_ext() only sniffs images and a short list | |
| 677 | + // of text formats; for A/V it hands back the name-based guess or | |
| 678 | + // nothing at all. finfo is the actual byte check. | |
| 679 | + $finfo = finfo_open( FILEINFO_MIME_TYPE ); | |
| 680 | + if ( $finfo ) { | |
| 681 | + $sniffed = finfo_file( $finfo, (string) $file['tmp_name'] ); | |
| 682 | + finfo_close( $finfo ); | |
| 683 | + $mime = is_string( $sniffed ) ? $sniffed : ''; | |
| 684 | + } | |
| 685 | + } | |
| 686 | + | |
| 687 | + if ( 0 !== strpos( $mime, 'audio/' ) && 0 !== strpos( $mime, 'video/' ) ) { | |
| 688 | + return new \WP_Error( 'ai_bad_media', __( 'That file is not a readable audio or video recording.', 'betterdocs' ) ); | |
| 689 | + } | |
| 690 | + | |
| 691 | + return array( | |
| 692 | + 'name' => sanitize_file_name( (string) $file['name'] ), | |
| 693 | + 'kind' => 'media', | |
| 694 | + 'mime' => $mime, | |
| 695 | + 'path' => (string) $file['tmp_name'], | |
| 696 | + 'size' => (int) $file['size'], | |
| 697 | + ); | |
| 698 | + } | |
| 699 | + | |
| 700 | + $text = $this->extract_attachment_text( (string) $file['tmp_name'], $ext ); | |
| 701 | + if ( is_wp_error( $text ) ) { | |
| 702 | + return $text; | |
| 703 | + } | |
| 704 | + | |
| 705 | + return array( | |
| 706 | + 'name' => sanitize_file_name( (string) $file['name'] ), | |
| 707 | + 'kind' => 'text', | |
| 708 | + 'text' => $text, | |
| 709 | + ); | |
| 710 | + } | |
| 711 | + | |
| 712 | + /** | |
| 713 | + * Extension → MIME map for the recording formats OpenAI's transcription | |
| 714 | + * endpoint accepts. The video containers are here on purpose: the endpoint | |
| 715 | + * reads their audio track, which is what lets this feature work without | |
| 716 | + * ffmpeg on the host. | |
| 717 | + * | |
| 718 | + * @since 4.9.4 | |
| 719 | + * | |
| 720 | + * @return array<string,string> | |
| 721 | + */ | |
| 722 | + public static function media_mimes() { | |
| 723 | + return array( | |
| 724 | + 'mp3|mpga' => 'audio/mpeg', | |
| 725 | + 'm4a' => 'audio/mp4', | |
| 726 | + 'wav' => 'audio/wav', | |
| 727 | + 'flac' => 'audio/flac', | |
| 728 | + 'ogg|oga' => 'audio/ogg', | |
| 729 | + 'mp4' => 'video/mp4', | |
| 730 | + 'mpeg|mpg' => 'video/mpeg', | |
| 731 | + 'webm' => 'video/webm', | |
| 732 | + ); | |
| 733 | + } | |
| 734 | + | |
| 735 | + /** | |
| 736 | + * Flat list of accepted recording extensions. | |
| 737 | + * | |
| 738 | + * @since 4.9.4 | |
| 739 | + * | |
| 740 | + * @return string[] | |
| 741 | + */ | |
| 742 | + public static function media_exts() { | |
| 743 | + $exts = array(); | |
| 744 | + foreach ( array_keys( self::media_mimes() ) as $group ) { | |
| 745 | + foreach ( explode( '|', $group ) as $ext ) { | |
| 746 | + $exts[] = $ext; | |
| 747 | + } | |
| 748 | + } | |
| 749 | + return $exts; | |
| 750 | + } | |
| 751 | + | |
| 752 | + /** | |
| 753 | + * Effective upload ceiling for a kind of attachment, in bytes. | |
| 754 | + * | |
| 755 | + * Always clamped to what the host will accept. A site with | |
| 756 | + * `upload_max_filesize = 8M` cannot receive 25 MB no matter what we allow — | |
| 757 | + * PHP discards the body and the request arrives empty — so advertising the | |
| 758 | + * higher number would just produce an unexplained failure. | |
| 759 | + * | |
| 760 | + * @since 4.9.4 | |
| 761 | + * | |
| 762 | + * @param string $kind `media` | `file` | |
| 763 | + * @return int | |
| 764 | + */ | |
| 765 | + public static function upload_cap( $kind = 'file' ) { | |
| 766 | + $cap = ( 'media' === $kind ) ? self::media_cap_for_platform() : self::MAX_UPLOAD_BYTES; | |
| 767 | + $host = (int) wp_max_upload_size(); | |
| 768 | + | |
| 769 | + return ( $host > 0 && $host < $cap ) ? $host : $cap; | |
| 770 | + } | |
| 771 | + | |
| 772 | + /** | |
| 773 | + * The recording cap for the configured platform, before the host clamp. | |
| 774 | + * | |
| 775 | + * Gemini carries the media inline as base64 inside the JSON request, which | |
| 776 | + * inflates it by roughly a third, so its practical ceiling is lower than | |
| 777 | + * OpenAI's, where the file is a real multipart part. | |
| 778 | + * | |
| 779 | + * @since 4.9.4 | |
| 780 | + * | |
| 781 | + * @return int | |
| 782 | + */ | |
| 783 | + public static function media_cap_for_platform() { | |
| 784 | + $factory = new ProviderFactory( betterdocs()->settings ); | |
| 785 | + $platform = $factory->active_platform(); | |
| 786 | + | |
| 787 | + return ( 'gemini' === $platform ) ? self::MAX_MEDIA_BYTES_INLINE : self::MAX_MEDIA_BYTES; | |
| 788 | + } | |
| 789 | + | |
| 790 | + /** | |
| 791 | + * Extract plain text from a supported uploaded file. TXT/MD are read as-is; | |
| 792 | + * DOCX is unzipped natively (ZipArchive) and its document body flattened; | |
| 793 | + * PDF text is pulled natively from FlateDecode content streams — all without | |
| 794 | + * a third-party parser dependency. Images/scanned PDFs (no embedded text) are | |
| 795 | + * not handled here (that would need OCR / a multimodal model). | |
| 796 | + * | |
| 797 | + * @param string $path Local tmp upload path (already is_uploaded_file-verified). | |
| 798 | + * @param string $ext Allow-listed extension. | |
| 799 | + * @return string|\WP_Error | |
| 800 | + */ | |
| 801 | + protected function extract_attachment_text( $path, $ext ) { | |
| 802 | + if ( in_array( $ext, array( 'txt', 'md', 'markdown' ), true ) ) { | |
| 803 | + $raw = file_get_contents( $path ); // phpcs:ignore WordPressVIPMinimum.Performance.FetchingRemoteData.FileGetContentsUnknown -- local tmp upload. | |
| 804 | + return false === $raw ? new \WP_Error( 'ai_read_failed', __( 'Could not read the file.', 'betterdocs' ) ) : $raw; | |
| 805 | + } | |
| 806 | + | |
| 807 | + if ( 'docx' === $ext ) { | |
| 808 | + if ( ! class_exists( '\ZipArchive' ) ) { | |
| 809 | + return new \WP_Error( 'ai_no_zip', __( 'Reading .docx files needs the PHP Zip extension, which is not available on this server. Upload a .txt or .md instead.', 'betterdocs' ) ); | |
| 810 | + } | |
| 811 | + $zip = new \ZipArchive(); | |
| 812 | + if ( true !== $zip->open( $path ) ) { | |
| 813 | + return new \WP_Error( 'ai_bad_docx', __( 'Could not open that .docx file — it may be corrupt.', 'betterdocs' ) ); | |
| 814 | + } | |
| 815 | + $xml = $zip->getFromName( 'word/document.xml' ); | |
| 816 | + $zip->close(); | |
| 817 | + | |
| 818 | + if ( false === $xml || '' === $xml ) { | |
| 819 | + return new \WP_Error( 'ai_bad_docx', __( 'That .docx file has no readable document body.', 'betterdocs' ) ); | |
| 820 | + } | |
| 821 | + | |
| 822 | + // Turn Word paragraph/break/tab elements into whitespace, then strip | |
| 823 | + // every remaining tag so only the run text (<w:t>) survives, and decode | |
| 824 | + // XML entities. Keeps paragraph structure the model can read. | |
| 825 | + $xml = preg_replace( '#</w:p>#', "\n\n", $xml ); | |
| 826 | + $xml = preg_replace( '#<w:br\b[^>]*/?>#', "\n", $xml ); | |
| 827 | + $xml = preg_replace( '#<w:tab\b[^>]*/?>#', "\t", $xml ); | |
| 828 | + $text = wp_strip_all_tags( (string) $xml ); | |
| 829 | + $text = html_entity_decode( $text, ENT_QUOTES | ENT_XML1, 'UTF-8' ); | |
| 830 | + | |
| 831 | + return trim( preg_replace( "/\n{3,}/", "\n\n", $text ) ); | |
| 832 | + } | |
| 833 | + | |
| 834 | + if ( 'pdf' === $ext ) { | |
| 835 | + return $this->extract_pdf_text( $path ); | |
| 836 | + } | |
| 837 | + | |
| 838 | + return new \WP_Error( 'ai_bad_filetype', __( 'Unsupported file type.', 'betterdocs' ) ); | |
| 839 | + } | |
| 840 | + | |
| 841 | + /** | |
| 842 | + * Extract text from a PDF natively — no library. PDFs keep their page text in | |
| 843 | + * "content streams" (usually zlib/FlateDecode-compressed); we inflate each one | |
| 844 | + * and pull the operands of the text-showing operators (Tj / TJ / ' / "). This | |
| 845 | + * covers the common case (real, text-based documents). It intentionally does | |
| 846 | + * NOT handle: | |
| 847 | + * - encrypted PDFs (no key) — reported so the user knows why, | |
| 848 | + * - scanned/image-only PDFs (there is no embedded text to read) — reported, | |
| 849 | + * - exotic font encodings (CID/Type0 with custom CMaps) — those decode to | |
| 850 | + * garbled text, so we drop a stream whose result looks non-textual. | |
| 851 | + * The extracted text is prompt-bound source only; it is never rendered, and | |
| 852 | + * the model output still passes wp_kses_post downstream. | |
| 853 | + * | |
| 854 | + * @param string $path Local tmp upload path. | |
| 855 | + * @return string|\WP_Error | |
| 856 | + */ | |
| 857 | + protected function extract_pdf_text( $path ) { | |
| 858 | + $data = file_get_contents( $path ); // phpcs:ignore WordPressVIPMinimum.Performance.FetchingRemoteData.FileGetContentsUnknown -- local tmp upload. | |
| 859 | + if ( false === $data || 0 !== strncmp( $data, '%PDF', 4 ) ) { | |
| 860 | + return new \WP_Error( 'ai_bad_pdf', __( 'That does not look like a valid PDF file.', 'betterdocs' ) ); | |
| 861 | + } | |
| 862 | + | |
| 863 | + // An encrypted PDF's streams won't inflate to readable text without the | |
| 864 | + // key. Detect the Encrypt entry up front and say so, rather than return | |
| 865 | + // empty. (An /Encrypt inside a literal string is a rare false positive we | |
| 866 | + // accept — worst case the user gets the "no text" message below instead.) | |
| 867 | + if ( preg_match( '/\/Encrypt\b/', $data ) ) { | |
| 868 | + return new \WP_Error( 'ai_pdf_encrypted', __( 'This PDF is password-protected or encrypted, so its text can\'t be read. Remove the protection, or paste the text instead.', 'betterdocs' ) ); | |
| 869 | + } | |
| 870 | + | |
| 871 | + $out = ''; | |
| 872 | + $cap = self::MAX_SOURCE_LENGTH + 4000; // stop early; the source is clipped later anyway. | |
| 873 | + | |
| 874 | + if ( preg_match_all( '/stream\r?\n(.*?)\r?\nendstream/s', $data, $streams ) ) { | |
| 875 | + foreach ( $streams[1] as $chunk ) { | |
| 876 | + // Try zlib (FlateDecode) first, then raw-deflate, then treat as | |
| 877 | + // already-plain. @-silenced: a binary (image/font) stream simply | |
| 878 | + // fails to inflate and is skipped below. | |
| 879 | + $decoded = @gzuncompress( $chunk ); | |
| 880 | + if ( false === $decoded ) { | |
| 881 | + $decoded = @gzinflate( $chunk ); | |
| 882 | + } | |
| 883 | + $content = ( is_string( $decoded ) && '' !== $decoded ) ? $decoded : $chunk; | |
| 884 | + | |
| 885 | + // Only content streams carry text-showing operators; skip the rest | |
| 886 | + // (images, fonts) so we don't scrape binary noise. | |
| 887 | + if ( false === strpos( $content, 'Tj' ) && false === strpos( $content, 'TJ' ) ) { | |
| 888 | + continue; | |
| 889 | + } | |
| 890 | + | |
| 891 | + $piece = $this->pdf_stream_text( $content ); | |
| 892 | + // Guard against garbled CID/font-encoded streams: if the decoded | |
| 893 | + // "text" is mostly non-printable, drop it rather than inject noise. | |
| 894 | + if ( '' !== $piece && $this->mostly_printable( $piece ) ) { | |
| 895 | + $out .= $piece . "\n"; | |
| 896 | + if ( strlen( $out ) > $cap ) { | |
| 897 | + break; | |
| 898 | + } | |
| 899 | + } | |
| 900 | + } | |
| 901 | + } | |
| 902 | + | |
| 903 | + $out = preg_replace( "/[ \t]+/", ' ', $out ); | |
| 904 | + $out = trim( preg_replace( "/\n{3,}/", "\n\n", $out ) ); | |
| 905 | + | |
| 906 | + // Subsetted LaTeX/CID fonts emit control bytes (ligatures) and non-UTF-8 | |
| 907 | + // sequences among the readable text. Strip them and coerce to valid UTF-8 — | |
| 908 | + // otherwise the caller's wp_check_invalid_utf8() discards the ENTIRE string | |
| 909 | + // on the first bad byte and an 8-page paper looks empty ("no readable text"). | |
| 910 | + $out = $this->to_clean_utf8( $out ); | |
| 911 | + | |
| 912 | + if ( '' === $out ) { | |
| 913 | + return new \WP_Error( | |
| 914 | + 'ai_pdf_no_text', | |
| 915 | + __( 'No selectable text was found in that PDF — it may be a scanned image. Try a text-based PDF, or paste the content into the prompt.', 'betterdocs' ) | |
| 916 | + ); | |
| 917 | + } | |
| 918 | + | |
| 919 | + return $out; | |
| 920 | + } | |
| 921 | + | |
| 922 | + /** | |
| 923 | + * Pull the visible text out of one decoded PDF content stream. Positioning | |
| 924 | + * operators (Td/TD/T*) become newlines; the literal `( … )` and hex `< … >` | |
| 925 | + * operands of Tj/TJ/'/'' become the text. Kerning numbers inside TJ arrays are | |
| 926 | + * ignored (their effect on spacing is cosmetic for our purposes). | |
| 927 | + */ | |
| 928 | + protected function pdf_stream_text( $content ) { | |
| 929 | + // Text-positioning operators (new line / new paragraph) become newlines so | |
| 930 | + // words on different lines don't run together. | |
| 931 | + $content = preg_replace( '/\b(?:T\*|Td|TD)\b/', " \n ", $content ); | |
| 932 | + | |
| 933 | + // Walk TJ arrays and Tj/'/'" strings in document order. Inside a TJ array | |
| 934 | + // pdfTeX (LaTeX) renders an inter-word space as a large negative kerning | |
| 935 | + // number, not a literal space in the string — so we synthesise a space when | |
| 936 | + // the kerning passes a threshold, otherwise every word runs together | |
| 937 | + // ("FormallyVerifiedand…"). Small kerning (letter pairs) is ignored. | |
| 938 | + if ( ! preg_match_all( | |
| 939 | + '/\[((?:\\\\.|[^\]\\\\])*)\]\s*TJ|(\((?:\\\\.|[^\\\\()])*\)|<[0-9A-Fa-f\s]+>)\s*(?:Tj|\'|")|(\n)/s', | |
| 940 | + $content, | |
| 941 | + $matches, | |
| 942 | + PREG_SET_ORDER | |
| 943 | + ) ) { | |
| 944 | + return ''; | |
| 945 | + } | |
| 946 | + | |
| 947 | + $text = ''; | |
| 948 | + foreach ( $matches as $tok ) { | |
| 949 | + if ( isset( $tok[3] ) && "\n" === $tok[3] ) { | |
| 950 | + $text .= "\n"; | |
| 951 | + continue; | |
| 952 | + } | |
| 953 | + if ( isset( $tok[1] ) && '' !== $tok[1] ) { | |
| 954 | + // TJ array: alternating string operands and kerning numbers. | |
| 955 | + preg_match_all( '/\((?:\\\\.|[^\\\\()])*\)|<[0-9A-Fa-f\s]+>|-?\d+(?:\.\d+)?/s', $tok[1], $parts ); | |
| 956 | + foreach ( $parts[0] as $part ) { | |
| 957 | + if ( '(' === $part[0] || '<' === $part[0] ) { | |
| 958 | + $text .= $this->pdf_token_text( $part ); | |
| 959 | + } elseif ( (float) $part < -100 ) { | |
| 960 | + $text .= ' '; | |
| 961 | + } | |
| 962 | + } | |
| 963 | + $text .= ' '; | |
| 964 | + } elseif ( isset( $tok[2] ) && '' !== $tok[2] ) { | |
| 965 | + $text .= $this->pdf_token_text( $tok[2] ) . ' '; | |
| 966 | + } | |
| 967 | + } | |
| 968 | + | |
| 969 | + return preg_replace( '/[^\S\n]+/', ' ', $text ); | |
| 970 | + } | |
| 971 | + | |
| 972 | + /** | |
| 973 | + * Decode one PDF string operand — a literal `( … )` (with escapes) or a hex | |
| 974 | + * `< … >` string — into its raw bytes. | |
| 975 | + */ | |
| 976 | + protected function pdf_token_text( $token ) { | |
| 977 | + if ( '(' === $token[0] ) { | |
| 978 | + return $this->pdf_unescape( substr( $token, 1, -1 ) ); | |
| 979 | + } | |
| 980 | + $hex = preg_replace( '/[^0-9A-Fa-f]/', '', $token ); | |
| 981 | + return ( '' === $hex ) ? '' : (string) @hex2bin( strlen( $hex ) % 2 ? substr( $hex, 0, -1 ) : $hex ); | |
| 982 | + } | |
| 983 | + | |
| 984 | + /** | |
| 985 | + * Resolve PDF string escapes: \( \) \\ \n \r \t \b \f and \ddd octal codes. | |
| 986 | + */ | |
| 987 | + protected function pdf_unescape( $string ) { | |
| 988 | + return preg_replace_callback( | |
| 989 | + '/\\\\(?:([nrtbf()\\\\])|([0-7]{1,3}))/', | |
| 990 | + function ( $mm ) { | |
| 991 | + if ( isset( $mm[1] ) && '' !== $mm[1] ) { | |
| 992 | + $map = array( 'n' => "\n", 'r' => "\r", 't' => "\t", 'b' => "\x08", 'f' => "\x0C", '(' => '(', ')' => ')', '\\' => '\\' ); | |
| 993 | + return isset( $map[ $mm[1] ] ) ? $map[ $mm[1] ] : $mm[1]; | |
| 994 | + } | |
| 995 | + return chr( octdec( $mm[2] ) & 0xFF ); | |
| 996 | + }, | |
| 997 | + $string | |
| 998 | + ); | |
| 999 | + } | |
| 1000 | + | |
| 1001 | + /** | |
| 1002 | + * Is this decoded string mostly readable text? Used to drop font/CID streams | |
| 1003 | + * that decode to binary-looking garbage. Counts printable + common whitespace. | |
| 1004 | + */ | |
| 1005 | + protected function mostly_printable( $string ) { | |
| 1006 | + $len = strlen( $string ); | |
| 1007 | + if ( 0 === $len ) { | |
| 1008 | + return false; | |
| 1009 | + } | |
| 1010 | + $printable = preg_match_all( '/[\P{Cc}\t\n\r]/u', $string ); | |
| 1011 | + // Fallback for non-UTF-8 payloads where \p{} may not match cleanly. | |
| 1012 | + if ( false === $printable ) { | |
| 1013 | + $printable = strlen( preg_replace( '/[^\x09\x0A\x0D\x20-\x7E]/', '', $string ) ); | |
| 1014 | + } | |
| 1015 | + return ( $printable / $len ) >= 0.7; | |
| 1016 | + } | |
| 1017 | + | |
| 414 | 1018 | protected function clip( $value, $max ) { |
| 415 | - return strlen( $value ) > $max ? substr( $value, 0, $max ) : $value; | |
| 1019 | + if ( strlen( $value ) <= $max ) { | |
| 1020 | + return $value; | |
| 1021 | + } | |
| 1022 | + | |
| 1023 | + // Byte-limited but never mid-character: a split multi-byte sequence is | |
| 1024 | + // invalid UTF-8, which wp_json_encode() refuses — the whole request | |
| 1025 | + // body would go out empty. | |
| 1026 | + if ( function_exists( 'mb_strcut' ) ) { | |
| 1027 | + return mb_strcut( $value, 0, $max, 'UTF-8' ); | |
| 1028 | + } | |
| 1029 | + | |
| 1030 | + return wp_check_invalid_utf8( substr( $value, 0, $max ), true ); | |
| 1031 | + } | |
| 1032 | + | |
| 1033 | + /** | |
| 1034 | + * Coerce extracted PDF bytes to clean, valid UTF-8: drop C0/C1 control bytes | |
| 1035 | + * (except tab/newline) and any byte sequence that isn't valid UTF-8. This keeps | |
| 1036 | + * the readable text intact for the downstream wp_check_invalid_utf8(), which | |
| 1037 | + * would otherwise discard the whole string on a single invalid byte. | |
| 1038 | + */ | |
| 1039 | + protected function to_clean_utf8( $string ) { | |
| 1040 | + $string = preg_replace( '/[\x00-\x08\x0B\x0C\x0E-\x1F\x7F]/', '', (string) $string ); | |
| 1041 | + if ( '' !== $string && ! preg_match( '//u', $string ) ) { | |
| 1042 | + $converted = @iconv( 'UTF-8', 'UTF-8//IGNORE', $string ); | |
| 1043 | + $string = ( false !== $converted ) ? $converted : preg_replace( '/[^\x09\x0A\x20-\x7E]/', '', $string ); | |
| 1044 | + } | |
| 1045 | + return (string) $string; | |
| 416 | 1046 | } |
| 417 | 1047 | |
| 418 | 1048 | /** |
| 419 | 1049 | * Frame the user's free-form request as a documentation instruction. Returns |