PluginProbe
BetterDocs – AI Documentation, Knowledge Base, MCP Server, Docs, Wikis, FAQ & Chatbot / 4.9.4
BetterDocs – AI Documentation, Knowledge Base, MCP Server, Docs, Wikis, FAQ & Chatbot v4.9.4
4.9.4 4.9.3 4.9.2 4.9.1 4.9.0 4.8.2 4.8.1 4.8.0 4.7.0 4.6.2 4.6.1 4.6.0 4.5.6 4.5.5 4.5.4 4.5.3 4.5.2 4.5.1 4.5.0 4.4.1 4.4.0 3.3.4 3.4.0 3.4.1 3.4.2 All 202 releases
← All changes | includes/REST/WriteWithAI.php +637 -8 4.7.0 → 4.9.4 View file →
@@ -4,8 +4,9 @@
4 4
5 5 use WP_REST_Request;
6 6 use WPDeveloper\BetterDocs\Core\BaseAPI;
7 7 use WPDeveloper\BetterDocs\Utils\AIUsage;
8 +use WPDeveloper\BetterDocs\AI\ProviderFactory;
8 9
9 10 /**
10 11 * REST surface for the redesigned "Write with AI" modal.
11 12 *
@@ -20,8 +21,25 @@
20 21
21 22 const MAX_SOURCE_LENGTH = 12000;
22 23 const MAX_PROMPT_LENGTH = 4000;
23 24
25 + // "From Attachment" source: server-side text extraction from an uploaded file.
26 + // 5 MB is generous for text/markdown/DOCX while capping abuse; the extracted
27 + // text is still clipped to MAX_SOURCE_LENGTH before it reaches the model.
28 + const MAX_UPLOAD_BYTES = 5242880; // 5 MB
29 +
30 + // Recordings get their own, larger cap: 25 MB is OpenAI's transcription
31 + // limit — roughly 25 minutes of mono MP3 — and there is no point accepting
32 + // a file the provider will refuse. Gemini is capped lower still (see
33 + // media_cap_for_platform): its media rides inline as base64, which inflates
34 + // the payload by about a third.
35 + //
36 + // The cap sits 1 MB under that limit, not on it: the limit applies to the
37 + // whole request, and a file of exactly 25 MB plus the multipart fields and
38 + // boundaries would be refused after the full upload.
39 + const MAX_MEDIA_BYTES = 25165824; // 24 MB
40 + const MAX_MEDIA_BYTES_INLINE = 15728640; // 15 MB — inline-data platforms
41 +
24 42 public function register() {
25 43 $this->post(
26 44 '/write-with-ai',
27 45 array( $this, 'generate' ),
@@ -121,10 +139,9 @@
121 139 }
122 140
123 141 public function permission_check() {
124 142 // Gate on edit_others_posts to match the sibling FAQ/Glossary AI endpoints
125 - // (AIFaq/AIGlossary). This keeps Author-role users — who can create their own
126 - // posts but not others' — from spending the site's OpenAI budget.
143 + // (AIFaq/AIGlossary) and keep Author-role users from spending the AI budget.
127 144 return current_user_can( 'edit_others_posts' );
128 145 }
129 146
130 147 public function generate( WP_REST_Request $request ) {
@@ -140,9 +157,9 @@
140 157
141 158 if ( empty( $write_ai->get_api_key() ) ) {
142 159 return $this->error(
143 160 'missing_key',
144 - __( 'OpenAI API key is missing. Add one in BetterDocs settings.', 'betterdocs' ),
161 + __( 'AI API key is missing. Add one in BetterDocs settings.', 'betterdocs' ),
145 162 400
146 163 );
147 164 }
148 165
@@ -199,8 +216,9 @@
199 216 $src_labels = array(
200 217 'transcript' => __( 'support transcript', 'betterdocs' ),
201 218 'forum' => __( 'forum thread', 'betterdocs' ),
202 219 'notes' => __( 'raw notes', 'betterdocs' ),
220 + 'recording' => __( 'recording transcript', 'betterdocs' ),
203 221 );
204 222 $src_label = isset( $src_labels[ $src_type ] ) ? $src_labels[ $src_type ] : __( 'source material', 'betterdocs' );
205 223
206 224 // Light per-type framing: a one-line system hint steering how to treat
@@ -208,8 +226,12 @@
208 226 $src_frames = array(
209 227 'transcript' => __( 'The source below is a customer-support conversation. Focus on the user\'s problem and its resolution; ignore greetings and small talk.', 'betterdocs' ),
210 228 'forum' => __( 'The source below is a forum discussion among multiple people. Treat the accepted or most-supported answer as authoritative and skip off-topic replies.', 'betterdocs' ),
211 229 'notes' => __( 'The source below is rough notes. Expand them into clear, complete prose.', 'betterdocs' ),
230 + // Speech-to-text output reads nothing like written source: it
231 + // has no punctuation discipline, keeps every "um", and may
232 + // label speakers. Say so, or the model documents the filler.
233 + 'recording' => __( 'The source below is a machine transcript of an audio or video recording. It may contain filler words, false starts, repetition and speaker labels — ignore those and document only the substance. The author has already reviewed it, so keep their spelling of names and product terms exactly as written.', 'betterdocs' ),
212 234 );
213 235 if ( isset( $src_frames[ $src_type ] ) ) {
214 236 array_unshift( $extra_system, array( 'role' => 'system', 'content' => $src_frames[ $src_type ] ) );
215 237 }
@@ -221,10 +243,124 @@
221 243 $src_label
222 244 )
223 245 . "\n---\n" . $source . "\n---" )
224 246 . $this->build_directives( $tone, $doc_size, $generate_title );
225 - return $this->handle_doc( $write_ai, $post_id, $source_prompt, $keywords, $action, $doc_size, $extra_system );
247 + // A doc written from a recording's transcript counts as a recording
248 + // in the insights, once — the transcribe step records nothing.
249 + $usage_action = 'recording' === $src_type ? 'from-recording' : null;
250 + return $this->handle_doc( $write_ai, $post_id, $source_prompt, $keywords, $action, $doc_size, $extra_system, null, $usage_action );
226 251
252 + case 'transcribe-attachment':
253 + // Step one of the recording flow: upload → transcript. No doc is
254 + // written here. The transcript goes back to the modal so the
255 + // author can fix mis-heard product names, and generation then
256 + // runs through the ordinary `from-source` path with that text —
257 + // which is why there is no media-shaped generation path at all.
258 + $media = $this->read_uploaded_attachment( $request );
259 + if ( is_wp_error( $media ) ) {
260 + return $this->error( $media->get_error_code() ?: 'ai_attachment_failed', $media->get_error_message(), 400 );
261 + }
262 +
263 + if ( ! isset( $media['kind'] ) || 'media' !== $media['kind'] ) {
264 + return $this->error( 'ai_not_media', __( 'That file is not an audio or video recording.', 'betterdocs' ), 400 );
265 + }
266 +
267 + $transcript = $write_ai->transcribe( array(
268 + 'path' => $media['path'],
269 + 'filename' => $media['name'],
270 + 'mime' => $media['mime'],
271 + ) );
272 +
273 + if ( is_wp_error( $transcript ) ) {
274 + $code = $transcript->get_error_code() ?: 'ai_transcribe_failed';
275 + // A platform that cannot transcribe is the user's setting to
276 + // change, not an upstream fault — 400, like ai_no_vision.
277 + return $this->error( $code, $transcript->get_error_message(), 'ai_no_transcription' === $code ? 400 : 502 );
278 + }
279 +
280 + // Clip before returning, not after editing: the author should be
281 + // correcting exactly the text the model will receive, rather than
282 + // polishing a tail that gets silently cut on the way out.
283 + //
284 + // A clipped transcript says so in its own text: the author sees
285 + // where it stops, and the marker travels with the text to the
286 + // model, which would otherwise document half a recording as if it
287 + // were the whole of it.
288 + $transcript = wp_check_invalid_utf8( (string) $transcript, true );
289 + $clipped = strlen( $transcript ) > self::MAX_SOURCE_LENGTH;
290 +
291 + if ( $clipped ) {
292 + $marker = "\n\n" . __( '[Transcript cut off here: the recording is longer than BetterDocs AI can use at once.]', 'betterdocs' );
293 + $transcript = rtrim( $this->clip( $transcript, self::MAX_SOURCE_LENGTH - strlen( $marker ) ) ) . $marker;
294 + }
295 +
296 + if ( '' === trim( $transcript ) ) {
297 + return $this->error( 'ai_no_speech', __( 'No speech was found in that recording.', 'betterdocs' ), 400 );
298 + }
299 +
300 + // No AIUsage::record() here. Nothing has been written yet — the doc
301 + // is generated by the `from-source` call that follows, which records
302 + // it (as a recording, via source_type). Counting both made every
303 + // recording-based doc show up twice in the insights.
304 +
305 + return $this->success( array(
306 + 'transcript' => $transcript,
307 + 'clipped' => $clipped,
308 + 'name' => $media['name'],
309 + 'action' => $action,
310 + ) );
311 +
312 + case 'from-attachment':
313 + // Upload a file; extract its text server-side and treat it exactly
314 + // like from-source (same "use only what it contains" contract and
315 + // the same handle_doc → wp_kses_post output path). The file itself
316 + // is never stored or rendered — only its extracted text is used as
317 + // grounded prompt context.
318 + $extracted = $this->read_uploaded_attachment( $request );
319 + if ( is_wp_error( $extracted ) ) {
320 + return $this->error( $extracted->get_error_code() ?: 'ai_attachment_failed', $extracted->get_error_message(), 400 );
321 + }
322 +
323 + // A recording has no text to extract; it goes through
324 + // `transcribe-attachment` first. Sent here directly (an older modal,
325 + // or a hand-made request) it used to fall through to the text path
326 + // and read an undefined `text` key.
327 + if ( isset( $extracted['kind'] ) && 'media' === $extracted['kind'] ) {
328 + return $this->error( 'ai_needs_transcript', __( 'Recordings are transcribed first. Use "Transcribe recording", then generate from the transcript.', 'betterdocs' ), 400 );
329 + }
330 +
331 + // Image attachment → send the picture to a vision-capable model
332 + // instead of extracting text (there is none). Same handle_doc
333 + // output path (wp_kses_post), just a multimodal request.
334 + if ( isset( $extracted['kind'] ) && 'image' === $extracted['kind'] ) {
335 + $image_prompt = trim( $prompt . "\n\n"
336 + . sprintf(
337 + /* translators: %s: the uploaded image file name. */
338 + __( 'Read the attached image "%s" and turn what it shows — its text, tables, diagrams, UI or screenshots — into structured documentation. Describe only what is actually visible in the image; do not invent details:', 'betterdocs' ),
339 + $extracted['name']
340 + ) )
341 + . $this->build_directives( $tone, $doc_size, $generate_title );
342 +
343 + return $this->handle_doc( $write_ai, $post_id, $image_prompt, $keywords, $action, $doc_size, $extra_system, $extracted );
344 + }
345 +
346 + // Extracted file text is prompt-bound source (not rendered as HTML),
347 + // so preserve angle brackets like from-source/from-git do.
348 + $file_text = $this->clip( wp_check_invalid_utf8( (string) $extracted['text'], true ), self::MAX_SOURCE_LENGTH );
349 + if ( '' === trim( $file_text ) ) {
350 + return $this->error( 'ai_empty_attachment', __( 'No readable text was found in that file.', 'betterdocs' ), 400 );
351 + }
352 +
353 + $file_prompt = trim( $prompt . "\n\n"
354 + . sprintf(
355 + /* translators: %s: the uploaded file name. */
356 + __( 'Turn the content of the uploaded file "%s" into structured documentation. Use only the information it contains; do not invent details:', 'betterdocs' ),
357 + $extracted['name']
358 + )
359 + . "\n---\n" . $file_text . "\n---" )
360 + . $this->build_directives( $tone, $doc_size, $generate_title );
361 + return $this->handle_doc( $write_ai, $post_id, $file_prompt, $keywords, $action, $doc_size, $extra_system );
362 +
227 363 case 'git-repos':
228 364 case 'git-items':
229 365 case 'git-contents':
230 366 // "Browse repository" data for the From Git tab. Read-only listing
@@ -327,9 +463,9 @@
327 463
328 464 /**
329 465 * Full-doc generation (generate-doc, expand-outline, from-source all land here).
330 466 */
331 - protected function handle_doc( $write_ai, $post_id, $prompt, $keywords, $action, $doc_size = 'any', $extra_system = array() ) {
467 + protected function handle_doc( $write_ai, $post_id, $prompt, $keywords, $action, $doc_size = 'any', $extra_system = array(), $image = null, $usage_action = null ) {
332 468 if ( '' === trim( $prompt ) ) {
333 469 return $this->error( 'ai_empty_prompt', __( 'Please provide a prompt for the AI.', 'betterdocs' ), 400 );
334 470 }
335 471
@@ -335,9 +471,20 @@
335 471
336 472 // A "long" doc can outrun the default 2500-token cap; give it headroom.
337 473 $max_tokens = 'long' === $doc_size ? 4000 : null;
338 474
339 - $content = $write_ai->generate_openai_response( $prompt, $keywords, $max_tokens, $extra_system );
475 + if ( null !== $image ) {
476 + // Image attachment: send the picture to a vision model. Returns a
477 + // WP_Error when the configured model can't read images (guard) — surface
478 + // that as a 400 so the user knows to switch models, not a 502.
479 + $content = $write_ai->generate_vision_response( $prompt, $image, $max_tokens, $extra_system );
480 + if ( is_wp_error( $content ) ) {
481 + $code = $content->get_error_code() ?: 'ai_vision_failed';
482 + return $this->error( $code, $content->get_error_message(), 'ai_no_vision' === $code ? 400 : 502 );
483 + }
484 + } else {
485 + $content = $write_ai->generate_openai_response( $prompt, $keywords, $max_tokens, $extra_system );
486 + }
340 487
341 488 if ( ! is_string( $content ) || '' === trim( $content ) ) {
342 489 return $this->error( 'empty', __( 'The AI returned no content. Try again or rephrase your prompt.', 'betterdocs' ), 502 );
343 490 }
@@ -352,9 +499,11 @@
352 499 // prompt asks the model to avoid these, but that is a soft constraint — this
353 500 // is the enforcement (a prompt-injected source/Git payload can't inject XSS).
354 501 $content = wp_kses_post( $content );
355 502
356 - AIUsage::record( 'write_with_ai', $post_id, $action );
503 + // `$usage_action` lets a caller count the generation under a different
504 + // mode than the action it answers to (a recording arrives as from-source).
505 + AIUsage::record( 'write_with_ai', $post_id, null !== $usage_action ? $usage_action : $action );
357 506
358 507 return $this->success( array( 'content' => $content, 'action' => $action ) );
359 508 }
360 509
@@ -411,10 +560,490 @@
411 560 }
412 561 return implode( "\n", $lines );
413 562 }
414 563
564 + /**
565 + * Validate the uploaded "From Attachment" file and return its extracted text.
566 + *
567 + * Security: enforces is_uploaded_file (a real HTTP upload, not an arbitrary
568 + * server path), a byte cap, and a strict extension + MIME allow-list via
569 + * wp_check_filetype(). The file is read for text only — never moved into the
570 + * uploads dir, stored, or rendered — so there is no persisted attack surface.
571 + *
572 + * @param WP_REST_Request $request
573 + * @return array{name:string,text:string}|\WP_Error
574 + */
575 + protected function read_uploaded_attachment( WP_REST_Request $request ) {
576 + $files = $request->get_file_params();
577 + if ( empty( $files['file'] ) || ! is_array( $files['file'] ) ) {
578 + return new \WP_Error( 'ai_no_file', __( 'No file was received. Choose a file to write from.', 'betterdocs' ) );
579 + }
580 +
581 + $file = $files['file'];
582 +
583 + if ( ! empty( $file['error'] ) || empty( $file['tmp_name'] ) || ! is_uploaded_file( $file['tmp_name'] ) ) {
584 + return new \WP_Error( 'ai_upload_failed', __( 'The upload did not complete — please try again.', 'betterdocs' ) );
585 + }
586 +
587 + // Strict extension + MIME allow-list. wp_check_filetype() validates the
588 + // name against exactly these types; anything else yields an empty ext.
589 + $allowed = array_merge(
590 + array(
591 + 'txt' => 'text/plain',
592 + 'md|markdown' => 'text/markdown',
593 + 'docx' => 'application/vnd.openxmlformats-officedocument.wordprocessingml.document',
594 + 'pdf' => 'application/pdf',
595 + 'png' => 'image/png',
596 + 'jpg|jpeg' => 'image/jpeg',
597 + 'webp' => 'image/webp',
598 + ),
599 + self::media_mimes()
600 + );
601 + $check = wp_check_filetype( (string) $file['name'], $allowed );
602 + $ext = strtolower( (string) $check['ext'] );
603 +
604 + $image_exts = array( 'png', 'jpg', 'jpeg', 'webp' );
605 + $text_exts = array( 'txt', 'md', 'markdown', 'docx', 'pdf' );
606 + $media_exts = self::media_exts();
607 +
608 + if ( ! in_array( $ext, array_merge( $text_exts, $image_exts, $media_exts ), true ) ) {
609 + // .mov and .avi are the two formats people actually try and that no
610 + // provider accepts, so name the fix rather than listing types again.
611 + $tried = strtolower( (string) pathinfo( (string) $file['name'], PATHINFO_EXTENSION ) );
612 + if ( in_array( $tried, array( 'mov', 'avi', 'wmv', 'mkv' ), true ) ) {
613 + return new \WP_Error(
614 + 'ai_bad_filetype',
615 + sprintf(
616 + /* translators: %s: the uploaded file's extension, e.g. "mov". */
617 + __( '.%s recordings are not supported. Export or convert it to MP4 and upload that.', 'betterdocs' ),
618 + $tried
619 + )
620 + );
621 + }
622 +
623 + return new \WP_Error( 'ai_bad_filetype', __( 'Unsupported file type. Upload a .pdf, .docx, .txt, .md, an image (.png, .jpg, .webp), or a recording (.mp3, .m4a, .wav, .flac, .ogg, .mp4, .mpeg, .webm).', 'betterdocs' ) );
624 + }
625 +
626 + // Size is capped per kind: a recording is legitimately much larger than a
627 + // text file, but the cap can never exceed what the host will actually
628 + // accept — PHP truncates a POST over post_max_size before we see it.
629 + $is_media = in_array( $ext, $media_exts, true );
630 + $cap = self::upload_cap( $is_media ? 'media' : 'file' );
631 +
632 + if ( (int) $file['size'] > $cap ) {
633 + return new \WP_Error(
634 + 'ai_file_too_large',
635 + sprintf(
636 + /* translators: %s: maximum allowed size, e.g. "25 MB". */
637 + __( 'The file exceeds the %s limit.', 'betterdocs' ),
638 + size_format( $cap )
639 + )
640 + );
641 + }
642 +
643 + // Image → send the picture itself to a vision model (there is no text to
644 + // extract). Verify it is a real image by its bytes, not just its name,
645 + // then hand back a base64 data URI for the multimodal request.
646 + if ( in_array( $ext, $image_exts, true ) ) {
647 + $raw = file_get_contents( (string) $file['tmp_name'] ); // phpcs:ignore WordPressVIPMinimum.Performance.FetchingRemoteData.FileGetContentsUnknown -- local tmp upload.
648 + if ( false === $raw || '' === $raw ) {
649 + return new \WP_Error( 'ai_read_failed', __( 'Could not read the image file.', 'betterdocs' ) );
650 + }
651 +
652 + $info = @getimagesize( (string) $file['tmp_name'] );
653 + $mime = ( is_array( $info ) && ! empty( $info['mime'] ) ) ? (string) $info['mime'] : '';
654 +
655 + if ( ! in_array( $mime, array( 'image/png', 'image/jpeg', 'image/webp' ), true ) ) {
656 + return new \WP_Error( 'ai_bad_image', __( 'That file is not a valid PNG, JPG or WEBP image.', 'betterdocs' ) );
657 + }
658 +
659 + return array(
660 + 'name' => sanitize_file_name( (string) $file['name'] ),
661 + 'kind' => 'image',
662 + 'mime' => $mime,
663 + 'data_uri' => 'data:' . $mime . ';base64,' . base64_encode( $raw ),
664 + );
665 + }
666 +
667 + // Recording → nothing to extract here; it goes to a speech-to-text model
668 + // whole. Verify the bytes really are audio/video before handing a file to
669 + // a paid endpoint: wp_check_filetype() above only read the *name*, so a
670 + // renamed binary would otherwise sail through.
671 + if ( $is_media ) {
672 + $real = wp_check_filetype_and_ext( (string) $file['tmp_name'], (string) $file['name'], $allowed );
673 + $mime = ! empty( $real['type'] ) ? (string) $real['type'] : '';
674 +
675 + if ( '' === $mime && function_exists( 'finfo_open' ) ) {
676 + // wp_check_filetype_and_ext() only sniffs images and a short list
677 + // of text formats; for A/V it hands back the name-based guess or
678 + // nothing at all. finfo is the actual byte check.
679 + $finfo = finfo_open( FILEINFO_MIME_TYPE );
680 + if ( $finfo ) {
681 + $sniffed = finfo_file( $finfo, (string) $file['tmp_name'] );
682 + finfo_close( $finfo );
683 + $mime = is_string( $sniffed ) ? $sniffed : '';
684 + }
685 + }
686 +
687 + if ( 0 !== strpos( $mime, 'audio/' ) && 0 !== strpos( $mime, 'video/' ) ) {
688 + return new \WP_Error( 'ai_bad_media', __( 'That file is not a readable audio or video recording.', 'betterdocs' ) );
689 + }
690 +
691 + return array(
692 + 'name' => sanitize_file_name( (string) $file['name'] ),
693 + 'kind' => 'media',
694 + 'mime' => $mime,
695 + 'path' => (string) $file['tmp_name'],
696 + 'size' => (int) $file['size'],
697 + );
698 + }
699 +
700 + $text = $this->extract_attachment_text( (string) $file['tmp_name'], $ext );
701 + if ( is_wp_error( $text ) ) {
702 + return $text;
703 + }
704 +
705 + return array(
706 + 'name' => sanitize_file_name( (string) $file['name'] ),
707 + 'kind' => 'text',
708 + 'text' => $text,
709 + );
710 + }
711 +
712 + /**
713 + * Extension → MIME map for the recording formats OpenAI's transcription
714 + * endpoint accepts. The video containers are here on purpose: the endpoint
715 + * reads their audio track, which is what lets this feature work without
716 + * ffmpeg on the host.
717 + *
718 + * @since 4.9.4
719 + *
720 + * @return array<string,string>
721 + */
722 + public static function media_mimes() {
723 + return array(
724 + 'mp3|mpga' => 'audio/mpeg',
725 + 'm4a' => 'audio/mp4',
726 + 'wav' => 'audio/wav',
727 + 'flac' => 'audio/flac',
728 + 'ogg|oga' => 'audio/ogg',
729 + 'mp4' => 'video/mp4',
730 + 'mpeg|mpg' => 'video/mpeg',
731 + 'webm' => 'video/webm',
732 + );
733 + }
734 +
735 + /**
736 + * Flat list of accepted recording extensions.
737 + *
738 + * @since 4.9.4
739 + *
740 + * @return string[]
741 + */
742 + public static function media_exts() {
743 + $exts = array();
744 + foreach ( array_keys( self::media_mimes() ) as $group ) {
745 + foreach ( explode( '|', $group ) as $ext ) {
746 + $exts[] = $ext;
747 + }
748 + }
749 + return $exts;
750 + }
751 +
752 + /**
753 + * Effective upload ceiling for a kind of attachment, in bytes.
754 + *
755 + * Always clamped to what the host will accept. A site with
756 + * `upload_max_filesize = 8M` cannot receive 25 MB no matter what we allow —
757 + * PHP discards the body and the request arrives empty — so advertising the
758 + * higher number would just produce an unexplained failure.
759 + *
760 + * @since 4.9.4
761 + *
762 + * @param string $kind `media` | `file`
763 + * @return int
764 + */
765 + public static function upload_cap( $kind = 'file' ) {
766 + $cap = ( 'media' === $kind ) ? self::media_cap_for_platform() : self::MAX_UPLOAD_BYTES;
767 + $host = (int) wp_max_upload_size();
768 +
769 + return ( $host > 0 && $host < $cap ) ? $host : $cap;
770 + }
771 +
772 + /**
773 + * The recording cap for the configured platform, before the host clamp.
774 + *
775 + * Gemini carries the media inline as base64 inside the JSON request, which
776 + * inflates it by roughly a third, so its practical ceiling is lower than
777 + * OpenAI's, where the file is a real multipart part.
778 + *
779 + * @since 4.9.4
780 + *
781 + * @return int
782 + */
783 + public static function media_cap_for_platform() {
784 + $factory = new ProviderFactory( betterdocs()->settings );
785 + $platform = $factory->active_platform();
786 +
787 + return ( 'gemini' === $platform ) ? self::MAX_MEDIA_BYTES_INLINE : self::MAX_MEDIA_BYTES;
788 + }
789 +
790 + /**
791 + * Extract plain text from a supported uploaded file. TXT/MD are read as-is;
792 + * DOCX is unzipped natively (ZipArchive) and its document body flattened;
793 + * PDF text is pulled natively from FlateDecode content streams — all without
794 + * a third-party parser dependency. Images/scanned PDFs (no embedded text) are
795 + * not handled here (that would need OCR / a multimodal model).
796 + *
797 + * @param string $path Local tmp upload path (already is_uploaded_file-verified).
798 + * @param string $ext Allow-listed extension.
799 + * @return string|\WP_Error
800 + */
801 + protected function extract_attachment_text( $path, $ext ) {
802 + if ( in_array( $ext, array( 'txt', 'md', 'markdown' ), true ) ) {
803 + $raw = file_get_contents( $path ); // phpcs:ignore WordPressVIPMinimum.Performance.FetchingRemoteData.FileGetContentsUnknown -- local tmp upload.
804 + return false === $raw ? new \WP_Error( 'ai_read_failed', __( 'Could not read the file.', 'betterdocs' ) ) : $raw;
805 + }
806 +
807 + if ( 'docx' === $ext ) {
808 + if ( ! class_exists( '\ZipArchive' ) ) {
809 + return new \WP_Error( 'ai_no_zip', __( 'Reading .docx files needs the PHP Zip extension, which is not available on this server. Upload a .txt or .md instead.', 'betterdocs' ) );
810 + }
811 + $zip = new \ZipArchive();
812 + if ( true !== $zip->open( $path ) ) {
813 + return new \WP_Error( 'ai_bad_docx', __( 'Could not open that .docx file — it may be corrupt.', 'betterdocs' ) );
814 + }
815 + $xml = $zip->getFromName( 'word/document.xml' );
816 + $zip->close();
817 +
818 + if ( false === $xml || '' === $xml ) {
819 + return new \WP_Error( 'ai_bad_docx', __( 'That .docx file has no readable document body.', 'betterdocs' ) );
820 + }
821 +
822 + // Turn Word paragraph/break/tab elements into whitespace, then strip
823 + // every remaining tag so only the run text (<w:t>) survives, and decode
824 + // XML entities. Keeps paragraph structure the model can read.
825 + $xml = preg_replace( '#</w:p>#', "\n\n", $xml );
826 + $xml = preg_replace( '#<w:br\b[^>]*/?>#', "\n", $xml );
827 + $xml = preg_replace( '#<w:tab\b[^>]*/?>#', "\t", $xml );
828 + $text = wp_strip_all_tags( (string) $xml );
829 + $text = html_entity_decode( $text, ENT_QUOTES | ENT_XML1, 'UTF-8' );
830 +
831 + return trim( preg_replace( "/\n{3,}/", "\n\n", $text ) );
832 + }
833 +
834 + if ( 'pdf' === $ext ) {
835 + return $this->extract_pdf_text( $path );
836 + }
837 +
838 + return new \WP_Error( 'ai_bad_filetype', __( 'Unsupported file type.', 'betterdocs' ) );
839 + }
840 +
841 + /**
842 + * Extract text from a PDF natively — no library. PDFs keep their page text in
843 + * "content streams" (usually zlib/FlateDecode-compressed); we inflate each one
844 + * and pull the operands of the text-showing operators (Tj / TJ / ' / "). This
845 + * covers the common case (real, text-based documents). It intentionally does
846 + * NOT handle:
847 + * - encrypted PDFs (no key) — reported so the user knows why,
848 + * - scanned/image-only PDFs (there is no embedded text to read) — reported,
849 + * - exotic font encodings (CID/Type0 with custom CMaps) — those decode to
850 + * garbled text, so we drop a stream whose result looks non-textual.
851 + * The extracted text is prompt-bound source only; it is never rendered, and
852 + * the model output still passes wp_kses_post downstream.
853 + *
854 + * @param string $path Local tmp upload path.
855 + * @return string|\WP_Error
856 + */
857 + protected function extract_pdf_text( $path ) {
858 + $data = file_get_contents( $path ); // phpcs:ignore WordPressVIPMinimum.Performance.FetchingRemoteData.FileGetContentsUnknown -- local tmp upload.
859 + if ( false === $data || 0 !== strncmp( $data, '%PDF', 4 ) ) {
860 + return new \WP_Error( 'ai_bad_pdf', __( 'That does not look like a valid PDF file.', 'betterdocs' ) );
861 + }
862 +
863 + // An encrypted PDF's streams won't inflate to readable text without the
864 + // key. Detect the Encrypt entry up front and say so, rather than return
865 + // empty. (An /Encrypt inside a literal string is a rare false positive we
866 + // accept — worst case the user gets the "no text" message below instead.)
867 + if ( preg_match( '/\/Encrypt\b/', $data ) ) {
868 + return new \WP_Error( 'ai_pdf_encrypted', __( 'This PDF is password-protected or encrypted, so its text can\'t be read. Remove the protection, or paste the text instead.', 'betterdocs' ) );
869 + }
870 +
871 + $out = '';
872 + $cap = self::MAX_SOURCE_LENGTH + 4000; // stop early; the source is clipped later anyway.
873 +
874 + if ( preg_match_all( '/stream\r?\n(.*?)\r?\nendstream/s', $data, $streams ) ) {
875 + foreach ( $streams[1] as $chunk ) {
876 + // Try zlib (FlateDecode) first, then raw-deflate, then treat as
877 + // already-plain. @-silenced: a binary (image/font) stream simply
878 + // fails to inflate and is skipped below.
879 + $decoded = @gzuncompress( $chunk );
880 + if ( false === $decoded ) {
881 + $decoded = @gzinflate( $chunk );
882 + }
883 + $content = ( is_string( $decoded ) && '' !== $decoded ) ? $decoded : $chunk;
884 +
885 + // Only content streams carry text-showing operators; skip the rest
886 + // (images, fonts) so we don't scrape binary noise.
887 + if ( false === strpos( $content, 'Tj' ) && false === strpos( $content, 'TJ' ) ) {
888 + continue;
889 + }
890 +
891 + $piece = $this->pdf_stream_text( $content );
892 + // Guard against garbled CID/font-encoded streams: if the decoded
893 + // "text" is mostly non-printable, drop it rather than inject noise.
894 + if ( '' !== $piece && $this->mostly_printable( $piece ) ) {
895 + $out .= $piece . "\n";
896 + if ( strlen( $out ) > $cap ) {
897 + break;
898 + }
899 + }
900 + }
901 + }
902 +
903 + $out = preg_replace( "/[ \t]+/", ' ', $out );
904 + $out = trim( preg_replace( "/\n{3,}/", "\n\n", $out ) );
905 +
906 + // Subsetted LaTeX/CID fonts emit control bytes (ligatures) and non-UTF-8
907 + // sequences among the readable text. Strip them and coerce to valid UTF-8 —
908 + // otherwise the caller's wp_check_invalid_utf8() discards the ENTIRE string
909 + // on the first bad byte and an 8-page paper looks empty ("no readable text").
910 + $out = $this->to_clean_utf8( $out );
911 +
912 + if ( '' === $out ) {
913 + return new \WP_Error(
914 + 'ai_pdf_no_text',
915 + __( 'No selectable text was found in that PDF — it may be a scanned image. Try a text-based PDF, or paste the content into the prompt.', 'betterdocs' )
916 + );
917 + }
918 +
919 + return $out;
920 + }
921 +
922 + /**
923 + * Pull the visible text out of one decoded PDF content stream. Positioning
924 + * operators (Td/TD/T*) become newlines; the literal `( … )` and hex `< … >`
925 + * operands of Tj/TJ/'/'' become the text. Kerning numbers inside TJ arrays are
926 + * ignored (their effect on spacing is cosmetic for our purposes).
927 + */
928 + protected function pdf_stream_text( $content ) {
929 + // Text-positioning operators (new line / new paragraph) become newlines so
930 + // words on different lines don't run together.
931 + $content = preg_replace( '/\b(?:T\*|Td|TD)\b/', " \n ", $content );
932 +
933 + // Walk TJ arrays and Tj/'/'" strings in document order. Inside a TJ array
934 + // pdfTeX (LaTeX) renders an inter-word space as a large negative kerning
935 + // number, not a literal space in the string — so we synthesise a space when
936 + // the kerning passes a threshold, otherwise every word runs together
937 + // ("FormallyVerifiedand…"). Small kerning (letter pairs) is ignored.
938 + if ( ! preg_match_all(
939 + '/\[((?:\\\\.|[^\]\\\\])*)\]\s*TJ|(\((?:\\\\.|[^\\\\()])*\)|<[0-9A-Fa-f\s]+>)\s*(?:Tj|\'|")|(\n)/s',
940 + $content,
941 + $matches,
942 + PREG_SET_ORDER
943 + ) ) {
944 + return '';
945 + }
946 +
947 + $text = '';
948 + foreach ( $matches as $tok ) {
949 + if ( isset( $tok[3] ) && "\n" === $tok[3] ) {
950 + $text .= "\n";
951 + continue;
952 + }
953 + if ( isset( $tok[1] ) && '' !== $tok[1] ) {
954 + // TJ array: alternating string operands and kerning numbers.
955 + preg_match_all( '/\((?:\\\\.|[^\\\\()])*\)|<[0-9A-Fa-f\s]+>|-?\d+(?:\.\d+)?/s', $tok[1], $parts );
956 + foreach ( $parts[0] as $part ) {
957 + if ( '(' === $part[0] || '<' === $part[0] ) {
958 + $text .= $this->pdf_token_text( $part );
959 + } elseif ( (float) $part < -100 ) {
960 + $text .= ' ';
961 + }
962 + }
963 + $text .= ' ';
964 + } elseif ( isset( $tok[2] ) && '' !== $tok[2] ) {
965 + $text .= $this->pdf_token_text( $tok[2] ) . ' ';
966 + }
967 + }
968 +
969 + return preg_replace( '/[^\S\n]+/', ' ', $text );
970 + }
971 +
972 + /**
973 + * Decode one PDF string operand — a literal `( … )` (with escapes) or a hex
974 + * `< … >` string — into its raw bytes.
975 + */
976 + protected function pdf_token_text( $token ) {
977 + if ( '(' === $token[0] ) {
978 + return $this->pdf_unescape( substr( $token, 1, -1 ) );
979 + }
980 + $hex = preg_replace( '/[^0-9A-Fa-f]/', '', $token );
981 + return ( '' === $hex ) ? '' : (string) @hex2bin( strlen( $hex ) % 2 ? substr( $hex, 0, -1 ) : $hex );
982 + }
983 +
984 + /**
985 + * Resolve PDF string escapes: \( \) \\ \n \r \t \b \f and \ddd octal codes.
986 + */
987 + protected function pdf_unescape( $string ) {
988 + return preg_replace_callback(
989 + '/\\\\(?:([nrtbf()\\\\])|([0-7]{1,3}))/',
990 + function ( $mm ) {
991 + if ( isset( $mm[1] ) && '' !== $mm[1] ) {
992 + $map = array( 'n' => "\n", 'r' => "\r", 't' => "\t", 'b' => "\x08", 'f' => "\x0C", '(' => '(', ')' => ')', '\\' => '\\' );
993 + return isset( $map[ $mm[1] ] ) ? $map[ $mm[1] ] : $mm[1];
994 + }
995 + return chr( octdec( $mm[2] ) & 0xFF );
996 + },
997 + $string
998 + );
999 + }
1000 +
1001 + /**
1002 + * Is this decoded string mostly readable text? Used to drop font/CID streams
1003 + * that decode to binary-looking garbage. Counts printable + common whitespace.
1004 + */
1005 + protected function mostly_printable( $string ) {
1006 + $len = strlen( $string );
1007 + if ( 0 === $len ) {
1008 + return false;
1009 + }
1010 + $printable = preg_match_all( '/[\P{Cc}\t\n\r]/u', $string );
1011 + // Fallback for non-UTF-8 payloads where \p{} may not match cleanly.
1012 + if ( false === $printable ) {
1013 + $printable = strlen( preg_replace( '/[^\x09\x0A\x0D\x20-\x7E]/', '', $string ) );
1014 + }
1015 + return ( $printable / $len ) >= 0.7;
1016 + }
1017 +
415 1018 protected function clip( $value, $max ) {
416 - return strlen( $value ) > $max ? substr( $value, 0, $max ) : $value;
1019 + if ( strlen( $value ) <= $max ) {
1020 + return $value;
1021 + }
1022 +
1023 + // Byte-limited but never mid-character: a split multi-byte sequence is
1024 + // invalid UTF-8, which wp_json_encode() refuses — the whole request
1025 + // body would go out empty.
1026 + if ( function_exists( 'mb_strcut' ) ) {
1027 + return mb_strcut( $value, 0, $max, 'UTF-8' );
1028 + }
1029 +
1030 + return wp_check_invalid_utf8( substr( $value, 0, $max ), true );
1031 + }
1032 +
1033 + /**
1034 + * Coerce extracted PDF bytes to clean, valid UTF-8: drop C0/C1 control bytes
1035 + * (except tab/newline) and any byte sequence that isn't valid UTF-8. This keeps
1036 + * the readable text intact for the downstream wp_check_invalid_utf8(), which
1037 + * would otherwise discard the whole string on a single invalid byte.
1038 + */
1039 + protected function to_clean_utf8( $string ) {
1040 + $string = preg_replace( '/[\x00-\x08\x0B\x0C\x0E-\x1F\x7F]/', '', (string) $string );
1041 + if ( '' !== $string && ! preg_match( '//u', $string ) ) {
1042 + $converted = @iconv( 'UTF-8', 'UTF-8//IGNORE', $string );
1043 + $string = ( false !== $converted ) ? $converted : preg_replace( '/[^\x09\x0A\x20-\x7E]/', '', $string );
1044 + }
1045 + return (string) $string;
417 1046 }
418 1047
419 1048 /**
420 1049 * Frame the user's free-form request as a documentation instruction. Returns