| @@ -121,8 +121,81 @@ | ||
| 121 | 121 | return new Message( $message->getRole(), $kept ); |
| 122 | 122 | } |
| 123 | 123 | |
| 124 | 124 | /** |
| 125 | + * Output-token ceiling for every generation turn the site's | |
| 126 | + * `openstation_ai_model_config` filter leaves uncapped. | |
| 127 | + * | |
| 128 | + * The three default providers disagree about what "no ceiling" means. | |
| 129 | + * The OpenAI and Google providers send none, so the model's own maximum | |
| 130 | + * applies; the Anthropic provider must send one and falls back to a | |
| 131 | + * hard-coded 4096. That is enough for a chat answer and nowhere near | |
| 132 | + * enough for a tool call carrying a whole post: the model runs out of | |
| 133 | + * room inside the call's JSON, the API returns only the argument pairs | |
| 134 | + * that were complete before the cut, and the ability rejects the call | |
| 135 | + * for its missing `content`. The Localizer failed seven translations of | |
| 136 | + * one post in a row exactly that way, each attempt cut at the same place. | |
| 137 | + * | |
| 138 | + * 16384 is the largest value every current-generation model of the three | |
| 139 | + * providers accepts (OpenAI's gpt-4o family caps output at exactly that). | |
| 140 | + * A site that needs more, or pins an older model with a smaller limit, | |
| 141 | + * sets `max_tokens` in the filter: the filter's value always wins. | |
| 142 | + */ | |
| 143 | +const OPENSTATION_AI_DEFAULT_MAX_TOKENS = 16384; | |
| 144 | + | |
| 145 | +/** | |
| 146 | + * Whether the provider stopped because the reply hit the output-token | |
| 147 | + * ceiling. | |
| 148 | + * | |
| 149 | + * Anthropic's `max_tokens` and Google's `MAX_TOKENS` stop reasons both | |
| 150 | + * map to the SDK's LENGTH finish reason. The OpenAI provider throws | |
| 151 | + * instead and never builds a result; {@see openstation_ai_client_generate()} | |
| 152 | + * maps that path from the WP_Error Core turns the exception into. | |
| 153 | + * | |
| 154 | + * @param mixed $result GenerativeAiResult. | |
| 155 | + * @return bool | |
| 156 | + */ | |
| 157 | +function openstation_ai_result_is_truncated( $result ) { | |
| 158 | + try { | |
| 159 | + foreach ( $result->getCandidates() as $candidate ) { | |
| 160 | + if ( $candidate->getFinishReason()->isLength() ) { | |
| 161 | + return true; | |
| 162 | + } | |
| 163 | + } | |
| 164 | + } catch ( \Throwable $e ) { | |
| 165 | + return false; | |
| 166 | + } | |
| 167 | + return false; | |
| 168 | +} | |
| 169 | + | |
| 170 | +/** | |
| 171 | + * Builds the error for a turn the output-token ceiling cut short. | |
| 172 | + * | |
| 173 | + * A truncated reply is never usable. A JSON answer no longer parses, | |
| 174 | + * and a function call arrives with only the argument pairs that were | |
| 175 | + * complete before the cut, so the ability rejects it for a missing | |
| 176 | + * required field, and the model, reading its own truncated call back | |
| 177 | + * from history, sends the same call again to the same end. Failing the | |
| 178 | + * turn here turns a run of silent retries into one error that names | |
| 179 | + * the cause. | |
| 180 | + * | |
| 181 | + * @param string $detail Underlying provider detail, preserved for logs. | |
| 182 | + * @param array|null $usage Token usage of the truncated turn, if known. | |
| 183 | + * @return WP_Error | |
| 184 | + */ | |
| 185 | +function openstation_ai_output_truncated_error( $detail, $usage = null ) { | |
| 186 | + return new WP_Error( | |
| 187 | + 'openstation_ai_output_truncated', | |
| 188 | + __( 'The AI provider cut the reply short at the output-token limit.', 'desktop-mode' ), | |
| 189 | + array( | |
| 190 | + 'status' => 502, | |
| 191 | + 'detail' => (string) $detail, | |
| 192 | + 'completion_tokens' => is_array( $usage ) && isset( $usage['completion'] ) ? (int) $usage['completion'] : null, | |
| 193 | + ) | |
| 194 | + ); | |
| 195 | +} | |
| 196 | + | |
| 197 | +/** | |
| 125 | 198 | * Builds the error for a final turn that produced no answer text. |
| 126 | 199 | * |
| 127 | 200 | * Observed live with the Anthropic provider under agent runs: a hard task |
| 128 | 201 | * spends the entire `max_tokens` budget inside a thinking block |
| @@ -166,9 +239,11 @@ | ||
| 166 | 239 | |
| 167 | 240 | /** |
| 168 | 241 | * Filters the model config for one AI turn. |
| 169 | 242 | * |
| 170 | - * Defaults to empty. Recipe: `docs/examples/ai-model-config.md`. | |
| 243 | + * Defaults to empty; the only value OpenStation fills in afterwards is | |
| 244 | + * `max_tokens` ({@see OPENSTATION_AI_DEFAULT_MAX_TOKENS}), and only | |
| 245 | + * when the filter left it unset. Recipe: `docs/examples/ai-model-config.md`. | |
| 171 | 246 | * |
| 172 | 247 | * @param array $config { model?: string|ModelInterface, max_tokens?: int, temperature?: float, custom_options?: array<string, mixed> }. |
| 173 | 248 | * @param array $context { user_id, request_id, source, has_tools, has_schema }. |
| 174 | 249 | */ |
| @@ -173,16 +248,18 @@ | ||
| 173 | 248 | * @param array $context { user_id, request_id, source, has_tools, has_schema }. |
| 174 | 249 | */ |
| 175 | 250 | $config = apply_filters( 'openstation_ai_model_config', array(), $context ); |
| 176 | 251 | if ( ! is_array( $config ) ) { |
| 177 | - return $builder; | |
| 252 | + $config = array(); | |
| 178 | 253 | } |
| 179 | 254 | |
| 180 | 255 | $model_config = new ModelConfig(); |
| 181 | 256 | |
| 257 | + $max_tokens = OPENSTATION_AI_DEFAULT_MAX_TOKENS; | |
| 182 | 258 | if ( isset( $config['max_tokens'] ) && is_numeric( $config['max_tokens'] ) && (int) $config['max_tokens'] > 0 ) { |
| 183 | - $model_config->setMaxTokens( (int) $config['max_tokens'] ); | |
| 259 | + $max_tokens = (int) $config['max_tokens']; | |
| 184 | 260 | } |
| 261 | + $model_config->setMaxTokens( $max_tokens ); | |
| 185 | 262 | |
| 186 | 263 | // Unlike max_tokens, 0.0 is a legitimate temperature (deterministic). The |
| 187 | 264 | // 2.0 ceiling is the range the SDK's own schema declares. |
| 188 | 265 | if ( isset( $config['temperature'] ) && is_numeric( $config['temperature'] ) |
| @@ -275,14 +352,23 @@ | ||
| 275 | 352 | ); |
| 276 | 353 | |
| 277 | 354 | $result = $builder->generate_result(); |
| 278 | 355 | if ( is_wp_error( $result ) ) { |
| 356 | + // The OpenAI provider reports the output ceiling as an exception | |
| 357 | + // rather than a finish reason; Core maps it to this code. | |
| 358 | + if ( 'prompt_token_limit_reached' === $result->get_error_code() ) { | |
| 359 | + return openstation_ai_output_truncated_error( $result->get_error_message() ); | |
| 360 | + } | |
| 279 | 361 | return $result; |
| 280 | 362 | } |
| 281 | 363 | |
| 282 | 364 | $message = $result->toMessage(); |
| 283 | 365 | $function_calls = array(); |
| 366 | + $has_text = false; | |
| 284 | 367 | foreach ( $message->getParts() as $part ) { |
| 368 | + if ( $part->getType()->isText() && ! $part->getChannel()->isThought() ) { | |
| 369 | + $has_text = true; | |
| 370 | + } | |
| 285 | 371 | if ( ! $part->getType()->isFunctionCall() ) { |
| 286 | 372 | continue; |
| 287 | 373 | } |
| 288 | 374 | $call = $part->getFunctionCall(); |
| @@ -293,8 +379,19 @@ | ||
| 293 | 379 | $function_calls[] = array( |
| 294 | 380 | 'name' => (string) $call->getName(), |
| 295 | 381 | 'call_id' => (string) $call->getId(), |
| 296 | 382 | 'arguments' => wp_json_encode( is_array( $args ) ? $args : array() ), |
| 383 | + ); | |
| 384 | + } | |
| 385 | + | |
| 386 | + // Whatever the ceiling cut off is partial, and partial is unusable: | |
| 387 | + // a function call missing its longest argument, or a JSON answer | |
| 388 | + // missing its closing half. A budget spent entirely on reasoning | |
| 389 | + // leaves nothing written at all; that case keeps its own error below. | |
| 390 | + if ( ( ! empty( $function_calls ) || $has_text ) && openstation_ai_result_is_truncated( $result ) ) { | |
| 391 | + return openstation_ai_output_truncated_error( | |
| 392 | + 'The provider stopped at the output-token ceiling (finish reason: length).', | |
| 393 | + openstation_ai_result_token_usage( $result ) | |
| 297 | 394 | ); |
| 298 | 395 | } |
| 299 | 396 | |
| 300 | 397 | $text = null; |