| @@ -20,8 +20,10 @@ | ||
| 20 | 20 | |
| 21 | 21 | use WordPress\AiClient\Messages\DTO\Message; |
| 22 | 22 | use WordPress\AiClient\Messages\DTO\MessagePart; |
| 23 | 23 | use WordPress\AiClient\Messages\DTO\UserMessage; |
| 24 | +use WordPress\AiClient\Providers\Models\Contracts\ModelInterface; | |
| 25 | +use WordPress\AiClient\Providers\Models\DTO\ModelConfig; | |
| 24 | 26 | use WordPress\AiClient\Tools\DTO\FunctionCall; |
| 25 | 27 | use WordPress\AiClient\Tools\DTO\FunctionDeclaration; |
| 26 | 28 | use WordPress\AiClient\Tools\DTO\FunctionResponse; |
| 27 | 29 | |
| @@ -119,8 +121,185 @@ | ||
| 119 | 121 | return new Message( $message->getRole(), $kept ); |
| 120 | 122 | } |
| 121 | 123 | |
| 122 | 124 | /** |
| 125 | + * Output-token ceiling for every generation turn the site's | |
| 126 | + * `openstation_ai_model_config` filter leaves uncapped. | |
| 127 | + * | |
| 128 | + * The three default providers disagree about what "no ceiling" means. | |
| 129 | + * The OpenAI and Google providers send none, so the model's own maximum | |
| 130 | + * applies; the Anthropic provider must send one and falls back to a | |
| 131 | + * hard-coded 4096. That is enough for a chat answer and nowhere near | |
| 132 | + * enough for a tool call carrying a whole post: the model runs out of | |
| 133 | + * room inside the call's JSON, the API returns only the argument pairs | |
| 134 | + * that were complete before the cut, and the ability rejects the call | |
| 135 | + * for its missing `content`. The Localizer failed seven translations of | |
| 136 | + * one post in a row exactly that way, each attempt cut at the same place. | |
| 137 | + * | |
| 138 | + * 16384 is the largest value every current-generation model of the three | |
| 139 | + * providers accepts (OpenAI's gpt-4o family caps output at exactly that). | |
| 140 | + * A site that needs more, or pins an older model with a smaller limit, | |
| 141 | + * sets `max_tokens` in the filter: the filter's value always wins. | |
| 142 | + */ | |
| 143 | +const OPENSTATION_AI_DEFAULT_MAX_TOKENS = 16384; | |
| 144 | + | |
| 145 | +/** | |
| 146 | + * Whether the provider stopped because the reply hit the output-token | |
| 147 | + * ceiling. | |
| 148 | + * | |
| 149 | + * Anthropic's `max_tokens` and Google's `MAX_TOKENS` stop reasons both | |
| 150 | + * map to the SDK's LENGTH finish reason. The OpenAI provider throws | |
| 151 | + * instead and never builds a result; {@see openstation_ai_client_generate()} | |
| 152 | + * maps that path from the WP_Error Core turns the exception into. | |
| 153 | + * | |
| 154 | + * @param mixed $result GenerativeAiResult. | |
| 155 | + * @return bool | |
| 156 | + */ | |
| 157 | +function openstation_ai_result_is_truncated( $result ) { | |
| 158 | + try { | |
| 159 | + foreach ( $result->getCandidates() as $candidate ) { | |
| 160 | + if ( $candidate->getFinishReason()->isLength() ) { | |
| 161 | + return true; | |
| 162 | + } | |
| 163 | + } | |
| 164 | + } catch ( \Throwable $e ) { | |
| 165 | + return false; | |
| 166 | + } | |
| 167 | + return false; | |
| 168 | +} | |
| 169 | + | |
| 170 | +/** | |
| 171 | + * Builds the error for a turn the output-token ceiling cut short. | |
| 172 | + * | |
| 173 | + * A truncated reply is never usable. A JSON answer no longer parses, | |
| 174 | + * and a function call arrives with only the argument pairs that were | |
| 175 | + * complete before the cut, so the ability rejects it for a missing | |
| 176 | + * required field, and the model, reading its own truncated call back | |
| 177 | + * from history, sends the same call again to the same end. Failing the | |
| 178 | + * turn here turns a run of silent retries into one error that names | |
| 179 | + * the cause. | |
| 180 | + * | |
| 181 | + * @param string $detail Underlying provider detail, preserved for logs. | |
| 182 | + * @param array|null $usage Token usage of the truncated turn, if known. | |
| 183 | + * @return WP_Error | |
| 184 | + */ | |
| 185 | +function openstation_ai_output_truncated_error( $detail, $usage = null ) { | |
| 186 | + return new WP_Error( | |
| 187 | + 'openstation_ai_output_truncated', | |
| 188 | + __( 'The AI provider cut the reply short at the output-token limit.', 'desktop-mode' ), | |
| 189 | + array( | |
| 190 | + 'status' => 502, | |
| 191 | + 'detail' => (string) $detail, | |
| 192 | + 'completion_tokens' => is_array( $usage ) && isset( $usage['completion'] ) ? (int) $usage['completion'] : null, | |
| 193 | + ) | |
| 194 | + ); | |
| 195 | +} | |
| 196 | + | |
| 197 | +/** | |
| 198 | + * Builds the error for a final turn that produced no answer text. | |
| 199 | + * | |
| 200 | + * Observed live with the Anthropic provider under agent runs: a hard task | |
| 201 | + * spends the entire `max_tokens` budget inside a thinking block | |
| 202 | + * (`stop_reason: "max_tokens"`, a single text-less thought part), so the | |
| 203 | + * turn carries neither function calls nor extractable text. Callers that | |
| 204 | + * can meaningfully degrade instead (the command follow-up turn) match on | |
| 205 | + * this code and keep their own fallback. | |
| 206 | + * | |
| 207 | + * @param string $detail Underlying extraction failure, preserved for logs. | |
| 208 | + * @return WP_Error | |
| 209 | + */ | |
| 210 | +function openstation_ai_empty_answer_error( $detail ) { | |
| 211 | + return new WP_Error( | |
| 212 | + 'openstation_ai_empty_answer', | |
| 213 | + __( 'The AI provider returned no answer text.', 'desktop-mode' ), | |
| 214 | + array( | |
| 215 | + 'status' => 502, | |
| 216 | + 'detail' => (string) $detail, | |
| 217 | + ) | |
| 218 | + ); | |
| 219 | +} | |
| 220 | + | |
| 221 | +/** | |
| 222 | + * Applies the site's model config to a prompt builder. | |
| 223 | + * | |
| 224 | + * @param mixed $builder WP_AI_Client_Prompt_Builder. | |
| 225 | + * @param array $context Partial filter context; missing keys are defaulted. | |
| 226 | + * @return mixed | |
| 227 | + */ | |
| 228 | +function openstation_ai_apply_model_config( $builder, array $context ) { | |
| 229 | + $context = array_merge( | |
| 230 | + array( | |
| 231 | + 'user_id' => 0, | |
| 232 | + 'request_id' => '', | |
| 233 | + 'source' => '', | |
| 234 | + 'has_tools' => false, | |
| 235 | + 'has_schema' => false, | |
| 236 | + ), | |
| 237 | + $context | |
| 238 | + ); | |
| 239 | + | |
| 240 | + /** | |
| 241 | + * Filters the model config for one AI turn. | |
| 242 | + * | |
| 243 | + * Defaults to empty; the only value OpenStation fills in afterwards is | |
| 244 | + * `max_tokens` ({@see OPENSTATION_AI_DEFAULT_MAX_TOKENS}), and only | |
| 245 | + * when the filter left it unset. Recipe: `docs/examples/ai-model-config.md`. | |
| 246 | + * | |
| 247 | + * @param array $config { model?: string|ModelInterface, max_tokens?: int, temperature?: float, custom_options?: array<string, mixed> }. | |
| 248 | + * @param array $context { user_id, request_id, source, has_tools, has_schema }. | |
| 249 | + */ | |
| 250 | + $config = apply_filters( 'openstation_ai_model_config', array(), $context ); | |
| 251 | + if ( ! is_array( $config ) ) { | |
| 252 | + $config = array(); | |
| 253 | + } | |
| 254 | + | |
| 255 | + $model_config = new ModelConfig(); | |
| 256 | + | |
| 257 | + $max_tokens = OPENSTATION_AI_DEFAULT_MAX_TOKENS; | |
| 258 | + if ( isset( $config['max_tokens'] ) && is_numeric( $config['max_tokens'] ) && (int) $config['max_tokens'] > 0 ) { | |
| 259 | + $max_tokens = (int) $config['max_tokens']; | |
| 260 | + } | |
| 261 | + $model_config->setMaxTokens( $max_tokens ); | |
| 262 | + | |
| 263 | + // Unlike max_tokens, 0.0 is a legitimate temperature (deterministic). The | |
| 264 | + // 2.0 ceiling is the range the SDK's own schema declares. | |
| 265 | + if ( isset( $config['temperature'] ) && is_numeric( $config['temperature'] ) | |
| 266 | + && (float) $config['temperature'] >= 0.0 && (float) $config['temperature'] <= 2.0 ) { | |
| 267 | + $model_config->setTemperature( (float) $config['temperature'] ); | |
| 268 | + } | |
| 269 | + | |
| 270 | + $custom_options = array(); | |
| 271 | + if ( isset( $config['custom_options'] ) && is_array( $config['custom_options'] ) ) { | |
| 272 | + foreach ( $config['custom_options'] as $key => $value ) { | |
| 273 | + // A list would reach the provider as parameters named `0`, `1`, …. | |
| 274 | + if ( is_string( $key ) && '' !== $key ) { | |
| 275 | + $custom_options[ $key ] = $value; | |
| 276 | + } | |
| 277 | + } | |
| 278 | + } | |
| 279 | + | |
| 280 | + if ( ! empty( $custom_options ) ) { | |
| 281 | + $model_config->setCustomOptions( $custom_options ); | |
| 282 | + } | |
| 283 | + | |
| 284 | + $builder = $builder->using_model_config( $model_config ); | |
| 285 | + | |
| 286 | + // After the config: `using_model()` merges the model's own defaults under | |
| 287 | + // whatever the builder already carries, so ours has to land first. | |
| 288 | + $model = isset( $config['model'] ) ? $config['model'] : null; | |
| 289 | + if ( $model instanceof ModelInterface ) { | |
| 290 | + $builder = $builder->using_model( $model ); | |
| 291 | + } elseif ( is_string( $model ) && '' !== trim( $model ) ) { | |
| 292 | + // `using_model()` needs a ModelInterface, so a bare model id goes | |
| 293 | + // through `using_model_preference()`, which throws on anything that | |
| 294 | + // isn't a non-empty string. | |
| 295 | + $builder = $builder->using_model_preference( trim( $model ) ); | |
| 296 | + } | |
| 297 | + | |
| 298 | + return $builder; | |
| 299 | +} | |
| 300 | + | |
| 301 | +/** | |
| 123 | 302 | * Runs one generation turn through the AI Client. |
| 124 | 303 | * |
| 125 | 304 | * Rebuilds the prompt from the full ordered message list each turn (the |
| 126 | 305 | * builder's `with_history()` prepends, so it can't append turns in a loop), |
| @@ -128,19 +307,18 @@ | ||
| 128 | 307 | * answer to `$answer_schema` when given. Returns the assistant turn normalized |
| 129 | 308 | * to the shape the loop consumes; `message` has thought-channel parts stripped |
| 130 | 309 | * ({@see openstation_ai_strip_thought_parts()}) so it is safe to replay. |
| 131 | 310 | * |
| 132 | - * @param int $user_id Requesting user id. Currently unused — the | |
| 133 | - * provider comes from Connectors and no | |
| 134 | - * per-user preference is applied; retained for | |
| 135 | - * signature stability and future attribution. | |
| 311 | + * @param int $user_id Requesting user id. | |
| 136 | 312 | * @param array $messages Ordered conversation as SDK Message objects. |
| 137 | 313 | * @param array $tool_defs Tool definitions to advertise. |
| 138 | 314 | * @param array|null $answer_schema JSON Schema for the final answer, or null. |
| 139 | 315 | * @param string $instructions System instruction. |
| 316 | + * @param array $context Optional. `{ source?: string, request_id?: string }` | |
| 317 | + * for the model-config filter. | |
| 140 | 318 | * @return array{ text: ?string, function_calls: array, message: mixed, usage: ?array, model: ?array }|WP_Error |
| 141 | 319 | */ |
| 142 | -function openstation_ai_client_generate( $user_id, array $messages, array $tool_defs, $answer_schema, $instructions ) { | |
| 320 | +function openstation_ai_client_generate( $user_id, array $messages, array $tool_defs, $answer_schema, $instructions, array $context = array() ) { | |
| 143 | 321 | $builder = wp_ai_client_prompt( $messages ); |
| 144 | 322 | |
| 145 | 323 | if ( is_string( $instructions ) && '' !== $instructions ) { |
| 146 | 324 | $builder = $builder->using_system_instruction( $instructions ); |
| @@ -145,10 +323,10 @@ | ||
| 145 | 323 | if ( is_string( $instructions ) && '' !== $instructions ) { |
| 146 | 324 | $builder = $builder->using_system_instruction( $instructions ); |
| 147 | 325 | } |
| 148 | 326 | |
| 149 | - // Provider + model selection is delegated entirely to the Core AI Client | |
| 150 | - // (Connector-backed); OpenStation pins neither. | |
| 327 | + // Provider + model selection is delegated to the Core AI Client | |
| 328 | + // (Connector-backed) unless the model-config filter says otherwise. | |
| 151 | 329 | |
| 152 | 330 | $declarations = openstation_ai_build_function_declarations( $tool_defs ); |
| 153 | 331 | if ( ! empty( $declarations ) ) { |
| 154 | 332 | $builder = $builder->using_function_declarations( ...$declarations ); |
| @@ -160,16 +338,37 @@ | ||
| 160 | 338 | // whole turn. Normalize here so no schema author has to know that. |
| 161 | 339 | $builder = $builder->as_json_response( openstation_ai_normalize_response_schema( $answer_schema ) ); |
| 162 | 340 | } |
| 163 | 341 | |
| 342 | + $builder = openstation_ai_apply_model_config( | |
| 343 | + $builder, | |
| 344 | + array_merge( | |
| 345 | + $context, | |
| 346 | + array( | |
| 347 | + 'user_id' => (int) $user_id, | |
| 348 | + 'has_tools' => ! empty( $declarations ), | |
| 349 | + 'has_schema' => is_array( $answer_schema ), | |
| 350 | + ) | |
| 351 | + ) | |
| 352 | + ); | |
| 353 | + | |
| 164 | 354 | $result = $builder->generate_result(); |
| 165 | 355 | if ( is_wp_error( $result ) ) { |
| 356 | + // The OpenAI provider reports the output ceiling as an exception | |
| 357 | + // rather than a finish reason; Core maps it to this code. | |
| 358 | + if ( 'prompt_token_limit_reached' === $result->get_error_code() ) { | |
| 359 | + return openstation_ai_output_truncated_error( $result->get_error_message() ); | |
| 360 | + } | |
| 166 | 361 | return $result; |
| 167 | 362 | } |
| 168 | 363 | |
| 169 | 364 | $message = $result->toMessage(); |
| 170 | 365 | $function_calls = array(); |
| 366 | + $has_text = false; | |
| 171 | 367 | foreach ( $message->getParts() as $part ) { |
| 368 | + if ( $part->getType()->isText() && ! $part->getChannel()->isThought() ) { | |
| 369 | + $has_text = true; | |
| 370 | + } | |
| 172 | 371 | if ( ! $part->getType()->isFunctionCall() ) { |
| 173 | 372 | continue; |
| 174 | 373 | } |
| 175 | 374 | $call = $part->getFunctionCall(); |
| @@ -183,14 +382,32 @@ | ||
| 183 | 382 | 'arguments' => wp_json_encode( is_array( $args ) ? $args : array() ), |
| 184 | 383 | ); |
| 185 | 384 | } |
| 186 | 385 | |
| 386 | + // Whatever the ceiling cut off is partial, and partial is unusable: | |
| 387 | + // a function call missing its longest argument, or a JSON answer | |
| 388 | + // missing its closing half. A budget spent entirely on reasoning | |
| 389 | + // leaves nothing written at all; that case keeps its own error below. | |
| 390 | + if ( ( ! empty( $function_calls ) || $has_text ) && openstation_ai_result_is_truncated( $result ) ) { | |
| 391 | + return openstation_ai_output_truncated_error( | |
| 392 | + 'The provider stopped at the output-token ceiling (finish reason: length).', | |
| 393 | + openstation_ai_result_token_usage( $result ) | |
| 394 | + ); | |
| 395 | + } | |
| 396 | + | |
| 187 | 397 | $text = null; |
| 188 | 398 | if ( empty( $function_calls ) ) { |
| 399 | + // A turn with no function calls IS the final answer, so failing to | |
| 400 | + // extract its text is a failed generation, not a valid empty one. | |
| 401 | + // Swallowing it here used to surface as a "successful" run with an | |
| 402 | + // empty answer, invisible to the retry and error paths alike. | |
| 189 | 403 | try { |
| 190 | 404 | $text = $result->toText(); |
| 191 | 405 | } catch ( \Throwable $e ) { |
| 192 | - $text = null; | |
| 406 | + return openstation_ai_empty_answer_error( $e->getMessage() ); | |
| 407 | + } | |
| 408 | + if ( ! is_string( $text ) || '' === trim( $text ) ) { | |
| 409 | + return openstation_ai_empty_answer_error( 'The provider response contains no text part.' ); | |
| 193 | 410 | } |
| 194 | 411 | } |
| 195 | 412 | |
| 196 | 413 | return array( |