PluginProbe
OpenStation: Desktop Windows, Dock & Virtual Desktops for WP Admin / 1.1.12
OpenStation: Desktop Windows, Dock & Virtual Desktops for WP Admin v1.1.12
1.1.12 1.1.11 1.1.10 1.1.9 1.1.8 1.1.7 1.1.6 1.1.5 1.1.4 1.1.3 1.1.2 1.1.1 1.1.0 1.0.1 1.0.0 0.9.8 0.9.7 0.9.6 0.9.4 0.9.5 0.9.3 0.9.2 0.9.1 0.9.0 0.8.9 All 36 releases
← All changes | includes/ai-copilot/client.php +100 -3 1.1.0 → 1.1.12 View file →
@@ -121,8 +121,81 @@
121 121 return new Message( $message->getRole(), $kept );
122 122 }
123 123
124 124 /**
125 + * Output-token ceiling for every generation turn the site's
126 + * `openstation_ai_model_config` filter leaves uncapped.
127 + *
128 + * The three default providers disagree about what "no ceiling" means.
129 + * The OpenAI and Google providers send none, so the model's own maximum
130 + * applies; the Anthropic provider must send one and falls back to a
131 + * hard-coded 4096. That is enough for a chat answer and nowhere near
132 + * enough for a tool call carrying a whole post: the model runs out of
133 + * room inside the call's JSON, the API returns only the argument pairs
134 + * that were complete before the cut, and the ability rejects the call
135 + * for its missing `content`. The Localizer failed seven translations of
136 + * one post in a row exactly that way, each attempt cut at the same place.
137 + *
138 + * 16384 is the largest value every current-generation model of the three
139 + * providers accepts (OpenAI's gpt-4o family caps output at exactly that).
140 + * A site that needs more, or pins an older model with a smaller limit,
141 + * sets `max_tokens` in the filter: the filter's value always wins.
142 + */
143 +const OPENSTATION_AI_DEFAULT_MAX_TOKENS = 16384;
144 +
145 +/**
146 + * Whether the provider stopped because the reply hit the output-token
147 + * ceiling.
148 + *
149 + * Anthropic's `max_tokens` and Google's `MAX_TOKENS` stop reasons both
150 + * map to the SDK's LENGTH finish reason. The OpenAI provider throws
151 + * instead and never builds a result; {@see openstation_ai_client_generate()}
152 + * maps that path from the WP_Error Core turns the exception into.
153 + *
154 + * @param mixed $result GenerativeAiResult.
155 + * @return bool
156 + */
157 +function openstation_ai_result_is_truncated( $result ) {
158 + try {
159 + foreach ( $result->getCandidates() as $candidate ) {
160 + if ( $candidate->getFinishReason()->isLength() ) {
161 + return true;
162 + }
163 + }
164 + } catch ( \Throwable $e ) {
165 + return false;
166 + }
167 + return false;
168 +}
169 +
170 +/**
171 + * Builds the error for a turn the output-token ceiling cut short.
172 + *
173 + * A truncated reply is never usable. A JSON answer no longer parses,
174 + * and a function call arrives with only the argument pairs that were
175 + * complete before the cut, so the ability rejects it for a missing
176 + * required field, and the model, reading its own truncated call back
177 + * from history, sends the same call again to the same end. Failing the
178 + * turn here turns a run of silent retries into one error that names
179 + * the cause.
180 + *
181 + * @param string $detail Underlying provider detail, preserved for logs.
182 + * @param array|null $usage Token usage of the truncated turn, if known.
183 + * @return WP_Error
184 + */
185 +function openstation_ai_output_truncated_error( $detail, $usage = null ) {
186 + return new WP_Error(
187 + 'openstation_ai_output_truncated',
188 + __( 'The AI provider cut the reply short at the output-token limit.', 'desktop-mode' ),
189 + array(
190 + 'status' => 502,
191 + 'detail' => (string) $detail,
192 + 'completion_tokens' => is_array( $usage ) && isset( $usage['completion'] ) ? (int) $usage['completion'] : null,
193 + )
194 + );
195 +}
196 +
197 +/**
125 198 * Builds the error for a final turn that produced no answer text.
126 199 *
127 200 * Observed live with the Anthropic provider under agent runs: a hard task
128 201 * spends the entire `max_tokens` budget inside a thinking block
@@ -166,9 +239,11 @@
166 239
167 240 /**
168 241 * Filters the model config for one AI turn.
169 242 *
170 - * Defaults to empty. Recipe: `docs/examples/ai-model-config.md`.
243 + * Defaults to empty; the only value OpenStation fills in afterwards is
244 + * `max_tokens` ({@see OPENSTATION_AI_DEFAULT_MAX_TOKENS}), and only
245 + * when the filter left it unset. Recipe: `docs/examples/ai-model-config.md`.
171 246 *
172 247 * @param array $config { model?: string|ModelInterface, max_tokens?: int, temperature?: float, custom_options?: array<string, mixed> }.
173 248 * @param array $context { user_id, request_id, source, has_tools, has_schema }.
174 249 */
@@ -173,16 +248,18 @@
173 248 * @param array $context { user_id, request_id, source, has_tools, has_schema }.
174 249 */
175 250 $config = apply_filters( 'openstation_ai_model_config', array(), $context );
176 251 if ( ! is_array( $config ) ) {
177 - return $builder;
252 + $config = array();
178 253 }
179 254
180 255 $model_config = new ModelConfig();
181 256
257 + $max_tokens = OPENSTATION_AI_DEFAULT_MAX_TOKENS;
182 258 if ( isset( $config['max_tokens'] ) && is_numeric( $config['max_tokens'] ) && (int) $config['max_tokens'] > 0 ) {
183 - $model_config->setMaxTokens( (int) $config['max_tokens'] );
259 + $max_tokens = (int) $config['max_tokens'];
184 260 }
261 + $model_config->setMaxTokens( $max_tokens );
185 262
186 263 // Unlike max_tokens, 0.0 is a legitimate temperature (deterministic). The
187 264 // 2.0 ceiling is the range the SDK's own schema declares.
188 265 if ( isset( $config['temperature'] ) && is_numeric( $config['temperature'] )
@@ -275,14 +352,23 @@
275 352 );
276 353
277 354 $result = $builder->generate_result();
278 355 if ( is_wp_error( $result ) ) {
356 + // The OpenAI provider reports the output ceiling as an exception
357 + // rather than a finish reason; Core maps it to this code.
358 + if ( 'prompt_token_limit_reached' === $result->get_error_code() ) {
359 + return openstation_ai_output_truncated_error( $result->get_error_message() );
360 + }
279 361 return $result;
280 362 }
281 363
282 364 $message = $result->toMessage();
283 365 $function_calls = array();
366 + $has_text = false;
284 367 foreach ( $message->getParts() as $part ) {
368 + if ( $part->getType()->isText() && ! $part->getChannel()->isThought() ) {
369 + $has_text = true;
370 + }
285 371 if ( ! $part->getType()->isFunctionCall() ) {
286 372 continue;
287 373 }
288 374 $call = $part->getFunctionCall();
@@ -293,8 +379,19 @@
293 379 $function_calls[] = array(
294 380 'name' => (string) $call->getName(),
295 381 'call_id' => (string) $call->getId(),
296 382 'arguments' => wp_json_encode( is_array( $args ) ? $args : array() ),
383 + );
384 + }
385 +
386 + // Whatever the ceiling cut off is partial, and partial is unusable:
387 + // a function call missing its longest argument, or a JSON answer
388 + // missing its closing half. A budget spent entirely on reasoning
389 + // leaves nothing written at all; that case keeps its own error below.
390 + if ( ( ! empty( $function_calls ) || $has_text ) && openstation_ai_result_is_truncated( $result ) ) {
391 + return openstation_ai_output_truncated_error(
392 + 'The provider stopped at the output-token ceiling (finish reason: length).',
393 + openstation_ai_result_token_usage( $result )
297 394 );
298 395 }
299 396
300 397 $text = null;