PluginProbe
OpenStation: Desktop Windows, Dock & Virtual Desktops for WP Admin / 1.1.12
OpenStation: Desktop Windows, Dock & Virtual Desktops for WP Admin v1.1.12
1.1.12 1.1.11 1.1.10 1.1.9 1.1.8 1.1.7 1.1.6 1.1.5 1.1.4 1.1.3 1.1.2 1.1.1 1.1.0 1.0.1 1.0.0 0.9.8 0.9.7 0.9.6 0.9.4 0.9.5 0.9.3 0.9.2 0.9.1 0.9.0 0.8.9 All 36 releases
← All changes | includes/agents/runner.php +113 -12 1.0.1 → 1.1.12 View file →
@@ -67,8 +67,20 @@
67 67 */
68 68 const OPENSTATION_AGENT_RUNNER_MAX_TURNS = 8;
69 69
70 70 /**
71 + * Consecutive turns in which every tool call failed with the same
72 + * errors before the loop gives up on tools and asks for a final answer.
73 + *
74 + * A model reads a tool error and usually fixes its next call. One that
75 + * sends the same failing call a third time is not going to fix it on
76 + * the eighth: the Localizer once spent seven turns re-sending a
77 + * `create_post` call the ability rejected identically every time. Three
78 + * is one honest correction attempt plus proof it did not help.
79 + */
80 +const OPENSTATION_AGENT_RUNNER_STUCK_TURNS = 3;
81 +
82 +/**
71 83 * Seconds to allow one provider generation request, replacing the
72 84 * WordPress HTTP default of 5.
73 85 *
74 86 * The AI Client's HTTP adapter issues provider calls through
@@ -177,8 +189,9 @@
177 189 return $rate;
178 190 }
179 191
180 192 $instructions = openstation_agent_get_instructions( $user->ID );
193 + $instructions = openstation_agent_apply_vibes( $instructions, (int) $user->ID );
181 194 $abilities = openstation_agent_get_abilities( $user->ID );
182 195
183 196 list( $tool_defs, $slug_by_name ) = openstation_agent_runner_build_tools( $abilities );
184 197
@@ -503,10 +516,15 @@
503 516 'text' => (string) $message,
504 517 );
505 518 $tool_trace = array();
506 519
520 + $turns_used = 0;
521 + $last_failure = '';
522 + $repeated_failures = 0;
523 +
507 524 for ( $turn = 1; $turn <= OPENSTATION_AGENT_RUNNER_MAX_TURNS; $turn++ ) {
508 - $generated = openstation_agent_runner_generate( $agent_user_id, $history, $tool_defs, $instructions );
525 + $turns_used = $turn;
526 + $generated = openstation_agent_runner_generate( $agent_user_id, $history, $tool_defs, $instructions );
509 527 if ( is_wp_error( $generated ) && openstation_agent_generate_error_is_transient( $generated ) ) {
510 528 // One bounded retry for provider-side hiccups (a failed
511 529 // models-list fetch, a gateway timeout, a borderline
512 530 // refusal). A manual "try again" was already the working
@@ -608,15 +626,27 @@
608 626 $history[] = array(
609 627 'type' => 'tool_results',
610 628 'results' => $results,
611 629 );
630 +
631 + $failure = openstation_agent_runner_failure_signature( $results );
632 + if ( '' !== $failure && $failure === $last_failure ) {
633 + ++$repeated_failures;
634 + } else {
635 + $repeated_failures = '' === $failure ? 0 : 1;
636 + }
637 + $last_failure = $failure;
638 + if ( $repeated_failures >= OPENSTATION_AGENT_RUNNER_STUCK_TURNS ) {
639 + break;
640 + }
612 641 }
613 642
614 - // Cap reached with the model still asking for tools. Force one
615 - // last TOOL-LESS generate over the transcript so far: with nothing
616 - // to call, the model can only produce a final answer from what it
617 - // already gathered. A best-effort summary beats discarding the
618 - // whole run (observed on Anthropic: a model happily spends the cap
643 + // Cap reached (or the model stuck re-sending the same failing
644 + // call) with it still asking for tools. Force one last TOOL-LESS
645 + // generate over the transcript so far: with nothing to call, the
646 + // model can only produce a final answer from what it already
647 + // gathered. A best-effort summary beats discarding the whole run
648 + // (observed on Anthropic: a model happily spends the cap
619 649 // re-searching before it answers).
620 650 $generated = openstation_agent_runner_generate( $agent_user_id, $history, array(), $instructions );
621 651 if ( is_wp_error( $generated ) && openstation_agent_generate_error_is_transient( $generated ) ) {
622 652 $generated = openstation_agent_runner_generate( $agent_user_id, $history, array(), $instructions );
@@ -628,9 +658,9 @@
628 658 return array(
629 659 'text' => $answer['text'],
630 660 'callToActions' => $answer['callToActions'],
631 661 'toolCalls' => $tool_trace,
632 - 'turns' => OPENSTATION_AGENT_RUNNER_MAX_TURNS + 1,
662 + 'turns' => $turns_used + 1,
633 663 );
634 664 }
635 665
636 666 return new WP_Error(
@@ -635,16 +665,45 @@
635 665
636 666 return new WP_Error(
637 667 'openstation_agent_runner_max_turns',
638 668 sprintf(
639 - /* translators: %d is the max-turn cap. */
669 + /* translators: %d is the number of turns the run made. */
640 670 __( 'Agent stopped after %d turns without a final answer.', 'desktop-mode' ),
641 - OPENSTATION_AGENT_RUNNER_MAX_TURNS
671 + $turns_used
642 672 )
643 673 );
644 674 }
645 675
646 676 /**
677 + * Fingerprint of a turn in which every tool call failed: the sorted
678 + * tool names with their error messages. Empty when any call succeeded,
679 + * so a turn that got something done never counts as stuck.
680 + *
681 + * Arguments are deliberately left out. The failure that motivated this
682 + * had the model vary a title between attempts while the ability
683 + * rejected each one for the same missing field, and that IS the same
684 + * failure.
685 + *
686 + * @param array $results Tool results of one turn (`{ name, response }` rows).
687 + * @return string
688 + */
689 +function openstation_agent_runner_failure_signature( array $results ) {
690 + if ( empty( $results ) ) {
691 + return '';
692 + }
693 + $failures = array();
694 + foreach ( $results as $row ) {
695 + $response = isset( $row['response'] ) ? $row['response'] : null;
696 + if ( ! is_array( $response ) || ! isset( $response['error'] ) ) {
697 + return '';
698 + }
699 + $failures[] = ( isset( $row['name'] ) ? (string) $row['name'] : '' ) . "\0" . (string) $response['error'];
700 + }
701 + sort( $failures );
702 + return implode( "\n", $failures );
703 +}
704 +
705 +/**
647 706 * JSON Schema every agent's FINAL answer is constrained to (via the
648 707 * AI Client's structured output, `as_json_response()`): the markdown
649 708 * answer in `text`, plus optional `call_to_actions` the chat renders
650 709 * as buttons when the agent needs the user's confirmation instead of
@@ -895,8 +954,19 @@
895 954 'detail' => $error->get_error_message(),
896 955 )
897 956 );
898 957 }
958 + if ( 'openstation_ai_output_truncated' === $error->get_error_code() ) {
959 + $data = $error->get_error_data();
960 + return new WP_Error(
961 + 'openstation_agent_output_truncated',
962 + __( 'The reply ran past the output-token limit before it finished, so it was discarded rather than acted on incomplete. Ask for something shorter, or raise max_tokens with the openstation_ai_model_config filter.', 'desktop-mode' ),
963 + array(
964 + 'status' => 502,
965 + 'detail' => is_array( $data ) && isset( $data['detail'] ) ? (string) $data['detail'] : '',
966 + )
967 + );
968 + }
899 969 if ( 'openstation_ai_empty_answer' === $error->get_error_code() ) {
900 970 $data = $error->get_error_data();
901 971 return new WP_Error(
902 972 'openstation_agent_empty_answer',
@@ -967,9 +1037,10 @@
967 1037 // confirmations arrive as renderable buttons, not typed-reply
968 1038 // requests. Tool-call turns are unaffected — the model either
969 1039 // calls a function or emits the JSON answer.
970 1040 openstation_agent_answer_schema(),
971 - (string) $instructions . "\n\n" . openstation_agent_answer_prompt_appendix()
1041 + (string) $instructions . "\n\n" . openstation_agent_answer_prompt_appendix(),
1042 + array( 'source' => 'agents/runner' )
972 1043 );
973 1044 }
974 1045 );
975 1046 }
@@ -974,8 +1045,35 @@
974 1045 );
975 1046 }
976 1047
977 1048 /**
1049 + * Append an agent's voice line to its instructions.
1050 + *
1051 + * **After the instructions, never before.** The two can disagree — a
1052 + * voice that says "blunt" against a workflow that says "always explain
1053 + * your reasoning" — and when they do, the workflow should win. Later
1054 + * text is the one the model weights more heavily, so position is the
1055 + * whole mechanism here.
1056 + *
1057 + * The line is stored through `openstation_agent_sanitize_vibes()`,
1058 + * which strips line breaks. That matters more than it looks: the
1059 + * composed prompt marks operator turns, and a multi-line voice line
1060 + * could otherwise fake a turn boundary. `agentsSecurity.php` pins it.
1061 + *
1062 + * @param string $instructions The agent's system prompt.
1063 + * @param int $user_id Agent user id.
1064 + * @return string
1065 + */
1066 +function openstation_agent_apply_vibes( $instructions, $user_id ) {
1067 + $vibes = openstation_agent_get_vibes( $user_id );
1068 + if ( '' === $vibes ) {
1069 + return $instructions;
1070 + }
1071 + $line = 'Voice: ' . $vibes;
1072 + return '' === $instructions ? $line : $instructions . "\n\n" . $line;
1073 +}
1074 +
1075 +/**
978 1076 * Run a callback with the WordPress HTTP timeout raised for the
979 1077 * provider request it makes.
980 1078 *
981 1079 * Scoped to the generation call rather than the whole run: tool
@@ -1214,10 +1312,13 @@
1214 1312 return $ability->execute( $args );
1215 1313 }
1216 1314
1217 1315 /**
1218 - * Append one invocation to the agent's persistent log. Most-recent
1219 - * entries surface in the chat window's history strip.
1316 + * Append one invocation to the agent's persistent log: an audit trail
1317 + * of who ran the agent and what came back, capped at
1318 + * OPENSTATION_AGENT_RUNNER_LOG_CAP entries and readable from PHP with
1319 + * {@see openstation_agent_runner_get_log()}. No UI shows it; the chat
1320 + * window's history is the human's saved conversations, not this log.
1220 1321 *
1221 1322 * @param int $agent_user_id Agent user id.
1222 1323 * @param string $message Submitted message.
1223 1324 * @param array $result `{ text, toolCalls, turns }`.