| @@ -67,8 +67,20 @@ | ||
| 67 | 67 | */ |
| 68 | 68 | const OPENSTATION_AGENT_RUNNER_MAX_TURNS = 8; |
| 69 | 69 | |
| 70 | 70 | /** |
| 71 | + * Consecutive turns in which every tool call failed with the same | |
| 72 | + * errors before the loop gives up on tools and asks for a final answer. | |
| 73 | + * | |
| 74 | + * A model reads a tool error and usually fixes its next call. One that | |
| 75 | + * sends the same failing call a third time is not going to fix it on | |
| 76 | + * the eighth: the Localizer once spent seven turns re-sending a | |
| 77 | + * `create_post` call the ability rejected identically every time. Three | |
| 78 | + * is one honest correction attempt plus proof it did not help. | |
| 79 | + */ | |
| 80 | +const OPENSTATION_AGENT_RUNNER_STUCK_TURNS = 3; | |
| 81 | + | |
| 82 | +/** | |
| 71 | 83 | * Seconds to allow one provider generation request, replacing the |
| 72 | 84 | * WordPress HTTP default of 5. |
| 73 | 85 | * |
| 74 | 86 | * The AI Client's HTTP adapter issues provider calls through |
| @@ -177,8 +189,9 @@ | ||
| 177 | 189 | return $rate; |
| 178 | 190 | } |
| 179 | 191 | |
| 180 | 192 | $instructions = openstation_agent_get_instructions( $user->ID ); |
| 193 | + $instructions = openstation_agent_apply_vibes( $instructions, (int) $user->ID ); | |
| 181 | 194 | $abilities = openstation_agent_get_abilities( $user->ID ); |
| 182 | 195 | |
| 183 | 196 | list( $tool_defs, $slug_by_name ) = openstation_agent_runner_build_tools( $abilities ); |
| 184 | 197 | |
| @@ -503,10 +516,15 @@ | ||
| 503 | 516 | 'text' => (string) $message, |
| 504 | 517 | ); |
| 505 | 518 | $tool_trace = array(); |
| 506 | 519 | |
| 520 | + $turns_used = 0; | |
| 521 | + $last_failure = ''; | |
| 522 | + $repeated_failures = 0; | |
| 523 | + | |
| 507 | 524 | for ( $turn = 1; $turn <= OPENSTATION_AGENT_RUNNER_MAX_TURNS; $turn++ ) { |
| 508 | - $generated = openstation_agent_runner_generate( $agent_user_id, $history, $tool_defs, $instructions ); | |
| 525 | + $turns_used = $turn; | |
| 526 | + $generated = openstation_agent_runner_generate( $agent_user_id, $history, $tool_defs, $instructions ); | |
| 509 | 527 | if ( is_wp_error( $generated ) && openstation_agent_generate_error_is_transient( $generated ) ) { |
| 510 | 528 | // One bounded retry for provider-side hiccups (a failed |
| 511 | 529 | // models-list fetch, a gateway timeout, a borderline |
| 512 | 530 | // refusal). A manual "try again" was already the working |
| @@ -522,11 +540,20 @@ | ||
| 522 | 540 | ? $generated['function_calls'] |
| 523 | 541 | : array(); |
| 524 | 542 | |
| 525 | 543 | if ( empty( $function_calls ) ) { |
| 526 | - $answer = openstation_agent_parse_answer( | |
| 527 | - isset( $generated['text'] ) && is_string( $generated['text'] ) ? $generated['text'] : '' | |
| 528 | - ); | |
| 544 | + // Belt-and-braces behind the same check in | |
| 545 | + // openstation_ai_client_generate(): a final turn with no | |
| 546 | + // extractable text is a failed generation, never a valid | |
| 547 | + // empty answer — without this, the run reports success and | |
| 548 | + // the chat renders nothing. | |
| 549 | + $text = isset( $generated['text'] ) && is_string( $generated['text'] ) ? $generated['text'] : ''; | |
| 550 | + if ( '' === trim( $text ) ) { | |
| 551 | + return openstation_agent_humanize_generate_error( | |
| 552 | + openstation_ai_empty_answer_error( 'The generation produced neither function calls nor answer text.' ) | |
| 553 | + ); | |
| 554 | + } | |
| 555 | + $answer = openstation_agent_parse_answer( $text ); | |
| 529 | 556 | return array( |
| 530 | 557 | 'text' => $answer['text'], |
| 531 | 558 | 'callToActions' => $answer['callToActions'], |
| 532 | 559 | 'toolCalls' => $tool_trace, |
| @@ -599,15 +626,27 @@ | ||
| 599 | 626 | $history[] = array( |
| 600 | 627 | 'type' => 'tool_results', |
| 601 | 628 | 'results' => $results, |
| 602 | 629 | ); |
| 630 | + | |
| 631 | + $failure = openstation_agent_runner_failure_signature( $results ); | |
| 632 | + if ( '' !== $failure && $failure === $last_failure ) { | |
| 633 | + ++$repeated_failures; | |
| 634 | + } else { | |
| 635 | + $repeated_failures = '' === $failure ? 0 : 1; | |
| 636 | + } | |
| 637 | + $last_failure = $failure; | |
| 638 | + if ( $repeated_failures >= OPENSTATION_AGENT_RUNNER_STUCK_TURNS ) { | |
| 639 | + break; | |
| 640 | + } | |
| 603 | 641 | } |
| 604 | 642 | |
| 605 | - // Cap reached with the model still asking for tools. Force one | |
| 606 | - // last TOOL-LESS generate over the transcript so far: with nothing | |
| 607 | - // to call, the model can only produce a final answer from what it | |
| 608 | - // already gathered. A best-effort summary beats discarding the | |
| 609 | - // whole run (observed on Anthropic: a model happily spends the cap | |
| 643 | + // Cap reached (or the model stuck re-sending the same failing | |
| 644 | + // call) with it still asking for tools. Force one last TOOL-LESS | |
| 645 | + // generate over the transcript so far: with nothing to call, the | |
| 646 | + // model can only produce a final answer from what it already | |
| 647 | + // gathered. A best-effort summary beats discarding the whole run | |
| 648 | + // (observed on Anthropic: a model happily spends the cap | |
| 610 | 649 | // re-searching before it answers). |
| 611 | 650 | $generated = openstation_agent_runner_generate( $agent_user_id, $history, array(), $instructions ); |
| 612 | 651 | if ( is_wp_error( $generated ) && openstation_agent_generate_error_is_transient( $generated ) ) { |
| 613 | 652 | $generated = openstation_agent_runner_generate( $agent_user_id, $history, array(), $instructions ); |
| @@ -619,9 +658,9 @@ | ||
| 619 | 658 | return array( |
| 620 | 659 | 'text' => $answer['text'], |
| 621 | 660 | 'callToActions' => $answer['callToActions'], |
| 622 | 661 | 'toolCalls' => $tool_trace, |
| 623 | - 'turns' => OPENSTATION_AGENT_RUNNER_MAX_TURNS + 1, | |
| 662 | + 'turns' => $turns_used + 1, | |
| 624 | 663 | ); |
| 625 | 664 | } |
| 626 | 665 | |
| 627 | 666 | return new WP_Error( |
| @@ -626,16 +665,45 @@ | ||
| 626 | 665 | |
| 627 | 666 | return new WP_Error( |
| 628 | 667 | 'openstation_agent_runner_max_turns', |
| 629 | 668 | sprintf( |
| 630 | - /* translators: %d is the max-turn cap. */ | |
| 669 | + /* translators: %d is the number of turns the run made. */ | |
| 631 | 670 | __( 'Agent stopped after %d turns without a final answer.', 'desktop-mode' ), |
| 632 | - OPENSTATION_AGENT_RUNNER_MAX_TURNS | |
| 671 | + $turns_used | |
| 633 | 672 | ) |
| 634 | 673 | ); |
| 635 | 674 | } |
| 636 | 675 | |
| 637 | 676 | /** |
| 677 | + * Fingerprint of a turn in which every tool call failed: the sorted | |
| 678 | + * tool names with their error messages. Empty when any call succeeded, | |
| 679 | + * so a turn that got something done never counts as stuck. | |
| 680 | + * | |
| 681 | + * Arguments are deliberately left out. The failure that motivated this | |
| 682 | + * had the model vary a title between attempts while the ability | |
| 683 | + * rejected each one for the same missing field, and that IS the same | |
| 684 | + * failure. | |
| 685 | + * | |
| 686 | + * @param array $results Tool results of one turn (`{ name, response }` rows). | |
| 687 | + * @return string | |
| 688 | + */ | |
| 689 | +function openstation_agent_runner_failure_signature( array $results ) { | |
| 690 | + if ( empty( $results ) ) { | |
| 691 | + return ''; | |
| 692 | + } | |
| 693 | + $failures = array(); | |
| 694 | + foreach ( $results as $row ) { | |
| 695 | + $response = isset( $row['response'] ) ? $row['response'] : null; | |
| 696 | + if ( ! is_array( $response ) || ! isset( $response['error'] ) ) { | |
| 697 | + return ''; | |
| 698 | + } | |
| 699 | + $failures[] = ( isset( $row['name'] ) ? (string) $row['name'] : '' ) . "\0" . (string) $response['error']; | |
| 700 | + } | |
| 701 | + sort( $failures ); | |
| 702 | + return implode( "\n", $failures ); | |
| 703 | +} | |
| 704 | + | |
| 705 | +/** | |
| 638 | 706 | * JSON Schema every agent's FINAL answer is constrained to (via the |
| 639 | 707 | * AI Client's structured output, `as_json_response()`): the markdown |
| 640 | 708 | * answer in `text`, plus optional `call_to_actions` the chat renders |
| 641 | 709 | * as buttons when the agent needs the user's confirmation instead of |
| @@ -886,8 +954,30 @@ | ||
| 886 | 954 | 'detail' => $error->get_error_message(), |
| 887 | 955 | ) |
| 888 | 956 | ); |
| 889 | 957 | } |
| 958 | + if ( 'openstation_ai_output_truncated' === $error->get_error_code() ) { | |
| 959 | + $data = $error->get_error_data(); | |
| 960 | + return new WP_Error( | |
| 961 | + 'openstation_agent_output_truncated', | |
| 962 | + __( 'The reply ran past the output-token limit before it finished, so it was discarded rather than acted on incomplete. Ask for something shorter, or raise max_tokens with the openstation_ai_model_config filter.', 'desktop-mode' ), | |
| 963 | + array( | |
| 964 | + 'status' => 502, | |
| 965 | + 'detail' => is_array( $data ) && isset( $data['detail'] ) ? (string) $data['detail'] : '', | |
| 966 | + ) | |
| 967 | + ); | |
| 968 | + } | |
| 969 | + if ( 'openstation_ai_empty_answer' === $error->get_error_code() ) { | |
| 970 | + $data = $error->get_error_data(); | |
| 971 | + return new WP_Error( | |
| 972 | + 'openstation_agent_empty_answer', | |
| 973 | + __( 'The model ran out of room before writing its answer — it most likely spent the whole output budget reasoning. Try a narrower request, or try again.', 'desktop-mode' ), | |
| 974 | + array( | |
| 975 | + 'status' => 502, | |
| 976 | + 'detail' => is_array( $data ) && isset( $data['detail'] ) ? (string) $data['detail'] : '', | |
| 977 | + ) | |
| 978 | + ); | |
| 979 | + } | |
| 890 | 980 | return $error; |
| 891 | 981 | } |
| 892 | 982 | |
| 893 | 983 | /** |
| @@ -947,9 +1037,10 @@ | ||
| 947 | 1037 | // confirmations arrive as renderable buttons, not typed-reply |
| 948 | 1038 | // requests. Tool-call turns are unaffected — the model either |
| 949 | 1039 | // calls a function or emits the JSON answer. |
| 950 | 1040 | openstation_agent_answer_schema(), |
| 951 | - (string) $instructions . "\n\n" . openstation_agent_answer_prompt_appendix() | |
| 1041 | + (string) $instructions . "\n\n" . openstation_agent_answer_prompt_appendix(), | |
| 1042 | + array( 'source' => 'agents/runner' ) | |
| 952 | 1043 | ); |
| 953 | 1044 | } |
| 954 | 1045 | ); |
| 955 | 1046 | } |
| @@ -954,8 +1045,35 @@ | ||
| 954 | 1045 | ); |
| 955 | 1046 | } |
| 956 | 1047 | |
| 957 | 1048 | /** |
| 1049 | + * Append an agent's voice line to its instructions. | |
| 1050 | + * | |
| 1051 | + * **After the instructions, never before.** The two can disagree — a | |
| 1052 | + * voice that says "blunt" against a workflow that says "always explain | |
| 1053 | + * your reasoning" — and when they do, the workflow should win. Later | |
| 1054 | + * text is the one the model weights more heavily, so position is the | |
| 1055 | + * whole mechanism here. | |
| 1056 | + * | |
| 1057 | + * The line is stored through `openstation_agent_sanitize_vibes()`, | |
| 1058 | + * which strips line breaks. That matters more than it looks: the | |
| 1059 | + * composed prompt marks operator turns, and a multi-line voice line | |
| 1060 | + * could otherwise fake a turn boundary. `agentsSecurity.php` pins it. | |
| 1061 | + * | |
| 1062 | + * @param string $instructions The agent's system prompt. | |
| 1063 | + * @param int $user_id Agent user id. | |
| 1064 | + * @return string | |
| 1065 | + */ | |
| 1066 | +function openstation_agent_apply_vibes( $instructions, $user_id ) { | |
| 1067 | + $vibes = openstation_agent_get_vibes( $user_id ); | |
| 1068 | + if ( '' === $vibes ) { | |
| 1069 | + return $instructions; | |
| 1070 | + } | |
| 1071 | + $line = 'Voice: ' . $vibes; | |
| 1072 | + return '' === $instructions ? $line : $instructions . "\n\n" . $line; | |
| 1073 | +} | |
| 1074 | + | |
| 1075 | +/** | |
| 958 | 1076 | * Run a callback with the WordPress HTTP timeout raised for the |
| 959 | 1077 | * provider request it makes. |
| 960 | 1078 | * |
| 961 | 1079 | * Scoped to the generation call rather than the whole run: tool |
| @@ -1194,10 +1312,13 @@ | ||
| 1194 | 1312 | return $ability->execute( $args ); |
| 1195 | 1313 | } |
| 1196 | 1314 | |
| 1197 | 1315 | /** |
| 1198 | - * Append one invocation to the agent's persistent log. Most-recent | |
| 1199 | - * entries surface in the chat window's history strip. | |
| 1316 | + * Append one invocation to the agent's persistent log: an audit trail | |
| 1317 | + * of who ran the agent and what came back, capped at | |
| 1318 | + * OPENSTATION_AGENT_RUNNER_LOG_CAP entries and readable from PHP with | |
| 1319 | + * {@see openstation_agent_runner_get_log()}. No UI shows it; the chat | |
| 1320 | + * window's history is the human's saved conversations, not this log. | |
| 1200 | 1321 | * |
| 1201 | 1322 | * @param int $agent_user_id Agent user id. |
| 1202 | 1323 | * @param string $message Submitted message. |
| 1203 | 1324 | * @param array $result `{ text, toolCalls, turns }`. |